e2e, demos, ate-setup: create substrate ActorTemplates directly and match actors by the actor_template ref

Stop relying on the ActorTemplate CRD path and the legacy actor_template
fields on the consumer side: ate-setup matches demo actors and the e2e
suites name probe templates by the actor_template ref, the egress and
MITM egress demos and the e2e fixtures (probe family, networking
ingress) ship substrate ActorTemplate manifests, and the e2e harness
gains helpers to deploy substrate fixtures.
This commit is contained in:
zoezhao
2026-08-31 17:45:53 -07:00
parent 594e8beb60
commit 1d5d8f066f
50 changed files with 1299 additions and 1033 deletions
@@ -47,8 +47,8 @@ func init() {
DemoName: "demo-autoscaled-workerpool",
Short: "A WorkerPool scaled by an HPA over custom metrics (Kind only)",
Template: "demos/autoscaled-workerpool/autoscaled-workerpool.yaml.tmpl",
Deployments: []steps.TemplateRef{{Namespace: namespace, Name: "counter"}},
ActorTemplates: []steps.TemplateRef{{Namespace: namespace, Name: "counter"}},
Deployments: []steps.TemplateRef{{Atespace: namespace, Name: "counter"}},
ActorTemplates: []steps.TemplateRef{{Atespace: namespace, Name: "counter"}},
}})
}
@@ -50,9 +50,9 @@ const (
// agents are the actors the demo template declares. Their actors are removed
// before the manifests at delete time.
var agents = []steps.TemplateRef{
{Namespace: namespace, Name: "agent-luna"},
{Namespace: namespace, Name: "agent-mars"},
{Namespace: namespace, Name: "agent-orion"},
{Atespace: namespace, Name: "agent-luna"},
{Atespace: namespace, Name: "agent-mars"},
{Atespace: namespace, Name: "agent-orion"},
}
type demo struct{}
@@ -61,8 +61,8 @@ func init() {
DemoName: "demo-counter",
Short: "A counter actor exercising snapshot, resume, and atenet ingress",
Template: template,
Deployments: []steps.TemplateRef{{Namespace: namespace, Name: "counter"}},
ActorTemplates: []steps.TemplateRef{{Namespace: namespace, Name: "counter"}},
Deployments: []steps.TemplateRef{{Atespace: namespace, Name: "counter"}},
ActorTemplates: []steps.TemplateRef{{Atespace: namespace, Name: "counter"}},
}})
}
@@ -28,7 +28,7 @@ func init() {
DemoName: "demo-egress",
Short: "Egress policy enforcement through atenet",
Template: "demos/egress/egress.yaml.tmpl",
Deployments: []steps.TemplateRef{{Namespace: namespace, Name: "egress"}},
ActorTemplates: []steps.TemplateRef{{Namespace: namespace, Name: "egress"}},
Deployments: []steps.TemplateRef{{Atespace: namespace, Name: "egress"}},
ActorTemplates: []steps.TemplateRef{{Atespace: namespace, Name: "egress"}},
})
}
@@ -26,10 +26,10 @@ func init() {
DemoName: "demo-multi-template",
Short: "Two ActorTemplates sharing one WorkerPool",
Template: "demos/multi-template/multi-template.yaml.tmpl",
Deployments: []steps.TemplateRef{{Namespace: "ate-demo-multi-template-pool", Name: "shared-pool"}},
Deployments: []steps.TemplateRef{{Atespace: "ate-demo-multi-template-pool", Name: "shared-pool"}},
ActorTemplates: []steps.TemplateRef{
{Namespace: "ate-demo-multi-template-counter", Name: "counter"},
{Namespace: "ate-demo-multi-template-fspersist", Name: "fspersist"},
{Atespace: "ate-demo-multi-template-counter", Name: "counter"},
{Atespace: "ate-demo-multi-template-fspersist", Name: "fspersist"},
},
})
}
@@ -28,7 +28,7 @@ func init() {
DemoName: "demo-parking",
Short: "Actor parking and unparking on a small WorkerPool",
Template: "demos/parking/parking.yaml.tmpl",
Deployments: []steps.TemplateRef{{Namespace: namespace, Name: "parking"}},
ActorTemplates: []steps.TemplateRef{{Namespace: namespace, Name: "parking"}},
Deployments: []steps.TemplateRef{{Atespace: namespace, Name: "parking"}},
ActorTemplates: []steps.TemplateRef{{Atespace: namespace, Name: "parking"}},
})
}
@@ -28,7 +28,7 @@ func init() {
DemoName: "demo-sandbox",
Short: "An on-demand sandbox actor driven by the sandbox client",
Template: "demos/sandbox/sandbox.yaml.tmpl",
ActorTemplates: []steps.TemplateRef{{Namespace: namespace, Name: "sandbox-template"}},
ActorTemplates: []steps.TemplateRef{{Atespace: namespace, Name: "sandbox-template"}},
// There is no workload to come up, and the template is exercised on
// demand, so the install does not block on readiness.
SkipReadinessWait: true,
+2 -2
View File
@@ -142,12 +142,12 @@ func (d *Simple) WaitReady(ctx context.Context, e *steps.Env) error {
}
log.Stepf("Waiting for %s to be ready...", d.DemoName)
for _, ref := range d.Deployments {
if err := e.Kube.RolloutStatus(ctx, kube.KindDeployment, ref.Namespace, ref.Name, steps.DemoTimeout); err != nil {
if err := e.Kube.RolloutStatus(ctx, kube.KindDeployment, ref.Atespace, ref.Name, steps.DemoTimeout); err != nil {
return err
}
}
for _, ref := range d.ActorTemplates {
if err := WaitActorTemplateReady(ctx, e, ref.Namespace, ref.Name); err != nil {
if err := WaitActorTemplateReady(ctx, e, ref.Atespace, ref.Name); err != nil {
return err
}
}
+6 -4
View File
@@ -27,8 +27,8 @@ import (
// against it because the demo manifests own the template, not the actors that
// were created from it.
type TemplateRef struct {
Namespace string
Name string
Atespace string
Name string
}
// DeleteDemoActors removes every actor created from the given ActorTemplates.
@@ -66,9 +66,11 @@ func (e *Env) DeleteDemoActors(ctx context.Context, refs ...TemplateRef) error {
}
for _, ref := range refs {
log.Stepf("Deleting actors for %s/%s", ref.Namespace, ref.Name)
log.Stepf("Deleting actors for %s/%s", ref.Atespace, ref.Name)
for _, actor := range actors {
if actor.GetActorTemplateNamespace() != ref.Namespace || actor.GetActorTemplateName() != ref.Name {
// Actors name their template through the actor_template ref; demo
// templates keep the CRD namespace as the ref's atespace.
if actor.GetActorTemplate().GetAtespace() != ref.Atespace || actor.GetActorTemplate().GetName() != ref.Name {
continue
}
actorRef := resources.ActorRefFromActor(actor)
+15 -8
View File
@@ -75,7 +75,14 @@ intercepted and carried over mTLS to a gateway that verifies who is making the r
```bash
./hack/install-ate.sh --deploy-demo-egress
kubectl wait --for=condition=Ready actortemplate/egress -n ate-demo-egress --timeout=5m
```
The install applies the worker pool, creates the `ate-demo-egress` atespace and
the `egress` ActorTemplate (a substrate resource, not a CRD) through the ate
API, and blocks until the template's golden snapshot is built:
```bash
kubectl ate get actor-template egress -a ate-demo-egress
```
## Run the automated test (easiest)
@@ -104,15 +111,15 @@ kubectl -n egress-target create deployment whoami --image=traefik/whoami
kubectl -n egress-target expose deployment whoami --port=80
TARGET_IP=$(kubectl -n egress-target get svc whoami -o jsonpath='{.spec.clusterIP}')
# 2. Create and resume an Actor.
kubectl ate create atespace demo
kubectl ate create actor egress-demo -a demo --template ate-demo-egress/egress
kubectl ate resume actor egress-demo -a demo # wait for ACTOR_STATE_RUNNING
# 2. Create and resume an Actor in the demo's atespace: --template-ref
# resolves the template by name within the actor's own atespace.
kubectl ate create actor egress-demo -a ate-demo-egress --template-ref egress
kubectl ate resume actor egress-demo -a ate-demo-egress # wait for ACTOR_STATE_RUNNING
# 3. Drive the Actor's egress through the ingress gateway.
kubectl -n ate-system port-forward service/atenet-router 8000:80 &
curl -s -X POST http://localhost:8000/ \
-H 'Host: egress-demo.demo.actors.resources.substrate.ate.dev' \
-H 'Host: egress-demo.ate-demo-egress.actors.resources.substrate.ate.dev' \
-H 'Content-Type: application/json' \
-d "{\"url\":\"http://${TARGET_IP}:80/\"}"
```
@@ -122,11 +129,11 @@ curl -s -X POST http://localhost:8000/ \
```bash
# The egress gateway logs each tunneled CONNECT against the verified peer certificate:
kubectl -n ate-system logs deploy/atenet-egress | grep '\[egress\]'
# [egress] authority=<TARGET_IP>:80 peer_san=spiffe://substrate-actor.local/atespace/demo/actor/egress-demo … code=200 …
# [egress] authority=<TARGET_IP>:80 peer_san=spiffe://substrate-actor.local/atespace/ate-demo-egress/actor/egress-demo … code=200 …
# The co-located ext_proc sidecar logs the identity decision, including the UID it authorized on:
kubectl -n ate-system logs deploy/atenet-egress -c ext-proc | grep -i 'egress identity\|egress denied'
# egress identity authenticated atespace=demo actor=egress-demo actorUid=… destination=<TARGET_IP>:80
# egress identity authenticated atespace=ate-demo-egress actor=egress-demo actorUid=… destination=<TARGET_IP>:80
```
The `whoami` body shows `RemoteAddr: <atenet-egress pod IP>` — proof the request egressed
@@ -0,0 +1,86 @@
# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# Micro-VM MITM variant of the egress demo template. Delivery of
# the projected bundle into the sandbox is what differs by class — a read-only
# bind for gVisor, the unified virtio-fs share for the guest here — so the
# projection is worth proving on both.
#
# Kept in step with egress-microvm-template.yaml.tmpl; the deltas
# are the names, the projected trust bundle, and the two SSL_CERT_* variables.
metadata:
atespace: ate-demo-egress-microvm-mitm
name: egress-microvm-mitm
workerSelector:
matchLabels:
workload: egress-microvm-mitm
containers:
- name: egress
image: ko://github.com/agent-substrate/substrate/demos/egress
command: ["/ko-app/egress"]
# The demo fetches with a stock net/http client, so its roots come from
# crypto/x509's system pool — which on Unix reads both of these.
#
# SSL_CERT_FILE alone is not enough to make the projected bundle the whole
# story: it replaces the default cert FILE list, but the default cert
# DIRECTORY list is still scanned, and the base image keeps its public
# roots in /etc/ssl/certs. Pointing SSL_CERT_DIR at the projection too
# makes the anchor set exactly the gateway CA, so a successful HTTPS fetch
# proves the projected bundle validated the minted leaf. Under sdsmint the
# public roots are useless anyway: every origin is fronted by the gateway.
env:
- name: SSL_CERT_FILE
value: /run/ate/trust-bundle.pem
- name: SSL_CERT_DIR
value: /run/ate
volumeMounts:
# the bundle lands at /run/ate/trust-bundle.pem
- name: system-info
mountPath: /run/ate
readyz:
httpGet:
path: /readyz
port: 80
# See counter-substrate-microvm-template.yaml.tmpl: a micro-VM guest needs
# more than readyz's 30s default once CI's single kind node is under load.
timeoutSeconds: 120
volumes:
- name: system-info
type: SystemInfo
systemInfo:
dataSources:
# The trust anchors for the per-SNI leaves the MITM egress gateway mints.
# Only allowlisted bundle names resolve; this is the one supported today.
- trustBundle:
name: egress-mitm.ate.dev
path: trust-bundle.pem
# Sandbox size: without these the guest boots at the kata config's default
# (2GiB). ateom applies them to the VM (see internal/sizing). Quantities are
# strings in the proto.
resources:
limits:
- name: cpu
quantity: "1"
- name: memory
quantity: 512Mi
sandboxConfig:
sandboxClass: SANDBOX_CLASS_MICROVM
# Deliberately not the class default; installed cluster-wide by
# hack/install-microvm-deps.sh, so a missing or stale install fails loudly.
configName: microvm
snapshotsConfig:
onPause: SNAPSHOT_CONTENT_SCOPE_FULL
onCommit: SNAPSHOT_CONTENT_SCOPE_FULL
storageLocation: gs://${BUCKET_NAME}/ate-demo-egress-microvm-mitm/
+12 -76
View File
@@ -12,19 +12,12 @@
# See the License for the specific language governing permissions and
# limitations under the License.
# Micro-VM (kata + cloud-hypervisor) variant of the MITM egress demo, so the
# networking suite's egress tests run against the sdsmint gateway on both
# sandbox classes. Delivery of the projected bundle into the sandbox is what
# differs by class — a read-only bind for gVisor, the unified virtio-fs share
# for the guest here — so the projection is worth proving on both.
#
# As in egress-mitm.yaml.tmpl, this REQUIRES an sdsmint install
# (--experimental-use-sdsmint).
#
# Kept in step with egress-microvm.yaml.tmpl; the deltas are the names, the
# projected trust bundle, and the two SSL_CERT_* variables. The cluster-wide
# `microvm` SandboxConfig referenced below by name is installed by
# hack/install-microvm-deps.sh.
# Worker pool for the micro-VM MITM variant of the egress demo, so
# the networking suite's egress tests run against the sdsmint gateway on both
# sandbox classes. As in egress-mitm.yaml.tmpl, this REQUIRES an
# sdsmint install (--experimental-use-sdsmint). The cluster-wide `microvm`
# SandboxConfig referenced by the pool is installed by
# hack/install-microvm-deps.sh
apiVersion: v1
kind: Namespace
@@ -45,66 +38,9 @@ spec:
sandboxClass: microvm
sandboxConfigName: microvm
workerImage: ko://github.com/agent-substrate/substrate/cmd/ateom-microvm
# No template.resources, unlike counter-microvm: an unlimited ateom container
# reports no capacity, which the scheduler reads as unconstrained, and the pod
# requests nothing so it stays placeable next to the counter-microvm pool's
# reservation on CI's single kind node. What actually bounds the memory here
# is the ActorTemplate's resources below, which size the guest.
---
apiVersion: ate.dev/v1alpha1
kind: ActorTemplate
metadata:
name: egress-microvm-mitm
namespace: ate-demo-egress-microvm-mitm
spec:
# Must match the WorkerPool's sandboxClass: a snapshot is not portable across
# sandbox classes, so only pools of the same class are eligible to run this
# template's actors.
sandboxClass: microvm
volumes:
- name: system-info
systemInfo:
dataSources:
# The trust anchors for the per-SNI leaves the MITM egress gateway mints.
# Only allowlisted bundle names resolve; this is the one supported today.
- trustBundle:
name: egress-mitm.ate.dev
path: trust-bundle.pem
containers:
- name: egress
image: ko://github.com/agent-substrate/substrate/demos/egress
command: ["/ko-app/egress"]
# Both variables, for the reason egress-mitm.yaml.tmpl spells out:
# SSL_CERT_FILE replaces the default cert file list but not the default cert
# DIRECTORY scan, so SSL_CERT_DIR has to point at the projection too for the
# anchor set to be exactly the gateway CA.
env:
- name: SSL_CERT_FILE
value: /run/ate/trust-bundle.pem
- name: SSL_CERT_DIR
value: /run/ate
volumeMounts:
- name: system-info
mountPath: /run/ate # the bundle lands at /run/ate/trust-bundle.pem
readyz:
httpGet:
path: /readyz
port: 80
# See counter-microvm.yaml.tmpl: a micro-VM guest needs more than
# readyz's 30s default once CI's single kind node is under load.
timeoutSeconds: 120
# Sandbox size: without these the guest boots at the kata config's default
# (2GiB). ateom applies them to the VM (see internal/sizing).
resources:
limits:
cpu: "1"
memory: 512Mi
workerSelector:
matchLabels:
workload: egress-microvm-mitm
snapshotsConfig:
onPause: Full
onCommit: Full
location: gs://${BUCKET_NAME}/ate-demo-egress-microvm-mitm/
# No template.resources, unlike counter-substrate-microvm: an unlimited
# ateom container reports no capacity, which the scheduler reads as
# unconstrained, and the pod requests nothing so it stays placeable next to
# the counter pool's reservation on CI's single kind node. What actually
# bounds the memory here is the ActorTemplate's resources, which size the
# guest.
@@ -0,0 +1,58 @@
# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# Substrate ActorTemplate resource (protojson-shaped ateapipb.ActorTemplate,
# NOT the ate.dev/v1alpha1 CRD) for the micro-VM egress demo. The actor's
# outbound TCP is redirected into atunnel by nftables in the ateom
# (prepareActorEgress), which the micro-VM ateom implements the same way the
# gVisor one does — this fixture is what proves it end to end.
#
# Kept in step with egress-template.yaml.tmpl; the deltas are the
# runtime fields and the sandbox size.
metadata:
atespace: ate-demo-egress-microvm
name: egress-microvm
workerSelector:
matchLabels:
workload: egress-microvm
containers:
- name: egress
image: ko://github.com/agent-substrate/substrate/demos/egress
command: ["/ko-app/egress"]
readyz:
httpGet:
path: /readyz
port: 80
# See counter-substrate-microvm-template.yaml.tmpl: a micro-VM guest needs
# more than readyz's 30s default once CI's single kind node is under load.
timeoutSeconds: 120
# Sandbox size: without these the guest boots at the kata config's default
# (2GiB). ateom applies them to the VM (see internal/sizing). Quantities are
# strings in the proto.
resources:
limits:
- name: cpu
quantity: "1"
- name: memory
quantity: 512Mi
sandboxConfig:
sandboxClass: SANDBOX_CLASS_MICROVM
# Deliberately not the class default; installed cluster-wide by
# hack/install-microvm-deps.sh, so a missing or stale install fails loudly.
configName: microvm
snapshotsConfig:
onPause: SNAPSHOT_CONTENT_SCOPE_FULL
onCommit: SNAPSHOT_CONTENT_SCOPE_FULL
storageLocation: gs://${BUCKET_NAME}/ate-demo-egress-microvm/
+13 -51
View File
@@ -12,15 +12,13 @@
# See the License for the specific language governing permissions and
# limitations under the License.
# Micro-VM (kata + cloud-hypervisor) variant of the egress demo, so the
# networking suite's egress tests run against both sandbox classes. The actor's
# outbound TCP is redirected into atunnel by nftables in the ateom
# (prepareActorEgress), which the micro-VM ateom implements the same way the
# gVisor one does — this fixture is what proves it end to end.
#
# Kept in step with egress.yaml.tmpl; the deltas are the runtime fields and the
# sandbox size. The cluster-wide `microvm` SandboxConfig referenced below by
# name is installed by hack/install-microvm-deps.sh.
# Worker pool for the micro-VM (kata + cloud-hypervisor) variant of the
# egress demo, so the networking suite's egress tests run against both
# sandbox classes. The ActorTemplate is a substrate resource, not a CRD: it
# lives in egress-microvm-template.yaml.tmpl and is created through the
# ate API with `kubectl ate create actor-template`. The cluster-wide `microvm`
# SandboxConfig referenced by the pool is installed by
# hack/install-microvm-deps.sh
apiVersion: v1
kind: Namespace
@@ -41,45 +39,9 @@ spec:
sandboxClass: microvm
sandboxConfigName: microvm
workerImage: ko://github.com/agent-substrate/substrate/cmd/ateom-microvm
# No template.resources, unlike counter-microvm: an unlimited ateom container
# reports no capacity, which the scheduler reads as unconstrained, and the pod
# requests nothing so it stays placeable next to the counter-microvm pool's
# reservation on CI's single kind node. What actually bounds the memory here
# is the ActorTemplate's resources below, which size the guest.
---
apiVersion: ate.dev/v1alpha1
kind: ActorTemplate
metadata:
name: egress-microvm
namespace: ate-demo-egress-microvm
spec:
# Must match the WorkerPool's sandboxClass: a snapshot is not portable across
# sandbox classes, so only pools of the same class are eligible to run this
# template's actors.
sandboxClass: microvm
containers:
- name: egress
image: ko://github.com/agent-substrate/substrate/demos/egress
command: ["/ko-app/egress"]
readyz:
httpGet:
path: /readyz
port: 80
# See counter-microvm.yaml.tmpl: a micro-VM guest needs more than
# readyz's 30s default once CI's single kind node is under load.
timeoutSeconds: 120
# Sandbox size: without these the guest boots at the kata config's default
# (2GiB). ateom applies them to the VM (see internal/sizing).
resources:
limits:
cpu: "1"
memory: 512Mi
workerSelector:
matchLabels:
workload: egress-microvm
snapshotsConfig:
onPause: Full
onCommit: Full
location: gs://${BUCKET_NAME}/ate-demo-egress-microvm/
# No template.resources, unlike counter-substrate-microvm: an unlimited
# ateom container reports no capacity, which the scheduler reads as
# unconstrained, and the pod requests nothing so it stays placeable next to
# the counter pool's reservation on CI's single kind node. What actually
# bounds the memory here is the ActorTemplate's resources, which size the
# guest.
@@ -0,0 +1,74 @@
# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# MITM variant of the egress demo template: the same workload as
# egress-template.yaml.tmpl, but trusting only the egress gateway
# CA. The networking suite builds its egress actors from this fixture instead
# of the plain one when E2E_EGRESS_MITM is set, because under an sdsmint
# gateway TestActorEgressHTTPS's origin certificate is a per-SNI leaf the
# gateway minted, which chains to no public CA.
#
# Kept in step with egress-template.yaml.tmpl; the deltas are the
# names, the projected trust bundle, and the two SSL_CERT_* variables.
metadata:
atespace: ate-demo-egress-mitm
name: egress-mitm
workerSelector:
matchLabels:
workload: egress-mitm
containers:
- name: egress
image: ko://github.com/agent-substrate/substrate/demos/egress
command: ["/ko-app/egress"]
# The demo fetches with a stock net/http client, so its roots come from
# crypto/x509's system pool — which on Unix reads both of these.
#
# SSL_CERT_FILE alone is not enough to make the projected bundle the whole
# story: it replaces the default cert FILE list, but the default cert
# DIRECTORY list is still scanned, and the base image keeps its public
# roots in /etc/ssl/certs. Pointing SSL_CERT_DIR at the projection too
# makes the anchor set exactly the gateway CA, so a successful HTTPS fetch
# proves the projected bundle validated the minted leaf. Under sdsmint the
# public roots are useless anyway: every origin is fronted by the gateway.
env:
- name: SSL_CERT_FILE
value: /run/ate/trust-bundle.pem
- name: SSL_CERT_DIR
value: /run/ate
volumeMounts:
# the bundle lands at /run/ate/trust-bundle.pem
- name: system-info
mountPath: /run/ate
readyz:
httpGet:
path: /readyz
port: 80
volumes:
- name: system-info
type: SystemInfo
systemInfo:
dataSources:
# The trust anchors for the per-SNI leaves the MITM egress gateway mints.
# Only allowlisted bundle names resolve; this is the one supported today.
- trustBundle:
name: egress-mitm.ate.dev
path: trust-bundle.pem
sandboxConfig:
sandboxClass: SANDBOX_CLASS_GVISOR
configName: gvisor-default
snapshotsConfig:
onPause: SNAPSHOT_CONTENT_SCOPE_FULL
onCommit: SNAPSHOT_CONTENT_SCOPE_FULL
storageLocation: gs://${BUCKET_NAME}/ate-demo-egress-mitm/
+5 -62
View File
@@ -12,17 +12,11 @@
# See the License for the specific language governing permissions and
# limitations under the License.
# MITM variant of the egress demo: the same workload as egress.yaml.tmpl, but
# trusting only the egress gateway CA. The networking suite builds its egress
# actors from this fixture instead of the plain one when E2E_EGRESS_MITM is set,
# because under an sdsmint gateway TestActorEgressHTTPS's origin certificate is
# a per-SNI leaf the gateway minted, which chains to no public CA.
#
# Kept as its own fixture rather than a flag on the egress demo: it REQUIRES an
# sdsmint install (--experimental-use-sdsmint).
#
# Kept in step with egress.yaml.tmpl; the deltas are the names, the projected
# trust bundle, and the two SSL_CERT_* variables.
# Worker pool for the MITM variant of the egress demo: the same
# workload as egress.yaml.tmpl, but its template (in
# egress-mitm-template.yaml.tmpl) trusts only the egress gateway CA.
# Kept as its own fixture rather than a flag on the egress demo: it REQUIRES
# an sdsmint install (--experimental-use-sdsmint).
apiVersion: v1
kind: Namespace
@@ -41,54 +35,3 @@ metadata:
spec:
replicas: 2
workerImage: ko://github.com/agent-substrate/substrate/cmd/ateom-gvisor
---
apiVersion: ate.dev/v1alpha1
kind: ActorTemplate
metadata:
name: egress-mitm
namespace: ate-demo-egress-mitm
spec:
volumes:
- name: system-info
systemInfo:
dataSources:
# The trust anchors for the per-SNI leaves the MITM egress gateway mints.
# Only allowlisted bundle names resolve; this is the one supported today.
- trustBundle:
name: egress-mitm.ate.dev
path: trust-bundle.pem
containers:
- name: egress
image: ko://github.com/agent-substrate/substrate/demos/egress
command: ["/ko-app/egress"]
# The demo fetches with a stock net/http client, so its roots come from
# crypto/x509's system pool — which on Unix reads both of these.
#
# SSL_CERT_FILE alone is not enough to make the projected bundle the whole
# story: it replaces the default cert FILE list, but the default cert
# DIRECTORY list is still scanned, and the base image keeps its public
# roots in /etc/ssl/certs. Pointing SSL_CERT_DIR at the projection too
# makes the anchor set exactly the gateway CA, so a successful HTTPS fetch
# proves the projected bundle validated the minted leaf. Under sdsmint the
# public roots are useless anyway: every origin is fronted by the gateway.
env:
- name: SSL_CERT_FILE
value: /run/ate/trust-bundle.pem
- name: SSL_CERT_DIR
value: /run/ate
volumeMounts:
- name: system-info
mountPath: /run/ate # the bundle lands at /run/ate/trust-bundle.pem
readyz:
httpGet:
path: /readyz
port: 80
workerSelector:
matchLabels:
workload: egress-mitm
snapshotsConfig:
onPause: Full
onCommit: Full
location: gs://${BUCKET_NAME}/ate-demo-egress-mitm/
+40
View File
@@ -0,0 +1,40 @@
# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# Substrate ActorTemplate resource (protojson-shaped ateapipb.ActorTemplate,
# NOT the ate.dev/v1alpha1 CRD) for the egress demo. Applied with
# `kubectl ate create actor-template -f -` after `ko resolve` replaces the
# ko:// image reference; the ate-demo-egress atespace must exist.
metadata:
atespace: ate-demo-egress
name: egress
workerSelector:
matchLabels:
workload: egress
containers:
- name: egress
image: ko://github.com/agent-substrate/substrate/demos/egress
command: ["/ko-app/egress"]
readyz:
httpGet:
path: /readyz
port: 80
sandboxConfig:
sandboxClass: SANDBOX_CLASS_GVISOR
configName: gvisor-default
snapshotsConfig:
onPause: SNAPSHOT_CONTENT_SCOPE_FULL
onCommit: SNAPSHOT_CONTENT_SCOPE_FULL
storageLocation: gs://${BUCKET_NAME}/ate-demo-egress/
+4 -24
View File
@@ -12,6 +12,10 @@
# See the License for the specific language governing permissions and
# limitations under the License.
# Worker pool for the egress demo. The ActorTemplate is a substrate resource,
# not a CRD: it lives in egress-template.yaml.tmpl and is created through the
# ate API with `kubectl ate create actor-template`.
apiVersion: v1
kind: Namespace
metadata:
@@ -29,27 +33,3 @@ metadata:
spec:
replicas: 2
workerImage: ko://github.com/agent-substrate/substrate/cmd/ateom-gvisor
---
apiVersion: ate.dev/v1alpha1
kind: ActorTemplate
metadata:
name: egress
namespace: ate-demo-egress
spec:
containers:
- name: egress
image: ko://github.com/agent-substrate/substrate/demos/egress
command: ["/ko-app/egress"]
readyz:
httpGet:
path: /readyz
port: 80
workerSelector:
matchLabels:
workload: egress
snapshotsConfig:
onPause: Full
onCommit: Full
location: gs://${BUCKET_NAME}/ate-demo-egress/
+5 -3
View File
@@ -33,9 +33,11 @@
set -o errexit -o nounset -o pipefail
CTX="${KUBECTL_CONTEXT:-kind-kind}"
ATESPACE="${ATESPACE:-demo}"
# The actor lives in the demo's atespace: --template-ref resolves the
# template by name within the actor's own atespace.
ATESPACE="${ATESPACE:-ate-demo-egress}"
ACTOR="${ACTOR:-egress-demo}"
TEMPLATE="${TEMPLATE:-ate-demo-egress/egress}"
TEMPLATE="${TEMPLATE:-egress}"
TARGET_NS="${TARGET_NS:-egress-target}"
PROBE_POD="egress-identity-probe"
@@ -78,7 +80,7 @@ info "target ClusterIP = ${TARGET_IP}"
log "create + resume Actor ${ATESPACE}/${ACTOR}"
${KATE} create atespace "${ATESPACE}" >/dev/null 2>&1 || true
${KATE} create actor "${ACTOR}" -a "${ATESPACE}" --template "${TEMPLATE}" >/dev/null 2>&1 || true
${KATE} create actor "${ACTOR}" -a "${ATESPACE}" --template-ref "${TEMPLATE}" >/dev/null 2>&1 || true
${KATE} resume actor "${ACTOR}" -a "${ATESPACE}" >/dev/null 2>&1 || true
for _ in $(seq 1 30); do
${KATE} get actors -a "${ATESPACE}" 2>/dev/null | grep -q "ACTOR_STATE_RUNNING" && break
+96
View File
@@ -878,6 +878,102 @@ wait_actortemplate_ready() {
return 1
}
# deploy_substrate_demo deploys a demo whose ActorTemplate is a substrate
# resource: apply the pool manifest, wait for the pool rollout, create the
# template through the ate API, and block on its golden snapshot.
# deploy_substrate_demo <demo> <pool_manifest> <template_manifest> \
# <atespace> <pool> <template> <golden_timeout> [extra sed exprs...]
# The atespace doubles as the pool's k8s namespace, keeping the substrate
# naming parallel to the CRD-era namespace/template pairs. An empty <pool>
# skips the rollout wait; a golden_timeout of 0 creates the template without
# blocking on its golden snapshot. Extra sed expressions are applied to the
# template manifest, for demos with optional template stanzas.
deploy_substrate_demo() {
local demo="$1" pool_manifest="$2" template_manifest="$3"
local atespace="$4" pool="$5" template="$6" golden_timeout="${7:-300}"
shift 7
log_step "${demo}_deploy (${atespace}/${template})"
ensure_crds
sed -e "s|\${BUCKET_NAME}|${BUCKET_NAME}|g" "${pool_manifest}" \
| run_ko apply -f -
if [[ -n "${pool}" ]]; then
log_step "Waiting for the ${pool} worker pool rollout..."
wait_for_pool_rollout_fatal "${pool}" "${atespace}"
fi
create_substrate_template "${template_manifest}" "${atespace}" "${template}" "$@"
if [[ "${golden_timeout}" != "0" ]]; then
# Mirrors the CRD era's `kubectl wait --for=condition=Ready
# actortemplate/...` (there is no kubectl wait for substrate resources).
log_step "Waiting for the ${atespace}/${template} golden snapshot..."
if ! wait_actortemplate_ready "${atespace}" "${template}" "${golden_timeout}"; then
exit 1
fi
fi
}
# create_substrate_template renders a protojson ActorTemplate manifest and
# creates it through the ate API:
# create_substrate_template <manifest> <atespace> <template> [extra sed exprs...]
create_substrate_template() {
local template_manifest="$1" atespace="$2" template="$3"
shift 3
# The store enforces that the template's atespace exists at create time.
if ! run_kubectl_ate create atespace "${atespace}" >/dev/null 2>&1 \
&& ! run_kubectl_ate get atespace "${atespace}" >/dev/null 2>&1; then
echo "error: failed to create atespace ${atespace}" >&2
exit 1
fi
# ko resolve builds the ko:// image references and replaces them with pushed
# digests before the manifest reaches kubectl-ate. Actor templates are
# immutable (no update RPC), so an existing template is left in place:
# delete the demo and redeploy to change it.
if ! sed -e "s|\${BUCKET_NAME}|${BUCKET_NAME}|g" "$@" "${template_manifest}" \
| run_ko resolve -f - \
| run_kubectl_ate create actor-template -f -; then
if run_kubectl_ate get actor-template "${template}" -a "${atespace}" >/dev/null 2>&1; then
log_step "actor template ${atespace}/${template} already exists; keeping it (delete the demo to replace it)"
else
echo "error: failed to create actor template ${atespace}/${template}" >&2
exit 1
fi
fi
}
# delete_substrate_templates removes a demo's actors, its templates, and then
# their shared atespace:
# delete_substrate_templates <atespace> <template...>
delete_substrate_templates() {
local atespace="$1"
shift
local template
for template in "$@"; do
delete_demo_actors_substrate "${atespace}" "${template}"
# Also removes the template's golden actor and golden snapshot server-side.
run_kubectl_ate delete actor-template "${template}" -a "${atespace}" 2>/dev/null \
|| log_step "actor template ${atespace}/${template} not deleted (may not exist)"
done
run_kubectl_ate delete atespace "${atespace}" 2>/dev/null \
|| log_step "atespace ${atespace} not deleted (may not exist or is not empty)"
}
# delete_substrate_demo tears down one substrate demo: its actors, templates,
# atespace, and pool manifest.
# delete_substrate_demo <demo> <pool_manifest> <atespace> <template...>
delete_substrate_demo() {
local demo="$1" pool_manifest="$2" atespace="$3"
shift 3
log_step "${demo}_delete (${atespace})"
delete_substrate_templates "${atespace}" "$@"
sed -e "s|\${BUCKET_NAME}|${BUCKET_NAME}|g" "${pool_manifest}" \
| run_kubectl delete --ignore-not-found -f -
}
delete_ate_system() {
log_step "delete_ate_system"
if [[ "${ATE_INSTALL_KIND:-false}" == "true" ]]; then
+40 -60
View File
@@ -15,6 +15,14 @@
# limitations under the License.
#
# This is sourced as part of install-ate.sh. Do not run directly.
#
# The egress demo: each variant applies a worker pool manifest, then creates
# its ActorTemplate as a substrate resource through the ate API with
# `kubectl ate create actor-template`. The micro-VM variants need the
# cluster-wide `microvm` SandboxConfig from hack/install-microvm-deps.sh
# --install; the MITM variants need an sdsmint install
# (--experimental-use-sdsmint), because their actors project the egress
# gateway trust bundle, which does not resolve otherwise.
ATE_DEMOS+=(demo-egress) # register demo-egress
# The micro-VM variant is its own demo rather than a flag on demo-egress: that
@@ -74,24 +82,16 @@ demo-egress-microvm-mitm_cmdline() {
}
demo-egress_deploy() {
log_step "demo-egress_deploy"
ensure_crds
sed "s|\${BUCKET_NAME}|${BUCKET_NAME}|g" demos/egress/egress.yaml.tmpl \
| run_ko apply -f -
log_step "Waiting for egress demo to be ready..."
# The WorkerPool controller names the Deployment after the WorkerPool
# ("egress"), the same way demo-counter gets "deployment/counter". The old
# "egress-deployment" name was NotFound on every successful deploy.
wait_for_pool_rollout egress ate-demo-egress
run_kubectl wait --for=condition=Ready actortemplate/egress -n ate-demo-egress --timeout=300s
deploy_substrate_demo demo-egress \
demos/egress/egress.yaml.tmpl \
demos/egress/egress-template.yaml.tmpl \
ate-demo-egress egress egress 300
}
demo-egress_delete() {
log_step "demo-egress_delete"
delete_demo_actors ate-demo-egress egress
sed "s|\${BUCKET_NAME}|${BUCKET_NAME}|g" demos/egress/egress.yaml.tmpl \
| run_kubectl delete --ignore-not-found -f -
delete_substrate_demo demo-egress \
demos/egress/egress.yaml.tmpl \
ate-demo-egress egress
}
demo-egress-microvm_usage() {
@@ -99,24 +99,18 @@ demo-egress-microvm_usage() {
}
demo-egress-microvm_deploy() {
log_step "demo-egress-microvm_deploy"
ensure_crds
sed "s|\${BUCKET_NAME}|${BUCKET_NAME}|g" demos/egress/egress-microvm.yaml.tmpl \
| run_ko apply -f -
log_step "Waiting for micro-VM egress demo to be ready..."
wait_for_pool_rollout egress-microvm ate-demo-egress-microvm
# A micro-VM golden is a cloud-hypervisor cold boot plus a checkpoint, on
# nested KVM in CI, so it needs a longer budget than the gVisor one above.
run_kubectl wait --for=condition=Ready actortemplate/egress-microvm \
-n ate-demo-egress-microvm --timeout=600s
# 600s golden budget: a micro-VM golden is a cloud-hypervisor cold boot
# plus checkpoint, on nested KVM in CI.
deploy_substrate_demo demo-egress-microvm \
demos/egress/egress-microvm.yaml.tmpl \
demos/egress/egress-microvm-template.yaml.tmpl \
ate-demo-egress-microvm egress-microvm egress-microvm 600
}
demo-egress-microvm_delete() {
log_step "demo-egress-microvm_delete"
delete_demo_actors ate-demo-egress-microvm egress-microvm
sed "s|\${BUCKET_NAME}|${BUCKET_NAME}|g" demos/egress/egress-microvm.yaml.tmpl \
| run_kubectl delete --ignore-not-found -f -
delete_substrate_demo demo-egress-microvm \
demos/egress/egress-microvm.yaml.tmpl \
ate-demo-egress-microvm egress-microvm
}
demo-egress-mitm_usage() {
@@ -125,25 +119,19 @@ demo-egress-mitm_usage() {
}
demo-egress-mitm_deploy() {
log_step "demo-egress-mitm_deploy"
ensure_crds
sed "s|\${BUCKET_NAME}|${BUCKET_NAME}|g" demos/egress/egress-mitm.yaml.tmpl \
| run_ko apply -f -
log_step "Waiting for MITM egress demo to be ready..."
wait_for_pool_rollout egress-mitm ate-demo-egress-mitm
# The golden snapshot only becomes Ready once an actor starts, and an actor
# whose trust bundle does not resolve never does — so a timeout here is the
# The golden snapshot only exists once an actor starts, and an actor whose
# trust bundle does not resolve never does — so a timeout here is the
# symptom of a missing sdsmint install (see demo-egress-mitm_usage).
run_kubectl wait --for=condition=Ready actortemplate/egress-mitm \
-n ate-demo-egress-mitm --timeout=300s
deploy_substrate_demo demo-egress-mitm \
demos/egress/egress-mitm.yaml.tmpl \
demos/egress/egress-mitm-template.yaml.tmpl \
ate-demo-egress-mitm egress-mitm egress-mitm 300
}
demo-egress-mitm_delete() {
log_step "demo-egress-mitm_delete"
delete_demo_actors ate-demo-egress-mitm egress-mitm
sed "s|\${BUCKET_NAME}|${BUCKET_NAME}|g" demos/egress/egress-mitm.yaml.tmpl \
| run_kubectl delete --ignore-not-found -f -
delete_substrate_demo demo-egress-mitm \
demos/egress/egress-mitm.yaml.tmpl \
ate-demo-egress-mitm egress-mitm
}
demo-egress-microvm-mitm_usage() {
@@ -152,22 +140,14 @@ demo-egress-microvm-mitm_usage() {
}
demo-egress-microvm-mitm_deploy() {
log_step "demo-egress-microvm-mitm_deploy"
ensure_crds
sed "s|\${BUCKET_NAME}|${BUCKET_NAME}|g" demos/egress/egress-microvm-mitm.yaml.tmpl \
| run_ko apply -f -
log_step "Waiting for micro-VM MITM egress demo to be ready..."
wait_for_pool_rollout egress-microvm-mitm ate-demo-egress-microvm-mitm
# A micro-VM golden is a cloud-hypervisor cold boot plus a checkpoint, on
# nested KVM in CI, so it needs a longer budget than the gVisor one above.
run_kubectl wait --for=condition=Ready actortemplate/egress-microvm-mitm \
-n ate-demo-egress-microvm-mitm --timeout=600s
deploy_substrate_demo demo-egress-microvm-mitm \
demos/egress/egress-microvm-mitm.yaml.tmpl \
demos/egress/egress-microvm-mitm-template.yaml.tmpl \
ate-demo-egress-microvm-mitm egress-microvm-mitm egress-microvm-mitm 600
}
demo-egress-microvm-mitm_delete() {
log_step "demo-egress-microvm-mitm_delete"
delete_demo_actors ate-demo-egress-microvm-mitm egress-microvm-mitm
sed "s|\${BUCKET_NAME}|${BUCKET_NAME}|g" demos/egress/egress-microvm-mitm.yaml.tmpl \
| run_kubectl delete --ignore-not-found -f -
delete_substrate_demo demo-egress-microvm-mitm \
demos/egress/egress-microvm-mitm.yaml.tmpl \
ate-demo-egress-microvm-mitm egress-microvm-mitm
}
+7 -3
View File
@@ -29,7 +29,9 @@ ROOT="$(git rev-parse --show-toplevel)"; cd "${ROOT}"
CTX="${KUBECTL_CONTEXT:-kind-kind}"
K="kubectl --context ${CTX}"
ATESPACE="${ATESPACE:-demo}"
# The actor lives in the demo's atespace: --template-ref resolves the
# template by name within the actor's own atespace.
ATESPACE="${ATESPACE:-ate-demo-egress}"
ACTOR="${ACTOR:-egress-demo}"
TARGET_URL="${TARGET_URL:-http://example.com/}"
@@ -39,8 +41,10 @@ ${K} -n ate-system rollout status deployment/atenet-egress --timeout=120s
echo "== create atespace + actor =="
kubectl-ate --context "${CTX}" create atespace "${ATESPACE}" 2>/dev/null || true
kubectl-ate --context "${CTX}" create actor "${ACTOR}" \
--atespace "${ATESPACE}" --template ate-demo-egress/egress 2>/dev/null || true
${K} -n ate-system wait --for=condition=Ready "actor/${ACTOR}" 2>/dev/null || sleep 10
--atespace "${ATESPACE}" --template-ref egress 2>/dev/null || true
# The router resumes the actor on demand; give the control plane a beat to
# register it before driving traffic.
sleep 10
echo "== snapshot gateway authentication log offset =="
BEFORE=$(${K} -n ate-system logs deployment/atenet-egress -c ext-proc --tail=-1 2>/dev/null | wc -l | tr -d ' ')
+193
View File
@@ -0,0 +1,193 @@
// Copyright 2026 Google LLC
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
package e2e
import (
"context"
"strings"
"testing"
"time"
"github.com/agent-substrate/substrate/pkg/proto/ateapipb"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
"google.golang.org/protobuf/encoding/protojson"
"sigs.k8s.io/yaml"
)
// substrateTemplateSubstitutions is the placeholder set the protojson-shaped
// ActorTemplate fixture templates carry: the substrate counterpart of
// fixtureSubstitutions, whose block fragments are CRD-shaped and now serve
// only the pool manifests. The inline placeholders match the pool manifests'
// values so one (bucket, name) pair renders both halves of a fixture.
func substrateTemplateSubstitutions(bucket, name string, trustBundle bool) (inline, blocks map[string]string) {
inline = map[string]string{
"${BUCKET_NAME}": bucket,
"${FIXTURE_SUFFIX}": "-" + name,
}
blocks = map[string]string{
// gvisor-default is the cluster-wide default SandboxConfig
// manifests/ate-install ships; config_name is required, so the
// templates name it explicitly even though the gVisor WorkerPools
// leave sandboxConfigName empty and resolve to the same object.
"${TEMPLATE_SANDBOX_CONFIG}": "sandboxConfig:\n sandboxClass: SANDBOX_CLASS_GVISOR\n configName: gvisor-default",
"${TEMPLATE_RESOURCES}": "",
// Off unless the caller opts in; see WithTrustBundle.
"${TEMPLATE_TRUST_BUNDLE}": "",
}
if trustBundle {
// Indented to sit in a template's systemInfo dataSources list. The
// name must be on atelet's supported-bundle allowlist.
blocks["${TEMPLATE_TRUST_BUNDLE}"] = " - trustBundle:\n name: egress-mitm.ate.dev\n path: trust-bundle.pem"
}
if !IsMicroVM() {
return inline, blocks
}
inline["${FIXTURE_SUFFIX}"] = "-" + SandboxClassMicroVM + "-" + name
// The cluster-wide SandboxConfig hack/install-microvm-deps.sh installs.
// It is deliberately not the class default, so a missing or stale one
// fails loudly.
blocks["${TEMPLATE_SANDBOX_CONFIG}"] = "sandboxConfig:\n sandboxClass: SANDBOX_CLASS_MICROVM\n configName: microvm"
// Only for fixtures that declare no limits of their own. Without them the
// guest boots at the kata config's default (2GiB), and several of those
// do not fit beside the demo pools on CI's single kind node. These size
// the VM itself — see internal/sizing. Quantities are strings.
blocks["${TEMPLATE_RESOURCES}"] = "resources:\n limits:\n - name: cpu\n quantity: \"1\"\n - name: memory\n quantity: 512Mi"
return inline, blocks
}
// decodeSubstrateTemplates strict-decodes a rendered manifest of one or more
// ---separated protojson-shaped ActorTemplate documents. Strict, so a block
// placeholder landing at the wrong depth or a misspelled field fails here
// rather than applying cleanly and doing nothing — the same contract
// `kubectl ate create actor-template` enforces.
func decodeSubstrateTemplates(t *testing.T, rendered []byte) []*ateapipb.ActorTemplate {
t.Helper()
var templates []*ateapipb.ActorTemplate
for doc := range strings.SplitSeq(string(rendered), "\n---") {
jsonData, err := yaml.YAMLToJSON([]byte(doc))
if err != nil {
t.Fatalf("invalid YAML in ActorTemplate manifest: %v", err)
}
if string(jsonData) == "null" {
continue
}
tmpl := &ateapipb.ActorTemplate{}
if err := protojson.Unmarshal(jsonData, tmpl); err != nil {
t.Fatalf("decoding protojson ActorTemplate: %v", err)
}
templates = append(templates, tmpl)
}
if len(templates) == 0 {
t.Fatalf("ActorTemplate manifest holds no documents")
}
return templates
}
// SubstrateFixtureManifests names the two manifest templates a substrate
// fixture is built from (both repo-relative, under internal/e2e/fixtures).
type SubstrateFixtureManifests struct {
// Pool declares the k8s side: a Namespace plus the WorkerPool CRD.
Pool string
// Template declares one or more protojson-shaped ActorTemplate
// documents, separated by ---.
Template string
}
// DeploySubstrateFixture installs a fixture for the sandbox class under test:
// it ko-applies the pool manifest, creates the fixture's atespace and
// ActorTemplates through the ate API (ko-resolving the templates' ko:// image
// references first), and blocks until every template's golden snapshot
// exists. name distinguishes the caller (by convention its suite name): each
// suite gets its own copy of the fixture, so no suite's cleanup can delete it
// out from under another running concurrently.
//
// Everything is removed when the test ends. The substrate resources need
// explicit cleanup — unlike the CRD templates they replaced, they do not ride
// the k8s namespace GC — and a template leaked by an interrupted earlier run
// is cleared before creating its replacement, since templates are immutable.
//
// Returns the fixture's atespace (the same string that names the k8s
// namespace holding the pool) and the created templates.
func DeploySubstrateFixture(t *testing.T, ctx context.Context, clients *Clients, manifests SubstrateFixtureManifests, bucket, name string, trustBundle bool) (string, []*ateapipb.ActorTemplate) {
t.Helper()
// The pool manifest also carries the namespace. Its cleanup is registered
// first so it runs last (t.Cleanup is LIFO), after the templates that
// select the pool's workers are gone.
inline, blocks := fixtureSubstitutions(bucket, name)
poolManifest := renderManifest(t, manifests.Pool, inline, blocks)
koApply(t, poolManifest)
t.Cleanup(func() {
delArgs := []string{"delete", "--ignore-not-found", "-f", poolManifest}
if KubeContext != "" {
delArgs = append([]string{"--context=" + KubeContext}, delArgs...)
}
RunCmd(t, "kubectl", delArgs...)
})
tmplInline, tmplBlocks := substrateTemplateSubstitutions(bucket, name, trustBundle)
rendered := koResolve(t, renderManifest(t, manifests.Template, tmplInline, tmplBlocks))
templates := decodeSubstrateTemplates(t, rendered)
atespace := templates[0].GetMetadata().GetAtespace()
for _, tmpl := range templates {
if got := tmpl.GetMetadata().GetAtespace(); got != atespace {
t.Fatalf("fixture %s declares templates in different atespaces (%q and %q)", manifests.Template, atespace, got)
}
}
if _, err := clients.SubstrateAPI.CreateAtespace(ctx, &ateapipb.CreateAtespaceRequest{Atespace: &ateapipb.Atespace{Metadata: &ateapipb.ResourceMetadata{Name: atespace}}}); err != nil && status.Code(err) != codes.AlreadyExists {
t.Fatalf("failed to create atespace %q: %v", atespace, err)
}
t.Cleanup(func() {
// Best-effort: a leaked actor keeps the atespace non-empty, and the
// next run tolerates AlreadyExists anyway.
cleanupCtx, cancel := context.WithTimeout(context.Background(), time.Minute)
defer cancel()
if _, err := clients.SubstrateAPI.DeleteAtespace(cleanupCtx, &ateapipb.DeleteAtespaceRequest{Atespace: &ateapipb.ObjectRef{Name: atespace}}); err != nil && status.Code(err) != codes.NotFound {
t.Logf("failed to delete atespace %q: %v", atespace, err)
}
})
created := make([]*ateapipb.ActorTemplate, 0, len(templates))
for _, tmpl := range templates {
ref := &ateapipb.ObjectRef{Atespace: atespace, Name: tmpl.GetMetadata().GetName()}
if _, err := clients.SubstrateAPI.DeleteActorTemplate(ctx, &ateapipb.DeleteActorTemplateRequest{ActorTemplate: ref}); err != nil && status.Code(err) != codes.NotFound {
t.Fatalf("failed to clear leaked ActorTemplate %s/%s: %v", atespace, ref.GetName(), err)
}
c, err := clients.SubstrateAPI.CreateActorTemplate(ctx, &ateapipb.CreateActorTemplateRequest{ActorTemplate: tmpl})
if err != nil {
t.Fatalf("failed to create ActorTemplate %s/%s: %v", atespace, tmpl.GetMetadata().GetName(), err)
}
// Registered before the golden wait so a template whose golden never
// builds still gets cleaned up.
t.Cleanup(func() {
cleanupCtx, cancel := context.WithTimeout(context.Background(), time.Minute)
defer cancel()
if _, err := clients.SubstrateAPI.DeleteActorTemplate(cleanupCtx, &ateapipb.DeleteActorTemplateRequest{ActorTemplate: ref}); err != nil && status.Code(err) != codes.NotFound {
t.Logf("failed to delete ActorTemplate %s/%s: %v", atespace, ref.GetName(), err)
}
})
created = append(created, c)
}
for _, tmpl := range created {
t.Logf("Waiting for ActorTemplate %s/%s golden snapshot...", atespace, tmpl.GetMetadata().GetName())
WaitForSubstrateTemplateReady(ctx, t, clients, atespace, tmpl.GetMetadata().GetName())
}
return atespace, created
}
@@ -0,0 +1,74 @@
# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# The capabilities fixture's ActorTemplates: protojson-shaped
# ateapipb.ActorTemplate documents (NOT the ate.dev/v1alpha1 CRD), created
# through the ate API by e2e.DeploySubstrateFixture. Two templates run the
# same probe image and differ only in securityContext.capabilities, so the
# suite can compare what the kernel reports inside the sandbox against what
# the template asked for.
# No securityContext: pins the default capability set, so a change to the
# default is caught here and not only in atelet's unit tests.
metadata:
atespace: ate-e2e${FIXTURE_SUFFIX}
name: caps-default
workerSelector:
matchLabels:
workload: caps
containers:
- name: probe
image: ko://github.com/agent-substrate/substrate/internal/e2e/fixtures/probe
command: ["/ko-app/probe"]
readyz:
httpGet:
path: /healthz
port: 80
timeoutSeconds: 60
${TEMPLATE_RESOURCES}
${TEMPLATE_SANDBOX_CONFIG}
snapshotsConfig:
storageLocation: gs://${BUCKET_NAME}/ate-e2e-caps-default${FIXTURE_SUFFIX}/
---
# drop ALL + add: the resulting set is exact rather than relative, so asserting
# it proves both halves at once — everything default was dropped, and only the
# named capability was granted back.
metadata:
atespace: ate-e2e${FIXTURE_SUFFIX}
name: caps-exact
workerSelector:
matchLabels:
workload: caps
containers:
- name: probe
image: ko://github.com/agent-substrate/substrate/internal/e2e/fixtures/probe
command: ["/ko-app/probe"]
# The probe binds :80, which is exactly what NET_BIND_SERVICE permits, so
# this template also proves the granted capability is usable and not merely
# present in the mask: without it the container could not become ready.
securityContext:
capabilities:
drop: ["ALL"]
add: ["NET_BIND_SERVICE"]
readyz:
httpGet:
path: /healthz
port: 80
timeoutSeconds: 60
${TEMPLATE_RESOURCES}
${TEMPLATE_SANDBOX_CONFIG}
snapshotsConfig:
storageLocation: gs://${BUCKET_NAME}/ate-e2e-caps-exact${FIXTURE_SUFFIX}/
@@ -12,10 +12,8 @@
# See the License for the specific language governing permissions and
# limitations under the License.
# Fixture for the capabilities e2e suite. Two templates run the same probe
# image and differ only in securityContext.capabilities, so the suite can
# compare what the kernel reports inside the sandbox against what the template
# asked for.
# The k8s half of the capabilities fixture: the namespace and the WorkerPool
# both templates in capabilities-templates.yaml.tmpl select.
#
# Rendered by e2e.RenderFixtureManifest, which fills the ${...} placeholders for
# the sandbox class under test: one manifest serves both runtimes so the gVisor
@@ -42,65 +40,3 @@ spec:
replicas: 4
workerImage: ${ATEOM_IMAGE}
${WORKERPOOL_RUNTIME}
---
# No securityContext: pins the default capability set, so a change to the
# default is caught here and not only in atelet's unit tests.
apiVersion: ate.dev/v1alpha1
kind: ActorTemplate
metadata:
name: caps-default
namespace: ate-e2e${FIXTURE_SUFFIX}
spec:
${TEMPLATE_SANDBOX_CLASS}
containers:
- name: probe
image: ko://github.com/agent-substrate/substrate/internal/e2e/fixtures/probe
command: ["/ko-app/probe"]
readyz:
httpGet:
path: /healthz
port: 80
timeoutSeconds: 60
${TEMPLATE_RESOURCES}
workerSelector:
matchLabels:
workload: caps
snapshotsConfig:
location: gs://${BUCKET_NAME}/ate-e2e-caps-default${FIXTURE_SUFFIX}/
---
# drop ALL + add: the resulting set is exact rather than relative, so asserting
# it proves both halves at once — everything default was dropped, and only the
# named capability was granted back.
apiVersion: ate.dev/v1alpha1
kind: ActorTemplate
metadata:
name: caps-exact
namespace: ate-e2e${FIXTURE_SUFFIX}
spec:
${TEMPLATE_SANDBOX_CLASS}
containers:
- name: probe
image: ko://github.com/agent-substrate/substrate/internal/e2e/fixtures/probe
command: ["/ko-app/probe"]
# The probe binds :80, which is exactly what NET_BIND_SERVICE permits, so
# this template also proves the granted capability is usable and not merely
# present in the mask: without it the container could not become ready.
securityContext:
capabilities:
drop: ["ALL"]
add: ["NET_BIND_SERVICE"]
readyz:
httpGet:
path: /healthz
port: 80
timeoutSeconds: 60
${TEMPLATE_RESOURCES}
workerSelector:
matchLabels:
workload: caps
snapshotsConfig:
location: gs://${BUCKET_NAME}/ate-e2e-caps-exact${FIXTURE_SUFFIX}/
@@ -0,0 +1,52 @@
# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# Sized variant of the probe template: a protojson-shaped
# ateapipb.ActorTemplate declaring its own resources.limits, so the sizing e2e
# suite can assert the actor's sandbox is shaped to those limits. Reuses the
# probe image (it serves the /resources endpoint). Unlike
# probe-template.yaml.tmpl this carries no ${TEMPLATE_RESOURCES} placeholder —
# the limits are the thing under test.
metadata:
atespace: ate-e2e${FIXTURE_SUFFIX}
name: probe-sized
workerSelector:
matchLabels:
workload: probe-sized
containers:
- name: probe
image: ko://github.com/agent-substrate/substrate/internal/e2e/fixtures/probe
command: ["/ko-app/probe"]
# Gate the golden snapshot on the probe actually serving, so the single
# (un-retried) GET /resources the suite makes after resuming cannot race a
# sandbox that was checkpointed before the listener was up.
readyz:
httpGet:
path: /healthz
port: 80
# The feature under test: these limits size the sandbox (and, when the worker
# advertises capacity, gate scheduling). CPU=2 makes NumCPU() inside the
# sandbox a distinct, assertable value — the vCPU count of the micro-VM guest,
# or what runsc --cpu-num-from-quota provisions the gVisor sentry with.
# Quantities are strings in the proto.
resources:
limits:
- name: cpu
quantity: "2"
- name: memory
quantity: 512Mi
${TEMPLATE_SANDBOX_CONFIG}
snapshotsConfig:
storageLocation: gs://${BUCKET_NAME}/ate-e2e${FIXTURE_SUFFIX}/
@@ -12,16 +12,12 @@
# See the License for the specific language governing permissions and
# limitations under the License.
# Sized variant of the probe fixture: the ActorTemplate declares
# spec.resources.limits, so the sizing e2e suite can assert the actor's
# sandbox is shaped to those limits. Reuses the probe image (it serves the
# /resources endpoint). Kept in its own namespace so it never collides with
# The k8s half of the sized probe fixture (see probe-sized-template.yaml.tmpl
# for the ActorTemplate). Kept in its own namespace so it never collides with
# the plain probe fixture.
#
# Rendered by e2e.RenderFixtureManifest, which fills the ${...} placeholders for
# the sandbox class under test. Unlike probe.yaml.tmpl this declares its own
# resources (they are the thing under test), so it carries no
# ${TEMPLATE_RESOURCES} placeholder.
# Rendered by e2e.RenderFixtureManifest, which fills the ${...} placeholders
# for the sandbox class under test.
apiVersion: v1
kind: Namespace
@@ -41,37 +37,3 @@ spec:
replicas: 3
workerImage: ${ATEOM_IMAGE}
${WORKERPOOL_RUNTIME}
---
apiVersion: ate.dev/v1alpha1
kind: ActorTemplate
metadata:
name: probe-sized
namespace: ate-e2e${FIXTURE_SUFFIX}
spec:
${TEMPLATE_SANDBOX_CLASS}
containers:
- name: probe
image: ko://github.com/agent-substrate/substrate/internal/e2e/fixtures/probe
command: ["/ko-app/probe"]
# Gate the golden snapshot on the probe actually serving, so the single
# (un-retried) GET /resources the suite makes after resuming cannot race a
# sandbox that was checkpointed before the listener was up.
readyz:
httpGet:
path: /healthz
port: 80
# The feature under test: these limits size the sandbox (and, when the worker
# advertises capacity, gate scheduling). CPU=2 makes NumCPU() inside the
# sandbox a distinct, assertable value — the vCPU count of the micro-VM guest,
# or what runsc --cpu-num-from-quota provisions the gVisor sentry with.
resources:
limits:
cpu: "2"
memory: 512Mi
workerSelector:
matchLabels:
workload: probe-sized
snapshotsConfig:
location: gs://${BUCKET_NAME}/ate-e2e${FIXTURE_SUFFIX}/
@@ -0,0 +1,63 @@
# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# The probe fixture's ActorTemplate: a protojson-shaped ateapipb.ActorTemplate
# (NOT the ate.dev/v1alpha1 CRD), created through the ate API by
# e2e.DeploySubstrateFixture after ko resolve pins the probe image. The
# ate-e2e-probe${FIXTURE_SUFFIX} atespace doubles as the pool's k8s namespace
# name in probe.yaml.tmpl.
metadata:
atespace: ate-e2e-probe${FIXTURE_SUFFIX}
name: probe
workerSelector:
matchLabels:
workload: probe${FIXTURE_SUFFIX}
containers:
- name: probe
image: ko://github.com/agent-substrate/substrate/internal/e2e/fixtures/probe
command: ["/ko-app/probe"]
volumeMounts:
# the probe reads actor metadata and trust bundles under /run/ate
- name: system-info
mountPath: /run/ate
# The probe binary binds :80 immediately, so this gates actor start on a
# readiness signal rather than a guess, and carries a non-default
# timeoutSeconds so e2e covers the value crossing ateapi -> atelet -> ateom
# instead of only the ateom's built-in default.
readyz:
httpGet:
path: /healthz
port: 80
timeoutSeconds: 60
volumes:
- name: system-info
type: SystemInfo
systemInfo:
dataSources:
- actorMetadata:
items:
- field: ACTOR_METADATA_FIELD_NAME
path: actor-id
- field: ACTOR_METADATA_FIELD_ATESPACE
path: atespace
- field: ACTOR_METADATA_FIELD_UID
path: actor-uid
# Only for suites that pass WithTrustBundle: the bundle is a cluster
# singleton, so a suite that merely needs a probe must not depend on it.
${TEMPLATE_TRUST_BUNDLE}
${TEMPLATE_RESOURCES}
${TEMPLATE_SANDBOX_CONFIG}
snapshotsConfig:
storageLocation: gs://${BUCKET_NAME}/ate-e2e-probe${FIXTURE_SUFFIX}/
+4 -47
View File
@@ -12,6 +12,10 @@
# See the License for the specific language governing permissions and
# limitations under the License.
# The k8s half of the probe fixture: the namespace and the WorkerPool. The
# ActorTemplate is the substrate resource in probe-template.yaml.tmpl, created
# through the ate API by e2e.DeploySubstrateFixture.
#
# Rendered by e2e.RenderFixtureManifest, which fills the ${...} placeholders for
# the sandbox class under test: one manifest serves both runtimes so the gVisor
# and micro-VM variants cannot drift apart. A placeholder that has no value for
@@ -35,50 +39,3 @@ spec:
replicas: 3
workerImage: ${ATEOM_IMAGE}
${WORKERPOOL_RUNTIME}
---
apiVersion: ate.dev/v1alpha1
kind: ActorTemplate
metadata:
name: probe
namespace: ate-e2e-probe${FIXTURE_SUFFIX}
spec:
${TEMPLATE_SANDBOX_CLASS}
volumes:
- name: system-info
systemInfo:
dataSources:
- actorMetadata:
items:
- field: name
path: actor-id
- field: atespace
path: atespace
- field: uid
path: actor-uid
# Only for suites that pass WithTrustBundle: the bundle is a cluster
# singleton, so a suite that merely needs a probe must not depend on it.
${TEMPLATE_TRUST_BUNDLE}
containers:
- name: probe
image: ko://github.com/agent-substrate/substrate/internal/e2e/fixtures/probe
command: ["/ko-app/probe"]
volumeMounts:
- name: system-info
mountPath: /run/ate # the probe reads actor metadata and trust bundles under /run/ate
# The probe binary binds :80 immediately, so this gates actor start on a
# readiness signal rather than a guess, and carries a non-default
# timeoutSeconds so e2e covers the value crossing ateapi -> atelet -> ateom
# instead of only the ateom's built-in default.
readyz:
httpGet:
path: /healthz
port: 80
timeoutSeconds: 60
${TEMPLATE_RESOURCES}
workerSelector:
matchLabels:
workload: probe${FIXTURE_SUFFIX}
snapshotsConfig:
location: gs://${BUCKET_NAME}/ate-e2e-probe${FIXTURE_SUFFIX}/
@@ -0,0 +1,51 @@
# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# The gRPC Actor the ingress suite reaches through atenet-router, running the
# same `testserver grpc` echo origin the egress suite deploys as a plain pod.
# One binary for both directions, so what counts as a working RPC cannot differ
# between them.
#
# A protojson-shaped ateapipb.ActorTemplate (NOT the ate.dev/v1alpha1 CRD),
# created through the ate API by e2e.DeploySubstrateFixture.
metadata:
atespace: ate-e2e-grpcecho${FIXTURE_SUFFIX}
name: grpcecho
workerSelector:
matchLabels:
workload: grpcecho${FIXTURE_SUFFIX}
containers:
- name: grpcecho
image: ko://github.com/agent-substrate/substrate/internal/e2e/fixtures/testserver
command: ["/ko-app/testserver"]
# gRPC on the Actor's primary port, so the ingress test reaches it at the
# Actor's DNS name with no CONNECT and no port juggling -- the plain path
# every other ingress assertion in the suite uses.
args:
- "grpc"
- "--listen=:80"
- "--health-listen=:8080"
# readyz is an HTTP GET and nothing else, and a gRPC server answers one with
# a protocol error -- hence the fixture's second, HTTP-only port. Probing :80
# here would leave the Actor stuck out of PhaseReady forever.
readyz:
httpGet:
path: /readyz
port: 8080
timeoutSeconds: 60
${TEMPLATE_RESOURCES}
${TEMPLATE_SANDBOX_CONFIG}
snapshotsConfig:
storageLocation: gs://${BUCKET_NAME}/ate-e2e-grpcecho${FIXTURE_SUFFIX}/
@@ -12,10 +12,9 @@
# See the License for the specific language governing permissions and
# limitations under the License.
# The gRPC Actor the ingress suite reaches through atenet-router, running the
# same `testserver grpc` echo origin the egress suite deploys as a plain pod.
# One binary for both directions, so what counts as a working RPC cannot differ
# between them.
# The k8s half of the grpcecho fixture; the ActorTemplate is the substrate
# resource in grpcecho-template.yaml.tmpl, created through the ate API by
# e2e.DeploySubstrateFixture.
#
# Rendered by e2e.RenderFixtureManifest, which fills the ${...} placeholders for
# the sandbox class under test: one manifest serves both runtimes so the gVisor
@@ -42,38 +41,3 @@ spec:
replicas: 2
workerImage: ${ATEOM_IMAGE}
${WORKERPOOL_RUNTIME}
---
apiVersion: ate.dev/v1alpha1
kind: ActorTemplate
metadata:
name: grpcecho
namespace: ate-e2e-grpcecho${FIXTURE_SUFFIX}
spec:
${TEMPLATE_SANDBOX_CLASS}
containers:
- name: grpcecho
image: ko://github.com/agent-substrate/substrate/internal/e2e/fixtures/testserver
command: ["/ko-app/testserver"]
# gRPC on the Actor's primary port, so the ingress test reaches it at the
# Actor's DNS name with no CONNECT and no port juggling -- the plain path
# every other ingress assertion in the suite uses.
args:
- "grpc"
- "--listen=:80"
- "--health-listen=:8080"
# readyz is an HTTP GET and nothing else, and a gRPC server answers one with
# a protocol error -- hence the fixture's second, HTTP-only port. Probing :80
# here would leave the Actor stuck out of PhaseReady forever.
readyz:
httpGet:
path: /readyz
port: 8080
timeoutSeconds: 60
${TEMPLATE_RESOURCES}
workerSelector:
matchLabels:
workload: grpcecho${FIXTURE_SUFFIX}
snapshotsConfig:
location: gs://${BUCKET_NAME}/ate-e2e-grpcecho${FIXTURE_SUFFIX}/
@@ -0,0 +1,37 @@
# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# The websocket fixture's ActorTemplate: a protojson-shaped
# ateapipb.ActorTemplate (NOT the ate.dev/v1alpha1 CRD), created through the
# ate API by e2e.DeploySubstrateFixture.
metadata:
atespace: ate-e2e${FIXTURE_SUFFIX}
name: websocket
workerSelector:
matchLabels:
workload: websocket${FIXTURE_SUFFIX}
containers:
- name: websocket
image: ko://github.com/agent-substrate/substrate/internal/e2e/fixtures/testserver
args: ["websocket"]
readyz:
httpGet:
path: /readyz
port: 80
timeoutSeconds: 30
${TEMPLATE_RESOURCES}
${TEMPLATE_SANDBOX_CONFIG}
snapshotsConfig:
storageLocation: gs://${BUCKET_NAME}/ate-e2e-websocket${FIXTURE_SUFFIX}/
@@ -12,6 +12,10 @@
# See the License for the specific language governing permissions and
# limitations under the License.
# The k8s half of the websocket fixture; the ActorTemplate is the substrate
# resource in websocket-template.yaml.tmpl, created through the ate API by
# e2e.DeploySubstrateFixture.
apiVersion: v1
kind: Namespace
metadata:
@@ -30,28 +34,3 @@ spec:
replicas: 1
workerImage: ${ATEOM_IMAGE}
${WORKERPOOL_RUNTIME}
---
apiVersion: ate.dev/v1alpha1
kind: ActorTemplate
metadata:
name: websocket
namespace: ate-e2e${FIXTURE_SUFFIX}
spec:
${TEMPLATE_SANDBOX_CLASS}
containers:
- name: websocket
image: ko://github.com/agent-substrate/substrate/internal/e2e/fixtures/testserver
args: ["websocket"]
readyz:
httpGet:
path: /readyz
port: 80
timeoutSeconds: 30
${TEMPLATE_RESOURCES}
workerSelector:
matchLabels:
workload: websocket${FIXTURE_SUFFIX}
snapshotsConfig:
location: gs://${BUCKET_NAME}/ate-e2e-websocket${FIXTURE_SUFFIX}/
+14
View File
@@ -115,3 +115,17 @@ func koApply(t *testing.T, manifest string) {
}
RunCmdWithEnv(t, []string{"KO_CONFIG_PATH=" + root}, filepath.Join(root, "hack/run-tool.sh"), applyArgs...)
}
// koResolve builds and pushes the ko:// images named in manifest and returns
// the manifest with those references replaced by pushed digests. Same pinned
// ko and KO_CONFIG_PATH rules as koApply; resolve is how a manifest that is
// not destined for kubectl (a protojson ActorTemplate document) still gets
// its image built.
func koResolve(t *testing.T, manifest string) []byte {
t.Helper()
root, err := FindRepoRoot()
if err != nil {
t.Fatalf("FindRepoRoot: %v", err)
}
return RunCmdOutput(t, []string{"KO_CONFIG_PATH=" + root}, filepath.Join(root, "hack/run-tool.sh"), "ko", "resolve", "-f", manifest)
}
+19 -47
View File
@@ -17,21 +17,20 @@ package e2e
import (
"context"
"testing"
"github.com/agent-substrate/substrate/pkg/proto/ateapipb"
)
// ProbeName is the name of the probe fixture's WorkerPool and ActorTemplate,
// inside the namespace DeployProbe returns.
// inside the atespace (and matching k8s namespace) DeployProbe returns.
const ProbeName = "probe"
// probeManifest is the fixture template DeployProbe renders.
const probeManifest = "internal/e2e/fixtures/probe/probe.yaml.tmpl"
// trustBundleDataSource fills ${TEMPLATE_TRUST_BUNDLE}, indented to sit in the
// template's dataSources list. The name must be on atelet's supported-bundle
// allowlist.
const trustBundleDataSource = ` - trustBundle:
name: egress-mitm.ate.dev
path: trust-bundle.pem`
// probeManifests are the fixture templates DeployProbe deploys: the k8s pool
// half and the substrate ActorTemplate half.
var probeManifests = SubstrateFixtureManifests{
Pool: "internal/e2e/fixtures/probe/probe.yaml.tmpl",
Template: "internal/e2e/fixtures/probe/probe-template.yaml.tmpl",
}
// ProbeOption adjusts what DeployProbe installs.
type ProbeOption func(*probeConfig)
@@ -49,12 +48,14 @@ type probeConfig struct{ trustBundle bool }
// step, leaving the identity suite the only opt-in in the standard lanes.
func WithTrustBundle() ProbeOption { return func(c *probeConfig) { c.trustBundle = true } }
// DeployProbe builds the probe fixture image and applies its manifest for the
// sandbox class under test, removing it when the test ends. name distinguishes
// the caller (by convention its suite name): each suite gets its own copy of
// the fixture, so no suite's cleanup can delete the fixture out from under
// another running concurrently. It returns the fixture's namespace.
func DeployProbe(t *testing.T, bucket, name string, opts ...ProbeOption) string {
// DeployProbe builds the probe fixture image and installs the fixture for the
// sandbox class under test, removing it when the test ends. name
// distinguishes the caller (by convention its suite name): each suite gets
// its own copy of the fixture, so no suite's cleanup can delete the fixture
// out from under another running concurrently. It returns the fixture's
// atespace (which also names the k8s namespace holding the pool) and the
// created ActorTemplate, already golden-snapshotted.
func DeployProbe(t *testing.T, bucket, name string, opts ...ProbeOption) (string, *ateapipb.ActorTemplate) {
t.Helper()
var cfg probeConfig
@@ -67,35 +68,6 @@ func DeployProbe(t *testing.T, bucket, name string, opts ...ProbeOption) string
EnsureEgressTrustBundle(t, context.Background(), GetClients())
}
// One manifest, rendered for the sandbox class under test, so both apply
// and delete consume the same file without any shell involved.
manifest := renderProbeManifest(t, bucket, name, cfg)
koApply(t, manifest)
// Unlike the fixtures that live in a namespace CreateNamespace tears down,
// this one installs into a fixed namespace it shares with nothing, so it has
// to clean up after itself.
t.Cleanup(func() {
// Deletion needs no image build, so go straight to kubectl. `ko delete`
// rejects this arg shape ("you may not specify resource arguments as
// well").
delArgs := []string{"delete", "--ignore-not-found", "-f", manifest}
if KubeContext != "" {
delArgs = append([]string{"--context=" + KubeContext}, delArgs...)
}
RunCmd(t, "kubectl", delArgs...)
})
return FixtureName("ate-e2e-probe") + "-" + name
}
// renderProbeManifest renders the probe fixture for cfg. Split out of
// DeployProbe so the rendering has a unit test that needs no cluster.
func renderProbeManifest(t *testing.T, bucket, name string, cfg probeConfig) string {
t.Helper()
inline, blocks := fixtureSubstitutions(bucket, name)
if cfg.trustBundle {
blocks["${TEMPLATE_TRUST_BUNDLE}"] = trustBundleDataSource
}
return renderManifest(t, probeManifest, inline, blocks)
atespace, templates := DeploySubstrateFixture(t, context.Background(), GetClients(), probeManifests, bucket, name, cfg.trustBundle)
return atespace, templates[0]
}
+24 -18
View File
@@ -15,17 +15,18 @@
package e2e
import (
"os"
"strings"
"testing"
"github.com/agent-substrate/substrate/pkg/api/v1alpha1"
"github.com/agent-substrate/substrate/pkg/proto/ateapipb"
)
// TestRenderProbeManifest_TrustBundle pins the opt-in. The bundle is derived
// from one cluster-wide Secret, so a probe suite that does not ask for the
// TestProbeTemplate_TrustBundle pins the opt-in. The bundle is derived from
// one cluster-wide Secret, so a probe suite that does not ask for the
// projection must not carry it: it would otherwise fail whenever the suite
// that owns the pool finishes and takes the bundle with it.
func TestRenderProbeManifest_TrustBundle(t *testing.T) {
func TestProbeTemplate_TrustBundle(t *testing.T) {
t.Setenv(sandboxClassEnv, "")
for _, tc := range []struct {
name string
@@ -37,18 +38,23 @@ func TestRenderProbeManifest_TrustBundle(t *testing.T) {
} {
t.Run(tc.name, func(t *testing.T) {
// Strict decoding is what proves the fragment landed at the right
// depth: misindented, it would parse as some other field.
_, template := decodeFixture(t, probeManifest,
renderProbeManifest(t, "test-bucket", "render", tc.cfg))
// depth: misindented, it would fail the protojson decode or parse
// as some other field.
inline, blocks := substrateTemplateSubstitutions("test-bucket", "render", tc.cfg.trustBundle)
rendered, err := os.ReadFile(renderManifest(t, probeManifests.Template, inline, blocks))
if err != nil {
t.Fatalf("reading rendered manifest: %v", err)
}
templates := decodeSubstrateTemplates(t, rendered)
if len(templates) != 1 {
t.Fatalf("probe template manifest yields %d documents, want 1", len(templates))
}
var source *v1alpha1.TrustBundleDataSource
for _, vol := range template.Spec.Volumes {
if vol.SystemInfo == nil {
continue
}
for _, ds := range vol.SystemInfo.DataSources {
if ds.TrustBundle != nil {
source = ds.TrustBundle
var source *ateapipb.TrustBundleDataSource
for _, vol := range templates[0].GetVolumes() {
for _, ds := range vol.GetSystemInfo().GetDataSources() {
if ds.GetTrustBundle() != nil {
source = ds.GetTrustBundle()
}
}
}
@@ -60,10 +66,10 @@ func TestRenderProbeManifest_TrustBundle(t *testing.T) {
}
// The projected name must select the bundle atecontroller
// publishes, or actors fail closed on a name atelet rejects.
if !strings.HasPrefix(EgressTrustBundleObjectName, source.Name+":") {
t.Errorf("trustBundle name = %q, want the bundle backing %q", source.Name, EgressTrustBundleObjectName)
if !strings.HasPrefix(EgressTrustBundleObjectName, source.GetName()+":") {
t.Errorf("trustBundle name = %q, want the bundle backing %q", source.GetName(), EgressTrustBundleObjectName)
}
if source.Path == "" {
if source.GetPath() == "" {
t.Error("trustBundle projection has no path")
}
})
+20
View File
@@ -15,6 +15,7 @@
package e2e
import (
"bytes"
"fmt"
"os"
"os/exec"
@@ -57,6 +58,25 @@ func RunCmd(t *testing.T, name string, args ...string) {
}
}
// RunCmdOutput executes the given command with custom environment variables
// appended to the current process environment, streaming stderr like RunCmd
// does, and returns the captured stdout. Fails the test if the command
// returns an error.
func RunCmdOutput(t *testing.T, env []string, name string, args ...string) []byte {
t.Helper()
t.Logf("Running command: %s %s", name, strings.Join(args, " "))
cmd := exec.Command(name, args...)
cmd.Env = append(os.Environ(), env...)
var stdout bytes.Buffer
cmd.Stdout = &stdout
stderrColor := &ColorWriter{W: os.Stderr, ANSI: ansiRed}
cmd.Stderr = NewIndentWriter(stderrColor, " ")
if err := cmd.Run(); err != nil {
t.Fatalf("Command failed: %s %s: %v", name, strings.Join(args, " "), err)
}
return stdout.Bytes()
}
// RunCmdWithEnv executes the given command with custom environment variables
// appended to the current process environment, and fails the test if it returns an error.
func RunCmdWithEnv(t *testing.T, env []string, name string, args ...string) {
+130 -70
View File
@@ -20,41 +20,56 @@ import (
"testing"
"github.com/agent-substrate/substrate/pkg/api/v1alpha1"
"github.com/agent-substrate/substrate/pkg/proto/ateapipb"
"sigs.k8s.io/yaml"
)
// fixtureManifests are every template RenderFixtureManifest is asked to render.
var fixtureManifests = []string{
"internal/e2e/fixtures/probe/probe.yaml.tmpl",
"internal/e2e/fixtures/probe/probe-sized.yaml.tmpl",
"internal/e2e/fixtures/capabilities/capabilities.yaml.tmpl",
"internal/e2e/fixtures/testserver/websocket.yaml.tmpl",
"internal/e2e/fixtures/testserver/grpcecho.yaml.tmpl",
// substrateFixtures are every fixture DeploySubstrateFixture is asked to
// deploy: the pool half rendered by RenderFixtureManifest and the substrate
// ActorTemplate half rendered with substrateTemplateSubstitutions, plus how
// many template documents the latter must yield.
var substrateFixtures = []struct {
manifests SubstrateFixtureManifests
templates int
}{
{SubstrateFixtureManifests{
Pool: "internal/e2e/fixtures/probe/probe.yaml.tmpl",
Template: "internal/e2e/fixtures/probe/probe-template.yaml.tmpl",
}, 1},
{SubstrateFixtureManifests{
Pool: "internal/e2e/fixtures/probe/probe-sized.yaml.tmpl",
Template: "internal/e2e/fixtures/probe/probe-sized-template.yaml.tmpl",
}, 1},
{SubstrateFixtureManifests{
Pool: "internal/e2e/fixtures/capabilities/capabilities.yaml.tmpl",
Template: "internal/e2e/fixtures/capabilities/capabilities-templates.yaml.tmpl",
}, 2},
{SubstrateFixtureManifests{
Pool: "internal/e2e/fixtures/testserver/websocket.yaml.tmpl",
Template: "internal/e2e/fixtures/testserver/websocket-template.yaml.tmpl",
}, 1},
{SubstrateFixtureManifests{
Pool: "internal/e2e/fixtures/testserver/grpcecho.yaml.tmpl",
Template: "internal/e2e/fixtures/testserver/grpcecho-template.yaml.tmpl",
}, 1},
}
// renderFixture renders a manifest and decodes the two resources the
// assertions below care about (the third document is a Namespace).
// renderPool renders a fixture's pool manifest and strict-decodes its
// WorkerPool.
//
// Strict decoding against the real API types is the point: the runtime blocks
// Strict decoding against the real API type is the point: the runtime blocks
// are injected as pre-indented text, so a placeholder that lands at the wrong
// depth yields YAML that still parses but hangs the field off the wrong parent
// — which strict mode reports as an unknown field instead of silently applying
// a WorkerPool that never gets a micro-VM worker.
func renderFixture(t *testing.T, relPath string) (*v1alpha1.WorkerPool, *v1alpha1.ActorTemplate) {
func renderPool(t *testing.T, relPath string) *v1alpha1.WorkerPool {
t.Helper()
return decodeFixture(t, relPath, RenderFixtureManifest(t, relPath, "test-bucket", "render"))
}
// decodeFixture strict-decodes an already-rendered manifest, for callers that
// render a fixture some way other than RenderFixtureManifest.
func decodeFixture(t *testing.T, relPath, rendered string) (*v1alpha1.WorkerPool, *v1alpha1.ActorTemplate) {
t.Helper()
raw, err := os.ReadFile(rendered)
raw, err := os.ReadFile(RenderFixtureManifest(t, relPath, "test-bucket", "render"))
if err != nil {
t.Fatalf("reading the rendered %s: %v", relPath, err)
}
pool, template := &v1alpha1.WorkerPool{}, &v1alpha1.ActorTemplate{}
pool := &v1alpha1.WorkerPool{}
for doc := range strings.SplitSeq(string(raw), "\n---\n") {
if strings.TrimSpace(doc) == "" {
continue
@@ -65,34 +80,51 @@ func decodeFixture(t *testing.T, relPath, rendered string) (*v1alpha1.WorkerPool
if err := yaml.Unmarshal([]byte(doc), &meta); err != nil {
t.Fatalf("rendered %s is not valid YAML: %v\n%s", relPath, err, doc)
}
var into any
switch meta.Kind {
case "WorkerPool":
into = pool
case "ActorTemplate":
into = template
default:
if meta.Kind != "WorkerPool" {
continue
}
if err := yaml.UnmarshalStrict([]byte(doc), into); err != nil {
t.Fatalf("rendered %s %s does not match the API type: %v\n%s", relPath, meta.Kind, err, doc)
if err := yaml.UnmarshalStrict([]byte(doc), pool); err != nil {
t.Fatalf("rendered %s WorkerPool does not match the API type: %v\n%s", relPath, meta.Kind, doc)
}
}
if pool.Name == "" || template.Name == "" {
t.Fatalf("rendered %s is missing a WorkerPool or an ActorTemplate", relPath)
if pool.Name == "" {
t.Fatalf("rendered %s is missing a WorkerPool", relPath)
}
return pool, template
return pool
}
// TestRenderFixtureManifest_GVisor pins the default rendering: every micro-VM
// block is gone and no placeholder survives, so the gVisor lane keeps applying
// exactly what it applied before the templates were parameterized.
func TestRenderFixtureManifest_GVisor(t *testing.T) {
t.Setenv(sandboxClassEnv, "")
for _, relPath := range fixtureManifests {
t.Run(relPath, func(t *testing.T) {
pool, template := renderFixture(t, relPath)
// renderTemplates renders a fixture's substrate template manifest and
// strict-decodes its ActorTemplate documents; the protojson decode plays the
// same misplaced-placeholder tripwire renderPool's strict mode does.
func renderTemplates(t *testing.T, relPath string) []*ateapipb.ActorTemplate {
t.Helper()
inline, blocks := substrateTemplateSubstitutions("test-bucket", "render", false)
rendered, err := os.ReadFile(renderManifest(t, relPath, inline, blocks))
if err != nil {
t.Fatalf("reading the rendered %s: %v", relPath, err)
}
return decodeSubstrateTemplates(t, rendered)
}
// memoryLimit returns the template's spec-level memory limit quantity, "" if
// none is declared.
func memoryLimit(tmpl *ateapipb.ActorTemplate) string {
for _, l := range tmpl.GetResources().GetLimits() {
if l.GetName() == "memory" {
return l.GetQuantity()
}
}
return ""
}
// TestRenderSubstrateFixtures_GVisor pins the default rendering: every
// micro-VM block is gone, no placeholder survives, and the templates name the
// cluster-wide default SandboxConfig.
func TestRenderSubstrateFixtures_GVisor(t *testing.T) {
t.Setenv(sandboxClassEnv, "")
for _, fixture := range substrateFixtures {
t.Run(fixture.manifests.Pool, func(t *testing.T) {
pool := renderPool(t, fixture.manifests.Pool)
if !strings.HasSuffix(pool.Spec.WorkerImage, "/cmd/ateom-gvisor") {
t.Errorf("WorkerPool workerImage = %q, want the gVisor ateom", pool.Spec.WorkerImage)
}
@@ -101,32 +133,48 @@ func TestRenderFixtureManifest_GVisor(t *testing.T) {
pool.Spec.SandboxClass, pool.Spec.SandboxConfigName)
}
if template.Spec.SandboxClass != "" {
t.Errorf("ActorTemplate sandboxClass = %q, want unset for gVisor", template.Spec.SandboxClass)
templates := renderTemplates(t, fixture.manifests.Template)
if len(templates) != fixture.templates {
t.Fatalf("rendered %s yields %d templates, want %d", fixture.manifests.Template, len(templates), fixture.templates)
}
// An inline placeholder with an empty value must substitute, not
// delete its line: the location is what the golden snapshot needs.
if want := "gs://test-bucket/"; !strings.HasPrefix(template.Spec.SnapshotsConfig.Location, want) {
t.Errorf("ActorTemplate snapshot location = %q, want it to start with %q",
template.Spec.SnapshotsConfig.Location, want)
}
if strings.HasSuffix(template.Spec.SnapshotsConfig.Location, "-microvm/") {
t.Errorf("ActorTemplate snapshot location = %q, want no micro-VM suffix",
template.Spec.SnapshotsConfig.Location)
for _, tmpl := range templates {
name := tmpl.GetMetadata().GetName()
if got := tmpl.GetSandboxConfig().GetSandboxClass(); got != ateapipb.SandboxClass_SANDBOX_CLASS_GVISOR {
t.Errorf("template %s sandboxClass = %v, want GVISOR", name, got)
}
// The templates name the cluster-wide default SandboxConfig
// explicitly: config_name is required.
if got := tmpl.GetSandboxConfig().GetConfigName(); got != "gvisor-default" {
t.Errorf("template %s configName = %q, want gvisor-default", name, got)
}
// An inline placeholder with an empty value must substitute, not
// delete its line: the location is what the golden snapshot needs.
location := tmpl.GetSnapshotsConfig().GetStorageLocation()
if want := "gs://test-bucket/"; !strings.HasPrefix(location, want) {
t.Errorf("template %s snapshot location = %q, want it to start with %q", name, location, want)
}
if strings.Contains(location, "-microvm") {
t.Errorf("template %s snapshot location = %q, want no micro-VM suffix", name, location)
}
// The selector is what ties the template to its fixture's pool.
for k, v := range tmpl.GetWorkerSelector().GetMatchLabels() {
if pool.Labels[k] != v {
t.Errorf("template %s selects %s=%s, which the pool's labels %v do not carry", name, k, v, pool.Labels)
}
}
}
})
}
}
// TestRenderFixtureManifest_MicroVM pins the micro-VM rendering: the pool names
// the cluster-wide SandboxConfig, the template matches its class, and the
// snapshots land under their own prefix.
func TestRenderFixtureManifest_MicroVM(t *testing.T) {
// TestRenderSubstrateFixtures_MicroVM pins the micro-VM rendering: the pool
// names the cluster-wide SandboxConfig, the templates match its class and
// carry limits, and the snapshots land under their own prefix.
func TestRenderSubstrateFixtures_MicroVM(t *testing.T) {
t.Setenv(sandboxClassEnv, SandboxClassMicroVM)
for _, relPath := range fixtureManifests {
t.Run(relPath, func(t *testing.T) {
pool, template := renderFixture(t, relPath)
for _, fixture := range substrateFixtures {
t.Run(fixture.manifests.Pool, func(t *testing.T) {
pool := renderPool(t, fixture.manifests.Pool)
if !strings.HasSuffix(pool.Spec.WorkerImage, "/cmd/ateom-microvm") {
t.Errorf("WorkerPool workerImage = %q, want the micro-VM ateom", pool.Spec.WorkerImage)
}
@@ -135,18 +183,30 @@ func TestRenderFixtureManifest_MicroVM(t *testing.T) {
pool.Spec.SandboxClass, pool.Spec.SandboxConfigName)
}
if template.Spec.SandboxClass != SandboxClassMicroVM {
t.Errorf("ActorTemplate sandboxClass = %q, want %q — it must match the pool's or no worker is eligible",
template.Spec.SandboxClass, SandboxClassMicroVM)
templates := renderTemplates(t, fixture.manifests.Template)
if len(templates) != fixture.templates {
t.Fatalf("rendered %s yields %d templates, want %d", fixture.manifests.Template, len(templates), fixture.templates)
}
// Undeclared limits boot the guest at the kata config default
// (2GiB), which does not fit beside the demo pools on one kind node.
if template.Spec.Resources.Limits.Memory().IsZero() {
t.Errorf("ActorTemplate declares no memory limit, so the guest would boot at the kata default: %+v", template.Spec.Resources)
}
if want := "-microvm-render/"; !strings.HasSuffix(template.Spec.SnapshotsConfig.Location, want) {
t.Errorf("ActorTemplate snapshot location = %q, want it to end with %q",
template.Spec.SnapshotsConfig.Location, want)
for _, tmpl := range templates {
name := tmpl.GetMetadata().GetName()
if got := tmpl.GetSandboxConfig().GetSandboxClass(); got != ateapipb.SandboxClass_SANDBOX_CLASS_MICROVM {
t.Errorf("template %s sandboxClass = %v, want MICROVM — it must match the pool's or no worker is eligible", name, got)
}
// Deliberately not the class default (see fixture.go), so a
// missing or stale microvm install fails loudly.
if got := tmpl.GetSandboxConfig().GetConfigName(); got != "microvm" {
t.Errorf("template %s configName = %q, want microvm", name, got)
}
// Undeclared limits boot the guest at the kata config default
// (2GiB), which does not fit beside the demo pools on one kind
// node.
if memoryLimit(tmpl) == "" {
t.Errorf("template %s declares no memory limit, so the guest would boot at the kata default", name)
}
location := tmpl.GetSnapshotsConfig().GetStorageLocation()
if want := "-microvm-render/"; !strings.HasSuffix(location, want) {
t.Errorf("template %s snapshot location = %q, want it to end with %q", name, location, want)
}
}
})
}
@@ -19,16 +19,12 @@ import (
"encoding/json"
"io"
"net/http"
"path/filepath"
"slices"
"testing"
"time"
"github.com/agent-substrate/substrate/internal/e2e"
"github.com/agent-substrate/substrate/internal/resources"
"github.com/agent-substrate/substrate/pkg/api/v1alpha1"
"github.com/agent-substrate/substrate/pkg/proto/ateapipb"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
// defaultCapabilities mirrors atelet's default set (cmd/atelet/oci.go). It is
@@ -62,7 +58,7 @@ func TestActorCapabilities(t *testing.T) {
ctx := context.Background()
clients := e2e.GetClients()
namespace := deployFixture(t, env["BUCKET_NAME"])
namespace := deployFixture(t, ctx, clients, env["BUCKET_NAME"])
tests := []struct {
name string
@@ -84,7 +80,6 @@ func TestActorCapabilities(t *testing.T) {
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
waitForGolden(t, ctx, clients, namespace, tt.template)
actor := tt.template + "-actor"
createAndResumeActor(t, ctx, clients, namespace, tt.template, actor)
@@ -131,74 +126,27 @@ func assertSameCapabilities(t *testing.T, set string, got, want []string) {
}
}
// deployFixture renders and applies the fixture for the sandbox class under
// test and returns the namespace it created. The namespace carries the class
// suffix so the gVisor and micro-VM lanes never share one.
func deployFixture(t *testing.T, bucket string) string {
// deployFixture installs the fixture for the sandbox class under test and
// returns its atespace (which also names the namespace it created, carrying
// the class suffix so the gVisor and micro-VM lanes never share one). Both
// templates are golden-snapshotted when this returns; a template whose
// container cannot start — for example because a needed capability was
// dropped — fails the deploy with the template's error message rather than
// timing out per subtest.
func deployFixture(t *testing.T, ctx context.Context, clients *e2e.Clients, bucket string) string {
t.Helper()
root, err := e2e.FindRepoRoot()
if err != nil {
t.Fatalf("FindRepoRoot: %v", err)
}
namespace := e2e.FixtureName("ate-e2e") + "-capabilities"
// One manifest, rendered for the sandbox class under test (mirrors the
// sizing suite).
manifest := e2e.RenderFixtureManifest(t, "internal/e2e/fixtures/capabilities/capabilities.yaml.tmpl", bucket, "capabilities")
// Build/push the probe image and apply through the repo's pinned ko, as the
// identity suite does; CI does not install ko on PATH, and KO_CONFIG_PATH is
// required because ko resolves .ko.yaml from its working directory.
applyArgs := []string{"ko", "apply", "-f", manifest}
if e2e.KubeContext != "" {
applyArgs = append(applyArgs, "--", "--context="+e2e.KubeContext)
}
e2e.RunCmdWithEnv(t, []string{"KO_CONFIG_PATH=" + root}, filepath.Join(root, "hack/run-tool.sh"), applyArgs...)
t.Cleanup(func() {
delArgs := []string{"delete", "--ignore-not-found", "-f", manifest}
if e2e.KubeContext != "" {
delArgs = append([]string{"--context=" + e2e.KubeContext}, delArgs...)
}
e2e.RunCmd(t, "kubectl", delArgs...)
})
return namespace
}
func waitForGolden(t *testing.T, ctx context.Context, clients *e2e.Clients, namespace, template string) {
t.Helper()
deadline := time.Now().Add(10 * time.Minute)
for time.Now().Before(deadline) {
at, err := clients.SubstrateK8s.ApiV1alpha1().ActorTemplates(namespace).Get(ctx, template, metav1.GetOptions{})
if err == nil {
switch at.Status.Phase {
case v1alpha1.PhaseReady:
t.Logf("ActorTemplate %s ready, golden=%s", template, at.Status.GoldenActorID)
return
case v1alpha1.PhaseFailed:
// A template whose container cannot start — for example because
// a needed capability was dropped — lands here rather than
// timing out, so say so plainly.
t.Fatalf("ActorTemplate %s entered PhaseFailed; its container never became ready", template)
}
}
time.Sleep(2 * time.Second)
}
t.Fatalf("timed out waiting for ActorTemplate %s to be Ready", template)
atespace, _ := e2e.DeploySubstrateFixture(t, ctx, clients, e2e.SubstrateFixtureManifests{
Pool: "internal/e2e/fixtures/capabilities/capabilities.yaml.tmpl",
Template: "internal/e2e/fixtures/capabilities/capabilities-templates.yaml.tmpl",
}, bucket, "capabilities", false)
return atespace
}
func createAndResumeActor(t *testing.T, ctx context.Context, clients *e2e.Clients, namespace, template, id string) {
t.Helper()
// CreateActor requires the atespace to exist first.
_, _ = clients.SubstrateAPI.CreateAtespace(ctx, &ateapipb.CreateAtespaceRequest{
Atespace: &ateapipb.Atespace{Metadata: &ateapipb.ResourceMetadata{Name: namespace}},
})
if _, err := clients.SubstrateAPI.CreateActor(ctx, &ateapipb.CreateActorRequest{Actor: &ateapipb.Actor{
Metadata: &ateapipb.ResourceMetadata{Atespace: namespace, Name: id},
ActorTemplateNamespace: namespace,
ActorTemplateName: template,
Metadata: &ateapipb.ResourceMetadata{Atespace: namespace, Name: id},
ActorTemplate: &ateapipb.ObjectRef{Atespace: namespace, Name: template},
}}); err != nil {
t.Fatalf("CreateActor %q: %v", id, err)
}
+1 -1
View File
@@ -1070,7 +1070,7 @@ func createActorTemplateInternal(ctx context.Context, t *testing.T, clients *e2e
// The whole suite shares demoAtespace, so the per-test suffix keeps
// template names unique.
name := base + "-" + nsObj.Name
at := e2e.CreateSubstrateCounterTemplate(ctx, t, clients, nsObj.Name, e2e.SubstrateCounterTemplateOptions{
at := e2e.CreateSubstrateCounterTemplate(ctx, t, clients, nsObj.Name, e2e.SubstrateTemplateOptions{
Atespace: demoAtespace,
Name: name,
PoolName: base,
@@ -32,9 +32,7 @@ import (
"github.com/agent-substrate/substrate/internal/e2e"
"github.com/agent-substrate/substrate/internal/resources"
"github.com/agent-substrate/substrate/pkg/api/v1alpha1"
"github.com/agent-substrate/substrate/pkg/proto/ateapipb"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
const probeTemplate = "probe"
@@ -87,8 +85,7 @@ func TestActorEgressMITMTrust(t *testing.T) {
// missing.)
e2e.EnsureEgressTrustBundle(t, ctx, clients)
probeNamespace = e2e.DeployProbe(t, env["BUCKET_NAME"], "egressmitm", e2e.WithTrustBundle())
waitForGolden(t, ctx, clients)
probeNamespace, _ = e2e.DeployProbe(t, env["BUCKET_NAME"], "egressmitm", e2e.WithTrustBundle())
const id = "probe-mitm"
createAndResumeActor(t, ctx, clients, id)
@@ -172,37 +169,18 @@ func probeFetch(t *testing.T, ctx context.Context, rc *e2e.RouterClient, id, ori
}
}
// The helpers below mirror the identity suite's: fixture golden wait and a
// self-healing actor lifecycle (actor records outlive the fixture namespace).
func waitForGolden(t *testing.T, ctx context.Context, clients *e2e.Clients) {
t.Helper()
deadline := time.Now().Add(e2e.TemplateReadyTimeout(t))
for time.Now().Before(deadline) {
at, err := clients.SubstrateK8s.ApiV1alpha1().ActorTemplates(probeNamespace).Get(ctx, probeTemplate, metav1.GetOptions{})
if err == nil {
switch at.Status.Phase {
case v1alpha1.PhaseReady:
return
case v1alpha1.PhaseFailed:
t.Fatalf("probe ActorTemplate entered PhaseFailed")
}
}
time.Sleep(2 * time.Second)
}
t.Fatalf("timed out waiting for probe ActorTemplate to be Ready")
}
// createAndResumeActor mirrors the identity suite's self-healing actor
// lifecycle (actor records outlive the fixture namespace); DeployProbe has
// already waited for the template's golden snapshot.
func createAndResumeActor(t *testing.T, ctx context.Context, clients *e2e.Clients, id string) {
t.Helper()
_, _ = clients.SubstrateAPI.CreateAtespace(ctx, &ateapipb.CreateAtespaceRequest{Atespace: &ateapipb.Atespace{Metadata: &ateapipb.ResourceMetadata{Name: probeNamespace}}})
ref := &ateapipb.ObjectRef{Atespace: probeNamespace, Name: id}
_, _ = clients.SubstrateAPI.SuspendActor(ctx, &ateapipb.SuspendActorRequest{Actor: ref})
_, _ = clients.SubstrateAPI.DeleteActor(ctx, &ateapipb.DeleteActorRequest{Actor: ref})
if _, err := clients.SubstrateAPI.CreateActor(ctx, &ateapipb.CreateActorRequest{Actor: &ateapipb.Actor{
Metadata: &ateapipb.ResourceMetadata{Atespace: probeNamespace, Name: id},
ActorTemplateNamespace: probeNamespace,
ActorTemplateName: probeTemplate,
Metadata: &ateapipb.ResourceMetadata{Atespace: probeNamespace, Name: id},
ActorTemplate: &ateapipb.ObjectRef{Atespace: probeNamespace, Name: probeTemplate},
}}); err != nil {
t.Fatalf("CreateActor %q: %v", id, err)
}
+9 -29
View File
@@ -24,9 +24,7 @@ import (
"github.com/agent-substrate/substrate/internal/e2e"
"github.com/agent-substrate/substrate/internal/resources"
"github.com/agent-substrate/substrate/pkg/api/v1alpha1"
"github.com/agent-substrate/substrate/pkg/proto/ateapipb"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
const probeTemplate = "probe"
@@ -79,8 +77,13 @@ func TestActorIdentity_AfterRestore_IsOwnID_NotGolden(t *testing.T) {
// ensures a bundle EXISTS): the assertions below compare the projected
// file against this run's CA, and rotation later replaces it again.
wantTrust := e2e.ReplaceEgressTrustPool(t, ctx, clients, "ate-e2e-probe-trust")
probeNamespace = e2e.DeployProbe(t, env["BUCKET_NAME"], "identity", e2e.WithTrustBundle())
golden := waitForGolden(t, ctx, clients)
var tmpl *ateapipb.ActorTemplate
probeNamespace, tmpl = e2e.DeployProbe(t, env["BUCKET_NAME"], "identity", e2e.WithTrustBundle())
// The golden actor's id, for the not-golden assertion below. Coupled to
// the reconciler's naming: the golden actor is named after the template's
// UID (cmd/ateapi template reconciler), so a naming change there weakens
// this check to a no-op rather than false-failing it.
golden := tmpl.GetMetadata().GetUid()
// Two distinct actors from the same golden snapshot.
ids := []string{"probe-alpha", "probe-beta"}
@@ -216,30 +219,8 @@ func waitForActorState(t *testing.T, ctx context.Context, clients *e2e.Clients,
t.Fatalf("timed out waiting for actor %q to reach state %v", actorName, want)
}
func waitForGolden(t *testing.T, ctx context.Context, clients *e2e.Clients) string {
t.Helper()
deadline := time.Now().Add(e2e.TemplateReadyTimeout(t))
for time.Now().Before(deadline) {
at, err := clients.SubstrateK8s.ApiV1alpha1().ActorTemplates(probeNamespace).Get(ctx, probeTemplate, metav1.GetOptions{})
if err == nil {
switch at.Status.Phase {
case v1alpha1.PhaseReady:
t.Logf("probe ActorTemplate ready, golden=%s", at.Status.GoldenActorID)
return at.Status.GoldenActorID
case v1alpha1.PhaseFailed:
t.Fatalf("probe ActorTemplate entered PhaseFailed")
}
}
time.Sleep(2 * time.Second)
}
t.Fatalf("timed out waiting for probe ActorTemplate to be Ready")
return ""
}
func createAndResumeActor(t *testing.T, ctx context.Context, clients *e2e.Clients, id string) {
t.Helper()
// CreateActor requires the atespace to exist first.
_, _ = clients.SubstrateAPI.CreateAtespace(ctx, &ateapipb.CreateAtespaceRequest{Atespace: &ateapipb.Atespace{Metadata: &ateapipb.ResourceMetadata{Name: probeNamespace}}})
ref := &ateapipb.ObjectRef{Atespace: probeNamespace, Name: id}
// The actor record lives in the ateapi store and outlives the fixture
// namespace, so a failed prior run can leak it and wedge every rerun on
@@ -248,9 +229,8 @@ func createAndResumeActor(t *testing.T, ctx context.Context, clients *e2e.Client
_, _ = clients.SubstrateAPI.SuspendActor(ctx, &ateapipb.SuspendActorRequest{Actor: ref})
_, _ = clients.SubstrateAPI.DeleteActor(ctx, &ateapipb.DeleteActorRequest{Actor: ref})
if _, err := clients.SubstrateAPI.CreateActor(ctx, &ateapipb.CreateActorRequest{Actor: &ateapipb.Actor{
Metadata: &ateapipb.ResourceMetadata{Atespace: probeNamespace, Name: id},
ActorTemplateNamespace: probeNamespace,
ActorTemplateName: probeTemplate,
Metadata: &ateapipb.ResourceMetadata{Atespace: probeNamespace, Name: id},
ActorTemplate: &ateapipb.ObjectRef{Atespace: probeNamespace, Name: probeTemplate},
}}); err != nil {
t.Fatalf("CreateActor %q: %v", id, err)
}
@@ -24,22 +24,20 @@ import (
"io"
"net/http"
"os"
"slices"
"strings"
"testing"
"time"
"github.com/agent-substrate/substrate/internal/e2e"
"github.com/agent-substrate/substrate/internal/resources"
"github.com/agent-substrate/substrate/pkg/api/v1alpha1"
"github.com/agent-substrate/substrate/pkg/proto/ateapipb"
"github.com/google/go-containerregistry/pkg/authn"
"github.com/google/go-containerregistry/pkg/name"
v1 "github.com/google/go-containerregistry/pkg/v1"
"github.com/google/go-containerregistry/pkg/v1/empty"
"github.com/google/go-containerregistry/pkg/v1/mutate"
"github.com/google/go-containerregistry/pkg/v1/remote"
"github.com/google/go-containerregistry/pkg/v1/tarball"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
const (
@@ -116,7 +114,10 @@ func buildFixtureImage(t *testing.T, repo string) string {
if err != nil {
t.Fatalf("parsing %q: %v", ref, err)
}
if err := remote.Write(tag, img); err != nil {
// The default keychain reads the local docker config, so the push works
// against an authenticated registry (a GKE dev cluster's gcr.io) as well
// as CI's anonymous kind registry.
if err := remote.Write(tag, img, remote.WithAuthFromKeychain(authn.DefaultKeychain)); err != nil {
t.Fatalf("pushing %q: %v", ref, err)
}
@@ -129,7 +130,9 @@ func buildFixtureImage(t *testing.T, repo string) string {
// createTemplate builds a probe ActorTemplate with the fixture attached as an
// image volume, copying the resolved runtime from the shared probe template.
func createTemplate(ctx context.Context, t *testing.T, clients *e2e.Clients, ns *e2e.Namespace, fixtureImage string) *v1alpha1.ActorTemplate {
// The template's name is suffixed per test run: it lives in the suite's
// shared atespace, which outlives the per-test k8s namespace.
func createTemplate(ctx context.Context, t *testing.T, clients *e2e.Clients, ns *e2e.Namespace, fixtureImage string) *ateapipb.ActorTemplate {
t.Helper()
env, err := e2e.CheckEnv("BUCKET_NAME")
@@ -138,60 +141,36 @@ func createTemplate(ctx context.Context, t *testing.T, clients *e2e.Clients, ns
}
// The probe supplies this suite's container image and resolved runtime.
probeNamespace := e2e.DeployProbe(t, env["BUCKET_NAME"], "imagevolume")
srcPool, err := clients.SubstrateK8s.ApiV1alpha1().WorkerPools(probeNamespace).Get(ctx, probeName, metav1.GetOptions{})
if err != nil {
t.Fatalf("getting WorkerPool %s/%s: %v", probeNamespace, probeName, err)
}
srcTemplate, err := clients.SubstrateK8s.ApiV1alpha1().ActorTemplates(probeNamespace).Get(ctx, probeName, metav1.GetOptions{})
if err != nil {
t.Fatalf("getting ActorTemplate %s/%s: %v", probeNamespace, probeName, err)
probeAtespace, _ := e2e.DeployProbe(t, env["BUCKET_NAME"], "imagevolume")
src := e2e.SubstrateFixture{
Atespace: probeAtespace,
Name: probeName,
PoolNamespace: probeAtespace,
PoolName: probeName,
DeployWith: "the imagevolume suite's own DeployProbe",
}
// The pool is labeled uniquely to this namespace so the cluster-wide
// scheduler cannot hand its workers to another suite's actors.
poolLabels := map[string]string{"imagevolume": ns.Name}
pool := &v1alpha1.WorkerPool{
ObjectMeta: metav1.ObjectMeta{Name: probeName, Namespace: ns.Name, Labels: poolLabels},
Spec: v1alpha1.WorkerPoolSpec{
Replicas: 2,
WorkerImage: srcPool.Spec.WorkerImage,
SandboxClass: srcPool.Spec.SandboxClass,
SandboxConfigName: srcPool.Spec.SandboxConfigName,
return e2e.CreateSubstrateTemplateFrom(ctx, t, clients, ns.Name, src, e2e.SubstrateTemplateOptions{
Atespace: atespace,
Name: "probe-" + ns.Name,
PoolName: probeName,
PoolReplicas: 2,
// The pool is labeled uniquely to this namespace so the cluster-wide
// scheduler cannot hand its workers to another suite's actors.
Labels: map[string]string{"imagevolume": ns.Name},
SnapshotsConfig: &ateapipb.SnapshotsConfig{
StorageLocation: fmt.Sprintf("gs://%s/%s/", env["BUCKET_NAME"], ns.Name),
},
Modify: func(tmpl *ateapipb.ActorTemplate) {
tmpl.Containers[0].VolumeMounts = append(tmpl.Containers[0].VolumeMounts,
&ateapipb.VolumeMount{Name: "fixture", MountPath: mountPath})
tmpl.Volumes = append(tmpl.Volumes, &ateapipb.Volume{
Name: "fixture",
Type: "Image",
Image: &ateapipb.ImageVolumeSource{Reference: fixtureImage},
})
},
}
if _, err := clients.SubstrateK8s.ApiV1alpha1().WorkerPools(ns.Name).Create(ctx, pool, metav1.CreateOptions{}); err != nil {
t.Fatalf("creating WorkerPool: %v", err)
}
container := srcTemplate.Spec.Containers[0]
container.VolumeMounts = append(container.VolumeMounts, v1alpha1.VolumeMount{Name: "fixture", MountPath: mountPath})
volumes := append(slices.Clone(srcTemplate.Spec.Volumes), v1alpha1.Volume{
Name: "fixture",
VolumeSource: v1alpha1.VolumeSource{Image: &v1alpha1.ImageVolumeSource{Reference: fixtureImage}},
})
at := &v1alpha1.ActorTemplate{
ObjectMeta: metav1.ObjectMeta{Name: probeName, Namespace: ns.Name},
Spec: v1alpha1.ActorTemplateSpec{
Containers: []v1alpha1.Container{container},
WorkerSelector: &metav1.LabelSelector{MatchLabels: poolLabels},
SandboxClass: srcTemplate.Spec.SandboxClass,
Volumes: volumes,
SnapshotsConfig: v1alpha1.SnapshotsConfig{
Location: fmt.Sprintf("gs://%s/%s/", env["BUCKET_NAME"], ns.Name),
},
},
}
created, err := clients.SubstrateK8s.ApiV1alpha1().ActorTemplates(ns.Name).Create(ctx, at, metav1.CreateOptions{})
if err != nil {
t.Fatalf("creating ActorTemplate: %v", err)
}
e2e.WaitForTemplateReady(ctx, t, clients, ns.Name, probeName)
return created
}
// probeJSON calls a probe endpoint through the router and decodes its reply.
@@ -227,20 +206,13 @@ func TestImageVolume(t *testing.T) {
fixtureImage := buildFixtureImage(t, repo)
t.Logf("fixture image: %s", fixtureImage)
createTemplate(ctx, t, clients, ns, fixtureImage)
if _, err := clients.SubstrateAPI.CreateAtespace(ctx, &ateapipb.CreateAtespaceRequest{
Atespace: &ateapipb.Atespace{Metadata: &ateapipb.ResourceMetadata{Name: atespace}},
}); err != nil {
t.Logf("CreateAtespace (may already exist): %v", err)
}
tmpl := createTemplate(ctx, t, clients, ns, fixtureImage)
actorRef := resources.ActorRef{Atespace: atespace, Name: "iv-" + ns.Name}
if _, err := clients.SubstrateAPI.CreateActor(ctx, &ateapipb.CreateActorRequest{
Actor: &ateapipb.Actor{
Metadata: &ateapipb.ResourceMetadata{Atespace: actorRef.Atespace, Name: actorRef.Name},
ActorTemplateNamespace: ns.Name,
ActorTemplateName: probeName,
Metadata: &ateapipb.ResourceMetadata{Atespace: actorRef.Atespace, Name: actorRef.Name},
ActorTemplate: e2e.TemplateRef(tmpl),
},
}); err != nil {
t.Fatalf("CreateActor: %v", err)
@@ -20,7 +20,6 @@ import (
"fmt"
"io"
"net/http"
"path/filepath"
"strings"
"testing"
"time"
@@ -36,11 +35,14 @@ import (
"k8s.io/client-go/kubernetes"
)
// grpcEchoFixtureManifest is the ActorTemplate this suite installs to get a
// grpcEchoFixtureManifests name the fixture this suite installs to get a
// gRPC-speaking Actor. It runs the same `testserver grpc` echo origin
// grpcegress_test.go deploys as a plain pod; see
// internal/e2e/fixtures/testserver.
const grpcEchoFixtureManifest = "internal/e2e/fixtures/testserver/grpcecho.yaml.tmpl"
var grpcEchoFixtureManifests = e2e.SubstrateFixtureManifests{
Pool: "internal/e2e/fixtures/testserver/grpcecho.yaml.tmpl",
Template: "internal/e2e/fixtures/testserver/grpcecho-template.yaml.tmpl",
}
// TestIngressProtocolDowngrade pins the ingress protocol contract end to end:
// a client that negotiates HTTP/2 with the router must still be able to reach
@@ -137,7 +139,7 @@ func TestIngressGRPC(t *testing.T) {
ctx := context.Background()
fixture := deployGRPCEchoTemplate(t, ctx, env["BUCKET_NAME"])
actorName, _ := createAndResumeActor(t, ctx, "grpcingress", fixture)
actorName, _ := createAndResumeSubstrateActor(t, ctx, "grpcingress", fixture)
actorRef := resources.ActorRef{Atespace: networkingAtespace, Name: actorName}
// Cleartext h2c to the router's HTTP port, with the Actor's DNS name as the
@@ -251,45 +253,17 @@ func TestIngressGRPC(t *testing.T) {
}
// deployGRPCEchoTemplate installs the gRPC Actor fixture for the sandbox class
// under test, waits for its golden snapshot and returns it. Mirrors the
// capabilities and sizing suites: render one manifest, build and apply it
// through the repo's pinned ko, delete the same file on the way out.
func deployGRPCEchoTemplate(t *testing.T, ctx context.Context, bucket string) e2e.Fixture {
// under test, waits for its golden snapshot and returns it. The suite gets its
// own copy of the fixture: suite packages run as concurrent processes, so a
// shared one would be deleted out from under another.
func deployGRPCEchoTemplate(t *testing.T, ctx context.Context, bucket string) e2e.SubstrateFixture {
t.Helper()
root, err := e2e.FindRepoRoot()
if err != nil {
t.Fatalf("FindRepoRoot: %v", err)
}
// The suite's own copy of the fixture: suite packages run as concurrent
// processes, so a shared one would be deleted out from under another.
manifest := e2e.RenderFixtureManifest(t, grpcEchoFixtureManifest, bucket, "networking")
// KO_CONFIG_PATH is required because ko resolves .ko.yaml from its working
// directory, which here is this package rather than the repo root.
applyArgs := []string{"ko", "apply", "-f", manifest}
if e2e.KubeContext != "" {
applyArgs = append(applyArgs, "--", "--context="+e2e.KubeContext)
}
e2e.RunCmdWithEnv(t, []string{"KO_CONFIG_PATH=" + root}, filepath.Join(root, "hack/run-tool.sh"), applyArgs...)
t.Cleanup(func() {
// Deletion needs no image build, so go straight to kubectl; `ko delete`
// rejects this arg shape.
delArgs := []string{"delete", "--ignore-not-found", "-f", manifest}
if e2e.KubeContext != "" {
delArgs = append([]string{"--context=" + e2e.KubeContext}, delArgs...)
}
e2e.RunCmd(t, "kubectl", delArgs...)
})
fixture := e2e.Fixture{
Namespace: e2e.FixtureName("ate-e2e-grpcecho") + "-networking",
atespace, _ := e2e.DeploySubstrateFixture(t, ctx, e2e.GetClients(), grpcEchoFixtureManifests, bucket, "networking", false)
return e2e.SubstrateFixture{
Atespace: atespace,
Name: "grpcecho",
DeployWith: "the networking suite itself (see deployGRPCEchoTemplate)",
}
e2e.WaitForTemplateReady(ctx, t, e2e.GetClients(), fixture.Namespace, fixture.Name)
return fixture
}
// waitForGRPCRouteReady retries a unary Echo until it succeeds, riding out the
@@ -310,7 +310,7 @@ func accessLogField(line, key string) (string, bool) {
func createAndResumeActor(t *testing.T, ctx context.Context, prefix string, template e2e.Fixture) (string, *ateapipb.Actor) {
t.Helper()
actor := &ateapipb.Actor{ActorTemplateNamespace: template.Namespace, ActorTemplateName: template.Name}
actor := &ateapipb.Actor{ActorTemplate: &ateapipb.ObjectRef{Atespace: template.Namespace, Name: template.Name}}
return createAndResume(t, ctx, prefix, actor, template.Namespace+"/"+template.Name, template.DeployWith)
}
@@ -18,7 +18,6 @@ import (
"context"
"net/http"
"net/url"
"path/filepath"
"strings"
"testing"
"time"
@@ -28,49 +27,33 @@ import (
"github.com/gorilla/websocket"
)
// deployWebsocketFixture renders and applies the websocket fixture for the sandbox class.
func deployWebsocketFixture(t *testing.T) string {
// deployWebsocketFixture installs the websocket fixture for the sandbox class
// under test and waits for its golden snapshot.
func deployWebsocketFixture(t *testing.T, ctx context.Context) e2e.SubstrateFixture {
t.Helper()
root, err := e2e.FindRepoRoot()
if err != nil {
t.Fatalf("FindRepoRoot: %v", err)
}
env, err := e2e.CheckEnv("BUCKET_NAME", "KO_DOCKER_REPO")
if err != nil {
t.Fatalf("CheckEnv failed: %v", err)
}
namespace := e2e.FixtureName("ate-e2e") + "-websocket"
bucket := env["BUCKET_NAME"]
manifest := e2e.RenderFixtureManifest(t, "internal/e2e/fixtures/testserver/websocket.yaml.tmpl", bucket, "websocket")
atespace, _ := e2e.DeploySubstrateFixture(t, ctx, e2e.GetClients(), e2e.SubstrateFixtureManifests{
Pool: "internal/e2e/fixtures/testserver/websocket.yaml.tmpl",
Template: "internal/e2e/fixtures/testserver/websocket-template.yaml.tmpl",
}, env["BUCKET_NAME"], "websocket", false)
applyArgs := []string{"ko", "apply", "-f", manifest}
if e2e.KubeContext != "" {
applyArgs = append(applyArgs, "--", "--context="+e2e.KubeContext)
return e2e.SubstrateFixture{
Atespace: atespace,
Name: "websocket",
DeployWith: "the networking suite itself (see deployWebsocketFixture)",
}
e2e.RunCmdWithEnv(t, []string{"KO_CONFIG_PATH=" + root}, filepath.Join(root, "hack/run-tool.sh"), applyArgs...)
t.Cleanup(func() {
delArgs := []string{"delete", "--ignore-not-found", "-f", manifest}
if e2e.KubeContext != "" {
delArgs = append([]string{"--context=" + e2e.KubeContext}, delArgs...)
}
e2e.RunCmd(t, "kubectl", delArgs...)
})
return namespace
}
func TestWebsocketIngressPing(t *testing.T) {
ctx := context.Background()
clients := e2e.GetClients()
namespace := deployWebsocketFixture(t)
fixture := deployWebsocketFixture(t, ctx)
// Wait for the websocket ActorTemplate golden snapshot to be ready
e2e.WaitForTemplateReady(ctx, t, clients, namespace, "websocket")
actorName, _ := createAndResumeActor(t, ctx, "websocket", e2e.Fixture{Namespace: namespace, Name: "websocket"})
actorName, _ := createAndResumeSubstrateActor(t, ctx, "websocket", fixture)
rc := mustRouterClient(t, ctx)
defer rc.Close()
@@ -278,7 +278,7 @@ func TestNetworkPolicyDataPlaneEnforcement(t *testing.T) {
func setupDemoCounterTemplate(ctx context.Context, t *testing.T, clients *e2e.Clients, ns string) (string, *ateapipb.ActorTemplate) {
t.Helper()
const poolName = "counter"
at := e2e.CreateSubstrateCounterTemplate(ctx, t, clients, ns, e2e.SubstrateCounterTemplateOptions{
at := e2e.CreateSubstrateCounterTemplate(ctx, t, clients, ns, e2e.SubstrateTemplateOptions{
Atespace: ns,
Name: "counter",
PoolName: poolName,
+1 -1
View File
@@ -197,7 +197,7 @@ func createParkingFixture(ctx context.Context, t *testing.T, clients *e2e.Client
t.Fatalf("CheckEnv failed: %v", err)
}
return e2e.CreateSubstrateCounterTemplate(ctx, t, clients, nsObj.Name, e2e.SubstrateCounterTemplateOptions{
return e2e.CreateSubstrateCounterTemplate(ctx, t, clients, nsObj.Name, e2e.SubstrateTemplateOptions{
Atespace: parkingAtespace,
// Unique within the suite-shared atespace.
Name: "parking-" + nsObj.Name,
+13 -59
View File
@@ -19,23 +19,19 @@ import (
"encoding/json"
"io"
"net/http"
"path/filepath"
"testing"
"time"
"github.com/agent-substrate/substrate/internal/e2e"
"github.com/agent-substrate/substrate/internal/resources"
"github.com/agent-substrate/substrate/pkg/api/v1alpha1"
"github.com/agent-substrate/substrate/pkg/proto/ateapipb"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
const (
sizingTemplate = "probe-sized"
// The limits declared in probe-sized.yaml.tmpl. Keep these in sync with the
// manifest: the whole point of the suite is to assert the sandbox observes
// exactly what the ActorTemplate declared.
// The limits declared in probe-sized-template.yaml.tmpl. Keep these in sync
// with the manifest: the whole point of the suite is to assert the sandbox
// observes exactly what the ActorTemplate declared.
wantCPU = 2
wantMemBytes = 512 * 1024 * 1024 // 512Mi
)
@@ -73,8 +69,7 @@ func TestActorSizing_SandboxObservesDeclaredLimits(t *testing.T) {
ctx := context.Background()
clients := e2e.GetClients()
deploySizedProbe(t, env["BUCKET_NAME"])
waitForTemplateReady(t, ctx, clients)
deploySizedProbe(t, ctx, clients, env["BUCKET_NAME"])
const id = "sized-actor"
createAndResumeActor(t, ctx, clients, id)
@@ -112,62 +107,21 @@ func TestActorSizing_SandboxObservesDeclaredLimits(t *testing.T) {
}
}
func deploySizedProbe(t *testing.T, bucket string) {
// deploySizedProbe installs the sized probe fixture for the sandbox class
// under test and waits for its golden snapshot.
func deploySizedProbe(t *testing.T, ctx context.Context, clients *e2e.Clients, bucket string) {
t.Helper()
root, err := e2e.FindRepoRoot()
if err != nil {
t.Fatalf("FindRepoRoot: %v", err)
}
// One manifest, rendered for the sandbox class under test (mirrors the
// identity suite).
manifest := e2e.RenderFixtureManifest(t, "internal/e2e/fixtures/probe/probe-sized.yaml.tmpl", bucket, "sizing")
// Build/push the probe image and apply through the repo's pinned ko. See the
// identity suite's deployProbe for why KO_CONFIG_PATH and the trailing
// `-- --context=...` are required.
applyArgs := []string{"ko", "apply", "-f", manifest}
if e2e.KubeContext != "" {
applyArgs = append(applyArgs, "--", "--context="+e2e.KubeContext)
}
e2e.RunCmdWithEnv(t, []string{"KO_CONFIG_PATH=" + root}, filepath.Join(root, "hack/run-tool.sh"), applyArgs...)
t.Cleanup(func() {
delArgs := []string{"delete", "--ignore-not-found", "-f", manifest}
if e2e.KubeContext != "" {
delArgs = append([]string{"--context=" + e2e.KubeContext}, delArgs...)
}
e2e.RunCmd(t, "kubectl", delArgs...)
})
}
func waitForTemplateReady(t *testing.T, ctx context.Context, clients *e2e.Clients) {
t.Helper()
deadline := time.Now().Add(e2e.TemplateReadyTimeout(t))
for time.Now().Before(deadline) {
at, err := clients.SubstrateK8s.ApiV1alpha1().ActorTemplates(sizingNamespace).Get(ctx, sizingTemplate, metav1.GetOptions{})
if err == nil {
switch at.Status.Phase {
case v1alpha1.PhaseReady:
t.Logf("sized probe ActorTemplate ready, golden=%s", at.Status.GoldenActorID)
return
case v1alpha1.PhaseFailed:
t.Fatalf("sized probe ActorTemplate entered PhaseFailed")
}
}
time.Sleep(2 * time.Second)
}
t.Fatalf("timed out waiting for sized probe ActorTemplate to be Ready")
e2e.DeploySubstrateFixture(t, ctx, clients, e2e.SubstrateFixtureManifests{
Pool: "internal/e2e/fixtures/probe/probe-sized.yaml.tmpl",
Template: "internal/e2e/fixtures/probe/probe-sized-template.yaml.tmpl",
}, bucket, "sizing", false)
}
func createAndResumeActor(t *testing.T, ctx context.Context, clients *e2e.Clients, id string) {
t.Helper()
// CreateActor requires the atespace to exist first.
_, _ = clients.SubstrateAPI.CreateAtespace(ctx, &ateapipb.CreateAtespaceRequest{Atespace: &ateapipb.Atespace{Metadata: &ateapipb.ResourceMetadata{Name: sizingNamespace}}})
if _, err := clients.SubstrateAPI.CreateActor(ctx, &ateapipb.CreateActorRequest{Actor: &ateapipb.Actor{
Metadata: &ateapipb.ResourceMetadata{Atespace: sizingNamespace, Name: id},
ActorTemplateNamespace: sizingNamespace,
ActorTemplateName: sizingTemplate,
Metadata: &ateapipb.ResourceMetadata{Atespace: sizingNamespace, Name: id},
ActorTemplate: &ateapipb.ObjectRef{Atespace: sizingNamespace, Name: sizingTemplate},
}}); err != nil {
t.Fatalf("CreateActor %q: %v", id, err)
}
+16 -49
View File
@@ -16,7 +16,6 @@ package e2e
import (
"context"
"os"
"testing"
"time"
@@ -27,50 +26,13 @@ import (
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
// WaitForTemplateReady blocks until the ActorTemplate's golden actor has
// booted and been snapshotted. The default 5 minute timeout can be
// overridden with E2E_TEMPLATE_READY_TIMEOUT.
func WaitForTemplateReady(ctx context.Context, t *testing.T, clients *Clients, namespace, name string) {
t.Helper()
timeout := 5 * time.Minute
if v := os.Getenv("E2E_TEMPLATE_READY_TIMEOUT"); v != "" {
d, err := time.ParseDuration(v)
if err != nil {
t.Fatalf("invalid E2E_TEMPLATE_READY_TIMEOUT %q: %v", v, err)
}
timeout = d
}
ctx, cancel := context.WithTimeout(ctx, timeout)
defer cancel()
var lastPhase v1alpha1.PhaseType
for {
at, err := clients.SubstrateK8s.ApiV1alpha1().ActorTemplates(namespace).Get(ctx, name, metav1.GetOptions{})
if err == nil {
lastPhase = at.Status.Phase
if lastPhase == v1alpha1.PhaseReady {
return
}
if lastPhase == v1alpha1.PhaseFailed {
t.Fatalf("ActorTemplate %s/%s transitioned to Failed", namespace, name)
}
}
select {
case <-ctx.Done():
t.Fatalf("timed out after %v waiting for ActorTemplate %s/%s to be Ready (last phase %q, err %v)", timeout, namespace, name, lastPhase, err)
case <-time.After(time.Second):
}
}
}
// TemplateRef builds the substrate template reference an Actor carries.
func TemplateRef(at *ateapipb.ActorTemplate) *ateapipb.ObjectRef {
return &ateapipb.ObjectRef{Atespace: at.GetMetadata().GetAtespace(), Name: at.GetMetadata().GetName()}
}
// SubstrateCounterTemplateOptions shapes CreateSubstrateCounterTemplate.
type SubstrateCounterTemplateOptions struct {
// SubstrateTemplateOptions shapes CreateSubstrateTemplateFrom.
type SubstrateTemplateOptions struct {
// Atespace and Name locate the new template. The atespace is created if
// missing; Name must be unique within it (atespaces are shared across a
// suite's tests, unlike the k8s namespaces the CRD templates lived in).
@@ -90,15 +52,20 @@ type SubstrateCounterTemplateOptions struct {
}
// CreateSubstrateCounterTemplate creates a per-test WorkerPool CRD plus a
// substrate ActorTemplate copying the resolved runtime (sandbox config, ateom
// image, container images, sandbox size) from the substrate counter demo for
// the sandbox class under test. It registers cleanup of the template (which
// does not ride the k8s namespace GC the CRD templates did) and blocks until
// the golden snapshot exists.
func CreateSubstrateCounterTemplate(ctx context.Context, t *testing.T, clients *Clients, namespace string, opts SubstrateCounterTemplateOptions) *ateapipb.ActorTemplate {
// substrate ActorTemplate copying the resolved runtime from the substrate
// counter demo for the sandbox class under test.
func CreateSubstrateCounterTemplate(ctx context.Context, t *testing.T, clients *Clients, namespace string, opts SubstrateTemplateOptions) *ateapipb.ActorTemplate {
t.Helper()
return CreateSubstrateTemplateFrom(ctx, t, clients, namespace, SubstrateCounterFixture(), opts)
}
src := SubstrateCounterFixture()
// CreateSubstrateTemplateFrom creates a per-test WorkerPool CRD plus a
// substrate ActorTemplate copying the resolved runtime (sandbox config, ateom
// image, container images, sandbox size) from the installed fixture src. It
// registers cleanup of the template (which does not ride the k8s namespace GC
// the CRD templates did) and blocks until the golden snapshot exists.
func CreateSubstrateTemplateFrom(ctx context.Context, t *testing.T, clients *Clients, namespace string, src SubstrateFixture, opts SubstrateTemplateOptions) *ateapipb.ActorTemplate {
t.Helper()
existingWp, err := clients.SubstrateK8s.ApiV1alpha1().WorkerPools(src.PoolNamespace).Get(ctx, src.PoolName, metav1.GetOptions{})
if err != nil {
@@ -180,8 +147,8 @@ func CreateSubstrateCounterTemplate(ctx context.Context, t *testing.T, clients *
}
// WaitForSubstrateTemplateReady blocks until the substrate ActorTemplate's
// golden snapshot exists, the same readiness the CRD phase poll above proves.
// The timeout follows the sandbox class under test (see TemplateReadyTimeout).
// golden snapshot exists. The timeout follows the sandbox class under test
// (see TemplateReadyTimeout).
func WaitForSubstrateTemplateReady(ctx context.Context, t *testing.T, clients *Clients, atespace, name string) {
t.Helper()