mirror of
https://github.com/NVIDIA/OpenShell.git
synced 2026-10-02 07:34:45 +08:00
* feat(isolation): add RFC 0012 backend contract Signed-off-by: Drew Newberry <385+drew@users.noreply.github.com> * refactor(isolation): name the interface crate explicitly Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(isolation): expose trusted host gateway Signed-off-by: Drew Newberry <anewberry@nvidia.com> * docs(agents): inventory the MXC driver Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(isolation): add mediated DNS transport Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): tighten interface error and digest contracts Signed-off-by: Drew Newberry <anewberry@nvidia.com> * docs(isolation): remove unrelated driver inventory Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(isolation): define capability-free launch contract Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): seal confirmed boundary state Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): validate confirmation for external backend implementations Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): clarify mediated DNS identity Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(isolation): generalize loopback connector Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(isolation): unify typed network mediation Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(isolation): bind launches to sandbox sessions Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(mxc): initialize extended sandbox status Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(isolation): add boundary protocol and Linux primitives Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): harden signals and separate process status from transport Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): validate remote confirmation through public contract Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): validate wire state and propagate snapshot failures Signed-off-by: Drew Newberry <anewberry@nvidia.com> * test(isolation): import owned agent specification explicitly Signed-off-by: Drew Newberry <anewberry@nvidia.com> * docs(isolation): describe mediated DNS channel Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): bound mediation attach without nested retries Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(isolation): generalize loopback protocol Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(isolation): add transport-neutral session authentication Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(isolation): separate sandbox backend protocol Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): harden runtime boundary controls Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(isolation): add terminal boundary operation Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(isolation): split supervisor and sandbox runtimes Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(sandbox): harden boundary isolation and lifecycle ownership Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(sandbox): reject private root redirects and adopt typed errors Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(sandbox): preserve accept thread ownership on musl Signed-off-by: Drew Newberry <anewberry@nvidia.com> * test(sandbox): isolate credential probes from filtered threads Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(sandbox): return retained exec exit status to independent waiters Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(sandbox): bound network mediation and preserve socket authorization Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(sandbox): bound control admission and retire stale mediation Signed-off-by: Drew Newberry <anewberry@nvidia.com> * ci(e2e): select migrated drivers per stack layer Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(sandbox): implement loopback connector Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(isolation): authenticate the Sandbox Protocol Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(supervisor): rotate launch-scoped authentication Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(sandbox): consume dedicated backend crate Signed-off-by: Drew Newberry <anewberry@nvidia.com> * test(sandbox): align topology session fixture Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(sandbox): align projected bootstrap bundle Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(auth): validate refreshed credentials before rotation Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(sandbox): fail closed across supervisor disconnects Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): repair rebased sandbox CI Signed-off-by: Drew Newberry <anewberry@nvidia.com> * build(runtime): publish separate sandbox and supervisor images Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(config): configure the sandbox runtime image Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(ci): validate sandbox binary linkage Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(isolation): use backend and runtime terminology Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(sandbox): use a scratch runtime image Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(ci): refresh schema and dependency policy Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(sandbox): bind reconnects to supervisor process Signed-off-by: Drew Newberry <anewberry@nvidia.com> * docs: align runtime split operational guidance Signed-off-by: Drew Newberry <anewberry@nvidia.com> * chore(security): document Kubernetes runtime RBAC Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): enforce runtime lifecycle invariants Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(compute): identify sandbox start generations Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(server): restore sandbox launch sessions Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): support authenticated runtime replacement Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(auth): bind sandbox session successors Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(auth): retry pending sandbox successors Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(vm): run the supervisor outside the guest workload Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(vm): use sandbox backend protocol Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): repair rebase integration Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): use unified build toolchain Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(vm): use sandbox runtime terminology Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(vm): own guest network bootstrap Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): expose guest init version Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): select native supervisor artifacts Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): guard guest init Linux symbols Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): scope Linux test imports Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): avoid guest interface casts Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): reconcile admitted sandbox identity Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): share resolved sandbox identity Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): surface host supervisor failures Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): include guest logs on supervisor exit Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): rotate and clean runtime generations Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): make sandbox starts generation-aware Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): rotate restored sandbox sessions Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(vm): keep shared paths in the base layer Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): bind sandbox session lineage Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(docker): isolate workloads behind the host supervisor Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(docker): rotate launch-scoped authentication Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(docker): use sandbox backend protocol Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(docker): use host networking for supervisor Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(docker): preserve host gateway alias resolution Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(docker): use separate sandbox and supervisor images Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(docker): restore startup validation after rebase Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(docker): name the sandbox runtime directly Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(docker): narrow supervisor CA runtime storage Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(docker): close companion isolation gaps Signed-off-by: Drew Newberry <anewberry@nvidia.com> * test(docker): align mediated network expectations Signed-off-by: Drew Newberry <anewberry@nvidia.com> * test(docker): exercise mediated network paths Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(docker): attach supervisor to managed network Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(docker): defer supervisor recovery until gateway is ready Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(docker): make sandbox starts generation-aware Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(docker): rotate restored sandbox sessions Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(docker): preserve workloads during session rotation Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(docker): remove unrelated configuration RFC changes Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(docker): bind sandbox session lineage Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(kubernetes): add proxy-pod isolation topology Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(kubernetes): use sandbox backend protocol Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(kubernetes): use stable sandbox service authority Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(kubernetes): split sandbox and supervisor images Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(kubernetes): adapt proxy pods to current runtime APIs Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(kubernetes): describe the single runtime placement Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(kubernetes): simplify sandbox orchestration Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(kubernetes): validate deployment prerequisites Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(kubernetes): update Trivy Helm profile inventory Signed-off-by: Drew Newberry <anewberry@nvidia.com> * test(kubernetes): update Trivy scan inventory count Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(kubernetes): reuse preloaded runtime images in e2e Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(kubernetes): type and clean runtime resources Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(kubernetes): make sandbox restarts recoverable Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(kubernetes): rotate restored sandbox sessions Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(kubernetes): preserve supervisor egress Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(kubernetes): bind sandbox session lineage Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(podman): adopt isolated sandbox and supervisor containers Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): stage bootstrap archives at named volume destinations Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(podman): rotate launch-scoped authentication Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(podman): use sandbox backend protocol Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(podman): use host networking for supervisor Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(podman): split sandbox and supervisor images Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): repair rebase integration Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(podman): name the sandbox runtime directly Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): provision supervisor CA runtime storage Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): address isolation review findings Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): inspect Debian supervisor provenance Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): use libpod-compatible tmpfs options Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): bind verified sandbox runtime binary Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): provide external driver data directory Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): start sandbox before joining user namespace Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): separate supervisor user namespace Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): make sandbox starts generation-aware Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): rotate restored sandbox sessions Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): bind sandbox session lineage Signed-off-by: Drew Newberry <anewberry@nvidia.com> * perf(isolation): add TCP and DNS benchmark harnesses Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(perf): align benchmark timing and supported protocols Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(perf): report TCP benchmark metrics accurately Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(perf): cancel failed worker startup Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(docker): build matching local supervisor image Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(podman): make local sandbox smoke test runnable Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(kubernetes): wire local sandbox runtime image Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(kubernetes): narrow sandbox service RBAC Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(ci): validate split runtime artifacts Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): harden runtime session handling Signed-off-by: Drew Newberry <anewberry@nvidia.com> * feat(supervisor): add standalone network proxy role Signed-off-by: Drew Newberry <anewberry@nvidia.com> * docs(rfc): remove implementation companion notes Signed-off-by: Drew Newberry <anewberry@nvidia.com> * refactor(vm): standardize runtime release name Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(vm): pin renamed runtime artifacts Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(auth): persist sandbox runtime identity Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(runtime): restore branch validation Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(isolation): reconcile main after rebase Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(network): close unframed HTTP 1.0 responses Signed-off-by: Drew Newberry <anewberry@nvidia.com> * chore(isolation): preserve upstream OCSF updates Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(security): close credential and TLS replay paths Signed-off-by: Drew Newberry <anewberry@nvidia.com> * fix(auth): make sandbox refresh retries idempotent Signed-off-by: Drew Newberry <anewberry@nvidia.com> --------- Signed-off-by: Drew Newberry <385+drew@users.noreply.github.com> Signed-off-by: Drew Newberry <anewberry@nvidia.com>
231 lines
8.3 KiB
Rust
231 lines
8.3 KiB
Rust
// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
|
// SPDX-License-Identifier: Apache-2.0
|
|
|
|
#![cfg(feature = "e2e")]
|
|
|
|
//! E2E coverage for reconciling Docker sandboxes after a standalone gateway restart.
|
|
//!
|
|
//! This intentionally targets the Docker-driver gateway started by
|
|
//! `e2e/with-docker-gateway.sh`. Existing-endpoint E2E runs do not own the
|
|
//! gateway process, so they skip this restart-only coverage.
|
|
|
|
use std::process::{Command, Stdio};
|
|
use std::time::Duration;
|
|
|
|
use openshell_e2e::harness::cli::{
|
|
run_cli, sandbox_names, wait_for_healthy, wait_for_sandbox_exec_contains,
|
|
wait_for_sandbox_phase,
|
|
};
|
|
use openshell_e2e::harness::gateway::ManagedGateway;
|
|
use openshell_e2e::harness::sandbox::SandboxGuard;
|
|
use tokio::time::sleep;
|
|
|
|
const MANAGED_BY_LABEL_FILTER: &str = "label=openshell.ai/managed-by=openshell";
|
|
const READY_MARKER: &str = "gateway-start-ready";
|
|
const STOPPED_READY_MARKER: &str = "gateway-start-stopped-ready";
|
|
const START_FILE: &str = "/sandbox/gateway-start-state";
|
|
const SANDBOX_NAMESPACE_LABEL: &str = "openshell.ai/sandbox-namespace";
|
|
const SANDBOX_NAME_LABEL: &str = "openshell.ai/sandbox-name";
|
|
const SANDBOX_ROLE_LABEL_FILTER: &str = "label=openshell.ai/isolation-role=sandbox";
|
|
|
|
fn sandbox_container_id(namespace: &str, sandbox_name: &str) -> Result<String, String> {
|
|
let namespace_filter = format!("label={SANDBOX_NAMESPACE_LABEL}={namespace}");
|
|
let sandbox_name_filter = format!("label={SANDBOX_NAME_LABEL}={sandbox_name}");
|
|
let output = Command::new("docker")
|
|
.args([
|
|
"ps",
|
|
"-aq",
|
|
"--filter",
|
|
MANAGED_BY_LABEL_FILTER,
|
|
"--filter",
|
|
SANDBOX_ROLE_LABEL_FILTER,
|
|
"--filter",
|
|
])
|
|
.arg(namespace_filter)
|
|
.args(["--filter"])
|
|
.arg(sandbox_name_filter)
|
|
.stdout(Stdio::piped())
|
|
.stderr(Stdio::piped())
|
|
.output()
|
|
.map_err(|err| format!("failed to run docker ps: {err}"))?;
|
|
let stdout = String::from_utf8_lossy(&output.stdout);
|
|
let stderr = String::from_utf8_lossy(&output.stderr);
|
|
let combined = format!("{stdout}{stderr}");
|
|
if !output.status.success() {
|
|
return Err(format!(
|
|
"docker ps failed (exit {:?}):\n{combined}",
|
|
output.status.code()
|
|
));
|
|
}
|
|
|
|
let ids = stdout
|
|
.lines()
|
|
.map(str::trim)
|
|
.filter(|line| !line.is_empty())
|
|
.collect::<Vec<_>>();
|
|
match ids.as_slice() {
|
|
[id] => Ok((*id).to_string()),
|
|
[] => Err(format!(
|
|
"no Docker container found for sandbox '{sandbox_name}' in namespace '{namespace}'"
|
|
)),
|
|
_ => Err(format!(
|
|
"multiple Docker containers found for sandbox '{sandbox_name}' in namespace '{namespace}': {ids:?}"
|
|
)),
|
|
}
|
|
}
|
|
|
|
fn sandbox_container_running(namespace: &str, sandbox_name: &str) -> Result<bool, String> {
|
|
let container_id = sandbox_container_id(namespace, sandbox_name)?;
|
|
let output = Command::new("docker")
|
|
.args(["inspect", "-f", "{{.State.Running}}", &container_id])
|
|
.stdout(Stdio::piped())
|
|
.stderr(Stdio::piped())
|
|
.output()
|
|
.map_err(|err| format!("failed to run docker inspect: {err}"))?;
|
|
let stdout = String::from_utf8_lossy(&output.stdout);
|
|
let stderr = String::from_utf8_lossy(&output.stderr);
|
|
let combined = format!("{stdout}{stderr}");
|
|
if !output.status.success() {
|
|
return Err(format!(
|
|
"docker inspect failed (exit {:?}):\n{combined}",
|
|
output.status.code()
|
|
));
|
|
}
|
|
|
|
match stdout.trim() {
|
|
"true" => Ok(true),
|
|
"false" => Ok(false),
|
|
other => Err(format!(
|
|
"unexpected Docker running state for container {container_id}: {other}"
|
|
)),
|
|
}
|
|
}
|
|
|
|
async fn wait_for_container_running(
|
|
namespace: &str,
|
|
sandbox_name: &str,
|
|
expected: bool,
|
|
timeout: Duration,
|
|
) -> Result<(), String> {
|
|
let start = std::time::Instant::now();
|
|
let mut last_state: String;
|
|
|
|
loop {
|
|
match sandbox_container_running(namespace, sandbox_name) {
|
|
Ok(running) if running == expected => return Ok(()),
|
|
Ok(running) => last_state = format!("running={running}"),
|
|
Err(err) => last_state = err,
|
|
}
|
|
|
|
if start.elapsed() > timeout {
|
|
return Err(format!(
|
|
"sandbox container '{sandbox_name}' did not reach running={expected} within {}s. Last state: {last_state}",
|
|
timeout.as_secs()
|
|
));
|
|
}
|
|
sleep(Duration::from_secs(1)).await;
|
|
}
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn docker_gateway_restart_preserves_running_and_stopped_intent() {
|
|
let Some(gateway) = ManagedGateway::from_env().expect("load managed e2e gateway metadata")
|
|
else {
|
|
eprintln!("Skipping gateway start test: e2e gateway is not managed by this test run");
|
|
return;
|
|
};
|
|
let Some(namespace) = std::env::var("OPENSHELL_E2E_DOCKER_NETWORK_NAME")
|
|
.ok()
|
|
.filter(|value| !value.trim().is_empty())
|
|
else {
|
|
eprintln!("Skipping gateway start test: Docker e2e namespace is unavailable");
|
|
return;
|
|
};
|
|
|
|
wait_for_healthy(Duration::from_secs(30))
|
|
.await
|
|
.expect("gateway should start healthy");
|
|
|
|
let script = format!(
|
|
"echo before-restart > {START_FILE}; echo {READY_MARKER}; while true; do sleep 1; done"
|
|
);
|
|
let mut sandbox = SandboxGuard::create_keep(&["sh", "-lc", &script], READY_MARKER)
|
|
.await
|
|
.expect("create long-running sandbox");
|
|
|
|
let before_restart = sandbox
|
|
.exec(&["cat", START_FILE])
|
|
.await
|
|
.expect("read sandbox state before restart");
|
|
assert!(
|
|
before_restart.contains("before-restart"),
|
|
"sandbox state was not written before restart:\n{before_restart}"
|
|
);
|
|
|
|
wait_for_container_running(&namespace, &sandbox.name, true, Duration::from_secs(60))
|
|
.await
|
|
.expect("sandbox container should be running before gateway restart");
|
|
|
|
let stopped_script = format!("echo {STOPPED_READY_MARKER}; while true; do sleep 1; done");
|
|
let mut stopped_sandbox =
|
|
SandboxGuard::create_keep(&["sh", "-lc", &stopped_script], STOPPED_READY_MARKER)
|
|
.await
|
|
.expect("create Docker sandbox that will remain stopped");
|
|
let (stop_output, stop_code) = run_cli(&["sandbox", "stop", &stopped_sandbox.name]).await;
|
|
assert_eq!(stop_code, 0, "sandbox stop should succeed:\n{stop_output}");
|
|
wait_for_sandbox_phase(&stopped_sandbox.name, "Stopped", Duration::from_secs(30))
|
|
.await
|
|
.expect("sandbox should be stopped before gateway restart");
|
|
wait_for_container_running(
|
|
&namespace,
|
|
&stopped_sandbox.name,
|
|
false,
|
|
Duration::from_secs(30),
|
|
)
|
|
.await
|
|
.expect("stopped Docker sandbox container should not be running");
|
|
|
|
gateway.stop().expect("stop e2e gateway");
|
|
wait_for_container_running(&namespace, &sandbox.name, false, Duration::from_secs(30))
|
|
.await
|
|
.expect("gateway shutdown should stop a running-intent Docker sandbox");
|
|
|
|
gateway.start().expect("restart e2e gateway");
|
|
wait_for_healthy(Duration::from_secs(120))
|
|
.await
|
|
.expect("gateway should become healthy after restart");
|
|
wait_for_container_running(&namespace, &sandbox.name, true, Duration::from_secs(120))
|
|
.await
|
|
.expect("gateway startup should restart the running-intent Docker sandbox container");
|
|
wait_for_sandbox_phase(&stopped_sandbox.name, "Stopped", Duration::from_secs(120))
|
|
.await
|
|
.expect("explicitly stopped Docker sandbox should remain stopped after restart");
|
|
wait_for_container_running(
|
|
&namespace,
|
|
&stopped_sandbox.name,
|
|
false,
|
|
Duration::from_secs(30),
|
|
)
|
|
.await
|
|
.expect("gateway startup should not start an explicitly stopped Docker sandbox");
|
|
|
|
let names = sandbox_names().await.expect("list sandboxes after restart");
|
|
assert!(
|
|
names.contains(&sandbox.name),
|
|
"sandbox '{}' should still be listed after gateway restart. Names: {names:?}",
|
|
sandbox.name
|
|
);
|
|
|
|
wait_for_sandbox_exec_contains(
|
|
&sandbox.name,
|
|
&["cat", START_FILE],
|
|
"before-restart",
|
|
Duration::from_secs(240),
|
|
)
|
|
.await
|
|
.expect("sandbox should become ready again with its state preserved");
|
|
|
|
sandbox.cleanup().await;
|
|
stopped_sandbox.cleanup().await;
|
|
}
|