mirror of
https://github.com/NVIDIA/OpenShell.git
synced 2026-10-04 08:28:19 +08:00
feat(kubernetes): add sidecar supervisor topology (#2076)
* feat(kubernetes): add sidecar supervisor topology Add the Kubernetes sidecar supervisor topology, its Helm/Skaffold configuration, topology documentation, and sidecar e2e matrix coverage. Skip root-only sandbox identity rewriting when process enforcement is network-only so the low-permission sidecar process container can start successfully. Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(supervisor): avoid similar process id names Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(supervisor): avoid similar process id names Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(sandbox): avoid similar proxy id names Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * docs(kubernetes): clarify sidecar topology limits Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(kubernetes): keep sidecar process leaf capless Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(kubernetes): refresh sidecar provider env snapshots Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * test(supervisor): align hot-swap identity regression Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(kubernetes): stage sidecar mtls files before proxy chown Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(kubernetes): simplify sidecar supervisor topology Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * chore(helm): reuse sidecar skaffold values Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(supervisor): avoid similar iptables helper names Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(e2e): harden kube gateway wrapper setup Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(supervisor): avoid nft batch rollback on OCP Run nftables setup as individual commands so optional conntrack and log expressions can fail without rolling back required table, chain, and reject rules. Signed-off-by: Seth Jennings <sjenning@redhat.com> Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(kubernetes): preserve process identity in sidecar topology Render sidecar pods with a shared process namespace, keep binary-aware network policy enabled, and move Kubernetes sidecar settings under the nested sidecar config table. Also apply unprivileged Landlock/seccomp setup in NetworkOnly supervisor mode so sidecar topology keeps sandbox child hardening without privileged process setup. Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * refactor(kubernetes): replace sidecar snapshots with control socket Coordinate sidecar policy and provider bootstrap over a local Unix socket so the process leaf no longer reads policy/provider snapshot files. Report entrypoint startup through the control channel and keep gateway credentials confined to the network sidecar. Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * feat(kubernetes): support relaxed sidecar network identity Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(sandbox): satisfy sidecar clippy lint Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * refactor(kubernetes): standardize topology naming Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(sandbox): satisfy linux clippy timeout import Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(kubernetes): support kata sidecar on ipv4 pods Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(kubernetes): satisfy linux clippy for sidecar fallback Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * chore(kubernetes): remove stale supervisor topology references Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(kubernetes): enable sidecar binary policy inspection Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(kubernetes): harden sidecar control boundary Signed-off-by: Taylor Mutch <taylormutch@gmail.com> * fix(kubernetes): couple sidecar supervisor lifecycles Signed-off-by: Taylor Mutch <taylormutch@gmail.com> --------- Signed-off-by: Taylor Mutch <taylormutch@gmail.com> Signed-off-by: Seth Jennings <sjenning@redhat.com> Co-authored-by: Seth Jennings <sjenning@redhat.com>
This commit is contained in:
co-authored by
Seth Jennings
parent
bebf440b25
commit
8eacb4779f
@@ -268,6 +268,57 @@ kubectl -n openshell get configmap openshell-config -o jsonpath='{.data.gateway\
|
||||
kubectl -n <sandbox-namespace> get sandbox <sandbox-name> -o jsonpath='{.spec.template.spec.serviceAccountName}{"\n"}'
|
||||
```
|
||||
|
||||
If `topology = "sidecar"` is rendered under `[openshell.drivers.kubernetes]`,
|
||||
sandbox pods should have an `openshell-network-init` init container running
|
||||
`--mode=network-init`, an `agent` container running
|
||||
`openshell-sandbox --mode=process`, and an `openshell-supervisor-network`
|
||||
container running `--mode=network`. The init container owns nftables setup and
|
||||
should be the only sidecar topology container with `NET_ADMIN`. It also needs
|
||||
`CHOWN`/`FOWNER` to hand shared emptyDir state to the effective sidecar UID. The
|
||||
default binary-aware network sidecar runs as UID 0 with primary GID
|
||||
`sandbox_gid` and adds `SYS_PTRACE` plus `DAC_READ_SEARCH`. When
|
||||
`process_binary_aware_network_policy = false`, it runs as the configured
|
||||
non-root `proxy_uid` without those inspection capabilities. The pod `fsGroup`
|
||||
is set to `sandbox_gid` in both modes.
|
||||
|
||||
In sidecar topology only the network sidecar should mount the gateway bootstrap
|
||||
credentials (`openshell-sa-token` and `openshell-client-tls`). The process
|
||||
container should not receive `OPENSHELL_ENDPOINT`, gateway TLS env vars, the
|
||||
sandbox token file, or those credential mounts. Instead, the network sidecar
|
||||
serves policy and provider environment state over the Unix control socket from
|
||||
`OPENSHELL_SIDECAR_CONTROL_SOCKET` (`/run/openshell-sidecar/control.sock` by
|
||||
default). The process supervisor must be the first and only client. After
|
||||
validating its peer UID, GID, and PID, the sidecar unlinks the listener. If the
|
||||
connection later closes, the network sidecar exits non-zero so Kubernetes can
|
||||
restart it with a fresh listener. If the process supervisor fails before
|
||||
launching the workload,
|
||||
inspect both containers for control-socket bind, connect, bootstrap, or update
|
||||
errors. If new SSH/exec sessions do not pick up refreshed provider environment,
|
||||
inspect the network sidecar settings-poll logs and the process container logs
|
||||
for provider environment update handling; the process container should consume
|
||||
newer provider-env revisions without receiving gateway credentials.
|
||||
|
||||
The process container reports the workload entrypoint PID over the same control
|
||||
socket, and the network sidecar uses that PID for binary-scoped policy
|
||||
decisions through `/proc`. If rules with `policy.binaries` are unexpectedly
|
||||
denied, inspect the sidecar control logs and confirm the pod has
|
||||
`shareProcessNamespace: true`.
|
||||
The shared state directory should preserve `sandbox_gid` inheritance
|
||||
(`02775`). Sidecar SSH uses the Linux abstract socket
|
||||
`@openshell-sidecar-ssh`; the network sidecar verifies its peer PID before
|
||||
bridging gateway relay requests. No `ssh.sock` file should appear in the shared
|
||||
state directory.
|
||||
Inspect all three when sandbox registration or egress enforcement fails:
|
||||
|
||||
```bash
|
||||
kubectl -n openshell get configmap openshell-config -o jsonpath='{.data.gateway\.toml}' | grep -E '^\[openshell\.drivers\.kubernetes\]|^topology\s*='
|
||||
kubectl -n <sandbox-namespace> get pod <sandbox-pod> -o jsonpath='{range .spec.initContainers[*]}{.name}{" "}{.command}{"\n"}{end}'
|
||||
kubectl -n <sandbox-namespace> get pod <sandbox-pod> -o jsonpath='{range .spec.containers[*]}{.name}{" "}{.command}{"\n"}{end}'
|
||||
kubectl -n <sandbox-namespace> logs <sandbox-pod> -c openshell-network-init --tail=200
|
||||
kubectl -n <sandbox-namespace> logs <sandbox-pod> -c openshell-supervisor-network --tail=200
|
||||
kubectl -n <sandbox-namespace> logs <sandbox-pod> -c agent --tail=200
|
||||
```
|
||||
|
||||
### Step 6: Check VM-Backed Gateways
|
||||
|
||||
Use the VM driver logs and host diagnostics available in the user's environment. Verify:
|
||||
|
||||
@@ -60,9 +60,25 @@ mise run helm:skaffold:dev
|
||||
mise run helm:skaffold:run
|
||||
```
|
||||
|
||||
**Supervisor sidecar topology** (build once and leave running):
|
||||
```bash
|
||||
mise run helm:skaffold:run:sidecar
|
||||
```
|
||||
|
||||
**Supervisor sidecar topology with TLS/mTLS enabled** (build once and leave running):
|
||||
```bash
|
||||
mise run helm:skaffold:run:sidecar-mtls
|
||||
```
|
||||
|
||||
Both commands build the `gateway` and `supervisor` images and deploy the OpenShell Helm
|
||||
chart. The `pkiInitJob` hook (a pre-install Job that runs `openshell-gateway generate-certs`)
|
||||
generates mTLS secrets on first install. Envoy Gateway opt-in; see the Optional Add-ons section below.
|
||||
chart. The sidecar profile renders an `openshell-network-init` init container for
|
||||
nftables setup and an `openshell-supervisor-network` runtime sidecar for proxying.
|
||||
Binary-aware policy mode runs that sidecar as UID 0 with `SYS_PTRACE` and
|
||||
`DAC_READ_SEARCH`; relaxed mode can run it as the configured proxy UID. The
|
||||
sidecar-mTLS profile reuses `ci/values-sidecar.yaml` and restores
|
||||
`server.disableTls=false` inline for Skaffold. The `pkiInitJob` hook (a pre-install
|
||||
Job that runs `openshell-gateway generate-certs`) generates mTLS secrets on first
|
||||
install. Envoy Gateway opt-in; see the Optional Add-ons section below.
|
||||
|
||||
The gateway Service uses ClusterIP. Access is via Envoy Gateway (port `8080`) or `kubectl port-forward`.
|
||||
|
||||
@@ -74,7 +90,8 @@ create the Secret named `openshell-ha-pg` with a `uri` key, then run
|
||||
### TLS behaviour
|
||||
|
||||
`ci/values-skaffold.yaml` sets `server.disableTls: true`, so Skaffold-based deploys run
|
||||
plaintext by default. To test with TLS enabled, comment out that line and redeploy.
|
||||
plaintext by default. To test sidecar topology with TLS enabled, use
|
||||
`mise run helm:skaffold:run:sidecar-mtls`.
|
||||
|
||||
| Mode | `server.disableTls` | Gateway scheme |
|
||||
|------|---------------------|----------------|
|
||||
@@ -126,6 +143,12 @@ openshell sandbox list --gateway-endpoint https://localhost:8090
|
||||
mise run helm:skaffold:delete
|
||||
```
|
||||
|
||||
For a sidecar-profile deployment:
|
||||
|
||||
```bash
|
||||
mise run helm:skaffold:delete:sidecar
|
||||
```
|
||||
|
||||
### Delete the cluster entirely
|
||||
|
||||
```bash
|
||||
@@ -250,6 +273,7 @@ for dependencies still declared in `Chart.yaml`.
|
||||
| `deploy/helm/openshell/ci/values-gateway.yaml` | Envoy Gateway GRPCRoute + Gateway overlay |
|
||||
| `deploy/helm/openshell/ci/values-high-availability.yaml` | HA test overlay (`replicaCount: 2` with external PostgreSQL Secret) |
|
||||
| `deploy/helm/openshell/ci/values-keycloak.yaml` | Keycloak OIDC overlay |
|
||||
| `deploy/helm/openshell/ci/values-sidecar.yaml` | Supervisor sidecar topology overlay for Kubernetes e2e/dev |
|
||||
| `deploy/helm/openshell/ci/values-spire.yaml` | SPIFFE/SPIRE provider token grant overlay |
|
||||
| `deploy/helm/openshell/ci/values-spire-stack.yaml` | SPIRE hardened chart values for local dev |
|
||||
| `deploy/helm/openshell/ci/values-tls-disabled.yaml` | Lint-only: TLS + auth disabled (reverse-proxy edge termination) |
|
||||
|
||||
@@ -121,16 +121,25 @@ jobs:
|
||||
include:
|
||||
- agent_sandbox_api: v1beta1
|
||||
agent_sandbox_version: v0.5.0
|
||||
topology: combined
|
||||
extra_helm_values: ""
|
||||
- agent_sandbox_api: v1alpha1
|
||||
agent_sandbox_version: v0.4.6
|
||||
topology: combined
|
||||
extra_helm_values: ""
|
||||
- agent_sandbox_api: v1beta1
|
||||
agent_sandbox_version: v0.5.0
|
||||
topology: sidecar
|
||||
extra_helm_values: deploy/helm/openshell/ci/values-sidecar.yaml
|
||||
permissions:
|
||||
contents: read
|
||||
packages: read
|
||||
uses: ./.github/workflows/e2e-kubernetes-test.yml
|
||||
with:
|
||||
image-tag: ${{ github.sha }}
|
||||
job-name: Kubernetes E2E (Rust smoke, Agent Sandbox ${{ matrix.agent_sandbox_api }})
|
||||
job-name: Kubernetes E2E (Rust smoke, ${{ matrix.topology }}, Agent Sandbox ${{ matrix.agent_sandbox_api }})
|
||||
agent-sandbox-version: ${{ matrix.agent_sandbox_version }}
|
||||
extra-helm-values: ${{ matrix.extra_helm_values }}
|
||||
|
||||
kubernetes-ha-e2e:
|
||||
needs: [pr_metadata, build-gateway, build-supervisor]
|
||||
|
||||
Generated
+3
@@ -3827,12 +3827,15 @@ dependencies = [
|
||||
"clap",
|
||||
"futures",
|
||||
"miette",
|
||||
"nix",
|
||||
"openshell-core",
|
||||
"openshell-ocsf",
|
||||
"openshell-policy",
|
||||
"openshell-supervisor-network",
|
||||
"openshell-supervisor-process",
|
||||
"prost",
|
||||
"rustls",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"temp-env",
|
||||
"tempfile",
|
||||
|
||||
@@ -91,10 +91,11 @@ Runtime layout:
|
||||
as a release artifact. Linux GNU VM driver binaries must not reference
|
||||
`GLIBC_*` symbols newer than `GLIBC_2.28`; release workflows verify this
|
||||
before publishing artifacts.
|
||||
- **Supervisor**: `scratch` base, static musl binary at `/openshell-sandbox`.
|
||||
Static linkage is required because the image is mounted/extracted into
|
||||
sandbox environments (Docker extraction, Podman image volumes, Kubernetes
|
||||
init-container copy-self) and cannot rely on a dynamic loader.
|
||||
- **Supervisor**: Alpine base with `nftables`, static musl binary at
|
||||
`/openshell-sandbox`. Static linkage keeps the binary usable when the image
|
||||
is mounted/extracted into sandbox environments (Docker extraction, Podman
|
||||
image volumes, Kubernetes init-container copy-self), while `nftables` supports
|
||||
Kubernetes supervisor sidecar egress enforcement.
|
||||
|
||||
Gateway image builds bake the corresponding supervisor image tag into the
|
||||
gateway binary so Docker sandboxes do not depend on `:latest` by default.
|
||||
|
||||
@@ -81,7 +81,7 @@ The supervisor must be available inside each sandbox workload:
|
||||
|---|---|
|
||||
| Docker | Bind-mounted local supervisor binary, or a binary extracted from the configured supervisor image. |
|
||||
| Podman | Read-only OCI image volume containing the supervisor binary. |
|
||||
| Kubernetes | Sandbox pod image or pod template configuration. |
|
||||
| Kubernetes | Supervisor image side-loaded into the sandbox pod by image volume or init container. |
|
||||
| VM | Embedded in the guest rootfs bundle. |
|
||||
| Extension | Defined by the out-of-tree driver. |
|
||||
|
||||
@@ -89,6 +89,44 @@ Driver-controlled environment variables must override sandbox image or template
|
||||
values for sandbox ID, sandbox name, gateway endpoint, relay socket path, TLS
|
||||
paths, and command metadata.
|
||||
|
||||
Kubernetes can run the supervisor in the default combined topology or in a
|
||||
sidecar topology. Combined mode keeps network and process supervision in the
|
||||
agent container. Sidecar mode runs network enforcement, the proxy, and gateway
|
||||
session in a dedicated sidecar, while the agent container runs only the
|
||||
process-supervision leaf and launches the user workload after the sidecar
|
||||
serves bootstrap state over a local control socket. The network sidecar owns
|
||||
gateway credentials and sends policy plus workload-facing provider environment
|
||||
state to the process leaf over that socket. It also streams provider
|
||||
environment updates after settings polls so future process sessions see
|
||||
updated provider env without giving the process leaf gateway access. The
|
||||
pre-workload process supervisor is the only accepted control client: the
|
||||
network sidecar verifies its UID, GID, and PID with peer credentials, removes
|
||||
the listener after accepting it, and ignores workload-supplied relay targets.
|
||||
SSH relays use a Linux abstract socket and verify its peer PID against that
|
||||
authenticated process-supervisor connection, so workload filesystem access
|
||||
cannot replace the relay endpoint. Either supervisor exits when this control
|
||||
connection closes. This couples their restart lifecycle and prevents a workload
|
||||
that survives an isolated network-sidecar restart from becoming the next
|
||||
authoritative control client. In sidecar mode, an init container performs the
|
||||
privileged pod-network nftables setup with
|
||||
`NET_ADMIN`. The default binary-aware network sidecar runs as UID 0 without
|
||||
`NET_ADMIN` and adds `SYS_PTRACE` plus `DAC_READ_SEARCH` so it can resolve
|
||||
cross-UID workload process/binary identity through shared `/proc`. Operators
|
||||
can set the sidecar `process_binary_aware_network_policy` flag false to run the
|
||||
sidecar as the configured non-root proxy UID, omit both inspection capabilities,
|
||||
and downgrade network policy to endpoint/L7 matching without `policy.binaries`.
|
||||
The init path applies nftables as individual commands so optional conntrack and
|
||||
log expressions can fail without rolling back the required table, chain, and
|
||||
reject rules.
|
||||
The agent container runs as the resolved sandbox UID/GID with no added Linux
|
||||
capabilities. Sidecar mode preserves gateway session and SSH behavior, but
|
||||
treats the process leaf as network-only: Landlock filesystem policy and child
|
||||
seccomp still apply where supported, while process privilege dropping and
|
||||
supervisor identity mount isolation do not run because the agent container is
|
||||
already unprivileged. Sidecar pods use a shared process namespace so the
|
||||
network sidecar can resolve workload process and binary identity through
|
||||
`/proc/<entrypoint-pid>`.
|
||||
|
||||
## Images
|
||||
|
||||
The gateway image and Helm chart are built from this repository. Sandbox images
|
||||
|
||||
@@ -167,9 +167,14 @@ async fn build_plain_channel(endpoint: &str) -> Result<Channel> {
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| format!("failed to read client key from {key_path}"))?;
|
||||
|
||||
let tls_config = ClientTlsConfig::new()
|
||||
let mut tls_config = ClientTlsConfig::new()
|
||||
.ca_certificate(Certificate::from_pem(ca_pem))
|
||||
.identity(Identity::from_pem(cert_pem, key_pem));
|
||||
if let Ok(server_name) = std::env::var(sandbox_env::GATEWAY_TLS_SERVER_NAME)
|
||||
&& !server_name.is_empty()
|
||||
{
|
||||
tls_config = tls_config.domain_name(server_name);
|
||||
}
|
||||
|
||||
ep = ep
|
||||
.tls_config(tls_config)
|
||||
|
||||
@@ -63,6 +63,63 @@ impl ProviderCredentialState {
|
||||
}
|
||||
}
|
||||
|
||||
/// Build a static provider state from an already-prepared child
|
||||
/// environment snapshot.
|
||||
///
|
||||
/// Kubernetes sidecar topology uses this in the process-only supervisor:
|
||||
/// the network sidecar owns provider credential resolvers and sends the
|
||||
/// workload-facing env map over a local control channel. The process leaf
|
||||
/// must inject that map into child processes without re-placeholderizing it
|
||||
/// or holding the gateway-side resolver material.
|
||||
pub fn from_child_env_snapshot(revision: u64, child_env: HashMap<String, String>) -> Self {
|
||||
let snapshot = Arc::new(ProviderCredentialSnapshot {
|
||||
revision,
|
||||
child_env,
|
||||
dynamic_credentials: HashMap::new(),
|
||||
});
|
||||
|
||||
Self {
|
||||
inner: Arc::new(RwLock::new(ProviderCredentialStateInner {
|
||||
current: snapshot,
|
||||
generations: VecDeque::new(),
|
||||
current_resolver: None,
|
||||
combined_resolver: None,
|
||||
suppressed_keys: HashSet::new(),
|
||||
})),
|
||||
}
|
||||
}
|
||||
|
||||
/// Install an already-prepared child environment snapshot.
|
||||
///
|
||||
/// This is intentionally narrower than [`Self::install_environment`]: it
|
||||
/// updates only the workload-facing env map and clears resolver state so a
|
||||
/// process that does not own gateway/provider resolver material can still
|
||||
/// pick up refreshed provider env for future child processes.
|
||||
pub fn install_child_env_snapshot(
|
||||
&self,
|
||||
revision: u64,
|
||||
mut child_env: HashMap<String, String>,
|
||||
) -> usize {
|
||||
let mut inner = self
|
||||
.inner
|
||||
.write()
|
||||
.expect("provider credential state poisoned");
|
||||
|
||||
for key in &inner.suppressed_keys {
|
||||
child_env.remove(key);
|
||||
}
|
||||
|
||||
inner.current = Arc::new(ProviderCredentialSnapshot {
|
||||
revision,
|
||||
child_env,
|
||||
dynamic_credentials: HashMap::new(),
|
||||
});
|
||||
inner.generations.clear();
|
||||
inner.current_resolver = None;
|
||||
inner.combined_resolver = None;
|
||||
inner.current.child_env.len()
|
||||
}
|
||||
|
||||
pub fn snapshot(&self) -> Arc<ProviderCredentialSnapshot> {
|
||||
self.inner
|
||||
.read()
|
||||
@@ -594,6 +651,39 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn child_env_snapshot_install_updates_env_without_resolver_material() {
|
||||
let state = ProviderCredentialState::from_child_env_snapshot(
|
||||
1,
|
||||
HashMap::from([
|
||||
("GITHUB_TOKEN".to_string(), "old".to_string()),
|
||||
("GCE_METADATA_HOST".to_string(), "marker".to_string()),
|
||||
]),
|
||||
);
|
||||
state.remove_env_key("GCE_METADATA_HOST");
|
||||
|
||||
let env_count = state.install_child_env_snapshot(
|
||||
2,
|
||||
HashMap::from([
|
||||
("GITHUB_TOKEN".to_string(), "new".to_string()),
|
||||
("GCE_METADATA_HOST".to_string(), "marker".to_string()),
|
||||
]),
|
||||
);
|
||||
|
||||
let snapshot = state.snapshot();
|
||||
assert_eq!(snapshot.revision, 2);
|
||||
assert_eq!(env_count, 1);
|
||||
assert_eq!(
|
||||
snapshot.child_env.get("GITHUB_TOKEN").map(String::as_str),
|
||||
Some("new")
|
||||
);
|
||||
assert!(!snapshot.child_env.contains_key("GCE_METADATA_HOST"));
|
||||
assert!(
|
||||
state.resolver().is_none(),
|
||||
"child-env snapshots must not install provider resolver material"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stale_generation_falls_back_to_current_credential_after_retention_window() {
|
||||
let state = ProviderCredentialState::from_environment(
|
||||
|
||||
@@ -29,6 +29,34 @@ pub const SANDBOX_COMMAND: &str = "OPENSHELL_SANDBOX_COMMAND";
|
||||
/// Deployment-controlled telemetry toggle propagated to the sandbox supervisor.
|
||||
pub const TELEMETRY_ENABLED: &str = "OPENSHELL_TELEMETRY_ENABLED";
|
||||
|
||||
/// Supervisor pod/runtime topology. Kubernetes sidecar mode sets this to
|
||||
/// `"sidecar"`; the default combined supervisor path omits it.
|
||||
pub const SUPERVISOR_TOPOLOGY: &str = "OPENSHELL_SUPERVISOR_TOPOLOGY";
|
||||
|
||||
/// Network enforcement backend selected by the compute driver.
|
||||
pub const NETWORK_ENFORCEMENT_MODE: &str = "OPENSHELL_NETWORK_ENFORCEMENT_MODE";
|
||||
|
||||
/// Whether network policy evaluation must bind requests to the peer binary.
|
||||
///
|
||||
/// The default when unset is `"required"`. Kubernetes sidecar experiments may
|
||||
/// set this to `"relaxed"` to enforce endpoint and L7 policy without per-binary
|
||||
/// `/proc` identity binding.
|
||||
pub const NETWORK_BINARY_IDENTITY: &str = "OPENSHELL_NETWORK_BINARY_IDENTITY";
|
||||
|
||||
/// Unix socket used by Kubernetes sidecar topology for local coordination.
|
||||
///
|
||||
/// The network sidecar owns gateway credentials and serves policy/provider
|
||||
/// state over this socket instead of exposing gateway credentials to the agent
|
||||
/// container.
|
||||
pub const SIDECAR_CONTROL_SOCKET: &str = "OPENSHELL_SIDECAR_CONTROL_SOCKET";
|
||||
|
||||
/// Optional TLS server name override used when connecting to the gateway.
|
||||
pub const GATEWAY_TLS_SERVER_NAME: &str = "OPENSHELL_GATEWAY_TLS_SERVER_NAME";
|
||||
|
||||
/// Directory where the network supervisor writes the proxy CA files consumed
|
||||
/// by workload child processes.
|
||||
pub const PROXY_TLS_DIR: &str = "OPENSHELL_PROXY_TLS_DIR";
|
||||
|
||||
/// Path to the CA certificate for mTLS communication with the gateway.
|
||||
pub const TLS_CA: &str = "OPENSHELL_TLS_CA";
|
||||
|
||||
|
||||
@@ -53,9 +53,43 @@ pods do not need direct external ingress for SSH.
|
||||
|
||||
## Container Security Context
|
||||
|
||||
The driver grants the sandbox agent container the Linux capabilities the
|
||||
supervisor needs for namespace setup and policy enforcement. It can also request
|
||||
a Kubernetes AppArmor profile through `app_armor_profile`.
|
||||
The default `combined` supervisor topology grants the sandbox agent container
|
||||
the Linux capabilities the supervisor needs for namespace setup and process,
|
||||
filesystem, and network policy enforcement.
|
||||
|
||||
The `sidecar` supervisor topology moves pod-level network setup into a root init
|
||||
container. In the default process/binary-aware mode, the long-lived network
|
||||
sidecar runs as UID 0 with `allowPrivilegeEscalation: false`, drops default
|
||||
Linux capabilities, and adds only `SYS_PTRACE` plus `DAC_READ_SEARCH` for
|
||||
cross-UID workload `/proc` inspection. The agent container also runs as the
|
||||
resolved sandbox UID/GID with `allowPrivilegeEscalation: false` and
|
||||
`capabilities.drop: ["ALL"]`.
|
||||
Set `sidecar.process_binary_aware_network_policy = false` to run the network
|
||||
sidecar as the configured non-root `sidecar.proxy_uid`, omit the extra `/proc`
|
||||
inspection capabilities, and enforce endpoint/L7 network policy without
|
||||
matching `policy.binaries`.
|
||||
In this mode OpenShell preserves gateway session and SSH behavior, but the
|
||||
process supervisor does not perform root-to-sandbox privilege dropping or
|
||||
supervisor identity mount isolation. It still applies Landlock filesystem policy
|
||||
and child seccomp filters where the kernel/runtime supports them. Network
|
||||
endpoint and L7 policy remain enforced by the network sidecar, and
|
||||
sidecar pods use a shared process namespace so the network sidecar can resolve
|
||||
process/binary identity through `/proc/<entrypoint-pid>`.
|
||||
|
||||
Sidecar mode keeps gateway credentials in the network sidecar. The agent
|
||||
container does not mount the projected service-account token used for sandbox
|
||||
token bootstrap, does not mount the sandbox client TLS secret, and does not get
|
||||
gateway callback environment variables. The process supervisor receives policy
|
||||
and provider environment state from the sidecar over a local control socket in
|
||||
the shared sidecar state volume. The sidecar accepts only the pre-workload
|
||||
process-supervisor connection, authenticates its UID/GID/PID with peer
|
||||
credentials, and removes the listener afterward. SSH relays use a Linux
|
||||
abstract socket whose peer PID must match that authenticated supervisor. Both
|
||||
supervisors exit if the control connection closes, coupling their container
|
||||
restart lifecycle before a new authoritative client can be established.
|
||||
|
||||
The driver can request a Kubernetes AppArmor profile through
|
||||
`app_armor_profile`.
|
||||
|
||||
Supported values are `Unconfined`, `RuntimeDefault`, and
|
||||
`Localhost/<profile-name>`. An empty or unset value omits
|
||||
|
||||
@@ -15,6 +15,9 @@ pub const DEFAULT_SANDBOX_SERVICE_ACCOUNT_NAME: &str = "default";
|
||||
/// Default storage size for the workspace PVC.
|
||||
pub const DEFAULT_WORKSPACE_STORAGE_SIZE: &str = "2Gi";
|
||||
|
||||
/// Default non-root UID for relaxed Kubernetes network supervisor sidecars.
|
||||
pub const DEFAULT_PROXY_UID: u32 = 1337;
|
||||
|
||||
/// How the supervisor binary is delivered into sandbox pods.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
|
||||
#[serde(rename_all = "kebab-case")]
|
||||
@@ -59,12 +62,16 @@ pub enum SupervisorTopology {
|
||||
/// Run networking and process supervision in the agent container.
|
||||
#[default]
|
||||
Combined,
|
||||
/// Run network supervision in a privileged sidecar and process supervision
|
||||
/// as a low-capability wrapper in the agent container.
|
||||
Sidecar,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for SupervisorTopology {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
match self {
|
||||
Self::Combined => f.write_str("combined"),
|
||||
Self::Sidecar => f.write_str("sidecar"),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -75,11 +82,49 @@ impl FromStr for SupervisorTopology {
|
||||
fn from_str(s: &str) -> Result<Self, Self::Err> {
|
||||
match s {
|
||||
"combined" => Ok(Self::Combined),
|
||||
other => Err(format!("unknown supervisor topology '{other}'")),
|
||||
"sidecar" => Ok(Self::Sidecar),
|
||||
other => Err(format!("unknown topology '{other}'")),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
#[serde(default, deny_unknown_fields)]
|
||||
pub struct KubernetesSidecarConfig {
|
||||
/// UID used by relaxed long-running network sidecars in `sidecar`
|
||||
/// topology. The network init container installs nftables rules that
|
||||
/// exempt this UID, so it must not match the sandbox workload UID.
|
||||
/// Strict process/binary-aware sidecars run as UID 0 so Kubernetes grants
|
||||
/// the requested `/proc` inspection capabilities into the effective set.
|
||||
pub proxy_uid: u32,
|
||||
/// Require process/binary-aware network policy enforcement in sidecar
|
||||
/// topology. When disabled, the network sidecar runs as `proxy_uid`,
|
||||
/// drops the extra `/proc` inspection permissions, and evaluates
|
||||
/// endpoint/L7 policy without matching `policy.binaries`.
|
||||
pub process_binary_aware_network_policy: bool,
|
||||
}
|
||||
|
||||
impl Default for KubernetesSidecarConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
proxy_uid: DEFAULT_PROXY_UID,
|
||||
process_binary_aware_network_policy: true,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl KubernetesSidecarConfig {
|
||||
pub fn validate_proxy_uid(&self) -> Result<(), String> {
|
||||
if self.proxy_uid < openshell_policy::MIN_SANDBOX_UID {
|
||||
return Err(format!(
|
||||
"sidecar.proxy_uid must be at least {}",
|
||||
openshell_policy::MIN_SANDBOX_UID
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// Kubernetes `AppArmor` profile requested for the sandbox agent container.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum AppArmorProfile {
|
||||
@@ -205,7 +250,9 @@ pub struct KubernetesComputeConfig {
|
||||
/// How the supervisor binary is delivered into sandbox pods.
|
||||
pub supervisor_sideload_method: SupervisorSideloadMethod,
|
||||
/// How the supervisor is arranged for Kubernetes sandbox pods.
|
||||
pub supervisor_topology: SupervisorTopology,
|
||||
pub topology: SupervisorTopology,
|
||||
/// Sidecar-only settings used when `topology = "sidecar"`.
|
||||
pub sidecar: KubernetesSidecarConfig,
|
||||
pub grpc_endpoint: String,
|
||||
pub ssh_socket_path: String,
|
||||
pub client_tls_secret_name: String,
|
||||
@@ -291,7 +338,8 @@ impl Default for KubernetesComputeConfig {
|
||||
supervisor_image: config::default_supervisor_image(),
|
||||
supervisor_image_pull_policy: String::new(),
|
||||
supervisor_sideload_method: SupervisorSideloadMethod::default(),
|
||||
supervisor_topology: SupervisorTopology::default(),
|
||||
topology: SupervisorTopology::default(),
|
||||
sidecar: KubernetesSidecarConfig::default(),
|
||||
grpc_endpoint: String::new(),
|
||||
ssh_socket_path: "/run/openshell/ssh.sock".to_string(),
|
||||
client_tls_secret_name: String::new(),
|
||||
@@ -336,6 +384,10 @@ impl KubernetesComputeConfig {
|
||||
)
|
||||
}
|
||||
|
||||
pub fn validate_proxy_uid(&self) -> Result<(), String> {
|
||||
self.sidecar.validate_proxy_uid()
|
||||
}
|
||||
|
||||
/// Resolve the sandbox UID/GID pair.
|
||||
///
|
||||
/// Resolution order:
|
||||
@@ -351,6 +403,7 @@ impl KubernetesComputeConfig {
|
||||
if let Some(uid) = self.sandbox_uid {
|
||||
return uid;
|
||||
}
|
||||
// Try OpenShift SCC annotation.
|
||||
if let Some(anns) = namespace_annotations
|
||||
&& let Some(range) = anns.get(ANNOTATION_SCC_UID_RANGE)
|
||||
&& let Some(uid) = Self::from_open_shift_uid_range(range)
|
||||
@@ -461,6 +514,132 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn default_topology_is_combined() {
|
||||
let cfg = KubernetesComputeConfig::default();
|
||||
assert_eq!(cfg.topology, SupervisorTopology::Combined);
|
||||
assert_eq!(cfg.topology.to_string(), "combined");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn default_proxy_uid_is_dedicated_non_root_uid() {
|
||||
let cfg = KubernetesComputeConfig::default();
|
||||
assert_eq!(cfg.sidecar.proxy_uid, DEFAULT_PROXY_UID);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn default_sidecar_requires_process_binary_aware_network_policy() {
|
||||
let cfg = KubernetesComputeConfig::default();
|
||||
assert!(cfg.sidecar.process_binary_aware_network_policy);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serde_override_topology_sidecar() {
|
||||
let json = serde_json::json!({
|
||||
"topology": "sidecar"
|
||||
});
|
||||
let cfg: KubernetesComputeConfig = serde_json::from_value(json).unwrap();
|
||||
assert_eq!(cfg.topology, SupervisorTopology::Sidecar);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serde_override_topology_combined() {
|
||||
let json = serde_json::json!({
|
||||
"topology": "combined"
|
||||
});
|
||||
let cfg: KubernetesComputeConfig = serde_json::from_value(json).unwrap();
|
||||
assert_eq!(cfg.topology, SupervisorTopology::Combined);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serde_rejects_sidecar_binary_identity_field() {
|
||||
let json = serde_json::json!({
|
||||
"sidecar": {
|
||||
"binary_identity": "shared-pid"
|
||||
}
|
||||
});
|
||||
let err = serde_json::from_value::<KubernetesComputeConfig>(json).unwrap_err();
|
||||
assert!(err.to_string().contains("unknown field"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serde_override_sidecar_process_binary_aware_network_policy_nested() {
|
||||
let json = serde_json::json!({
|
||||
"sidecar": {
|
||||
"process_binary_aware_network_policy": false
|
||||
}
|
||||
});
|
||||
let cfg: KubernetesComputeConfig = serde_json::from_value(json).unwrap();
|
||||
assert!(!cfg.sidecar.process_binary_aware_network_policy);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serde_override_sidecar_proxy_uid_nested() {
|
||||
let json = serde_json::json!({
|
||||
"sidecar": {
|
||||
"proxy_uid": 2000
|
||||
}
|
||||
});
|
||||
let cfg: KubernetesComputeConfig = serde_json::from_value(json).unwrap();
|
||||
assert_eq!(cfg.sidecar.proxy_uid, 2000);
|
||||
cfg.validate_proxy_uid().unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_proxy_uid_rejects_privileged_uid() {
|
||||
let cfg = KubernetesComputeConfig {
|
||||
sidecar: KubernetesSidecarConfig {
|
||||
proxy_uid: 999,
|
||||
..KubernetesSidecarConfig::default()
|
||||
},
|
||||
..KubernetesComputeConfig::default()
|
||||
};
|
||||
let err = cfg.validate_proxy_uid().unwrap_err();
|
||||
assert!(err.contains("proxy_uid"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serde_rejects_invalid_topology() {
|
||||
let json = serde_json::json!({
|
||||
"topology": "unsupported"
|
||||
});
|
||||
let err = serde_json::from_value::<KubernetesComputeConfig>(json).unwrap_err();
|
||||
assert!(err.to_string().contains("unknown variant"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serde_rejects_removed_topology_alias_field() {
|
||||
let mut json = serde_json::Map::new();
|
||||
json.insert(
|
||||
["supervisor", "topology"].join("_"),
|
||||
serde_json::json!("sidecar"),
|
||||
);
|
||||
let err =
|
||||
serde_json::from_value::<KubernetesComputeConfig>(serde_json::Value::Object(json))
|
||||
.unwrap_err();
|
||||
assert!(err.to_string().contains("unknown field"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serde_rejects_removed_flat_sidecar_fields() {
|
||||
for json in [
|
||||
serde_json::json!({ "sidecar_binary_identity": "shared-pid" }),
|
||||
serde_json::json!({ "proxy_uid": 2000 }),
|
||||
] {
|
||||
let err = serde_json::from_value::<KubernetesComputeConfig>(json).unwrap_err();
|
||||
assert!(err.to_string().contains("unknown field"));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serde_rejects_removed_process_enforcement_field() {
|
||||
let json = serde_json::json!({
|
||||
"process_enforcement": "network-only"
|
||||
});
|
||||
let err = serde_json::from_value::<KubernetesComputeConfig>(json).unwrap_err();
|
||||
assert!(err.to_string().contains("unknown field"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn default_service_account_name_is_default() {
|
||||
let cfg = KubernetesComputeConfig::default();
|
||||
@@ -470,31 +649,6 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn default_supervisor_topology_is_combined() {
|
||||
let cfg = KubernetesComputeConfig::default();
|
||||
assert_eq!(cfg.supervisor_topology, SupervisorTopology::Combined);
|
||||
assert_eq!(cfg.supervisor_topology.to_string(), "combined");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serde_override_supervisor_topology_combined() {
|
||||
let json = serde_json::json!({
|
||||
"supervisor_topology": "combined"
|
||||
});
|
||||
let cfg: KubernetesComputeConfig = serde_json::from_value(json).unwrap();
|
||||
assert_eq!(cfg.supervisor_topology, SupervisorTopology::Combined);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serde_rejects_invalid_supervisor_topology() {
|
||||
let json = serde_json::json!({
|
||||
"supervisor_topology": "unsupported"
|
||||
});
|
||||
let err = serde_json::from_value::<KubernetesComputeConfig>(json).unwrap_err();
|
||||
assert!(err.to_string().contains("unknown variant"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serde_override_workspace_storage_size() {
|
||||
let json = serde_json::json!({
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -6,8 +6,9 @@ pub mod driver;
|
||||
pub mod grpc;
|
||||
|
||||
pub use config::{
|
||||
AppArmorProfile, DEFAULT_SANDBOX_SERVICE_ACCOUNT_NAME, DEFAULT_WORKSPACE_STORAGE_SIZE,
|
||||
KubernetesComputeConfig, SupervisorSideloadMethod, SupervisorTopology,
|
||||
AppArmorProfile, DEFAULT_PROXY_UID, DEFAULT_SANDBOX_SERVICE_ACCOUNT_NAME,
|
||||
DEFAULT_WORKSPACE_STORAGE_SIZE, KubernetesComputeConfig, KubernetesSidecarConfig,
|
||||
SupervisorSideloadMethod, SupervisorTopology,
|
||||
};
|
||||
pub use driver::{KubernetesComputeDriver, KubernetesDriverError};
|
||||
pub use grpc::ComputeDriverService;
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
use clap::Parser;
|
||||
use clap::{ArgAction, Parser};
|
||||
use miette::{IntoDiagnostic, Result};
|
||||
use std::net::SocketAddr;
|
||||
use tracing::info;
|
||||
@@ -10,8 +10,9 @@ use tracing_subscriber::EnvFilter;
|
||||
use openshell_core::VERSION;
|
||||
use openshell_core::proto::compute::v1::compute_driver_server::ComputeDriverServer;
|
||||
use openshell_driver_kubernetes::{
|
||||
AppArmorProfile, ComputeDriverService, DEFAULT_SANDBOX_SERVICE_ACCOUNT_NAME,
|
||||
KubernetesComputeConfig, KubernetesComputeDriver, SupervisorSideloadMethod, SupervisorTopology,
|
||||
AppArmorProfile, ComputeDriverService, DEFAULT_PROXY_UID, DEFAULT_SANDBOX_SERVICE_ACCOUNT_NAME,
|
||||
KubernetesComputeConfig, KubernetesComputeDriver, KubernetesSidecarConfig,
|
||||
SupervisorSideloadMethod, SupervisorTopology,
|
||||
};
|
||||
|
||||
#[derive(Parser, Debug)]
|
||||
@@ -80,12 +81,24 @@ struct Args {
|
||||
)]
|
||||
supervisor_sideload_method: SupervisorSideloadMethod,
|
||||
|
||||
#[arg(long, env = "OPENSHELL_K8S_TOPOLOGY", default_value = "combined")]
|
||||
topology: SupervisorTopology,
|
||||
|
||||
#[arg(
|
||||
long,
|
||||
env = "OPENSHELL_SUPERVISOR_TOPOLOGY",
|
||||
default_value = "combined"
|
||||
long = "sidecar-proxy-uid",
|
||||
alias = "proxy-uid",
|
||||
env = "OPENSHELL_K8S_SIDECAR_PROXY_UID",
|
||||
default_value_t = DEFAULT_PROXY_UID
|
||||
)]
|
||||
supervisor_topology: SupervisorTopology,
|
||||
sidecar_proxy_uid: u32,
|
||||
|
||||
#[arg(
|
||||
long = "sidecar-process-binary-aware-network-policy",
|
||||
env = "OPENSHELL_K8S_SIDECAR_PROCESS_BINARY_AWARE_NETWORK_POLICY",
|
||||
default_value_t = true,
|
||||
action = ArgAction::Set
|
||||
)]
|
||||
sidecar_process_binary_aware_network_policy: bool,
|
||||
|
||||
#[arg(long, env = "OPENSHELL_ENABLE_USER_NAMESPACES")]
|
||||
enable_user_namespaces: bool,
|
||||
@@ -130,7 +143,11 @@ async fn main() -> Result<()> {
|
||||
.unwrap_or_else(openshell_core::config::default_supervisor_image),
|
||||
supervisor_image_pull_policy: args.supervisor_image_pull_policy.unwrap_or_default(),
|
||||
supervisor_sideload_method: args.supervisor_sideload_method,
|
||||
supervisor_topology: args.supervisor_topology,
|
||||
topology: args.topology,
|
||||
sidecar: KubernetesSidecarConfig {
|
||||
proxy_uid: args.sidecar_proxy_uid,
|
||||
process_binary_aware_network_policy: args.sidecar_process_binary_aware_network_policy,
|
||||
},
|
||||
grpc_endpoint: args.grpc_endpoint.unwrap_or_default(),
|
||||
ssh_socket_path: args.sandbox_ssh_socket_path,
|
||||
client_tls_secret_name: args.client_tls_secret_name.unwrap_or_default(),
|
||||
|
||||
@@ -130,8 +130,8 @@ sequenceDiagram
|
||||
C->>C: entrypoint: /opt/openshell/bin/openshell-sandbox
|
||||
```
|
||||
|
||||
The supervisor image from `deploy/docker/Dockerfile.supervisor` copies the static
|
||||
`openshell-sandbox` binary to `/openshell-sandbox`.
|
||||
The supervisor image from `deploy/docker/Dockerfile.supervisor` provides the
|
||||
static `openshell-sandbox` binary at `/openshell-sandbox`.
|
||||
Mounting that image at `/opt/openshell/bin` makes the binary available as
|
||||
`/opt/openshell/bin/openshell-sandbox`.
|
||||
|
||||
|
||||
@@ -888,9 +888,8 @@ pub fn build_container_spec_with_token_and_gpu_devices(
|
||||
// Side-load the supervisor binary from a standalone OCI image.
|
||||
// Podman resolves image_volumes at the libpod layer, mounting the
|
||||
// image's filesystem at the destination path without starting a
|
||||
// container from it. The supervisor image is FROM scratch with just
|
||||
// the binary at /openshell-sandbox, so it appears at
|
||||
// /opt/openshell/bin/openshell-sandbox.
|
||||
// container from it. The supervisor image exposes the binary at
|
||||
// /openshell-sandbox, so it appears at /opt/openshell/bin/openshell-sandbox.
|
||||
image_volumes,
|
||||
hostname: format!("sandbox-{}", sandbox.name),
|
||||
// Override the image's ENTRYPOINT so the supervisor binary runs
|
||||
|
||||
@@ -33,11 +33,16 @@ clap = { workspace = true }
|
||||
# Error handling
|
||||
miette = { workspace = true }
|
||||
|
||||
# Unix ownership for Kubernetes sidecar init setup
|
||||
nix = { workspace = true }
|
||||
|
||||
# TLS crypto provider install (main.rs)
|
||||
rustls = { workspace = true }
|
||||
|
||||
# Serialization (serde_json::json! for OCSF unmapped fields)
|
||||
serde = { workspace = true }
|
||||
serde_json = { workspace = true }
|
||||
prost = { workspace = true }
|
||||
|
||||
# Logging
|
||||
tracing = { workspace = true }
|
||||
|
||||
@@ -12,8 +12,9 @@ mod google_cloud_metadata;
|
||||
mod mechanistic_mapper;
|
||||
#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
|
||||
mod metadata_server;
|
||||
mod sidecar_control;
|
||||
|
||||
use miette::Result;
|
||||
use miette::{IntoDiagnostic, Result, WrapErr};
|
||||
use std::future::Future;
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::AtomicU32;
|
||||
@@ -64,12 +65,20 @@ use openshell_core::denial::DenialEvent;
|
||||
use openshell_core::policy::{NetworkMode, NetworkPolicy, ProxyPolicy, SandboxPolicy};
|
||||
use openshell_core::provider_credentials::ProviderCredentialState;
|
||||
use openshell_supervisor_network::opa::OpaEngine;
|
||||
use openshell_supervisor_process::process::ProcessEnforcementMode;
|
||||
pub use openshell_supervisor_process::process::{ProcessHandle, ProcessStatus};
|
||||
use openshell_supervisor_process::skills;
|
||||
use tokio::sync::mpsc::UnboundedSender;
|
||||
#[cfg(target_os = "linux")]
|
||||
#[cfg(any(test, target_os = "linux"))]
|
||||
use tokio::time::timeout;
|
||||
|
||||
const SIDECAR_NETWORK_ENFORCEMENT_MODE: &str = "sidecar-nftables";
|
||||
const SIDECAR_TLS_DIR: &str = "/etc/openshell-tls/proxy";
|
||||
const SIDECAR_CA_CERT: &str = "openshell-ca.pem";
|
||||
const SIDECAR_CA_BUNDLE: &str = "ca-bundle.pem";
|
||||
const SIDECAR_PROCESS_PROXY_ADDR: &str = "127.0.0.1:3128";
|
||||
const SIDECAR_READY_TIMEOUT_SECS: u64 = 120;
|
||||
|
||||
/// Run a command in the sandbox.
|
||||
///
|
||||
/// # Errors
|
||||
@@ -125,17 +134,45 @@ pub async fn run_sandbox(
|
||||
}
|
||||
}
|
||||
|
||||
let sidecar_network_enforcement = sidecar_network_enforcement_enabled();
|
||||
let process_enforcement_mode = process_enforcement_mode();
|
||||
let process_uses_sidecar_control =
|
||||
process_enabled && !network_enabled && sidecar_network_enforcement;
|
||||
let mut process_control_connection = None;
|
||||
let sidecar_bootstrap = if process_uses_sidecar_control {
|
||||
let socket = sidecar_control_socket().ok_or_else(|| {
|
||||
miette::miette!(
|
||||
"{} is required for process-only sidecar topology",
|
||||
openshell_core::sandbox_env::SIDECAR_CONTROL_SOCKET
|
||||
)
|
||||
})?;
|
||||
let (bootstrap, connection) = sidecar_control::connect_process_client(
|
||||
&socket,
|
||||
Duration::from_secs(SIDECAR_READY_TIMEOUT_SECS),
|
||||
)
|
||||
.await?;
|
||||
process_control_connection = Some(connection);
|
||||
Some(bootstrap)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
// Load policy and initialize OPA engine
|
||||
let openshell_endpoint_for_proxy = openshell_endpoint.clone();
|
||||
let sandbox_name_for_agg = sandbox.clone();
|
||||
let (mut policy, opa_engine, retained_proto, loaded_policy_origin) = load_policy(
|
||||
sandbox_id.clone(),
|
||||
sandbox,
|
||||
openshell_endpoint.clone(),
|
||||
policy_rules,
|
||||
policy_data,
|
||||
)
|
||||
.await?;
|
||||
let (mut policy, opa_engine, retained_proto, loaded_policy_origin) =
|
||||
if let Some(bootstrap) = sidecar_bootstrap.as_ref() {
|
||||
load_policy_from_sidecar_bootstrap(bootstrap)?
|
||||
} else {
|
||||
load_policy(
|
||||
sandbox_id.clone(),
|
||||
sandbox,
|
||||
openshell_endpoint.clone(),
|
||||
policy_rules,
|
||||
policy_data,
|
||||
)
|
||||
.await?
|
||||
};
|
||||
|
||||
// Override the policy's process identity with the driver-resolved UID/GID
|
||||
// from the pod environment. The policy defaults to the name "sandbox" which
|
||||
@@ -170,71 +207,89 @@ pub async fn run_sandbox(
|
||||
policy.process.run_as_group = Some(gid);
|
||||
}
|
||||
|
||||
// Fetch provider environment variables from the server.
|
||||
// This is done after loading the policy so the sandbox can still start
|
||||
// even if provider env fetch fails (graceful degradation).
|
||||
let (
|
||||
provider_env_revision,
|
||||
provider_env,
|
||||
provider_credential_expires_at_ms,
|
||||
dynamic_credentials,
|
||||
) = if let (Some(id), Some(endpoint)) = (&sandbox_id, &openshell_endpoint) {
|
||||
match openshell_core::grpc_client::fetch_provider_environment(endpoint, id).await {
|
||||
Ok(result) => {
|
||||
ocsf_emit!(
|
||||
ConfigStateChangeBuilder::new(ocsf_ctx())
|
||||
.severity(SeverityId::Informational)
|
||||
.status(StatusId::Success)
|
||||
.state(StateId::Enabled, "loaded")
|
||||
.message(format!(
|
||||
"Fetched provider environment [env_count:{}]",
|
||||
result.environment.len()
|
||||
))
|
||||
.build()
|
||||
);
|
||||
(
|
||||
result.provider_env_revision,
|
||||
result.environment,
|
||||
result.credential_expires_at_ms,
|
||||
result.dynamic_credentials,
|
||||
)
|
||||
}
|
||||
Err(e) => {
|
||||
ocsf_emit!(
|
||||
ConfigStateChangeBuilder::new(ocsf_ctx())
|
||||
.severity(SeverityId::Medium)
|
||||
.status(StatusId::Failure)
|
||||
.state(StateId::Other, "degraded")
|
||||
.message(format!(
|
||||
"Failed to fetch provider environment, continuing without: {e}"
|
||||
))
|
||||
.build()
|
||||
);
|
||||
#[cfg_attr(not(target_os = "linux"), allow(unused_mut))]
|
||||
let (provider_credentials, mut provider_env) =
|
||||
if let Some(bootstrap) = sidecar_bootstrap.as_ref() {
|
||||
let provider_credentials = ProviderCredentialState::from_child_env_snapshot(
|
||||
bootstrap.provider_env_revision,
|
||||
bootstrap.provider_child_env.clone(),
|
||||
);
|
||||
(provider_credentials, bootstrap.provider_child_env.clone())
|
||||
} else {
|
||||
// Fetch provider environment variables from the server.
|
||||
// This is done after loading the policy so the sandbox can still start
|
||||
// even if provider env fetch fails (graceful degradation).
|
||||
let (
|
||||
provider_env_revision,
|
||||
provider_env,
|
||||
provider_credential_expires_at_ms,
|
||||
dynamic_credentials,
|
||||
) = if let (Some(id), Some(endpoint)) = (&sandbox_id, &openshell_endpoint) {
|
||||
match openshell_core::grpc_client::fetch_provider_environment(endpoint, id).await {
|
||||
Ok(result) => {
|
||||
ocsf_emit!(
|
||||
ConfigStateChangeBuilder::new(ocsf_ctx())
|
||||
.severity(SeverityId::Informational)
|
||||
.status(StatusId::Success)
|
||||
.state(StateId::Enabled, "loaded")
|
||||
.message(format!(
|
||||
"Fetched provider environment [env_count:{}]",
|
||||
result.environment.len()
|
||||
))
|
||||
.build()
|
||||
);
|
||||
(
|
||||
result.provider_env_revision,
|
||||
result.environment,
|
||||
result.credential_expires_at_ms,
|
||||
result.dynamic_credentials,
|
||||
)
|
||||
}
|
||||
Err(e) => {
|
||||
ocsf_emit!(
|
||||
ConfigStateChangeBuilder::new(ocsf_ctx())
|
||||
.severity(SeverityId::Medium)
|
||||
.status(StatusId::Failure)
|
||||
.state(StateId::Other, "degraded")
|
||||
.message(format!(
|
||||
"Failed to fetch provider environment, continuing without: {e}"
|
||||
))
|
||||
.build()
|
||||
);
|
||||
(
|
||||
0,
|
||||
std::collections::HashMap::new(),
|
||||
std::collections::HashMap::new(),
|
||||
std::collections::HashMap::new(),
|
||||
)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
(
|
||||
0,
|
||||
std::collections::HashMap::new(),
|
||||
std::collections::HashMap::new(),
|
||||
std::collections::HashMap::new(),
|
||||
)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
(
|
||||
0,
|
||||
std::collections::HashMap::new(),
|
||||
std::collections::HashMap::new(),
|
||||
std::collections::HashMap::new(),
|
||||
)
|
||||
};
|
||||
};
|
||||
|
||||
let provider_credentials = ProviderCredentialState::from_environment(
|
||||
provider_env_revision,
|
||||
provider_env,
|
||||
provider_credential_expires_at_ms,
|
||||
dynamic_credentials,
|
||||
);
|
||||
#[cfg_attr(not(target_os = "linux"), allow(unused_mut))]
|
||||
let mut provider_env = provider_credentials.child_env_with_gcp_resolved();
|
||||
let provider_credentials = ProviderCredentialState::from_environment(
|
||||
provider_env_revision,
|
||||
provider_env,
|
||||
provider_credential_expires_at_ms,
|
||||
dynamic_credentials,
|
||||
);
|
||||
let provider_env = provider_credentials.child_env_with_gcp_resolved();
|
||||
(provider_credentials, provider_env)
|
||||
};
|
||||
let process_control_writer = process_control_connection
|
||||
.as_ref()
|
||||
.map(|connection| connection.writer.clone());
|
||||
let mut process_control_closed = None;
|
||||
if let Some(connection) = process_control_connection {
|
||||
process_control_closed = Some(connection.closed);
|
||||
spawn_sidecar_control_update_watcher(connection.updates, provider_credentials.clone());
|
||||
}
|
||||
|
||||
// Initialize the agent-proposals feature flag. Default false until the
|
||||
// initial settings fetch (or the poll loop) tells us otherwise. The flag
|
||||
@@ -258,7 +313,7 @@ pub async fn run_sandbox(
|
||||
// it via setns(). The RAII handle lives in this frame for the duration
|
||||
// of the sandbox.
|
||||
#[cfg(target_os = "linux")]
|
||||
let netns = if network_enabled {
|
||||
let netns = if network_enabled && !sidecar_network_enforcement {
|
||||
openshell_supervisor_process::netns::create_netns_for_proxy(&policy)?
|
||||
} else {
|
||||
None
|
||||
@@ -328,6 +383,80 @@ pub async fn run_sandbox(
|
||||
None
|
||||
};
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
let sidecar_control_server = if network_enabled && sidecar_network_enforcement {
|
||||
if !matches!(policy.network.mode, NetworkMode::Proxy) {
|
||||
return Err(miette::miette!(
|
||||
"sidecar network enforcement requires proxy network mode"
|
||||
));
|
||||
}
|
||||
let socket = sidecar_control_socket().ok_or_else(|| {
|
||||
miette::miette!(
|
||||
"{} is required for sidecar topology",
|
||||
openshell_core::sandbox_env::SIDECAR_CONTROL_SOCKET
|
||||
)
|
||||
})?;
|
||||
let proto = retained_proto.as_ref().ok_or_else(|| {
|
||||
miette::miette!(
|
||||
"sidecar topology requires gateway policy data for the process supervisor"
|
||||
)
|
||||
})?;
|
||||
let ca_paths = networking.as_ref().and_then(|n| n.ca_file_paths.clone());
|
||||
Some(sidecar_control::spawn_server(
|
||||
&socket,
|
||||
sidecar_control::BootstrapData {
|
||||
policy_proto: proto.clone(),
|
||||
provider_env_revision: provider_credentials.snapshot().revision,
|
||||
provider_child_env: provider_env.clone(),
|
||||
proxy_ca_cert_path: ca_paths.as_ref().map(|paths| paths.0.clone()),
|
||||
proxy_ca_bundle_path: ca_paths.as_ref().map(|paths| paths.1.clone()),
|
||||
},
|
||||
sidecar_expected_peer()?,
|
||||
)?)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
let sidecar_control_server: Option<sidecar_control::ServerHandle> = None;
|
||||
|
||||
let sidecar_control_publisher = sidecar_control_server
|
||||
.as_ref()
|
||||
.map(sidecar_control::ServerHandle::publisher);
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
let mut sidecar_control_task = None;
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
if network_enabled
|
||||
&& sidecar_network_enforcement
|
||||
&& let Some(server) = sidecar_control_server
|
||||
{
|
||||
let trusted_ssh_socket_path = ssh_socket_path.clone().ok_or_else(|| {
|
||||
miette::miette!(
|
||||
"{} is required for sidecar network topology",
|
||||
openshell_core::sandbox_env::SSH_SOCKET_PATH
|
||||
)
|
||||
})?;
|
||||
let (entrypoint_rx, connection_task) = server.into_runtime_parts();
|
||||
sidecar_control_task = Some(connection_task);
|
||||
spawn_sidecar_entrypoint_handler(
|
||||
entrypoint_rx,
|
||||
entrypoint_pid.clone(),
|
||||
opa_engine.clone(),
|
||||
retained_proto.clone(),
|
||||
openshell_endpoint.clone(),
|
||||
sandbox_id.clone(),
|
||||
std::path::PathBuf::from(trusted_ssh_socket_path),
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
if network_enabled && sidecar_network_enforcement {
|
||||
return Err(miette::miette!(
|
||||
"sidecar network enforcement is only supported on Linux"
|
||||
));
|
||||
}
|
||||
|
||||
// Spawn the denial-aggregator flush task. The aggregator drains denial
|
||||
// events from the proxy + bypass monitor, batches them, and ships
|
||||
// summaries to the gateway via `SubmitPolicyAnalysis`.
|
||||
@@ -398,11 +527,13 @@ pub async fn run_sandbox(
|
||||
}
|
||||
|
||||
// Spawn background policy poll task (gRPC mode only).
|
||||
if let (Some(id), Some(endpoint), Some(engine)) = (
|
||||
sandbox_id.as_deref(),
|
||||
openshell_endpoint.as_deref(),
|
||||
opa_engine.as_ref(),
|
||||
) {
|
||||
if !process_uses_sidecar_control
|
||||
&& let (Some(id), Some(endpoint), Some(engine)) = (
|
||||
sandbox_id.as_deref(),
|
||||
openshell_endpoint.as_deref(),
|
||||
opa_engine.as_ref(),
|
||||
)
|
||||
{
|
||||
let poll_id = id.to_string();
|
||||
let poll_endpoint = endpoint.to_string();
|
||||
let poll_engine = engine.clone();
|
||||
@@ -424,6 +555,7 @@ pub async fn run_sandbox(
|
||||
ocsf_enabled: poll_ocsf_enabled,
|
||||
provider_credentials: poll_provider_credentials,
|
||||
policy_local_ctx: poll_policy_local,
|
||||
sidecar_control_publisher: sidecar_control_publisher.clone(),
|
||||
};
|
||||
|
||||
tokio::spawn(async move {
|
||||
@@ -479,10 +611,51 @@ pub async fn run_sandbox(
|
||||
}
|
||||
}
|
||||
|
||||
let exit_code = if process_enabled {
|
||||
let ca_file_paths = networking.as_ref().and_then(|n| n.ca_file_paths.clone());
|
||||
let process_policy = process_policy_for_topology(&policy, sidecar_network_enforcement)?;
|
||||
let sidecar_bootstrap_ca_file_paths = sidecar_bootstrap.as_ref().and_then(|bootstrap| {
|
||||
bootstrap
|
||||
.proxy_ca_cert_path
|
||||
.clone()
|
||||
.zip(bootstrap.proxy_ca_bundle_path.clone())
|
||||
});
|
||||
|
||||
openshell_supervisor_process::run::run_process(
|
||||
let exit_code = if process_enabled {
|
||||
let ca_file_paths = networking
|
||||
.as_ref()
|
||||
.and_then(|n| n.ca_file_paths.clone())
|
||||
.or_else(|| {
|
||||
if sidecar_network_enforcement {
|
||||
sidecar_bootstrap_ca_file_paths
|
||||
.clone()
|
||||
.or_else(sidecar_ca_file_paths)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
});
|
||||
|
||||
let entrypoint_started_tx =
|
||||
if process_uses_sidecar_control && let Some(writer) = process_control_writer.clone() {
|
||||
let (tx, rx) = tokio::sync::oneshot::channel();
|
||||
tokio::spawn(async move {
|
||||
match rx.await {
|
||||
Ok(pid) => {
|
||||
if let Err(err) =
|
||||
sidecar_control::send_entrypoint_started(&writer, pid).await
|
||||
{
|
||||
warn!(error = %err, "Failed to send sidecar entrypoint event");
|
||||
}
|
||||
}
|
||||
Err(_closed) => {
|
||||
debug!("Entrypoint exited before sidecar entrypoint event was sent");
|
||||
}
|
||||
}
|
||||
});
|
||||
Some(tx)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
let process = openshell_supervisor_process::run::run_process(
|
||||
program,
|
||||
args,
|
||||
workdir.as_deref(),
|
||||
@@ -491,8 +664,11 @@ pub async fn run_sandbox(
|
||||
sandbox_id.as_deref(),
|
||||
openshell_endpoint.as_deref(),
|
||||
ssh_socket_path,
|
||||
&policy,
|
||||
sidecar_network_enforcement,
|
||||
&process_policy,
|
||||
process_enforcement_mode,
|
||||
entrypoint_pid,
|
||||
entrypoint_started_tx,
|
||||
provider_credentials,
|
||||
provider_env,
|
||||
ca_file_paths,
|
||||
@@ -502,14 +678,54 @@ pub async fn run_sandbox(
|
||||
bypass_denial_tx,
|
||||
#[cfg(target_os = "linux")]
|
||||
bypass_activity_tx,
|
||||
)
|
||||
.await?
|
||||
);
|
||||
|
||||
if let Some(control_closed) = process_control_closed.as_mut() {
|
||||
tokio::select! {
|
||||
result = process => result?,
|
||||
_ = control_closed => {
|
||||
ocsf_emit!(
|
||||
AppLifecycleBuilder::new(ocsf_ctx())
|
||||
.activity(ActivityId::Fail)
|
||||
.severity(SeverityId::High)
|
||||
.status(StatusId::Failure)
|
||||
.message(
|
||||
"Authoritative network-sidecar control channel closed; terminating process container"
|
||||
)
|
||||
.build()
|
||||
);
|
||||
return Err(miette::miette!(
|
||||
"authoritative network-sidecar control channel closed"
|
||||
));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
process.await?
|
||||
}
|
||||
} else {
|
||||
// Network-only sidecar mode: keep the proxy and its background
|
||||
// tasks alive (held via the `networking` value) until SIGINT or
|
||||
// SIGTERM. Exit 0 on clean shutdown.
|
||||
wait_for_shutdown_signal().await;
|
||||
0
|
||||
// tasks alive (held via the `networking` value) until shutdown. If the
|
||||
// sole authenticated process-supervisor control connection closes,
|
||||
// exit non-zero so Kubernetes restarts the network sidecar and creates
|
||||
// a fresh one-client bootstrap listener for the restarted agent.
|
||||
#[cfg(target_os = "linux")]
|
||||
if let Some(control_task) = sidecar_control_task {
|
||||
tokio::select! {
|
||||
() = wait_for_shutdown_signal() => 0,
|
||||
result = control_task => {
|
||||
warn!(?result, "Authoritative sidecar control channel exited; restarting sidecar");
|
||||
1
|
||||
}
|
||||
}
|
||||
} else {
|
||||
wait_for_shutdown_signal().await;
|
||||
0
|
||||
}
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
{
|
||||
wait_for_shutdown_signal().await;
|
||||
0
|
||||
}
|
||||
};
|
||||
|
||||
// Drop networking explicitly so the proxy + bypass monitor RAII
|
||||
@@ -552,6 +768,206 @@ async fn wait_for_shutdown_signal() {
|
||||
}
|
||||
}
|
||||
|
||||
fn sidecar_network_enforcement_enabled() -> bool {
|
||||
std::env::var(openshell_core::sandbox_env::NETWORK_ENFORCEMENT_MODE)
|
||||
.is_ok_and(|value| value == SIDECAR_NETWORK_ENFORCEMENT_MODE)
|
||||
}
|
||||
|
||||
fn process_enforcement_mode() -> ProcessEnforcementMode {
|
||||
match std::env::var(openshell_core::sandbox_env::SUPERVISOR_TOPOLOGY)
|
||||
.ok()
|
||||
.as_deref()
|
||||
{
|
||||
Some("sidecar") => ProcessEnforcementMode::NetworkOnly,
|
||||
_ => ProcessEnforcementMode::Full,
|
||||
}
|
||||
}
|
||||
|
||||
fn sidecar_control_socket() -> Option<std::path::PathBuf> {
|
||||
std::env::var(openshell_core::sandbox_env::SIDECAR_CONTROL_SOCKET)
|
||||
.ok()
|
||||
.filter(|path| !path.is_empty())
|
||||
.map(std::path::PathBuf::from)
|
||||
}
|
||||
|
||||
#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
|
||||
fn sidecar_expected_peer() -> Result<sidecar_control::ExpectedPeer> {
|
||||
fn required_numeric_env(name: &str) -> Result<u32> {
|
||||
let value = std::env::var(name)
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| format!("{name} is required for sidecar control authentication"))?;
|
||||
value.parse::<u32>().into_diagnostic().wrap_err_with(|| {
|
||||
format!("{name} must be a numeric ID for sidecar control authentication")
|
||||
})
|
||||
}
|
||||
|
||||
Ok(sidecar_control::ExpectedPeer {
|
||||
uid: required_numeric_env(openshell_core::sandbox_env::SANDBOX_UID)?,
|
||||
gid: required_numeric_env(openshell_core::sandbox_env::SANDBOX_GID)?,
|
||||
})
|
||||
}
|
||||
|
||||
type LoadedPolicyBundle = (
|
||||
SandboxPolicy,
|
||||
Option<Arc<OpaEngine>>,
|
||||
Option<openshell_core::proto::SandboxPolicy>,
|
||||
LoadedPolicyOrigin,
|
||||
);
|
||||
|
||||
fn load_policy_from_sidecar_bootstrap(
|
||||
bootstrap: &sidecar_control::BootstrapData,
|
||||
) -> Result<LoadedPolicyBundle> {
|
||||
let proto = bootstrap.policy_proto.clone();
|
||||
let opa_engine = Some(Arc::new(OpaEngine::from_proto(&proto)?));
|
||||
let policy = SandboxPolicy::try_from(proto.clone())?;
|
||||
info!("Loaded sidecar policy from control socket bootstrap");
|
||||
Ok((
|
||||
policy,
|
||||
opa_engine,
|
||||
Some(proto),
|
||||
LoadedPolicyOrigin::Gateway { revision: None },
|
||||
))
|
||||
}
|
||||
|
||||
fn spawn_sidecar_control_update_watcher(
|
||||
mut updates: tokio::sync::mpsc::UnboundedReceiver<sidecar_control::ControlUpdate>,
|
||||
provider_credentials: ProviderCredentialState,
|
||||
) -> tokio::task::JoinHandle<()> {
|
||||
tokio::spawn(async move {
|
||||
while let Some(update) = updates.recv().await {
|
||||
match update {
|
||||
sidecar_control::ControlUpdate::ProviderEnvUpdated {
|
||||
revision,
|
||||
provider_child_env,
|
||||
} => {
|
||||
if revision <= provider_credentials.snapshot().revision {
|
||||
continue;
|
||||
}
|
||||
let env_count = provider_credentials
|
||||
.install_child_env_snapshot(revision, provider_child_env);
|
||||
ocsf_emit!(
|
||||
ConfigStateChangeBuilder::new(ocsf_ctx())
|
||||
.severity(SeverityId::Informational)
|
||||
.status(StatusId::Success)
|
||||
.state(StateId::Enabled, "loaded")
|
||||
.unmapped("provider_env_revision", serde_json::json!(revision))
|
||||
.message(format!(
|
||||
"Sidecar provider environment refreshed [revision:{revision} env_count:{env_count}]"
|
||||
))
|
||||
.build()
|
||||
);
|
||||
}
|
||||
sidecar_control::ControlUpdate::PolicyUpdated {
|
||||
policy_proto,
|
||||
policy_hash,
|
||||
config_revision,
|
||||
} => {
|
||||
debug!(
|
||||
version = policy_proto.version,
|
||||
policy_hash,
|
||||
config_revision,
|
||||
"Received sidecar policy update for process supervisor"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn spawn_sidecar_entrypoint_handler(
|
||||
mut entrypoint_rx: tokio::sync::mpsc::Receiver<sidecar_control::EntrypointStarted>,
|
||||
entrypoint_pid: Arc<AtomicU32>,
|
||||
opa_engine: Option<Arc<OpaEngine>>,
|
||||
retained_proto: Option<openshell_core::proto::SandboxPolicy>,
|
||||
openshell_endpoint: Option<String>,
|
||||
sandbox_id: Option<String>,
|
||||
trusted_ssh_socket_path: std::path::PathBuf,
|
||||
) {
|
||||
tokio::spawn(async move {
|
||||
let mut session_started = false;
|
||||
let mut trusted_supervisor_pid = None;
|
||||
while let Some(started) = entrypoint_rx.recv().await {
|
||||
entrypoint_pid.store(started.pid, std::sync::atomic::Ordering::Release);
|
||||
if started.start_session {
|
||||
info!(
|
||||
pid = started.pid,
|
||||
ssh_socket = %trusted_ssh_socket_path.display(),
|
||||
"Sidecar process supervisor reported entrypoint start"
|
||||
);
|
||||
} else {
|
||||
trusted_supervisor_pid = Some(started.pid);
|
||||
info!(
|
||||
pid = started.pid,
|
||||
"Sidecar process supervisor reported initial process anchor"
|
||||
);
|
||||
}
|
||||
|
||||
if let (Some(engine), Some(proto)) = (opa_engine.as_ref(), retained_proto.as_ref()) {
|
||||
match engine.reload_from_proto_with_pid(proto, started.pid) {
|
||||
Ok(()) => info!(
|
||||
pid = started.pid,
|
||||
"Policy binary symlink resolution complete for sidecar process anchor"
|
||||
),
|
||||
Err(err) => warn!(
|
||||
error = %err,
|
||||
pid = started.pid,
|
||||
"Failed to rebuild OPA engine with sidecar process anchor PID"
|
||||
),
|
||||
}
|
||||
}
|
||||
|
||||
if started.start_session
|
||||
&& !session_started
|
||||
&& let (Some(endpoint), Some(id)) =
|
||||
(openshell_endpoint.as_ref(), sandbox_id.as_ref())
|
||||
{
|
||||
let Some(supervisor_pid) = trusted_supervisor_pid else {
|
||||
warn!(
|
||||
pid = started.pid,
|
||||
"Ignoring sidecar entrypoint event before authenticated supervisor anchor"
|
||||
);
|
||||
continue;
|
||||
};
|
||||
openshell_supervisor_process::supervisor_session::spawn(
|
||||
endpoint.clone(),
|
||||
id.clone(),
|
||||
trusted_ssh_socket_path.clone(),
|
||||
None,
|
||||
Some(supervisor_pid),
|
||||
);
|
||||
session_started = true;
|
||||
info!("sidecar supervisor session task spawned");
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
fn sidecar_ca_file_paths() -> Option<(std::path::PathBuf, std::path::PathBuf)> {
|
||||
let tls_dir = std::env::var(openshell_core::sandbox_env::PROXY_TLS_DIR)
|
||||
.unwrap_or_else(|_| SIDECAR_TLS_DIR.to_string());
|
||||
let cert = std::path::Path::new(&tls_dir).join(SIDECAR_CA_CERT);
|
||||
let bundle = std::path::Path::new(&tls_dir).join(SIDECAR_CA_BUNDLE);
|
||||
(cert.exists() && bundle.exists()).then_some((cert, bundle))
|
||||
}
|
||||
|
||||
fn process_policy_for_topology(
|
||||
policy: &SandboxPolicy,
|
||||
sidecar_network_enforcement: bool,
|
||||
) -> Result<SandboxPolicy> {
|
||||
let mut process_policy = policy.clone();
|
||||
if sidecar_network_enforcement && matches!(process_policy.network.mode, NetworkMode::Proxy) {
|
||||
let proxy = process_policy
|
||||
.network
|
||||
.proxy
|
||||
.get_or_insert(ProxyPolicy { http_addr: None });
|
||||
if proxy.http_addr.is_none() {
|
||||
proxy.http_addr = Some(SIDECAR_PROCESS_PROXY_ADDR.parse().into_diagnostic()?);
|
||||
}
|
||||
}
|
||||
Ok(process_policy)
|
||||
}
|
||||
|
||||
/// Flush aggregated denial summaries to the gateway via `SubmitPolicyAnalysis`.
|
||||
async fn flush_proposals_to_gateway(
|
||||
endpoint: &str,
|
||||
@@ -1947,6 +2363,7 @@ struct PolicyPollLoopContext {
|
||||
ocsf_enabled: Arc<std::sync::atomic::AtomicBool>,
|
||||
provider_credentials: ProviderCredentialState,
|
||||
policy_local_ctx: Option<Arc<openshell_supervisor_network::policy_local::PolicyLocalContext>>,
|
||||
sidecar_control_publisher: Option<sidecar_control::Publisher>,
|
||||
}
|
||||
|
||||
async fn run_policy_poll_loop(ctx: PolicyPollLoopContext) -> Result<()> {
|
||||
@@ -2058,12 +2475,20 @@ async fn run_policy_poll_loop(ctx: PolicyPollLoopContext) -> Result<()> {
|
||||
.await
|
||||
{
|
||||
Ok(env_result) => {
|
||||
let env_count = ctx.provider_credentials.install_environment(
|
||||
ctx.provider_credentials.install_environment(
|
||||
env_result.provider_env_revision,
|
||||
env_result.environment,
|
||||
env_result.credential_expires_at_ms,
|
||||
env_result.dynamic_credentials,
|
||||
);
|
||||
let child_env = ctx.provider_credentials.child_env_with_gcp_resolved();
|
||||
let env_count = child_env.len();
|
||||
if let Some(publisher) = ctx.sidecar_control_publisher.as_ref() {
|
||||
publisher.publish_provider_env(
|
||||
env_result.provider_env_revision,
|
||||
child_env.clone(),
|
||||
);
|
||||
}
|
||||
current_provider_env_revision = env_result.provider_env_revision;
|
||||
ocsf_emit!(
|
||||
ConfigStateChangeBuilder::new(ocsf_ctx())
|
||||
@@ -2072,10 +2497,11 @@ async fn run_policy_poll_loop(ctx: PolicyPollLoopContext) -> Result<()> {
|
||||
.state(StateId::Enabled, "loaded")
|
||||
.unmapped(
|
||||
"provider_env_revision",
|
||||
serde_json::json!(current_provider_env_revision)
|
||||
serde_json::json!(env_result.provider_env_revision)
|
||||
)
|
||||
.message(format!(
|
||||
"Provider environment refreshed [revision:{current_provider_env_revision} env_count:{env_count}]"
|
||||
"Provider environment refreshed [revision:{} env_count:{env_count}]",
|
||||
env_result.provider_env_revision
|
||||
))
|
||||
.build()
|
||||
);
|
||||
@@ -2111,6 +2537,13 @@ async fn run_policy_poll_loop(ctx: PolicyPollLoopContext) -> Result<()> {
|
||||
if let Some(policy_local_ctx) = ctx.policy_local_ctx.as_ref() {
|
||||
policy_local_ctx.set_current_policy(policy.clone()).await;
|
||||
}
|
||||
if let Some(publisher) = ctx.sidecar_control_publisher.as_ref() {
|
||||
publisher.publish_policy(
|
||||
policy.clone(),
|
||||
result.policy_hash.clone(),
|
||||
result.config_revision,
|
||||
);
|
||||
}
|
||||
if result.global_policy_version > 0 {
|
||||
ocsf_emit!(ConfigStateChangeBuilder::new(ocsf_ctx())
|
||||
.severity(SeverityId::Informational)
|
||||
@@ -2317,8 +2750,24 @@ fn format_setting_value(es: &openshell_core::proto::EffectiveSetting) -> String
|
||||
)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use openshell_core::policy::{
|
||||
FilesystemPolicy, LandlockPolicy, NetworkMode, NetworkPolicy, ProcessPolicy, ProxyPolicy,
|
||||
};
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
|
||||
fn proxy_policy(http_addr: Option<std::net::SocketAddr>) -> SandboxPolicy {
|
||||
SandboxPolicy {
|
||||
version: 1,
|
||||
filesystem: FilesystemPolicy::default(),
|
||||
network: NetworkPolicy {
|
||||
mode: NetworkMode::Proxy,
|
||||
proxy: Some(ProxyPolicy { http_addr }),
|
||||
},
|
||||
landlock: LandlockPolicy::default(),
|
||||
process: ProcessPolicy::default(),
|
||||
}
|
||||
}
|
||||
|
||||
fn effective_bool(value: bool) -> openshell_core::proto::EffectiveSetting {
|
||||
openshell_core::proto::EffectiveSetting {
|
||||
value: Some(openshell_core::proto::SettingValue {
|
||||
@@ -2330,6 +2779,100 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sidecar_process_policy_sets_loopback_proxy_addr() {
|
||||
let policy = proxy_policy(None);
|
||||
|
||||
let process_policy = process_policy_for_topology(&policy, true).unwrap();
|
||||
|
||||
let http_addr = process_policy
|
||||
.network
|
||||
.proxy
|
||||
.and_then(|proxy| proxy.http_addr)
|
||||
.expect("sidecar process policy should set proxy address");
|
||||
assert_eq!(http_addr.to_string(), SIDECAR_PROCESS_PROXY_ADDR);
|
||||
assert!(
|
||||
policy
|
||||
.network
|
||||
.proxy
|
||||
.as_ref()
|
||||
.expect("original policy should keep proxy config")
|
||||
.http_addr
|
||||
.is_none(),
|
||||
"process policy normalization must not mutate the network policy"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_sidecar_process_policy_preserves_proxy_addr() {
|
||||
let policy = proxy_policy(None);
|
||||
|
||||
let process_policy = process_policy_for_topology(&policy, false).unwrap();
|
||||
|
||||
assert!(
|
||||
process_policy
|
||||
.network
|
||||
.proxy
|
||||
.and_then(|proxy| proxy.http_addr)
|
||||
.is_none()
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn sidecar_control_provider_env_update_installs_newer_revision() {
|
||||
let (tx, rx) = tokio::sync::mpsc::unbounded_channel();
|
||||
let provider_credentials = ProviderCredentialState::from_child_env_snapshot(
|
||||
1,
|
||||
std::collections::HashMap::from([("TOKEN".to_string(), "old".to_string())]),
|
||||
);
|
||||
let handle = spawn_sidecar_control_update_watcher(rx, provider_credentials.clone());
|
||||
|
||||
tx.send(sidecar_control::ControlUpdate::ProviderEnvUpdated {
|
||||
revision: 2,
|
||||
provider_child_env: std::collections::HashMap::from([(
|
||||
"TOKEN".to_string(),
|
||||
"new".to_string(),
|
||||
)]),
|
||||
})
|
||||
.unwrap();
|
||||
|
||||
timeout(Duration::from_secs(1), async {
|
||||
loop {
|
||||
if provider_credentials.snapshot().revision == 2 {
|
||||
break;
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(10)).await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
let snapshot = provider_credentials.snapshot();
|
||||
assert_eq!(snapshot.revision, 2);
|
||||
assert_eq!(
|
||||
snapshot.child_env.get("TOKEN").map(String::as_str),
|
||||
Some("new")
|
||||
);
|
||||
|
||||
tx.send(sidecar_control::ControlUpdate::ProviderEnvUpdated {
|
||||
revision: 1,
|
||||
provider_child_env: std::collections::HashMap::from([(
|
||||
"TOKEN".to_string(),
|
||||
"stale".to_string(),
|
||||
)]),
|
||||
})
|
||||
.unwrap();
|
||||
tokio::time::sleep(Duration::from_millis(20)).await;
|
||||
assert_eq!(
|
||||
provider_credentials
|
||||
.snapshot()
|
||||
.child_env
|
||||
.get("TOKEN")
|
||||
.map(String::as_str),
|
||||
Some("new")
|
||||
);
|
||||
handle.abort();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn apply_ocsf_json_setting_enables_from_initial_settings_snapshot() {
|
||||
let enabled = AtomicBool::new(false);
|
||||
|
||||
@@ -35,15 +35,36 @@ const DEBUG_RPC_SUBCOMMAND: &str = "debug-rpc";
|
||||
|
||||
/// Default `--mode` value: run both supervisor leaves in a single binary.
|
||||
const DEFAULT_MODE: &str = "network,process";
|
||||
const SIDECAR_STATE_DIR: &str = "/run/openshell-sidecar";
|
||||
const SIDECAR_TLS_DIR: &str = "/etc/openshell-tls/proxy";
|
||||
#[cfg(target_os = "linux")]
|
||||
const CLIENT_TLS_DIR: &str = "/etc/openshell-tls/client";
|
||||
#[cfg(target_os = "linux")]
|
||||
const SIDECAR_CLIENT_TLS_SUBDIR: &str = "client";
|
||||
#[cfg(target_os = "linux")]
|
||||
const CLIENT_TLS_FILES: [&str; 3] = ["ca.crt", "tls.crt", "tls.key"];
|
||||
#[cfg(target_os = "linux")]
|
||||
const SIDECAR_STATE_DIR_MODE: u32 = 0o2775;
|
||||
#[cfg(target_os = "linux")]
|
||||
const SIDECAR_TLS_DIR_MODE: u32 = 0o755;
|
||||
#[cfg(target_os = "linux")]
|
||||
const SIDECAR_TLS_STAGING_DIR_MODE: u32 = 0o700;
|
||||
#[cfg(target_os = "linux")]
|
||||
const SIDECAR_CLIENT_TLS_DIR_MODE: u32 = 0o750;
|
||||
#[cfg(target_os = "linux")]
|
||||
const SIDECAR_CLIENT_TLS_FILE_MODE: u32 = 0o400;
|
||||
|
||||
/// Which supervisor leaves are enabled in this process.
|
||||
///
|
||||
/// Parsed from a comma-separated `--mode` value, e.g. `network`,
|
||||
/// `process`, or `network,process`. At least one must be set.
|
||||
/// `process`, or `network,process`. `network-init` is a one-shot setup mode
|
||||
/// used by the Kubernetes sidecar topology and cannot be combined with other
|
||||
/// mode components. At least one must be set.
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
struct Mode {
|
||||
network: bool,
|
||||
process: bool,
|
||||
network_init: bool,
|
||||
}
|
||||
|
||||
impl std::str::FromStr for Mode {
|
||||
@@ -53,20 +74,27 @@ impl std::str::FromStr for Mode {
|
||||
let mut mode = Self {
|
||||
network: false,
|
||||
process: false,
|
||||
network_init: false,
|
||||
};
|
||||
for part in s.split(',').map(str::trim).filter(|p| !p.is_empty()) {
|
||||
match part {
|
||||
"network" => mode.network = true,
|
||||
"process" => mode.process = true,
|
||||
"network-init" => mode.network_init = true,
|
||||
other => {
|
||||
return Err(format!(
|
||||
"unknown mode component '{other}' (expected 'network' and/or 'process')"
|
||||
"unknown mode component '{other}' (expected 'network', 'process', or 'network-init')"
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
if !mode.network && !mode.process {
|
||||
return Err("--mode must enable at least one of: network, process".into());
|
||||
if mode.network_init && (mode.network || mode.process) {
|
||||
return Err("--mode=network-init cannot be combined with other components".into());
|
||||
}
|
||||
if !mode.network && !mode.process && !mode.network_init {
|
||||
return Err(
|
||||
"--mode must enable at least one of: network, process, network-init".into(),
|
||||
);
|
||||
}
|
||||
Ok(mode)
|
||||
}
|
||||
@@ -125,7 +153,8 @@ struct Args {
|
||||
#[arg(long, default_value = "warn", env = openshell_core::sandbox_env::LOG_LEVEL)]
|
||||
log_level: String,
|
||||
|
||||
/// Filesystem path to the Unix socket the embedded SSH daemon binds.
|
||||
/// Unix socket the embedded SSH daemon binds. On Linux, a value beginning
|
||||
/// with `@` selects an abstract socket in the network namespace.
|
||||
/// The supervisor bridges `RelayStream` traffic from the gateway onto
|
||||
/// this socket; nothing else should connect to it.
|
||||
#[arg(long, env = openshell_core::sandbox_env::SSH_SOCKET_PATH)]
|
||||
@@ -149,9 +178,28 @@ struct Args {
|
||||
/// "network" and/or "process". Defaults to both (single-binary
|
||||
/// topology). Use --mode=network for a network-only sidecar, or
|
||||
/// --mode=process for a process-only supervisor when network
|
||||
/// enforcement runs in another pod.
|
||||
/// enforcement runs in another pod. Use --mode=network-init only in
|
||||
/// the Kubernetes init container that prepares sidecar nftables.
|
||||
#[arg(long, default_value = DEFAULT_MODE)]
|
||||
mode: Mode,
|
||||
|
||||
/// UID that the long-running Kubernetes network sidecar will run as.
|
||||
/// `--mode=network-init` installs nftables rules that exempt this UID.
|
||||
#[arg(long, env = "OPENSHELL_PROXY_UID", default_value_t = 1337)]
|
||||
proxy_uid: u32,
|
||||
|
||||
/// GID assigned to shared sidecar state directories. Defaults to
|
||||
/// `--proxy-uid` when omitted.
|
||||
#[arg(long, env = "OPENSHELL_PROXY_GID")]
|
||||
proxy_gid: Option<u32>,
|
||||
|
||||
/// Shared state directory between the network init container and sidecar.
|
||||
#[arg(long, env = "OPENSHELL_SIDECAR_STATE_DIR", default_value = SIDECAR_STATE_DIR)]
|
||||
sidecar_state_dir: String,
|
||||
|
||||
/// Shared TLS work directory between the network init container and sidecar.
|
||||
#[arg(long, env = "OPENSHELL_PROXY_TLS_DIR", default_value = SIDECAR_TLS_DIR)]
|
||||
sidecar_tls_dir: String,
|
||||
}
|
||||
|
||||
/// Copy the running executable to `dest`, creating parent directories as
|
||||
@@ -194,6 +242,189 @@ fn copy_self(dest: &str) -> Result<()> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn prepare_sidecar_directory(path: &Path, uid: u32, gid: u32, mode: u32) -> Result<()> {
|
||||
use miette::Context as _;
|
||||
use nix::unistd::{Gid, Uid, chown};
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
|
||||
std::fs::create_dir_all(path)
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| format!("failed to create sidecar directory {}", path.display()))?;
|
||||
let mut perms = std::fs::metadata(path).into_diagnostic()?.permissions();
|
||||
perms.set_mode(mode);
|
||||
std::fs::set_permissions(path, perms)
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| format!("failed to chmod sidecar directory {}", path.display()))?;
|
||||
chown(path, Some(Uid::from_raw(uid)), Some(Gid::from_raw(gid)))
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| {
|
||||
format!(
|
||||
"failed to chown sidecar directory {} to {uid}:{gid}",
|
||||
path.display()
|
||||
)
|
||||
})?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn prepare_sidecar_directory_for_current_user(path: &Path, mode: u32) -> Result<()> {
|
||||
use miette::Context as _;
|
||||
use nix::unistd::{Gid, Uid, chown};
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
|
||||
let uid = Uid::current();
|
||||
let gid = Gid::current();
|
||||
std::fs::create_dir_all(path)
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| format!("failed to create sidecar directory {}", path.display()))?;
|
||||
chown(path, Some(uid), Some(gid))
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| {
|
||||
format!(
|
||||
"failed to chown sidecar directory {} to {}:{}",
|
||||
path.display(),
|
||||
uid.as_raw(),
|
||||
gid.as_raw()
|
||||
)
|
||||
})?;
|
||||
let mut perms = std::fs::metadata(path).into_diagnostic()?.permissions();
|
||||
perms.set_mode(mode);
|
||||
std::fs::set_permissions(path, perms)
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| format!("failed to chmod sidecar directory {}", path.display()))?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn copy_sidecar_client_tls_if_present(
|
||||
source_dir: &Path,
|
||||
sidecar_tls_dir: &Path,
|
||||
uid: u32,
|
||||
gid: u32,
|
||||
) -> Result<()> {
|
||||
use miette::Context as _;
|
||||
use nix::unistd::{Gid, Uid, chown};
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
|
||||
if !source_dir.exists() {
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let dest_dir = sidecar_tls_dir.join(SIDECAR_CLIENT_TLS_SUBDIR);
|
||||
prepare_sidecar_directory_for_current_user(&dest_dir, SIDECAR_TLS_STAGING_DIR_MODE)?;
|
||||
for file_name in CLIENT_TLS_FILES {
|
||||
let source = source_dir.join(file_name);
|
||||
if !source.exists() {
|
||||
return Err(miette::miette!(
|
||||
"client TLS source file is missing: {}",
|
||||
source.display()
|
||||
));
|
||||
}
|
||||
let dest = dest_dir.join(file_name);
|
||||
if dest.exists() {
|
||||
std::fs::remove_file(&dest)
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| {
|
||||
format!("failed to remove stale client TLS file {}", dest.display())
|
||||
})?;
|
||||
}
|
||||
std::fs::copy(&source, &dest)
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| {
|
||||
format!(
|
||||
"failed to copy client TLS file {} to {}",
|
||||
source.display(),
|
||||
dest.display()
|
||||
)
|
||||
})?;
|
||||
let mut perms = std::fs::metadata(&dest).into_diagnostic()?.permissions();
|
||||
perms.set_mode(SIDECAR_CLIENT_TLS_FILE_MODE);
|
||||
std::fs::set_permissions(&dest, perms)
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| {
|
||||
format!("failed to chmod copied client TLS file {}", dest.display())
|
||||
})?;
|
||||
chown(&dest, Some(Uid::from_raw(uid)), Some(Gid::from_raw(gid)))
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| {
|
||||
format!(
|
||||
"failed to chown copied client TLS file {} to {uid}:{gid}",
|
||||
dest.display()
|
||||
)
|
||||
})?;
|
||||
}
|
||||
|
||||
prepare_sidecar_directory(&dest_dir, uid, gid, SIDECAR_CLIENT_TLS_DIR_MODE)?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn run_network_init(
|
||||
proxy_user_id: u32,
|
||||
proxy_primary_group_id: u32,
|
||||
sidecar_state_dir: &str,
|
||||
sidecar_tls_dir: &str,
|
||||
) -> Result<()> {
|
||||
validate_network_init_ids(proxy_user_id, proxy_primary_group_id)?;
|
||||
|
||||
let sidecar_state_dir = Path::new(sidecar_state_dir);
|
||||
let sidecar_tls_dir = Path::new(sidecar_tls_dir);
|
||||
prepare_sidecar_directory(
|
||||
sidecar_state_dir,
|
||||
proxy_user_id,
|
||||
proxy_primary_group_id,
|
||||
SIDECAR_STATE_DIR_MODE,
|
||||
)?;
|
||||
// The init container runs as uid 0 with CAP_DAC_OVERRIDE dropped. Keep the
|
||||
// TLS work directory owned by the init user until the client cert copy is
|
||||
// complete, then hand it to the long-running proxy UID.
|
||||
prepare_sidecar_directory_for_current_user(sidecar_tls_dir, SIDECAR_TLS_DIR_MODE)?;
|
||||
copy_sidecar_client_tls_if_present(
|
||||
Path::new(CLIENT_TLS_DIR),
|
||||
sidecar_tls_dir,
|
||||
proxy_user_id,
|
||||
proxy_primary_group_id,
|
||||
)?;
|
||||
prepare_sidecar_directory(
|
||||
sidecar_tls_dir,
|
||||
proxy_user_id,
|
||||
proxy_primary_group_id,
|
||||
SIDECAR_TLS_DIR_MODE,
|
||||
)?;
|
||||
openshell_supervisor_process::netns::install_sidecar_bypass_rules(proxy_user_id)
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn validate_network_init_ids(proxy_user_id: u32, proxy_primary_group_id: u32) -> Result<()> {
|
||||
if proxy_user_id != 0 && proxy_user_id < openshell_policy::MIN_SANDBOX_UID {
|
||||
return Err(miette::miette!(
|
||||
"--proxy-uid must be 0 or at least {}",
|
||||
openshell_policy::MIN_SANDBOX_UID
|
||||
));
|
||||
}
|
||||
if proxy_primary_group_id < openshell_policy::MIN_SANDBOX_UID {
|
||||
return Err(miette::miette!(
|
||||
"--proxy-gid must be at least {}",
|
||||
openshell_policy::MIN_SANDBOX_UID
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
fn run_network_init(
|
||||
_proxy_uid: u32,
|
||||
_proxy_gid: u32,
|
||||
_sidecar_state_dir: &str,
|
||||
_sidecar_tls_dir: &str,
|
||||
) -> Result<()> {
|
||||
Err(miette::miette!(
|
||||
"--mode=network-init is only supported on Linux"
|
||||
))
|
||||
}
|
||||
|
||||
fn main() -> Result<()> {
|
||||
// Handle `copy-self <DEST>` before clap so it works without any of the
|
||||
// sandbox flags. Kubernetes init containers invoke this path to seed an
|
||||
@@ -222,6 +453,16 @@ fn main() -> Result<()> {
|
||||
|
||||
let args = Args::parse();
|
||||
|
||||
if args.mode.network_init {
|
||||
let proxy_gid = args.proxy_gid.unwrap_or(args.proxy_uid);
|
||||
return run_network_init(
|
||||
args.proxy_uid,
|
||||
proxy_gid,
|
||||
&args.sidecar_state_dir,
|
||||
&args.sidecar_tls_dir,
|
||||
);
|
||||
}
|
||||
|
||||
// Try to open a rolling log file; fall back to stderr-only logging if it fails
|
||||
// (e.g., /var/log is not writable in custom workload images).
|
||||
// Rotates daily, keeps the 3 most recent files to bound disk usage.
|
||||
@@ -421,4 +662,50 @@ mod tests {
|
||||
let final_path = dest_dir.join("openshell-sandbox");
|
||||
assert!(final_path.exists(), "binary should land inside dest dir");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mode_parses_network_init_standalone() {
|
||||
let mode = "network-init".parse::<Mode>().unwrap();
|
||||
assert!(mode.network_init);
|
||||
assert!(!mode.network);
|
||||
assert!(!mode.process);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mode_rejects_combined_network_init() {
|
||||
let err = "network-init,network".parse::<Mode>().unwrap_err();
|
||||
assert!(err.contains("cannot be combined"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mode_rejects_empty_value() {
|
||||
let err = "".parse::<Mode>().unwrap_err();
|
||||
assert!(err.contains("at least one"));
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
#[test]
|
||||
fn sidecar_tls_modes_preserve_proxy_owned_parent_and_private_client_dir() {
|
||||
assert_eq!(SIDECAR_TLS_DIR_MODE, 0o755);
|
||||
assert_eq!(SIDECAR_TLS_STAGING_DIR_MODE, 0o700);
|
||||
assert_eq!(SIDECAR_CLIENT_TLS_DIR_MODE, 0o750);
|
||||
assert_eq!(SIDECAR_CLIENT_TLS_FILE_MODE, 0o400);
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
#[test]
|
||||
fn network_init_accepts_root_proxy_uid_for_binary_aware_sidecar() {
|
||||
validate_network_init_ids(0, openshell_policy::MIN_SANDBOX_UID).unwrap();
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
#[test]
|
||||
fn network_init_still_rejects_low_non_root_proxy_ids() {
|
||||
let uid_err =
|
||||
validate_network_init_ids(999, openshell_policy::MIN_SANDBOX_UID).unwrap_err();
|
||||
assert!(uid_err.to_string().contains("--proxy-uid"));
|
||||
|
||||
let gid_err = validate_network_init_ids(0, 999).unwrap_err();
|
||||
assert!(gid_err.to_string().contains("--proxy-gid"));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,783 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
//! Local control channel for Kubernetes sidecar topology.
|
||||
//!
|
||||
//! The network sidecar owns gateway credentials. The process supervisor in the
|
||||
//! agent container connects over this Unix socket to receive policy/provider
|
||||
//! state without mounting gateway credentials into the agent container.
|
||||
|
||||
use miette::{IntoDiagnostic, Result, WrapErr};
|
||||
use prost::Message;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::collections::HashMap;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::{Arc, RwLock};
|
||||
use std::time::Duration;
|
||||
use tokio::io::{AsyncBufReadExt, AsyncWrite, AsyncWriteExt, BufReader};
|
||||
use tokio::net::UnixListener;
|
||||
use tokio::net::unix::OwnedWriteHalf;
|
||||
use tokio::sync::{Mutex, broadcast, mpsc};
|
||||
use tracing::{debug, info, warn};
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct BootstrapData {
|
||||
pub policy_proto: openshell_core::proto::SandboxPolicy,
|
||||
pub provider_env_revision: u64,
|
||||
pub provider_child_env: HashMap<String, String>,
|
||||
pub proxy_ca_cert_path: Option<PathBuf>,
|
||||
pub proxy_ca_bundle_path: Option<PathBuf>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
|
||||
pub struct EntrypointStarted {
|
||||
pub pid: u32,
|
||||
pub start_session: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct ExpectedPeer {
|
||||
pub uid: u32,
|
||||
pub gid: u32,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum ControlUpdate {
|
||||
ProviderEnvUpdated {
|
||||
revision: u64,
|
||||
provider_child_env: HashMap<String, String>,
|
||||
},
|
||||
PolicyUpdated {
|
||||
policy_proto: openshell_core::proto::SandboxPolicy,
|
||||
policy_hash: String,
|
||||
config_revision: u64,
|
||||
},
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct Publisher {
|
||||
state: Arc<RwLock<BootstrapData>>,
|
||||
updates: broadcast::Sender<WireServerMessage>,
|
||||
}
|
||||
|
||||
impl Publisher {
|
||||
pub fn publish_provider_env(&self, revision: u64, provider_child_env: HashMap<String, String>) {
|
||||
{
|
||||
let mut state = self.state.write().expect("sidecar control state poisoned");
|
||||
if revision <= state.provider_env_revision {
|
||||
return;
|
||||
}
|
||||
state.provider_env_revision = revision;
|
||||
state.provider_child_env.clone_from(&provider_child_env);
|
||||
}
|
||||
|
||||
let _ = self.updates.send(WireServerMessage::ProviderEnvUpdated {
|
||||
revision,
|
||||
provider_child_env,
|
||||
});
|
||||
}
|
||||
|
||||
pub fn publish_policy(
|
||||
&self,
|
||||
policy_proto: openshell_core::proto::SandboxPolicy,
|
||||
policy_hash: String,
|
||||
config_revision: u64,
|
||||
) {
|
||||
{
|
||||
let mut state = self.state.write().expect("sidecar control state poisoned");
|
||||
state.policy_proto = policy_proto.clone();
|
||||
}
|
||||
|
||||
let _ = self.updates.send(WireServerMessage::PolicyUpdated {
|
||||
policy_proto: policy_proto.encode_to_vec(),
|
||||
policy_hash,
|
||||
config_revision,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
pub struct ServerHandle {
|
||||
publisher: Publisher,
|
||||
#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
|
||||
entrypoint_rx: mpsc::Receiver<EntrypointStarted>,
|
||||
connection_task: tokio::task::JoinHandle<()>,
|
||||
}
|
||||
|
||||
impl ServerHandle {
|
||||
pub fn publisher(&self) -> Publisher {
|
||||
self.publisher.clone()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub fn into_entrypoint_receiver(self) -> mpsc::Receiver<EntrypointStarted> {
|
||||
self.entrypoint_rx
|
||||
}
|
||||
|
||||
#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
|
||||
pub fn into_runtime_parts(
|
||||
self,
|
||||
) -> (
|
||||
mpsc::Receiver<EntrypointStarted>,
|
||||
tokio::task::JoinHandle<()>,
|
||||
) {
|
||||
(self.entrypoint_rx, self.connection_task)
|
||||
}
|
||||
}
|
||||
|
||||
pub struct ProcessConnection {
|
||||
pub writer: Arc<Mutex<OwnedWriteHalf>>,
|
||||
pub updates: mpsc::UnboundedReceiver<ControlUpdate>,
|
||||
pub closed: tokio::sync::oneshot::Receiver<()>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Serialize, Deserialize)]
|
||||
#[serde(tag = "type", rename_all = "snake_case")]
|
||||
enum WireClientMessage {
|
||||
BootstrapRequest { supervisor_pid: u32 },
|
||||
EntrypointStarted { pid: u32 },
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
#[serde(tag = "type", rename_all = "snake_case")]
|
||||
enum WireServerMessage {
|
||||
BootstrapResponse {
|
||||
policy_proto: Vec<u8>,
|
||||
provider_env_revision: u64,
|
||||
provider_child_env: HashMap<String, String>,
|
||||
proxy_ca_cert_path: Option<String>,
|
||||
proxy_ca_bundle_path: Option<String>,
|
||||
},
|
||||
ProviderEnvUpdated {
|
||||
revision: u64,
|
||||
provider_child_env: HashMap<String, String>,
|
||||
},
|
||||
PolicyUpdated {
|
||||
policy_proto: Vec<u8>,
|
||||
policy_hash: String,
|
||||
config_revision: u64,
|
||||
},
|
||||
}
|
||||
|
||||
impl BootstrapData {
|
||||
#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
|
||||
fn to_wire(&self) -> WireServerMessage {
|
||||
WireServerMessage::BootstrapResponse {
|
||||
policy_proto: self.policy_proto.encode_to_vec(),
|
||||
provider_env_revision: self.provider_env_revision,
|
||||
provider_child_env: self.provider_child_env.clone(),
|
||||
proxy_ca_cert_path: self
|
||||
.proxy_ca_cert_path
|
||||
.as_ref()
|
||||
.map(|path| path.display().to_string()),
|
||||
proxy_ca_bundle_path: self
|
||||
.proxy_ca_bundle_path
|
||||
.as_ref()
|
||||
.map(|path| path.display().to_string()),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl TryFrom<WireServerMessage> for BootstrapData {
|
||||
type Error = miette::Report;
|
||||
|
||||
fn try_from(message: WireServerMessage) -> Result<Self> {
|
||||
let WireServerMessage::BootstrapResponse {
|
||||
policy_proto,
|
||||
provider_env_revision,
|
||||
provider_child_env,
|
||||
proxy_ca_cert_path,
|
||||
proxy_ca_bundle_path,
|
||||
} = message
|
||||
else {
|
||||
return Err(miette::miette!(
|
||||
"expected sidecar bootstrap response, received update message"
|
||||
));
|
||||
};
|
||||
|
||||
let policy_proto = openshell_core::proto::SandboxPolicy::decode(policy_proto.as_slice())
|
||||
.into_diagnostic()
|
||||
.wrap_err("failed to decode sidecar bootstrap policy")?;
|
||||
|
||||
Ok(Self {
|
||||
policy_proto,
|
||||
provider_env_revision,
|
||||
provider_child_env,
|
||||
proxy_ca_cert_path: proxy_ca_cert_path.map(PathBuf::from),
|
||||
proxy_ca_bundle_path: proxy_ca_bundle_path.map(PathBuf::from),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
impl TryFrom<WireServerMessage> for ControlUpdate {
|
||||
type Error = miette::Report;
|
||||
|
||||
fn try_from(message: WireServerMessage) -> Result<Self> {
|
||||
match message {
|
||||
WireServerMessage::ProviderEnvUpdated {
|
||||
revision,
|
||||
provider_child_env,
|
||||
} => Ok(Self::ProviderEnvUpdated {
|
||||
revision,
|
||||
provider_child_env,
|
||||
}),
|
||||
WireServerMessage::PolicyUpdated {
|
||||
policy_proto,
|
||||
policy_hash,
|
||||
config_revision,
|
||||
} => {
|
||||
let policy_proto =
|
||||
openshell_core::proto::SandboxPolicy::decode(policy_proto.as_slice())
|
||||
.into_diagnostic()
|
||||
.wrap_err("failed to decode sidecar policy update")?;
|
||||
Ok(Self::PolicyUpdated {
|
||||
policy_proto,
|
||||
policy_hash,
|
||||
config_revision,
|
||||
})
|
||||
}
|
||||
WireServerMessage::BootstrapResponse { .. } => Err(miette::miette!(
|
||||
"unexpected sidecar bootstrap response after initial handshake"
|
||||
)),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
|
||||
pub fn spawn_server(
|
||||
path: &Path,
|
||||
bootstrap: BootstrapData,
|
||||
expected_peer: ExpectedPeer,
|
||||
) -> Result<ServerHandle> {
|
||||
if let Some(parent) = path.parent() {
|
||||
std::fs::create_dir_all(parent)
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| {
|
||||
format!(
|
||||
"failed to create sidecar control socket dir {}",
|
||||
parent.display()
|
||||
)
|
||||
})?;
|
||||
}
|
||||
match std::fs::remove_file(path) {
|
||||
Ok(()) => {}
|
||||
Err(err) if err.kind() == std::io::ErrorKind::NotFound => {}
|
||||
Err(err) => {
|
||||
return Err(err).into_diagnostic().wrap_err_with(|| {
|
||||
format!(
|
||||
"failed to remove stale sidecar control socket {}",
|
||||
path.display()
|
||||
)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
let listener = UnixListener::bind(path)
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| format!("failed to bind sidecar control socket {}", path.display()))?;
|
||||
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o660))
|
||||
.into_diagnostic()
|
||||
.wrap_err_with(|| {
|
||||
format!(
|
||||
"failed to set permissions on sidecar control socket {}",
|
||||
path.display()
|
||||
)
|
||||
})?;
|
||||
}
|
||||
|
||||
let state = Arc::new(RwLock::new(bootstrap));
|
||||
let (updates, _) = broadcast::channel(32);
|
||||
let (entrypoint_tx, entrypoint_rx) = mpsc::channel(8);
|
||||
let publisher = Publisher {
|
||||
state: state.clone(),
|
||||
updates: updates.clone(),
|
||||
};
|
||||
|
||||
let connection_task = tokio::spawn(accept_authoritative_connection(
|
||||
listener,
|
||||
path.to_path_buf(),
|
||||
expected_peer,
|
||||
state,
|
||||
updates,
|
||||
entrypoint_tx,
|
||||
));
|
||||
info!(path = %path.display(), "Sidecar control socket listening");
|
||||
|
||||
Ok(ServerHandle {
|
||||
publisher,
|
||||
entrypoint_rx,
|
||||
connection_task,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
|
||||
async fn accept_authoritative_connection(
|
||||
listener: UnixListener,
|
||||
socket_path: PathBuf,
|
||||
expected_peer: ExpectedPeer,
|
||||
state: Arc<RwLock<BootstrapData>>,
|
||||
updates: broadcast::Sender<WireServerMessage>,
|
||||
entrypoint_tx: mpsc::Sender<EntrypointStarted>,
|
||||
) {
|
||||
let stream = match listener.accept().await {
|
||||
Ok((stream, _addr)) => stream,
|
||||
Err(err) => {
|
||||
warn!(error = %err, "Failed to accept authoritative sidecar control connection");
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
// The process supervisor connects before it launches the workload. Drop
|
||||
// the listener and unlink its pathname after that first accept so workload
|
||||
// processes can neither open a second control channel nor impersonate a
|
||||
// restarted server at the trusted path.
|
||||
drop(listener);
|
||||
if let Err(err) = std::fs::remove_file(&socket_path)
|
||||
&& err.kind() != std::io::ErrorKind::NotFound
|
||||
{
|
||||
warn!(
|
||||
path = %socket_path.display(),
|
||||
error = %err,
|
||||
"Failed to unlink accepted sidecar control socket"
|
||||
);
|
||||
}
|
||||
|
||||
if let Err(err) = handle_connection(stream, expected_peer, state, updates, entrypoint_tx).await
|
||||
{
|
||||
warn!(error = %err, "Authoritative sidecar control connection closed");
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
|
||||
async fn handle_connection(
|
||||
stream: tokio::net::UnixStream,
|
||||
expected_peer: ExpectedPeer,
|
||||
state: Arc<RwLock<BootstrapData>>,
|
||||
updates: broadcast::Sender<WireServerMessage>,
|
||||
entrypoint_tx: mpsc::Sender<EntrypointStarted>,
|
||||
) -> Result<()> {
|
||||
let credentials = stream
|
||||
.peer_cred()
|
||||
.into_diagnostic()
|
||||
.wrap_err("failed to read sidecar control peer credentials")?;
|
||||
if credentials.uid() != expected_peer.uid || credentials.gid() != expected_peer.gid {
|
||||
return Err(miette::miette!(
|
||||
"sidecar control peer identity mismatch: expected uid:gid {}:{}, got {}:{}",
|
||||
expected_peer.uid,
|
||||
expected_peer.gid,
|
||||
credentials.uid(),
|
||||
credentials.gid(),
|
||||
));
|
||||
}
|
||||
let peer_pid = credentials
|
||||
.pid()
|
||||
.and_then(|pid| u32::try_from(pid).ok())
|
||||
.ok_or_else(|| miette::miette!("sidecar control peer PID is unavailable"))?;
|
||||
|
||||
let (reader, mut writer) = stream.into_split();
|
||||
let mut lines = BufReader::new(reader).lines();
|
||||
|
||||
let first_line =
|
||||
lines.next_line().await.into_diagnostic()?.ok_or_else(|| {
|
||||
miette::miette!("sidecar control client disconnected before bootstrap")
|
||||
})?;
|
||||
match decode_client_message(&first_line)? {
|
||||
WireClientMessage::BootstrapRequest { supervisor_pid } => {
|
||||
if supervisor_pid == 0 || supervisor_pid != peer_pid {
|
||||
return Err(miette::miette!(
|
||||
"sidecar bootstrap PID mismatch: peer PID {peer_pid}, claimed PID {supervisor_pid}"
|
||||
));
|
||||
}
|
||||
entrypoint_tx
|
||||
.send(EntrypointStarted {
|
||||
pid: supervisor_pid,
|
||||
start_session: false,
|
||||
})
|
||||
.await
|
||||
.map_err(|_| miette::miette!("sidecar entrypoint receiver closed"))?;
|
||||
}
|
||||
WireClientMessage::EntrypointStarted { .. } => {
|
||||
return Err(miette::miette!(
|
||||
"sidecar control client sent entrypoint event before bootstrap"
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
let bootstrap = {
|
||||
let state = state.read().expect("sidecar control state poisoned");
|
||||
state.to_wire()
|
||||
};
|
||||
write_json_line(&mut writer, &bootstrap).await?;
|
||||
|
||||
let mut update_rx = updates.subscribe();
|
||||
loop {
|
||||
tokio::select! {
|
||||
line = lines.next_line() => {
|
||||
let Some(line) = line.into_diagnostic()? else {
|
||||
return Ok(());
|
||||
};
|
||||
match decode_client_message(&line)? {
|
||||
WireClientMessage::BootstrapRequest { .. } => {
|
||||
debug!("Ignoring duplicate sidecar bootstrap request");
|
||||
}
|
||||
WireClientMessage::EntrypointStarted { pid } => {
|
||||
if pid == 0 {
|
||||
warn!("Ignoring sidecar entrypoint event with pid=0");
|
||||
continue;
|
||||
}
|
||||
entrypoint_tx
|
||||
.send(EntrypointStarted {
|
||||
pid,
|
||||
start_session: true,
|
||||
})
|
||||
.await
|
||||
.map_err(|_| miette::miette!("sidecar entrypoint receiver closed"))?;
|
||||
}
|
||||
}
|
||||
}
|
||||
update = update_rx.recv() => {
|
||||
match update {
|
||||
Ok(message) => write_json_line(&mut writer, &message).await?,
|
||||
Err(broadcast::error::RecvError::Lagged(skipped)) => {
|
||||
warn!(skipped, "Sidecar control client lagged behind updates");
|
||||
}
|
||||
Err(broadcast::error::RecvError::Closed) => return Ok(()),
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn connect_process_client(
|
||||
path: &Path,
|
||||
timeout: Duration,
|
||||
) -> Result<(BootstrapData, ProcessConnection)> {
|
||||
let stream = connect_with_retry(path, timeout).await?;
|
||||
let (reader, mut writer) = stream.into_split();
|
||||
write_json_line(
|
||||
&mut writer,
|
||||
&WireClientMessage::BootstrapRequest {
|
||||
supervisor_pid: std::process::id(),
|
||||
},
|
||||
)
|
||||
.await?;
|
||||
|
||||
let mut lines = BufReader::new(reader).lines();
|
||||
let first_line = lines
|
||||
.next_line()
|
||||
.await
|
||||
.into_diagnostic()?
|
||||
.ok_or_else(|| miette::miette!("sidecar control closed before bootstrap response"))?;
|
||||
let bootstrap = BootstrapData::try_from(decode_server_message(&first_line)?)?;
|
||||
|
||||
let (update_tx, updates) = mpsc::unbounded_channel();
|
||||
let (closed_tx, closed) = tokio::sync::oneshot::channel();
|
||||
tokio::spawn(async move {
|
||||
while let Ok(Some(line)) = lines.next_line().await {
|
||||
match decode_server_message(&line).and_then(ControlUpdate::try_from) {
|
||||
Ok(update) => {
|
||||
if update_tx.send(update).is_err() {
|
||||
break;
|
||||
}
|
||||
}
|
||||
Err(err) => {
|
||||
warn!(error = %err, "Ignoring invalid sidecar control update");
|
||||
}
|
||||
}
|
||||
}
|
||||
let _ = closed_tx.send(());
|
||||
});
|
||||
|
||||
Ok((
|
||||
bootstrap,
|
||||
ProcessConnection {
|
||||
writer: Arc::new(Mutex::new(writer)),
|
||||
updates,
|
||||
closed,
|
||||
},
|
||||
))
|
||||
}
|
||||
|
||||
async fn connect_with_retry(path: &Path, timeout: Duration) -> Result<tokio::net::UnixStream> {
|
||||
let deadline = tokio::time::Instant::now() + timeout;
|
||||
loop {
|
||||
match tokio::net::UnixStream::connect(path).await {
|
||||
Ok(stream) => return Ok(stream),
|
||||
Err(err) if tokio::time::Instant::now() < deadline => {
|
||||
debug!(
|
||||
path = %path.display(),
|
||||
error = %err,
|
||||
"Waiting for sidecar control socket"
|
||||
);
|
||||
tokio::time::sleep(Duration::from_millis(100)).await;
|
||||
}
|
||||
Err(err) => {
|
||||
return Err(err).into_diagnostic().wrap_err_with(|| {
|
||||
format!(
|
||||
"timed out waiting for sidecar control socket {}",
|
||||
path.display()
|
||||
)
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn send_entrypoint_started(writer: &Arc<Mutex<OwnedWriteHalf>>, pid: u32) -> Result<()> {
|
||||
let message = WireClientMessage::EntrypointStarted { pid };
|
||||
let mut writer = writer.lock().await;
|
||||
write_json_line(&mut *writer, &message).await
|
||||
}
|
||||
|
||||
async fn write_json_line<W, T>(writer: &mut W, value: &T) -> Result<()>
|
||||
where
|
||||
W: AsyncWrite + Unpin + Send,
|
||||
T: Serialize + Sync,
|
||||
{
|
||||
let bytes = serde_json::to_vec(value).into_diagnostic()?;
|
||||
writer.write_all(&bytes).await.into_diagnostic()?;
|
||||
writer.write_all(b"\n").await.into_diagnostic()?;
|
||||
writer.flush().await.into_diagnostic()?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
|
||||
fn decode_client_message(line: &str) -> Result<WireClientMessage> {
|
||||
serde_json::from_str(line)
|
||||
.into_diagnostic()
|
||||
.wrap_err("failed to decode sidecar client message")
|
||||
}
|
||||
|
||||
fn decode_server_message(line: &str) -> Result<WireServerMessage> {
|
||||
serde_json::from_str(line)
|
||||
.into_diagnostic()
|
||||
.wrap_err("failed to decode sidecar server message")
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use openshell_core::proto::SandboxPolicy;
|
||||
|
||||
fn current_peer() -> ExpectedPeer {
|
||||
ExpectedPeer {
|
||||
uid: nix::unistd::Uid::current().as_raw(),
|
||||
gid: nix::unistd::Gid::current().as_raw(),
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn bootstrap_round_trips_policy_and_provider_env() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let socket = dir.path().join("control.sock");
|
||||
let mut env = HashMap::new();
|
||||
env.insert("GITHUB_TOKEN".to_string(), "secret".to_string());
|
||||
let bootstrap = BootstrapData {
|
||||
policy_proto: SandboxPolicy {
|
||||
version: 7,
|
||||
..SandboxPolicy::default()
|
||||
},
|
||||
provider_env_revision: 3,
|
||||
provider_child_env: env.clone(),
|
||||
proxy_ca_cert_path: Some(PathBuf::from("/tmp/ca.pem")),
|
||||
proxy_ca_bundle_path: Some(PathBuf::from("/tmp/bundle.pem")),
|
||||
};
|
||||
|
||||
let _server = spawn_server(&socket, bootstrap, current_peer()).unwrap();
|
||||
let (received, _connection) = connect_process_client(&socket, Duration::from_secs(1))
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(received.policy_proto.version, 7);
|
||||
assert_eq!(received.provider_env_revision, 3);
|
||||
assert_eq!(received.provider_child_env, env);
|
||||
assert_eq!(
|
||||
received.proxy_ca_cert_path,
|
||||
Some(PathBuf::from("/tmp/ca.pem"))
|
||||
);
|
||||
assert_eq!(
|
||||
received.proxy_ca_bundle_path,
|
||||
Some(PathBuf::from("/tmp/bundle.pem"))
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn entrypoint_started_is_delivered_to_server() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let socket = dir.path().join("control.sock");
|
||||
let server = spawn_server(
|
||||
&socket,
|
||||
BootstrapData {
|
||||
policy_proto: SandboxPolicy::default(),
|
||||
provider_env_revision: 0,
|
||||
provider_child_env: HashMap::new(),
|
||||
proxy_ca_cert_path: None,
|
||||
proxy_ca_bundle_path: None,
|
||||
},
|
||||
current_peer(),
|
||||
)
|
||||
.unwrap();
|
||||
let mut entrypoint_rx = server.into_entrypoint_receiver();
|
||||
let (_bootstrap, connection) = connect_process_client(&socket, Duration::from_secs(1))
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let anchor = tokio::time::timeout(Duration::from_secs(1), entrypoint_rx.recv())
|
||||
.await
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
assert_eq!(anchor.pid, std::process::id());
|
||||
assert!(!anchor.start_session);
|
||||
|
||||
send_entrypoint_started(&connection.writer, 4242)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let started = tokio::time::timeout(Duration::from_secs(1), entrypoint_rx.recv())
|
||||
.await
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
assert_eq!(started.pid, 4242);
|
||||
assert!(started.start_session);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn second_control_client_is_rejected_after_authoritative_bootstrap() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let socket = dir.path().join("control.sock");
|
||||
let _server = spawn_server(
|
||||
&socket,
|
||||
BootstrapData {
|
||||
policy_proto: SandboxPolicy::default(),
|
||||
provider_env_revision: 0,
|
||||
provider_child_env: HashMap::new(),
|
||||
proxy_ca_cert_path: None,
|
||||
proxy_ca_bundle_path: None,
|
||||
},
|
||||
current_peer(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let (_bootstrap, _connection) = connect_process_client(&socket, Duration::from_secs(1))
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let err = tokio::net::UnixStream::connect(&socket)
|
||||
.await
|
||||
.expect_err("control listener must be removed after the first bootstrap");
|
||||
assert!(
|
||||
matches!(
|
||||
err.kind(),
|
||||
std::io::ErrorKind::NotFound | std::io::ErrorKind::ConnectionRefused
|
||||
),
|
||||
"unexpected second-client error: {err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn authoritative_connection_task_ends_when_process_supervisor_disconnects() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let socket = dir.path().join("control.sock");
|
||||
let server = spawn_server(
|
||||
&socket,
|
||||
BootstrapData {
|
||||
policy_proto: SandboxPolicy::default(),
|
||||
provider_env_revision: 0,
|
||||
provider_child_env: HashMap::new(),
|
||||
proxy_ca_cert_path: None,
|
||||
proxy_ca_bundle_path: None,
|
||||
},
|
||||
current_peer(),
|
||||
)
|
||||
.unwrap();
|
||||
let (_entrypoint_rx, connection_task) = server.into_runtime_parts();
|
||||
let (_bootstrap, connection) = connect_process_client(&socket, Duration::from_secs(1))
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
drop(connection);
|
||||
tokio::time::timeout(Duration::from_secs(1), connection_task)
|
||||
.await
|
||||
.expect("server must observe authoritative client disconnect")
|
||||
.expect("control task must not panic");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn process_client_reports_network_sidecar_restart() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let socket = dir.path().join("control.sock");
|
||||
let server = spawn_server(
|
||||
&socket,
|
||||
BootstrapData {
|
||||
policy_proto: SandboxPolicy::default(),
|
||||
provider_env_revision: 0,
|
||||
provider_child_env: HashMap::new(),
|
||||
proxy_ca_cert_path: None,
|
||||
proxy_ca_bundle_path: None,
|
||||
},
|
||||
current_peer(),
|
||||
)
|
||||
.unwrap();
|
||||
let (_entrypoint_rx, connection_task) = server.into_runtime_parts();
|
||||
let (_bootstrap, connection) = connect_process_client(&socket, Duration::from_secs(1))
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
connection_task.abort();
|
||||
let _ = connection_task.await;
|
||||
tokio::time::timeout(Duration::from_secs(1), connection.closed)
|
||||
.await
|
||||
.expect("process supervisor must observe network sidecar disconnect")
|
||||
.expect("disconnect notifier must remain live");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn bootstrap_rejects_claimed_pid_that_does_not_match_peer_credentials() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let socket = dir.path().join("control.sock");
|
||||
let server = spawn_server(
|
||||
&socket,
|
||||
BootstrapData {
|
||||
policy_proto: SandboxPolicy::default(),
|
||||
provider_env_revision: 0,
|
||||
provider_child_env: HashMap::new(),
|
||||
proxy_ca_cert_path: None,
|
||||
proxy_ca_bundle_path: None,
|
||||
},
|
||||
current_peer(),
|
||||
)
|
||||
.unwrap();
|
||||
let mut entrypoint_rx = server.into_entrypoint_receiver();
|
||||
|
||||
let mut stream = tokio::net::UnixStream::connect(&socket).await.unwrap();
|
||||
write_json_line(
|
||||
&mut stream,
|
||||
&WireClientMessage::BootstrapRequest {
|
||||
supervisor_pid: std::process::id().saturating_add(1),
|
||||
},
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
assert!(
|
||||
tokio::time::timeout(Duration::from_secs(1), entrypoint_rx.recv())
|
||||
.await
|
||||
.unwrap()
|
||||
.is_none(),
|
||||
"mismatched bootstrap must not publish a process anchor"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn malformed_client_message_is_rejected() {
|
||||
let err = decode_client_message("not-json").unwrap_err();
|
||||
assert!(
|
||||
err.to_string()
|
||||
.contains("failed to decode sidecar client message")
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -19,6 +19,10 @@ allow_network if {
|
||||
network_policy_for_request
|
||||
}
|
||||
|
||||
binary_identity_required if {
|
||||
object.get(object.get(data, "runtime", {}), "require_binary_identity", true)
|
||||
}
|
||||
|
||||
# --- Deny reasons (specific diagnostics for debugging policy denials) ---
|
||||
|
||||
deny_reason := "missing input.network" if {
|
||||
@@ -131,6 +135,12 @@ endpoint_allowed(policy, network) if {
|
||||
endpoint.ports[_] == network.port
|
||||
}
|
||||
|
||||
# Binary matching can be relaxed by trusted runtime configuration. In that
|
||||
# mode, network policies are endpoint/L7 scoped and ignore policy.binaries.
|
||||
binary_allowed(_, _) if {
|
||||
not binary_identity_required
|
||||
}
|
||||
|
||||
# Binary matching: exact path.
|
||||
# SHA256 integrity is enforced in Rust via trust-on-first-use (TOFU) cache,
|
||||
# not in Rego. The proxy computes and caches binary hashes at runtime.
|
||||
@@ -161,6 +171,10 @@ binary_allowed(policy, exec) if {
|
||||
glob.match(b.path, ["/"], p)
|
||||
}
|
||||
|
||||
user_declared_binary_allowed(_, _) if {
|
||||
not binary_identity_required
|
||||
}
|
||||
|
||||
user_declared_binary_allowed(policy, exec) if {
|
||||
some b
|
||||
b := policy.binaries[_]
|
||||
|
||||
@@ -100,23 +100,34 @@ impl BinaryIdentityCache {
|
||||
/// Returns `Ok(hash)` if it matches, `Err` if the hash changed (binary tampered).
|
||||
#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
|
||||
pub fn verify_or_cache(&self, path: &Path) -> Result<String> {
|
||||
self.verify_or_cache_with_hasher(path, procfs::file_sha256)
|
||||
self.verify_or_cache_with_paths(path, path, procfs::file_sha256)
|
||||
}
|
||||
|
||||
fn verify_or_cache_with_hasher<F>(&self, path: &Path, mut hash_file: F) -> Result<String>
|
||||
#[cfg(target_os = "linux")]
|
||||
pub fn verify_or_cache_process_exe(&self, display_path: &Path, pid: u32) -> Result<String> {
|
||||
let proc_exe = PathBuf::from(format!("/proc/{pid}/exe"));
|
||||
self.verify_or_cache_with_paths(display_path, &proc_exe, procfs::file_sha256)
|
||||
}
|
||||
|
||||
fn verify_or_cache_with_paths<F>(
|
||||
&self,
|
||||
cache_path: &Path,
|
||||
access_path: &Path,
|
||||
mut hash_file: F,
|
||||
) -> Result<String>
|
||||
where
|
||||
F: FnMut(&Path) -> Result<String>,
|
||||
{
|
||||
let start = std::time::Instant::now();
|
||||
let metadata = std::fs::metadata(path)
|
||||
.map_err(|error| miette::miette!("Failed to stat {}: {error}", path.display()))?;
|
||||
let metadata = std::fs::metadata(access_path)
|
||||
.map_err(|error| miette::miette!("Failed to stat {}: {error}", cache_path.display()))?;
|
||||
let fingerprint = FileFingerprint::from_metadata(&metadata);
|
||||
|
||||
let cached = self
|
||||
.hashes
|
||||
.lock()
|
||||
.map_err(|_| miette::miette!("Binary identity cache lock poisoned"))?
|
||||
.get(path)
|
||||
.get(cache_path)
|
||||
.cloned();
|
||||
|
||||
if let Some(cached_binary) = &cached
|
||||
@@ -125,7 +136,7 @@ impl BinaryIdentityCache {
|
||||
debug!(
|
||||
" verify_or_cache: {}ms CACHE HIT path={}",
|
||||
start.elapsed().as_millis(),
|
||||
path.display()
|
||||
cache_path.display()
|
||||
);
|
||||
return Ok(cached_binary.hash.clone());
|
||||
}
|
||||
@@ -133,29 +144,29 @@ impl BinaryIdentityCache {
|
||||
debug!(
|
||||
" verify_or_cache: CACHE MISS size={} path={}",
|
||||
metadata.len(),
|
||||
path.display()
|
||||
cache_path.display()
|
||||
);
|
||||
|
||||
let current_hash = hash_file(path)?;
|
||||
let current_hash = hash_file(access_path)?;
|
||||
|
||||
let mut hashes = self
|
||||
.hashes
|
||||
.lock()
|
||||
.map_err(|_| miette::miette!("Binary identity cache lock poisoned"))?;
|
||||
|
||||
if let Some(existing) = hashes.get(path)
|
||||
if let Some(existing) = hashes.get(cache_path)
|
||||
&& existing.hash != current_hash
|
||||
{
|
||||
return Err(miette::miette!(
|
||||
"Binary integrity violation: {} hash changed (cached: {}, current: {})",
|
||||
path.display(),
|
||||
cache_path.display(),
|
||||
existing.hash,
|
||||
current_hash
|
||||
));
|
||||
}
|
||||
|
||||
hashes.insert(
|
||||
path.to_path_buf(),
|
||||
cache_path.to_path_buf(),
|
||||
CachedBinary {
|
||||
hash: current_hash.clone(),
|
||||
fingerprint,
|
||||
@@ -165,7 +176,7 @@ impl BinaryIdentityCache {
|
||||
debug!(
|
||||
" verify_or_cache TOTAL (cold): {}ms path={}",
|
||||
start.elapsed().as_millis(),
|
||||
path.display()
|
||||
cache_path.display()
|
||||
);
|
||||
|
||||
Ok(current_hash)
|
||||
@@ -212,13 +223,13 @@ mod tests {
|
||||
let mut hash_calls = 0;
|
||||
|
||||
let hash1 = cache
|
||||
.verify_or_cache_with_hasher(tmp.path(), |path| {
|
||||
.verify_or_cache_with_paths(tmp.path(), tmp.path(), |path| {
|
||||
hash_calls += 1;
|
||||
procfs::file_sha256(path)
|
||||
})
|
||||
.unwrap();
|
||||
let hash2 = cache
|
||||
.verify_or_cache_with_hasher(tmp.path(), |path| {
|
||||
.verify_or_cache_with_paths(tmp.path(), tmp.path(), |path| {
|
||||
hash_calls += 1;
|
||||
procfs::file_sha256(path)
|
||||
})
|
||||
@@ -238,7 +249,7 @@ mod tests {
|
||||
let mut hash_calls = 0;
|
||||
|
||||
let hash1 = cache
|
||||
.verify_or_cache_with_hasher(tmp.path(), |path| {
|
||||
.verify_or_cache_with_paths(tmp.path(), tmp.path(), |path| {
|
||||
hash_calls += 1;
|
||||
procfs::file_sha256(path)
|
||||
})
|
||||
@@ -254,7 +265,7 @@ mod tests {
|
||||
.unwrap();
|
||||
|
||||
let hash2 = cache
|
||||
.verify_or_cache_with_hasher(tmp.path(), |path| {
|
||||
.verify_or_cache_with_paths(tmp.path(), tmp.path(), |path| {
|
||||
hash_calls += 1;
|
||||
procfs::file_sha256(path)
|
||||
})
|
||||
@@ -275,7 +286,7 @@ mod tests {
|
||||
let mut hash_calls = 0;
|
||||
|
||||
cache
|
||||
.verify_or_cache_with_hasher(&path, |path| {
|
||||
.verify_or_cache_with_paths(&path, &path, |path| {
|
||||
hash_calls += 1;
|
||||
procfs::file_sha256(path)
|
||||
})
|
||||
@@ -292,7 +303,7 @@ mod tests {
|
||||
.set_modified(original_mtime)
|
||||
.unwrap();
|
||||
|
||||
let result = cache.verify_or_cache_with_hasher(&path, |path| {
|
||||
let result = cache.verify_or_cache_with_paths(&path, &path, |path| {
|
||||
hash_calls += 1;
|
||||
procfs::file_sha256(path)
|
||||
});
|
||||
@@ -301,6 +312,28 @@ mod tests {
|
||||
assert_eq!(hash_calls, 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn display_path_can_differ_from_access_path() {
|
||||
let mut tmp = tempfile::NamedTempFile::new().unwrap();
|
||||
tmp.write_all(b"binary content").unwrap();
|
||||
tmp.flush().unwrap();
|
||||
let display_path = Path::new("/usr/bin/python3");
|
||||
|
||||
let cache = BinaryIdentityCache::new();
|
||||
let hash = cache
|
||||
.verify_or_cache_with_paths(display_path, tmp.path(), procfs::file_sha256)
|
||||
.unwrap();
|
||||
|
||||
assert!(!hash.is_empty());
|
||||
assert!(
|
||||
cache
|
||||
.hashes
|
||||
.lock()
|
||||
.unwrap()
|
||||
.contains_key(Path::new("/usr/bin/python3"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hash_mismatch_returns_error() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
|
||||
@@ -18,6 +18,7 @@ use std::sync::{
|
||||
Arc, Mutex,
|
||||
atomic::{AtomicU64, Ordering},
|
||||
};
|
||||
use tracing::info;
|
||||
|
||||
/// Baked-in rego rules for OPA policy evaluation.
|
||||
/// These rules define the network access decision logic and static config
|
||||
@@ -55,6 +56,49 @@ pub struct NetworkInput {
|
||||
pub cmdline_paths: Vec<PathBuf>,
|
||||
}
|
||||
|
||||
pub(crate) fn network_binary_identity_required() -> bool {
|
||||
std::env::var(openshell_core::sandbox_env::NETWORK_BINARY_IDENTITY).map_or(true, |value| {
|
||||
!matches!(
|
||||
value.as_str(),
|
||||
"relaxed" | "disabled" | "endpoint-only" | "false" | "0"
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
fn inject_runtime_policy_data(data: &mut serde_json::Value, require_binary_identity: bool) {
|
||||
let Some(obj) = data.as_object_mut() else {
|
||||
return;
|
||||
};
|
||||
obj.insert(
|
||||
"runtime".to_string(),
|
||||
serde_json::json!({
|
||||
"require_binary_identity": require_binary_identity,
|
||||
}),
|
||||
);
|
||||
}
|
||||
|
||||
fn emit_binary_identity_mode(require_binary_identity: bool, source: &str) {
|
||||
info!(
|
||||
require_binary_identity,
|
||||
source, "Configured OPA runtime binary identity mode"
|
||||
);
|
||||
openshell_ocsf::ocsf_emit!(
|
||||
openshell_ocsf::ConfigStateChangeBuilder::new(openshell_ocsf::ctx::ctx())
|
||||
.severity(openshell_ocsf::SeverityId::Informational)
|
||||
.status(openshell_ocsf::StatusId::Success)
|
||||
.state(openshell_ocsf::StateId::Enabled, "configured")
|
||||
.unmapped(
|
||||
"require_binary_identity",
|
||||
serde_json::json!(require_binary_identity)
|
||||
)
|
||||
.unmapped("source", serde_json::json!(source))
|
||||
.message(format!(
|
||||
"OPA runtime binary identity mode configured [source:{source} require_binary_identity:{require_binary_identity}]"
|
||||
))
|
||||
.build()
|
||||
);
|
||||
}
|
||||
|
||||
/// Sandbox configuration extracted from OPA data at startup.
|
||||
pub struct SandboxConfig {
|
||||
pub filesystem: FilesystemPolicy,
|
||||
@@ -146,7 +190,9 @@ impl OpaEngine {
|
||||
engine
|
||||
.add_policy_from_file(policy_path)
|
||||
.map_err(|e| miette::miette!("{e}"))?;
|
||||
let data_json = preprocess_yaml_data(&yaml_str)?;
|
||||
let require_binary_identity = network_binary_identity_required();
|
||||
emit_binary_identity_mode(require_binary_identity, "files");
|
||||
let data_json = preprocess_yaml_data(&yaml_str, require_binary_identity)?;
|
||||
engine
|
||||
.add_data_json(&data_json)
|
||||
.map_err(|e| miette::miette!("{e}"))?;
|
||||
@@ -160,11 +206,24 @@ impl OpaEngine {
|
||||
///
|
||||
/// Preprocesses the YAML data to expand access presets and validate L7 config.
|
||||
pub fn from_strings(policy: &str, data_yaml: &str) -> Result<Self> {
|
||||
Self::from_strings_with_binary_identity_required(
|
||||
policy,
|
||||
data_yaml,
|
||||
network_binary_identity_required(),
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn from_strings_with_binary_identity_required(
|
||||
policy: &str,
|
||||
data_yaml: &str,
|
||||
require_binary_identity: bool,
|
||||
) -> Result<Self> {
|
||||
let mut engine = regorus::Engine::new();
|
||||
engine
|
||||
.add_policy("policy.rego".into(), policy.into())
|
||||
.map_err(|e| miette::miette!("{e}"))?;
|
||||
let data_json = preprocess_yaml_data(data_yaml)?;
|
||||
emit_binary_identity_mode(require_binary_identity, "strings");
|
||||
let data_json = preprocess_yaml_data(data_yaml, require_binary_identity)?;
|
||||
engine
|
||||
.add_data_json(&data_json)
|
||||
.map_err(|e| miette::miette!("{e}"))?;
|
||||
@@ -193,11 +252,25 @@ impl OpaEngine {
|
||||
/// gap between user-specified symlink paths (e.g., `/usr/bin/python3`) and
|
||||
/// kernel-resolved canonical paths (e.g., `/usr/bin/python3.11`).
|
||||
pub fn from_proto_with_pid(proto: &ProtoSandboxPolicy, entrypoint_pid: u32) -> Result<Self> {
|
||||
Self::from_proto_with_pid_and_binary_identity_required(
|
||||
proto,
|
||||
entrypoint_pid,
|
||||
network_binary_identity_required(),
|
||||
)
|
||||
}
|
||||
|
||||
fn from_proto_with_pid_and_binary_identity_required(
|
||||
proto: &ProtoSandboxPolicy,
|
||||
entrypoint_pid: u32,
|
||||
require_binary_identity: bool,
|
||||
) -> Result<Self> {
|
||||
emit_binary_identity_mode(require_binary_identity, "proto");
|
||||
let data_json_str = proto_to_opa_data_json(proto, entrypoint_pid);
|
||||
|
||||
// Parse back to Value for preprocessing, then re-serialize
|
||||
let mut data: serde_json::Value = serde_json::from_str(&data_json_str)
|
||||
.map_err(|e| miette::miette!("internal: failed to parse proto JSON: {e}"))?;
|
||||
inject_runtime_policy_data(&mut data, require_binary_identity);
|
||||
|
||||
// Validate BEFORE expanding presets
|
||||
let (errors, warnings) = crate::l7::validate_l7_policies(&data);
|
||||
@@ -720,9 +793,10 @@ fn parse_process_policy(val: ®orus::Value) -> ProcessPolicy {
|
||||
}
|
||||
|
||||
/// Preprocess YAML policy data: parse, normalize, validate, expand access presets, return JSON.
|
||||
fn preprocess_yaml_data(yaml_str: &str) -> Result<String> {
|
||||
fn preprocess_yaml_data(yaml_str: &str, require_binary_identity: bool) -> Result<String> {
|
||||
let mut data: serde_json::Value = serde_yml::from_str(yaml_str)
|
||||
.map_err(|e| miette::miette!("failed to parse YAML data: {e}"))?;
|
||||
inject_runtime_policy_data(&mut data, require_binary_identity);
|
||||
|
||||
// Normalize port → ports for all endpoints so Rego always sees "ports" array.
|
||||
normalize_endpoint_ports(&mut data);
|
||||
@@ -2298,6 +2372,88 @@ process:
|
||||
assert!(eval_l7(&engine, &input));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn l7_get_allowed_by_rules_when_binary_identity_relaxed() {
|
||||
let engine =
|
||||
OpaEngine::from_strings_with_binary_identity_required(TEST_POLICY, L7_TEST_DATA, false)
|
||||
.expect("Failed to load relaxed L7 test data");
|
||||
let mut input = l7_input("api.example.com", 8080, "GET", "/repos/myorg/foo");
|
||||
input["exec"]["path"] = "".into();
|
||||
assert!(eval_l7(&engine, &input));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn relaxed_binary_identity_preserves_matched_policy_and_l7_for_proto() {
|
||||
let mut network_policies = std::collections::HashMap::new();
|
||||
network_policies.insert(
|
||||
"test_l7".to_string(),
|
||||
NetworkPolicyRule {
|
||||
name: "test_l7".to_string(),
|
||||
endpoints: vec![NetworkEndpoint {
|
||||
host: "host.k3d.internal".to_string(),
|
||||
port: 56123,
|
||||
protocol: "rest".to_string(),
|
||||
enforcement: "enforce".to_string(),
|
||||
rules: vec![L7Rule {
|
||||
allow: Some(L7Allow {
|
||||
method: "GET".to_string(),
|
||||
path: "/allowed".to_string(),
|
||||
command: String::new(),
|
||||
query: std::collections::HashMap::new(),
|
||||
operation_type: String::new(),
|
||||
operation_name: String::new(),
|
||||
fields: Vec::new(),
|
||||
params: std::collections::HashMap::new(),
|
||||
}),
|
||||
}],
|
||||
allowed_ips: vec!["192.168.0.0/16".to_string()],
|
||||
..Default::default()
|
||||
}],
|
||||
binaries: vec![NetworkBinary {
|
||||
path: "/usr/bin/curl".to_string(),
|
||||
..Default::default()
|
||||
}],
|
||||
},
|
||||
);
|
||||
let proto = ProtoSandboxPolicy {
|
||||
version: 1,
|
||||
filesystem: Some(ProtoFs {
|
||||
include_workdir: true,
|
||||
read_only: vec![],
|
||||
read_write: vec![],
|
||||
}),
|
||||
landlock: Some(openshell_core::proto::LandlockPolicy {
|
||||
compatibility: "best_effort".to_string(),
|
||||
}),
|
||||
process: Some(ProtoProc {
|
||||
run_as_user: "sandbox".to_string(),
|
||||
run_as_group: "sandbox".to_string(),
|
||||
}),
|
||||
network_policies,
|
||||
};
|
||||
let engine = OpaEngine::from_proto_with_pid_and_binary_identity_required(&proto, 0, false)
|
||||
.expect("engine from relaxed proto");
|
||||
let network_input = NetworkInput {
|
||||
host: "host.k3d.internal".into(),
|
||||
port: 56123,
|
||||
binary_path: PathBuf::new(),
|
||||
binary_sha256: String::new(),
|
||||
ancestors: vec![],
|
||||
cmdline_paths: vec![],
|
||||
};
|
||||
let action = engine.evaluate_network_action(&network_input).unwrap();
|
||||
assert_eq!(
|
||||
action,
|
||||
NetworkAction::Allow {
|
||||
matched_policy: Some("test_l7".to_string())
|
||||
}
|
||||
);
|
||||
|
||||
let mut input = l7_input("host.k3d.internal", 56123, "GET", "/allowed");
|
||||
input["exec"]["path"] = "".into();
|
||||
assert!(eval_l7(&engine, &input));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn l7_post_allowed_by_rules() {
|
||||
let engine = l7_engine();
|
||||
@@ -4626,6 +4782,46 @@ process:
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn relaxed_binary_identity_allows_declared_endpoint_without_binary_match() {
|
||||
let engine = OpaEngine::from_strings_with_binary_identity_required(
|
||||
TEST_POLICY,
|
||||
INFERENCE_TEST_DATA,
|
||||
false,
|
||||
)
|
||||
.expect("Failed to load relaxed binary identity test data");
|
||||
let input = NetworkInput {
|
||||
host: "api.anthropic.com".into(),
|
||||
port: 443,
|
||||
binary_path: PathBuf::from("/tmp/unlisted-agent"),
|
||||
binary_sha256: "unused".into(),
|
||||
ancestors: vec![],
|
||||
cmdline_paths: vec![],
|
||||
};
|
||||
|
||||
let action = engine.evaluate_network_action(&input).unwrap();
|
||||
assert_eq!(
|
||||
action,
|
||||
NetworkAction::Allow {
|
||||
matched_policy: Some("claude_code".to_string())
|
||||
},
|
||||
);
|
||||
assert!(
|
||||
engine.query_exact_declared_endpoint_host(&input).unwrap(),
|
||||
"relaxed identity should preserve exact declared endpoint handling"
|
||||
);
|
||||
|
||||
let undeclared = NetworkInput {
|
||||
host: "api.openai.com".into(),
|
||||
..input
|
||||
};
|
||||
let action = engine.evaluate_network_action(&undeclared).unwrap();
|
||||
assert!(
|
||||
matches!(action, NetworkAction::Deny { .. }),
|
||||
"relaxed identity must not allow undeclared endpoints"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unknown_endpoint_returns_deny() {
|
||||
let engine = inference_engine();
|
||||
|
||||
@@ -42,6 +42,8 @@ const TUNNEL_PROTOCOL_PEEK_POLL: std::time::Duration = std::time::Duration::from
|
||||
const TUNNEL_PROTOCOL_PEEK_POLL: std::time::Duration = std::time::Duration::from_millis(1);
|
||||
const INFERENCE_LOCAL_HOST: &str = "inference.local";
|
||||
const INFERENCE_LOCAL_PORT: u16 = 443;
|
||||
#[cfg(target_os = "linux")]
|
||||
const SIDECAR_SUPERVISOR_TOPOLOGY: &str = "sidecar";
|
||||
|
||||
/// Hostnames injected by compute drivers as `/etc/hosts` aliases for the host
|
||||
/// machine. Traffic to these names is eligible for the trusted-gateway SSRF
|
||||
@@ -1426,7 +1428,7 @@ fn resolve_owner_identity(
|
||||
})?;
|
||||
|
||||
let bin_hash = identity_cache
|
||||
.verify_or_cache(&bin_path)
|
||||
.verify_or_cache_process_exe(&bin_path, owner_pid)
|
||||
.map_err(|e| IdentityError {
|
||||
reason: format!("binary integrity check failed: {e}"),
|
||||
binary: Some(bin_path.clone()),
|
||||
@@ -1434,11 +1436,15 @@ fn resolve_owner_identity(
|
||||
ancestors: vec![],
|
||||
})?;
|
||||
|
||||
let ancestors = crate::procfs::collect_ancestor_binaries(owner_pid, entrypoint_pid);
|
||||
let ancestor_identities = collect_ancestor_identities(owner_pid, entrypoint_pid);
|
||||
let ancestors: Vec<PathBuf> = ancestor_identities
|
||||
.iter()
|
||||
.map(|(_, path)| path.clone())
|
||||
.collect();
|
||||
|
||||
for ancestor in &ancestors {
|
||||
for (ancestor_pid, ancestor) in &ancestor_identities {
|
||||
identity_cache
|
||||
.verify_or_cache(ancestor)
|
||||
.verify_or_cache_process_exe(ancestor, *ancestor_pid)
|
||||
.map_err(|e| IdentityError {
|
||||
reason: format!(
|
||||
"ancestor integrity check failed for {}: {e}",
|
||||
@@ -1463,6 +1469,31 @@ fn resolve_owner_identity(
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn collect_ancestor_identities(start_pid: u32, stop_pid: u32) -> Vec<(u32, PathBuf)> {
|
||||
const MAX_DEPTH: usize = 64;
|
||||
let mut ancestors = Vec::new();
|
||||
let mut current = start_pid;
|
||||
|
||||
for _ in 0..MAX_DEPTH {
|
||||
let parent_pid = match crate::procfs::read_ppid(current) {
|
||||
Some(parent) if parent > 0 && parent != current => parent,
|
||||
_ => break,
|
||||
};
|
||||
|
||||
if let Ok(path) = crate::procfs::binary_path(parent_pid.cast_signed()) {
|
||||
ancestors.push((parent_pid, path));
|
||||
}
|
||||
|
||||
if parent_pid == stop_pid || parent_pid == 1 {
|
||||
break;
|
||||
}
|
||||
current = parent_pid;
|
||||
}
|
||||
|
||||
ancestors
|
||||
}
|
||||
|
||||
/// Resolve the identity of the process owning a TCP peer connection.
|
||||
///
|
||||
/// Walks `/proc/<entrypoint_pid>/net/tcp` to find the socket inode, locates
|
||||
@@ -1472,10 +1503,10 @@ fn resolve_owner_identity(
|
||||
///
|
||||
/// This is the identity-resolution block of [`evaluate_opa_tcp`] extracted
|
||||
/// into a standalone helper so it can be exercised by Linux-only regression
|
||||
/// tests without a full OPA engine. The key invariant under test is that on
|
||||
/// a hot-swap of the peer binary, the failure mode is
|
||||
/// `"Binary integrity violation"` (from the identity cache) rather than
|
||||
/// `"Failed to stat ... (deleted)"` (from the kernel-tainted path).
|
||||
/// tests without a full OPA engine. The key hot-swap invariant under test is
|
||||
/// that display paths are stripped for policy/logging, while integrity hashing
|
||||
/// reads the live executable via `/proc/<pid>/exe` instead of the replacement
|
||||
/// file that now exists at the display path.
|
||||
#[cfg(target_os = "linux")]
|
||||
fn resolve_process_identity(
|
||||
entrypoint_pid: u32,
|
||||
@@ -1573,8 +1604,17 @@ fn evaluate_opa_tcp(
|
||||
}
|
||||
};
|
||||
|
||||
let pid = entrypoint_pid.load(Ordering::Acquire);
|
||||
if pid == 0 {
|
||||
if !crate::opa::network_binary_identity_required() {
|
||||
let result = evaluate_endpoint_only_opa(engine, host, port);
|
||||
debug!(
|
||||
"evaluate_opa_tcp endpoint-only: host={host} port={port} action={:?}",
|
||||
result.action
|
||||
);
|
||||
return result;
|
||||
}
|
||||
|
||||
let entrypoint_pid = entrypoint_pid.load(Ordering::Acquire);
|
||||
let Some(proc_net_anchor_pid) = proc_net_anchor_pid(entrypoint_pid) else {
|
||||
return deny(
|
||||
"entrypoint process not yet spawned".into(),
|
||||
None,
|
||||
@@ -1582,12 +1622,12 @@ fn evaluate_opa_tcp(
|
||||
vec![],
|
||||
vec![],
|
||||
);
|
||||
}
|
||||
};
|
||||
|
||||
let total_start = std::time::Instant::now();
|
||||
let peer_port = peer_addr.port();
|
||||
|
||||
let identity = match resolve_process_identity(pid, peer_port, identity_cache) {
|
||||
let identity = match resolve_process_identity(proc_net_anchor_pid, peer_port, identity_cache) {
|
||||
Ok(id) => id,
|
||||
Err(err) => {
|
||||
return deny(
|
||||
@@ -1641,6 +1681,52 @@ fn evaluate_opa_tcp(
|
||||
result
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn proc_net_anchor_pid(entrypoint_pid: u32) -> Option<u32> {
|
||||
if entrypoint_pid != 0 {
|
||||
return Some(entrypoint_pid);
|
||||
}
|
||||
sidecar_topology_enabled().then(std::process::id)
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn sidecar_topology_enabled() -> bool {
|
||||
std::env::var(openshell_core::sandbox_env::SUPERVISOR_TOPOLOGY)
|
||||
.is_ok_and(|value| value == SIDECAR_SUPERVISOR_TOPOLOGY)
|
||||
}
|
||||
|
||||
fn evaluate_endpoint_only_opa(engine: &OpaEngine, host: &str, port: u16) -> ConnectDecision {
|
||||
let input = crate::opa::NetworkInput {
|
||||
host: host.to_string(),
|
||||
port,
|
||||
binary_path: PathBuf::new(),
|
||||
binary_sha256: String::new(),
|
||||
ancestors: vec![],
|
||||
cmdline_paths: vec![],
|
||||
};
|
||||
|
||||
match engine.evaluate_network_action_with_generation(&input) {
|
||||
Ok((action, generation)) => ConnectDecision {
|
||||
action,
|
||||
generation,
|
||||
binary: None,
|
||||
binary_pid: None,
|
||||
ancestors: vec![],
|
||||
cmdline_paths: vec![],
|
||||
},
|
||||
Err(e) => ConnectDecision {
|
||||
action: NetworkAction::Deny {
|
||||
reason: format!("policy evaluation error: {e}"),
|
||||
},
|
||||
generation: engine.current_generation(),
|
||||
binary: None,
|
||||
binary_pid: None,
|
||||
ancestors: vec![],
|
||||
cmdline_paths: vec![],
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// Non-Linux stub: OPA identity binding requires /proc.
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
fn evaluate_opa_tcp(
|
||||
@@ -1648,9 +1734,13 @@ fn evaluate_opa_tcp(
|
||||
engine: &OpaEngine,
|
||||
_identity_cache: &BinaryIdentityCache,
|
||||
_entrypoint_pid: &AtomicU32,
|
||||
_host: &str,
|
||||
_port: u16,
|
||||
host: &str,
|
||||
port: u16,
|
||||
) -> ConnectDecision {
|
||||
if !crate::opa::network_binary_identity_required() {
|
||||
return evaluate_endpoint_only_opa(engine, host, port);
|
||||
}
|
||||
|
||||
ConnectDecision {
|
||||
action: NetworkAction::Deny {
|
||||
reason: "identity binding unavailable on this platform".into(),
|
||||
@@ -2152,14 +2242,24 @@ fn query_l7_route_snapshot(
|
||||
};
|
||||
|
||||
match engine.query_endpoint_configs_with_generation(&input) {
|
||||
Ok((vals, generation)) => Some(L7RouteSnapshot {
|
||||
configs: vals
|
||||
Ok((vals, generation)) => {
|
||||
let configs: Vec<_> = vals
|
||||
.into_iter()
|
||||
.filter_map(|val| crate::l7::parse_l7_config(&val))
|
||||
.map(|config| L7ConfigSnapshot { config })
|
||||
.collect(),
|
||||
generation,
|
||||
}),
|
||||
.collect();
|
||||
debug!(
|
||||
host,
|
||||
port,
|
||||
generation,
|
||||
config_count = configs.len(),
|
||||
"Forward proxy L7 route lookup complete"
|
||||
);
|
||||
Some(L7RouteSnapshot {
|
||||
configs,
|
||||
generation,
|
||||
})
|
||||
}
|
||||
Err(e) => {
|
||||
let event = NetworkActivityBuilder::new(openshell_ocsf::ctx::ctx())
|
||||
.activity(ActivityId::Fail)
|
||||
@@ -3337,10 +3437,29 @@ async fn handle_forward_proxy(
|
||||
}
|
||||
};
|
||||
let policy_str = matched_policy.as_deref().unwrap_or("-");
|
||||
debug!(
|
||||
host = %host_lc,
|
||||
port,
|
||||
binary = %binary_str,
|
||||
binary_pid = %pid_str,
|
||||
matched_policy = %policy_str,
|
||||
decision_generation = decision.generation,
|
||||
current_generation = opa_engine.current_generation(),
|
||||
action = ?decision.action,
|
||||
"Forward proxy L4 policy decision"
|
||||
);
|
||||
let sandbox_entrypoint_pid = entrypoint_pid.load(Ordering::Acquire);
|
||||
let forward_generation_guard = match opa_engine.generation_guard(decision.generation) {
|
||||
Ok(guard) => guard,
|
||||
Err(e) => {
|
||||
warn!(
|
||||
host = %host_lc,
|
||||
port,
|
||||
decision_generation = decision.generation,
|
||||
current_generation = opa_engine.current_generation(),
|
||||
error = %e,
|
||||
"Forward proxy rejected request because policy generation changed after L4 decision"
|
||||
);
|
||||
emit_l7_tunnel_close_after_policy_change(&host_lc, port, e);
|
||||
emit_activity_simple(activity_tx, true, "policy_stale");
|
||||
respond(
|
||||
@@ -3401,6 +3520,15 @@ async fn handle_forward_proxy(
|
||||
&& !route.configs.is_empty()
|
||||
{
|
||||
if route.generation != forward_generation_guard.captured_generation() {
|
||||
warn!(
|
||||
host = %host_lc,
|
||||
port,
|
||||
decision_generation = decision.generation,
|
||||
guard_generation = forward_generation_guard.captured_generation(),
|
||||
route_generation = route.generation,
|
||||
current_generation = opa_engine.current_generation(),
|
||||
"Forward proxy rejected request because L7 route lookup used a different policy generation"
|
||||
);
|
||||
emit_l7_tunnel_close_after_policy_change(
|
||||
&host_lc,
|
||||
port,
|
||||
@@ -3426,6 +3554,14 @@ async fn handle_forward_proxy(
|
||||
let tunnel_engine = match opa_engine.clone_engine_for_tunnel(route.generation) {
|
||||
Ok(engine) => engine,
|
||||
Err(e) => {
|
||||
warn!(
|
||||
host = %host_lc,
|
||||
port,
|
||||
route_generation = route.generation,
|
||||
current_generation = opa_engine.current_generation(),
|
||||
error = %e,
|
||||
"Forward proxy rejected request because L7 tunnel engine could not be cloned"
|
||||
);
|
||||
emit_l7_tunnel_close_after_policy_change(&host_lc, port, e);
|
||||
emit_activity_simple(activity_tx, true, "policy_stale");
|
||||
respond(
|
||||
@@ -4105,6 +4241,14 @@ async fn handle_forward_proxy(
|
||||
};
|
||||
|
||||
if let Err(e) = forward_generation_guard.ensure_current() {
|
||||
warn!(
|
||||
host = %host_lc,
|
||||
port,
|
||||
captured_generation = forward_generation_guard.captured_generation(),
|
||||
current_generation = forward_generation_guard.current_generation(),
|
||||
error = %e,
|
||||
"Forward proxy rejected request because policy changed before upstream connect"
|
||||
);
|
||||
emit_l7_tunnel_close_after_policy_change(&host_lc, port, e);
|
||||
emit_activity_simple(activity_tx, true, "policy_stale");
|
||||
respond(
|
||||
@@ -4243,6 +4387,14 @@ async fn handle_forward_proxy(
|
||||
};
|
||||
|
||||
if let Err(e) = forward_generation_guard.ensure_current() {
|
||||
warn!(
|
||||
host = %host_lc,
|
||||
port,
|
||||
captured_generation = forward_generation_guard.captured_generation(),
|
||||
current_generation = forward_generation_guard.current_generation(),
|
||||
error = %e,
|
||||
"Forward proxy rejected request because policy changed before relay"
|
||||
);
|
||||
emit_l7_tunnel_close_after_policy_change(&host_lc, port, e);
|
||||
respond(
|
||||
client,
|
||||
@@ -4379,6 +4531,46 @@ mod tests {
|
||||
use tokio::io::{AsyncRead, AsyncReadExt, AsyncWriteExt};
|
||||
use tokio::net::{TcpListener, TcpStream};
|
||||
|
||||
#[test]
|
||||
fn endpoint_only_opa_allows_declared_endpoint_without_process_identity() {
|
||||
let policy = include_str!("../data/sandbox-policy.rego");
|
||||
let data = r#"
|
||||
version: 1
|
||||
network_policies:
|
||||
test_l7:
|
||||
name: test_l7
|
||||
endpoints:
|
||||
- host: host.k3d.internal
|
||||
port: 56123
|
||||
protocol: rest
|
||||
enforcement: enforce
|
||||
rules:
|
||||
- allow:
|
||||
method: GET
|
||||
path: /allowed
|
||||
binaries:
|
||||
- path: /usr/bin/curl
|
||||
"#;
|
||||
let engine = OpaEngine::from_strings_with_binary_identity_required(policy, data, false)
|
||||
.expect("relaxed engine");
|
||||
|
||||
let decision = evaluate_endpoint_only_opa(&engine, "host.k3d.internal", 56123);
|
||||
assert_eq!(
|
||||
decision.action,
|
||||
NetworkAction::Allow {
|
||||
matched_policy: Some("test_l7".to_string()),
|
||||
}
|
||||
);
|
||||
assert!(decision.binary.is_none());
|
||||
assert!(decision.ancestors.is_empty());
|
||||
|
||||
let denied = evaluate_endpoint_only_opa(&engine, "api.example.com", 443);
|
||||
assert!(
|
||||
matches!(denied.action, NetworkAction::Deny { .. }),
|
||||
"endpoint-only mode must still deny undeclared endpoints"
|
||||
);
|
||||
}
|
||||
|
||||
fn websocket_l7_config(
|
||||
protocol: crate::l7::L7Protocol,
|
||||
websocket_credential_rewrite: bool,
|
||||
@@ -7992,27 +8184,23 @@ network_policies:
|
||||
assert_eq!(resp_str[body_start..].len(), cl);
|
||||
}
|
||||
|
||||
/// End-to-end regression for the `docker cp` hot-swap hazard that
|
||||
/// motivated `binary_path()` stripping the kernel's `" (deleted)"`
|
||||
/// suffix (PR #844).
|
||||
/// End-to-end regression for the `docker cp` hot-swap hazard around
|
||||
/// unlinked process executables.
|
||||
///
|
||||
/// Before the strip, the identity-resolution chain inside
|
||||
/// `evaluate_opa_tcp` failed with `"Failed to stat
|
||||
/// /opt/openshell/bin/openshell-sandbox (deleted)"` because
|
||||
/// `BinaryIdentityCache::verify_or_cache()` tried to `metadata()` the
|
||||
/// tainted path. That masked the real security signal: a live process
|
||||
/// was now bound to a *different* binary on disk than the one that was
|
||||
/// TOFU-cached. After the strip, `binary_path()` returns a path that
|
||||
/// stats fine, the cache rehashes the new bytes, and the hash mismatch
|
||||
/// surfaces as a `Binary integrity violation` error — the contract this
|
||||
/// PR is trying to establish.
|
||||
/// `binary_path()` strips the kernel's `" (deleted)"` suffix so policy
|
||||
/// identity and logs use a clean display path. Integrity verification must
|
||||
/// not hash that display path after a hot-swap, because it may now point to
|
||||
/// unrelated replacement bytes. It hashes `/proc/<pid>/exe` instead, which
|
||||
/// resolves to the live executable inode even after the original path was
|
||||
/// unlinked.
|
||||
///
|
||||
/// Test shape (from the review comment on the initial PR):
|
||||
/// 1. Start a `TcpListener` in the test process.
|
||||
/// 2. Copy `/bin/bash` to a temp path we control.
|
||||
/// 3. Prime `BinaryIdentityCache` with that temp binary's hash.
|
||||
/// 4. Spawn the temp bash as a child with a `/dev/tcp` one-liner that
|
||||
/// opens a real TCP connection to the listener and holds it open.
|
||||
/// opens a real TCP connection to the listener and holds it open
|
||||
/// inside the bash process.
|
||||
/// 5. Accept the connection on the listener side and capture the peer's
|
||||
/// ephemeral port — that's what `resolve_process_identity` uses to
|
||||
/// walk `/proc/net/tcp` back to the child PID.
|
||||
@@ -8022,13 +8210,12 @@ network_policies:
|
||||
/// now readlink to `" (deleted)"` OR the overwritten file, depending
|
||||
/// on whether the filesystem reused the inode.
|
||||
/// 7. Call `resolve_process_identity` and assert:
|
||||
/// - the error reason contains `"Binary integrity violation"` (the
|
||||
/// cache detected the tampered on-disk bytes), and
|
||||
/// - the error reason does NOT contain `"Failed to stat"` or
|
||||
/// `"(deleted)"` (the old pre-strip failure mode).
|
||||
/// - identity resolution succeeds using the live executable hash, and
|
||||
/// - the returned display path does not contain the kernel-added
|
||||
/// `"(deleted)"` suffix.
|
||||
#[cfg(target_os = "linux")]
|
||||
#[test]
|
||||
fn resolve_process_identity_surfaces_binary_integrity_violation_on_hot_swap() {
|
||||
fn resolve_process_identity_hashes_live_exe_after_hot_swap() {
|
||||
use crate::identity::BinaryIdentityCache;
|
||||
use std::io::Read;
|
||||
use std::net::TcpListener;
|
||||
@@ -8060,9 +8247,12 @@ network_policies:
|
||||
assert!(!v1_hash.is_empty());
|
||||
|
||||
// 4. Spawn the temp bash with a /dev/tcp one-liner that opens a real
|
||||
// connection to the listener and sleeps to keep it open. The
|
||||
// `read -t` blocks on stdin so the shell stays resident.
|
||||
let script = format!("exec 3<>/dev/tcp/127.0.0.1/{listener_port}; sleep 30 <&3");
|
||||
// connection to the listener and blocks in bash's `read` builtin
|
||||
// to keep it open. Do not use an external command like `sleep`:
|
||||
// it inherits the socket fd and intentionally trips the shared
|
||||
// socket ambiguity guard instead of exercising the hot-swap path.
|
||||
let script =
|
||||
format!("exec 3<>/dev/tcp/127.0.0.1/{listener_port}; read -r -t 30 _ <&3 || true");
|
||||
let mut child = Command::new(&bash_v1)
|
||||
.arg("-c")
|
||||
.arg(&script)
|
||||
@@ -8106,10 +8296,11 @@ network_policies:
|
||||
std::fs::write(&bash_v1, tampered_bytes).expect("write replacement bytes");
|
||||
|
||||
// 7. Resolve identity through the real helper and assert the
|
||||
// contract: we want "Binary integrity violation", not
|
||||
// "Failed to stat ... (deleted)".
|
||||
// contract: hash the live executable via /proc/<pid>/exe while
|
||||
// returning a clean display path for policy/logging.
|
||||
let test_pid = std::process::id();
|
||||
let result = resolve_process_identity(test_pid, peer_port, &cache);
|
||||
let child_pid = child.id();
|
||||
|
||||
// Always clean up the child before asserting so a failure doesn't
|
||||
// leak a sleeping process across test runs.
|
||||
@@ -8117,40 +8308,29 @@ network_policies:
|
||||
let _ = child.wait();
|
||||
|
||||
match result {
|
||||
Ok(_) => panic!(
|
||||
"resolve_process_identity unexpectedly succeeded after hot-swap; \
|
||||
the cache should have detected the tampered on-disk bytes"
|
||||
),
|
||||
Err(err) => {
|
||||
assert!(
|
||||
err.reason.contains("Binary integrity violation"),
|
||||
"expected 'Binary integrity violation' error, got: {}",
|
||||
err.reason
|
||||
Ok(identity) => {
|
||||
assert_eq!(
|
||||
identity.binary_pid, child_pid,
|
||||
"expected the hot-swapped bash child to own the socket"
|
||||
);
|
||||
assert_eq!(
|
||||
identity.bin_path, bash_v1,
|
||||
"expected stripped display path to remain the original binary path"
|
||||
);
|
||||
assert!(
|
||||
!err.reason.contains("Failed to stat"),
|
||||
"pre-PR-#844 failure mode leaked: {}",
|
||||
err.reason
|
||||
!identity.bin_path.to_string_lossy().contains("(deleted)"),
|
||||
"resolved binary path still tainted: {}",
|
||||
identity.bin_path.display()
|
||||
);
|
||||
assert!(
|
||||
!err.reason.contains("(deleted)"),
|
||||
"resolved path still contains '(deleted)' suffix: {}",
|
||||
err.reason
|
||||
assert_eq!(
|
||||
identity.bin_hash, v1_hash,
|
||||
"expected integrity hash from the live executable, not replacement bytes"
|
||||
);
|
||||
// The binary field should be populated — we did resolve a
|
||||
// path before failing.
|
||||
assert!(
|
||||
err.binary.is_some(),
|
||||
"expected resolved binary path on integrity failure"
|
||||
);
|
||||
if let Some(path) = &err.binary {
|
||||
assert!(
|
||||
!path.to_string_lossy().contains("(deleted)"),
|
||||
"resolved binary path still tainted: {}",
|
||||
path.display()
|
||||
);
|
||||
}
|
||||
}
|
||||
Err(err) => panic!(
|
||||
"resolve_process_identity failed after hot-swap; expected live-exe identity: {}",
|
||||
err.reason
|
||||
),
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -201,7 +201,9 @@ pub async fn run_networking(
|
||||
let (tls_state, ca_file_paths) = if matches!(policy.network.mode, NetworkMode::Proxy) {
|
||||
match SandboxCa::generate() {
|
||||
Ok(ca) => {
|
||||
let tls_dir = std::path::Path::new("/etc/openshell-tls");
|
||||
let tls_dir = std::env::var(openshell_core::sandbox_env::PROXY_TLS_DIR)
|
||||
.unwrap_or_else(|_| "/etc/openshell-tls".to_string());
|
||||
let tls_dir = std::path::Path::new(&tls_dir);
|
||||
let system_ca_bundle = read_system_ca_bundle();
|
||||
match write_ca_files(&ca, tls_dir, &system_ca_bundle) {
|
||||
Ok(paths) => {
|
||||
|
||||
@@ -19,6 +19,8 @@ pub mod skills;
|
||||
pub mod ssh;
|
||||
pub mod supervisor_session;
|
||||
|
||||
mod unix_socket;
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
pub mod bypass_monitor;
|
||||
#[cfg(target_os = "linux")]
|
||||
|
||||
@@ -285,43 +285,22 @@ impl NetworkNamespace {
|
||||
// monitor can see log entries from the sandbox namespace.
|
||||
enable_nf_log_all_netns();
|
||||
|
||||
// Try combined ruleset with log rules first. Log rules must appear
|
||||
// before reject rules in the chain so packets are logged before being
|
||||
// rejected. If the kernel lacks nft_log support, fall back to the
|
||||
// reject-only ruleset.
|
||||
let ruleset_with_log =
|
||||
nft_ruleset::generate_bypass_ruleset(&host_ip_str, proxy_port, Some(&log_prefix));
|
||||
let commands =
|
||||
nft_ruleset::generate_bypass_commands(&host_ip_str, proxy_port, Some(&log_prefix));
|
||||
|
||||
if let Err(e) = run_nft_netns(&self.name, &nft_path, &ruleset_with_log) {
|
||||
if let Err(e) = run_nft_commands_netns(&self.name, &nft_path, &commands) {
|
||||
openshell_ocsf::ocsf_emit!(
|
||||
openshell_ocsf::ConfigStateChangeBuilder::new(openshell_ocsf::ctx::ctx())
|
||||
.severity(openshell_ocsf::SeverityId::Low)
|
||||
.severity(openshell_ocsf::SeverityId::Medium)
|
||||
.status(openshell_ocsf::StatusId::Failure)
|
||||
.state(openshell_ocsf::StateId::Other, "degraded")
|
||||
.state(openshell_ocsf::StateId::Disabled, "failed")
|
||||
.message(format!(
|
||||
"Failed to install bypass log rules (non-fatal), falling back to reject-only [ns:{}]: {e}",
|
||||
"Failed to install bypass detection rules [ns:{}]: {e}",
|
||||
self.name
|
||||
))
|
||||
.build()
|
||||
);
|
||||
|
||||
let ruleset_no_log =
|
||||
nft_ruleset::generate_bypass_ruleset(&host_ip_str, proxy_port, None);
|
||||
|
||||
if let Err(e) = run_nft_netns(&self.name, &nft_path, &ruleset_no_log) {
|
||||
openshell_ocsf::ocsf_emit!(
|
||||
openshell_ocsf::ConfigStateChangeBuilder::new(openshell_ocsf::ctx::ctx())
|
||||
.severity(openshell_ocsf::SeverityId::Medium)
|
||||
.status(openshell_ocsf::StatusId::Failure)
|
||||
.state(openshell_ocsf::StateId::Disabled, "failed")
|
||||
.message(format!(
|
||||
"Failed to install bypass detection rules [ns:{}]: {e}",
|
||||
self.name
|
||||
))
|
||||
.build()
|
||||
);
|
||||
return Err(e);
|
||||
}
|
||||
return Err(e);
|
||||
}
|
||||
|
||||
openshell_ocsf::ocsf_emit!(
|
||||
@@ -467,6 +446,193 @@ pub fn create_netns_for_proxy(
|
||||
}
|
||||
}
|
||||
|
||||
/// Install pod-network bypass enforcement for Kubernetes sidecar topology.
|
||||
///
|
||||
/// This runs in the current network namespace, not in a per-workload netns.
|
||||
/// The rules allow loopback and the sidecar proxy UID, then reject direct
|
||||
/// TCP/UDP egress from other UIDs so traffic must use the sidecar's local
|
||||
/// proxy.
|
||||
///
|
||||
/// # Errors
|
||||
///
|
||||
/// Returns an error when `nft` is unavailable or the ruleset cannot be loaded.
|
||||
pub fn install_sidecar_bypass_rules(proxy_uid: u32) -> Result<()> {
|
||||
match install_sidecar_nft_bypass_rules(proxy_uid) {
|
||||
Ok(()) => Ok(()),
|
||||
Err(nft_error) => {
|
||||
warn!(
|
||||
error = %nft_error,
|
||||
"Failed to install nftables sidecar rules; trying iptables-legacy fallback"
|
||||
);
|
||||
install_sidecar_iptables_legacy_bypass_rules(proxy_uid).map_err(|iptables_error| {
|
||||
miette::miette!(
|
||||
"sidecar nft ruleset load failed: {nft_error}; sidecar iptables-legacy fallback failed: {iptables_error}"
|
||||
)
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn install_sidecar_nft_bypass_rules(proxy_uid: u32) -> Result<()> {
|
||||
let nft_cmd = find_nft().ok_or_else(|| {
|
||||
miette::miette!(
|
||||
"trusted nft helper not found; sidecar network enforcement requires nftables"
|
||||
)
|
||||
})?;
|
||||
let log_prefix = Some("openshell:sidecar-bypass:");
|
||||
let commands = nft_ruleset::generate_sidecar_bypass_commands(proxy_uid, log_prefix);
|
||||
run_nft_commands_current_namespace(&nft_cmd, &commands)
|
||||
}
|
||||
|
||||
const SIDECAR_IPTABLES_CHAIN: &str = "OPENSHELL_SIDECAR_BYPASS";
|
||||
const PROC_NET_IF_INET6_PATH: &str = "/proc/net/if_inet6";
|
||||
|
||||
fn install_sidecar_iptables_legacy_bypass_rules(proxy_uid: u32) -> Result<()> {
|
||||
let ipv4_filter_tool = find_iptables_legacy().ok_or_else(|| {
|
||||
miette::miette!(
|
||||
"trusted iptables-legacy helper not found; sidecar network enforcement fallback unavailable"
|
||||
)
|
||||
})?;
|
||||
|
||||
let ipv6_fence_tool = if current_namespace_has_non_loopback_ipv6()? {
|
||||
Some(find_ip6tables_legacy().ok_or_else(|| {
|
||||
miette::miette!(
|
||||
"trusted ip6tables-legacy helper not found; sidecar network enforcement fallback cannot fence IPv6"
|
||||
)
|
||||
})?)
|
||||
} else {
|
||||
warn!(
|
||||
"Skipping IPv6 sidecar iptables-legacy fallback because the current namespace has no non-loopback IPv6 interface"
|
||||
);
|
||||
None
|
||||
};
|
||||
|
||||
cleanup_sidecar_iptables_legacy_rule_families(&ipv4_filter_tool, ipv6_fence_tool.as_deref());
|
||||
|
||||
if let Err(e) = install_sidecar_iptables_legacy_family_rules(
|
||||
&ipv4_filter_tool,
|
||||
proxy_uid,
|
||||
"icmp-port-unreachable",
|
||||
) {
|
||||
cleanup_sidecar_iptables_legacy_rule_families(
|
||||
&ipv4_filter_tool,
|
||||
ipv6_fence_tool.as_deref(),
|
||||
);
|
||||
return Err(e);
|
||||
}
|
||||
|
||||
if let Some(ipv6_fence_tool) = ipv6_fence_tool
|
||||
&& let Err(e) = install_sidecar_iptables_legacy_family_rules(
|
||||
&ipv6_fence_tool,
|
||||
proxy_uid,
|
||||
"icmp6-port-unreachable",
|
||||
)
|
||||
{
|
||||
cleanup_sidecar_iptables_legacy_rule_families(&ipv4_filter_tool, Some(&ipv6_fence_tool));
|
||||
return Err(e);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn current_namespace_has_non_loopback_ipv6() -> Result<bool> {
|
||||
match std::fs::read_to_string(PROC_NET_IF_INET6_PATH) {
|
||||
Ok(content) => Ok(has_non_loopback_ipv6_interface(&content)),
|
||||
Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(false),
|
||||
Err(e) => Err(miette::miette!(
|
||||
"failed to inspect {PROC_NET_IF_INET6_PATH} before installing sidecar IPv6 fence: {e}"
|
||||
)),
|
||||
}
|
||||
}
|
||||
|
||||
fn has_non_loopback_ipv6_interface(content: &str) -> bool {
|
||||
content.lines().any(|line| {
|
||||
line.split_whitespace()
|
||||
.nth(5)
|
||||
.is_some_and(|iface| iface != "lo")
|
||||
})
|
||||
}
|
||||
|
||||
fn install_sidecar_iptables_legacy_family_rules(
|
||||
cmd: &str,
|
||||
proxy_uid: u32,
|
||||
udp_reject_with: &str,
|
||||
) -> Result<()> {
|
||||
let proxy_uid_arg = proxy_uid.to_string();
|
||||
let commands: Vec<Vec<&str>> = vec![
|
||||
vec!["-N", SIDECAR_IPTABLES_CHAIN],
|
||||
vec!["-A", SIDECAR_IPTABLES_CHAIN, "-o", "lo", "-j", "ACCEPT"],
|
||||
vec![
|
||||
"-A",
|
||||
SIDECAR_IPTABLES_CHAIN,
|
||||
"-m",
|
||||
"conntrack",
|
||||
"--ctstate",
|
||||
"ESTABLISHED,RELATED",
|
||||
"-j",
|
||||
"ACCEPT",
|
||||
],
|
||||
vec![
|
||||
"-A",
|
||||
SIDECAR_IPTABLES_CHAIN,
|
||||
"-m",
|
||||
"owner",
|
||||
"--uid-owner",
|
||||
&proxy_uid_arg,
|
||||
"-j",
|
||||
"ACCEPT",
|
||||
],
|
||||
vec![
|
||||
"-A",
|
||||
SIDECAR_IPTABLES_CHAIN,
|
||||
"-p",
|
||||
"tcp",
|
||||
"-j",
|
||||
"REJECT",
|
||||
"--reject-with",
|
||||
"tcp-reset",
|
||||
],
|
||||
vec![
|
||||
"-A",
|
||||
SIDECAR_IPTABLES_CHAIN,
|
||||
"-p",
|
||||
"udp",
|
||||
"-j",
|
||||
"REJECT",
|
||||
"--reject-with",
|
||||
udp_reject_with,
|
||||
],
|
||||
vec!["-A", "OUTPUT", "-j", SIDECAR_IPTABLES_CHAIN],
|
||||
];
|
||||
|
||||
for args in commands {
|
||||
if let Err(e) = run_iptables_legacy_current_namespace(cmd, &args) {
|
||||
cleanup_sidecar_iptables_legacy_rules(cmd);
|
||||
return Err(e);
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn cleanup_sidecar_iptables_legacy_rules(iptables_cmd: &str) {
|
||||
while run_iptables_legacy_current_namespace(
|
||||
iptables_cmd,
|
||||
&["-D", "OUTPUT", "-j", SIDECAR_IPTABLES_CHAIN],
|
||||
)
|
||||
.is_ok()
|
||||
{}
|
||||
let _ = run_iptables_legacy_current_namespace(iptables_cmd, &["-F", SIDECAR_IPTABLES_CHAIN]);
|
||||
let _ = run_iptables_legacy_current_namespace(iptables_cmd, &["-X", SIDECAR_IPTABLES_CHAIN]);
|
||||
}
|
||||
|
||||
fn cleanup_sidecar_iptables_legacy_rule_families(ipv4_cmd: &str, ipv6_cmd: Option<&str>) {
|
||||
cleanup_sidecar_iptables_legacy_rules(ipv4_cmd);
|
||||
if let Some(ipv6_cmd) = ipv6_cmd {
|
||||
cleanup_sidecar_iptables_legacy_rules(ipv6_cmd);
|
||||
}
|
||||
}
|
||||
|
||||
/// Run an `ip` command on the host.
|
||||
fn run_ip(args: &[&str]) -> Result<()> {
|
||||
let ip_path = find_trusted_binary("ip", IP_SEARCH_PATHS)?;
|
||||
@@ -490,6 +656,68 @@ fn run_ip(args: &[&str]) -> Result<()> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn run_iptables_legacy_current_namespace(iptables_cmd: &str, args: &[&str]) -> Result<()> {
|
||||
debug!(
|
||||
command = %format!("{iptables_cmd} {}", args.join(" ")),
|
||||
"Running iptables-legacy sidecar command"
|
||||
);
|
||||
|
||||
let output = Command::new(iptables_cmd)
|
||||
.args(args)
|
||||
.output()
|
||||
.into_diagnostic()?;
|
||||
|
||||
if !output.status.success() {
|
||||
let stderr = String::from_utf8_lossy(&output.stderr);
|
||||
return Err(miette::miette!(
|
||||
"{iptables_cmd} {} failed: {}",
|
||||
args.join(" "),
|
||||
stderr.trim()
|
||||
));
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Run a sequence of nft commands in the current network namespace.
|
||||
///
|
||||
/// Each command is executed as a separate `nft` invocation to avoid atomic
|
||||
/// batch rollback (where one unsupported expression like `ct state` or `log`
|
||||
/// causes the entire transaction, including table creation, to fail).
|
||||
///
|
||||
/// Commands marked as non-required are allowed to fail with a warning.
|
||||
/// Required commands that fail abort the sequence immediately.
|
||||
fn run_nft_commands_current_namespace(
|
||||
nft_cmd: &str,
|
||||
commands: &[nft_ruleset::NftCommand],
|
||||
) -> Result<()> {
|
||||
for cmd in commands {
|
||||
let args_str = cmd.args.join(" ");
|
||||
debug!(command = %format!("{nft_cmd} {args_str}"), "Running nft command");
|
||||
|
||||
let output = Command::new(nft_cmd)
|
||||
.args(&cmd.args)
|
||||
.output()
|
||||
.into_diagnostic()?;
|
||||
|
||||
if !output.status.success() {
|
||||
let stderr = String::from_utf8_lossy(&output.stderr);
|
||||
if cmd.required {
|
||||
return Err(miette::miette!(
|
||||
"{nft_cmd} {args_str} failed: {}",
|
||||
stderr.trim()
|
||||
));
|
||||
}
|
||||
warn!(
|
||||
command = %args_str,
|
||||
error = %stderr.trim(),
|
||||
"non-required nft command failed (continuing)"
|
||||
);
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Run an `ip` command inside a network namespace via `nsenter --net=`.
|
||||
///
|
||||
/// We use `nsenter` instead of `ip netns exec` because `ip netns exec`
|
||||
@@ -532,47 +760,51 @@ fn run_ip_netns(netns: &str, args: &[&str]) -> Result<()> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Load an nftables ruleset inside a network namespace via `nsenter --net=`.
|
||||
/// Run a sequence of nft commands inside a network namespace via `nsenter --net=`.
|
||||
///
|
||||
/// Writes the ruleset to a temp file and loads it with `nft -f <path>`.
|
||||
/// A temp file is used instead of piping to stdin (`nft -f -`) because
|
||||
/// `nft` resolves `-` to `/dev/stdin`, which may not exist in minimal
|
||||
/// VM guest environments (e.g. virtiofs rootfs without /proc mounted
|
||||
/// at nft invocation time).
|
||||
fn run_nft_netns(netns: &str, nft_cmd: &str, ruleset: &str) -> Result<()> {
|
||||
use std::io::Write;
|
||||
let mut tmp = tempfile::Builder::new()
|
||||
.prefix("openshell-nft-")
|
||||
.suffix(".conf")
|
||||
.tempfile()
|
||||
.into_diagnostic()?;
|
||||
tmp.write_all(ruleset.as_bytes()).into_diagnostic()?;
|
||||
let ruleset_path = tmp.path().to_string_lossy().to_string();
|
||||
|
||||
/// Each command is executed as a separate invocation to avoid atomic batch
|
||||
/// rollback. See [`run_nft_commands_current_namespace`] for rationale.
|
||||
fn run_nft_commands_netns(
|
||||
netns: &str,
|
||||
nft_cmd: &str,
|
||||
commands: &[nft_ruleset::NftCommand],
|
||||
) -> Result<()> {
|
||||
let nsenter_path = find_trusted_binary("nsenter", NSENTER_SEARCH_PATHS)?;
|
||||
let ns_path = format!("/var/run/netns/{netns}");
|
||||
let net_flag = format!("--net={ns_path}");
|
||||
|
||||
debug!(
|
||||
command = %format!("{nsenter_path} {net_flag} -- {nft_cmd} -f {ruleset_path}"),
|
||||
"Loading nftables ruleset in namespace"
|
||||
);
|
||||
for cmd in commands {
|
||||
let args_str = cmd.args.join(" ");
|
||||
debug!(
|
||||
command = %format!("{nsenter_path} {net_flag} -- {nft_cmd} {args_str}"),
|
||||
"Running nft command in namespace"
|
||||
);
|
||||
|
||||
let output = Command::new(nsenter_path)
|
||||
.args([net_flag.as_str(), "--", nft_cmd, "-f", &ruleset_path])
|
||||
.output()
|
||||
.into_diagnostic()?;
|
||||
let mut full_args = vec![net_flag.as_str(), "--", nft_cmd];
|
||||
let arg_refs: Vec<&str> = cmd.args.iter().map(String::as_str).collect();
|
||||
full_args.extend(&arg_refs);
|
||||
|
||||
drop(tmp);
|
||||
let output = Command::new(nsenter_path)
|
||||
.args(&full_args)
|
||||
.output()
|
||||
.into_diagnostic()?;
|
||||
|
||||
if !output.status.success() {
|
||||
let stderr = String::from_utf8_lossy(&output.stderr);
|
||||
return Err(miette::miette!(
|
||||
"nft ruleset load failed in netns {netns}: {}",
|
||||
stderr.trim()
|
||||
));
|
||||
if !output.status.success() {
|
||||
let stderr = String::from_utf8_lossy(&output.stderr);
|
||||
if cmd.required {
|
||||
return Err(miette::miette!(
|
||||
"nft {args_str} failed in netns {netns}: {}",
|
||||
stderr.trim()
|
||||
));
|
||||
}
|
||||
warn!(
|
||||
command = %args_str,
|
||||
error = %stderr.trim(),
|
||||
netns = %netns,
|
||||
"non-required nft command failed in namespace (continuing)"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -605,6 +837,16 @@ fn enable_nf_log_all_netns() {
|
||||
|
||||
/// Well-known paths where nft may be installed.
|
||||
const NFT_SEARCH_PATHS: &[&str] = &["/usr/sbin/nft", "/sbin/nft", "/usr/bin/nft"];
|
||||
const IPTABLES_LEGACY_SEARCH_PATHS: &[&str] = &[
|
||||
"/usr/sbin/iptables-legacy",
|
||||
"/sbin/iptables-legacy",
|
||||
"/usr/bin/iptables-legacy",
|
||||
];
|
||||
const IP6TABLES_LEGACY_SEARCH_PATHS: &[&str] = &[
|
||||
"/usr/sbin/ip6tables-legacy",
|
||||
"/sbin/ip6tables-legacy",
|
||||
"/usr/bin/ip6tables-legacy",
|
||||
];
|
||||
|
||||
fn find_trusted_binary<'a>(name: &str, paths: &'a [&str]) -> Result<&'a str> {
|
||||
paths
|
||||
@@ -629,6 +871,18 @@ fn find_nft() -> Option<String> {
|
||||
.map(String::from)
|
||||
}
|
||||
|
||||
fn find_iptables_legacy() -> Option<String> {
|
||||
find_trusted_binary("iptables-legacy", IPTABLES_LEGACY_SEARCH_PATHS)
|
||||
.ok()
|
||||
.map(String::from)
|
||||
}
|
||||
|
||||
fn find_ip6tables_legacy() -> Option<String> {
|
||||
find_trusted_binary("ip6tables-legacy", IP6TABLES_LEGACY_SEARCH_PATHS)
|
||||
.ok()
|
||||
.map(String::from)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -668,6 +922,49 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn iptables_legacy_search_paths_are_absolute() {
|
||||
for path in IPTABLES_LEGACY_SEARCH_PATHS {
|
||||
assert!(
|
||||
path.starts_with('/'),
|
||||
"IPTABLES_LEGACY_SEARCH_PATHS entry must be absolute: {path}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ip6tables_legacy_search_paths_are_absolute() {
|
||||
for path in IP6TABLES_LEGACY_SEARCH_PATHS {
|
||||
assert!(
|
||||
path.starts_with('/'),
|
||||
"IP6TABLES_LEGACY_SEARCH_PATHS entry must be absolute: {path}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_loopback_ipv6_detector_ignores_empty_input() {
|
||||
assert!(!has_non_loopback_ipv6_interface(""));
|
||||
assert!(!has_non_loopback_ipv6_interface("\n\n"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_loopback_ipv6_detector_ignores_loopback() {
|
||||
let content = "00000000000000000000000000000001 01 80 10 80 lo\n";
|
||||
|
||||
assert!(!has_non_loopback_ipv6_interface(content));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_loopback_ipv6_detector_detects_pod_interface() {
|
||||
let content = "\
|
||||
00000000000000000000000000000001 01 80 10 80 lo
|
||||
fe800000000000000000000000000001 02 40 20 80 eth0
|
||||
";
|
||||
|
||||
assert!(has_non_loopback_ipv6_interface(content));
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[ignore = "requires root privileges"]
|
||||
fn test_create_and_drop_namespace() {
|
||||
|
||||
@@ -6,85 +6,417 @@
|
||||
//! This module provides pure functions to generate nftables rulesets that enforce
|
||||
//! the sandbox network policy: all traffic must go through the proxy, with bypass
|
||||
//! attempts logged and rejected.
|
||||
//!
|
||||
//! Rulesets are returned as a sequence of individual nft commands rather than a
|
||||
//! monolithic file. Running each command as a separate `nft` invocation avoids
|
||||
//! `nft -f` atomic batch semantics, where a single unsupported expression (e.g.
|
||||
//! `ct state` without `nf_conntrack`, `log` without `nf_log`) rolls back the
|
||||
//! entire transaction including table/chain creation.
|
||||
|
||||
/// Generate a complete nftables ruleset for sandbox network bypass enforcement.
|
||||
/// A single nft command with metadata about whether it is required.
|
||||
pub struct NftCommand {
|
||||
/// The nft command arguments (e.g. `["add", "table", "inet", "openshell_bypass"]`).
|
||||
pub args: Vec<String>,
|
||||
/// When false, failure of this command is non-fatal; the caller should
|
||||
/// log a warning and continue with the remaining commands.
|
||||
pub required: bool,
|
||||
}
|
||||
|
||||
/// Generate nft commands for sandbox network bypass enforcement.
|
||||
///
|
||||
/// Creates an `inet` family table (handles both IPv4 and IPv6) with rules that:
|
||||
/// 1. Accept traffic to the proxy (IPv4 only)
|
||||
/// 2. Accept loopback traffic
|
||||
/// 3. Accept established/related connections
|
||||
/// 3. Accept established/related connections (optional; requires `nf_conntrack`)
|
||||
/// 4. Reject TCP and UDP bypass attempts (both IPv4 and IPv6)
|
||||
///
|
||||
/// If `log_prefix` is provided, log rules are inserted before each reject rule
|
||||
/// so that bypass attempts are recorded in the kernel ring buffer before being
|
||||
/// rejected. The `log` expression requires kernel `nft_log` module support;
|
||||
/// pass `None` for `log_prefix` as a fallback when that module is unavailable.
|
||||
pub fn generate_bypass_ruleset(host_ip: &str, proxy_port: u16, log_prefix: Option<&str>) -> String {
|
||||
let log_tcp = log_prefix
|
||||
.map(|p| {
|
||||
format!(
|
||||
"\n tcp flags syn limit rate 5/second burst 10 packets log prefix \"{p}\" flags skuid"
|
||||
)
|
||||
})
|
||||
.unwrap_or_default();
|
||||
let log_udp = log_prefix
|
||||
.map(|p| {
|
||||
format!(
|
||||
"\n meta l4proto udp limit rate 5/second burst 10 packets log prefix \"{p}\" flags skuid"
|
||||
)
|
||||
})
|
||||
.unwrap_or_default();
|
||||
/// rejected. Log rules are always non-required since they need `nf_log` support.
|
||||
pub fn generate_bypass_commands(
|
||||
host_ip: &str,
|
||||
proxy_port: u16,
|
||||
log_prefix: Option<&str>,
|
||||
) -> Vec<NftCommand> {
|
||||
let table = "openshell_bypass";
|
||||
let mut cmds = vec![
|
||||
nft_cmd(true, &["add", "table", "inet", table]),
|
||||
nft_cmd(true, &["flush", "table", "inet", table]),
|
||||
nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add",
|
||||
"chain",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"{ type filter hook output priority 0; policy accept; }",
|
||||
],
|
||||
),
|
||||
nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add",
|
||||
"rule",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"ip",
|
||||
"daddr",
|
||||
host_ip,
|
||||
"tcp",
|
||||
"dport",
|
||||
&proxy_port.to_string(),
|
||||
"accept",
|
||||
],
|
||||
),
|
||||
nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add", "rule", "inet", table, "output", "oifname", "lo", "accept",
|
||||
],
|
||||
),
|
||||
nft_cmd(
|
||||
false,
|
||||
&[
|
||||
"add",
|
||||
"rule",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"ct",
|
||||
"state",
|
||||
"established,related",
|
||||
"accept",
|
||||
],
|
||||
),
|
||||
];
|
||||
|
||||
format!(
|
||||
r#"table inet openshell_bypass {{
|
||||
chain output {{
|
||||
type filter hook output priority 0; policy accept;
|
||||
if let Some(prefix) = log_prefix {
|
||||
cmds.push(nft_cmd(
|
||||
false,
|
||||
&[
|
||||
"add", "rule", "inet", table, "output", "tcp", "flags", "syn", "limit", "rate",
|
||||
"5/second", "burst", "10", "packets", "log", "prefix", prefix, "flags", "skuid",
|
||||
],
|
||||
));
|
||||
}
|
||||
|
||||
ip daddr {host_ip} tcp dport {proxy_port} accept
|
||||
oifname "lo" accept
|
||||
ct state established,related accept{log_tcp}
|
||||
meta nfproto ipv4 meta l4proto tcp reject with icmp type port-unreachable
|
||||
meta nfproto ipv6 meta l4proto tcp reject with icmpv6 type port-unreachable{log_udp}
|
||||
meta nfproto ipv4 meta l4proto udp reject with icmp type port-unreachable
|
||||
meta nfproto ipv6 meta l4proto udp reject with icmpv6 type port-unreachable
|
||||
}}
|
||||
}}
|
||||
"#
|
||||
)
|
||||
cmds.push(nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add",
|
||||
"rule",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"meta",
|
||||
"nfproto",
|
||||
"ipv4",
|
||||
"meta",
|
||||
"l4proto",
|
||||
"tcp",
|
||||
"reject",
|
||||
"with",
|
||||
"icmp",
|
||||
"type",
|
||||
"port-unreachable",
|
||||
],
|
||||
));
|
||||
cmds.push(nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add",
|
||||
"rule",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"meta",
|
||||
"nfproto",
|
||||
"ipv6",
|
||||
"meta",
|
||||
"l4proto",
|
||||
"tcp",
|
||||
"reject",
|
||||
"with",
|
||||
"icmpv6",
|
||||
"type",
|
||||
"port-unreachable",
|
||||
],
|
||||
));
|
||||
|
||||
if let Some(prefix) = log_prefix {
|
||||
cmds.push(nft_cmd(
|
||||
false,
|
||||
&[
|
||||
"add", "rule", "inet", table, "output", "meta", "l4proto", "udp", "limit", "rate",
|
||||
"5/second", "burst", "10", "packets", "log", "prefix", prefix, "flags", "skuid",
|
||||
],
|
||||
));
|
||||
}
|
||||
|
||||
cmds.push(nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add",
|
||||
"rule",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"meta",
|
||||
"nfproto",
|
||||
"ipv4",
|
||||
"meta",
|
||||
"l4proto",
|
||||
"udp",
|
||||
"reject",
|
||||
"with",
|
||||
"icmp",
|
||||
"type",
|
||||
"port-unreachable",
|
||||
],
|
||||
));
|
||||
cmds.push(nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add",
|
||||
"rule",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"meta",
|
||||
"nfproto",
|
||||
"ipv6",
|
||||
"meta",
|
||||
"l4proto",
|
||||
"udp",
|
||||
"reject",
|
||||
"with",
|
||||
"icmpv6",
|
||||
"type",
|
||||
"port-unreachable",
|
||||
],
|
||||
));
|
||||
|
||||
cmds
|
||||
}
|
||||
|
||||
/// Generate nft commands for Kubernetes sidecar enforcement.
|
||||
///
|
||||
/// The network sidecar and the process supervisor share a pod network
|
||||
/// namespace. The sidecar runs as `proxy_uid` and owns external egress;
|
||||
/// sandbox traffic must use loopback services hosted by that sidecar
|
||||
/// (gateway forward and HTTP CONNECT proxy). The generated fence rejects
|
||||
/// TCP/UDP bypass attempts from non-proxy UIDs; other L4 protocols are outside
|
||||
/// the sidecar policy fence.
|
||||
pub fn generate_sidecar_bypass_commands(
|
||||
proxy_uid: u32,
|
||||
log_prefix: Option<&str>,
|
||||
) -> Vec<NftCommand> {
|
||||
let table = "openshell_sidecar_bypass";
|
||||
let uid_str = proxy_uid.to_string();
|
||||
let mut cmds = vec![
|
||||
nft_cmd(true, &["add", "table", "inet", table]),
|
||||
nft_cmd(true, &["flush", "table", "inet", table]),
|
||||
nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add",
|
||||
"chain",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"{ type filter hook output priority 0; policy accept; }",
|
||||
],
|
||||
),
|
||||
nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add", "rule", "inet", table, "output", "oifname", "lo", "accept",
|
||||
],
|
||||
),
|
||||
nft_cmd(
|
||||
false,
|
||||
&[
|
||||
"add",
|
||||
"rule",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"ct",
|
||||
"state",
|
||||
"established,related",
|
||||
"accept",
|
||||
],
|
||||
),
|
||||
nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add", "rule", "inet", table, "output", "meta", "skuid", &uid_str, "accept",
|
||||
],
|
||||
),
|
||||
];
|
||||
|
||||
if let Some(prefix) = log_prefix {
|
||||
cmds.push(nft_cmd(
|
||||
false,
|
||||
&[
|
||||
"add", "rule", "inet", table, "output", "tcp", "flags", "syn", "limit", "rate",
|
||||
"5/second", "burst", "10", "packets", "log", "prefix", prefix, "flags", "skuid",
|
||||
],
|
||||
));
|
||||
}
|
||||
|
||||
cmds.push(nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add",
|
||||
"rule",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"meta",
|
||||
"nfproto",
|
||||
"ipv4",
|
||||
"meta",
|
||||
"l4proto",
|
||||
"tcp",
|
||||
"reject",
|
||||
"with",
|
||||
"icmp",
|
||||
"type",
|
||||
"port-unreachable",
|
||||
],
|
||||
));
|
||||
cmds.push(nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add",
|
||||
"rule",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"meta",
|
||||
"nfproto",
|
||||
"ipv6",
|
||||
"meta",
|
||||
"l4proto",
|
||||
"tcp",
|
||||
"reject",
|
||||
"with",
|
||||
"icmpv6",
|
||||
"type",
|
||||
"port-unreachable",
|
||||
],
|
||||
));
|
||||
|
||||
if let Some(prefix) = log_prefix {
|
||||
cmds.push(nft_cmd(
|
||||
false,
|
||||
&[
|
||||
"add", "rule", "inet", table, "output", "meta", "l4proto", "udp", "limit", "rate",
|
||||
"5/second", "burst", "10", "packets", "log", "prefix", prefix, "flags", "skuid",
|
||||
],
|
||||
));
|
||||
}
|
||||
|
||||
cmds.push(nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add",
|
||||
"rule",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"meta",
|
||||
"nfproto",
|
||||
"ipv4",
|
||||
"meta",
|
||||
"l4proto",
|
||||
"udp",
|
||||
"reject",
|
||||
"with",
|
||||
"icmp",
|
||||
"type",
|
||||
"port-unreachable",
|
||||
],
|
||||
));
|
||||
cmds.push(nft_cmd(
|
||||
true,
|
||||
&[
|
||||
"add",
|
||||
"rule",
|
||||
"inet",
|
||||
table,
|
||||
"output",
|
||||
"meta",
|
||||
"nfproto",
|
||||
"ipv6",
|
||||
"meta",
|
||||
"l4proto",
|
||||
"udp",
|
||||
"reject",
|
||||
"with",
|
||||
"icmpv6",
|
||||
"type",
|
||||
"port-unreachable",
|
||||
],
|
||||
));
|
||||
|
||||
cmds
|
||||
}
|
||||
|
||||
fn nft_cmd(required: bool, args: &[&str]) -> NftCommand {
|
||||
NftCommand {
|
||||
args: args.iter().map(|s| (*s).to_string()).collect(),
|
||||
required,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn generates_bypass_ruleset_with_proxy_rule() {
|
||||
let ruleset = generate_bypass_ruleset("10.0.2.2", 8080, None);
|
||||
assert!(ruleset.contains("table inet openshell_bypass"));
|
||||
assert!(ruleset.contains("chain output"));
|
||||
assert!(ruleset.contains("ip daddr 10.0.2.2 tcp dport 8080 accept"));
|
||||
fn cmd_str(cmd: &NftCommand) -> String {
|
||||
cmd.args.join(" ")
|
||||
}
|
||||
|
||||
fn all_strs(cmds: &[NftCommand]) -> String {
|
||||
cmds.iter().map(cmd_str).collect::<Vec<_>>().join("\n")
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ruleset_has_inet_family_table_and_output_chain() {
|
||||
let ruleset = generate_bypass_ruleset("192.168.1.1", 3128, None);
|
||||
assert!(ruleset.contains("table inet openshell_bypass"));
|
||||
assert!(ruleset.contains("type filter hook output priority 0; policy accept;"));
|
||||
fn generates_bypass_commands_with_proxy_rule() {
|
||||
let cmds = generate_bypass_commands("10.0.2.2", 8080, None);
|
||||
let text = all_strs(&cmds);
|
||||
assert!(text.contains("add table inet openshell_bypass"));
|
||||
assert!(text.contains("add chain inet openshell_bypass output"));
|
||||
assert!(text.contains("ip daddr 10.0.2.2 tcp dport 8080 accept"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bypass_commands_have_table_and_chain() {
|
||||
let cmds = generate_bypass_commands("192.168.1.1", 3128, None);
|
||||
let text = all_strs(&cmds);
|
||||
assert!(text.contains("add table inet openshell_bypass"));
|
||||
assert!(text.contains("type filter hook output priority 0; policy accept;"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn proxy_accept_rule_uses_provided_ip_and_port() {
|
||||
let ruleset = generate_bypass_ruleset("172.16.0.1", 9999, None);
|
||||
assert!(ruleset.contains("ip daddr 172.16.0.1 tcp dport 9999 accept"));
|
||||
let cmds = generate_bypass_commands("172.16.0.1", 9999, None);
|
||||
let text = all_strs(&cmds);
|
||||
assert!(text.contains("ip daddr 172.16.0.1 tcp dport 9999 accept"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rules_are_ordered_accept_then_reject() {
|
||||
let ruleset = generate_bypass_ruleset("10.0.2.2", 8080, None);
|
||||
let proxy_pos = ruleset.find("ip daddr").unwrap();
|
||||
let lo_pos = ruleset.find("oifname \"lo\"").unwrap();
|
||||
let ct_pos = ruleset.find("ct state established,related").unwrap();
|
||||
let reject_pos = ruleset.find("reject with icmp type").unwrap();
|
||||
let cmds = generate_bypass_commands("10.0.2.2", 8080, None);
|
||||
let text = all_strs(&cmds);
|
||||
let proxy_pos = text.find("ip daddr").unwrap();
|
||||
let lo_pos = text.find("oifname lo").unwrap();
|
||||
let ct_pos = text.find("ct state established,related").unwrap();
|
||||
let reject_pos = text.find("reject with icmp type").unwrap();
|
||||
|
||||
assert!(proxy_pos < lo_pos);
|
||||
assert!(lo_pos < ct_pos);
|
||||
@@ -93,11 +425,12 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn both_ipv4_and_ipv6_reject_types_are_present() {
|
||||
let ruleset = generate_bypass_ruleset("10.0.2.2", 8080, None);
|
||||
let icmp_count = ruleset
|
||||
let cmds = generate_bypass_commands("10.0.2.2", 8080, None);
|
||||
let text = all_strs(&cmds);
|
||||
let icmp_count = text
|
||||
.matches("reject with icmp type port-unreachable")
|
||||
.count();
|
||||
let icmpv6_count = ruleset
|
||||
let icmpv6_count = text
|
||||
.matches("reject with icmpv6 type port-unreachable")
|
||||
.count();
|
||||
assert_eq!(icmp_count, 2, "need IPv4 ICMP rejects for TCP + UDP");
|
||||
@@ -105,34 +438,35 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn no_log_ruleset_omits_log_rules() {
|
||||
let ruleset = generate_bypass_ruleset("10.0.2.2", 8080, None);
|
||||
fn no_log_commands_omit_log_rules() {
|
||||
let cmds = generate_bypass_commands("10.0.2.2", 8080, None);
|
||||
let text = all_strs(&cmds);
|
||||
assert!(
|
||||
!ruleset.contains("log prefix"),
|
||||
"no-log ruleset must not contain log rules"
|
||||
!text.contains("log prefix"),
|
||||
"no-log commands must not contain log rules"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn log_ruleset_contains_prefix_for_tcp_and_udp() {
|
||||
let ruleset = generate_bypass_ruleset("10.0.2.2", 8080, Some("openshell:bypass:test:"));
|
||||
let count = ruleset
|
||||
.matches("log prefix \"openshell:bypass:test:\"")
|
||||
.count();
|
||||
fn log_commands_contain_prefix_for_tcp_and_udp() {
|
||||
let cmds = generate_bypass_commands("10.0.2.2", 8080, Some("openshell:bypass:test:"));
|
||||
let text = all_strs(&cmds);
|
||||
let count = text.matches("log prefix openshell:bypass:test:").count();
|
||||
assert_eq!(count, 2, "need log rules for both TCP and UDP");
|
||||
assert!(ruleset.contains("tcp flags syn limit rate 5/second burst 10 packets"));
|
||||
assert!(ruleset.contains("meta l4proto udp limit rate 5/second burst 10 packets"));
|
||||
assert!(text.contains("tcp flags syn limit rate 5/second burst 10 packets"));
|
||||
assert!(text.contains("meta l4proto udp limit rate 5/second burst 10 packets"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn log_rules_appear_before_reject_rules() {
|
||||
let ruleset = generate_bypass_ruleset("10.0.2.2", 8080, Some("openshell:bypass:test:"));
|
||||
let tcp_log_pos = ruleset.find("tcp flags syn").unwrap();
|
||||
let tcp_reject_pos = ruleset
|
||||
let cmds = generate_bypass_commands("10.0.2.2", 8080, Some("openshell:bypass:test:"));
|
||||
let text = all_strs(&cmds);
|
||||
let tcp_log_pos = text.find("tcp flags syn").unwrap();
|
||||
let tcp_reject_pos = text
|
||||
.find("meta nfproto ipv4 meta l4proto tcp reject")
|
||||
.unwrap();
|
||||
let udp_log_pos = ruleset.find("meta l4proto udp limit rate").unwrap();
|
||||
let udp_reject_pos = ruleset
|
||||
let udp_log_pos = text.find("meta l4proto udp limit rate").unwrap();
|
||||
let udp_reject_pos = text
|
||||
.find("meta nfproto ipv4 meta l4proto udp reject")
|
||||
.unwrap();
|
||||
|
||||
@@ -145,4 +479,53 @@ mod tests {
|
||||
"UDP log rule must come before UDP reject rule"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ct_state_rule_is_not_required() {
|
||||
let cmds = generate_bypass_commands("10.0.2.2", 8080, None);
|
||||
let ct_cmd = cmds
|
||||
.iter()
|
||||
.find(|c| cmd_str(c).contains("ct state"))
|
||||
.unwrap();
|
||||
assert!(
|
||||
!ct_cmd.required,
|
||||
"ct state rule should be non-required (needs nf_conntrack)"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn log_rules_are_not_required() {
|
||||
let cmds = generate_bypass_commands("10.0.2.2", 8080, Some("openshell:bypass:test:"));
|
||||
for cmd in &cmds {
|
||||
if cmd_str(cmd).contains("log prefix") {
|
||||
assert!(
|
||||
!cmd.required,
|
||||
"log rules should be non-required (needs nf_log)"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sidecar_commands_allow_supervisor_uid_and_loopback() {
|
||||
let cmds = generate_sidecar_bypass_commands(1337, None);
|
||||
let text = all_strs(&cmds);
|
||||
assert!(text.contains("add table inet openshell_sidecar_bypass"));
|
||||
assert!(text.contains("oifname lo accept"));
|
||||
assert!(text.contains("meta skuid 1337 accept"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sidecar_commands_reject_tcp_and_udp_egress() {
|
||||
let cmds = generate_sidecar_bypass_commands(0, Some("openshell:sidecar:test:"));
|
||||
let text = all_strs(&cmds);
|
||||
assert!(text.contains("meta nfproto ipv4 meta l4proto tcp reject"));
|
||||
assert!(text.contains("meta nfproto ipv6 meta l4proto tcp reject"));
|
||||
assert!(text.contains("meta nfproto ipv4 meta l4proto udp reject"));
|
||||
assert!(text.contains("meta nfproto ipv6 meta l4proto udp reject"));
|
||||
assert_eq!(
|
||||
text.matches("log prefix openshell:sidecar:test:").count(),
|
||||
2
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,10 +28,55 @@ use std::sync::OnceLock;
|
||||
use tokio::process::{Child, Command};
|
||||
use tracing::{debug, info};
|
||||
|
||||
/// Process/filesystem enforcement performed by the process supervisor.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum ProcessEnforcementMode {
|
||||
/// Preserve the existing supervisor behavior: prepare filesystem policy,
|
||||
/// drop privileges, and apply Landlock/seccomp to workload processes.
|
||||
Full,
|
||||
/// Preserve process launch and SSH/session behavior, but skip controls
|
||||
/// that require root or extra Linux capabilities. Kubernetes sidecar mode
|
||||
/// uses this when network policy is enforced by the network sidecar.
|
||||
NetworkOnly,
|
||||
}
|
||||
|
||||
impl ProcessEnforcementMode {
|
||||
#[must_use]
|
||||
pub const fn uses_privileged_process_setup(self) -> bool {
|
||||
matches!(self, Self::Full)
|
||||
}
|
||||
|
||||
#[must_use]
|
||||
pub const fn enforces_child_sandbox(self) -> bool {
|
||||
matches!(self, Self::Full | Self::NetworkOnly)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
pub(crate) fn prepare_child_sandbox(
|
||||
policy: &SandboxPolicy,
|
||||
workdir: Option<&str>,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
) -> Result<Option<sandbox::linux::PreparedSandbox>> {
|
||||
if !enforcement_mode.enforces_child_sandbox() {
|
||||
return Ok(None);
|
||||
}
|
||||
|
||||
let prepared = if enforcement_mode.uses_privileged_process_setup() {
|
||||
sandbox::linux::prepare(policy, workdir)
|
||||
} else {
|
||||
sandbox::linux::prepare_current_user(policy, workdir)
|
||||
}?;
|
||||
Ok(Some(prepared))
|
||||
}
|
||||
|
||||
const SUPERVISOR_ONLY_ENV_VARS: &[&str] = &[
|
||||
openshell_core::sandbox_env::SANDBOX_TOKEN,
|
||||
openshell_core::sandbox_env::SANDBOX_TOKEN_FILE,
|
||||
openshell_core::sandbox_env::K8S_SA_TOKEN_FILE,
|
||||
openshell_core::sandbox_env::TLS_CA,
|
||||
openshell_core::sandbox_env::TLS_CERT,
|
||||
openshell_core::sandbox_env::TLS_KEY,
|
||||
openshell_core::sandbox_env::PROVIDER_SPIFFE_WORKLOAD_API_SOCKET,
|
||||
];
|
||||
|
||||
@@ -443,6 +488,7 @@ impl ProcessHandle {
|
||||
workdir: Option<&str>,
|
||||
interactive: bool,
|
||||
policy: &SandboxPolicy,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
netns: Option<&NetworkNamespace>,
|
||||
ca_paths: Option<&(PathBuf, PathBuf)>,
|
||||
provider_env: &HashMap<String, String>,
|
||||
@@ -453,6 +499,7 @@ impl ProcessHandle {
|
||||
workdir,
|
||||
interactive,
|
||||
policy,
|
||||
enforcement_mode,
|
||||
netns.and_then(NetworkNamespace::ns_fd),
|
||||
ca_paths,
|
||||
provider_env,
|
||||
@@ -465,12 +512,14 @@ impl ProcessHandle {
|
||||
///
|
||||
/// Returns an error if the process fails to start.
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn spawn(
|
||||
program: &str,
|
||||
args: &[String],
|
||||
workdir: Option<&str>,
|
||||
interactive: bool,
|
||||
policy: &SandboxPolicy,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
ca_paths: Option<&(PathBuf, PathBuf)>,
|
||||
provider_env: &HashMap<String, String>,
|
||||
) -> Result<Self> {
|
||||
@@ -480,6 +529,7 @@ impl ProcessHandle {
|
||||
workdir,
|
||||
interactive,
|
||||
policy,
|
||||
enforcement_mode,
|
||||
ca_paths,
|
||||
provider_env,
|
||||
)
|
||||
@@ -493,6 +543,7 @@ impl ProcessHandle {
|
||||
workdir: Option<&str>,
|
||||
interactive: bool,
|
||||
policy: &SandboxPolicy,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
netns_fd: Option<RawFd>,
|
||||
ca_paths: Option<&(PathBuf, PathBuf)>,
|
||||
provider_env: &HashMap<String, String>,
|
||||
@@ -552,18 +603,26 @@ impl ProcessHandle {
|
||||
// process where the tracing subscriber is functional. The child's
|
||||
// pre_exec context cannot reliably emit structured logs.
|
||||
#[cfg(target_os = "linux")]
|
||||
sandbox::linux::log_sandbox_readiness(policy, workdir);
|
||||
if enforcement_mode.enforces_child_sandbox() {
|
||||
sandbox::linux::log_sandbox_readiness(policy, workdir);
|
||||
}
|
||||
|
||||
// Phase 1 (as root): Prepare Landlock ruleset by opening PathFds.
|
||||
// This MUST happen before drop_privileges() so that root-only paths
|
||||
// (e.g. mode 700 directories) can be opened. See issue #803.
|
||||
// Phase 1: Prepare Landlock ruleset by opening PathFds.
|
||||
// In full mode this runs before drop_privileges() so root-only paths
|
||||
// can be opened. In sidecar network-only mode the container already
|
||||
// runs as the sandbox UID, so inaccessible paths are unavailable to
|
||||
// the workload and best-effort compatibility skips them.
|
||||
#[cfg(target_os = "linux")]
|
||||
let prepared_sandbox = sandbox::linux::prepare(policy, workdir)
|
||||
let prepared_sandbox = prepare_child_sandbox(policy, workdir, enforcement_mode)
|
||||
.map_err(|err| miette::miette!("Failed to prepare sandbox: {err}"))?;
|
||||
#[cfg(target_os = "linux")]
|
||||
let supervisor_identity_mount = supervisor_identity_mount_from_env().map_err(|err| {
|
||||
miette::miette!("Failed to prepare supervisor identity isolation: {err}")
|
||||
})?;
|
||||
let supervisor_identity_mount = if enforcement_mode.uses_privileged_process_setup() {
|
||||
supervisor_identity_mount_from_env().map_err(|err| {
|
||||
miette::miette!("Failed to prepare supervisor identity isolation: {err}")
|
||||
})?
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
// Set up process group for signal handling (non-interactive mode only).
|
||||
// In interactive mode, we inherit the parent's process group to maintain
|
||||
@@ -575,7 +634,7 @@ impl ProcessHandle {
|
||||
// Wrap in Option so we can .take() it out of the FnMut closure.
|
||||
// pre_exec is only called once (after fork, before exec).
|
||||
#[cfg(target_os = "linux")]
|
||||
let mut prepared_sandbox = Some(prepared_sandbox);
|
||||
let mut prepared_sandbox = prepared_sandbox;
|
||||
#[allow(unsafe_code)]
|
||||
unsafe {
|
||||
cmd.pre_exec(move || {
|
||||
@@ -600,8 +659,10 @@ impl ProcessHandle {
|
||||
// Drop privileges. initgroups/setgid/setuid need access to
|
||||
// /etc/group and /etc/passwd which would be blocked if
|
||||
// Landlock were already enforced.
|
||||
drop_privileges(&policy)
|
||||
.map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
if enforcement_mode.uses_privileged_process_setup() {
|
||||
drop_privileges(&policy)
|
||||
.map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
}
|
||||
|
||||
harden_child_process().map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
|
||||
@@ -629,12 +690,14 @@ impl ProcessHandle {
|
||||
}
|
||||
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn spawn_impl(
|
||||
program: &str,
|
||||
args: &[String],
|
||||
workdir: Option<&str>,
|
||||
interactive: bool,
|
||||
policy: &SandboxPolicy,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
ca_paths: Option<&(PathBuf, PathBuf)>,
|
||||
provider_env: &HashMap<String, String>,
|
||||
) -> Result<Self> {
|
||||
@@ -697,13 +760,17 @@ impl ProcessHandle {
|
||||
// Drop privileges before applying sandbox restrictions.
|
||||
// initgroups/setgid/setuid need access to /etc/group and /etc/passwd
|
||||
// which may be blocked by Landlock.
|
||||
drop_privileges(&policy)
|
||||
.map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
if enforcement_mode.uses_privileged_process_setup() {
|
||||
drop_privileges(&policy)
|
||||
.map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
}
|
||||
|
||||
harden_child_process().map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
|
||||
sandbox::apply(&policy, workdir.as_deref())
|
||||
.map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
if enforcement_mode.enforces_child_sandbox() {
|
||||
sandbox::apply(&policy, workdir.as_deref())
|
||||
.map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
}
|
||||
|
||||
Ok(())
|
||||
});
|
||||
@@ -1456,6 +1523,18 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn full_enforcement_uses_privileged_setup_and_child_sandbox() {
|
||||
assert!(ProcessEnforcementMode::Full.uses_privileged_process_setup());
|
||||
assert!(ProcessEnforcementMode::Full.enforces_child_sandbox());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn network_only_enforcement_keeps_child_sandbox_without_privileged_setup() {
|
||||
assert!(!ProcessEnforcementMode::NetworkOnly.uses_privileged_process_setup());
|
||||
assert!(ProcessEnforcementMode::NetworkOnly.enforces_child_sandbox());
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn capability_bounding_set_clear_available() -> bool {
|
||||
capctl::caps::CapState::get_current()
|
||||
|
||||
@@ -33,7 +33,7 @@ use openshell_core::denial::DenialEvent;
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
use crate::managed_children;
|
||||
use crate::process::ProcessHandle;
|
||||
use crate::process::{ProcessEnforcementMode, ProcessHandle};
|
||||
|
||||
fn ocsf_ctx() -> &'static openshell_ocsf::SandboxContext {
|
||||
openshell_ocsf::ctx::ctx()
|
||||
@@ -56,8 +56,11 @@ pub async fn run_process(
|
||||
sandbox_id: Option<&str>,
|
||||
openshell_endpoint: Option<&str>,
|
||||
ssh_socket_path: Option<String>,
|
||||
shared_ssh_socket: bool,
|
||||
policy: &SandboxPolicy,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
entrypoint_pid: Arc<AtomicU32>,
|
||||
entrypoint_started_tx: Option<tokio::sync::oneshot::Sender<u32>>,
|
||||
provider_credentials: ProviderCredentialState,
|
||||
provider_env: std::collections::HashMap<String, String>,
|
||||
ca_file_paths: Option<(std::path::PathBuf, std::path::PathBuf)>,
|
||||
@@ -71,21 +74,26 @@ pub async fn run_process(
|
||||
// /etc/group so the "sandbox" entry matches. Must run before
|
||||
// validate_sandbox_user so passwd lookups see the correct identity.
|
||||
#[cfg(unix)]
|
||||
crate::process::update_sandbox_passwd_entries()?;
|
||||
if enforcement_mode.uses_privileged_process_setup() {
|
||||
crate::process::update_sandbox_passwd_entries()?;
|
||||
}
|
||||
|
||||
// Validate that the sandbox user exists in the image. All sandbox images
|
||||
// must include a "sandbox" user for privilege dropping; failing fast here
|
||||
// beats silently running children as root.
|
||||
#[cfg(unix)]
|
||||
crate::process::validate_sandbox_user(policy)?;
|
||||
#[cfg(unix)]
|
||||
crate::process::validate_sandbox_group(policy)?;
|
||||
if enforcement_mode.uses_privileged_process_setup() {
|
||||
crate::process::validate_sandbox_user(policy)?;
|
||||
crate::process::validate_sandbox_group(policy)?;
|
||||
}
|
||||
|
||||
// Create read_write directories and chown newly-created ones to the
|
||||
// sandbox user/group. Runs as the supervisor (root) before the child
|
||||
// is forked so the workload sees writable paths it owns.
|
||||
#[cfg(unix)]
|
||||
crate::process::prepare_filesystem(policy)?;
|
||||
if enforcement_mode.uses_privileged_process_setup() {
|
||||
crate::process::prepare_filesystem(policy)?;
|
||||
}
|
||||
|
||||
// Eagerly fetch initial settings and install the agent skill if the
|
||||
// proposals flag is on at startup, rather than waiting for the policy
|
||||
@@ -206,31 +214,10 @@ pub async fn run_process(
|
||||
// their env so cooperative tools (curl, npm, Node) route through the
|
||||
// CONNECT proxy. Linux uses the netns host_ip; on other targets fall back
|
||||
// to the policy-declared http_addr directly.
|
||||
let ssh_proxy_url = if matches!(policy.network.mode, NetworkMode::Proxy) {
|
||||
#[cfg(target_os = "linux")]
|
||||
{
|
||||
netns.map(|ns| {
|
||||
let port = policy
|
||||
.network
|
||||
.proxy
|
||||
.as_ref()
|
||||
.and_then(|p| p.http_addr)
|
||||
.map_or(3128, |addr| addr.port());
|
||||
format!("http://{}:{port}", ns.host_ip())
|
||||
})
|
||||
}
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
{
|
||||
policy
|
||||
.network
|
||||
.proxy
|
||||
.as_ref()
|
||||
.and_then(|p| p.http_addr)
|
||||
.map(|addr| format!("http://{addr}"))
|
||||
}
|
||||
} else {
|
||||
None
|
||||
};
|
||||
#[cfg(target_os = "linux")]
|
||||
let ssh_proxy_url = ssh_proxy_url_for_policy(policy, netns.map(NetworkNamespace::host_ip));
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
let ssh_proxy_url = ssh_proxy_url_for_policy(policy, None);
|
||||
|
||||
let ssh_socket_path: Option<std::path::PathBuf> = ssh_socket_path.map(std::path::PathBuf::from);
|
||||
if let Some(listen_path) = ssh_socket_path.clone() {
|
||||
@@ -259,6 +246,8 @@ pub async fn run_process(
|
||||
ca_paths,
|
||||
provider_credentials_clone,
|
||||
user_env_clone,
|
||||
enforcement_mode,
|
||||
shared_ssh_socket,
|
||||
)
|
||||
.await
|
||||
{
|
||||
@@ -314,6 +303,7 @@ pub async fn run_process(
|
||||
id.to_string(),
|
||||
socket.clone(),
|
||||
ssh_netns_fd,
|
||||
None,
|
||||
);
|
||||
info!("supervisor session task spawned");
|
||||
}
|
||||
@@ -325,6 +315,7 @@ pub async fn run_process(
|
||||
workdir,
|
||||
interactive,
|
||||
policy,
|
||||
enforcement_mode,
|
||||
netns,
|
||||
ca_file_paths.as_ref(),
|
||||
&provider_env,
|
||||
@@ -337,12 +328,16 @@ pub async fn run_process(
|
||||
workdir,
|
||||
interactive,
|
||||
policy,
|
||||
enforcement_mode,
|
||||
ca_file_paths.as_ref(),
|
||||
&provider_env,
|
||||
)?;
|
||||
|
||||
// Store the entrypoint PID so the proxy can resolve TCP peer identity
|
||||
entrypoint_pid.store(handle.pid(), Ordering::Release);
|
||||
if let Some(tx) = entrypoint_started_tx {
|
||||
let _ = tx.send(handle.pid());
|
||||
}
|
||||
ocsf_emit!(
|
||||
ProcessActivityBuilder::new(ocsf_ctx())
|
||||
.activity(ActivityId::Open)
|
||||
@@ -395,6 +390,23 @@ pub async fn run_process(
|
||||
Ok(status.code())
|
||||
}
|
||||
|
||||
fn ssh_proxy_url_for_policy(
|
||||
policy: &SandboxPolicy,
|
||||
netns_proxy_host: Option<std::net::IpAddr>,
|
||||
) -> Option<String> {
|
||||
if !matches!(policy.network.mode, NetworkMode::Proxy) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let proxy = policy.network.proxy.as_ref()?;
|
||||
if let Some(host) = netns_proxy_host {
|
||||
let port = proxy.http_addr.map_or(3128, |addr| addr.port());
|
||||
return Some(format!("http://{host}:{port}"));
|
||||
}
|
||||
|
||||
proxy.http_addr.map(|addr| format!("http://{addr}"))
|
||||
}
|
||||
|
||||
/// Eagerly fetch initial settings and install the agent-driven policy
|
||||
/// proposal skill if the flag is on at startup.
|
||||
///
|
||||
@@ -451,3 +463,53 @@ async fn install_initial_agent_skill(sandbox_id: Option<&str>, openshell_endpoin
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use openshell_core::policy::{
|
||||
FilesystemPolicy, LandlockPolicy, NetworkMode, NetworkPolicy, ProcessPolicy, ProxyPolicy,
|
||||
};
|
||||
|
||||
fn policy(mode: NetworkMode, http_addr: Option<std::net::SocketAddr>) -> SandboxPolicy {
|
||||
SandboxPolicy {
|
||||
version: 1,
|
||||
filesystem: FilesystemPolicy::default(),
|
||||
network: NetworkPolicy {
|
||||
mode,
|
||||
proxy: http_addr.map(|http_addr| ProxyPolicy {
|
||||
http_addr: Some(http_addr),
|
||||
}),
|
||||
},
|
||||
landlock: LandlockPolicy::default(),
|
||||
process: ProcessPolicy::default(),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ssh_proxy_url_uses_policy_addr_without_netns() {
|
||||
let policy = policy(NetworkMode::Proxy, Some(([127, 0, 0, 1], 3128).into()));
|
||||
|
||||
assert_eq!(
|
||||
ssh_proxy_url_for_policy(&policy, None).as_deref(),
|
||||
Some("http://127.0.0.1:3128")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ssh_proxy_url_prefers_netns_host_with_policy_port() {
|
||||
let policy = policy(NetworkMode::Proxy, Some(([127, 0, 0, 1], 8080).into()));
|
||||
|
||||
assert_eq!(
|
||||
ssh_proxy_url_for_policy(&policy, Some([10, 200, 0, 1].into())).as_deref(),
|
||||
Some("http://10.200.0.1:8080")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ssh_proxy_url_skips_non_proxy_mode() {
|
||||
let policy = policy(NetworkMode::Allow, Some(([127, 0, 0, 1], 3128).into()));
|
||||
|
||||
assert_eq!(ssh_proxy_url_for_policy(&policy, None), None);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -95,6 +95,12 @@ pub struct PreparedRuleset {
|
||||
compatibility: LandlockCompatibility,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
enum PathOpenMode {
|
||||
Privileged,
|
||||
CurrentUser,
|
||||
}
|
||||
|
||||
/// Phase 1: Open `PathFds` and build the Landlock ruleset **as root**.
|
||||
///
|
||||
/// This must run before `drop_privileges()` so that `PathFd::new()` can open
|
||||
@@ -103,6 +109,27 @@ pub struct PreparedRuleset {
|
||||
/// Returns `None` if there are no filesystem paths to restrict (no-op).
|
||||
/// Returns `Some(PreparedRuleset)` on success, or an error.
|
||||
pub fn prepare(policy: &SandboxPolicy, workdir: Option<&str>) -> Result<Option<PreparedRuleset>> {
|
||||
prepare_with_path_open_mode(policy, workdir, PathOpenMode::Privileged)
|
||||
}
|
||||
|
||||
/// Phase 1 for already-unprivileged workloads.
|
||||
///
|
||||
/// Kubernetes sidecar mode starts the process supervisor as the sandbox UID, so
|
||||
/// Landlock path FDs are opened as the same UID that will run the workload.
|
||||
/// Paths this UID cannot open are already unavailable to the workload; omit
|
||||
/// them from the allowlist and let the resulting ruleset deny everything else.
|
||||
pub fn prepare_current_user(
|
||||
policy: &SandboxPolicy,
|
||||
workdir: Option<&str>,
|
||||
) -> Result<Option<PreparedRuleset>> {
|
||||
prepare_with_path_open_mode(policy, workdir, PathOpenMode::CurrentUser)
|
||||
}
|
||||
|
||||
fn prepare_with_path_open_mode(
|
||||
policy: &SandboxPolicy,
|
||||
workdir: Option<&str>,
|
||||
path_open_mode: PathOpenMode,
|
||||
) -> Result<Option<PreparedRuleset>> {
|
||||
let read_only = policy.filesystem.read_only.clone();
|
||||
let mut read_write = policy.filesystem.read_write.clone();
|
||||
|
||||
@@ -188,7 +215,7 @@ pub fn prepare(policy: &SandboxPolicy, workdir: Option<&str>) -> Result<Option<P
|
||||
let mut rules_applied: usize = 0;
|
||||
|
||||
for path in &read_only {
|
||||
if let Some(path_fd) = try_open_path(path, compatibility)? {
|
||||
if let Some(path_fd) = try_open_path(path, compatibility, path_open_mode)? {
|
||||
debug!(path = %path.display(), "Landlock allow read-only");
|
||||
ruleset = ruleset
|
||||
.add_rule(PathBeneath::new(path_fd, access_read))
|
||||
@@ -198,7 +225,7 @@ pub fn prepare(policy: &SandboxPolicy, workdir: Option<&str>) -> Result<Option<P
|
||||
}
|
||||
|
||||
for path in &read_write {
|
||||
if let Some(path_fd) = try_open_path(path, compatibility)? {
|
||||
if let Some(path_fd) = try_open_path(path, compatibility, path_open_mode)? {
|
||||
debug!(path = %path.display(), "Landlock allow read-write");
|
||||
ruleset = ruleset
|
||||
.add_rule(PathBeneath::new(path_fd, access_all))
|
||||
@@ -325,7 +352,11 @@ pub fn apply(policy: &SandboxPolicy, workdir: Option<&str>) -> Result<()> {
|
||||
///
|
||||
/// In `HardRequirement` mode, any failure is fatal — the caller propagates the
|
||||
/// error, which ultimately aborts sandbox startup.
|
||||
fn try_open_path(path: &Path, compatibility: &LandlockCompatibility) -> Result<Option<PathFd>> {
|
||||
fn try_open_path(
|
||||
path: &Path,
|
||||
compatibility: &LandlockCompatibility,
|
||||
path_open_mode: PathOpenMode,
|
||||
) -> Result<Option<PathFd>> {
|
||||
match PathFd::new(path) {
|
||||
Ok(fd) => Ok(Some(fd)),
|
||||
Err(err) => {
|
||||
@@ -335,6 +366,28 @@ fn try_open_path(path: &Path, compatibility: &LandlockCompatibility) -> Result<O
|
||||
PathFdError::OpenCall { source, .. }
|
||||
if source.kind() == std::io::ErrorKind::NotFound
|
||||
);
|
||||
if matches!(path_open_mode, PathOpenMode::CurrentUser) {
|
||||
if is_not_found {
|
||||
debug!(
|
||||
path = %path.display(),
|
||||
reason,
|
||||
"Skipping non-existent Landlock path for current user"
|
||||
);
|
||||
} else {
|
||||
openshell_ocsf::ocsf_emit!(
|
||||
openshell_ocsf::ConfigStateChangeBuilder::new(openshell_ocsf::ctx::ctx())
|
||||
.severity(openshell_ocsf::SeverityId::Informational)
|
||||
.status(openshell_ocsf::StatusId::Success)
|
||||
.state(openshell_ocsf::StateId::Other, "already-denied")
|
||||
.message(format!(
|
||||
"Skipping inaccessible Landlock path for current user [path:{} error:{err}]",
|
||||
path.display()
|
||||
))
|
||||
.build()
|
||||
);
|
||||
}
|
||||
return Ok(None);
|
||||
}
|
||||
match compatibility {
|
||||
LandlockCompatibility::BestEffort => {
|
||||
// NotFound is expected for stale baseline paths (e.g.
|
||||
@@ -421,6 +474,7 @@ mod tests {
|
||||
let result = try_open_path(
|
||||
&PathBuf::from("/nonexistent/openshell/test/path"),
|
||||
&LandlockCompatibility::BestEffort,
|
||||
PathOpenMode::Privileged,
|
||||
);
|
||||
assert!(result.is_ok());
|
||||
assert!(result.unwrap().is_none());
|
||||
@@ -431,6 +485,7 @@ mod tests {
|
||||
let result = try_open_path(
|
||||
&PathBuf::from("/nonexistent/openshell/test/path"),
|
||||
&LandlockCompatibility::HardRequirement,
|
||||
PathOpenMode::Privileged,
|
||||
);
|
||||
assert!(result.is_err());
|
||||
let err_msg = result.unwrap_err().to_string();
|
||||
@@ -447,11 +502,26 @@ mod tests {
|
||||
#[test]
|
||||
fn try_open_path_succeeds_for_existing_path() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let result = try_open_path(dir.path(), &LandlockCompatibility::BestEffort);
|
||||
let result = try_open_path(
|
||||
dir.path(),
|
||||
&LandlockCompatibility::BestEffort,
|
||||
PathOpenMode::Privileged,
|
||||
);
|
||||
assert!(result.is_ok());
|
||||
assert!(result.unwrap().is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn try_open_path_current_user_skips_missing_path_in_hard_requirement() {
|
||||
let result = try_open_path(
|
||||
&PathBuf::from("/nonexistent/openshell/test/path"),
|
||||
&LandlockCompatibility::HardRequirement,
|
||||
PathOpenMode::CurrentUser,
|
||||
);
|
||||
assert!(result.is_ok());
|
||||
assert!(result.unwrap().is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn classify_not_found() {
|
||||
let err = std::io::Error::from_raw_os_error(libc::ENOENT);
|
||||
|
||||
@@ -12,7 +12,7 @@ use std::path::PathBuf;
|
||||
use std::sync::Once;
|
||||
|
||||
/// Opaque handle to a prepared-but-not-yet-enforced sandbox.
|
||||
/// Holds the Landlock ruleset with `PathFds` opened as root.
|
||||
/// Holds the Landlock ruleset with `PathFds` opened before child exec.
|
||||
pub struct PreparedSandbox {
|
||||
landlock: Option<landlock::PreparedRuleset>,
|
||||
policy: SandboxPolicy,
|
||||
@@ -30,6 +30,21 @@ pub fn prepare(policy: &SandboxPolicy, workdir: Option<&str>) -> Result<Prepared
|
||||
})
|
||||
}
|
||||
|
||||
/// Phase 1 for already-unprivileged workloads.
|
||||
///
|
||||
/// Opens Landlock `PathFds` as the current UID. This is used by Kubernetes
|
||||
/// sidecar mode, where the agent container already runs as the sandbox user.
|
||||
pub fn prepare_current_user(
|
||||
policy: &SandboxPolicy,
|
||||
workdir: Option<&str>,
|
||||
) -> Result<PreparedSandbox> {
|
||||
let landlock = landlock::prepare_current_user(policy, workdir)?;
|
||||
Ok(PreparedSandbox {
|
||||
landlock,
|
||||
policy: policy.clone(),
|
||||
})
|
||||
}
|
||||
|
||||
/// Phase 2: Enforce prepared sandbox restrictions (after `drop_privileges`).
|
||||
///
|
||||
/// Calls `restrict_self()` for Landlock and applies seccomp filters.
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
use crate::child_env;
|
||||
#[cfg(target_os = "linux")]
|
||||
use crate::managed_children;
|
||||
use crate::process::{drop_privileges, is_supervisor_only_env_var};
|
||||
use crate::process::{ProcessEnforcementMode, drop_privileges, is_supervisor_only_env_var};
|
||||
use crate::sandbox;
|
||||
use miette::{IntoDiagnostic, Result};
|
||||
use nix::pty::{Winsize, openpty};
|
||||
@@ -42,6 +42,8 @@ type SshServerInit = (
|
||||
fn ssh_server_init(
|
||||
listen_path: &Path,
|
||||
ca_file_paths: &Option<(PathBuf, PathBuf)>,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
shared_socket: bool,
|
||||
) -> Result<SshServerInit> {
|
||||
let mut rng = OsRng;
|
||||
let host_key = PrivateKey::random(&mut rng, Algorithm::Ed25519).into_diagnostic()?;
|
||||
@@ -55,13 +57,17 @@ fn ssh_server_init(
|
||||
let config = Arc::new(config);
|
||||
let ca_paths = ca_file_paths.as_ref().map(|p| Arc::new(p.clone()));
|
||||
|
||||
// Ensure the parent directory exists and is root-owned with 0700
|
||||
// permissions. The sandbox entrypoint runs as an unprivileged user; it
|
||||
// must not be able to enter this directory and connect to the socket.
|
||||
if let Some(parent) = listen_path.parent() {
|
||||
// In full enforcement mode the supervisor normally starts as root and can
|
||||
// isolate the SSH socket in a root-only directory before spawning
|
||||
// unprivileged children. Sidecar topology is different: the gateway relay
|
||||
// runs in the network sidecar as a different UID, so the shared sidecar
|
||||
// state directory must stay group-accessible. Sidecar mode uses a Linux
|
||||
// abstract socket instead, so the workload cannot unlink the relay target.
|
||||
let abstract_socket = crate::unix_socket::is_abstract(listen_path);
|
||||
if !abstract_socket && let Some(parent) = listen_path.parent() {
|
||||
std::fs::create_dir_all(parent).into_diagnostic()?;
|
||||
#[cfg(unix)]
|
||||
{
|
||||
if enforcement_mode.uses_privileged_process_setup() && !shared_socket {
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
let perms = std::fs::Permissions::from_mode(0o700);
|
||||
std::fs::set_permissions(parent, perms).into_diagnostic()?;
|
||||
@@ -69,19 +75,19 @@ fn ssh_server_init(
|
||||
}
|
||||
|
||||
// Remove any stale socket from a previous run before binding.
|
||||
if listen_path.exists() {
|
||||
if !abstract_socket && listen_path.exists() {
|
||||
std::fs::remove_file(listen_path).into_diagnostic()?;
|
||||
}
|
||||
let listener = UnixListener::bind(listen_path).into_diagnostic()?;
|
||||
let runtime_path = crate::unix_socket::runtime_path(listen_path);
|
||||
let listener = UnixListener::bind(runtime_path.as_ref()).into_diagnostic()?;
|
||||
|
||||
// Tighten permissions so only the supervisor (root) can connect. The
|
||||
// sandbox entrypoint runs as an unprivileged user and must not be able to
|
||||
// dial the SSH daemon directly — all access goes through the relay from
|
||||
// the gateway.
|
||||
// Tighten filesystem-socket permissions. Abstract sockets have no inode;
|
||||
// sidecar relay connections authenticate the listener with SO_PEERCRED.
|
||||
#[cfg(unix)]
|
||||
{
|
||||
if !abstract_socket {
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
let perms = std::fs::Permissions::from_mode(0o600);
|
||||
let mode = if shared_socket { 0o660 } else { 0o600 };
|
||||
let perms = std::fs::Permissions::from_mode(mode);
|
||||
std::fs::set_permissions(listen_path, perms).into_diagnostic()?;
|
||||
}
|
||||
|
||||
@@ -108,8 +114,15 @@ pub async fn run_ssh_server(
|
||||
ca_file_paths: Option<(PathBuf, PathBuf)>,
|
||||
provider_credentials: ProviderCredentialState,
|
||||
user_environment: HashMap<String, String>,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
shared_socket: bool,
|
||||
) -> Result<()> {
|
||||
let (listener, config, ca_paths) = match ssh_server_init(&listen_path, &ca_file_paths) {
|
||||
let (listener, config, ca_paths) = match ssh_server_init(
|
||||
&listen_path,
|
||||
&ca_file_paths,
|
||||
enforcement_mode,
|
||||
shared_socket,
|
||||
) {
|
||||
Ok(v) => {
|
||||
// Signal that the SSH server has bound the socket and is ready to
|
||||
// accept connections. The parent task awaits this before spawning
|
||||
@@ -145,6 +158,7 @@ pub async fn run_ssh_server(
|
||||
ca_paths,
|
||||
provider_credentials,
|
||||
user_environment,
|
||||
enforcement_mode,
|
||||
)
|
||||
.await
|
||||
{
|
||||
@@ -172,6 +186,7 @@ async fn handle_connection(
|
||||
ca_file_paths: Option<Arc<(PathBuf, PathBuf)>>,
|
||||
provider_credentials: ProviderCredentialState,
|
||||
user_environment: HashMap<String, String>,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
) -> Result<()> {
|
||||
// Access is gated by the Unix-socket filesystem permissions (root-only),
|
||||
// not by an application-level preface. The supervisor bridges the
|
||||
@@ -195,6 +210,7 @@ async fn handle_connection(
|
||||
ca_file_paths,
|
||||
provider_credentials,
|
||||
user_environment,
|
||||
enforcement_mode,
|
||||
);
|
||||
russh::server::run_stream(config, stream, handler)
|
||||
.await
|
||||
@@ -223,6 +239,7 @@ struct SshHandler {
|
||||
ca_file_paths: Option<Arc<(PathBuf, PathBuf)>>,
|
||||
provider_credentials: ProviderCredentialState,
|
||||
user_environment: HashMap<String, String>,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
channels: HashMap<ChannelId, ChannelState>,
|
||||
}
|
||||
|
||||
@@ -236,6 +253,7 @@ impl SshHandler {
|
||||
ca_file_paths: Option<Arc<(PathBuf, PathBuf)>>,
|
||||
provider_credentials: ProviderCredentialState,
|
||||
user_environment: HashMap<String, String>,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
) -> Self {
|
||||
Self {
|
||||
policy,
|
||||
@@ -245,6 +263,7 @@ impl SshHandler {
|
||||
ca_file_paths,
|
||||
provider_credentials,
|
||||
user_environment,
|
||||
enforcement_mode,
|
||||
channels: HashMap::new(),
|
||||
}
|
||||
}
|
||||
@@ -468,6 +487,7 @@ impl russh::server::Handler for SshHandler {
|
||||
self.ca_file_paths.clone(),
|
||||
&self.provider_credentials.child_env_with_gcp_resolved(),
|
||||
&self.user_environment,
|
||||
self.enforcement_mode,
|
||||
)?;
|
||||
let state = self.channels.get_mut(&channel).ok_or_else(|| {
|
||||
anyhow::anyhow!("subsystem_request on unknown channel {channel:?}")
|
||||
@@ -564,6 +584,7 @@ impl SshHandler {
|
||||
self.ca_file_paths.clone(),
|
||||
&provider_env,
|
||||
&self.user_environment,
|
||||
self.enforcement_mode,
|
||||
)?;
|
||||
state.pty_master = Some(pty_master);
|
||||
state.input_sender = Some(input_sender);
|
||||
@@ -582,6 +603,7 @@ impl SshHandler {
|
||||
self.ca_file_paths.clone(),
|
||||
&provider_env,
|
||||
&self.user_environment,
|
||||
self.enforcement_mode,
|
||||
)?;
|
||||
state.input_sender = Some(input_sender);
|
||||
}
|
||||
@@ -748,6 +770,7 @@ fn spawn_pty_shell(
|
||||
ca_file_paths: Option<Arc<(PathBuf, PathBuf)>>,
|
||||
provider_env: &HashMap<String, String>,
|
||||
user_environment: &HashMap<String, String>,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
) -> anyhow::Result<(std::fs::File, mpsc::Sender<Vec<u8>>)> {
|
||||
let winsize = Winsize {
|
||||
ws_row: to_u16(pty.row_height.max(1)),
|
||||
@@ -806,12 +829,15 @@ fn spawn_pty_shell(
|
||||
|
||||
// Probe Landlock availability from the parent process where tracing works.
|
||||
#[cfg(target_os = "linux")]
|
||||
sandbox::linux::log_sandbox_readiness(policy, workdir.as_deref());
|
||||
if enforcement_mode.enforces_child_sandbox() {
|
||||
sandbox::linux::log_sandbox_readiness(policy, workdir.as_deref());
|
||||
}
|
||||
|
||||
// Phase 1 (as root): Prepare Landlock ruleset before drop_privileges.
|
||||
// Phase 1: Prepare Landlock ruleset before the child applies it.
|
||||
#[cfg(target_os = "linux")]
|
||||
let prepared_sandbox = sandbox::linux::prepare(policy, workdir.as_deref())
|
||||
.map_err(|err| anyhow::anyhow!("Failed to prepare sandbox: {err}"))?;
|
||||
let prepared_sandbox =
|
||||
crate::process::prepare_child_sandbox(policy, workdir.as_deref(), enforcement_mode)
|
||||
.map_err(|err| anyhow::anyhow!("Failed to prepare sandbox: {err}"))?;
|
||||
|
||||
#[cfg(unix)]
|
||||
{
|
||||
@@ -821,6 +847,7 @@ fn spawn_pty_shell(
|
||||
workdir.clone(),
|
||||
slave_fd,
|
||||
netns_fd,
|
||||
enforcement_mode,
|
||||
#[cfg(target_os = "linux")]
|
||||
prepared_sandbox,
|
||||
)?;
|
||||
@@ -913,6 +940,7 @@ fn spawn_pipe_exec(
|
||||
ca_file_paths: Option<Arc<(PathBuf, PathBuf)>>,
|
||||
provider_env: &HashMap<String, String>,
|
||||
user_environment: &HashMap<String, String>,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
) -> anyhow::Result<mpsc::Sender<Vec<u8>>> {
|
||||
let mut cmd = command.map_or_else(
|
||||
|| {
|
||||
@@ -955,12 +983,15 @@ fn spawn_pipe_exec(
|
||||
|
||||
// Probe Landlock availability from the parent process where tracing works.
|
||||
#[cfg(target_os = "linux")]
|
||||
sandbox::linux::log_sandbox_readiness(policy, workdir.as_deref());
|
||||
if enforcement_mode.enforces_child_sandbox() {
|
||||
sandbox::linux::log_sandbox_readiness(policy, workdir.as_deref());
|
||||
}
|
||||
|
||||
// Phase 1 (as root): Prepare Landlock ruleset before drop_privileges.
|
||||
// Phase 1: Prepare Landlock ruleset before the child applies it.
|
||||
#[cfg(target_os = "linux")]
|
||||
let prepared_sandbox = sandbox::linux::prepare(policy, workdir.as_deref())
|
||||
.map_err(|err| anyhow::anyhow!("Failed to prepare sandbox: {err}"))?;
|
||||
let prepared_sandbox =
|
||||
crate::process::prepare_child_sandbox(policy, workdir.as_deref(), enforcement_mode)
|
||||
.map_err(|err| anyhow::anyhow!("Failed to prepare sandbox: {err}"))?;
|
||||
|
||||
#[cfg(unix)]
|
||||
{
|
||||
@@ -969,6 +1000,7 @@ fn spawn_pipe_exec(
|
||||
policy.clone(),
|
||||
workdir.clone(),
|
||||
netns_fd,
|
||||
enforcement_mode,
|
||||
#[cfg(target_os = "linux")]
|
||||
prepared_sandbox,
|
||||
)?;
|
||||
@@ -1068,7 +1100,9 @@ fn spawn_pipe_exec(
|
||||
mod unsafe_pty {
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
use super::sandbox;
|
||||
use super::{Command, RawFd, SandboxPolicy, Winsize, drop_privileges, setsid};
|
||||
use super::{
|
||||
Command, ProcessEnforcementMode, RawFd, SandboxPolicy, Winsize, drop_privileges, setsid,
|
||||
};
|
||||
#[cfg(unix)]
|
||||
use std::os::unix::process::CommandExt;
|
||||
|
||||
@@ -1107,17 +1141,21 @@ mod unsafe_pty {
|
||||
_workdir: Option<String>,
|
||||
slave_fd: RawFd,
|
||||
netns_fd: Option<RawFd>,
|
||||
#[cfg(target_os = "linux")] prepared: crate::sandbox::linux::PreparedSandbox,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
#[cfg(target_os = "linux")] prepared: Option<crate::sandbox::linux::PreparedSandbox>,
|
||||
) -> anyhow::Result<()> {
|
||||
// Wrap in Option so we can .take() it out of the FnMut closure.
|
||||
// pre_exec is only called once (after fork, before exec).
|
||||
#[cfg(target_os = "linux")]
|
||||
let mut prepared = Some(prepared);
|
||||
let mut prepared = prepared;
|
||||
#[cfg(target_os = "linux")]
|
||||
let supervisor_identity_mount = crate::process::supervisor_identity_mount_from_env()
|
||||
.map_err(|err| {
|
||||
let supervisor_identity_mount = if enforcement_mode.uses_privileged_process_setup() {
|
||||
crate::process::supervisor_identity_mount_from_env().map_err(|err| {
|
||||
anyhow::anyhow!("failed to prepare supervisor identity isolation: {err}")
|
||||
})?;
|
||||
})?
|
||||
} else {
|
||||
None
|
||||
};
|
||||
unsafe {
|
||||
cmd.pre_exec(move || {
|
||||
setsid().map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
@@ -1126,6 +1164,7 @@ mod unsafe_pty {
|
||||
enter_netns_and_sandbox(
|
||||
netns_fd,
|
||||
&policy,
|
||||
enforcement_mode,
|
||||
#[cfg(target_os = "linux")]
|
||||
supervisor_identity_mount,
|
||||
#[cfg(target_os = "linux")]
|
||||
@@ -1152,20 +1191,25 @@ mod unsafe_pty {
|
||||
policy: SandboxPolicy,
|
||||
_workdir: Option<String>,
|
||||
netns_fd: Option<RawFd>,
|
||||
#[cfg(target_os = "linux")] prepared: crate::sandbox::linux::PreparedSandbox,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
#[cfg(target_os = "linux")] prepared: Option<crate::sandbox::linux::PreparedSandbox>,
|
||||
) -> anyhow::Result<()> {
|
||||
#[cfg(target_os = "linux")]
|
||||
let mut prepared = Some(prepared);
|
||||
let mut prepared = prepared;
|
||||
#[cfg(target_os = "linux")]
|
||||
let supervisor_identity_mount = crate::process::supervisor_identity_mount_from_env()
|
||||
.map_err(|err| {
|
||||
let supervisor_identity_mount = if enforcement_mode.uses_privileged_process_setup() {
|
||||
crate::process::supervisor_identity_mount_from_env().map_err(|err| {
|
||||
anyhow::anyhow!("failed to prepare supervisor identity isolation: {err}")
|
||||
})?;
|
||||
})?
|
||||
} else {
|
||||
None
|
||||
};
|
||||
unsafe {
|
||||
cmd.pre_exec(move || {
|
||||
enter_netns_and_sandbox(
|
||||
netns_fd,
|
||||
&policy,
|
||||
enforcement_mode,
|
||||
#[cfg(target_os = "linux")]
|
||||
supervisor_identity_mount,
|
||||
#[cfg(target_os = "linux")]
|
||||
@@ -1179,6 +1223,7 @@ mod unsafe_pty {
|
||||
fn enter_netns_and_sandbox(
|
||||
netns_fd: Option<RawFd>,
|
||||
policy: &SandboxPolicy,
|
||||
enforcement_mode: ProcessEnforcementMode,
|
||||
#[cfg(target_os = "linux")] supervisor_identity_mount: Option<
|
||||
&crate::process::SupervisorIdentityMountNamespace,
|
||||
>,
|
||||
@@ -1207,7 +1252,9 @@ mod unsafe_pty {
|
||||
|
||||
// Drop privileges. initgroups/setgid/setuid need /etc/group and
|
||||
// /etc/passwd which would be blocked if Landlock were already enforced.
|
||||
drop_privileges(policy).map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
if enforcement_mode.uses_privileged_process_setup() {
|
||||
drop_privileges(policy).map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
}
|
||||
crate::process::harden_child_process()
|
||||
.map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
|
||||
@@ -1220,7 +1267,9 @@ mod unsafe_pty {
|
||||
}
|
||||
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
sandbox::apply(policy, None).map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
if enforcement_mode.enforces_child_sandbox() {
|
||||
sandbox::apply(policy, None).map_err(|err| std::io::Error::other(err.to_string()))?;
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
@@ -1275,6 +1324,71 @@ mod tests {
|
||||
use super::*;
|
||||
use std::process::Stdio;
|
||||
|
||||
#[cfg(unix)]
|
||||
fn file_mode(path: &Path) -> u32 {
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
std::fs::metadata(path).unwrap().permissions().mode() & 0o7777
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
fn set_file_mode(path: &Path, mode: u32) {
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
std::fs::set_permissions(path, std::fs::Permissions::from_mode(mode)).unwrap();
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn ssh_server_init_full_enforcement_keeps_private_socket() {
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let parent = temp.path().join("ssh");
|
||||
std::fs::create_dir_all(&parent).unwrap();
|
||||
set_file_mode(&parent, 0o775);
|
||||
let socket = parent.join("ssh.sock");
|
||||
|
||||
let (listener, _, _) =
|
||||
ssh_server_init(&socket, &None, ProcessEnforcementMode::Full, false).unwrap();
|
||||
drop(listener);
|
||||
|
||||
assert_eq!(file_mode(&parent), 0o700);
|
||||
assert_eq!(file_mode(&socket), 0o600);
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn ssh_server_init_shared_socket_keeps_group_access() {
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let parent = temp.path().join("ssh");
|
||||
std::fs::create_dir_all(&parent).unwrap();
|
||||
set_file_mode(&parent, 0o775);
|
||||
let socket = parent.join("ssh.sock");
|
||||
|
||||
let (listener, _, _) =
|
||||
ssh_server_init(&socket, &None, ProcessEnforcementMode::Full, true).unwrap();
|
||||
drop(listener);
|
||||
|
||||
assert_eq!(file_mode(&parent), 0o775);
|
||||
assert_eq!(file_mode(&socket), 0o660);
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
#[tokio::test]
|
||||
async fn ssh_server_abstract_socket_cannot_be_replaced_while_bound() {
|
||||
let socket = PathBuf::from(format!("@openshell-ssh-test-{}", uuid::Uuid::new_v4()));
|
||||
let (listener, _, _) =
|
||||
ssh_server_init(&socket, &None, ProcessEnforcementMode::NetworkOnly, true).unwrap();
|
||||
|
||||
assert!(
|
||||
!socket.exists(),
|
||||
"abstract socket must not create a filesystem inode"
|
||||
);
|
||||
let runtime_path = crate::unix_socket::runtime_path(&socket);
|
||||
let err = UnixListener::bind(runtime_path.as_ref())
|
||||
.expect_err("a workload must not be able to replace the bound abstract socket");
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::AddrInUse);
|
||||
|
||||
drop(listener);
|
||||
}
|
||||
|
||||
/// Verify that dropping the input sender (the operation `channel_eof`
|
||||
/// performs) causes the stdin writer loop to exit and close the child's
|
||||
/// stdin pipe. Without this, commands like `cat | tar xf -` used by
|
||||
@@ -1681,21 +1795,24 @@ mod tests {
|
||||
policy,
|
||||
None,
|
||||
None, // no netns fd
|
||||
ProcessEnforcementMode::Full,
|
||||
#[cfg(target_os = "linux")]
|
||||
sandbox::linux::prepare(
|
||||
&SandboxPolicy {
|
||||
version: 0,
|
||||
filesystem: FilesystemPolicy::default(),
|
||||
network: NetworkPolicy::default(),
|
||||
landlock: LandlockPolicy::default(),
|
||||
process: ProcessPolicy {
|
||||
run_as_user: None,
|
||||
run_as_group: None,
|
||||
Some(
|
||||
sandbox::linux::prepare(
|
||||
&SandboxPolicy {
|
||||
version: 0,
|
||||
filesystem: FilesystemPolicy::default(),
|
||||
network: NetworkPolicy::default(),
|
||||
landlock: LandlockPolicy::default(),
|
||||
process: ProcessPolicy {
|
||||
run_as_user: None,
|
||||
run_as_group: None,
|
||||
},
|
||||
},
|
||||
},
|
||||
None,
|
||||
)
|
||||
.expect("prepare should succeed in test environment"),
|
||||
None,
|
||||
)
|
||||
.expect("prepare should succeed in test environment"),
|
||||
),
|
||||
)
|
||||
.expect("install pre_exec should succeed");
|
||||
|
||||
|
||||
@@ -235,12 +235,14 @@ pub fn spawn(
|
||||
sandbox_id: String,
|
||||
ssh_socket_path: std::path::PathBuf,
|
||||
netns_fd: Option<i32>,
|
||||
expected_ssh_peer_pid: Option<u32>,
|
||||
) -> tokio::task::JoinHandle<()> {
|
||||
tokio::spawn(run_session_loop(
|
||||
endpoint,
|
||||
sandbox_id,
|
||||
ssh_socket_path,
|
||||
netns_fd,
|
||||
expected_ssh_peer_pid,
|
||||
))
|
||||
}
|
||||
|
||||
@@ -249,6 +251,7 @@ async fn run_session_loop(
|
||||
sandbox_id: String,
|
||||
ssh_socket_path: std::path::PathBuf,
|
||||
netns_fd: Option<i32>,
|
||||
expected_ssh_peer_pid: Option<u32>,
|
||||
) {
|
||||
let mut backoff = INITIAL_BACKOFF;
|
||||
let mut attempt: u64 = 0;
|
||||
@@ -256,7 +259,15 @@ async fn run_session_loop(
|
||||
loop {
|
||||
attempt += 1;
|
||||
|
||||
match run_single_session(&endpoint, &sandbox_id, &ssh_socket_path, netns_fd).await {
|
||||
match run_single_session(
|
||||
&endpoint,
|
||||
&sandbox_id,
|
||||
&ssh_socket_path,
|
||||
netns_fd,
|
||||
expected_ssh_peer_pid,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(()) => {
|
||||
let event =
|
||||
session_closed_event(openshell_ocsf::ctx::ctx(), &endpoint, &sandbox_id);
|
||||
@@ -283,6 +294,7 @@ async fn run_single_session(
|
||||
sandbox_id: &str,
|
||||
ssh_socket_path: &std::path::Path,
|
||||
netns_fd: Option<i32>,
|
||||
expected_ssh_peer_pid: Option<u32>,
|
||||
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
// Connect to the gateway. The same `Channel` is used for both the
|
||||
// long-lived control stream and all data-plane `RelayStream` calls, so
|
||||
@@ -352,6 +364,7 @@ async fn run_single_session(
|
||||
sandbox_id,
|
||||
ssh_socket_path,
|
||||
netns_fd,
|
||||
expected_ssh_peer_pid,
|
||||
&channel,
|
||||
&tx,
|
||||
);
|
||||
@@ -375,6 +388,7 @@ fn handle_gateway_message(
|
||||
sandbox_id: &str,
|
||||
ssh_socket_path: &std::path::Path,
|
||||
netns_fd: Option<i32>,
|
||||
expected_ssh_peer_pid: Option<u32>,
|
||||
channel: &grpc_client::AuthedChannel,
|
||||
tx: &mpsc::Sender<SupervisorMessage>,
|
||||
) {
|
||||
@@ -395,7 +409,16 @@ fn handle_gateway_message(
|
||||
|
||||
tokio::spawn(async move {
|
||||
let event_open = relay_open.clone();
|
||||
match handle_relay_open(relay_open, &ssh_socket_path, netns_fd, channel, tx).await {
|
||||
match handle_relay_open(
|
||||
relay_open,
|
||||
&ssh_socket_path,
|
||||
netns_fd,
|
||||
expected_ssh_peer_pid,
|
||||
channel,
|
||||
tx,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(()) => {
|
||||
let event = relay_closed_event(
|
||||
openshell_ocsf::ctx::ctx(),
|
||||
@@ -446,11 +469,19 @@ async fn handle_relay_open(
|
||||
relay_open: RelayOpen,
|
||||
ssh_socket_path: &std::path::Path,
|
||||
netns_fd: Option<i32>,
|
||||
expected_ssh_peer_pid: Option<u32>,
|
||||
channel: grpc_client::AuthedChannel,
|
||||
tx: mpsc::Sender<SupervisorMessage>,
|
||||
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
let channel_id = relay_open.channel_id.clone();
|
||||
let target = match open_target(&relay_open, ssh_socket_path, netns_fd).await {
|
||||
let target = match open_target(
|
||||
&relay_open,
|
||||
ssh_socket_path,
|
||||
netns_fd,
|
||||
expected_ssh_peer_pid,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(target) => target,
|
||||
Err(err) => {
|
||||
send_relay_open_result(&tx, &channel_id, false, err.to_string()).await;
|
||||
@@ -576,11 +607,23 @@ async fn open_target(
|
||||
relay_open: &RelayOpen,
|
||||
ssh_socket_path: &std::path::Path,
|
||||
netns_fd: Option<i32>,
|
||||
expected_ssh_peer_pid: Option<u32>,
|
||||
) -> Result<Box<dyn TargetStream>, Box<dyn std::error::Error + Send + Sync>> {
|
||||
match relay_open.target.as_ref() {
|
||||
Some(relay_open::Target::Tcp(target)) => open_tcp_target(target, netns_fd).await,
|
||||
Some(relay_open::Target::Ssh(_)) | None => {
|
||||
let stream = tokio::net::UnixStream::connect(ssh_socket_path).await?;
|
||||
let runtime_path = crate::unix_socket::runtime_path(ssh_socket_path);
|
||||
let stream = tokio::net::UnixStream::connect(runtime_path.as_ref()).await?;
|
||||
if let Some(expected_pid) = expected_ssh_peer_pid {
|
||||
let credentials = stream.peer_cred()?;
|
||||
let actual_pid = credentials.pid().and_then(|pid| u32::try_from(pid).ok());
|
||||
if actual_pid != Some(expected_pid) {
|
||||
return Err(format!(
|
||||
"SSH relay peer PID mismatch: expected {expected_pid}, got {actual_pid:?}"
|
||||
)
|
||||
.into());
|
||||
}
|
||||
}
|
||||
Ok(Box::new(stream))
|
||||
}
|
||||
}
|
||||
@@ -895,4 +938,37 @@ mod ocsf_event_tests {
|
||||
.expect_err("eof should force reconnect");
|
||||
assert_eq!(err.to_string(), "gateway closed stream");
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
#[tokio::test]
|
||||
async fn ssh_target_requires_authenticated_supervisor_peer_pid() {
|
||||
let socket =
|
||||
std::path::PathBuf::from(format!("@openshell-relay-test-{}", uuid::Uuid::new_v4()));
|
||||
let runtime_path = crate::unix_socket::runtime_path(&socket);
|
||||
let listener = tokio::net::UnixListener::bind(runtime_path.as_ref()).unwrap();
|
||||
let accept_task = tokio::spawn(async move {
|
||||
for _ in 0..2 {
|
||||
let (_stream, _) = listener.accept().await.unwrap();
|
||||
}
|
||||
});
|
||||
let relay = ssh_relay_open("peer-check");
|
||||
|
||||
let trusted = open_target(&relay, &socket, None, Some(std::process::id()))
|
||||
.await
|
||||
.expect("matching peer PID should be accepted");
|
||||
drop(trusted);
|
||||
|
||||
let Err(err) = open_target(
|
||||
&relay,
|
||||
&socket,
|
||||
None,
|
||||
Some(std::process::id().saturating_add(1)),
|
||||
)
|
||||
.await
|
||||
else {
|
||||
panic!("mismatched peer PID must be rejected");
|
||||
};
|
||||
assert!(err.to_string().contains("peer PID mismatch"));
|
||||
accept_task.await.unwrap();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
//! Unix-socket address helpers shared by the SSH server and relay client.
|
||||
|
||||
use std::borrow::Cow;
|
||||
use std::path::Path;
|
||||
#[cfg(target_os = "linux")]
|
||||
use std::path::PathBuf;
|
||||
|
||||
/// Return whether a configured socket name denotes a Linux abstract socket.
|
||||
///
|
||||
/// Environment variables cannot contain the leading NUL byte used by the
|
||||
/// kernel ABI, so configuration uses the conventional `@name` spelling.
|
||||
pub fn is_abstract(path: &Path) -> bool {
|
||||
#[cfg(target_os = "linux")]
|
||||
{
|
||||
use std::os::unix::ffi::OsStrExt;
|
||||
path.as_os_str().as_bytes().starts_with(b"@")
|
||||
}
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
{
|
||||
let _ = path;
|
||||
false
|
||||
}
|
||||
}
|
||||
|
||||
/// Translate the configured `@name` spelling into Tokio's NUL-prefixed Linux
|
||||
/// abstract-socket path representation.
|
||||
pub fn runtime_path(path: &Path) -> Cow<'_, Path> {
|
||||
#[cfg(target_os = "linux")]
|
||||
{
|
||||
use std::ffi::OsString;
|
||||
use std::os::unix::ffi::{OsStrExt, OsStringExt};
|
||||
|
||||
let bytes = path.as_os_str().as_bytes();
|
||||
if let Some(name) = bytes.strip_prefix(b"@") {
|
||||
let mut abstract_name = Vec::with_capacity(name.len() + 1);
|
||||
abstract_name.push(0);
|
||||
abstract_name.extend_from_slice(name);
|
||||
return Cow::Owned(PathBuf::from(OsString::from_vec(abstract_name)));
|
||||
}
|
||||
}
|
||||
Cow::Borrowed(path)
|
||||
}
|
||||
|
||||
#[cfg(all(test, target_os = "linux"))]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn at_name_maps_to_abstract_socket_path() {
|
||||
use std::os::unix::ffi::OsStrExt;
|
||||
|
||||
let configured = Path::new("@openshell-test");
|
||||
let runtime = runtime_path(configured);
|
||||
assert!(is_abstract(configured));
|
||||
assert_eq!(runtime.as_os_str().as_bytes(), b"\0openshell-test");
|
||||
}
|
||||
}
|
||||
@@ -5,10 +5,10 @@
|
||||
|
||||
# Supervisor image build.
|
||||
#
|
||||
# The final image is `scratch`: it only carries the static `openshell-sandbox`
|
||||
# binary used by Docker extraction, Podman image volumes, and the Kubernetes
|
||||
# init container copy-self path. A static musl binary lets the image stay
|
||||
# `scratch` while still being executable as an init container.
|
||||
# The final image carries the static `openshell-sandbox` binary used by Docker
|
||||
# extraction, Podman image volumes, and the Kubernetes init container copy-self
|
||||
# path. It also includes nftables so the Kubernetes supervisor sidecar can
|
||||
# install pod-namespace egress enforcement rules.
|
||||
#
|
||||
# The Rust binary is built natively before this image build runs and staged at:
|
||||
# deploy/docker/.build/prebuilt-binaries/<arch>/openshell-sandbox
|
||||
@@ -19,17 +19,16 @@
|
||||
# target) and uploads it as an artifact, which is downloaded into the same
|
||||
# staging directory before the image build job runs.
|
||||
|
||||
FROM scratch AS supervisor
|
||||
FROM alpine:3.22 AS supervisor
|
||||
|
||||
ARG TARGETARCH
|
||||
|
||||
# --chmod=0550 drops world-execute and survives the actions/upload-artifact
|
||||
# + download-artifact roundtrip (which strips exec perms). Ownership is left
|
||||
# at root (0:0) deliberately: the Podman driver mounts this image as a
|
||||
# read-only image volume into the sandbox container and drops DAC_OVERRIDE,
|
||||
# so the container's UID 0 must own the binary to read+exec it. Mode 0550
|
||||
# (r-xr-x---) is the security win; the chown to a non-root UID was breaking
|
||||
# Podman without buying anything since the container is always UID 0.
|
||||
COPY --chmod=0550 deploy/docker/.build/prebuilt-binaries/${TARGETARCH}/openshell-sandbox /openshell-sandbox
|
||||
RUN apk add --no-cache nftables iptables iptables-legacy
|
||||
|
||||
# --chmod=0555 restores execute bits after the actions/upload-artifact +
|
||||
# download-artifact roundtrip strips them. Ownership stays root (0:0) for
|
||||
# Podman image-volume mounts, while world-execute lets the Kubernetes
|
||||
# network sidecar run this binary as the dedicated non-root proxy UID.
|
||||
COPY --chmod=0555 deploy/docker/.build/prebuilt-binaries/${TARGETARCH}/openshell-sandbox /openshell-sandbox
|
||||
|
||||
ENTRYPOINT ["/openshell-sandbox"]
|
||||
|
||||
@@ -239,8 +239,10 @@ add `ci/values-spire.yaml` to the OpenShell release values files.
|
||||
| supervisor.image.pullPolicy | string | `""` | Supervisor image pull policy. Defaults to the gateway image pull policy when empty. |
|
||||
| supervisor.image.repository | string | `"ghcr.io/nvidia/openshell/supervisor"` | Supervisor image repository. Changing it uses the effective gateway image tag unless tag is also set. |
|
||||
| supervisor.image.tag | string | `""` | Supervisor image tag override. Empty uses the version pinned into the gateway unless repository is changed. |
|
||||
| supervisor.sidecar.processBinaryAwareNetworkPolicy | bool | `true` | Keep process/binary-aware network policy enabled in sidecar topology. When false, the network sidecar runs as proxyUid, drops the extra /proc inspection capabilities, and enforces endpoint/L7 policy without matching policy.binaries. |
|
||||
| supervisor.sidecar.proxyUid | int | `1337` | UID for relaxed long-running network sidecars in sidecar topology. Strict process/binary-aware sidecars run as UID 0 so Kubernetes grants the required /proc inspection capabilities into the effective set. The network init container installs nftables rules that exempt the effective sidecar UID. |
|
||||
| supervisor.sideloadMethod | string | `""` | How the supervisor binary is delivered into sandbox pods. Empty (default) = auto-detect from cluster version: K8s >= v1.35 -> "image-volume" (ImageVolume enabled by default; GA in v1.36) K8s < v1.35 -> "init-container" (copies via init container + emptyDir) On K8s v1.33-v1.34 with the ImageVolume feature gate manually enabled, set this to "image-volume" explicitly. |
|
||||
| supervisor.topology | string | `"combined"` | Supervisor pod topology for Kubernetes sandboxes. "combined" runs networking and process supervision in the agent container. |
|
||||
| supervisor.topology | string | `"combined"` | Supervisor pod topology for Kubernetes sandboxes. "combined" runs the current single supervisor container in the agent pod. "sidecar" runs network enforcement in a dedicated sidecar and the process supervisor as a low-capability wrapper in the agent container. |
|
||||
| tolerations | list | `[]` | Tolerations for the gateway pod. |
|
||||
| workload.allowMultiReplicaStatefulSet | bool | `false` | Allow replicaCount > 1 while rendering a StatefulSet. Prefer workload.kind=deployment for external database-backed multi-replica gateways; this override exists for operators who explicitly require StatefulSet identity or storage semantics. |
|
||||
| workload.kind | string | `"statefulset"` | Gateway workload controller kind. Use `statefulset` for the default SQLite database, or `deployment` when server.externalDbSecret points at an external database. |
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
# CI/dev overlay for exercising the Kubernetes supervisor sidecar topology under
|
||||
# a Kata RuntimeClass.
|
||||
#
|
||||
# Use with e2e/with-kube-gateway.sh by setting:
|
||||
# OPENSHELL_E2E_KUBE_EXTRA_VALUES=deploy/helm/openshell/ci/values-sidecar-kata.yaml
|
||||
# The e2e wrapper supplies the image repository and tag through OPENSHELL_REGISTRY
|
||||
# and IMAGE_TAG for existing-cluster runs.
|
||||
|
||||
supervisor:
|
||||
# Use the sidecar topology under Kata so network enforcement runs in the
|
||||
# sidecar and the sandbox agent container stays low-privilege.
|
||||
topology: sidecar
|
||||
sidecar:
|
||||
# Keep strict process/binary-aware network policy enabled for the Kata
|
||||
# validation path. Set this false only when intentionally validating the
|
||||
# documented endpoint/L7-only downgrade mode.
|
||||
processBinaryAwareNetworkPolicy: true
|
||||
|
||||
# Kata validation clusters normally install this RuntimeClass.
|
||||
server:
|
||||
defaultRuntimeClassName: kata-qemu
|
||||
@@ -0,0 +1,18 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
# CI/dev overlay for exercising the Kubernetes supervisor sidecar topology.
|
||||
#
|
||||
# Merge after values.yaml and ci/values-skaffold.yaml:
|
||||
# helm install ... -f values.yaml -f ci/values-skaffold.yaml -f ci/values-sidecar.yaml
|
||||
#
|
||||
# Or set:
|
||||
# OPENSHELL_E2E_KUBE_EXTRA_VALUES=deploy/helm/openshell/ci/values-sidecar.yaml
|
||||
# before running `mise run e2e:kubernetes`.
|
||||
supervisor:
|
||||
topology: sidecar
|
||||
sidecar:
|
||||
# The strict sidecar default requires cross-container /proc identity access.
|
||||
# CI/dev e2e uses the explicit downgraded mode so endpoint and L7 policy
|
||||
# coverage remains runnable on local k3d while that path is hardened.
|
||||
processBinaryAwareNetworkPolicy: false
|
||||
@@ -119,6 +119,8 @@ deploy:
|
||||
# To enable SPIFFE/SPIRE provider token grants (requires the
|
||||
# spire-crds and spire releases above):
|
||||
#- ci/values-spire.yaml
|
||||
# To exercise the Kubernetes supervisor sidecar topology:
|
||||
#- ci/values-sidecar.yaml
|
||||
# To test multi-replica external PostgreSQL behavior:
|
||||
#- ci/values-high-availability.yaml
|
||||
setValueTemplates:
|
||||
@@ -126,3 +128,18 @@ deploy:
|
||||
image.tag: '{{.IMAGE_TAG_openshell_gateway}}'
|
||||
supervisor.image.repository: '{{.IMAGE_REPO_openshell_supervisor}}'
|
||||
supervisor.image.tag: '{{.IMAGE_TAG_openshell_supervisor}}'
|
||||
profiles:
|
||||
- name: sidecar
|
||||
patches:
|
||||
- op: add
|
||||
path: /deploy/helm/releases/0/valuesFiles/-
|
||||
value: ci/values-sidecar.yaml
|
||||
- name: sidecar-mtls
|
||||
patches:
|
||||
- op: add
|
||||
path: /deploy/helm/releases/0/valuesFiles/-
|
||||
value: ci/values-sidecar.yaml
|
||||
- op: add
|
||||
path: /deploy/helm/releases/0/setValues
|
||||
value:
|
||||
server.disableTls: "false"
|
||||
|
||||
@@ -115,7 +115,7 @@ data:
|
||||
grpc_endpoint = {{ include "openshell.grpcEndpoint" . | quote }}
|
||||
service_account_name = {{ include "openshell.sandboxServiceAccountName" . | quote }}
|
||||
supervisor_sideload_method = {{ include "openshell.supervisorSideloadMethod" . | quote }}
|
||||
supervisor_topology = {{ .Values.supervisor.topology | default "combined" | quote }}
|
||||
topology = {{ .Values.supervisor.topology | default "combined" | quote }}
|
||||
sa_token_ttl_secs = {{ .Values.server.sandboxJwt.k8sSaTokenTtlSecs | default 3600 }}
|
||||
{{- if .Values.server.providerTokenGrants.spiffe.enabled }}
|
||||
provider_spiffe_workload_api_socket_path = {{ .Values.server.providerTokenGrants.spiffe.workloadApiSocketPath | quote }}
|
||||
@@ -144,3 +144,7 @@ data:
|
||||
{{- if .Values.supervisor.image.pullPolicy }}
|
||||
supervisor_image_pull_policy = {{ .Values.supervisor.image.pullPolicy | quote }}
|
||||
{{- end }}
|
||||
|
||||
[openshell.drivers.kubernetes.sidecar]
|
||||
proxy_uid = {{ .Values.supervisor.sidecar.proxyUid | default 1337 }}
|
||||
process_binary_aware_network_policy = {{ .Values.supervisor.sidecar.processBinaryAwareNetworkPolicy }}
|
||||
|
||||
@@ -88,7 +88,10 @@ tests:
|
||||
asserts:
|
||||
- matchRegex:
|
||||
path: data["gateway.toml"]
|
||||
pattern: '(?ms)\[openshell\.drivers\.kubernetes\].*?supervisor_topology\s*=\s*"combined"'
|
||||
pattern: '(?ms)\[openshell\.drivers\.kubernetes\].*?topology\s*=\s*"combined"'
|
||||
- notMatchRegex:
|
||||
path: data["gateway.toml"]
|
||||
pattern: 'supervisor[_]topology\s*='
|
||||
|
||||
- it: uses the gateway built-in supervisor image by default
|
||||
template: templates/gateway-config.yaml
|
||||
@@ -126,14 +129,35 @@ tests:
|
||||
path: data["gateway.toml"]
|
||||
pattern: 'supervisor_image\s*=\s*"registry\.example\.com/openshell/supervisor:supervisor-build"'
|
||||
|
||||
- it: renders explicit combined supervisor topology under [openshell.drivers.kubernetes]
|
||||
- it: renders sidecar supervisor topology under [openshell.drivers.kubernetes]
|
||||
template: templates/gateway-config.yaml
|
||||
set:
|
||||
supervisor.topology: combined
|
||||
supervisor.topology: sidecar
|
||||
asserts:
|
||||
- matchRegex:
|
||||
path: data["gateway.toml"]
|
||||
pattern: '(?ms)\[openshell\.drivers\.kubernetes\].*?supervisor_topology\s*=\s*"combined"'
|
||||
pattern: '(?ms)\[openshell\.drivers\.kubernetes\].*?topology\s*=\s*"sidecar"'
|
||||
- notMatchRegex:
|
||||
path: data["gateway.toml"]
|
||||
pattern: 'supervisor[_]topology\s*='
|
||||
|
||||
- it: renders proxy uid under [openshell.drivers.kubernetes.sidecar]
|
||||
template: templates/gateway-config.yaml
|
||||
set:
|
||||
supervisor.sidecar.proxyUid: 2200
|
||||
asserts:
|
||||
- matchRegex:
|
||||
path: data["gateway.toml"]
|
||||
pattern: '(?ms)\[openshell\.drivers\.kubernetes\.sidecar\].*?proxy_uid\s*=\s*2200'
|
||||
|
||||
- it: renders process binary aware network policy under [openshell.drivers.kubernetes.sidecar]
|
||||
template: templates/gateway-config.yaml
|
||||
set:
|
||||
supervisor.sidecar.processBinaryAwareNetworkPolicy: false
|
||||
asserts:
|
||||
- matchRegex:
|
||||
path: data["gateway.toml"]
|
||||
pattern: '(?ms)\[openshell\.drivers\.kubernetes\.sidecar\].*?process_binary_aware_network_policy\s*=\s*false'
|
||||
|
||||
- it: renders sandbox image pull secrets under [openshell.drivers.kubernetes]
|
||||
template: templates/gateway-config.yaml
|
||||
|
||||
@@ -45,8 +45,22 @@ supervisor:
|
||||
# set this to "image-volume" explicitly.
|
||||
sideloadMethod: ""
|
||||
# -- Supervisor pod topology for Kubernetes sandboxes.
|
||||
# "combined" runs networking and process supervision in the agent container.
|
||||
# "combined" runs the current single supervisor container in the agent pod.
|
||||
# "sidecar" runs network enforcement in a dedicated sidecar and the process
|
||||
# supervisor as a low-capability wrapper in the agent container.
|
||||
topology: "combined"
|
||||
sidecar:
|
||||
# -- UID for relaxed long-running network sidecars in sidecar topology.
|
||||
# Strict process/binary-aware sidecars run as UID 0 so Kubernetes grants
|
||||
# the required /proc inspection capabilities into the effective set. The
|
||||
# network init container installs nftables rules that exempt the effective
|
||||
# sidecar UID.
|
||||
proxyUid: 1337
|
||||
# -- Keep process/binary-aware network policy enabled in sidecar topology.
|
||||
# When false, the network sidecar runs as proxyUid, drops the extra /proc
|
||||
# inspection capabilities, and enforces endpoint/L7 policy without matching
|
||||
# policy.binaries.
|
||||
processBinaryAwareNetworkPolicy: true
|
||||
|
||||
# -- Image pull secrets attached to gateway and helper pods.
|
||||
imagePullSecrets: []
|
||||
|
||||
@@ -5,7 +5,7 @@ title: "Access Control"
|
||||
sidebar-title: "Access Control"
|
||||
description: "Configure OIDC user authentication or reverse-proxy auth termination for a Kubernetes-deployed OpenShell gateway."
|
||||
keywords: "Generative AI, Cybersecurity, Kubernetes, Authentication, mTLS, OIDC, Keycloak, Entra ID, Okta, Gateway Auth"
|
||||
position: 4
|
||||
position: 5
|
||||
---
|
||||
|
||||
The OpenShell gateway supports two access-control models for human callers on Kubernetes:
|
||||
|
||||
@@ -5,7 +5,7 @@ title: "Ingress"
|
||||
sidebar-title: "Ingress"
|
||||
description: "Expose the OpenShell gateway externally using the Kubernetes Gateway API and a GRPCRoute."
|
||||
keywords: "Generative AI, Cybersecurity, Kubernetes, Gateway API, Envoy Gateway, GRPCRoute, Ingress, External Access"
|
||||
position: 3
|
||||
position: 4
|
||||
---
|
||||
|
||||
By default, the OpenShell gateway is only reachable inside the cluster. To let CLI clients connect without a `kubectl port-forward`, expose the gateway through an ingress.
|
||||
|
||||
@@ -5,7 +5,7 @@ title: "Managing Certificates"
|
||||
sidebar-title: "Managing Certificates"
|
||||
description: "Configure the OpenShell Helm chart to use cert-manager for mTLS certificate issuance and automatic renewal."
|
||||
keywords: "Generative AI, Cybersecurity, Kubernetes, cert-manager, PKI, TLS, mTLS, Certificates"
|
||||
position: 2
|
||||
position: 3
|
||||
---
|
||||
|
||||
The OpenShell gateway uses mTLS certificates for transport between the gateway and sandbox supervisors. These certificates are not Kubernetes user authentication; configure OIDC or a trusted access proxy for user access. The Helm chart supports two ways to provision and manage the certificate bundle:
|
||||
|
||||
@@ -5,7 +5,7 @@ title: "OpenShift"
|
||||
sidebar-title: "OpenShift"
|
||||
description: "Install the OpenShell Helm chart on OpenShift, including the SCC binding and chart overrides required by OpenShift's Security Context Constraints."
|
||||
keywords: "Generative AI, Cybersecurity, Kubernetes, OpenShift, SCC, Security Context Constraints, Helm, Gateway, Installation"
|
||||
position: 5
|
||||
position: 6
|
||||
---
|
||||
|
||||
<Warning>
|
||||
@@ -14,6 +14,11 @@ The OpenShift install path is experimental. It currently requires running sandbo
|
||||
|
||||
OpenShift's [Security Context Constraints](https://docs.openshift.com/container-platform/latest/authentication/managing-security-context-constraints.html) reject the chart's default pod security settings. Installing on OpenShift requires precreating the namespace, granting the `privileged` SCC to the sandbox service account, and overriding a few chart values so the cluster admission controller can assign UIDs and FS groups itself.
|
||||
|
||||
OpenShell installs sandbox nftables rules as individual commands. On OpenShift
|
||||
nodes where optional conntrack or packet log expressions are unavailable, those
|
||||
optional rules can fail without rolling back the required proxy bypass reject
|
||||
rules.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- OpenShift 4.x cluster with `oc` configured
|
||||
|
||||
@@ -161,6 +161,7 @@ The most commonly changed values are:
|
||||
| `pkiInitJob.serverDnsNames` / `certManager.serverDnsNames` | Additional gateway server DNS SANs. Wildcard SANs also enable sandbox service URLs under that domain. |
|
||||
| `supervisor.sideloadMethod` | How the supervisor binary is delivered into sandbox pods. Leave empty to auto-detect based on cluster version: clusters running Kubernetes 1.35 or later use `image-volume` (ImageVolume GA in 1.36); older clusters use `init-container`. Set explicitly to `image-volume` on Kubernetes 1.33 or 1.34 with the ImageVolume feature gate enabled, or to `init-container` to force the legacy path on any version. |
|
||||
| `supervisor.topology` | Sandbox pod topology. Refer to [Topology](/kubernetes/topology). |
|
||||
| `supervisor.sidecar.proxyUid` | Non-root UID used when sidecar process/binary-aware network policy is disabled. The default binary-aware sidecar runs as UID 0 instead. The configured UID must not match the sandbox UID. |
|
||||
|
||||
Use a values file for repeatable deployments:
|
||||
|
||||
@@ -244,7 +245,7 @@ The gateway exposes `/healthz` for process liveness and `/readyz` for dependency
|
||||
|
||||
## Next Steps
|
||||
|
||||
- To review Kubernetes sandbox topology, refer to [Topology](/kubernetes/topology).
|
||||
- To choose between combined and sidecar sandbox pods, refer to [Topology](/kubernetes/topology).
|
||||
- To enable automatic certificate rotation with cert-manager, refer to [Managing Certificates](/kubernetes/managing-certificates).
|
||||
- To expose the gateway externally without port-forwarding, refer to [Ingress](/kubernetes/ingress).
|
||||
- To configure OIDC or reverse-proxy authentication, refer to [Access Control](/kubernetes/access-control).
|
||||
|
||||
+171
-24
@@ -3,37 +3,36 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
title: "Kubernetes Sandbox Topology"
|
||||
sidebar-title: "Topology"
|
||||
description: "Review the default combined supervisor topology for Kubernetes sandbox pods."
|
||||
keywords: "Generative AI, Cybersecurity, Kubernetes, Sandboxing, RuntimeClass"
|
||||
description: "Choose between combined and sidecar supervisor topology for Kubernetes sandbox pods."
|
||||
keywords: "Generative AI, Cybersecurity, Kubernetes, Sandboxing, Sidecar, Network Policy, RuntimeClass"
|
||||
position: 2
|
||||
---
|
||||
|
||||
Kubernetes sandbox pods run the OpenShell supervisor in `combined` topology by
|
||||
default. Combined topology keeps network, filesystem, and process controls in
|
||||
the agent pod so the supervisor can enforce the complete OpenShell sandbox
|
||||
contract before launching the workload.
|
||||
Kubernetes sandbox pods can run the OpenShell supervisor in `combined` or
|
||||
`sidecar` topology. Choose the topology based on which controls you need inside
|
||||
the pod and how much privilege your cluster allows on the agent container.
|
||||
|
||||
## Choose a Topology
|
||||
|
||||
The default `combined` topology preserves the full OpenShell enforcement model.
|
||||
Use it when you need OpenShell to apply all sandbox controls inside the workload
|
||||
pod and your cluster policy permits the required Linux capabilities.
|
||||
Use `sidecar` only when you accept network-focused enforcement in exchange for a
|
||||
lower-privilege agent container.
|
||||
|
||||
| Topology | Use when | Main tradeoff |
|
||||
|---|---|---|
|
||||
| `combined` | You need OpenShell network, filesystem, and process controls in the sandbox workload. | The agent container carries the Linux capabilities the supervisor needs. |
|
||||
|
||||
Additional Kubernetes sandbox topologies are still being designed. Until they
|
||||
are documented as supported configuration values, `combined` is the only
|
||||
supported value for `supervisor.topology`.
|
||||
| `sidecar` | You need the agent container to run as non-root without added Linux capabilities, and network policy is the primary control. | Privilege-dropping and supervisor mount isolation do not run in the agent container. |
|
||||
|
||||
## Privilege Model
|
||||
|
||||
The long-running container permissions for `combined` topology are:
|
||||
The long-running container permissions differ by topology:
|
||||
|
||||
| Topology | Pod or container | UID/GID | Privilege escalation | Capabilities | Result |
|
||||
|---|---|---|---|---|---|
|
||||
| `combined` | Agent container, which also runs the supervisor | Not forced by topology | Not explicitly disabled by the driver | Adds `SYS_ADMIN`, `NET_ADMIN`, `SYS_PTRACE`, and `SYSLOG`; adds `SETUID`, `SETGID`, and `DAC_READ_SEARCH` when user namespaces are enabled | Full supervisor controls run in the agent container. |
|
||||
| `sidecar` | Agent container, process-only supervisor (`network-only`) | `sandbox_uid:sandbox_gid` | `false` | Drops `ALL` | Agent and workload run without added Linux capabilities. |
|
||||
| `sidecar` | Network supervisor sidecar, binary-aware mode (default) | `0:sandbox_gid` | `false` | Drops `ALL`; adds `SYS_PTRACE` and `DAC_READ_SEARCH` | Root sidecar inspects cross-UID workload `/proc` entries. The nftables fence exempts UID 0, so do not inject other root containers into these pods. |
|
||||
| `sidecar` | Network supervisor sidecar, endpoint/L7-only mode | `proxyUid:sandbox_gid` | `false` | Drops `ALL` | Non-root sidecar enforces endpoint and L7 policy without matching `policy.binaries`. |
|
||||
|
||||
Short-lived setup containers still have the permissions needed to prepare the
|
||||
pod:
|
||||
@@ -41,6 +40,7 @@ pod:
|
||||
| Topology | Setup container | UID/GID | Privilege escalation | Capabilities | Purpose |
|
||||
|---|---|---|---|---|---|
|
||||
| `combined` | Supervisor install init container | `0` | Not set | Not set | Copies the supervisor binary into the agent container volume. |
|
||||
| `sidecar` | Network init container | `0` | `false` | Drops `ALL`; adds `NET_ADMIN`, `NET_RAW`, `CHOWN`, and `FOWNER` | Installs pod-local nftables rules and prepares shared sidecar state. |
|
||||
|
||||
## Combined Topology
|
||||
|
||||
@@ -48,6 +48,26 @@ Combined topology is the original Kubernetes mode and remains the default. The
|
||||
agent container starts the OpenShell supervisor, and the supervisor launches the
|
||||
workload after applying sandbox setup.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
Sandbox["agents.x-k8s.io Sandbox"]
|
||||
|
||||
subgraph Pod["Sandbox pod"]
|
||||
subgraph Agent["agent container"]
|
||||
Supervisor["OpenShell supervisor<br/>network + process + filesystem"]
|
||||
Workload["Agent workload"]
|
||||
end
|
||||
end
|
||||
|
||||
Gateway["OpenShell Gateway"]
|
||||
External["External services"]
|
||||
|
||||
Sandbox --> Pod
|
||||
Supervisor --> Workload
|
||||
Supervisor -->|"gateway callback / SSH relay"| Gateway
|
||||
Supervisor -->|"policy-enforced egress"| External
|
||||
```
|
||||
|
||||
Combined topology keeps these controls in one supervisor path:
|
||||
|
||||
- Network endpoint and L7 policy enforcement.
|
||||
@@ -61,12 +81,119 @@ controls from the agent container, Kubernetes grants that container elevated
|
||||
Linux capabilities. Use this mode when you need the complete OpenShell sandbox
|
||||
contract and your cluster policy permits those capabilities.
|
||||
|
||||
## Sidecar Topology
|
||||
|
||||
Sidecar topology splits the supervisor into a network sidecar and a
|
||||
low-privilege process supervisor in the agent container.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
Sandbox["agents.x-k8s.io Sandbox"]
|
||||
|
||||
subgraph Pod["Sandbox pod"]
|
||||
Init["network init container<br/>root setup capabilities"]
|
||||
State["shared state + TLS volumes"]
|
||||
NetNS["pod network namespace"]
|
||||
|
||||
subgraph Agent["agent container"]
|
||||
ProcessSupervisor["process supervisor<br/>network-only"]
|
||||
Workload["Agent workload"]
|
||||
end
|
||||
|
||||
NetworkSidecar["network supervisor sidecar<br/>UID 0 by default"]
|
||||
SshEndpoint["abstract SSH relay socket<br/>peer-PID authenticated"]
|
||||
end
|
||||
|
||||
Gateway["OpenShell Gateway"]
|
||||
External["External services"]
|
||||
|
||||
Sandbox --> Pod
|
||||
Init -->|"installs nftables rules"| NetNS
|
||||
ProcessSupervisor --> Workload
|
||||
Workload -->|"egress redirected on loopback"| NetworkSidecar
|
||||
NetworkSidecar -->|"gateway session + relays"| Gateway
|
||||
NetworkSidecar -->|"policy-enforced egress"| External
|
||||
NetworkSidecar -->|"control socket + proxy TLS"| State
|
||||
ProcessSupervisor -->|"bootstrap + updates"| State
|
||||
ProcessSupervisor --> SshEndpoint
|
||||
NetworkSidecar -->|"SSH relay"| SshEndpoint
|
||||
NetworkSidecar --- State
|
||||
```
|
||||
|
||||
The pod contains these OpenShell-managed pieces:
|
||||
|
||||
| Component | Runs as | Purpose |
|
||||
|---|---|---|
|
||||
| Network init container | Root with setup capabilities | Installs pod-level nftables rules and prepares shared sidecar state. |
|
||||
| Network sidecar | UID 0 by default; `supervisor.sidecar.proxyUid` when binary-aware policy is disabled | Runs the proxy, enforces network policy, owns gateway authentication and the gateway session, and serves local policy/provider state over the sidecar control socket. |
|
||||
| Agent container | Resolved sandbox UID/GID | Runs the process supervisor and launches the user workload. |
|
||||
|
||||
In this topology, the agent container defaults to `runAsNonRoot: true`,
|
||||
`allowPrivilegeEscalation: false`, and `capabilities.drop: ["ALL"]`. The
|
||||
default binary-aware network sidecar runs as UID 0, drops default Linux
|
||||
capabilities, and adds `SYS_PTRACE` plus `DAC_READ_SEARCH` for cross-UID workload
|
||||
process identity resolution. Setting
|
||||
`supervisor.sidecar.processBinaryAwareNetworkPolicy=false` runs the sidecar as
|
||||
the configured non-root `proxyUid`, omits both capabilities, and downgrades
|
||||
network policy to endpoint/L7 enforcement without binary matching. The root
|
||||
init container keeps the setup capabilities needed to configure pod networking.
|
||||
|
||||
Sidecar mode preserves gateway session behavior, including SSH connectivity,
|
||||
because the network sidecar owns the gateway session and bridges relay requests
|
||||
to a Linux abstract SSH socket owned by the process supervisor. The relay
|
||||
verifies the socket peer PID against the authenticated control connection, so
|
||||
the workload cannot replace the relay endpoint. The agent container does not get a
|
||||
gateway endpoint, gateway TLS material, or the sandbox bootstrap token in the
|
||||
default sidecar path.
|
||||
|
||||
<Warning>
|
||||
Sidecar mode runs the process supervisor in `network-only` mode. OpenShell still
|
||||
enforces network endpoint and L7 policy through the sidecar, and the process
|
||||
supervisor applies Landlock filesystem policy and child seccomp filters where
|
||||
the kernel/runtime supports them. The process supervisor does not perform
|
||||
root-to-sandbox privilege dropping because Kubernetes starts the container as
|
||||
the sandbox UID/GID, and it does not perform supervisor identity mount
|
||||
isolation because gateway credentials are not mounted into the agent container.
|
||||
Sidecar pods use `shareProcessNamespace: true` so the network sidecar can
|
||||
resolve workload process and binary identity through `/proc/<entrypoint-pid>`.
|
||||
</Warning>
|
||||
|
||||
## Credential Exposure
|
||||
|
||||
Sidecar topology keeps gateway credentials in the network sidecar. The agent
|
||||
container does not mount the projected ServiceAccount token used for sandbox
|
||||
token bootstrap, does not mount the sandbox client TLS secret, and does not get
|
||||
gateway callback environment variables.
|
||||
|
||||
The network sidecar serves the policy and workload-facing provider environment
|
||||
over a Unix control socket in the shared sidecar state volume. Before launching
|
||||
the workload, the process supervisor establishes the only accepted connection.
|
||||
The sidecar validates its UID, GID, and PID with peer credentials, unlinks the
|
||||
listener, derives the SSH target from trusted configuration, and rejects later
|
||||
clients. The connection receives bootstrap state and provider-environment
|
||||
updates after settings polls. If it closes, the network sidecar exits so
|
||||
Kubernetes recreates the one-client bootstrap listener, and the process
|
||||
supervisor exits so Kubernetes terminates the workload and restarts the agent
|
||||
container. This symmetric failure behavior prevents a surviving workload from
|
||||
claiming the new control listener after an isolated sidecar restart. Future
|
||||
child processes can see refreshed provider env without giving the agent
|
||||
container gateway authentication material. This does not mutate the environment
|
||||
of the already-running workload entrypoint. Use `combined` topology when you
|
||||
need the full single-supervisor enforcement path; use additional runtime
|
||||
isolation when you need a stronger container boundary around sidecar workloads.
|
||||
|
||||
## RuntimeClass Isolation
|
||||
|
||||
RuntimeClass isolation can add a stronger container boundary for the sandbox
|
||||
workload when the cluster supports it. Runtime classes do not replace the
|
||||
combined topology's supervisor controls; they add another isolation boundary
|
||||
around the same supervised workload.
|
||||
Sidecar topology has been validated with Kata Containers. It does not currently
|
||||
support gVisor because sidecar mode requires pod-local nftables setup, which
|
||||
gVisor does not provide to the init container. A supported sandboxed runtime
|
||||
strengthens the container boundary while OpenShell focuses on network policy
|
||||
enforcement from the sidecar.
|
||||
|
||||
Runtime classes do not re-enable the OpenShell privilege-drop or supervisor
|
||||
mount-isolation controls that sidecar mode relaxes. Use them as an additional
|
||||
workload boundary, not as a replacement for the combined topology's full
|
||||
supervisor controls.
|
||||
|
||||
You can set a default runtime class in the Kubernetes driver configuration or
|
||||
override it per sandbox with driver config:
|
||||
@@ -77,24 +204,44 @@ openshell sandbox create \
|
||||
-- claude
|
||||
```
|
||||
|
||||
## Configure Combined Mode
|
||||
## Enable Sidecar Mode
|
||||
|
||||
For direct gateway TOML configuration, leave `supervisor_topology` unset, or
|
||||
set it to `combined`, to use the default single-container supervisor path:
|
||||
For direct gateway TOML configuration, set the Kubernetes driver fields:
|
||||
|
||||
```toml
|
||||
[openshell.drivers.kubernetes]
|
||||
supervisor_topology = "combined"
|
||||
topology = "sidecar"
|
||||
|
||||
[openshell.drivers.kubernetes.sidecar]
|
||||
proxy_uid = 1337
|
||||
```
|
||||
|
||||
When the Helm chart renders `gateway.toml`, leave `supervisor.topology` unset,
|
||||
or set it to `combined`, to produce the same driver configuration:
|
||||
`proxy_uid` configures only the relaxed endpoint/L7-only sidecar. It must be a
|
||||
non-root UID and must not match the sandbox UID. The default binary-aware mode
|
||||
runs the sidecar as UID 0 instead. The network init container exempts the
|
||||
effective sidecar UID from proxy redirection so the sidecar can reach the
|
||||
gateway.
|
||||
|
||||
When the Helm chart renders `gateway.toml`, set the equivalent chart values:
|
||||
|
||||
```yaml
|
||||
supervisor:
|
||||
topology: combined
|
||||
topology: sidecar
|
||||
sidecar:
|
||||
proxyUid: 1337
|
||||
processBinaryAwareNetworkPolicy: true
|
||||
```
|
||||
|
||||
Leave `topology` unset, or set it to `combined`, to keep the original
|
||||
single-container supervisor path. For Helm installs, leave
|
||||
`supervisor.topology` unset or set it to `combined`.
|
||||
|
||||
Set `supervisor.sidecar.processBinaryAwareNetworkPolicy=false` only when you
|
||||
accept downgrading sidecar network policy to endpoint/L7 enforcement without
|
||||
matching `policy.binaries`. This changes the sidecar from UID 0 to `proxyUid`
|
||||
and removes its `SYS_PTRACE` and `DAC_READ_SEARCH` capabilities, which are used
|
||||
for cross-UID `/proc` inspection.
|
||||
|
||||
## Next Steps
|
||||
|
||||
- To install OpenShell on Kubernetes, refer to [Setup](/kubernetes/setup).
|
||||
|
||||
@@ -179,9 +179,10 @@ supervisor_image_pull_policy = "IfNotPresent"
|
||||
# Use the image volume on Kubernetes >= 1.35 (GA in 1.36); switch to "init-container"
|
||||
# on older clusters or where the ImageVolume feature gate is off.
|
||||
supervisor_sideload_method = "image-volume"
|
||||
# "combined" runs networking and process supervision in the sandbox agent
|
||||
# container and preserves the existing Kubernetes sandbox behavior.
|
||||
supervisor_topology = "combined"
|
||||
# "combined" runs the existing single supervisor container with full process,
|
||||
# filesystem, and network enforcement in the agent container. "sidecar" moves
|
||||
# pod-level network enforcement and gateway session handling into a network sidecar.
|
||||
topology = "combined"
|
||||
grpc_endpoint = "https://openshell-gateway.agents.svc:8080"
|
||||
ssh_socket_path = "/run/openshell/ssh.sock"
|
||||
client_tls_secret_name = "openshell-client-tls"
|
||||
@@ -205,6 +206,18 @@ provider_spiffe_workload_api_socket_path = "/spiffe-workload-api/spire-agent.soc
|
||||
# back to 1000 on non-OpenShift clusters.
|
||||
# sandbox_uid = 1500
|
||||
# sandbox_gid = 1500
|
||||
|
||||
[openshell.drivers.kubernetes.sidecar]
|
||||
# UID used by relaxed long-running network sidecars. Strict process/binary-aware
|
||||
# sidecars run as UID 0 so Kubernetes grants the required /proc inspection
|
||||
# capabilities into the effective set. In sidecar topology the network init
|
||||
# container installs nftables rules that exempt the effective sidecar UID.
|
||||
proxy_uid = 1337
|
||||
# Keep process/binary-aware network policy enabled in sidecar topology. Set
|
||||
# false to run the sidecar as proxy_uid, drop the sidecar's extra /proc
|
||||
# inspection capabilities, and enforce endpoint/L7 policy without matching
|
||||
# policy.binaries.
|
||||
process_binary_aware_network_policy = true
|
||||
```
|
||||
|
||||
### Docker
|
||||
|
||||
@@ -306,10 +306,38 @@ For maintainer-level implementation details, refer to the [Kubernetes driver REA
|
||||
| `supervisor_image` | `supervisor.image.repository` / `supervisor.image.tag` | Override the supervisor image that provides the `openshell-sandbox` binary. The default repository with an empty tag uses the version-pinned image built into the gateway. Changing the repository uses the effective gateway image tag, while setting a tag pins that version explicitly. |
|
||||
| `supervisor_image_pull_policy` | `supervisor.image.pullPolicy` | Set the Kubernetes image pull policy for the supervisor image. |
|
||||
| `supervisor_sideload_method` | `supervisor.sideloadMethod` | How the supervisor binary is delivered into sandbox pods. Leave empty to auto-detect from cluster version. Set to `image-volume` to mount the supervisor OCI image directly as a volume (requires Kubernetes 1.33+ with the ImageVolume feature gate; GA in 1.36), or `init-container` to copy it through an init container on older clusters. |
|
||||
| `topology` | `supervisor.topology` | Set `combined` for the default single supervisor path, or `sidecar` to move pod-level network enforcement and the gateway session into a dedicated sidecar. |
|
||||
| `sidecar.proxy_uid` | `supervisor.sidecar.proxyUid` | Non-root UID used by the relaxed sidecar when process/binary-aware network policy is disabled. The default binary-aware sidecar runs as UID 0. The network init container exempts the effective sidecar UID from proxy redirection. |
|
||||
| `sidecar.process_binary_aware_network_policy` | `supervisor.sidecar.processBinaryAwareNetworkPolicy` | Keep process/binary-aware network policy enabled in `sidecar` topology. The default runs the sidecar as UID 0 with `SYS_PTRACE` and `DAC_READ_SEARCH`. Set false to run as `proxy_uid`, drop both capabilities, and enforce endpoint/L7 policy without matching `policy.binaries`. |
|
||||
| `app_armor_profile` | `server.appArmorProfile` | Set the sandbox agent container's AppArmor profile. Helm defaults this to `Unconfined` so AppArmor-enabled nodes do not block supervisor network namespace setup. Set the Helm value to an empty string to omit the field, or use `RuntimeDefault` or `Localhost/<profile-name>` for operator-managed profiles. |
|
||||
| `workspace_default_storage_size` | `server.workspaceDefaultStorageSize` | Set the default workspace PVC size for new sandboxes. |
|
||||
| `sa_token_ttl_secs` | `server.sandboxJwt.k8sSaTokenTtlSecs` | Set the projected ServiceAccount token TTL used for the bootstrap token exchange. |
|
||||
|
||||
In `combined` topology, the agent container carries the Linux capabilities
|
||||
needed by the supervisor for network namespace setup, Landlock filesystem
|
||||
policy, process privilege changes, and network policy enforcement. In `sidecar`
|
||||
topology, the agent container runs as the resolved sandbox UID/GID with no added
|
||||
Linux capabilities. A root init container performs the nftables setup, and the
|
||||
long-running binary-aware sidecar runs as UID 0, drops default capabilities,
|
||||
and adds `SYS_PTRACE` plus `DAC_READ_SEARCH` for workload process identity
|
||||
resolution through shared `/proc`. The
|
||||
`sidecar.process_binary_aware_network_policy = false` setting runs it as the
|
||||
configured non-root `proxy_uid`, removes both capabilities, and relaxes network
|
||||
policy to endpoint/L7 matching only. The
|
||||
network sidecar owns gateway authentication and writes local policy/provider
|
||||
state to the process supervisor over a local control socket, so the agent
|
||||
container does not mount the sandbox bootstrap token or client TLS secret in
|
||||
the default sidecar path. The provider environment is refreshed by the network
|
||||
sidecar after settings polls and streamed to the process supervisor so future
|
||||
child processes can see updated provider env without gateway access in the
|
||||
agent container.
|
||||
Sidecar mode keeps gateway session and SSH behavior. The process supervisor
|
||||
applies Landlock filesystem policy and child seccomp filters where supported,
|
||||
but it does not perform root-to-sandbox privilege dropping or supervisor
|
||||
identity mount isolation. Network policy still runs in the sidecar, and sidecar
|
||||
pods set `shareProcessNamespace: true` so the network sidecar can resolve
|
||||
process/binary identity through `/proc/<entrypoint-pid>`.
|
||||
|
||||
The Kubernetes driver creates namespaced `agents.x-k8s.io` `Sandbox` resources from the Kubernetes SIG Apps [agent-sandbox](https://github.com/kubernetes-sigs/agent-sandbox) project. It detects the served Sandbox API at runtime, caches the selected API version for the gateway process, and uses `v1beta1` when available before falling back to `v1alpha1`, so supported Agent Sandbox installations work without version-specific operator configuration. The Agent Sandbox controller turns those resources into sandbox pods and related storage.
|
||||
|
||||
If Agent Sandbox is upgraded in place, restart the OpenShell gateway after the controller and CRD rollout completes so the gateway can detect the served API versions again.
|
||||
|
||||
@@ -393,6 +393,21 @@ require_cmd() {
|
||||
fi
|
||||
}
|
||||
|
||||
configure_fixture_container_engine() {
|
||||
[ -n "${CONTAINER_ENGINE:-}" ] || return 0
|
||||
local selected_engine
|
||||
selected_engine="$(printf '%s' "${CONTAINER_ENGINE}" | tr '[:upper:]' '[:lower:]')"
|
||||
case "${selected_engine}" in
|
||||
docker|podman)
|
||||
;;
|
||||
*)
|
||||
echo "ERROR: CONTAINER_ENGINE=${CONTAINER_ENGINE} is invalid; expected docker or podman" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
export CONTAINER_ENGINE="${selected_engine}"
|
||||
}
|
||||
|
||||
require_cmd helm
|
||||
require_cmd kubectl
|
||||
require_cmd curl
|
||||
@@ -423,6 +438,8 @@ else
|
||||
KUBE_CONTEXT="k3d-${CLUSTER_NAME}"
|
||||
fi
|
||||
|
||||
configure_fixture_container_engine
|
||||
|
||||
if [ -z "${OPENSHELL_E2E_KUBE_BUILD_IMAGES+x}" ]; then
|
||||
if [ "${CLUSTER_CREATED_BY_US}" = "1" ]; then
|
||||
OPENSHELL_E2E_KUBE_BUILD_IMAGES=1
|
||||
@@ -501,7 +518,7 @@ if [ -z "${HOST_GATEWAY_IP}" ] \
|
||||
# is unreachable for the typical test-host listener (0.0.0.0 bind).
|
||||
detected="$(docker network inspect "${net}" \
|
||||
-f '{{range .IPAM.Config}}{{.Gateway}}{{"\n"}}{{end}}' 2>/dev/null \
|
||||
| awk '/^[0-9.]+$/ { print; exit }')"
|
||||
| awk '/^[0-9.]+$/ { print; exit }' || true)"
|
||||
if [ -n "${detected}" ]; then
|
||||
HOST_GATEWAY_IP="${detected}"
|
||||
echo "Detected host gateway IP ${HOST_GATEWAY_IP} from docker network '${net}'."
|
||||
|
||||
@@ -55,16 +55,46 @@ description = "Run skaffold dev for deploy/helm/openshell (iterative deploy)"
|
||||
dir = "deploy/helm/openshell"
|
||||
run = "skaffold dev"
|
||||
|
||||
["helm:skaffold:dev:sidecar"]
|
||||
description = "Run skaffold dev with the Kubernetes supervisor sidecar topology"
|
||||
dir = "deploy/helm/openshell"
|
||||
run = "skaffold dev -p sidecar"
|
||||
|
||||
["helm:skaffold:dev:sidecar-mtls"]
|
||||
description = "Run skaffold dev with the Kubernetes supervisor sidecar topology and TLS/mTLS enabled"
|
||||
dir = "deploy/helm/openshell"
|
||||
run = "skaffold dev -p sidecar-mtls"
|
||||
|
||||
["helm:skaffold:run"]
|
||||
description = "Run skaffold run for deploy/helm/openshell (one-shot deploy)"
|
||||
dir = "deploy/helm/openshell"
|
||||
run = "skaffold run"
|
||||
|
||||
["helm:skaffold:run:sidecar"]
|
||||
description = "Run skaffold run with the Kubernetes supervisor sidecar topology"
|
||||
dir = "deploy/helm/openshell"
|
||||
run = "skaffold run -p sidecar"
|
||||
|
||||
["helm:skaffold:run:sidecar-mtls"]
|
||||
description = "Run skaffold run with the Kubernetes supervisor sidecar topology and TLS/mTLS enabled"
|
||||
dir = "deploy/helm/openshell"
|
||||
run = "skaffold run -p sidecar-mtls"
|
||||
|
||||
["helm:skaffold:delete"]
|
||||
description = "Run skaffold delete for deploy/helm/openshell"
|
||||
dir = "deploy/helm/openshell"
|
||||
run = "skaffold delete"
|
||||
|
||||
["helm:skaffold:delete:sidecar"]
|
||||
description = "Run skaffold delete for the Kubernetes supervisor sidecar topology"
|
||||
dir = "deploy/helm/openshell"
|
||||
run = "skaffold delete -p sidecar"
|
||||
|
||||
["helm:skaffold:delete:sidecar-mtls"]
|
||||
description = "Run skaffold delete for the Kubernetes supervisor sidecar topology with TLS/mTLS enabled"
|
||||
dir = "deploy/helm/openshell"
|
||||
run = "skaffold delete -p sidecar-mtls"
|
||||
|
||||
["helm:skaffold:diagnose"]
|
||||
description = "Run skaffold diagnose for deploy/helm/openshell"
|
||||
dir = "deploy/helm/openshell"
|
||||
|
||||
@@ -114,6 +114,11 @@ run = [
|
||||
"AGENT_SANDBOX_VERSION=v0.4.6 e2e/rust/e2e-kubernetes.sh",
|
||||
]
|
||||
|
||||
["e2e:kubernetes:sidecar"]
|
||||
description = "Run Kubernetes e2e with the supervisor sidecar topology overlay"
|
||||
env = { OPENSHELL_E2E_KUBE_EXTRA_VALUES = "deploy/helm/openshell/ci/values-sidecar.yaml" }
|
||||
run = "e2e/rust/e2e-kubernetes.sh"
|
||||
|
||||
["e2e:kubernetes:db"]
|
||||
description = "Run Kubernetes e2e with all database backend scenarios (SQLite and external PostgreSQL with existingSecret)"
|
||||
env = { OPENSHELL_E2E_KUBE_DB_SCENARIOS = "1" }
|
||||
|
||||
Reference in New Issue
Block a user