Files
OpenShell/e2e/with-kube-gateway.sh
Simon Scatton e21b7fd8cf chore(build): remove bundled Z3 support (#3275)
* chore(build): remove bundled Z3 support

Signed-off-by: Simon Scatton <sscatton@nvidia.com>
Signed-off-by: Piotr Mlocek <pmlocek@nvidia.com>

* fix(build): preserve vendored Z3 for local gateway artifacts

Signed-off-by: Simon Scatton <sscatton@nvidia.com>

---------

Signed-off-by: Simon Scatton <sscatton@nvidia.com>
Signed-off-by: Piotr Mlocek <pmlocek@nvidia.com>
2026-10-01 12:08:33 +00:00

1446 lines
59 KiB
Bash
Executable File

#!/usr/bin/env bash
# SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
# Run an e2e command against a Helm-deployed OpenShell gateway in Kubernetes.
#
# Modes:
# - OPENSHELL_E2E_KUBE_CONTEXT set:
# Target the named kubectl context, install the chart into an ephemeral
# namespace, and port-forward the gateway. Cluster lifecycle is the
# caller's responsibility (e.g. CI provisions kind via helm/kind-action).
# - OPENSHELL_E2E_KUBE_CONTEXT unset:
# Create a local k3d cluster via tasks/scripts/helm-k3s-local.sh, install
# the chart, port-forward, and tear the cluster down on exit.
#
# On vanilla Kubernetes, Helm e2e talks to the gateway in plaintext over
# `kubectl port-forward` (ci/values-skaffold.yaml). The certgen hook still runs
# so the gateway has sandbox JWT signing keys.
#
# On OpenShift, that port-forward is too slow for `sandbox connect`. That command
# opens an SSH session to the gateway, and SSH needs many small back-and-forth
# messages to set up. Each one has to travel through the port-forward tunnel, so
# the connection never finishes and the test times out. To avoid this, on
# OpenShift the harness reaches the gateway through a normal network path: a
# passthrough OpenShift Route, secured with mandatory mTLS
# (ci/values-openshift-e2e.yaml).
#
# Every OpenShift-specific branch below is gated on OPENSHIFT_DETECTED, so the
# vanilla-Kubernetes path stays exactly the same.
#
# Set OPENSHELL_E2E_KUBE_EXTRA_VALUES to one or more colon-separated Helm values
# files, relative to the repository root or absolute, to layer additional chart
# configuration on top of ci/values-skaffold.yaml.
#
# Image source:
# - Ephemeral k3d mode builds local
# `openshell/{gateway,sandbox,supervisor}:${IMAGE_TAG}`
# images by default, imports them into k3d, then installs the chart. This
# mirrors the Skaffold local-dev path.
# - Existing-context mode pulls from
# ${OPENSHELL_REGISTRY}/{gateway,sandbox,supervisor}:${IMAGE_TAG}
# (defaults: ghcr.io/nvidia/openshell, latest). CI sets IMAGE_TAG to the
# commit SHA and preloads or publishes the images before running this script.
#
# Database backend scenarios:
# Set OPENSHELL_E2E_KUBE_DB_SCENARIOS=1 to run the test command against
# the supported database configurations: SQLite and external PostgreSQL
# with an existing Secret. When unset, the default single-install behavior
# is unchanged.
#
# External PostgreSQL fixture:
# Set OPENSHELL_E2E_KUBE_EXTERNAL_POSTGRES_SECRET to create an ephemeral
# PostgreSQL Deployment and a matching Secret with a `uri` key before
# installing OpenShell. This is used by HA CI so the gateway can run multiple
# replicas without requiring the OpenShell chart to own a database.
#
# Credential-driver fixture:
# Set OPENSHELL_E2E_CREDENTIAL_DRIVERS=1 to enable one credential storage
# backend. Set OPENSHELL_E2E_CREDENTIAL_DRIVER to `kubernetes-secrets` or
# `vault`; the Rust `credential_drivers` e2e test validates the active
# backend. Vault mode installs a dev OpenBao fixture because it exposes the
# Vault-compatible API used by the driver.
set -euo pipefail
if [ "$#" -eq 0 ]; then
echo "Usage: e2e/with-kube-gateway.sh <command> [args...]" >&2
exit 2
fi
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
# shellcheck source=e2e/support/gateway-common.sh
source "${ROOT}/e2e/support/gateway-common.sh"
# Upstream agent-sandbox release. The Kubernetes driver supports the v1beta1
# Sandbox API introduced in v0.5.0 and falls back to v1alpha1 for v0.4.6
# clusters. Override this env var to exercise the v1alpha1 controller release.
AGENT_SANDBOX_VERSION="${AGENT_SANDBOX_VERSION:-v1.0.3}"
e2e_preserve_mise_dirs
e2e_align_docker_host_with_cli_context
WORKDIR_PARENT="${TMPDIR:-/tmp}"
WORKDIR_PARENT="${WORKDIR_PARENT%/}"
WORKDIR="$(mktemp -d "${WORKDIR_PARENT}/openshell-e2e-kube.XXXXXX")"
CLUSTER_CREATED_BY_US=0
CLUSTER_NAME=""
KUBE_CONTEXT=""
NAMESPACE="openshell"
RELEASE_NAME="openshell"
PORTFORWARD_PID=""
PORTFORWARD_LOG="${WORKDIR}/portforward.log"
PORTFORWARD_HEALTH_PID=""
PORTFORWARD_HEALTH_LOG="${WORKDIR}/portforward-health.log"
HELM_INSTALLED=0
EXTERNAL_PG_FIXTURE_DEPLOYED=0
EXTERNAL_PG_FIXTURE_SECRET=""
EXTERNAL_PG_FIXTURE_MANIFEST="${ROOT}/e2e/kubernetes/postgres-fixture.yaml"
EXTERNAL_PG_FIXTURE_SERVICE="openshell-e2e-postgres"
EXTERNAL_PG_FIXTURE_USER="openshell"
EXTERNAL_PG_FIXTURE_PASSWORD="openshell-e2e-postgres"
EXTERNAL_PG_FIXTURE_DATABASE="openshell"
ENVOY_RELEASE_NAME="${OPENSHELL_E2E_ENVOY_RELEASE_NAME:-envoy-gateway}"
ENVOY_NAMESPACE="${OPENSHELL_E2E_ENVOY_NAMESPACE:-envoy-gateway-system}"
ENVOY_CHART_VERSION="${OPENSHELL_E2E_ENVOY_VERSION:-v1.7.2}"
ENVOY_GATEWAY_MANIFEST="${ROOT}/deploy/kube/manifests/envoy-gateway-openshell.yaml"
ENVOY_HELM_INSTALLED=0
ENVOY_GATEWAY_CONFIG_APPLIED=0
VAULT_FIXTURE_DEPLOYED=0
VAULT_NAMESPACE="${OPENSHELL_E2E_VAULT_NAMESPACE:-openbao}"
VAULT_RELEASE_NAME="${OPENSHELL_E2E_VAULT_RELEASE_NAME:-openbao}"
VAULT_CHART_VERSION="${OPENSHELL_E2E_OPENBAO_CHART_VERSION:-0.28.3}"
VAULT_DEV_ROOT_TOKEN="${OPENSHELL_E2E_VAULT_DEV_ROOT_TOKEN:-root}"
VAULT_CA_CONFIG_MAP="openbao-ca"
VAULT_DNS_ALIAS="${VAULT_RELEASE_NAME}-0"
VAULT_CA_FILE="${WORKDIR}/openbao-ca.crt"
CORPORATE_PROXY_FIXTURE_DEPLOYED=0
CORPORATE_PROXY_FIXTURE_SECRET="openshell-e2e-proxy-auth"
CORPORATE_PROXY_FIXTURE_CA_CONFIGMAP="openshell-e2e-proxy-ca"
CORPORATE_PROXY_CA_FIXTURE_DEPLOYED=0
OPENSHIFT_DETECTED=0
OPENSHIFT_SANDBOX_SCC_GRANTED=0
OPENSHIFT_POSTGRES_SCC_GRANTED=0
OPENSHIFT_ROUTE_HOST=""
# Temp dir holding the client mTLS material extracted from openshell-client-tls
# for the OpenShift Route transport. Removed by cleanup().
OPENSHIFT_PKI_DIR="${WORKDIR}/openshift-pki"
# Isolate CLI/SDK gateway metadata from the developer's real config.
export XDG_CONFIG_HOME="${WORKDIR}/config"
export XDG_DATA_HOME="${WORKDIR}/data"
kctl() {
kubectl --context "${KUBE_CONTEXT}" "$@"
}
helmctl() {
helm --kube-context "${KUBE_CONTEXT}" "$@"
}
# Return the resource reference for the gateway workload installed by the chart.
# SQLite releases use a StatefulSet; external-database releases may use a Deployment.
kube_workload_ref() {
local name="$1"
local namespace="${2:-${NAMESPACE}}"
local resource
for resource in "statefulset/${name}" "deployment/${name}"; do
if kctl -n "${namespace}" get "${resource}" >/dev/null 2>&1; then
printf '%s\n' "${resource}"
return 0
fi
done
echo "ERROR: gateway workload ${name} was not found in namespace ${namespace}" >&2
return 1
}
deploy_postgres_fixture() {
local secret_name="$1"
local pg_uri
echo "Deploying external PostgreSQL fixture ${EXTERNAL_PG_FIXTURE_SERVICE}..."
if ! kctl get namespace "${NAMESPACE}" >/dev/null 2>&1; then
kctl create namespace "${NAMESPACE}"
fi
if [ "${OPENSHIFT_DETECTED}" = "1" ]; then
echo "Granting anyuid SCC to ${EXTERNAL_PG_FIXTURE_SERVICE} for OpenShift..."
oc adm policy add-scc-to-user anyuid \
--context "${KUBE_CONTEXT}" \
-z "${EXTERNAL_PG_FIXTURE_SERVICE}" -n "${NAMESPACE}"
# Record the grant before applying the fixture so cleanup revokes it even if
# the apply below fails and EXTERNAL_PG_FIXTURE_DEPLOYED is never set.
OPENSHIFT_POSTGRES_SCC_GRANTED=1
fi
kctl -n "${NAMESPACE}" apply -f "${EXTERNAL_PG_FIXTURE_MANIFEST}"
EXTERNAL_PG_FIXTURE_DEPLOYED=1
EXTERNAL_PG_FIXTURE_SECRET="${secret_name}"
kctl -n "${NAMESPACE}" rollout status "deployment/${EXTERNAL_PG_FIXTURE_SERVICE}" --timeout=120s
pg_uri="postgresql://${EXTERNAL_PG_FIXTURE_USER}:${EXTERNAL_PG_FIXTURE_PASSWORD}@${EXTERNAL_PG_FIXTURE_SERVICE}.${NAMESPACE}.svc.cluster.local:5432/${EXTERNAL_PG_FIXTURE_DATABASE}"
kctl -n "${NAMESPACE}" delete secret "${secret_name}" \
--ignore-not-found >/dev/null 2>&1 || true
kctl -n "${NAMESPACE}" create secret generic "${secret_name}" \
--from-literal=uri="${pg_uri}"
}
use_envoy_gateway() {
case "${OPENSHELL_E2E_KUBE_USE_ENVOY:-0}" in
1 | true | TRUE | yes | YES) return 0 ;;
*) return 1 ;;
esac
}
install_envoy_gateway() {
echo "Installing Envoy Gateway (${ENVOY_CHART_VERSION})..."
helmctl upgrade --install "${ENVOY_RELEASE_NAME}" \
oci://docker.io/envoyproxy/gateway-helm \
--version "${ENVOY_CHART_VERSION}" \
--namespace "${ENVOY_NAMESPACE}" --create-namespace \
--wait --timeout 5m
ENVOY_HELM_INSTALLED=1
if ! kctl get namespace "${NAMESPACE}" >/dev/null 2>&1; then
kctl create namespace "${NAMESPACE}"
fi
kctl apply -f "${ENVOY_GATEWAY_MANIFEST}"
ENVOY_GATEWAY_CONFIG_APPLIED=1
}
wait_for_envoy_service() {
local svc_ref=""
local svc_namespace=""
for _ in $(seq 1 60); do
svc_ref="$(kctl get svc -A \
-l "gateway.envoyproxy.io/owning-gateway-name=${RELEASE_NAME},gateway.envoyproxy.io/owning-gateway-namespace=${NAMESPACE}" \
-o jsonpath='{range .items[0]}{.metadata.namespace}{"/"}{.metadata.name}{end}' \
2>/dev/null || true)"
if [ -n "${svc_ref}" ]; then
svc_namespace="${svc_ref%%/*}"
if kctl -n "${svc_namespace}" wait --for=condition=Ready pod \
-l "gateway.envoyproxy.io/owning-gateway-name=${RELEASE_NAME},gateway.envoyproxy.io/owning-gateway-namespace=${NAMESPACE}" \
--timeout=5s >/dev/null 2>&1; then
printf '%s\n' "${svc_ref}"
return 0
fi
fi
sleep 2
done
echo "ERROR: Envoy proxy Service for Gateway ${RELEASE_NAME} was not ready." >&2
kctl -n "${NAMESPACE}" get gateway,grpcroute -o wide >&2 || true
kctl get svc -A \
-l "gateway.envoyproxy.io/owning-gateway-name=${RELEASE_NAME},gateway.envoyproxy.io/owning-gateway-namespace=${NAMESPACE}" \
-o wide >&2 || true
kctl get pods -A \
-l "gateway.envoyproxy.io/owning-gateway-name=${RELEASE_NAME},gateway.envoyproxy.io/owning-gateway-namespace=${NAMESPACE}" \
-o wide >&2 || true
return 1
}
start_gateway_portforward() {
local elapsed=0
local pf_timeout=30
local target_port=8080
local target_namespace="${NAMESPACE}"
local target_service="${RELEASE_NAME}"
local target_service_ref=""
LOCAL_PORT="$(e2e_pick_port)"
if use_envoy_gateway; then
target_service_ref="$(wait_for_envoy_service)"
target_namespace="${target_service_ref%%/*}"
target_service="${target_service_ref#*/}"
target_port=80
echo "Starting kubectl port-forward -n ${target_namespace} svc/${target_service} ${LOCAL_PORT}:${target_port} (Envoy Gateway)..."
else
echo "Starting kubectl port-forward svc/${target_service} ${LOCAL_PORT}:${target_port}..."
fi
kctl -n "${target_namespace}" port-forward "svc/${target_service}" \
"${LOCAL_PORT}:${target_port}" >"${PORTFORWARD_LOG}" 2>&1 &
PORTFORWARD_PID=$!
while [ "${elapsed}" -lt "${pf_timeout}" ]; do
if ! kill -0 "${PORTFORWARD_PID}" 2>/dev/null; then
echo "ERROR: kubectl port-forward exited before becoming reachable" >&2
cat "${PORTFORWARD_LOG}" >&2 || true
return 1
fi
if curl -s -o /dev/null --connect-timeout 1 "http://127.0.0.1:${LOCAL_PORT}"; then
return 0
fi
sleep 1
elapsed=$((elapsed + 1))
done
echo "ERROR: port-forward did not accept TCP within ${pf_timeout}s" >&2
cat "${PORTFORWARD_LOG}" >&2 || true
return 1
}
stop_gateway_portforward() {
local pid
local pid_var
for pid_var in PORTFORWARD_PID PORTFORWARD_HEALTH_PID; do
pid="${!pid_var}"
[ -n "${pid}" ] || continue
kill "${pid}" >/dev/null 2>&1 || true
for _ in $(seq 1 10); do
if ! kill -0 "${pid}" >/dev/null 2>&1; then
break
fi
sleep 0.5
done
kill -KILL "${pid}" >/dev/null 2>&1 || true
wait "${pid}" >/dev/null 2>&1 || true
printf -v "${pid_var}" '%s' ""
done
}
cleanup_postgres_fixture() {
local secret_name="$1"
[ -n "${KUBE_CONTEXT}" ] || return 0
[ -n "${NAMESPACE}" ] || return 0
kctl -n "${NAMESPACE}" delete -f "${EXTERNAL_PG_FIXTURE_MANIFEST}" \
--ignore-not-found >/dev/null 2>&1 || true
kctl -n "${NAMESPACE}" delete secret "${secret_name}" \
--ignore-not-found >/dev/null 2>&1 || true
if [ "${OPENSHIFT_POSTGRES_SCC_GRANTED}" = "1" ]; then
oc adm policy remove-scc-from-user anyuid \
--context "${KUBE_CONTEXT}" \
-z "${EXTERNAL_PG_FIXTURE_SERVICE}" -n "${NAMESPACE}" \
2>/dev/null || true
OPENSHIFT_POSTGRES_SCC_GRANTED=0
fi
EXTERNAL_PG_FIXTURE_DEPLOYED=0
EXTERNAL_PG_FIXTURE_SECRET=""
}
deploy_vault_fixture() {
echo "Deploying OpenBao fixture for Vault credential-driver validation..."
local openshift_flag="false"
if [ "${OPENSHIFT_DETECTED}" = "1" ]; then
echo "Enabling OpenBao chart OpenShift mode for restricted-v2 compatibility."
openshift_flag="true"
fi
helmctl repo add openbao https://openbao.github.io/openbao-helm \
>/dev/null 2>&1 || true
helmctl repo update openbao >/dev/null
helmctl upgrade --install "${VAULT_RELEASE_NAME}" openbao/openbao \
--namespace "${VAULT_NAMESPACE}" --create-namespace \
--version "${VAULT_CHART_VERSION}" \
--values "${ROOT}/e2e/kubernetes/openbao-tls-values.yaml" \
--set "server.dev.enabled=true" \
--set "server.dev.devRootToken=${VAULT_DEV_ROOT_TOKEN}" \
--set "injector.enabled=false" \
--set "global.openshift=${openshift_flag}" \
--wait --timeout 5m
VAULT_FIXTURE_DEPLOYED=1
kctl -n "${VAULT_NAMESPACE}" wait \
--for=condition=Ready pod \
-l "app.kubernetes.io/name=openbao,component=server" \
--timeout=300s
kctl -n "${VAULT_NAMESPACE}" exec "${VAULT_RELEASE_NAME}-0" -- \
cat /openbao/tls/vault-ca.pem >"${VAULT_CA_FILE}"
kctl create namespace "${NAMESPACE}" --dry-run=client -o yaml | kctl apply -f -
kctl -n "${NAMESPACE}" create configmap "${VAULT_CA_CONFIG_MAP}" \
--from-file="ca.crt=${VAULT_CA_FILE}" --dry-run=client -o yaml | kctl apply -f -
kctl -n "${NAMESPACE}" create service externalname "${VAULT_DNS_ALIAS}" \
--external-name="${VAULT_RELEASE_NAME}-0.${VAULT_RELEASE_NAME}-internal.${VAULT_NAMESPACE}.svc.cluster.local" \
--dry-run=client -o yaml | kctl apply -f -
provision_vault_auth
export OPENSHELL_E2E_VAULT_NAMESPACE="${VAULT_NAMESPACE}"
export OPENSHELL_E2E_VAULT_POD="${VAULT_RELEASE_NAME}-0"
export OPENSHELL_E2E_VAULT_TOKEN="${VAULT_DEV_ROOT_TOKEN}"
}
# Run a `bao` command in the fixture pod. Tolerates the "path is already in use"
# error from re-enabling a mount on rerun, but surfaces any other failure.
openbao_exec() {
local out
if out="$(kctl -n "${VAULT_NAMESPACE}" exec "${VAULT_RELEASE_NAME}-0" -- \
env "BAO_TOKEN=${VAULT_DEV_ROOT_TOKEN}" bao "$@" 2>&1)"; then
[ -n "${out}" ] && printf '%s\n' "${out}"
return 0
fi
case "${out}" in
*"path is already in use"*) return 0 ;;
*) printf '%s\n' "${out}" >&2; return 1 ;;
esac
}
# Provision the KV store, Kubernetes auth method, storage policy, and login role
# the gateway's Vault credential driver uses, so every provider-creating test in
# the suite can authenticate. The role binds ServiceAccount `openshell` in the
# gateway namespace, matching ci/values-credential-driver-vault.yaml.
provision_vault_auth() {
echo "Provisioning OpenBao Kubernetes auth for the gateway service account..."
openbao_exec secrets enable -path=secret kv-v2 >/dev/null
openbao_exec auth enable kubernetes >/dev/null
openbao_exec write auth/kubernetes/config \
kubernetes_host=https://kubernetes.default.svc \
kubernetes_ca_cert=@/var/run/secrets/kubernetes.io/serviceaccount/ca.crt \
>/dev/null
printf '%s\n' \
'path "secret/data/openshell/provider-credentials/*" {' \
' capabilities = ["create", "read", "update", "delete"]' \
'}' \
'path "secret/metadata/openshell/provider-credentials/*" {' \
' capabilities = ["read", "delete", "list"]' \
'}' \
| kctl -n "${VAULT_NAMESPACE}" exec -i "${VAULT_RELEASE_NAME}-0" -- \
env "BAO_TOKEN=${VAULT_DEV_ROOT_TOKEN}" \
bao policy write openshell-provider-storage - >/dev/null
openbao_exec write auth/kubernetes/role/openshell-gateway \
bound_service_account_names=openshell \
"bound_service_account_namespaces=${NAMESPACE}" \
policies=openshell-provider-storage \
ttl=1h >/dev/null
}
cleanup_vault_fixture() {
[ -n "${KUBE_CONTEXT}" ] || return 0
[ -n "${VAULT_NAMESPACE}" ] || return 0
kctl -n "${NAMESPACE}" delete service "${VAULT_DNS_ALIAS}" \
--ignore-not-found >/dev/null 2>&1 || true
kctl -n "${NAMESPACE}" delete configmap "${VAULT_CA_CONFIG_MAP}" \
--ignore-not-found >/dev/null 2>&1 || true
if command -v helm >/dev/null 2>&1; then
helmctl uninstall "${VAULT_RELEASE_NAME}" \
--namespace "${VAULT_NAMESPACE}" --wait --timeout 60s \
>/dev/null 2>&1 || true
fi
if command -v kubectl >/dev/null 2>&1; then
kctl delete namespace "${VAULT_NAMESPACE}" --wait=true --timeout=60s \
--ignore-not-found >/dev/null 2>&1 || true
fi
VAULT_FIXTURE_DEPLOYED=0
}
cleanup() {
local exit_code=$?
stop_gateway_portforward
if [ "${exit_code}" -ne 0 ] && [ -n "${KUBE_CONTEXT}" ] && [ -n "${NAMESPACE}" ]; then
if command -v kubectl >/dev/null 2>&1 \
&& kctl get namespace "${NAMESPACE}" >/dev/null 2>&1; then
echo "=== gateway pod state (preserved for debugging) ==="
kctl -n "${NAMESPACE}" get pods -o wide 2>&1 || true
echo "=== Agent Sandbox resources ==="
kctl -n "${NAMESPACE}" get sandboxes.agents.x-k8s.io -o yaml 2>&1 || true
echo "=== gateway sandbox records ==="
"${OPENSHELL_BIN:-${ROOT}/target/debug/openshell}" \
sandbox list --all-workspaces --output json 2>&1 || true
echo "=== sandbox-runtime supervisor Pods ==="
kctl -n "${NAMESPACE}" get pods \
-l "openshell.ai/boundary-role=supervisor" -o yaml 2>&1 || true
echo "=== sandbox-runtime supervisor logs (last 200 lines each) ==="
while IFS= read -r supervisor_pod; do
[ -n "${supervisor_pod}" ] || continue
echo "--- ${supervisor_pod} ---"
kctl -n "${NAMESPACE}" logs "${supervisor_pod}" \
--all-containers --prefix --tail=200 2>&1 || true
echo "--- ${supervisor_pod} (previous containers) ---"
kctl -n "${NAMESPACE}" logs "${supervisor_pod}" --previous \
--all-containers --prefix --tail=200 2>&1 || true
done < <(kctl -n "${NAMESPACE}" get pods \
-l "openshell.ai/boundary-role=supervisor" -o name 2>/dev/null || true)
echo "=== gateway events ==="
kctl -n "${NAMESPACE}" get events --sort-by=.lastTimestamp 2>&1 \
| tail -n 80 || true
echo "=== gateway lifecycle and supervisor-session logs ==="
kctl -n "${NAMESPACE}" logs "$(kube_workload_ref "${RELEASE_NAME}")" \
--since=20m \
--all-containers --prefix 2>&1 \
| grep -Ei "sandbox phase changed|start_sandbox|stop_sandbox|supervisor session|sandbox-runtime|bootstrap" \
|| true
echo "=== gateway logs (last 200 lines) ==="
kctl -n "${NAMESPACE}" logs \
-l "app.kubernetes.io/instance=${RELEASE_NAME}" --tail=200 \
--all-containers --prefix 2>&1 || true
echo "=== end gateway debug output ==="
fi
if [ -f "${PORTFORWARD_LOG}" ]; then
echo "=== port-forward log ==="
cat "${PORTFORWARD_LOG}" || true
echo "=== end port-forward log ==="
fi
if [ -f "${PORTFORWARD_HEALTH_LOG}" ]; then
echo "=== health port-forward log ==="
cat "${PORTFORWARD_HEALTH_LOG}" || true
echo "=== end health port-forward log ==="
fi
fi
if [ "${EXTERNAL_PG_FIXTURE_DEPLOYED}" = "1" ] \
|| [ "${OPENSHIFT_POSTGRES_SCC_GRANTED}" = "1" ]; then
cleanup_postgres_fixture "${EXTERNAL_PG_FIXTURE_SECRET}"
fi
if [ "${VAULT_FIXTURE_DEPLOYED}" = "1" ]; then
cleanup_vault_fixture
fi
if [ "${ENVOY_GATEWAY_CONFIG_APPLIED}" = "1" ] && [ -n "${KUBE_CONTEXT}" ]; then
if command -v kubectl >/dev/null 2>&1; then
kctl -n "${NAMESPACE}" delete backendtrafficpolicy.gateway.envoyproxy.io \
openshell-grpc-timeouts --ignore-not-found --wait=false \
>/dev/null 2>&1 || true
kctl delete gatewayclass.gateway.networking.k8s.io eg \
--ignore-not-found --wait=false >/dev/null 2>&1 || true
fi
ENVOY_GATEWAY_CONFIG_APPLIED=0
fi
if [ "${CORPORATE_PROXY_FIXTURE_DEPLOYED}" = "1" ]; then
kctl -n "${NAMESPACE}" delete secret "${CORPORATE_PROXY_FIXTURE_SECRET}" \
--ignore-not-found >/dev/null 2>&1 || true
fi
if [ "${CORPORATE_PROXY_CA_FIXTURE_DEPLOYED}" = "1" ]; then
kctl -n "${NAMESPACE}" delete configmap "${CORPORATE_PROXY_FIXTURE_CA_CONFIGMAP}" \
--ignore-not-found >/dev/null 2>&1 || true
fi
if [ "${OPENSHIFT_SANDBOX_SCC_GRANTED}" = "1" ]; then
oc adm policy remove-scc-from-user privileged \
--context "${KUBE_CONTEXT}" \
-z openshell-sandbox -n "${NAMESPACE}" \
2>/dev/null || true
OPENSHIFT_SANDBOX_SCC_GRANTED=0
fi
# Remove the extracted client mTLS material (also covered by the WORKDIR sweep
# below, but drop the private key promptly and explicitly).
if [ -n "${OPENSHIFT_PKI_DIR}" ]; then
rm -rf "${OPENSHIFT_PKI_DIR}" 2>/dev/null || true
fi
# Sweep managed-mode and operator-mode workspace namespaces before
# uninstalling the Helm release (ClusterRole still needed for deletion).
if command -v kubectl >/dev/null 2>&1 && [ -n "${KUBE_CONTEXT}" ]; then
for label in "openshell.ai/managed-by=openshell" \
"openshell.ai/e2e-operator-workspace=true"; do
ns_list="$(kctl get namespaces -l "${label}" -o name 2>/dev/null || true)"
if [ -n "${ns_list}" ]; then
echo "Cleaning up namespaces with label ${label}..."
echo "${ns_list}" | while read -r ns_ref; do
kctl delete "${ns_ref}" --wait=false --ignore-not-found \
2>/dev/null || true
done
fi
done
fi
if [ "${HELM_INSTALLED}" = "1" ] && [ -n "${KUBE_CONTEXT}" ] && [ -n "${NAMESPACE}" ]; then
if command -v helm >/dev/null 2>&1; then
helmctl uninstall "${RELEASE_NAME}" --namespace "${NAMESPACE}" --wait \
--timeout 60s >/dev/null 2>&1 || true
fi
if command -v kubectl >/dev/null 2>&1; then
# Wait for the namespace to fully delete so back-to-back runs don't hit
# "namespace is being terminated" when helm install creates it again.
kctl delete namespace "${NAMESPACE}" --wait=true --timeout=60s \
--ignore-not-found >/dev/null 2>&1 || true
fi
fi
if [ "${ENVOY_HELM_INSTALLED}" = "1" ] && [ -n "${KUBE_CONTEXT}" ]; then
if command -v helm >/dev/null 2>&1; then
helmctl uninstall "${ENVOY_RELEASE_NAME}" --namespace "${ENVOY_NAMESPACE}" \
--wait --timeout 60s >/dev/null 2>&1 || true
fi
if command -v kubectl >/dev/null 2>&1; then
kctl delete namespace "${ENVOY_NAMESPACE}" --wait=true --timeout=60s \
--ignore-not-found >/dev/null 2>&1 || true
fi
ENVOY_HELM_INSTALLED=0
fi
if [ "${CLUSTER_CREATED_BY_US}" = "1" ] && [ -n "${CLUSTER_NAME}" ]; then
if command -v k3d >/dev/null 2>&1 && k3d cluster list "${CLUSTER_NAME}" \
>/dev/null 2>&1; then
echo "Deleting ephemeral k3d cluster ${CLUSTER_NAME}..."
k3d cluster delete "${CLUSTER_NAME}" >/dev/null 2>&1 || true
fi
fi
rm -rf "${WORKDIR}" 2>/dev/null || true
}
trap cleanup EXIT
# --- DB-scenario helpers (used only when OPENSHELL_E2E_KUBE_DB_SCENARIOS=1) ---
scenario_stop_portforward() {
stop_gateway_portforward
}
scenario_cleanup_release() {
helmctl uninstall "${RELEASE_NAME}" --namespace "${NAMESPACE}" --wait \
--timeout 120s 2>/dev/null || true
HELM_INSTALLED=0
for _ in $(seq 1 30); do
remaining="$(kctl get pods -n "${NAMESPACE}" \
-l "app.kubernetes.io/instance=${RELEASE_NAME}" --no-headers 2>/dev/null || true)"
if [ -z "${remaining}" ]; then
break
fi
sleep 2
done
kctl delete pvc -n "${NAMESPACE}" \
-l "app.kubernetes.io/instance=${RELEASE_NAME}" --wait=false 2>/dev/null || true
}
scenario_record_failure() {
local scenario_label="$1"
local reason="$2"
DB_FAILED=$((DB_FAILED + 1))
DB_SCENARIOS_SUMMARY+=("FAIL ${scenario_label}: ${reason}")
scenario_stop_portforward
scenario_cleanup_release
}
scenario_deploy_external_pg() {
echo "==> Deploying standalone PostgreSQL as external database..."
deploy_postgres_fixture my-pg-credentials
}
scenario_cleanup_external_pg() {
echo "==> Cleaning up external PostgreSQL..."
cleanup_postgres_fixture my-pg-credentials
}
# Run a single DB-backend scenario: install chart → port-forward → run tests → cleanup.
# Usage: run_scenario "label" "type" [extra --set flags...]
# type: sqlite | external-pg
run_scenario() {
local scenario_label="$1"
shift 2
local scenario_exit=0
echo ""
echo "========================================"
echo "==> Scenario: ${scenario_label}"
echo "========================================"
helmctl install "${RELEASE_NAME}" "${ROOT}/deploy/helm/openshell" \
--namespace "${NAMESPACE}" --create-namespace \
"${helm_values_args[@]}" \
--set "fullnameOverride=openshell" \
"${GLOBAL_HELM_IMAGE_ARGS[@]}" \
"${GATEWAY_HELM_IMAGE_ARGS[@]}" \
"${SUPERVISOR_HELM_IMAGE_ARGS[@]}" \
"${SANDBOX_RUNTIME_HELM_IMAGE_ARGS[@]}" \
"${helm_post_renderer_args[@]}" \
"$@" \
--wait --timeout 5m
HELM_INSTALLED=1
if [ "${OPENSHIFT_DETECTED}" = "1" ]; then
# OpenShift: reach the gateway over the passthrough Route with mTLS instead
# of port-forward (which stalls the SSH-relay connect suites).
if ! openshift_register_route_gateway; then
scenario_record_failure "${scenario_label}" "Route/mTLS setup failed"
return
fi
else
# Vanilla Kubernetes: reach the gateway in plaintext over port-forward.
if ! start_gateway_portforward; then
scenario_record_failure "${scenario_label}" "port-forward failed"
return
fi
GATEWAY_NAME="openshell-e2e-kube-${LOCAL_PORT}"
GATEWAY_ENDPOINT="http://127.0.0.1:${LOCAL_PORT}"
e2e_register_plaintext_gateway \
"${XDG_CONFIG_HOME}" \
"${GATEWAY_NAME}" \
"${GATEWAY_ENDPOINT}" \
"${LOCAL_PORT}"
fi
if ! start_health_portforward; then
scenario_record_failure "${scenario_label}" "health port-forward failed"
return
fi
export OPENSHELL_GATEWAY="${GATEWAY_NAME}"
export OPENSHELL_E2E_DRIVER="kubernetes"
# Kubernetes e2e runs against k3d/kind-style Docker-backed clusters. Host
# fixture containers must use the same Docker host so published ports and
# cluster host-gateway aliases line up even on machines where Podman is also
# installed.
export CONTAINER_ENGINE="${CONTAINER_ENGINE:-docker}"
export OPENSHELL_E2E_KUBE_CONTEXT_ACTIVE="${KUBE_CONTEXT}"
export OPENSHELL_E2E_SANDBOX_NAMESPACE="${NAMESPACE}"
export OPENSHELL_E2E_KUBE_CONTEXT="${KUBE_CONTEXT}"
export OPENSHELL_E2E_KUBE_NAMESPACE="${NAMESPACE}"
export OPENSHELL_E2E_KUBE_RELEASE="${RELEASE_NAME}"
export OPENSHELL_PROVISION_TIMEOUT="${OPENSHELL_PROVISION_TIMEOUT:-300}"
e2e_import_example_provider_profiles \
"${OPENSHELL_BIN:-${ROOT}/target/debug/openshell}" "${ROOT}" || return 1
echo "Running e2e command against ${GATEWAY_ENDPOINT}: ${E2E_CMD[*]}"
"${E2E_CMD[@]}" || scenario_exit=$?
scenario_stop_portforward
scenario_cleanup_release
if [ "${scenario_exit}" -eq 0 ]; then
echo "==> PASS: ${scenario_label}"
DB_PASSED=$((DB_PASSED + 1))
DB_SCENARIOS_SUMMARY+=("PASS ${scenario_label}")
else
echo "==> FAIL: ${scenario_label} (exit code ${scenario_exit})"
DB_FAILED=$((DB_FAILED + 1))
DB_SCENARIOS_SUMMARY+=("FAIL ${scenario_label}: exit code ${scenario_exit}")
fi
}
# --- end DB-scenario helpers ---
require_cmd() {
if ! command -v "$1" >/dev/null 2>&1; then
echo "ERROR: $1 is required to run Helm-backed e2e tests" >&2
exit 2
fi
}
configure_fixture_container_engine() {
[ -n "${CONTAINER_ENGINE:-}" ] || return 0
local selected_engine
selected_engine="$(printf '%s' "${CONTAINER_ENGINE}" | tr '[:upper:]' '[:lower:]')"
case "${selected_engine}" in
docker|podman)
;;
*)
echo "ERROR: CONTAINER_ENGINE=${CONTAINER_ENGINE} is invalid; expected docker or podman" >&2
exit 2
;;
esac
export CONTAINER_ENGINE="${selected_engine}"
}
# OpenShift only: extract the client mTLS material, wait for the passthrough
# Route to serve mTLS, assert that a certless caller is rejected at the TLS
# handshake, and register an mTLS CLI gateway pointing at the Route.
#
# Sets GATEWAY_NAME and GATEWAY_ENDPOINT on success. Returns non-zero on failure
# (unreachable Route or a certless request that was NOT rejected — a security
# hole). Reads OPENSHIFT_ROUTE_HOST and OPENSHIFT_PKI_DIR.
openshift_register_route_gateway() {
local pki_dir="${OPENSHIFT_PKI_DIR}"
rm -rf "${pki_dir}"
mkdir -p "${pki_dir}/client"
echo "Extracting client mTLS material from secret openshell-client-tls..."
kctl -n "${NAMESPACE}" get secret openshell-client-tls \
-o jsonpath='{.data.ca\.crt}' | base64 -d >"${pki_dir}/ca.crt"
kctl -n "${NAMESPACE}" get secret openshell-client-tls \
-o jsonpath='{.data.tls\.crt}' | base64 -d >"${pki_dir}/client/tls.crt"
kctl -n "${NAMESPACE}" get secret openshell-client-tls \
-o jsonpath='{.data.tls\.key}' | base64 -d >"${pki_dir}/client/tls.key"
# Wait until an mTLS request to the Route completes the TLS handshake. Helm
# --wait already made the gateway pod Ready; this only covers the short window
# while the OpenShift router loads the new Route.
echo "Waiting for Route https://${OPENSHIFT_ROUTE_HOST} to serve mTLS..."
local elapsed=0 timeout=180
while [ "${elapsed}" -lt "${timeout}" ]; do
if curl -s --max-time 10 -o /dev/null \
--cacert "${pki_dir}/ca.crt" \
--cert "${pki_dir}/client/tls.crt" \
--key "${pki_dir}/client/tls.key" \
"https://${OPENSHIFT_ROUTE_HOST}/"; then
break
fi
sleep 3
elapsed=$((elapsed + 3))
done
if [ "${elapsed}" -ge "${timeout}" ]; then
echo "ERROR: Route ${OPENSHIFT_ROUTE_HOST} did not serve mTLS within ${timeout}s" >&2
return 1
fi
# Security gate: a caller with no client certificate MUST be rejected during
# the TLS handshake (clientCaSecretName set + no OIDC => mTLS mandatory).
#
# Validate the server cert with --cacert (no -k) and inspect curl's exit code
# so an unrelated TLS/DNS/timeout failure is not silently accepted as "certless
# rejected". Only a handshake abort by the server (no client cert presented)
# counts as the expected rejection.
local certless_rc=0
curl -s --max-time 10 -o /dev/null \
--cacert "${pki_dir}/ca.crt" \
"https://${OPENSHIFT_ROUTE_HOST}/" || certless_rc=$?
case "${certless_rc}" in
0)
echo "ERROR: SECURITY HOLE — gateway accepted a certless request over the Route" >&2
return 1
;;
35 | 56)
# 35 CURLE_SSL_CONNECT_ERROR / 56 CURLE_RECV_ERROR: the server aborted the
# TLS handshake because no client certificate was presented — the expected
# mTLS rejection.
echo "OK: Route reachable over mTLS; certless request rejected at TLS (curl ${certless_rc})."
;;
*)
echo "ERROR: certless probe to ${OPENSHIFT_ROUTE_HOST} failed with curl exit ${certless_rc}, not a TLS client-auth rejection; cannot confirm mTLS is enforced" >&2
return 1
;;
esac
GATEWAY_NAME="openshell-e2e-openshift"
GATEWAY_ENDPOINT="https://${OPENSHIFT_ROUTE_HOST}"
e2e_register_mtls_gateway \
"${XDG_CONFIG_HOME}" \
"${GATEWAY_NAME}" \
"${GATEWAY_ENDPOINT}" \
"$(e2e_endpoint_port "${GATEWAY_ENDPOINT}")" \
"${pki_dir}"
}
# Start `kubectl port-forward svc/openshell` for the gRPC endpoint and wait for
# it to accept TCP. Sets LOCAL_PORT and PORTFORWARD_PID. Prints the port-forward
# log and returns non-zero on failure. Used for the vanilla-Kubernetes transport
# (the OpenShift transport uses openshift_register_route_gateway instead).
start_grpc_portforward() {
LOCAL_PORT="$(e2e_pick_port)"
echo "Starting kubectl port-forward svc/openshell ${LOCAL_PORT}:8080..."
kctl -n "${NAMESPACE}" port-forward "svc/openshell" \
"${LOCAL_PORT}:8080" >"${PORTFORWARD_LOG}" 2>&1 &
PORTFORWARD_PID=$!
local elapsed=0 timeout=30
while [ "${elapsed}" -lt "${timeout}" ]; do
if ! kill -0 "${PORTFORWARD_PID}" 2>/dev/null; then
echo "ERROR: kubectl port-forward exited before becoming reachable" >&2
cat "${PORTFORWARD_LOG}" >&2 || true
return 1
fi
if curl -s -o /dev/null --connect-timeout 1 "http://127.0.0.1:${LOCAL_PORT}"; then
return 0
fi
sleep 1
elapsed=$((elapsed + 1))
done
echo "ERROR: port-forward did not accept TCP within ${timeout}s" >&2
cat "${PORTFORWARD_LOG}" >&2 || true
return 1
}
# Start `kubectl port-forward` for the health endpoint and wait for /healthz.
# Sets HEALTH_LOCAL_PORT and PORTFORWARD_HEALTH_PID and exports
# OPENSHELL_E2E_HEALTH_PORT. Used on both cluster types: the OpenShift Route
# targets grpc only, and the health endpoint is not the SSH path so port-forward
# is fine for it. Prints the log and returns non-zero on failure.
start_health_portforward() {
HEALTH_LOCAL_PORT="$(e2e_pick_port)"
local workload_ref
workload_ref="$(kube_workload_ref "${RELEASE_NAME}")"
echo "Starting kubectl port-forward ${workload_ref} ${HEALTH_LOCAL_PORT}:health..."
kctl -n "${NAMESPACE}" port-forward "${workload_ref}" \
"${HEALTH_LOCAL_PORT}:health" >"${PORTFORWARD_HEALTH_LOG}" 2>&1 &
PORTFORWARD_HEALTH_PID=$!
local elapsed=0 timeout=30
while [ "${elapsed}" -lt "${timeout}" ]; do
if ! kill -0 "${PORTFORWARD_HEALTH_PID}" 2>/dev/null; then
echo "ERROR: kubectl health port-forward exited before becoming reachable" >&2
cat "${PORTFORWARD_HEALTH_LOG}" >&2 || true
return 1
fi
if curl -s -o /dev/null --connect-timeout 1 "http://127.0.0.1:${HEALTH_LOCAL_PORT}/healthz"; then
export OPENSHELL_E2E_HEALTH_PORT="${HEALTH_LOCAL_PORT}"
return 0
fi
sleep 1
elapsed=$((elapsed + 1))
done
echo "ERROR: health port-forward did not accept TCP within ${timeout}s" >&2
cat "${PORTFORWARD_HEALTH_LOG}" >&2 || true
return 1
}
require_cmd helm
require_cmd kubectl
require_cmd curl
if [ -n "${OPENSHELL_E2E_KUBE_CONTEXT:-}" ]; then
KUBE_CONTEXT="${OPENSHELL_E2E_KUBE_CONTEXT}"
echo "Using existing kubectl context: ${KUBE_CONTEXT}"
if ! kctl cluster-info >/dev/null 2>&1; then
echo "ERROR: kubectl context '${KUBE_CONTEXT}' is not reachable." >&2
exit 2
fi
else
if ! command -v k3d >/dev/null 2>&1; then
if [ "$(uname -s)" = "Linux" ]; then
echo "ERROR: k3d is not installed by mise on Linux in this repo." >&2
echo "Set OPENSHELL_E2E_KUBE_CONTEXT to a kind/existing cluster, or install k3d explicitly." >&2
exit 2
fi
require_cmd k3d
fi
CLUSTER_NAME="oshe2e-$$-$(date +%s | tail -c 8)"
echo "Creating ephemeral k3d cluster ${CLUSTER_NAME}..."
HELM_K3S_CLUSTER_NAME="${CLUSTER_NAME}" \
HELM_K3S_KUBECONFIG="${WORKDIR}/kubeconfig" \
bash "${ROOT}/tasks/scripts/helm-k3s-local.sh" create
CLUSTER_CREATED_BY_US=1
export KUBECONFIG="${WORKDIR}/kubeconfig"
KUBE_CONTEXT="k3d-${CLUSTER_NAME}"
fi
configure_fixture_container_engine
if [ -z "${OPENSHELL_E2E_KUBE_BUILD_IMAGES+x}" ]; then
if [ "${CLUSTER_CREATED_BY_US}" = "1" ]; then
OPENSHELL_E2E_KUBE_BUILD_IMAGES=1
else
OPENSHELL_E2E_KUBE_BUILD_IMAGES=0
fi
fi
reuse_sandbox_image=0
reuse_supervisor_image=0
if [ "${OPENSHELL_E2E_KUBE_BUILD_IMAGES}" = "1" ]; then
REGISTRY_VALUE="${OPENSHELL_REGISTRY:-openshell}"
IMAGE_TAG_VALUE="${IMAGE_TAG:-e2e-${CLUSTER_NAME:-local}}"
else
REGISTRY_VALUE="${OPENSHELL_REGISTRY:-ghcr.io/nvidia/openshell}"
IMAGE_TAG_VALUE="${IMAGE_TAG:-latest}"
fi
REGISTRY_VALUE="${REGISTRY_VALUE%/}"
GATEWAY_IMAGE="$(e2e_resolve_image_reference "${GATEWAY_IMAGE:-${REGISTRY_VALUE}/gateway}" "${IMAGE_TAG_VALUE}")"
SUPERVISOR_IMAGE="$(e2e_resolve_image_reference "${SUPERVISOR_IMAGE:-${REGISTRY_VALUE}/supervisor}" "${IMAGE_TAG_VALUE}")"
SANDBOX_RUNTIME_IMAGE="$(e2e_resolve_image_reference "${SANDBOX_IMAGE:-${REGISTRY_VALUE}/sandbox}" "${IMAGE_TAG_VALUE}")"
BUILD_GATEWAY_IMAGE="${REGISTRY_VALUE}/gateway:${IMAGE_TAG_VALUE}"
BUILD_SUPERVISOR_IMAGE="${REGISTRY_VALUE}/supervisor:${IMAGE_TAG_VALUE}"
# Each image carries its own registry; clear the chart's default so a
# registry-less local tag is not rewritten to ghcr.io.
GLOBAL_HELM_IMAGE_ARGS=(--set-string "global.image.registry=")
GATEWAY_HELM_IMAGE_ARGS=(--set-string "gateway.image.registry=$(e2e_image_reference_registry "${GATEWAY_IMAGE}")" --set-string "gateway.image.repository=$(e2e_image_reference_repository_path "${GATEWAY_IMAGE}")" --set-string "gateway.image.tag=$(e2e_image_reference_tag "${GATEWAY_IMAGE}")" --set-string "gateway.image.digest=$(e2e_image_reference_digest "${GATEWAY_IMAGE}")")
SUPERVISOR_HELM_IMAGE_ARGS=(--set-string "supervisor.image.registry=$(e2e_image_reference_registry "${SUPERVISOR_IMAGE}")" --set-string "supervisor.image.repository=$(e2e_image_reference_repository_path "${SUPERVISOR_IMAGE}")" --set-string "supervisor.image.tag=$(e2e_image_reference_tag "${SUPERVISOR_IMAGE}")" --set-string "supervisor.image.digest=$(e2e_image_reference_digest "${SUPERVISOR_IMAGE}")")
SANDBOX_RUNTIME_HELM_IMAGE_ARGS=(--set-string "sandboxRuntime.image.registry=$(e2e_image_reference_registry "${SANDBOX_RUNTIME_IMAGE}")" --set-string "sandboxRuntime.image.repository=$(e2e_image_reference_repository_path "${SANDBOX_RUNTIME_IMAGE}")" --set-string "sandboxRuntime.image.tag=$(e2e_image_reference_tag "${SANDBOX_RUNTIME_IMAGE}")" --set-string "sandboxRuntime.image.digest=$(e2e_image_reference_digest "${SANDBOX_RUNTIME_IMAGE}")")
# Resolve a host-gateway IP that sandbox pods can dial to reach test fixtures
# running on the developer/CI host (HTTP fixtures bound to 0.0.0.0 plus sibling
# Docker containers with published ports). The Helm chart wires this into pod
# hostAliases for host.openshell.internal / host.docker.internal — without it,
# every test that relies on the alias has to skip on the kube driver.
#
# Preference order:
# 1. OPENSHELL_E2E_HOST_GATEWAY_IP — operator override (remote clusters where
# auto-detection has no signal).
# 2. k3d's CoreDNS host.k3d.internal entry. On Docker Desktop this is a
# host-routable address; the Docker network gateway is not.
# 3. Gateway of the cluster's Docker network (k3d-<cluster> for ephemeral
# clusters, `kind` for kind clusters used in CI). Pods SNAT through their
# node to this IP, which lands on the host's bridge interface and reaches
# any 0.0.0.0-bound listener / published container port.
HOST_GATEWAY_IP="${OPENSHELL_E2E_HOST_GATEWAY_IP:-}"
# k3d primes CoreDNS with `host.k3d.internal` pointing at the IP that pods can
# use to reach the host (Docker Desktop's gvisor-net loopback on macOS/Windows,
# the docker bridge gateway on Linux). That mapping handles Docker Desktop
# correctly; the docker network gateway alone does not.
if [ -z "${HOST_GATEWAY_IP}" ] && command -v kubectl >/dev/null 2>&1; then
for _ in {1..15}; do
detected="$(kctl -n kube-system get configmap coredns -o jsonpath='{.data.NodeHosts}' 2>/dev/null \
| awk '$2 == "host.k3d.internal" { print $1; exit }' || true)"
if [ -n "${detected}" ]; then
HOST_GATEWAY_IP="${detected}"
echo "Detected host gateway IP ${HOST_GATEWAY_IP} from CoreDNS host.k3d.internal entry."
break
fi
sleep 1
done
fi
# Fallback for non-k3d clusters (kind in CI, etc.): use the docker network
# gateway IP. Works on Linux where the bridge is reachable from pods; on macOS
# Docker Desktop without k3d, this will likely not route to the host.
use_docker_network_gateway=1
if [ "$(uname -s)" = "Darwin" ] \
&& { [ "${CLUSTER_CREATED_BY_US}" = "1" ] || [[ "${KUBE_CONTEXT}" == k3d-* ]]; }; then
use_docker_network_gateway=0
fi
if [ -z "${HOST_GATEWAY_IP}" ] \
&& [ "${use_docker_network_gateway}" = "1" ] \
&& command -v docker >/dev/null 2>&1; then
candidate_networks=()
if [ "${CLUSTER_CREATED_BY_US}" = "1" ]; then
candidate_networks+=("k3d-${CLUSTER_NAME}")
elif [[ "${KUBE_CONTEXT}" == k3d-* ]]; then
candidate_networks+=("k3d-${KUBE_CONTEXT#k3d-}")
elif [[ "${KUBE_CONTEXT}" == kind-* ]]; then
candidate_networks+=("kind")
else
candidate_networks+=("kind" "k3d-${KUBE_CONTEXT#k3d-}")
fi
for net in "${candidate_networks[@]}"; do
[ -n "${net}" ] || continue
# Prefer the IPv4 gateway — kind dual-stacks its network and the IPv6 entry
# is unreachable for the typical test-host listener (0.0.0.0 bind).
detected="$(docker network inspect "${net}" \
-f '{{range .IPAM.Config}}{{.Gateway}}{{"\n"}}{{end}}' 2>/dev/null \
| awk '/^[0-9.]+$/ { print; exit }' || true)"
if [ -n "${detected}" ]; then
HOST_GATEWAY_IP="${detected}"
echo "Detected host gateway IP ${HOST_GATEWAY_IP} from docker network '${net}'."
break
fi
done
fi
if [ -z "${HOST_GATEWAY_IP}" ]; then
echo "WARNING: could not resolve a host gateway IP for the active cluster." >&2
echo " Tests that require host.openshell.internal will be skipped." >&2
echo " Set OPENSHELL_E2E_HOST_GATEWAY_IP to override." >&2
fi
# Import locally available gateway, sandbox, and supervisor images into the k3d cluster so
# devs working off local builds don't depend on the configured registry. For
# kind clusters (used by CI), images must be loaded before this script runs —
# the workflow handles that via `kind load docker-image`. Best-effort: when an
# image isn't present locally, the cluster falls back to its pull behavior.
import_cluster_name=""
if [ "${CLUSTER_CREATED_BY_US}" = "1" ]; then
import_cluster_name="${CLUSTER_NAME}"
elif [[ "${KUBE_CONTEXT}" == k3d-* ]] && command -v k3d >/dev/null 2>&1; then
candidate="${KUBE_CONTEXT#k3d-}"
if k3d cluster list "${candidate}" >/dev/null 2>&1; then
import_cluster_name="${candidate}"
fi
fi
if [ "${OPENSHELL_E2E_KUBE_BUILD_IMAGES}" = "1" ]; then
require_cmd docker
echo "Building local Kubernetes e2e images (${BUILD_GATEWAY_IMAGE}, ${BUILD_SUPERVISOR_IMAGE})..."
if [ "${OPENSHELL_E2E_EXTERNAL_COMPUTE_DRIVER:-0}" = "1" ]; then
if [ "$(uname -s)" != "Linux" ]; then
echo "ERROR: external Kubernetes driver image composition currently requires a Linux build host." >&2
exit 2
fi
external_gateway="${OPENSHELL_GATEWAY_BIN:-${ROOT}/target/debug/openshell-gateway}"
external_driver="${OPENSHELL_EXTERNAL_DRIVER_BIN:-${ROOT}/target/debug/openshell-driver-kubernetes}"
if [ -z "${OPENSHELL_GATEWAY_BIN:-}" ]; then
cargo build -p openshell-gateway --bin openshell-gateway \
--no-default-features --features telemetry,vendored-z3
fi
if [ -z "${OPENSHELL_EXTERNAL_DRIVER_BIN:-}" ]; then
cargo build -p openshell-driver-kubernetes --bin openshell-driver-kubernetes
fi
case "$(uname -m)" in
x86_64) external_arch=amd64 ;;
aarch64|arm64) external_arch=arm64 ;;
*) echo "ERROR: unsupported external Kubernetes driver architecture: $(uname -m)" >&2; exit 2 ;;
esac
external_stage="${ROOT}/deploy/docker/.build/prebuilt-binaries/${external_arch}"
mkdir -p "${external_stage}"
cp "${external_gateway}" "${external_stage}/openshell-gateway"
cp "${external_driver}" "${external_stage}/openshell-driver-kubernetes"
docker build \
--build-arg "TARGETARCH=${external_arch}" \
--build-arg "SUPERVISOR_IMAGE=${BUILD_SUPERVISOR_IMAGE}" \
--build-arg "SANDBOX_RUNTIME_IMAGE=${REGISTRY_VALUE}/sandbox:${IMAGE_TAG_VALUE}" \
--tag "${BUILD_GATEWAY_IMAGE}" \
--file "${ROOT}/e2e/docker/Dockerfile.external-kubernetes-gateway" \
"${ROOT}"
else
CONTAINER_ENGINE=docker IMAGE_REGISTRY="${REGISTRY_VALUE}" IMAGE_TAG="${IMAGE_TAG_VALUE}" \
bash "${ROOT}/tasks/scripts/docker-build-image.sh" gateway
fi
sandbox_image="${REGISTRY_VALUE}/sandbox:${IMAGE_TAG_VALUE}"
if [ "${GATEWAY_IMAGE}" != "${BUILD_GATEWAY_IMAGE}" ]; then
if e2e_image_reference_has_digest "${GATEWAY_IMAGE}"; then echo "ERROR: digest-pinned GATEWAY_IMAGE requires OPENSHELL_E2E_KUBE_BUILD_IMAGES=0" >&2; exit 2; fi
docker tag "${BUILD_GATEWAY_IMAGE}" "${GATEWAY_IMAGE}"
fi
supervisor_image="${BUILD_SUPERVISOR_IMAGE}"
if [ "${OPENSHELL_E2E_EXTERNAL_COMPUTE_DRIVER:-0}" != "1" ] \
|| ! docker image inspect "${sandbox_image}" >/dev/null 2>&1; then
CONTAINER_ENGINE=docker IMAGE_REGISTRY="${REGISTRY_VALUE}" IMAGE_TAG="${IMAGE_TAG_VALUE}" \
bash "${ROOT}/tasks/scripts/docker-build-image.sh" sandbox
else
reuse_sandbox_image=1
echo "Reusing existing sandbox image ${sandbox_image}"
fi
if [ "${SANDBOX_RUNTIME_IMAGE}" != "${sandbox_image}" ]; then
if e2e_image_reference_has_digest "${SANDBOX_RUNTIME_IMAGE}"; then
echo "ERROR: digest-pinned SANDBOX_IMAGE requires OPENSHELL_E2E_KUBE_BUILD_IMAGES=0" >&2
exit 2
fi
docker tag "${sandbox_image}" "${SANDBOX_RUNTIME_IMAGE}"
fi
if [ "${OPENSHELL_E2E_EXTERNAL_COMPUTE_DRIVER:-0}" != "1" ] \
|| ! docker image inspect "${supervisor_image}" >/dev/null 2>&1; then
CONTAINER_ENGINE=docker IMAGE_REGISTRY="${REGISTRY_VALUE}" IMAGE_TAG="${IMAGE_TAG_VALUE}" \
bash "${ROOT}/tasks/scripts/docker-build-image.sh" supervisor
else
reuse_supervisor_image=1
echo "Reusing existing supervisor image ${supervisor_image}"
fi
if [ "${SUPERVISOR_IMAGE}" != "${BUILD_SUPERVISOR_IMAGE}" ]; then
if e2e_image_reference_has_digest "${SUPERVISOR_IMAGE}"; then echo "ERROR: digest-pinned SUPERVISOR_IMAGE requires OPENSHELL_E2E_KUBE_BUILD_IMAGES=0" >&2; exit 2; fi
docker tag "${BUILD_SUPERVISOR_IMAGE}" "${SUPERVISOR_IMAGE}"
fi
fi
if [ -n "${import_cluster_name}" ]; then
for image in \
"${GATEWAY_IMAGE}" \
"${REGISTRY_VALUE}/sandbox:${IMAGE_TAG_VALUE}" \
"${SUPERVISOR_IMAGE}" \
"${SANDBOX_RUNTIME_IMAGE}"; do
if docker image inspect "${image}" >/dev/null 2>&1; then
echo "Importing ${image} into k3d cluster ${import_cluster_name}..."
k3d image import "${image}" --cluster "${import_cluster_name}" \
--mode direct >/dev/null
fi
done
elif [ "${OPENSHELL_E2E_KUBE_BUILD_IMAGES}" = "1" ] \
&& [[ "${KUBE_CONTEXT}" == kind-* ]] \
&& command -v kind >/dev/null 2>&1; then
kind_cluster_name="${KUBE_CONTEXT#kind-}"
kind_images=("${GATEWAY_IMAGE}")
# The CI workflow loads its published sandbox archive before invoking this
# wrapper. Load a replacement only when this script rebuilt or retagged it.
if [ "${reuse_sandbox_image}" != "1" ] \
|| [ "${SANDBOX_RUNTIME_IMAGE}" != "${sandbox_image}" ]; then
kind_images+=("${SANDBOX_RUNTIME_IMAGE}")
fi
# The CI workflow loads its published supervisor archive before invoking this
# wrapper. Only load a supervisor image here when this script rebuilt it.
if [ "${reuse_supervisor_image}" != "1" ] \
|| [ "${SUPERVISOR_IMAGE}" != "${BUILD_SUPERVISOR_IMAGE}" ]; then
kind_images+=("${SUPERVISOR_IMAGE}")
fi
for image in "${kind_images[@]}"; do
echo "Loading ${image} into kind cluster ${kind_cluster_name}..."
kind load docker-image "${image}" --name "${kind_cluster_name}"
done
fi
# The Kubernetes compute driver creates and watches Sandbox CRs reconciled
# by the upstream agent-sandbox-controller. Without the CRD + controller,
# every gateway K8s call 404s and CreateSandbox never produces a Pod.
AGENT_SANDBOX_VERSION="${AGENT_SANDBOX_VERSION}" \
bash "${ROOT}/e2e/support/install-agent-sandbox.sh" --context "${KUBE_CONTEXT}"
# Detect OpenShift up front so fixtures deployed below can apply SCC-compatible
# handling; the gateway setup further down reuses this flag.
if kctl api-resources --api-group=route.openshift.io --no-headers 2>/dev/null | grep -q .; then
OPENSHIFT_DETECTED=1
if ! command -v oc >/dev/null 2>&1; then
echo "ERROR: oc CLI is required for OpenShift SCC management but was not found." >&2
exit 2
fi
fi
ACTIVE_CREDENTIAL_DRIVER="${OPENSHELL_E2E_CREDENTIAL_DRIVER:-kubernetes-secrets}"
if [ "${OPENSHELL_E2E_CREDENTIAL_DRIVERS:-0}" = "1" ] \
&& [ "${ACTIVE_CREDENTIAL_DRIVER}" = "vault" ]; then
deploy_vault_fixture
fi
helm_extra_args=()
helm_post_renderer_args=()
helm_extra_args+=(--set "server.telemetryEnabled=${OPENSHELL_TELEMETRY_ENABLED}")
if [ "${OPENSHELL_E2E_EXTERNAL_COMPUTE_DRIVER:-0}" = "1" ]; then
if [ "${OPENSHELL_E2E_KUBE_BUILD_IMAGES}" != "1" ]; then
echo "ERROR: external Kubernetes driver e2e requires OPENSHELL_E2E_KUBE_BUILD_IMAGES=1." >&2
exit 2
fi
export HELM_PLUGINS="${ROOT}/e2e/helm-plugins"
helm_post_renderer_args+=(
--post-renderer openshell-external-compute-driver
)
fi
if [ -n "${HOST_GATEWAY_IP}" ]; then
helm_extra_args+=(--set "server.hostGatewayIP=${HOST_GATEWAY_IP}")
fi
helm_values_args=(--values "${ROOT}/deploy/helm/openshell/ci/values-skaffold.yaml")
if [ "${OPENSHIFT_DETECTED}" = "1" ]; then
echo "OpenShift detected — applying SCC-compatible security context overrides."
helm_values_args+=(--values "${ROOT}/deploy/helm/openshell/ci/values-openshift-scc.yaml")
kctl create namespace "${NAMESPACE}" --dry-run=client -o yaml | kctl apply -f -
echo "Granting privileged SCC to openshell-sandbox in namespace ${NAMESPACE}..."
oc adm policy add-scc-to-user privileged \
--context "${KUBE_CONTEXT}" \
-z openshell-sandbox -n "${NAMESPACE}"
OPENSHIFT_SANDBOX_SCC_GRANTED=1
# Drive the gateway through a passthrough Route with mTLS instead of
# port-forward. The Route host is deterministic: OpenShift serves any name
# under the cluster ingress (apps) domain via the router's wildcard, so we
# bake "<release>-<namespace>.<apps-domain>" into the server cert SANs before
# the Route exists.
APPS_DOMAIN="$(kctl get ingresses.config/cluster -o jsonpath='{.spec.domain}')"
if [ -z "${APPS_DOMAIN}" ]; then
echo "ERROR: could not resolve the OpenShift cluster ingress domain." >&2
exit 2
fi
OPENSHIFT_ROUTE_HOST="${RELEASE_NAME}-${NAMESPACE}.${APPS_DOMAIN}"
echo "Using OpenShift Route host ${OPENSHIFT_ROUTE_HOST}."
helm_values_args+=(--values "${ROOT}/deploy/helm/openshell/ci/values-openshift-e2e.yaml")
helm_extra_args+=(--set "openshiftRoute.host=${OPENSHIFT_ROUTE_HOST}")
helm_extra_args+=(--set "pkiInitJob.serverDnsNames[0]=${OPENSHIFT_ROUTE_HOST}")
fi
if [ "${OPENSHELL_E2E_KUBE_CORPORATE_PROXY:-0}" = "1" ]; then
if [ -z "${HOST_GATEWAY_IP}" ]; then
echo "ERROR: corporate proxy e2e requires a host gateway IP for host.openshell.internal" >&2
exit 2
fi
CORPORATE_PROXY_PORT="$(e2e_pick_port)"
CORPORATE_PROXY_MODE="${OPENSHELL_E2E_KUBE_CORPORATE_PROXY_MODE:-authenticated}"
CORPORATE_PROXY_UPSTREAM_PORT=""
if [ "${CORPORATE_PROXY_MODE}" = "missing-secret" ] || [ "${CORPORATE_PROXY_MODE}" = "malformed" ]; then
export OPENSHELL_PROVISION_TIMEOUT="${OPENSHELL_PROVISION_TIMEOUT:-45}"
fi
CORPORATE_PROXY_VALUES="${WORKDIR}/corporate-proxy-values.yaml"
# `https-ca` terminates TLS on the proxy listener itself, so the supervisor
# must trust the operator CA bundle before it can even issue CONNECT.
CORPORATE_PROXY_SCHEME="http"
if [ "${CORPORATE_PROXY_MODE}" = "https-ca" ]; then
CORPORATE_PROXY_SCHEME="https"
fi
cat >"${CORPORATE_PROXY_VALUES}" <<EOF
upstreamProxy:
url: ${CORPORATE_PROXY_SCHEME}://host.openshell.internal:${CORPORATE_PROXY_PORT}
EOF
if [ "${CORPORATE_PROXY_MODE}" = "no-proxy" ]; then
CORPORATE_PROXY_UPSTREAM_PORT="$(e2e_pick_port)"
cat >>"${CORPORATE_PROXY_VALUES}" <<EOF
noProxy: host.openshell.internal
EOF
fi
case "${CORPORATE_PROXY_MODE}" in
authenticated|no-proxy)
kctl create namespace "${NAMESPACE}" --dry-run=client -o yaml | kctl apply -f -
kctl -n "${NAMESPACE}" create secret generic "${CORPORATE_PROXY_FIXTURE_SECRET}" \
--from-literal=proxy-auth=proxyuser:proxypass --dry-run=client -o yaml | kctl apply -f -
CORPORATE_PROXY_FIXTURE_DEPLOYED=1
;;
malformed)
kctl create namespace "${NAMESPACE}" --dry-run=client -o yaml | kctl apply -f -
kctl -n "${NAMESPACE}" create secret generic "${CORPORATE_PROXY_FIXTURE_SECRET}" \
--from-literal=proxy-auth=malformed --dry-run=client -o yaml | kctl apply -f -
CORPORATE_PROXY_FIXTURE_DEPLOYED=1
;;
https-ca)
kctl create namespace "${NAMESPACE}" --dry-run=client -o yaml | kctl apply -f -
kctl -n "${NAMESPACE}" create secret generic "${CORPORATE_PROXY_FIXTURE_SECRET}" \
--from-literal=proxy-auth=proxyuser:proxypass --dry-run=client -o yaml | kctl apply -f -
CORPORATE_PROXY_FIXTURE_DEPLOYED=1
# Mint a private CA and a listener leaf for the proxy. The leaf must be
# CA:FALSE with a serverAuth EKU, or rustls rejects it regardless of
# whether the issuing CA is trusted.
CORPORATE_PROXY_TLS_DIR="${WORKDIR}/corporate-proxy-tls"
mkdir -p "${CORPORATE_PROXY_TLS_DIR}"
openssl req -x509 -newkey rsa:2048 -nodes -days 1 \
-keyout "${CORPORATE_PROXY_TLS_DIR}/ca.key" \
-out "${CORPORATE_PROXY_TLS_DIR}/ca.crt" \
-subj "/CN=OpenShell E2E Corporate Proxy CA" >/dev/null 2>&1
openssl req -newkey rsa:2048 -nodes \
-keyout "${CORPORATE_PROXY_TLS_DIR}/proxy.key" \
-out "${CORPORATE_PROXY_TLS_DIR}/proxy.csr" \
-subj "/CN=host.openshell.internal" >/dev/null 2>&1
cat >"${CORPORATE_PROXY_TLS_DIR}/leaf.ext" <<EXT
basicConstraints=critical,CA:FALSE
keyUsage=critical,digitalSignature,keyEncipherment
extendedKeyUsage=serverAuth
subjectAltName=DNS:host.openshell.internal
EXT
openssl x509 -req -days 1 \
-in "${CORPORATE_PROXY_TLS_DIR}/proxy.csr" \
-CA "${CORPORATE_PROXY_TLS_DIR}/ca.crt" \
-CAkey "${CORPORATE_PROXY_TLS_DIR}/ca.key" \
-CAcreateserial \
-extfile "${CORPORATE_PROXY_TLS_DIR}/leaf.ext" \
-out "${CORPORATE_PROXY_TLS_DIR}/proxy.crt" >/dev/null 2>&1
# The CA ConfigMap belongs to the gateway release namespace: the gateway
# reads it and stages the bundle per sandbox, so it is never sourced
# from a workload namespace.
kctl -n "${NAMESPACE}" create configmap "${CORPORATE_PROXY_FIXTURE_CA_CONFIGMAP}" \
--from-file=ca.crt="${CORPORATE_PROXY_TLS_DIR}/ca.crt" \
--dry-run=client -o yaml | kctl apply -f -
CORPORATE_PROXY_CA_FIXTURE_DEPLOYED=1
cat >>"${CORPORATE_PROXY_VALUES}" <<EOF
caBundle:
configMapName: ${CORPORATE_PROXY_FIXTURE_CA_CONFIGMAP}
EOF
OPENSHELL_E2E_CORPORATE_PROXY_TLS_CERT="$(cat "${CORPORATE_PROXY_TLS_DIR}/proxy.crt")"
OPENSHELL_E2E_CORPORATE_PROXY_TLS_KEY="$(cat "${CORPORATE_PROXY_TLS_DIR}/proxy.key")"
export OPENSHELL_E2E_CORPORATE_PROXY_TLS_CERT
export OPENSHELL_E2E_CORPORATE_PROXY_TLS_KEY
;;
missing-secret) ;;
*) echo "ERROR: unknown corporate proxy e2e mode '${CORPORATE_PROXY_MODE}'" >&2; exit 2 ;;
esac
export OPENSHELL_E2E_CORPORATE_PROXY_PORT="${CORPORATE_PROXY_PORT}"
export OPENSHELL_E2E_CORPORATE_PROXY_UPSTREAM_PORT="${CORPORATE_PROXY_UPSTREAM_PORT}"
export OPENSHELL_E2E_CORPORATE_PROXY_MODE="${CORPORATE_PROXY_MODE}"
helm_values_args+=(--values "${ROOT}/deploy/helm/openshell/ci/values-corporate-proxy-e2e.yaml")
helm_values_args+=(--values "${CORPORATE_PROXY_VALUES}")
fi
if [ "${OPENSHELL_E2E_CREDENTIAL_DRIVERS:-0}" = "1" ]; then
case "${ACTIVE_CREDENTIAL_DRIVER}" in
kubernetes-secrets)
helm_values_args+=(--values "${ROOT}/deploy/helm/openshell/ci/values-credential-driver-kubernetes-secrets.yaml")
;;
vault)
helm_values_args+=(--values "${ROOT}/deploy/helm/openshell/ci/values-credential-driver-vault.yaml")
;;
*)
echo "ERROR: OPENSHELL_E2E_CREDENTIAL_DRIVER must be kubernetes-secrets or vault, got '${ACTIVE_CREDENTIAL_DRIVER}'" >&2
exit 2
;;
esac
export OPENSHELL_E2E_CREDENTIAL_DRIVER="${ACTIVE_CREDENTIAL_DRIVER}"
fi
if [ -n "${OPENSHELL_E2E_KUBE_EXTRA_VALUES:-}" ]; then
IFS=':' read -r -a extra_values_files <<< "${OPENSHELL_E2E_KUBE_EXTRA_VALUES}"
for values_file in "${extra_values_files[@]}"; do
[ -n "${values_file}" ] || continue
if [[ "${values_file}" != /* ]]; then
values_file="${ROOT}/${values_file}"
fi
helm_values_args+=(--values "${values_file}")
done
fi
if use_envoy_gateway; then
helm_values_args+=(--values "${ROOT}/deploy/helm/openshell/ci/values-gateway.yaml")
install_envoy_gateway
fi
if [ "${OPENSHELL_E2E_KUBE_DB_SCENARIOS:-0}" = "1" ]; then
# --- Multi-scenario mode: test all database backends ---
DB_PASSED=0
DB_FAILED=0
DB_SCENARIOS_SUMMARY=()
E2E_CMD=("$@")
run_scenario "SQLite (default)" sqlite \
"${helm_extra_args[@]}"
scenario_deploy_external_pg
run_scenario "External PostgreSQL (externalDbSecret)" external-pg \
"${helm_extra_args[@]}" \
--set server.externalDbSecret=my-pg-credentials
scenario_cleanup_external_pg
echo ""
echo "========================================"
echo " DB Scenario Test Summary"
echo "========================================"
for s in "${DB_SCENARIOS_SUMMARY[@]}"; do
echo " $s"
done
echo "----------------------------------------"
echo " Passed: $DB_PASSED Failed: $DB_FAILED"
echo "========================================"
if [ "$DB_FAILED" -gt 0 ]; then
exit 1
fi
else
# --- Single-install mode (default, existing behavior) ---
if [ -n "${OPENSHELL_E2E_KUBE_EXTERNAL_POSTGRES_SECRET:-}" ]; then
deploy_postgres_fixture "${OPENSHELL_E2E_KUBE_EXTERNAL_POSTGRES_SECRET}"
fi
echo "Installing Helm chart (release=${RELEASE_NAME}, namespace=${NAMESPACE}, tag=${IMAGE_TAG_VALUE})..."
helmctl install "${RELEASE_NAME}" "${ROOT}/deploy/helm/openshell" \
--namespace "${NAMESPACE}" --create-namespace \
"${helm_values_args[@]}" \
--set "fullnameOverride=openshell" \
"${GLOBAL_HELM_IMAGE_ARGS[@]}" \
"${GATEWAY_HELM_IMAGE_ARGS[@]}" \
"${SUPERVISOR_HELM_IMAGE_ARGS[@]}" \
"${SANDBOX_RUNTIME_HELM_IMAGE_ARGS[@]}" \
"${helm_extra_args[@]}" \
"${helm_post_renderer_args[@]}" \
--wait --timeout 5m
HELM_INSTALLED=1
if [ -n "${OPENSHELL_E2E_KUBE_IMAGE_PULL_SECRET:-}" ]; then
kctl -n "${NAMESPACE}" create secret docker-registry \
"${OPENSHELL_E2E_KUBE_IMAGE_PULL_SECRET}" \
--docker-server=registry.example.test \
--docker-username=e2e-user \
--docker-password=e2e-password
kctl -n "${NAMESPACE}" label secret \
"${OPENSHELL_E2E_KUBE_IMAGE_PULL_SECRET}" \
openshell.ai/sandbox-attachable=true
fi
if [ "${OPENSHIFT_DETECTED}" = "1" ]; then
# OpenShift: reach the gateway over the passthrough Route with mTLS so the
# SSH-relay `sandbox connect` suites work (port-forward stalls them).
openshift_register_route_gateway || exit 1
else
# Vanilla Kubernetes: reach the gateway in plaintext over port-forward.
start_gateway_portforward || exit 1
GATEWAY_NAME="openshell-e2e-kube-${LOCAL_PORT}"
GATEWAY_ENDPOINT="http://127.0.0.1:${LOCAL_PORT}"
e2e_register_plaintext_gateway \
"${XDG_CONFIG_HOME}" \
"${GATEWAY_NAME}" \
"${GATEWAY_ENDPOINT}" \
"${LOCAL_PORT}"
fi
start_health_portforward || exit 1
export OPENSHELL_GATEWAY="${GATEWAY_NAME}"
export OPENSHELL_E2E_DRIVER="kubernetes"
# Kubernetes e2e runs against k3d/kind-style Docker-backed clusters. Host
# fixture containers must use the same Docker host so published ports and
# cluster host-gateway aliases line up even on machines where Podman is also
# installed.
export CONTAINER_ENGINE="${CONTAINER_ENGINE:-docker}"
export OPENSHELL_E2E_KUBE_CONTEXT_ACTIVE="${KUBE_CONTEXT}"
export OPENSHELL_E2E_SANDBOX_NAMESPACE="${NAMESPACE}"
export OPENSHELL_E2E_KUBE_CONTEXT="${KUBE_CONTEXT}"
export OPENSHELL_E2E_KUBE_NAMESPACE="${NAMESPACE}"
export OPENSHELL_E2E_KUBE_RELEASE="${RELEASE_NAME}"
export OPENSHELL_PROVISION_TIMEOUT="${OPENSHELL_PROVISION_TIMEOUT:-300}"
e2e_import_example_provider_profiles \
"${OPENSHELL_BIN:-${ROOT}/target/debug/openshell}" "${ROOT}" || exit 1
echo "Running e2e command against ${GATEWAY_ENDPOINT}: $*"
"$@"
fi