diff --git a/crates/openshell-server/src/gateway_metrics.rs b/crates/openshell-server/src/gateway_metrics.rs index b18915864..28adda171 100644 --- a/crates/openshell-server/src/gateway_metrics.rs +++ b/crates/openshell-server/src/gateway_metrics.rs @@ -25,8 +25,6 @@ use tonic::{Code, Status}; pub const SUPERVISOR_SESSIONS: &str = "openshell_server_supervisor_sessions"; pub const RELAY_PENDING: &str = "openshell_server_relay_pending"; pub const RELAY_PENDING_CAPACITY: &str = "openshell_server_relay_pending_capacity"; -pub const RELAY_PENDING_PER_SANDBOX_CAPACITY: &str = - "openshell_server_relay_pending_per_sandbox_capacity"; // Counters pub const RELAY_REJECTED_TOTAL: &str = "openshell_server_relay_rejected_total"; pub const RELAY_EXPIRED_TOTAL: &str = "openshell_server_relay_expired_total"; @@ -36,11 +34,10 @@ pub const RELAY_CLAIM_DURATION_SECONDS: &str = "openshell_server_relay_claim_dur pub const PEER_REQUEST_DURATION_SECONDS: &str = "openshell_server_peer_request_duration_seconds"; const LABEL_REASON: &str = "reason"; -const LABEL_METHOD: &str = "method"; const LABEL_OPERATION: &str = "operation"; const LABEL_OUTCOME: &str = "outcome"; const LABEL_GRPC_CODE: &str = "grpc_code"; -const LABEL_TARGET: &str = "target"; +const LABEL_RELAY_KIND: &str = "relay_kind"; const LABEL_ROUTE: &str = "route"; /// Buckets for the new latency histograms, 1 ms to 15 s. The top buckets cover the 10 s relay @@ -56,12 +53,12 @@ const BUCKETED_HISTOGRAMS: [&str; 2] = /// Protocol the supervisor is asked to relay. Never label metrics with the target address. #[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum RelayTarget { +pub enum RelayKind { Ssh, Tcp, } -impl RelayTarget { +impl RelayKind { pub const ALL: [Self; 2] = [Self::Ssh, Self::Tcp]; pub const fn label(self) -> &'static str { @@ -108,8 +105,8 @@ impl RelayRejection { } } -/// Outbound peer RPC. The `method` label takes the same values as the `method` label of -/// `openshell_server_grpc_requests_total` on the owning replica. +/// Routed operation. For a peer request, the owning replica records the matching gRPC method in +/// `openshell_server_grpc_requests_total`, for example `PeerRelay` for `relay`. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum PeerRpc { Relay, @@ -126,15 +123,6 @@ impl PeerRpc { Self::GetSandboxProviderStatus, ]; - pub const fn label(self) -> &'static str { - match self { - Self::Relay => "PeerRelay", - Self::ReportProviderReadiness => "PeerReportProviderReadiness", - Self::ReportEndpointStatus => "PeerReportEndpointStatus", - Self::GetSandboxProviderStatus => "PeerGetSandboxProviderStatus", - } - } - const fn operation(self) -> &'static str { match self { Self::Relay => "relay", @@ -145,19 +133,23 @@ impl PeerRpc { } } +/// Where a routed attempt ended. A relay succeeds only when the supervisor claims it, on either +/// route, so the values mean the same thing for local and peer attempts. #[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum PeerOutcome { +enum AttemptOutcome { Success, - ClientError, - RpcError, + /// Failed on this replica, including cancellation by the caller. + LocalError, + /// The owning replica returned an error, or the open peer connection failed. + RemoteError, } -impl PeerOutcome { +impl AttemptOutcome { const fn label(self) -> &'static str { match self { Self::Success => "success", - Self::ClientError => "client_error", - Self::RpcError => "rpc_error", + Self::LocalError => "local_error", + Self::RemoteError => "remote_error", } } } @@ -187,15 +179,12 @@ const fn grpc_code_label(code: Code) -> &'static str { } } -/// Relay caps published as `openshell_server_relay_pending_capacity` and -/// `openshell_server_relay_pending_per_sandbox_capacity`. The caller passes the values that -/// enforce the caps, so this module does not depend on the relay registry. +/// Relay cap published as `openshell_server_relay_pending_capacity`. The caller passes the value +/// that enforces the cap, so this module does not depend on the relay registry. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct RelayCapacity { /// Pending relays allowed on one replica. - pub global: usize, - /// Pending relays allowed for one sandbox on one replica. - pub per_sandbox: usize, + pub per_replica: usize, } /// Apply the bucket overrides. Tests build local recorders from the same builder. @@ -236,11 +225,6 @@ pub fn describe_and_initialize(relay: RelayCapacity) { Unit::Count, "Maximum pending relay channels on one gateway replica." ); - describe_gauge!( - RELAY_PENDING_PER_SANDBOX_CAPACITY, - Unit::Count, - "Maximum pending relay channels for one sandbox on one gateway replica." - ); describe_counter!( ROUTED_REQUEST_ATTEMPTS_TOTAL, Unit::Count, @@ -264,22 +248,21 @@ pub fn describe_and_initialize(relay: RelayCapacity) { describe_histogram!( PEER_REQUEST_DURATION_SECONDS, Unit::Seconds, - "Latency of requests to the owning replica. For PeerRelay, until the owner's supervisor claimed the relay." + "Latency of outbound requests to the owning replica. For relays, until the owner's supervisor claimed the relay." ); // `increment(0)` registers a series without overwriting a value recorded earlier. gauge!(SUPERVISOR_SESSIONS).increment(0.0); gauge!(RELAY_PENDING).increment(0.0); - gauge!(RELAY_PENDING_CAPACITY).set(count_as_f64(relay.global)); - gauge!(RELAY_PENDING_PER_SANDBOX_CAPACITY).set(count_as_f64(relay.per_sandbox)); - for target in RelayTarget::ALL { + gauge!(RELAY_PENDING_CAPACITY).set(count_as_f64(relay.per_replica)); + for kind in RelayKind::ALL { for route in RelayRoute::ALL { counter!( ROUTED_REQUEST_ATTEMPTS_TOTAL, LABEL_OPERATION => PeerRpc::Relay.operation(), LABEL_ROUTE => route.label(), - LABEL_TARGET => target.label(), - LABEL_OUTCOME => PeerOutcome::Success.label(), + LABEL_RELAY_KIND => kind.label(), + LABEL_OUTCOME => AttemptOutcome::Success.label(), LABEL_GRPC_CODE => grpc_code_label(Code::Ok) ) .increment(0); @@ -297,8 +280,8 @@ pub fn describe_and_initialize(relay: RelayCapacity) { ROUTED_REQUEST_ATTEMPTS_TOTAL, LABEL_OPERATION => rpc.operation(), LABEL_ROUTE => RelayRoute::Peer.label(), - LABEL_TARGET => "none", - LABEL_OUTCOME => PeerOutcome::Success.label(), + LABEL_RELAY_KIND => "none", + LABEL_OUTCOME => AttemptOutcome::Success.label(), LABEL_GRPC_CODE => grpc_code_label(Code::Ok) ) .increment(0); @@ -357,14 +340,13 @@ pub fn record_relay_claimed(waited: Duration) { } /// Counts one local relay setup or outbound peer attempt exactly once, and times peer requests. -/// Local success means enqueued; peer relay success means the supervisor claimed it. -/// Dropping an unfinished timer (the caller's future was cancelled) records -/// `client_error` / `cancelled`. -#[must_use = "finish the timer with client_error() or finish()"] +/// A relay succeeds when the supervisor claims it, on either route. Dropping an unfinished +/// timer (the caller gave up) records `local_error` / `cancelled`. +#[must_use = "finish the timer with local_error() or finish()"] pub struct RoutedRequestTimer { rpc: PeerRpc, route: RelayRoute, - target: Option, + relay_kind: Option, started: Instant, recorded: bool, } @@ -374,8 +356,8 @@ impl RoutedRequestTimer { Self { rpc, route: RelayRoute::Peer, - target: if rpc == PeerRpc::Relay { - Some(RelayTarget::Ssh) + relay_kind: if rpc == PeerRpc::Relay { + Some(RelayKind::Ssh) } else { None }, @@ -384,32 +366,34 @@ impl RoutedRequestTimer { } } - pub fn relay(target: RelayTarget, route: RelayRoute) -> Self { + pub fn relay(kind: RelayKind, route: RelayRoute) -> Self { Self { rpc: PeerRpc::Relay, route, - target: Some(target), + relay_kind: Some(kind), started: Instant::now(), recorded: false, } } - /// The request failed before it reached the peer: token, channel, headers, or stream setup. - pub fn client_error(&mut self, status: &Status) { - self.record(PeerOutcome::ClientError, status.code()); + /// The attempt failed on this replica: local relay setup or claim (including the claim + /// window closing), or a peer request that failed before it reached the owner (token, + /// channel, headers, or stream setup). + pub fn local_error(&mut self, status: &Status) { + self.record(AttemptOutcome::LocalError, status.code()); } /// Record the raw tonic result of the RPC itself. Call this BEFORE any remap to /// `Unavailable`, so the owner's code (for example `resource_exhausted`) is kept. pub fn finish(&mut self, result: &Result) { match result { - Ok(_) => self.record(PeerOutcome::Success, Code::Ok), - Err(status) if self.route == RelayRoute::Local => self.client_error(status), - Err(status) => self.record(PeerOutcome::RpcError, status.code()), + Ok(_) => self.record(AttemptOutcome::Success, Code::Ok), + Err(status) if self.route == RelayRoute::Local => self.local_error(status), + Err(status) => self.record(AttemptOutcome::RemoteError, status.code()), } } - fn record(&mut self, outcome: PeerOutcome, code: Code) { + fn record(&mut self, outcome: AttemptOutcome, code: Code) { if self.recorded { return; } @@ -418,7 +402,7 @@ impl RoutedRequestTimer { ROUTED_REQUEST_ATTEMPTS_TOTAL, LABEL_OPERATION => self.rpc.operation(), LABEL_ROUTE => self.route.label(), - LABEL_TARGET => self.target.map_or("none", RelayTarget::label), + LABEL_RELAY_KIND => self.relay_kind.map_or("none", RelayKind::label), LABEL_OUTCOME => outcome.label(), LABEL_GRPC_CODE => grpc_code_label(code) ) @@ -426,7 +410,7 @@ impl RoutedRequestTimer { if self.route == RelayRoute::Peer { histogram!( PEER_REQUEST_DURATION_SECONDS, - LABEL_METHOD => self.rpc.label(), + LABEL_OPERATION => self.rpc.operation(), LABEL_OUTCOME => outcome.label() ) .record(self.started.elapsed()); @@ -436,7 +420,7 @@ impl RoutedRequestTimer { impl Drop for RoutedRequestTimer { fn drop(&mut self) { - self.record(PeerOutcome::ClientError, Code::Cancelled); + self.record(AttemptOutcome::LocalError, Code::Cancelled); } } @@ -504,14 +488,10 @@ mod tests { #[test] fn describe_and_initialize_exports_capacity_and_zero_series() { let metrics = MetricsCapture::install(); - describe_and_initialize(RelayCapacity { - global: 256, - per_sandbox: 32, - }); + describe_and_initialize(RelayCapacity { per_replica: 256 }); for (series, expected) in [ ("openshell_server_relay_pending_capacity", 256), - ("openshell_server_relay_pending_per_sandbox_capacity", 32), ("openshell_server_supervisor_sessions", 0), ("openshell_server_relay_pending", 0), ( @@ -526,10 +506,10 @@ mod tests { ] { assert_eq!(metrics.value(series), Some(expected), "{series}"); } - for target in ["ssh", "tcp"] { + for kind in ["ssh", "tcp"] { for route in ["local", "peer"] { let series = format!( - "openshell_server_routed_request_attempts_total{{operation=\"relay\",route=\"{route}\",target=\"{target}\",outcome=\"success\",grpc_code=\"ok\"}}" + "openshell_server_routed_request_attempts_total{{operation=\"relay\",route=\"{route}\",relay_kind=\"{kind}\",outcome=\"success\",grpc_code=\"ok\"}}" ); assert_eq!(metrics.value(&series), Some(0), "{series}"); } @@ -540,7 +520,7 @@ mod tests { "get_sandbox_provider_status", ] { let series = format!( - "openshell_server_routed_request_attempts_total{{operation=\"{operation}\",route=\"peer\",target=\"none\",outcome=\"success\",grpc_code=\"ok\"}}" + "openshell_server_routed_request_attempts_total{{operation=\"{operation}\",route=\"peer\",relay_kind=\"none\",outcome=\"success\",grpc_code=\"ok\"}}" ); assert_eq!(metrics.value(&series), Some(0), "{series}"); } @@ -554,23 +534,20 @@ mod tests { #[test] fn routed_attempts_have_seven_bounded_success_series_and_keep_counts_on_initialize() { let metrics = MetricsCapture::install(); - for target in RelayTarget::ALL { + for kind in RelayKind::ALL { for route in RelayRoute::ALL { - RoutedRequestTimer::relay(target, route).finish(&Ok::<(), Status>(())); + RoutedRequestTimer::relay(kind, route).finish(&Ok::<(), Status>(())); } } - RoutedRequestTimer::relay(RelayTarget::Tcp, RelayRoute::Peer).finish(&Ok::<(), Status>(())); - describe_and_initialize(RelayCapacity { - global: 256, - per_sandbox: 32, - }); + RoutedRequestTimer::relay(RelayKind::Tcp, RelayRoute::Peer).finish(&Ok::<(), Status>(())); + describe_and_initialize(RelayCapacity { per_replica: 256 }); - for target in ["ssh", "tcp"] { + for kind in ["ssh", "tcp"] { for route in ["local", "peer"] { let series = format!( - "openshell_server_routed_request_attempts_total{{operation=\"relay\",route=\"{route}\",target=\"{target}\",outcome=\"success\",grpc_code=\"ok\"}}" + "openshell_server_routed_request_attempts_total{{operation=\"relay\",route=\"{route}\",relay_kind=\"{kind}\",outcome=\"success\",grpc_code=\"ok\"}}" ); - let expected = if target == "tcp" && route == "peer" { + let expected = if kind == "tcp" && route == "peer" { 2 } else { 1 @@ -598,7 +575,7 @@ mod tests { histogram!(RELAY_CLAIM_DURATION_SECONDS).record(sample); histogram!( PEER_REQUEST_DURATION_SECONDS, - LABEL_METHOD => "PeerRelay", + LABEL_OPERATION => "relay", LABEL_OUTCOME => "success" ) .record(sample); @@ -682,23 +659,23 @@ mod tests { drop(relay); assert_eq!( metrics.value( - "openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"peer\",target=\"ssh\",outcome=\"rpc_error\",grpc_code=\"resource_exhausted\"}" + "openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"peer\",relay_kind=\"ssh\",outcome=\"remote_error\",grpc_code=\"resource_exhausted\"}" ), Some(1) ); assert_eq!( metrics.value( - "openshell_server_peer_request_duration_seconds_count{method=\"PeerRelay\",outcome=\"rpc_error\"}" + "openshell_server_peer_request_duration_seconds_count{operation=\"relay\",outcome=\"remote_error\"}" ), Some(1) ); let mut endpoint = RoutedRequestTimer::start(PeerRpc::ReportEndpointStatus); - endpoint.client_error(&Status::unavailable("x")); + endpoint.local_error(&Status::unavailable("x")); drop(endpoint); assert_eq!( metrics.value( - "openshell_server_routed_request_attempts_total{operation=\"report_endpoint_status\",route=\"peer\",target=\"none\",outcome=\"client_error\",grpc_code=\"unavailable\"}" + "openshell_server_routed_request_attempts_total{operation=\"report_endpoint_status\",route=\"peer\",relay_kind=\"none\",outcome=\"local_error\",grpc_code=\"unavailable\"}" ), Some(1) ); @@ -708,7 +685,7 @@ mod tests { drop(provider_status); assert_eq!( metrics.value( - "openshell_server_routed_request_attempts_total{operation=\"get_sandbox_provider_status\",route=\"peer\",target=\"none\",outcome=\"success\",grpc_code=\"ok\"}" + "openshell_server_routed_request_attempts_total{operation=\"get_sandbox_provider_status\",route=\"peer\",relay_kind=\"none\",outcome=\"success\",grpc_code=\"ok\"}" ), Some(1) ); @@ -720,7 +697,7 @@ mod tests { drop(RoutedRequestTimer::start(PeerRpc::ReportProviderReadiness)); assert_eq!( metrics.value( - "openshell_server_routed_request_attempts_total{operation=\"report_provider_readiness\",route=\"peer\",target=\"none\",outcome=\"client_error\",grpc_code=\"cancelled\"}" + "openshell_server_routed_request_attempts_total{operation=\"report_provider_readiness\",route=\"peer\",relay_kind=\"none\",outcome=\"local_error\",grpc_code=\"cancelled\"}" ), Some(1) ); @@ -731,25 +708,25 @@ mod tests { let metrics = MetricsCapture::install(); let mut timer = RoutedRequestTimer::start(PeerRpc::Relay); timer.finish(&Ok::<(), Status>(())); - timer.client_error(&Status::unavailable("x")); + timer.local_error(&Status::unavailable("x")); drop(timer); assert_eq!( metrics.value( - "openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"peer\",target=\"ssh\",outcome=\"success\",grpc_code=\"ok\"}" + "openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"peer\",relay_kind=\"ssh\",outcome=\"success\",grpc_code=\"ok\"}" ), Some(1) ); - assert!(!metrics.render().contains("outcome=\"client_error\"")); + assert!(!metrics.render().contains("outcome=\"local_error\"")); } #[test] fn local_relay_failure_records_status_without_peer_latency() { let metrics = MetricsCapture::install(); - let mut timer = RoutedRequestTimer::relay(RelayTarget::Tcp, RelayRoute::Local); + let mut timer = RoutedRequestTimer::relay(RelayKind::Tcp, RelayRoute::Local); timer.finish(&Err::<(), _>(Status::resource_exhausted("capacity"))); drop(timer); assert_eq!( - metrics.value("openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"local\",target=\"tcp\",outcome=\"client_error\",grpc_code=\"resource_exhausted\"}"), + metrics.value("openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"local\",relay_kind=\"tcp\",outcome=\"local_error\",grpc_code=\"resource_exhausted\"}"), Some(1) ); assert!(!metrics.render().contains(PEER_REQUEST_DURATION_SECONDS)); diff --git a/crates/openshell-server/src/supervisor_session.rs b/crates/openshell-server/src/supervisor_session.rs index ef1c34f11..494e0730f 100644 --- a/crates/openshell-server/src/supervisor_session.rs +++ b/crates/openshell-server/src/supervisor_session.rs @@ -28,7 +28,7 @@ use openshell_core::transport_errors::is_expected_transport_close_status; use crate::ServerState; use crate::auth::principal::Principal; use crate::gateway_metrics::{ - self, GaugeSlot, PeerRpc, RelayCapacity, RelayRejection, RelayRoute, RelayTarget, + self, GaugeSlot, PeerRpc, RelayCapacity, RelayKind, RelayRejection, RelayRoute, RoutedRequestTimer, }; use crate::grpc::provider_readiness::ProviderReadinessEvidence; @@ -52,10 +52,10 @@ const MAX_PENDING_RELAYS: usize = 256; /// consume the entire global budget. Sits above the SSH-tunnel per-sandbox /// cap (20) so tunnel-specific limits still fire first for that caller. const MAX_PENDING_RELAYS_PER_SANDBOX: usize = 32; -/// The relay caps above, published as capacity gauges when the metrics recorder is installed. +/// The replica relay cap above, published as a capacity gauge when the metrics recorder is +/// installed. pub(crate) const RELAY_CAPACITY: RelayCapacity = RelayCapacity { - global: MAX_PENDING_RELAYS, - per_sandbox: MAX_PENDING_RELAYS_PER_SANDBOX, + per_replica: MAX_PENDING_RELAYS, }; const PEER_TLS_CA_FILE_ENV: &str = "OPENSHELL_PEER_TLS_CA_FILE"; const PEER_TLS_CERT_FILE_ENV: &str = "OPENSHELL_PEER_TLS_CERT_FILE"; @@ -1429,7 +1429,7 @@ pub(crate) async fn forward_provider_readiness_to_owner( let mut timer = RoutedRequestTimer::start(PeerRpc::ReportProviderReadiness); let mut client = peer_rpc_client(state, &owner.owner_peer_endpoint) .await - .inspect_err(|status| timer.client_error(status))?; + .inspect_err(|status| timer.local_error(status))?; let result = client.peer_report_provider_readiness(request).await; timer.finish(&result); result.map(Response::into_inner).inspect_err(|_| { @@ -1447,7 +1447,7 @@ pub(crate) async fn forward_endpoint_status_to_owner( let mut timer = RoutedRequestTimer::start(PeerRpc::ReportEndpointStatus); let mut client = peer_rpc_client(state, &owner.owner_peer_endpoint) .await - .inspect_err(|status| timer.client_error(status))?; + .inspect_err(|status| timer.local_error(status))?; let result = client.peer_report_endpoint_status(request).await; timer.finish(&result); result.map(Response::into_inner).inspect_err(|_| { @@ -1465,7 +1465,7 @@ pub(crate) async fn forward_provider_status_query_to_owner( let mut timer = RoutedRequestTimer::start(PeerRpc::GetSandboxProviderStatus); let mut client = peer_rpc_client(state, &owner.owner_peer_endpoint) .await - .inspect_err(|status| timer.client_error(status))?; + .inspect_err(|status| timer.local_error(status))?; let result = client.peer_get_sandbox_provider_status(request).await; timer.finish(&result); result.map(Response::into_inner).inspect_err(|_| { @@ -1496,14 +1496,68 @@ pub async fn open_routed_relay_with_target( open_routed_relay_with_message(state, sandbox_id, relay_open, session_wait_timeout).await } -fn relay_target(relay_open: &RelayOpen) -> RelayTarget { +fn relay_kind(relay_open: &RelayOpen) -> RelayKind { // An absent target means SSH for compatibility with older callers. match relay_open.target.as_ref() { - Some(relay_open::Target::Ssh(_)) | None => RelayTarget::Ssh, - Some(relay_open::Target::Tcp(_)) => RelayTarget::Tcp, + Some(relay_open::Target::Ssh(_)) | None => RelayKind::Ssh, + Some(relay_open::Target::Tcp(_)) => RelayKind::Tcp, } } +/// Hand the caller a receiver that forwards the local relay's claim result unchanged, and count +/// the attempt when the supervisor claims the relay, the claim window closes, or the caller +/// gives up. The owner of a `PeerRelay` answers on the same events, so a relay outcome means +/// the same thing on the local and peer routes. +/// +/// The window is anchored before the caller can start its own wait, so a caller that times out +/// with the same 10 s budget is counted as an unclaimed relay, not as a cancellation. A claim +/// that lands just after the caller gave up still succeeds in the registry; the forwarder then +/// drops the stream and the supervisor sees it close, as when a caller drops right after a claim. +fn track_local_relay_claim( + mut timer: RoutedRequestTimer, + mut claimed: oneshot::Receiver>, +) -> oneshot::Receiver> { + let claim_deadline = tokio::time::Instant::now() + RELAY_PENDING_TIMEOUT; + let unclaimed = || Status::deadline_exceeded("relay was not claimed in time"); + let (mut forward_tx, forward_rx) = oneshot::channel(); + tokio::spawn(async move { + let claim_window = tokio::time::sleep_until(claim_deadline); + tokio::pin!(claim_window); + let mut window_open = true; + let result = loop { + tokio::select! { + biased; + result = &mut claimed => break result, + () = &mut claim_window, if window_open => { + window_open = false; + timer.local_error(&unclaimed()); + } + () = forward_tx.closed() => { + if tokio::time::Instant::now() >= claim_deadline { + timer.local_error(&unclaimed()); + } + // Otherwise dropping the timer records the cancellation. + return; + } + } + }; + match result { + Ok(claim) => { + timer.finish(&claim); + let _ = forward_tx.send(claim); + } + // The registry dropped the relay without an answer: it expired (reaper or a late + // claim), or the registry was torn down. Dropping `forward_tx` passes the same closed + // channel on to the caller. + Err(_) if tokio::time::Instant::now() >= claim_deadline => { + timer.local_error(&unclaimed()); + } + Err(_) => timer.local_error(&Status::unavailable("relay channel dropped")), + } + }); + forward_rx +} + pub async fn open_routed_relay_with_message( state: &Arc, sandbox_id: &str, @@ -1522,14 +1576,19 @@ pub async fn open_routed_relay_with_message( let owner_index = SupervisorOwnerIndex::new(state.store.clone(), OWNER_TTL); loop { if state.supervisor_sessions.has_session(sandbox_id) { - let mut timer = RoutedRequestTimer::relay(relay_target(&relay_open), RelayRoute::Local); + let mut timer = + RoutedRequestTimer::relay(relay_kind(&relay_open), RelayRoute::Local); let result = state .supervisor_sessions .open_relay_with_message_until(sandbox_id, relay_open.clone(), deadline, false) .await; - timer.finish(&result); + if let Err(status) = &result { + timer.local_error(status); + } match result { - Ok(relay) => return Ok(relay), + Ok((channel_id, relay_rx)) => { + return Ok((channel_id, track_local_relay_claim(timer, relay_rx))); + } Err(status) if status.code() == tonic::Code::Unavailable => { // The session can migrate after `has_session` but before // RelayOpen reaches its sender. Fall through and reread the @@ -1667,19 +1726,19 @@ async fn connect_peer_relay( sandbox_id: &str, relay_open: RelayOpen, ) -> Result { - let mut timer = RoutedRequestTimer::relay(relay_target(&relay_open), RelayRoute::Peer); + let mut timer = RoutedRequestTimer::relay(relay_kind(&relay_open), RelayRoute::Peer); let token = state .peer_routes .peer_token() .await - .inspect_err(|s| timer.client_error(s))?; + .inspect_err(|s| timer.local_error(s))?; let channel = state .peer_routes .channel(owner_peer_endpoint) .await - .inspect_err(|s| timer.client_error(s))?; + .inspect_err(|s| timer.local_error(s))?; let interceptor = PeerAuthInterceptor::new(&token, &state.replica_id) - .inspect_err(|s| timer.client_error(s))?; + .inspect_err(|s| timer.local_error(s))?; let mut client = open_shell_client::OpenShellClient::with_interceptor(channel, interceptor); let (out_tx, out_rx) = mpsc::channel::(16); @@ -1693,7 +1752,7 @@ async fn connect_peer_relay( }) .await .map_err(|_| Status::internal("failed to initialize peer relay stream")) - .inspect_err(|s| timer.client_error(s))?; + .inspect_err(|s| timer.local_error(s))?; let result = client.peer_relay(ReceiverStream::new(out_rx)).await; // Record the owner's code before the remap below hides it as `unavailable`. @@ -3979,7 +4038,7 @@ mod tests { target, ..peer_relay_open(&Uuid::new_v4().to_string()) }; - let (channel_id, _relay_rx) = open_routed_relay_with_message( + let (channel_id, relay_rx) = open_routed_relay_with_message( &state, "sbx-routing", relay_open.clone(), @@ -3992,12 +4051,17 @@ mod tests { rx.recv().await.unwrap().payload, Some(gateway_message::Payload::RelayOpen(relay_open)) ); - assert_eq!( - metrics.value(&format!( - "openshell_server_routed_request_attempts_total{{operation=\"relay\",route=\"local\",target=\"{label}\",outcome=\"success\",grpc_code=\"ok\"}}" - )), - Some(expected) + let series = format!( + "openshell_server_routed_request_attempts_total{{operation=\"relay\",route=\"local\",relay_kind=\"{label}\",outcome=\"success\",grpc_code=\"ok\"}}" ); + // Enqueued but not claimed yet: success waits for the supervisor. + assert_eq!(metrics.value(&series).unwrap_or(0), expected - 1); + let _claimed = state + .supervisor_sessions + .claim_relay(&channel_id, None) + .unwrap(); + relay_rx.await.unwrap().unwrap(); + assert_eq!(metrics.value(&series), Some(expected)); } let rendered = metrics.render(); assert!(!rendered.contains("route=\"peer\"")); @@ -4032,7 +4096,7 @@ mod tests { .expect_err("the local supervisor disconnected"); assert_eq!( metrics.value( - "openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"local\",target=\"ssh\",outcome=\"client_error\",grpc_code=\"unavailable\"}" + "openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"local\",relay_kind=\"ssh\",outcome=\"local_error\",grpc_code=\"unavailable\"}" ), Some(1) ); @@ -4067,13 +4131,13 @@ mod tests { .expect("the peer accepts after the local supervisor disconnects"); for route in ["local", "peer"] { let (outcome, code) = if route == "local" { - ("client_error", "unavailable") + ("local_error", "unavailable") } else { ("success", "ok") }; assert_eq!( metrics.value(&format!( - "openshell_server_routed_request_attempts_total{{operation=\"relay\",route=\"{route}\",target=\"ssh\",outcome=\"{outcome}\",grpc_code=\"{code}\"}}" + "openshell_server_routed_request_attempts_total{{operation=\"relay\",route=\"{route}\",relay_kind=\"ssh\",outcome=\"{outcome}\",grpc_code=\"{code}\"}}" )), Some(1) ); @@ -4111,7 +4175,7 @@ mod tests { .unwrap(); assert_eq!( metrics.value(&format!( - "openshell_server_routed_request_attempts_total{{operation=\"relay\",route=\"peer\",target=\"{label}\",outcome=\"success\",grpc_code=\"ok\"}}" + "openshell_server_routed_request_attempts_total{{operation=\"relay\",route=\"peer\",relay_kind=\"{label}\",outcome=\"success\",grpc_code=\"ok\"}}" )), Some(1) ); @@ -4149,12 +4213,12 @@ mod tests { .await .expect_err("the peer rejects every attempt"); let attempts = metrics - .value("openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"peer\",target=\"ssh\",outcome=\"rpc_error\",grpc_code=\"unavailable\"}") + .value("openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"peer\",relay_kind=\"ssh\",outcome=\"remote_error\",grpc_code=\"unavailable\"}") .unwrap(); assert!(attempts > 1, "the routing loop must have retried"); assert_eq!( metrics.value( - "openshell_server_peer_request_duration_seconds_count{method=\"PeerRelay\",outcome=\"rpc_error\"}" + "openshell_server_peer_request_duration_seconds_count{operation=\"relay\",outcome=\"remote_error\"}" ), Some(attempts) ); @@ -4209,7 +4273,7 @@ mod tests { ); drop(setup); assert_eq!( - metrics.value("openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"local\",target=\"ssh\",outcome=\"client_error\",grpc_code=\"cancelled\"}"), + metrics.value("openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"local\",relay_kind=\"ssh\",outcome=\"local_error\",grpc_code=\"cancelled\"}"), Some(1) ); assert!( @@ -4219,6 +4283,123 @@ mod tests { ); } + const LOCAL_RELAY_SERIES: &str = "openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"local\",relay_kind=\"ssh\""; + + fn local_relay_outcome(metrics: &MetricsCapture, outcome: &str, code: &str) -> Option { + metrics.value(&format!( + "{LOCAL_RELAY_SERIES},outcome=\"{outcome}\",grpc_code=\"{code}\"}}" + )) + } + + type RelayClaim = Result; + + fn tracked_local_relay() -> (oneshot::Sender, oneshot::Receiver) { + let (claim_tx, claim_rx) = oneshot::channel(); + let timer = RoutedRequestTimer::relay(RelayKind::Ssh, RelayRoute::Local); + (claim_tx, track_local_relay_claim(timer, claim_rx)) + } + + #[tokio::test] + async fn local_relay_claim_records_success_once() { + let metrics = MetricsCapture::install(); + let (claim_tx, forwarded) = tracked_local_relay(); + let (stream, _peer) = tokio::io::duplex(64); + claim_tx.send(Ok(stream)).unwrap(); + forwarded.await.unwrap().unwrap(); + assert_eq!(local_relay_outcome(&metrics, "success", "ok"), Some(1)); + assert!(!metrics.render().contains("local_error")); + } + + #[tokio::test(start_paused = true)] + async fn local_relay_unclaimed_in_window_records_deadline_exceeded_once() { + let metrics = MetricsCapture::install(); + let (claim_tx, forwarded) = tracked_local_relay(); + tokio::time::sleep(RELAY_PENDING_TIMEOUT + Duration::from_millis(1)).await; + assert_eq!( + local_relay_outcome(&metrics, "local_error", "deadline_exceeded"), + Some(1) + ); + // A late answer still reaches the caller without a second count. + claim_tx + .send(Err(Status::unavailable("supervisor gone"))) + .unwrap(); + let answer = forwarded.await.unwrap(); + assert_eq!(answer.unwrap_err().code(), tonic::Code::Unavailable); + assert_eq!( + local_relay_outcome(&metrics, "local_error", "unavailable"), + None + ); + } + + #[tokio::test(start_paused = true)] + async fn local_relay_caller_timeout_counts_as_unclaimed_not_cancelled() { + let metrics = MetricsCapture::install(); + let (claim_tx, forwarded) = tracked_local_relay(); + // Callers wait with the same 10 s budget and drop the receiver when it runs out. + assert!( + tokio::time::timeout(RELAY_PENDING_TIMEOUT, forwarded) + .await + .is_err() + ); + while !claim_tx.is_closed() { + tokio::task::yield_now().await; + } + assert_eq!( + local_relay_outcome(&metrics, "local_error", "deadline_exceeded"), + Some(1) + ); + assert_eq!( + local_relay_outcome(&metrics, "local_error", "cancelled"), + None + ); + } + + #[tokio::test] + async fn local_relay_failed_by_registry_records_its_status() { + let metrics = MetricsCapture::install(); + let (claim_tx, forwarded) = tracked_local_relay(); + claim_tx + .send(Err(Status::unavailable("supervisor session disconnected"))) + .unwrap(); + assert_eq!( + forwarded.await.unwrap().unwrap_err().code(), + tonic::Code::Unavailable + ); + assert_eq!( + local_relay_outcome(&metrics, "local_error", "unavailable"), + Some(1) + ); + } + + #[tokio::test] + async fn local_relay_dropped_by_registry_closes_the_caller_channel() { + let metrics = MetricsCapture::install(); + let (claim_tx, forwarded) = tracked_local_relay(); + drop(claim_tx); + assert!( + forwarded.await.is_err(), + "the caller sees the same closed channel" + ); + assert_eq!( + local_relay_outcome(&metrics, "local_error", "unavailable"), + Some(1) + ); + } + + #[tokio::test] + async fn local_relay_abandoned_by_caller_records_cancelled() { + let metrics = MetricsCapture::install(); + let (claim_tx, forwarded) = tracked_local_relay(); + drop(forwarded); + while !claim_tx.is_closed() { + tokio::task::yield_now().await; + } + assert_eq!( + local_relay_outcome(&metrics, "local_error", "cancelled"), + Some(1) + ); + } + #[tokio::test] async fn peer_relay_metrics_keep_owner_code_before_unavailable_remap() { let metrics = MetricsCapture::install(); @@ -4232,19 +4413,19 @@ mod tests { assert_eq!(err.code(), tonic::Code::Unavailable); assert_eq!( metrics.value( - "openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"peer\",target=\"ssh\",outcome=\"rpc_error\",grpc_code=\"resource_exhausted\"}" + "openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"peer\",relay_kind=\"ssh\",outcome=\"remote_error\",grpc_code=\"resource_exhausted\"}" ), Some(1) ); assert_eq!( metrics.value( - "openshell_server_peer_request_duration_seconds_count{method=\"PeerRelay\",outcome=\"rpc_error\"}" + "openshell_server_peer_request_duration_seconds_count{operation=\"relay\",outcome=\"remote_error\"}" ), Some(1) ); let rendered = metrics.render(); assert!(rendered.contains( - "openshell_server_peer_request_duration_seconds_bucket{method=\"PeerRelay\",outcome=\"rpc_error\",le=\"0.001\"}" + "openshell_server_peer_request_duration_seconds_bucket{operation=\"relay\",outcome=\"remote_error\",le=\"0.001\"}" )); assert!( !state @@ -4274,14 +4455,14 @@ mod tests { .expect("the owner accepted the relay"); assert_eq!( metrics.value( - "openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"peer\",target=\"ssh\",outcome=\"success\",grpc_code=\"ok\"}" + "openshell_server_routed_request_attempts_total{operation=\"relay\",route=\"peer\",relay_kind=\"ssh\",outcome=\"success\",grpc_code=\"ok\"}" ), Some(1) ); } #[tokio::test] - async fn peer_forward_metrics_record_client_error_when_owner_unreachable() { + async fn peer_forward_metrics_record_local_error_when_owner_unreachable() { let metrics = MetricsCapture::install(); let state = crate::grpc::test_support::test_server_state().await; seed_peer_token(&state); @@ -4298,19 +4479,19 @@ mod tests { assert_eq!(err.code(), tonic::Code::Unavailable); assert_eq!( metrics.value( - "openshell_server_routed_request_attempts_total{operation=\"get_sandbox_provider_status\",route=\"peer\",target=\"none\",outcome=\"client_error\",grpc_code=\"unavailable\"}" + "openshell_server_routed_request_attempts_total{operation=\"get_sandbox_provider_status\",route=\"peer\",relay_kind=\"none\",outcome=\"local_error\",grpc_code=\"unavailable\"}" ), Some(1) ); assert!( !metrics .render() - .contains("method=\"PeerGetSandboxProviderStatus\",outcome=\"rpc_error\"") + .contains("operation=\"get_sandbox_provider_status\",outcome=\"remote_error\"") ); } #[tokio::test] - async fn peer_forward_metrics_record_owner_rpc_error_code() { + async fn peer_forward_metrics_record_owner_remote_error_code() { let metrics = MetricsCapture::install(); let state = crate::grpc::test_support::test_server_state().await; seed_peer_token(&state); @@ -4330,7 +4511,7 @@ mod tests { assert_eq!(err.code(), tonic::Code::PermissionDenied); assert_eq!( metrics.value( - "openshell_server_routed_request_attempts_total{operation=\"report_endpoint_status\",route=\"peer\",target=\"none\",outcome=\"rpc_error\",grpc_code=\"permission_denied\"}" + "openshell_server_routed_request_attempts_total{operation=\"report_endpoint_status\",route=\"peer\",relay_kind=\"none\",outcome=\"remote_error\",grpc_code=\"permission_denied\"}" ), Some(1) ); @@ -4355,7 +4536,7 @@ mod tests { .expect("the owner accepted the report"); assert_eq!( metrics.value( - "openshell_server_routed_request_attempts_total{operation=\"report_provider_readiness\",route=\"peer\",target=\"none\",outcome=\"success\",grpc_code=\"ok\"}" + "openshell_server_routed_request_attempts_total{operation=\"report_provider_readiness\",route=\"peer\",relay_kind=\"none\",outcome=\"success\",grpc_code=\"ok\"}" ), Some(1) ); diff --git a/docs/kubernetes/high-availability.mdx b/docs/kubernetes/high-availability.mdx index c25b452d4..0153df050 100644 --- a/docs/kubernetes/high-availability.mdx +++ b/docs/kubernetes/high-availability.mdx @@ -202,8 +202,8 @@ configuration, and access control. | Sessions per replica | `openshell_server_supervisor_sessions` | Placement and skew. Do not scale on it. | | Relay utilization | `openshell_server_relay_pending`, `openshell_server_relay_pending_capacity` | Graph it. It is transient and usually near 0, so alert on rejections instead. | | Relay rejections | `openshell_server_relay_rejected_total` | Alert on any increase. | -| Relay claims | `openshell_server_relay_claim_duration_seconds`, `openshell_server_relay_expired_total` | Supervisor connect-back time for exec, SSH, forwarding, and service traffic, including relays requested through peers. A high 99th percentile while rejections stay flat points at the supervisor or its node, not at relay capacity. Relays not claimed within 10 seconds count only as expired, so alert on any increase. | -| Routed requests | `openshell_server_routed_request_attempts_total`, `openshell_server_peer_request_duration_seconds` | Local/peer relay setup mix and outbound peer rate, failures, and latency. Counts completed or cancelled attempts, including retries. Do not use as an HPA target. | +| Relay claims | `openshell_server_relay_claim_duration_seconds`, `openshell_server_relay_expired_total` | Supervisor connect-back time for exec, SSH, forwarding, and service traffic, including relays requested through peers. A high 99th percentile while rejections stay flat points at the supervisor or its node, not at relay capacity. Relays not claimed within 10 seconds are missing from the histogram and count as expired, so alert on any increase. | +| Routed requests | `openshell_server_routed_request_attempts_total`, `openshell_server_peer_request_duration_seconds` | Local/peer relay setup mix and outbound peer rate, failures, and latency. Counts completed or cancelled attempts, including retries. Spikes of `grpc_code="unavailable"` during rollouts are expected. Do not use as an HPA target. | These queries assume that Prometheus labels each series with `namespace` and `pod`: @@ -219,8 +219,8 @@ sum by (namespace, pod, reason) (increase(openshell_server_relay_rejected_total[ # 99th percentile time for supervisors to claim a relay on each replica. histogram_quantile(0.99, sum by (namespace, pod, le) (rate(openshell_server_relay_claim_duration_seconds_bucket[5m]))) -# Completed or cancelled relay setup attempts, by replica, target, and route. -sum by (namespace, pod, target, route) (rate(openshell_server_routed_request_attempts_total{operation="relay"}[5m])) +# Completed or cancelled relay setup attempts, by replica, relay kind, and route. +sum by (namespace, pod, relay_kind, route) (rate(openshell_server_routed_request_attempts_total{operation="relay"}[5m])) # Session skew across replicas. 1 means balanced. Starting and stopping pods # raise it during rollouts. @@ -232,14 +232,14 @@ sum by (operation) (rate(openshell_server_routed_request_attempts_total{route="p / sum by (operation) (rate(openshell_server_routed_request_attempts_total{route="peer"}[5m])) # 99th percentile peer request latency across the fleet. -histogram_quantile(0.99, sum by (le, method) (rate(openshell_server_peer_request_duration_seconds_bucket{outcome="success"}[5m]))) +histogram_quantile(0.99, sum by (le, operation) (rate(openshell_server_peer_request_duration_seconds_bucket{outcome="success"}[5m]))) ``` Relay capacity is used on the replica that owns the sandbox's supervisor session, including relays that other replicas request through peers, so adding replicas does not relieve a busy owner. A relay whose client gave up stays counted for up to about 40 seconds, the 10-second claim timeout plus the -30-second cleanup interval. Peer request counts include retries during +30-second cleanup interval. Routed request counts include retries during rollouts. ## Scale the Gateway diff --git a/docs/observability/gateway-metrics.mdx b/docs/observability/gateway-metrics.mdx index afdd910c1..aa849977d 100644 --- a/docs/observability/gateway-metrics.mdx +++ b/docs/observability/gateway-metrics.mdx @@ -163,7 +163,6 @@ Relays: |---|---|---|---| | `openshell_server_relay_pending` | Gauge | None | Relay channels on this replica waiting for the supervisor to connect back, including channels opened for peer replicas. A channel whose caller gave up stays counted until it is claimed or cleaned up, up to about 40 seconds. | | `openshell_server_relay_pending_capacity` | Gauge | None | Pending relay limit per replica, `256`. | -| `openshell_server_relay_pending_per_sandbox_capacity` | Gauge | None | Pending relay limit per sandbox on one replica, `32`. | | `openshell_server_relay_rejected_total` | Counter | `reason` | Relay opens rejected at a limit. Rejected relays fail the client request, so use this counter, not client error codes, to detect saturation. | | `openshell_server_relay_expired_total` | Counter | None | Pending relays dropped because the supervisor did not connect back within 10 seconds. | | `openshell_server_relay_claim_duration_seconds` | Histogram | None | Time from opening a relay to the supervisor claiming it. | @@ -173,15 +172,17 @@ that owns a sandbox's supervisor session: | Metric | Type | Labels | Description | |---|---|---|---| -| `openshell_server_routed_request_attempts_total` | Counter | `operation`, `route`, `target`, `outcome`, `grpc_code` | Local relay setup and outbound peer attempts, counted once at completion or cancellation. Each retry counts separately. | -| `openshell_server_peer_request_duration_seconds` | Histogram | `method`, `outcome` | Latency of those requests. For `PeerRelay`, until the owner's supervisor claimed the relay. | +| `openshell_server_routed_request_attempts_total` | Counter | `operation`, `route`, `relay_kind`, `outcome`, `grpc_code` | Local relay setup and outbound peer attempts, counted once when they finish or are cancelled. Each retry counts separately. | +| `openshell_server_peer_request_duration_seconds` | Histogram | `operation`, `outcome` | Latency of outbound peer requests only (`route="peer"`). For relays, until the owner's supervisor claimed the relay. | The labels take these values: - `operation` is `relay`, `report_provider_readiness`, `report_endpoint_status`, or `get_sandbox_provider_status`. Only relay setup includes the local route; - the other operations count outbound peer calls, not local API handling. -- `target` is `ssh` or `tcp` for relays, and `none` for other operations. An + the other operations count outbound peer calls, not local API handling. On the + owning replica, `openshell_server_grpc_requests_total` records the matching + `method`, such as `PeerRelay` for `relay`. +- `relay_kind` is `ssh` or `tcp` for relays, and `none` for other operations. An omitted relay target counts as `ssh`, matching the protocol's compatibility behavior. TCP hostnames and ports are not labels. - `route` is `local` when this replica tries its own supervisor session, or @@ -191,33 +192,29 @@ The labels take these values: they appear in that replica's gRPC request counter instead. This counter does not measure active connections, unique client requests, or unfinished attempts. - `reason` is `replica_capacity` when the gateway replica's pending-relay budget - is full, or `sandbox_capacity` when one sandbox's budget on that replica is full. -- `method` is `PeerRelay`, `PeerReportProviderReadiness`, - `PeerReportEndpointStatus`, or `PeerGetSandboxProviderStatus`, the same - values as the `method` label of `openshell_server_grpc_requests_total` on - the owning replica. -- `outcome` is `success`, `client_error`, or `rpc_error`. Local relay success - means the open was enqueued to the supervisor; peer relay success means the - supervisor claimed it and the peer returned response headers. Neither measures - the lifetime of the data stream. `client_error` means local relay setup failed, - or a peer request failed before it reached the owner (token, connection, or - stream setup), or the caller cancelled it at any point, which records - `grpc_code="cancelled"` even when the owner was already handling the request. - `rpc_error` means the call to the owner failed, either with the owner's gRPC - status or with a transport error on an already open peer connection. + is full, or `sandbox_capacity` when one sandbox's budget on that replica, 32 + pending relays, is full. +- `outcome` is `success`, `local_error`, or `remote_error`. A relay succeeds + when the supervisor claims it; this does not measure the data stream. + `local_error` means the attempt failed on this replica, and `remote_error` + means the owning replica returned an error or the open connection to it + failed. On `route="local"` this replica is the owner, so failures that appear + as `remote_error` on the peer route, such as an unclaimed relay + (`deadline_exceeded`) or a full relay budget (`resource_exhausted`), appear as + `local_error`. A cancelled attempt records `local_error` with + `grpc_code="cancelled"`, even when the owner was already handling it. - `grpc_code` is the lowercase gRPC status name, such as `ok`, `unavailable`, - or `resource_exhausted`, from local setup or the peer RPC. For `PeerRelay`, the - gateway then reports a failure to the client as `UNAVAILABLE`. The other - RPCs return the owner's status to the client unchanged. The label is not - named `code` because `openshell_server_grpc_requests_total` uses that name - for the numeric status code. + or `resource_exhausted`, from local setup, the local supervisor claim, or the + peer RPC. For a peer relay, the gateway then reports a failure to the client + as `UNAVAILABLE`. The other RPCs return the owner's status to the client + unchanged. The label is not named `code` because + `openshell_server_grpc_requests_total` uses that name for the numeric status + code. Gauges and counters exist from startup. They start at `0`, except -`openshell_server_relay_pending_capacity` and -`openshell_server_relay_pending_per_sandbox_capacity`, which start at their -limits. `openshell_server_routed_request_attempts_total` starts with seven -`outcome="success", grpc_code="ok"` series: four relay target/route combinations -and three other peer operations with `target="none"`. Other outcomes and codes, +`openshell_server_relay_pending_capacity`, which starts at its limit. `openshell_server_routed_request_attempts_total` starts with seven +`outcome="success", grpc_code="ok"` series: four relay kind and route +combinations, and three other peer operations with `relay_kind="none"`. Other outcomes and codes, and every histogram, appear after their first sample, so write alert expressions that tolerate absent series.