patch 0.6.2

This commit is contained in:
Hydra
2026-08-10 16:27:29 +03:00
parent 1101b4e309
commit 698893ecae
663 changed files with 60390 additions and 9007 deletions
+48 -4
View File
@@ -110,14 +110,58 @@ INTERNAL_TOKEN=change-me-32-byte-random-hex
# ─── Host operations from the container (optional) ────────
# The edge (routing/TLS) runs as the `edge` container; app containers run via
# the mounted docker socket. For the few HOST-OS ops a container can't do to its
# host (freeing a foreign proxy off :80/443, host system config), the API reaches
# the host over SSH via host.docker.internal (internal bridge, not the public IP).
# `openship up` provisions the key + these vars automatically; set them by hand
# only for a raw `docker compose` install that needs host ops.
# host (freeing a foreign proxy off :80/443, host system config, the mail engine,
# writing a catalog app's generated config file), the API reaches the host over
# SSH via host.docker.internal (internal bridge, not the public IP).
#
# Leaving this unset does NOT degrade to running those ops locally: inside a
# container "locally" is the container's own filesystem, so they REFUSE instead,
# naming this channel. Ordinary deploys are unaffected — they go through the
# docker socket. On a CLI install `openship doctor` reports which state you're
# in; on a raw `docker compose` install it can't see the stack, so the signal is
# the api's boot log (a `!!! HOST CONTROL …` banner, or silence when it's fine).
#
# `openship up` provisions all of it. By hand, on docker/docker-compose.yml, the
# vars are the LAST step, not the only one — all five are needed:
# 1. sudo mkdir -p /var/lib/openship/host-ssh
# sudo ssh-keygen -t ed25519 -N '' -C openship-host-executor \
# -f /var/lib/openship/host-ssh/id_ed25519
# 2. append the .pub to the authorized_keys of the user below — root, for the
# root-owned paths host ops touch — as ONE restricted line:
# printf 'from="172.16.0.0/12,192.168.0.0/16,10.0.0.0/8,127.0.0.1",restrict,pty %s\n' \
# "$(sudo cat /var/lib/openship/host-ssh/id_ed25519.pub)" \
# | sudo tee -a /root/.ssh/authorized_keys
# (`from=` matters: without it that key is a root login from anywhere sshd
# accepts. `pty` is added back because the host terminal needs one.)
# 3. sshd must be listening on an address the containers can reach — a
# ListenAddress pinned to 127.0.0.1 refuses this channel and nothing else.
# 4. allow container→host on the SSH port in the host's firewall: this
# address is host-local, so it traverses filter/INPUT where a default-deny
# ufw lives — published container ports are DNAT'd and skip it, which is
# why the rest of the stack looks healthy while this one hangs.
# 5. set the vars below, then recreate the api — `env_file:` is read when a
# container is CREATED, so a restart alone changes nothing:
# docker compose --env-file .env -f docker/docker-compose.yml \
# up -d --force-recreate --no-deps api
#
# Full walkthrough, including the repair for an install that reports success and
# then fails its first host operation:
# https://openship.io/docs/troubleshooting/host-channel
# OPENSHIP_HOST_SSH_HOST=host.docker.internal
# OPENSHIP_HOST_SSH_USER=root
# OPENSHIP_HOST_SSH_PORT=22
# In-container path of the key. Its SOURCE on the host is OPENSHIP_HOST_KEY_PATH
# below, which docker/docker-compose.yml mounts here — so no compose file needs
# editing. Unset OPENSHIP_HOST_KEY_PATH mounts /dev/null instead, which is what
# lets the stack start on a box with no key at all.
# OPENSHIP_HOST_SSH_KEY=/run/secrets/openship_host_key
# ABSOLUTE path, always: a relative one resolves against docker/, not the
# directory you run `docker compose` from.
# OPENSHIP_HOST_KEY_PATH=/var/lib/openship/host-ssh/id_ed25519
# Set to false to switch host control off deliberately: no key is used, host ops
# refuse, and this box stops being offered as a deploy target. The docker socket
# is still mounted (deploys need it), so this is defense in depth, not isolation.
# OPENSHIP_HOST_CONTROL=true
# ─── OAuth login (optional) ───────────────────────────────
# GITHUB_CLIENT_ID=
+24 -9
View File
@@ -1,9 +1,11 @@
name: Docker images
# Publishes the official pull-based images (openship-api + openship-dashboard +
# openship-edge) to GHCR. Deliberately SEPARATE from release.yml (the bun
# binary / npm CLI / desktop installers / GitHub release) so image publishing
# can be iterated + tested on its own.
# Publishes the official pull-based images to GHCR: the control plane
# (openship-api + openship-dashboard + openship-edge) and the two products that
# ship as images of their own (openship-mail, the iRedMail engine; and
# openship-webmail, the Zero client that the app catalog installs). Deliberately
# SEPARATE from release.yml (the bun binary / npm CLI / desktop installers /
# GitHub release) so image publishing can be iterated + tested on its own.
#
# Triggers:
# • tag push v*.*.* → real release: semver tags + :latest (RCs skip :latest)
@@ -39,13 +41,28 @@ jobs:
strategy:
fail-fast: false
matrix:
image: [api, dashboard, edge, mail]
image: [api, dashboard, edge, mail, webmail]
platform: [amd64, arm64]
# Two independent axes of `include`, both merged into every matching
# combination: platform → runner (native builds, no QEMU), and image →
# its Dockerfile. The image↔path map is spelled out rather than derived,
# because apps/email/ produces TWO images (the mail engine and the
# webmail client) — nothing shorter than a table can express that.
include:
- platform: amd64
runner: ubuntu-24.04
- platform: arm64
runner: ubuntu-24.04-arm
- image: api
dockerfile: apps/api/Dockerfile
- image: dashboard
dockerfile: apps/dashboard/Dockerfile
- image: edge
dockerfile: apps/edge/Dockerfile
- image: mail
dockerfile: apps/email/Dockerfile
- image: webmail
dockerfile: apps/email/Dockerfile.webmail
steps:
- name: Checkout
uses: actions/checkout@v7
@@ -73,9 +90,7 @@ jobs:
uses: docker/build-push-action@v6
with:
context: .
# The mail image is built from apps/email/ but published as openship-mail;
# every other image key matches its apps/<key>/ directory 1:1.
file: apps/${{ matrix.image == 'mail' && 'email' || matrix.image }}/Dockerfile
file: ${{ matrix.dockerfile }}
platforms: linux/${{ matrix.platform }}
labels: ${{ steps.meta.outputs.labels }}
outputs: type=image,name=ghcr.io/${{ github.repository_owner }}/openship-${{ matrix.image }},push-by-digest=true,name-canonical=true,push=true
@@ -104,7 +119,7 @@ jobs:
strategy:
fail-fast: false
matrix:
image: [api, dashboard, edge, mail]
image: [api, dashboard, edge, mail, webmail]
steps:
- name: Download digests
uses: actions/download-artifact@v8
+287
View File
@@ -3,6 +3,293 @@
All notable changes to Openship. Versions follow [semver](https://semver.org);
the in-app updater surfaces critical advisories from `release-advisories.json`.
## 0.6.2
A hardening and reach release. Reported security issues are fixed (see Security
below), the container→host control channel is provisioned and diagnosed end to end
instead of failing quietly, and Openship now installs on any mainstream Linux — and on hosts
where you don't log in as root. It also adds `openship edge` (the whole reverse
proxy from the terminal), real resource limits that respect the machine you own,
an Openship Mail product shell with split outbound delivery, webmail rebuilt as an
ordinary catalog app, and five new one-click apps.
### Security
- **Security fixes** — this release resolves a number of reported security issues,
with regression tests added for each. Areas touched: authorization on
instance-wide and GitHub operations, access-token minting, path handling,
the mail engine, webmail's HTML sanitizer, the desktop shell, and redaction in
stored build output. Details are published as advisories on the repository.
**Upgrading is recommended.**
- **Hardening** — host-control operations are pinned to the host channel rather
than defaulting to the API process's own container, cloud builds can never run
on the API host, and stored SSH key material is encrypted at rest and stripped
from any cross-host export.
### The host control channel
- **`openship up` provisions the container→host channel, and tells you when it
can't** — when Openship runs in Docker, a handful of operations are genuinely
host-level (freeing a foreign proxy off `:80`/`:443`, host system config, the
mail engine, writing a catalog app's generated config), and they reach the host
over SSH on the internal bridge. Install now generates the key, appends a
restricted `from=`-pinned `authorized_keys` line, checks that sshd listens on an
address containers can reach, opens the port in whichever firewall the host
runs, and verifies the round trip — instead of reporting success and failing on
the first host operation weeks later.
- **It probes every layer, because the failure looks like nothing** — this
address is host-local, so it traverses `filter/INPUT` where a default-deny `ufw`
lives; published container ports are DNAT'd and skip it, which is exactly why
the rest of the stack looks healthy while this one channel hangs. Firewall,
sshd listen address, key, and reachability are each probed and reported
separately.
- **One explanation, five surfaces** — the CLI preflight, the API's boot banner,
`openship doctor`, the dashboard's server banner, and the deploy log a host
operation dies in had drifted into different stories. They now read from one
shared vocabulary, because every line is either "your install is fine, this one
feature isn't" or "your install is broken" — and an operator who reads the
wrong one either tears down a working box or ignores a dead feature for months.
- **`openship update` repairs an unprovisioned channel** — an install from before
this work, or a raw `docker compose` install, can be fixed in place by updating,
rather than re-running `up` and risking the environment. A deploy that needs the
channel emits the notice once per deploy, before the fan-out, instead of once
per service.
- **The channel must be root, and says so** — an operation that needs a
root-owned path fails with the actual cause instead of
`mkdir: cannot create directory '/root': Permission denied`.
- **Documented for hand-rolled installs** — `.env.example` spells out all five
steps (key, restricted authorized_keys line, sshd listen address, firewall
rule, recreate the api because `env_file:` is read at container creation), and a
new [troubleshooting page](https://openship.io/docs/troubleshooting/host-channel)
walks the repair.
### Any Linux, and hosts that aren't root
- **Openship installs on the distros it always claimed to** — Docker installation
was `curl get.docker.com | sh` on every Linux, a script that hard-refuses
Amazon Linux, AlmaLinux, Oracle Linux and Alpine. Each call site had its own
package-manager table (there were five, and they disagreed), its own systemd
test, and its own reading of "host outside my allowlist" — always "nothing to
do". Host facts now come from one detector, the commands that act on them live
in one module, and adding a distro is a compile error rather than a silent
no-op.
- **A working Docker engine is never replaced implicitly** — only an explicit
reinstall overrides an engine that's already running, and an installer that
skips an already-working component says so instead of leaving the one
actionable line nowhere at all.
- **Deploying as a non-root user works** — three layers answered "may I do
root-owned work here?" differently: component installs gated properly,
toolchain installs never gated at all, and the server state store assumed the
login was root. There's now one privilege resolver, and it elevates the write
rather than the verify (so a `sudo` shell never resolves a different `$HOME`).
A forced-command host that can't report a uid is no longer told to "connect as
root", which was the opposite of its fix.
- **Language toolchains install per host, not per assumption** — the toolchain
catalog and installer were rewritten onto the same host profile, so a bare-metal
build on a non-Debian box gets its runtime installed instead of a
`not found` at build time.
### `openship edge`
- **The whole reverse proxy, from the terminal** — a new top-level command:
`edge up` stands the proxy up and serves `:80`/`:443`, `edge migrate` takes over
an existing nginx/Apache/Caddy and imports its sites, `edge takeover` and
`edge free` claim the ports, `edge sites` lists what could be imported, and
`edge repair` diagnoses why it isn't serving (`--fix` resolves a port
conflict).
- **Domains, rules and traffic without opening the dashboard** —
`edge domains add app.example.com --port 8080` registers a hostname against a
port and issues a certificate; `edge rules` manages per-route rate limits, bans
and geo/CIDR access; and `edge traffic`, `edge analytics` and `edge logs` read
what the proxy already records. Host operations are Linux-only; the
control-plane half works from any machine, including desktop, against whichever
context is active.
- **Documented as both reference and walkthrough** — a new
[The edge](https://openship.io/docs/guides/the-edge) guide for the tasks, and a
full [command reference](https://openship.io/docs/cli/edge).
### Resource limits
- **Self-hosted containers are unlimited by default** — a self-hosted project
silently inherited the cloud free tier (0.5 vCPU · 512 MB) and OOM-killed
memory-hungry images. On your own box the container's real ceiling is the
machine, so `0` (no limit) is now the default and every consumer tests for a
limit before applying a cap.
- **A custom cap is bounded by the actual machine** — caps were validated against
hardcoded constants (4 cores / 8192 MB), so a 64 GB box could not be told to
give a container more than 8 GB. The ceiling is read from the Docker daemon's
own `/info` (`NCPU`, `MemTotal`) — identical for the local socket and a remote
daemon over the pooled SSH bridge — falling back to the OS when Docker isn't
reachable. Cloud is unchanged: a metered workspace is still sized from the tier
table.
- **Deploys preflight the target's capacity** — a cap larger than the host can
allocate is rejected before any build work starts, with the machine's real
numbers in the message.
### Openship Mail
- **Run the instance as a mail product** — an instance can present itself as
Openship Mail (`OPENSHIP_PRODUCT`, or a toggle in Settings): the left rail
becomes the mail control plane — the ten admin tabs promoted to nav entries
across three headings, plus the host and settings rows an operator still needs —
and the platform nav is hidden. This is presentation only, never an
authorization boundary: webmail still deploys through the ordinary project
pipeline, so platform endpoints stay live. Cloud is always the full platform.
- **Switch between mail servers** — the mail surfaces are scoped to a selected
server rather than assuming one, with a switcher in the shell and the scope
carried through the admin tabs.
- **Outbound relay providers are real identities** — the relay code carried a
`"ses" | "custom"` union, so SendGrid, Mailgun and Postmark lost their identity
the moment they were saved: no SPF include, no round-trip in the UI, and adding
a provider meant editing an `if` in the service, the DNS builder and the
scanner. Per-provider facts are now data that all three read, and split
delivery (receive here, send through someone else) is a first-class setup.
- **"Is mail actually leaving this box?"** — the daemon sweep reports that nine
processes are running, which says nothing about whether Postfix can hand a
message to the next hop. One wrong character in a relay password gave nine
green daemons, green DNS, a Test-tab email that reported success (our Postfix
accepts and queues before the relay hop), and mail that quietly deferred
forever. Outbound delivery is now probed and surfaced on its own, with a
reworked Health tab and a summary that names the failing hop.
- **A mail console** — live engine output in the dashboard, so a setup that stalls
can be read rather than guessed at.
- **Backups you can schedule** — the mail admin Backup tab gains a real schedule
and retention plan, wired to the same pruning the rest of Openship uses.
- **Setup fails earlier and clearer** — a preflight checks the things that used to
surface mid-stream (password shape, firewall step, DNS), and setup errors render
in a banner instead of vanishing into the stream.
### Webmail
- **Webmail is an ordinary catalog app now** — it installs through the same
generic installer, image, volume and routing pipeline as every other app, and
installs from /apps just as well as from /emails. The mail-specific part is the
only part that stayed: which IMAP/SMTP backend it belongs to — one of your
Openship mail servers, or an external provider.
- **The link is stored, not parsed out of a slug** — the old lookup read
`webmail-<serverId>` back out of the project slug, which the generic installer
never produces, and which mislinked any project someone happened to name
"Webmail Prod". It's a real foreign key now, nulled (not orphaned) when the
webmail project is deleted, with a legacy-slug fallback that stamps the column
the first time it resolves a pre-existing install.
- **Its own image and database bootstrap** — a dedicated webmail Dockerfile and a
bootstrap script, so first boot provisions its schema instead of depending on
the mail engine's.
### App catalog
- **Five new one-click apps** — Meilisearch, Redis, PostHog, Umami and Neon
(Postgres), each declaring its internal connection so it can be linked into a
project over the internal network.
- **Badges that tell you what you're installing** — verified, unverified and
hosting-mode badges with explanatory tooltips, so a community template isn't
visually indistinguishable from one Openship has booted and checked.
- **Catalog schema updates** — the published `app.schema.json` gains the fields
behind the above, and the reference docs and "Add an app" guide were rewritten
around them.
### Routing & domains
- **A deploy stops wiping your `vercel.json` routing** — `registerRoute` replaces
the whole vhost, so a caller that omitted `cleanUrls` / `redirects` / `headers`
didn't leave them alone, it deleted them. Those fields were built inline in two
places, so a plain deploy neither applied them nor preserved them, and an
unrelated redeploy silently erased whatever a "Retry routing" had installed.
They're now compiled once, in a pure module every route-registration site
carries.
- **One decision about what the edge dials** — a single resolver owns the
`proxy_pass` target: a pinned loopback host port by default (stable across
restart, never internet-facing), the container's bridge IP as an advanced
option, and a transparent fallback to the container IP for an internal Compose
service that publishes no host port. Selectable per instance, with `auto` as the
default.
- **A domain redirect survives certificate issuance** — the redirect installed
for a hostname is no longer reverted the moment certbot succeeds, and it no
longer re-fires on internal rewrites.
- **Real client IPs at the edge** — the real-IP configuration is derived in one
place for every scope, so free subdomains and custom domains agree on what the
visitor's address is.
- **Route rules have an end-to-end suite** — per-route rate limits, bans and
access rules are covered by an e2e test that drives the real proxy.
### Servers
- **One add-server form** — the page and the modal were two 400-line forks that
had drifted; they're now one component with a variant, so a field added in one
place exists in both.
- **Paste or upload an SSH private key** — `ssh_key_path` points at a file on the
API host, which is useless on a remote or VPS instance where your key lives on
your own laptop. A key can now be pasted or uploaded from any browser, stored
encrypted at rest, never serialized back to the client, and preferred over the
on-disk path when present.
- **The server page's connection banner explains itself** — reachability, host
channel state, and what each one does or doesn't affect, rather than a single
red dot.
- **Deploy defaults can't be set to something derived** — the instance default
target is a server or cloud; "this machine" is derived (on a server-host box it
is already in the server list), so it's no longer an option that means two
different things.
### Projects & deploys
- **Rename a project from its heading** — a rename modal, plus copy-id and
pause/resume, on the heading where the name actually is. The slug stays
immutable (it's infrastructure identity); the display name is free.
- **Container actions go through one pipe** — pause, restart and logs run through
the same deployment-runtime path as everything else, so they can't disagree
with the platform about which host or runtime a service lives on.
- **Reconcile and teardown are steadier** — record-only deletes, orphan GC
scheduling, pending actions and port checks were tightened around the same
runtime path.
### Compose & install
- **`up` preserves your `.env`** — a variable you set by hand is kept and marked
as yours, instead of being regenerated out of a fixed key list.
- **A secret is never rotated out from under a running install** — `up` detects
that it would mint a secret an existing install already has (which would leave
encrypted env undecryptable), and refuses rather than silently replacing it.
The replaced `.env` is kept for recovery.
- **Image versions can be pinned** — an explicit `--image-version` wins over the
environment and the CLI's own version, for the case where a release reaches npm
ahead of its images reaching the registry.
- **A foreign Postgres data directory is refused, not adopted** — the volume's
contents are probed before a cluster is started on top of them.
- **`openship doctor` and `repair` cover more** — the host channel, an
unprovisioned install, and port conflicts, in the same run.
### Desktop
- **Local deploys are gated behind an explicit "coming soon"** — desktop mode is
built to control remote servers; running the workload on the desktop itself
isn't enabled yet, and the UI now says that once, in one place, instead of
offering a target that fails later.
- **Shell hardening and a safer updater** — see Security above.
### Docs & site
- **New and rewritten pages** — "The edge" guide, the `openship edge` command
reference, host-channel troubleshooting, a rewritten "Add an app" guide and app
catalog reference, plus notes on installation, updating and logs/monitoring.
- **Docs get social previews** — generated OG images per page.
- **A better changelog page** — entries collapse to headlines and expand to
detail, with controls to expand or filter, driven by this file.
### Fixes
- **Zero-auth login doesn't bounce a remote browser forever** — an instance with
no sign-in form grants its session only to a browser on the same machine, so
bouncing a remote one was guaranteed to fail, leaving a spinner and then either
a connection error for a host that isn't yours or an endless redirect. The page
now names the cause.
- **Mail retention pruning respects mail backups** — pruning no longer considers
a mail engine backup an ordinary project artifact.
- **A migrated container joins the network it's routed on** — a same-server
migration attached a container that was never published, leaving it unreachable
behind a verified domain.
- **Every new string is translated** — the release's new copy landed across all
shipped languages, with the parity test extended to the new namespaces.
## 0.6.1
A large release. It adds a full service-to-service networking plane, around-the-clock
+23
View File
@@ -160,6 +160,29 @@ If you're adding something that should only exist in the cloud version:
2. Make any required env vars (like Stripe keys) optional in `apps/api/src/config/env.ts`
3. Self-hosters should never see 500s from missing cloud config
## Adding an App to the Catalog
The one-click **Apps** catalog is data, not code — adding one is a small pull request that adds a
single JSON file. No TypeScript required.
1. Write `packages/core/src/apps/catalog/<id>.json` (start it with
`"$schema": "https://openship.io/app.schema.json"` for editor autocomplete)
2. Regenerate the merged artifact and validate:
```bash
cd packages/core
bun scripts/gen-catalog.ts # rewrites src/apps/catalog.json (a drift test fails CI without it)
bunx vitest run src/apps/catalog.test.ts # shape + referential validation for every app
```
3. Keep `"available": false` until it deploys cleanly end to end
Apps must be open-source, use an official image **pinned** to a version, and auto-generate any
credentials. Full walkthrough and field reference:
- **[Add an app](https://openship.io/docs/guides/add-an-app)** — builds a real two-service app step by step
- **[App catalog JSON](https://openship.io/docs/reference/app-catalog)** — every field
## Database
```bash
+3 -1
View File
@@ -157,7 +157,9 @@ docker compose --env-file .env -f docker/docker-compose.yml up -d
The stack is **postgres + redis + api + dashboard + edge**. The `edge` is OpenResty on **:80/:443** as a container (`network_mode: host`) — routing + Let's Encrypt, no bare host install. **Linux only** (host networking); on mac/win use `openship up` (bare). The `api` container mounts the host Docker socket so the control plane can build + run your apps as host containers — it's host-privileged through the socket, so run it only on a trusted host.
**Upgrade:** pin `OPENSHIP_VERSION` in `.env` for reproducible pulls, then `docker compose --env-file .env -f docker/docker-compose.yml pull && … up -d` (or just `openship update`). **Build from source instead:** add `-f docker/docker-compose.build.yml … up -d --build`.
**Upgrade:** pin `OPENSHIP_VERSION` in `.env` for reproducible pulls, then `docker compose --env-file .env -f docker/docker-compose.yml pull && … up -d`. `openship update` only reconciles a stack the CLI installed, and `openship up` would *adopt* this one — don't reach for either here. **Build from source instead:** add `-f docker/docker-compose.build.yml … up -d --build`.
**Host operations** (`:80`/`:443` takeover, the mail engine, host terminal/port scans) need the container→host SSH channel, which `openship up` provisions and this path does not — the five manual steps are in `.env.example` under *Host operations from the container*, and the failure it produces is [Troubleshooting → Host control channel](https://openship.io/docs/troubleshooting/host-channel). Everything else, including deploys, works without it.
> The **root** `docker-compose.yml` is a different file: it's the SaaS / from-source **control plane** (builds from source, ships the marketing site, no edge/socket). It does **not** self-host your apps — use `docker/docker-compose.yml` above or `openship up`.
+17
View File
@@ -196,6 +196,23 @@ const envSchema = z.object({
* - "desktop" → Bare runtime, no routing/SSL (desktop app)
*/
DEPLOY_MODE: z.enum(["docker", "bare", "cloud", "desktop"]).default("docker"),
/**
* Which PRODUCT this instance presents itself as — the INSTANCE DEFAULT, which
* `instance_settings.product_mode` may override (so an operator can flip it
* from the dashboard without editing env and restarting).
*
* - "platform" (default) → the full deploy platform
* - "mail" → Openship Mail: the dashboard's left rail becomes
* the mail control plane, the platform nav is hidden
*
* Orthogonal to DEPLOY_MODE: mail mode says what the UI presents, DEPLOY_MODE
* says how workloads run. Mail mode still needs the full deploy runtime, since
* the mail installer and webmail both ride it.
*
* Always read through resolveProductMode() (lib/product-mode.ts), never
* directly — that resolver owns the settings-override and CLOUD_MODE rules.
*/
OPENSHIP_PRODUCT: z.enum(["platform", "mail"]).default("platform"),
/* ---------- Auth (Better Auth) ---------- */
BETTER_AUTH_SECRET: z.string().default(DEFAULT_BETTER_AUTH_SECRET),
+7
View File
@@ -10,6 +10,7 @@ import { app } from "./app";
import { cloudRuntimeTarget, cloudRuntimeTargetId, env, runtimeTargetId } from "./config/env";
import { getAuthMode } from "./lib/auth-mode";
import { edgeBuildSpec, pinnedEdgeImage } from "./lib/edge-image";
import { reportHostChannelAtBoot } from "./lib/host-channel-banner";
import { mailBuildSpec, pinnedMailImage } from "./lib/mail-image";
import { getJobRunner } from "./lib/job-runner";
import { enforceRouteScanAtBoot } from "./lib/route-scanner";
@@ -82,6 +83,12 @@ void (async () => {
console.error("");
})();
// Same shape, for the container→host SSH channel (#490) — silent unless the channel
// is actually broken. At boot and not only at install: `openship up` probes it now,
// but a box provisioned before that existed never saw the check, and a firewall can
// change under a running install.
void reportHostChannelAtBoot().catch(() => {});
// Attach the tunnel agent lifecycle if this instance has been migrated
// via Path C (teamMode === "tunneled"). Local-API-only by design —
// CLOUD_MODE returns a no-op handle without touching state. Lives after
+2 -7
View File
@@ -1,4 +1,4 @@
import { db, schema, eq } from "@repo/db";
import { repos } from "@repo/db";
import { resolvesToLocalHost } from "./self-host";
@@ -21,12 +21,7 @@ let cachedBoxOrgId: string | null = null;
*/
export async function boxOwningOrgId(): Promise<string | null> {
if (cachedBoxOrgId) return cachedBoxOrgId;
const [admin] = await db
.select({ id: schema.user.id })
.from(schema.user)
.where(eq(schema.user.autoProvisioned, false))
.orderBy(schema.user.createdAt)
.limit(1);
const admin = await repos.user.findFoundingAdmin();
if (!admin?.id) return null;
cachedBoxOrgId = `org_${admin.id}`;
return cachedBoxOrgId;
+43 -1
View File
@@ -98,7 +98,7 @@ export { getPlatform as platform } from "@repo/adapters";
* a 404-shaped error if it doesn't, to avoid leaking existence across
* orgs (404, not 403 — IDOR-safe). NULL `organizationId` fails closed.
*/
import { NotFoundError } from "@repo/core";
import { ForbiddenError, NotFoundError } from "@repo/core";
export function assertResourceInOrg<T extends { organizationId?: string | null }>(
resource: T | null | undefined,
@@ -111,6 +111,47 @@ export function assertResourceInOrg<T extends { organizationId?: string | null }
}
}
/**
* Refuse a runtime action on the Openship control-plane self-app.
*
* The self-app IS the process serving the request, so "stop it" is a request to
* kill the thing that would report the result. Every mutating surface needs the
* same answer for the same reason — pausing the PROJECT stops the api container,
* deleting its `postgres` SERVICE drops the control plane's own database, and its
* DEPLOYMENT row is an adopt over a CLI-supervised host process, so rolling it
* back or pinning it would detach the live app. Read paths (status, logs, shell,
* container info) are deliberately NOT gated: showing the operator that state is
* the reason the self-app is linked at all.
*
* One definition, because it had grown three — a project copy, a services copy,
* and a deployments wrapper — and they had already drifted: one of them told
* operators to run `openship restart`, which is not a command (`restart` exists
* only under `deployment` and `service`).
*/
export function assertNotControlPlane(
project: { appTemplateId?: string | null } | null | undefined,
): void {
if (project?.appTemplateId === "openship") {
throw new ForbiddenError(
"The Openship control plane manages its own runtime — manage it with the CLI on the host " +
"(`openship up`, `openship stop`, `openship update`), not from the dashboard.",
);
}
}
/**
* The same policy for callers holding only a project id — a deployment row, a
* `projectId` route param.
*
* Exists so those callers have ONE shape instead of each resolving the project
* itself: that per-module resolve is what grew into two divergent copies of the
* check. Callers that already hold the project must use {@link assertNotControlPlane}
* directly rather than re-fetching it here.
*/
export async function assertNotControlPlaneById(projectId: string): Promise<void> {
assertNotControlPlane(await repos.project.findById(projectId));
}
/** Extract and validate a required route parameter */
export function param(c: Context, name: string): string {
const val = c.req.param(name);
@@ -197,6 +238,7 @@ export function resolvePlatformConfig(): PlatformConfig {
target: "cloud",
cloudClientId: env.OBLIEN_CLIENT_ID,
cloudClientSecret: env.OBLIEN_CLIENT_SECRET,
allowHostBuild: !env.CLOUD_MODE,
};
}
+23 -1
View File
@@ -20,7 +20,7 @@
* the source-built fields.
*/
import { STACKS, type StackDefinition, type StackId } from "@repo/core";
import { normalizeServiceLabel, STACKS, type StackDefinition, type StackId } from "@repo/core";
import type { ComposeService } from "./compose-parser";
/** Source-built sub-app fields. Only meaningful when `kind === "monorepo"`. */
@@ -162,3 +162,25 @@ export function resolveServicePort(
(typeof fallback === "number" && fallback > 0 ? fallback : null)
);
}
/**
* The operator's custom east-west DNS alias for a service, if any — the ONE rule
* for it.
*
* `advanced.alias` resolves ALONGSIDE the default (the row name), so this returns
* only the EXTRA names; the primary is added by `buildNetworkAliases`. Shared by
* the compose deploy (which passes it as `extraAliases` on the runtime config) and
* by the migration attach-live network join — a reused container has to answer to
* exactly the same names a natively-deployed one does, or east-west resolves for
* deployed services and silently fails for migrated ones.
*/
export function serviceAliasExtras(service: {
name: string;
advanced?: unknown;
}): string[] | undefined {
const raw = (service.advanced as { alias?: string } | null | undefined)?.alias;
if (!raw) return undefined;
const alias = normalizeServiceLabel(raw);
if (!alias || alias === normalizeServiceLabel(service.name)) return undefined;
return [alias];
}
+390 -8
View File
@@ -1,7 +1,10 @@
import {
createPlatform,
DockerRuntime,
isHostChannelUnavailableError,
peekPlatform,
resolveStaticOutputPath,
unavailableExecutor,
type CommandExecutor,
type DockerConnectionOptions,
type Platform,
@@ -10,7 +13,13 @@ import {
} from "@repo/adapters";
import type { Deployment } from "@repo/db";
import { repos } from "@repo/db";
import type { DeployTarget, RuntimeMode } from "@repo/core";
import {
HOST_CHANNEL_UNAFFECTED,
HostUnreachableError,
safeErrorMessage,
type DeployTarget,
type RuntimeMode,
} from "@repo/core";
import { env } from "../config";
import { cloudClient, getOrgCloudToken } from "./cloud/client";
import { resolveOrgCloudUserId } from "./cloud/transport";
@@ -18,7 +27,9 @@ import { platform } from "./controller-helpers";
import { buildSshConfig, sshManager } from "./ssh-manager";
import { createProvisionLock } from "./provision-lock";
import { isLocalHostRow } from "./box-org";
import { isConnectionLoss } from "./remote-state";
import { resolveAcmeProviderOptions } from "./acme-config";
import { findLocalServer } from "./startup/self-server";
/**
* The shape of `deployment.meta` JSONB. Snapshotted per-deploy —
@@ -204,10 +215,11 @@ async function resolveOrgServer(
// existence/name oracle. So we explain the likely cause + recovery without
// revealing whether the id exists elsewhere.
throw new Error(
"The selected deploy target isn't in this project's organization. This usually " +
"happens after re-deploying Openship at the same URL (a stale session) or when " +
"your active organization differs from the project's. Re-open the deploy target " +
"picker and reselect a server, or switch your active organization to match, then redeploy.",
"This project's deploy target is no longer available. That server may have been " +
"removed from Openship (deleting one unbinds its projects), or this is a stale " +
"session after re-deploying Openship at the same URL, or your active organization " +
"differs from the project's. Re-open the deploy target picker and reselect a " +
"server, or switch your active organization to match, then redeploy.",
);
}
return server;
@@ -294,6 +306,7 @@ async function resolveCloudPlatformForOrg(organizationId?: string): Promise<Plat
return createPlatform({
target: "cloud",
cloudToken: result.token,
allowHostBuild: !env.CLOUD_MODE,
cloudAdminProxy: {
createPage: (input) => cloudClient({ organizationId }).pages.create(input),
disablePage: (slug) => cloudClient({ organizationId }).pages.disable(slug),
@@ -395,6 +408,7 @@ export async function resolveTargetPlatform(
target: "selfhosted",
runtime: runtimeMode,
executor,
localHost: true,
docker: runtimeMode === "docker" ? { transport: "socket" as const } : undefined,
nginx: resolveAcmeProviderOptions(),
provisionLock: createProvisionLock("provision:local"),
@@ -414,15 +428,47 @@ export async function resolveTargetPlatform(
});
}
// Local target - no SSH, no pooling needed. Still serialize provisioning: two
// local deploys share the same host's openresty/docker/state.
// "local" is not a destination anyone picks — it is the ABSENCE of a binding
// (no cloud workspace, no serverId), so it always means "this box". Nothing
// offers it: `project.server_id` is ON DELETE SET NULL, so deleting a server is
// enough to make the next deploy for that project derive it.
//
// Which is why it resolves through the SAME executor as the isLocal "This Server"
// row above. One machine, one path — whether the deploy arrived with that row
// picked in the wizard or with no binding left at all. Before this, the branch
// took `createPlatform`'s default `createExecutor()`, a plain local executor: on a
// compose install that ran every host-side step inside the API CONTAINER, against
// the wrong filesystem, which is precisely the hazard `createHostExecutor` exists
// to refuse — reached through a different door. Now a dead channel refuses out
// loud, with the remedy, and the container workload deploys as before.
//
// READ the row, never create it. `findLocalServer` shares its gates with
// `ensureLocalServer` (self-server.ts), so there is no second definition of "this
// box's row" to drift from — but registering a server is preparation, not part of
// resolving a deploy. Calling the ensure here made every deploy on a row-less box
// responsible for an insert plus the creation path's public-IP lookup (an outbound
// request), on a function that also runs for plain runtime reads.
//
// Only the id is wanted, and only as bookkeeping: null and non-null both resolve to
// the same pooled host channel below, so a missing row degrades into "no borrow
// marker", never into a different machine.
const localRow = await findLocalServer().catch(() => null);
return createPlatform({
target: "selfhosted",
runtime: runtimeMode,
executor: await acquireLocalHostExecutor(localRow?.id),
// Explicit, and load-bearing now that an executor is injected: `createPlatform`
// infers "this machine" from `localHost ?? !executor`, so an injected executor
// would otherwise read as REMOTE — turning off the containerized edge provider
// and the same-path-mount rule (`sharedMountExecutor`) for the local box.
localHost: true,
docker: runtimeMode === "docker"
? { transport: "socket" as const }
: undefined,
nginx: resolveAcmeProviderOptions(),
// Still serialize provisioning: two local deploys share the same host's
// openresty/docker/state. Same lock name as the isLocal row's branch, because
// it is the same host being provisioned.
provisionLock: createProvisionLock("provision:local"),
});
}
@@ -454,6 +500,121 @@ export async function createServerDockerRuntime(
return DockerRuntime.create(toDockerSshTransport(ssh!, executor));
}
/**
* Executors we handed back REFUSING, and the reason, so the fact travels with the
* handle instead of in a cache someone has to invalidate (#509).
*
* Keyed by the executor object on purpose: a deploy already holds the very executor
* that was demoted, so identity answers "were host operations available to THIS
* deploy, on THIS box?" with no key, no TTL, and no way to describe a different
* target's channel. Entries die with the executor.
*/
const hostChannelRefusals = new WeakMap<CommandExecutor, string>();
/** Last demotion reason we logged, so the decision is logged once per outage and
* not once per resolve — see the `console.warn` below. Cleared on recovery. */
let lastLoggedRefusal: string | null = null;
/**
* One line for a deploy log: host operations were skipped, why, and that the deploy
* itself is unaffected.
*
* Callers emit this ONCE per deploy, before the fan-out. Without it the #509 box does
* not fail any more — it degrades in silence, because each host touchpoint absorbs the
* refusal on its own terms: `allocateHostPort` reports an unscanned host (and only
* under `loopback-port` routing), and the edge/routing step logs "deploy continues".
* Neither names the channel, so the first legible symptom is a container that dies
* later over a config file that never landed.
*
* Returns null unless this executor is one we demoted — the notice is EVIDENCE, so it
* is never printed for a target whose channel nothing has decided anything about.
*/
export function hostChannelDeployNotice(executor?: CommandExecutor | null): string | null {
const reason = executor ? hostChannelRefusals.get(executor) : undefined;
if (!reason) return null;
return (
"Host operations are unavailable on this deploy target, so this deploy skips them: " +
"live host port-occupancy scans and host-side edge/routing steps. Anything that MUST " +
"be written on the host — an app template's generated config file — still fails.\n" +
`${reason}\n${HOST_CHANNEL_UNAFFECTED}`
);
}
/**
* This box's host executor, with "this box has no host channel" demoted from a
* resolve-time throw to a use-time one.
*
* `serverId` is the canonical isLocal row when the box has one, and the executor is
* then POOLED — `sshManager.acquire` hands back the shared host channel, which is what
* stops one deploy from leaving behind an sshd session (#291). Without a row (desktop,
* the SaaS, `--no-host-control`) there is nothing to pool against, so the channel is
* constructed directly. Same box either way, so it must be the same policy: one
* function, so a target that arrives by the derived `local` door cannot end up with a
* gentler rule than the one that arrives as a picked server row.
*
* A local row's WORKLOAD lives behind the mounted Docker socket; the executor is
* for host-side extras (static file serve, port scans, host config). So a box with
* host control off — or containerized with no channel provisioned — should still
* deploy containers, and only fail on the extras. Before this, `createHostExecutor`
* throwing at construction meant every deploy to "This Server" died the moment host
* control was switched off, which is exactly what a blocked-channel banner used to
* recommend (#490).
*
* A channel that is configured but UNREACHABLE arrives here as the same typed error,
* raised by the manager once it has watched the channel fail — and it is demoted for
* the same reason. That is not pretending it works: the executor handed back refuses
* every call with the firewall remedy attached, and the banner and server health both
* report host control as unavailable. The alternative is what #490 actually did —
* every container deploy to "This Server" dying on a channel it never needed.
*
* The demotion is RECORDED and LOGGED here, because this is where it is decided:
* before, "this box cannot drive its host" was concluded silently and the operator's
* first evidence was a symptom several steps downstream (#509). `hostChannelRefusals`
* carries it forward to the deploy log; the log line covers everything that never
* reaches a deploy log at all.
*
* Any other acquire failure still propagates.
*/
async function acquireLocalHostExecutor(serverId?: string): Promise<CommandExecutor> {
try {
// One pooled channel either way — `acquire(localRow)` resolves to the very same
// executor `acquireHostChannel()` returns, and the row id only adds the borrow
// marker that lets `probeReachable`/idle cleanup see the row. So the branch is
// bookkeeping, never a difference in WHICH executor this box gets.
//
// The no-row door used to call `createHostExecutor()` here instead, which is a
// fresh SshExecutor outside the pool, outside the concurrent-acquire dedup and
// outside the channel-health gate — the unpooled idiom that reached 8,000+
// orphaned sshd sessions (#291), reintroduced on the one path that has no row to
// launder through `acquire`. Both doors now take the pooled channel.
const executor = serverId
? await sshManager.acquire(serverId)
: await sshManager.acquireHostChannel();
// Recovered — re-arm the log so the NEXT outage is reported rather than deduped
// against the last one.
lastLoggedRefusal = null;
return executor;
} catch (err) {
if (!isHostChannelUnavailableError(err)) throw err;
// Carry the CODE through, not just the prose: the refusal this stands in for is the
// same fact as the acquire that failed, so a `disabled` channel must not be re-labelled
// `not_configured` when the refusal is finally raised at use time.
const executor = unavailableExecutor(err.message, err.code);
hostChannelRefusals.set(executor, err.message);
// Once per outage, not per resolve: this sits on every deploy AND on read paths
// (logs, status polls), so an unconditional line here would bury the log it is
// meant to be found in. The reason carries the remedy.
if (lastLoggedRefusal !== err.message) {
lastLoggedRefusal = err.message;
console.warn(
`[host-channel] host operations unavailable on this box (${err.code}). ` +
`${HOST_CHANNEL_UNAFFECTED} ${err.message}`,
);
}
return executor;
}
}
/**
* THE single server → {executor, endpoint, transport} resolver. One place
* decides how to reach a server, so resolveTargetPlatform (deploy),
@@ -502,7 +663,7 @@ export async function resolveServerExecutor(
// stops one deploy from leaving behind an sshd session (#291).
return {
id: server.id,
executor: await sshManager.acquire(server.id),
executor: await acquireLocalHostExecutor(server.id),
conn,
isLocal: true,
ssh: null,
@@ -605,6 +766,227 @@ export async function resolveDeploymentRuntime(
* Callers own `runtime.dispose()` (tears down the SSH loopback bridge; no-op on
* the socket transport).
*/
/**
* Every container an existing deployment owns: its services' containers when it
* has any, else its own single container.
*
* Shared because "the deployment's container" is plural for a compose project and
* singular everywhere else, and each caller that re-derived it got a different
* answer — `disableProject` read `deployment.containerId` alone, so pausing a
* compose project stopped the app and left every sidecar running.
*/
export async function deploymentContainerIds(
dep: Pick<Deployment, "id" | "containerId">,
): Promise<string[]> {
// Deliberately NOT error-swallowing: if we can't read the service rows we don't
// know how many containers this deployment has, and falling back to the single
// `containerId` would quietly act on one of them.
const rows = await repos.service.listByDeployment(dep.id);
const serviceIds = [...new Set(rows.map((r) => r.containerId).filter((id): id is string => !!id))];
if (serviceIds.length > 0) return serviceIds;
return dep.containerId ? [dep.containerId] : [];
}
/**
* Run one container action against a deployment's runtime — THE entry point for
* every non-deploy runtime operation (enable/disable, restart, logs, info, usage).
*
* It exists because each of those call sites used to open its own transport, and
* each got a different subset of the three things all of them need:
*
* 1. the READ resolver, not a full platform. A full platform builds the infra
* provider, which runs `detectOpenRestyPaths` plus the edge-Lua self-heal
* inside the `provision:local` provision lock — so a status read or a pause
* queued behind any in-flight deploy. That is the "the action takes forever"
* half of the service-panel timeouts, and the project actions still had it.
* 2. `dispose()`, always. The SSH branch mints a NEW loopback bridge per
* runtime (see docker.ts `watchContainerEvents`), so a call site that
* forgets leaks a listening socket plus an ssh client per click — and the
* bridge's own accept path warns about exactly the fd pressure that causes.
* 3. one error classification. A refused key or an unreachable box is not a
* client error; mapping it here means every caller reports 503 with the real
* cause instead of each inventing a status.
*
* For a LONG-LIVED runtime (log streaming, where the transport must outlive this
* call) use `resolveDeploymentRuntimeForRead` directly and dispose in the
* stream's cleanup — this helper's whole contract is that the runtime is dead
* when it returns.
*/
export async function withDeploymentRuntime<T>(
dep: Pick<Deployment, "meta" | "organizationId">,
fn: (runtime: RuntimeAdapter, serverId: string | null) => Promise<T>,
): Promise<T> {
const { runtime, serverId } = await resolveDeploymentRuntimeForRead(dep);
try {
return await fn(runtime, serverId);
} catch (err) {
throw asHostUnreachable(err);
} finally {
disposeRuntime(runtime);
}
}
/**
* THE disposal step, so "how do we release a transport" has one answer.
*
* Best-effort and non-blocking on purpose: a transport that is already dead can't
* be closed politely, and a teardown failure must never replace the caller's real
* error. Optional-called because a bare/cloud runtime has nothing to release —
* calling it on those is a deliberate no-op, which is what lets every call site
* dispose unconditionally instead of first asking what kind of runtime it got.
*/
export function disposeRuntime(runtime: RuntimeAdapter | null | undefined): void {
release(runtime);
}
/** The one place `dispose()` is actually invoked. Structural rather than typed to
* `RuntimeAdapter` because the platform loop below releases whichever layers we
* decided to release, and those don't share an interface. */
function release(layer: { dispose?: () => Promise<void> } | null | undefined): void {
if (!layer || ownedByProcessPlatform(layer)) return;
void Promise.resolve(layer.dispose?.()).catch(() => {});
}
/**
* Is this layer one the process-wide platform owns, rather than one this resolve built?
*
* `resolveDeploymentPlatform` returns `basePlatform` itself — the `getPlatform()` singleton —
* whenever the effective target is cloud and no org-scoped platform is needed, which on the
* SaaS (`CLOUD_MODE`) is every such request. `withDeploymentPlatform`'s `finally` then hands
* the singleton's own layers to `release()`, so one deploy's teardown would dispose the
* transport every other request in the process is using. It is a no-op today only because a
* cloud runtime happens to have no `dispose()` — which is precisely the assumption
* `PLATFORM_DISPOSAL` exists to stop us from making, and it dies the day one gains one.
*
* Identity, not a flag: the caller cannot know whether the resolver handed it a fresh
* platform or the shared one, so asking it to declare ownership reintroduces the bug at
* fourteen call sites. `peekPlatform` rather than `platform()` because disposal must still
* work before startup and in unit tests, where "there is no singleton" means "not it".
*/
function ownedByProcessPlatform(layer: object): boolean {
const shared = peekPlatform();
if (!shared) return false;
return (Object.keys(PLATFORM_DISPOSAL) as PlatformDisposableField[]).some(
(field) => shared[field] === layer,
);
}
/**
* Every `Platform` field that CAN be disposed — read off the type, not listed by
* hand, so a provider that gains a `dispose()` cannot stay invisible here.
*
* The probe is `"dispose" extends keyof T` rather than `T extends { dispose?: … }`
* because an OPTIONAL member is satisfied by every object type: that predicate
* matches all seven fields and asserts nothing.
*/
type PlatformDisposableField = {
[K in keyof Platform]-?: "dispose" extends keyof NonNullable<Platform[K]> ? K : never;
}[keyof Platform];
/**
* Release it, or keep it and say who owns it instead. Total over
* `PlatformDisposableField`, so a `dispose()` added to `RoutingProvider`,
* `SslProvider` or `SystemManager` is a missing-key error here (TS2739) rather than
* a silent leak at all fourteen `disposePlatform` sites — a leak that surfaces as fd
* exhaustion hours later, nowhere near the resolve that caused it. The union value
* type is what makes it a decision: you cannot satisfy the key with `undefined`.
*/
const PLATFORM_DISPOSAL: Record<PlatformDisposableField, "release" | { keep: string }> = {
runtime: "release",
// NEVER released. `executor` is the pooled per-server SSH executor that
// `sshManager` owns and that concurrent deploys, routing applies and cert
// issuance on that box all share — disposing it here would tear the transport out
// from under every one of them, and the three `.ssl`-only sites in domain-ssl.ts
// depend on surviving exactly this call. The runtime's Docker-over-SSH bridge is a
// per-resolve loopback listener, which is why that one is ours to close.
executor: { keep: "pooled per server by sshManager; shared with concurrent work" },
};
/**
* `disposeRuntime` for anything that carries a platform. Use in a `finally` on
* flows too long to wrap in `withDeploymentPlatform`.
*
* Takes either shape the resolvers hand back — `resolveTargetPlatform` returns a
* bare `Platform`, `resolveDeploymentPlatform` wraps it next to the effective target
* and server id. Both are real and both need releasing, so accepting both beats
* making ten call sites reach through `.platform`; `Platform` declares no `platform`
* field, so the narrowing is exact.
*/
export function disposePlatform(
resolved: Platform | { platform: Platform } | null | undefined,
): void {
if (!resolved) return;
const p = "platform" in resolved ? resolved.platform : resolved;
const decisions = Object.entries(PLATFORM_DISPOSAL) as [
PlatformDisposableField,
(typeof PLATFORM_DISPOSAL)[PlatformDisposableField],
][];
for (const [field, decision] of decisions) {
if (decision === "release") release(p[field]);
}
}
/**
* `withDeploymentRuntime`'s twin for the FULL platform — routing + ssl + system
* alongside the runtime.
*
* Separate function rather than a flag because the two have genuinely different
* costs and the choice must stay visible at the call site: this one builds the
* infra provider (OpenResty detect + edge-Lua self-heal, under the provision
* lock), which is right for a route apply and wrong for a status read.
*
* The reason it exists at all: `createPlatform` builds its Docker runtime
* EAGERLY, and an SSH one binds a loopback bridge in the constructor path — so
* even a caller that only wanted `.routing` and never touches `.runtime` had
* already bound a listener that only `dispose()` closes.
*/
export async function withDeploymentPlatform<T>(
dep: Pick<Deployment, "meta" | "organizationId">,
fn: (resolved: {
runtime: RuntimeAdapter;
routing: Platform["routing"];
ssl: Platform["ssl"];
effectiveTarget: DeployTarget;
serverId: string | null;
}) => Promise<T>,
): Promise<T> {
const resolved = await resolveDeploymentPlatform((dep.meta ?? {}) as DeploymentMeta, {
organizationId: dep.organizationId,
});
try {
return await fn({
runtime: resolved.platform.runtime,
routing: resolved.platform.routing,
ssl: resolved.platform.ssl,
effectiveTarget: resolved.effectiveTarget,
serverId: resolved.serverId,
});
} catch (err) {
throw asHostUnreachable(err);
} finally {
disposePlatform(resolved);
}
}
/**
* Re-label "we could not reach the host" as a 503 `HostUnreachableError`, keeping
* the underlying message (which already names the target and the fix — see
* ssh-support.ts). Anything else passes through untouched.
*
* The distinction is the whole point: a 400 tells the operator they sent a bad
* request, when in fact their request was fine and the server was not.
*/
function asHostUnreachable(err: unknown): unknown {
if (err instanceof HostUnreachableError) return err;
// isConnectionLoss, not the raw adapter predicate: lib/remote-state.ts is this
// app's one present/absent/unreachable classifier, and it additionally catches
// the executor's lowercase command-timeout string.
if (isHostChannelUnavailableError(err) || isConnectionLoss(err)) {
return new HostUnreachableError(safeErrorMessage(err));
}
return err;
}
export async function resolveDeploymentRuntimeForRead(
dep: Pick<Deployment, "meta" | "organizationId">,
): Promise<{ runtime: RuntimeAdapter; serverId: string | null }> {
@@ -0,0 +1,232 @@
import { beforeEach, describe, expect, it, vi } from "vitest";
/**
* A DERIVED `local` target and the `isLocal` "This Server" row are the SAME machine,
* so they must resolve to the same executor.
*
* `local` is not a destination anyone picks — it is the absence of a binding (no
* cloud workspace, no serverId), and `project.server_id` is ON DELETE SET NULL, so
* deleting a server is enough to make the next deploy for that project derive it.
* That branch used to take `createPlatform`'s default local executor, which inside
* the api container runs host-side steps against the CONTAINER's filesystem — the
* exact hazard `createHostExecutor` exists to refuse, reached through a second door.
*
* What's pinned here: that both doors take the ONE pooled host channel, that resolving
* a target never REGISTERS one (creating a server row is preparation, not part of a
* deploy), and that a dead channel refuses out loud at USE while the container workload
* still deploys.
*/
const h = vi.hoisted(() => ({
/** Every createPlatform config, in order. */
configs: [] as Array<Record<string, unknown>>,
/**
* The pooled host channel. ONE object, handed back by both doors, because in
* production they are one: `acquire(localRow)` resolves through
* `acquireLocalHost` → `acquireHostChannel`, and the row id only adds the borrow
* marker. A mock that returned two distinct executors would hide the very
* convergence under test.
*/
channel: { tag: "pooled-host-channel" } as unknown,
acquire: vi.fn(async (_id: string) => h.channel),
acquireHostChannel: vi.fn(async () => h.channel),
/** The UNPOOLED idiom. Nothing here may reach it; #291 is what happens when it does. */
hostExecutor: vi.fn((): unknown => ({ tag: "direct-host-executor" })),
localRow: null as { id: string } | null,
findCalls: 0,
findRejects: false,
ensureCalls: 0,
}));
vi.mock("@repo/adapters", async (importOriginal) => ({
// Real `unavailableExecutor` / `HostChannelUnavailableError` / detection — the
// degrade under test is theirs, and stubbing it would only assert the stub.
...(await importOriginal<Record<string, unknown>>()),
createPlatform: async (config: Record<string, unknown>) => {
h.configs.push(config);
return { target: "selfhosted" };
},
createHostExecutor: () => h.hostExecutor(),
}));
vi.mock("./startup/self-server", () => ({
findLocalServer: async () => {
h.findCalls++;
if (h.findRejects) throw new Error("db unavailable");
return h.localRow;
},
// Mocked only so the "never on a resolve" assertion below can see a call if one
// ever appears.
ensureLocalServer: async () => {
h.ensureCalls++;
return h.localRow;
},
}));
vi.mock("@repo/db", () => ({
repos: {
server: {
getInOrganization: async (id: string) => ({
id,
isLocal: true,
sshHost: "127.0.0.1",
sshPort: 22,
sshUser: "root",
}),
update: async () => {},
},
},
}));
// The row IS this box; keyed off the flag so the test doesn't depend on loopback
// resolution or env.
vi.mock("./box-org", () => ({
isLocalHostRow: async (row: { isLocal?: boolean }) => Boolean(row?.isLocal),
}));
vi.mock("./ssh-manager", () => ({
sshManager: { acquire: h.acquire, acquireHostChannel: h.acquireHostChannel },
buildSshConfig: async () => ({ host: "127.0.0.1", port: 22, username: "root" }),
}));
vi.mock("./provision-lock", () => ({
createProvisionLock: (name: string) => ({ name, run: (f: () => unknown) => f() }),
}));
const { resolveTargetPlatform } = await import("./deployment-runtime");
const { HostChannelUnavailableError } = await import("@repo/adapters");
const last = () => h.configs[h.configs.length - 1] as Record<string, unknown>;
beforeEach(() => {
h.configs = [];
h.localRow = null;
h.findCalls = 0;
h.findRejects = false;
h.ensureCalls = 0;
h.acquire.mockReset();
h.acquire.mockImplementation(async () => h.channel);
h.acquireHostChannel.mockReset();
h.acquireHostChannel.mockImplementation(async () => h.channel);
h.hostExecutor.mockReset();
h.hostExecutor.mockImplementation(() => ({ tag: "direct-host-executor" }));
});
describe("derived local target — one machine, one executor path", () => {
it("resolves THROUGH this box's isLocal row when it has one", async () => {
h.localRow = { id: "srv-local" };
await resolveTargetPlatform("local", "docker");
// Pooled, keyed on the canonical row: that pool is what stops one deploy from
// leaving an sshd session behind (#291), and it is the row's own id — never a
// caller-supplied one, since the target already means "here".
expect(h.acquire).toHaveBeenCalledWith("srv-local");
expect(h.hostExecutor).not.toHaveBeenCalled();
expect(last().executor).toBe(h.channel);
});
it("hands the platform the same executor a PICKED 'This Server' row does", async () => {
h.localRow = { id: "srv-local" };
await resolveTargetPlatform("server", "docker", "srv-local", "org1");
const picked = last();
await resolveTargetPlatform("local", "docker");
const derived = last();
// The whole point of the convergence: the wizard door and the no-binding door
// cannot end up on different rules for the same box.
expect(derived.executor).toBe(picked.executor);
expect(derived.localHost).toBe(true);
expect(derived.docker).toEqual(picked.docker);
// Same host being provisioned → same lock, or two "local" deploys would race
// openresty/docker/state on one machine.
expect((derived.provisionLock as { name: string }).name).toBe(
(picked.provisionLock as { name: string }).name,
);
});
it("keeps the box as THIS machine, not a remote one", async () => {
h.localRow = { id: "srv-local" };
await resolveTargetPlatform("local", "docker");
// `createPlatform` infers "this machine" from `localHost ?? !executor`, so an
// injected executor without this reads as REMOTE — which switches off the
// containerized edge provider and the same-path-mount rule for the local box.
expect(last().localHost).toBe(true);
expect(last().ssh).toBeUndefined();
// The container workload still goes over the mounted socket, as before.
expect(last().docker).toEqual({ transport: "socket" });
});
it("bare runtime asks for no docker transport at all", async () => {
h.localRow = { id: "srv-local" };
await resolveTargetPlatform("local", "bare");
expect(last().docker).toBeUndefined();
expect(last().localHost).toBe(true);
});
});
describe("derived local target — boxes with no isLocal row", () => {
it("takes the same pooled host channel, never an unpooled executor", async () => {
// Desktop, the SaaS, and `--no-host-control` all legitimately have no row. The
// row id is bookkeeping (the borrow marker); WHICH executor this box gets must
// not depend on it, and it must never be a fresh `createHostExecutor()` outside
// the pool, the concurrent-acquire dedup and the channel-health gate.
await resolveTargetPlatform("local", "docker");
expect(h.acquireHostChannel).toHaveBeenCalledTimes(1);
expect(h.acquire).not.toHaveBeenCalled();
expect(h.hostExecutor).not.toHaveBeenCalled();
expect(last().executor).toBe(h.channel);
expect(last().localHost).toBe(true);
});
it("a failed row lookup degrades to that same channel, it does not fail the deploy", async () => {
h.findRejects = true;
await resolveTargetPlatform("local", "docker");
expect(h.acquireHostChannel).toHaveBeenCalledTimes(1);
expect(last().executor).toBe(h.channel);
});
it("READS the row and never registers one", async () => {
// Resolving a target is not preparation. Creating here would make a deploy
// responsible for inserting a server row — and drag the creation path's
// public-IP detection (an outbound request) onto a resolve that also runs for
// plain runtime reads. `ensureLocalServer` belongs to boot and the
// admin-establishing endpoints.
await resolveTargetPlatform("local", "docker");
await resolveTargetPlatform("local", "bare");
expect(h.findCalls).toBe(2);
expect(h.ensureCalls).toBe(0);
});
});
describe("derived local target — a dead host channel", () => {
it("still resolves, and refuses host operations at USE with the remedy", async () => {
// The containerized-compose state: the resolve must survive (the workload is
// reached through the mounted socket), and the host must be refused out loud
// instead of quietly writing inside the api container.
h.acquireHostChannel.mockRejectedValue(
new HostChannelUnavailableError(
"not_configured",
"no host channel is configured. Re-run `openship up` to provision the host channel.",
),
);
const platform = await resolveTargetPlatform("local", "docker");
expect(platform).toBeTruthy();
const executor = last().executor as {
exec: (cmd: string) => Promise<unknown>;
exists: (p: string) => Promise<unknown>;
};
await expect(executor.exec("uname -a")).rejects.toThrow(/openship up/);
// Not a silent `false`: an unproven answer that reads as fact is how a blocked
// channel produced wrong work instead of a visible failure.
await expect(executor.exists("/opt/openship")).rejects.toThrow(/no host channel/);
});
it("an AUTH failure is still a real failure, not a degrade", async () => {
h.localRow = { id: "srv-local" };
h.acquire.mockRejectedValue(new Error("All configured authentication methods failed"));
await expect(resolveTargetPlatform("local", "docker")).rejects.toThrow(
/authentication methods failed/,
);
});
});
+30 -11
View File
@@ -5,7 +5,7 @@ import { repos } from "@repo/db";
import { env } from "../config/env";
import { platform } from "./controller-helpers";
import { createProvisionLock } from "./provision-lock";
import { resolveDeploymentPlatform, type DeploymentMeta } from "./deployment-runtime";
import { disposePlatform, resolveDeploymentPlatform, type DeploymentMeta } from "./deployment-runtime";
/**
* The per-domain issuance lock key. EVERY path that can open an ACME order
@@ -341,6 +341,28 @@ async function recoverIssuedCert(
return onDisk;
}
/**
* `resolveDeploymentPlatform` for a caller that wants ONLY `.ssl`.
*
* The transport goes back before we return. Resolving a platform for a remote server
* eagerly binds a Docker-over-SSH bridge — one loopback listener — and this function
* is reached per issuance AND per renewal, so holding it is a leak on a schedule.
* Releasing it is safe rather than lucky: `createInfraProvider` (platform.ts:293) is
* handed the executor and the edge container and is never given the runtime, so
* `.ssl` cannot hold anything `disposePlatform` closes, and the pooled executor it
* does drive certbot through is explicitly kept (see `PLATFORM_DISPOSAL`).
*
* One function because the resolve/dispose/take-`.ssl` triple was written out at all
* three branches below, and the failure mode of forgetting the middle step there is
* invisible: certs still issue, the box just accumulates listeners until it runs out
* of descriptors.
*/
async function resolveSslOnly(meta: DeploymentMeta, organizationId: string): Promise<SslProvider> {
const resolved = await resolveDeploymentPlatform(meta, { organizationId });
disposePlatform(resolved);
return resolved.platform.ssl;
}
/**
* Resolve the SSL provider that runs on the SAME host that serves the domain.
*
@@ -361,11 +383,8 @@ async function resolveSslProvider(owner: SslOwner): Promise<ResolvedSslProvider>
// `lockScope` is the server id so mail issuance takes the same per-box ACME lock
// as the apps sharing that edge (they contend for one standalone challenge port).
if (owner.kind === "mail") {
const resolved = await resolveDeploymentPlatform(
{ serverId: owner.serverId } as DeploymentMeta,
{ organizationId: owner.organizationId },
);
return { ssl: resolved.platform.ssl, lockScope: owner.serverId };
const ssl = await resolveSslOnly({ serverId: owner.serverId } as DeploymentMeta, owner.organizationId);
return { ssl, lockScope: owner.serverId };
}
const project = owner.project;
@@ -375,8 +394,8 @@ async function resolveSslProvider(owner: SslOwner): Promise<ResolvedSslProvider>
if (dep) {
const meta = (dep.meta ?? {}) as DeploymentMeta;
try {
const resolved = await resolveDeploymentPlatform(meta, { organizationId: dep.organizationId });
return { ssl: resolved.platform.ssl, lockScope: meta.serverId ?? LOCAL_ACME_SCOPE };
const ssl = await resolveSslOnly(meta, dep.organizationId);
return { ssl, lockScope: meta.serverId ?? LOCAL_ACME_SCOPE };
} catch (err) {
// Deploy target unresolvable — fall through to the host-anchored fallback.
// But say so with the real cause: for a REMOTE-target project (meta.serverId
@@ -408,11 +427,11 @@ async function resolveSslProvider(owner: SslOwner): Promise<ResolvedSslProvider>
const local = await repos.server.findLocal(project.organizationId).catch(() => null);
if (local) {
try {
const resolved = await resolveDeploymentPlatform(
const ssl = await resolveSslOnly(
{ serverId: local.id } as DeploymentMeta,
{ organizationId: project.organizationId },
project.organizationId,
);
return { ssl: resolved.platform.ssl, lockScope: local.id };
return { ssl, lockScope: local.id };
} catch (err) {
// Host-server unresolvable — last resort below.
console.warn(
+40 -8
View File
@@ -10,8 +10,9 @@
*/
import { repos, type Project } from "@repo/db";
import type { CommandExecutor } from "@repo/adapters";
import { createExecutor, type CommandExecutor } from "@repo/adapters";
import { isLocalHostRow } from "./box-org";
import type { DeploymentMeta } from "./deployment-runtime";
import { sshManager } from "./ssh-manager";
@@ -24,13 +25,16 @@ export async function resolveServerIdForProject(project: Project): Promise<strin
/**
* Run `fn` with an executor that reaches the BOX the project's edge lives on —
* the same host the bare/containerized OpenResty + certbot + /etc/letsencrypt sit
* on. For the auto-registered "this server" (server-host mode) that's the LOCAL
* host (SSH-to-host when the API is itself containerized); for a real remote
* server it's the pooled SSH executor. Returns null when there's no server or the
* box is unreachable. This is what lets cert reuse read the HOST's
* /etc/letsencrypt even when the API runs in a container whose own
* /etc/letsencrypt is a different (empty) volume.
* the same host the bare/containerized OpenResty + certbot sit on. For the
* auto-registered "this server" (server-host mode) that's the LOCAL host
* (SSH-to-host when the API is itself containerized); for a real remote server it's
* the pooled SSH executor. Returns null when there's no server or the box is
* unreachable.
*
* Use this for anything that is genuinely the HOST's: the vhost tree (bind-mounted
* into the api container at a DIFFERENT path), a foreign proxy's config, host
* binaries. For a path the container shares 1:1 with its host, use
* {@link withCertStoreExecutor} instead.
*/
export async function withServerHostExecutor<T>(
project: Project,
@@ -43,3 +47,31 @@ export async function withServerHostExecutor<T>(
// never closed it — one leaked sshd session per domain/SSL status read (#291).
return sshManager.withExecutor(serverId, fn).catch(() => null);
}
/**
* Same, but for certbot's store — which the api container shares with its host at the
* SAME path (`/etc/letsencrypt`, see `EDGE_CONTAINER_MOUNTS`).
*
* So on the local box `fn` reads the same files with no host channel in the path. That
* matters more here than anywhere else the rule applies (`sharedMountExecutor`): when
* the channel is firewalled or switched off, a cert that is sitting on disk stops being
* adoptable and the domain falls through to ACME — which is rate-limited and fails
* behind Cloudflare. The container runs as root, so the 0600 privkey.pem is readable.
*
* ONLY for that store. Nothing else project-scoped is a same-path mount, and reading a
* translated path locally lands somewhere else entirely.
*/
export async function withCertStoreExecutor<T>(
project: Project,
fn: (exec: CommandExecutor) => Promise<T>,
): Promise<T | null> {
const serverId = await resolveServerIdForProject(project);
if (!serverId) return null;
const row = await repos.server.get(serverId).catch(() => null);
// isLocalHostRow, not `row.isLocal`: a plain loopback/SERVER_IP row for this box is
// this box too, and it carries its own org gate so a teammate's org can't claim it.
if (row && (await isLocalHostRow(row))) {
return fn(createExecutor()).catch(() => null);
}
return sshManager.withExecutor(serverId, fn).catch(() => null);
}
+2 -2
View File
@@ -43,8 +43,8 @@ export function withPinnedEdgeImage(config: InstallerConfig = {}): InstallerConf
/**
* apps/edge/ holds the edge's Dockerfile; the image it produces is published as
* `openship-edge`. Same image↔directory rename exception the mail engine has, encoded
* identically in .github/workflows/docker-images.yml — keep the two in step.
* `openship-edge`. The image→Dockerfile map lives in
* .github/workflows/docker-images.yml — keep the two in step.
*/
const EDGE_DOCKERFILE = join("apps", "edge", "Dockerfile");
+5 -1
View File
@@ -23,7 +23,7 @@ import { repos } from "@repo/db";
import type { Platform } from "@repo/adapters";
import { cloudClient } from "./cloud/client";
import { canonicalEdgeTarget } from "./edge-target";
import { resolveTargetPlatform } from "./deployment-runtime";
import { disposePlatform, resolveTargetPlatform } from "./deployment-runtime";
type Routing = Platform["routing"];
@@ -278,6 +278,10 @@ export async function resolveRoutingFor(
try {
const target = serverId ? "server" : "local";
const resolved = await resolveTargetPlatform(target, "docker", serverId, organizationId);
// Only `.routing` is wanted, but "docker" mode already bound a
// Docker-over-SSH bridge for a remote server. Routing drives the box through
// the pooled SSH executor, so it survives the release.
disposePlatform(resolved);
return resolved.routing;
} catch (err) {
console.warn(
+27
View File
@@ -142,6 +142,33 @@ export async function getHostCapacity(
return run;
}
/**
* Capacity, but only when the reading actually describes the machine asked about.
*
* The fallback above is deliberate for a resource CEILING (a number the operator
* typed is better accepted than blocked), yet wrong for a decision that can
* REFUSE: `probe` falls back to this API host's own `os.*`, and the cache is
* keyed org+serverId, so a `source: "local"` reading can legitimately come back
* for a remote box whose daemon was briefly unreachable. Believing it would match
* an app's requirement against the orchestrator's RAM and refuse a deploy to a
* 64 GB server. A `local` number is the truth only when the deploy lands here.
*
* Anything else collapses to UNKNOWN, which every consumer treats as "don't
* enforce" — the one safe direction for a check that can say no.
*/
export async function getTrustedHostCapacity(
serverId: string | undefined,
organizationId: string,
opts: { isLocalTarget: boolean },
): Promise<HostCapacity> {
const capacity = await getHostCapacity(serverId, organizationId, {
localFallback: opts.isLocalTarget,
}).catch(() => ({ ...UNKNOWN_CAPACITY }));
if (capacity.source === "docker") return capacity;
if (capacity.source === "local" && opts.isLocalTarget) return capacity;
return { ...UNKNOWN_CAPACITY };
}
/** Drop cached capacity for a server (or a whole org). Call after a server is
* resized/removed so the next read re-probes. */
export async function invalidateHostCapacity(
@@ -0,0 +1,199 @@
import { describe, expect, it, vi, beforeEach } from "vitest";
/**
* The boot banner exists for boxes installed BEFORE the CLI preflight shipped:
* nothing else on those instances ever tells the operator the host channel is dead
* (#490). So what's pinned here is the operator-facing contract — silence when
* there is nothing to say, and a copy-pasteable rule when there is — plus the
* property that a failed CHECK is never reported as a failed CHANNEL.
*/
const h = vi.hoisted(() => ({ env: { CLOUD_MODE: false, DEPLOY_MODE: "docker" as string } }));
vi.mock("../config/env", () => ({ env: h.env }));
// Only the probe is stubbed: the impact copy the banner prints is shared (#490), and a
// test that asserted against a mocked copy would pass while the real lines said anything.
vi.mock("@repo/adapters", async (importOriginal) => ({
...(await importOriginal<Record<string, unknown>>()),
hostChannelHealth: vi.fn(async () => ({ ok: true, code: "ok" })),
}));
import { HOST_CHANNEL_NOT_PROVISIONED, HOST_CHANNEL_UNPROVISIONED } from "@repo/core";
import { reportHostChannelAtBoot } from "./host-channel-banner";
/**
* Shaped like real `hostChannelHealth` output: the rule is NOT inside the hint. The
* two are separate fields because prose gets wrapped to the terminal and a command
* must not be, and this fixture is the reason the banner can't quietly go back to
* concatenating them.
*/
const UNREACHABLE = {
ok: false,
code: "unreachable" as const,
host: "host.docker.internal",
port: 22,
target: "root@host.docker.internal:22",
rule: "sudo ufw allow from 172.18.0.0/16 to any port 22 proto tcp",
hint:
"root@host.docker.internal:22 — no response, not even a refusal (ETIMEDOUT). That is " +
"what a dropped packet looks like. This address is host-local, so the connection " +
"traverses the host's filter/INPUT chain, where a default-deny firewall applies.",
};
function lines() {
const out: string[] = [];
return { out, log: (line: string) => out.push(line) };
}
beforeEach(() => {
h.env.CLOUD_MODE = false;
h.env.DEPLOY_MODE = "docker";
});
describe("reportHostChannelAtBoot — when it stays quiet", () => {
it("does not even probe in cloud mode", async () => {
h.env.CLOUD_MODE = true;
const { out, log } = lines();
const health = vi.fn();
expect(await reportHostChannelAtBoot({ health, log })).toBe("cloud");
expect(health).not.toHaveBeenCalled();
expect(out).toEqual([]);
});
it("does not probe when DEPLOY_MODE is cloud without CLOUD_MODE", async () => {
h.env.DEPLOY_MODE = "cloud";
const health = vi.fn();
expect(await reportHostChannelAtBoot({ health, log: () => {} })).toBe("cloud");
expect(health).not.toHaveBeenCalled();
});
it("says nothing when the channel answers", async () => {
const { out, log } = lines();
const health = async () => ({ ok: true, code: "ok" as const });
expect(await reportHostChannelAtBoot({ health, log })).toBe("ok");
expect(out).toEqual([]);
});
it("says nothing on a bare install, where there is no channel to dial", async () => {
const { out, log } = lines();
const health = async () => ({ ok: true, code: "not_applicable" as const });
expect(await reportHostChannelAtBoot({ health, log })).toBe("not_applicable");
expect(out).toEqual([]);
});
it("does not nag about a deliberate --no-host-control opt-out", async () => {
const { out, log } = lines();
const health = async () => ({
ok: false,
code: "disabled" as const,
hint: "Host control is off (OPENSHIP_HOST_CONTROL=false).",
});
expect(await reportHostChannelAtBoot({ health, log })).toBe("disabled");
expect(out).toEqual([]);
});
});
describe("reportHostChannelAtBoot — a blocked channel", () => {
it("names the endpoint and prints the rule once, with the blast radius", async () => {
const { out, log } = lines();
expect(await reportHostChannelAtBoot({ health: async () => UNREACHABLE, log })).toBe(
"unreachable",
);
const text = out.join("\n");
expect(text).toContain("HOST CONTROL UNREACHABLE — root@host.docker.internal:22");
// Exactly one copy of the rule, and it is the one under "Fix:".
expect(text.split(UNREACHABLE.rule)).toHaveLength(2);
expect(text).toContain("Fix:");
// Deploys keep working — an operator must not read this as "my box is broken".
expect(text).toContain("deploys to this box still work");
expect(text).toContain('reads "Offline"');
expect(text).toContain("openship up --no-host-control");
// …and the repair comes before the switch that only hides the warning.
expect(text.indexOf(UNREACHABLE.rule)).toBeLessThan(text.indexOf("--no-host-control"));
// Every body line is prefixed so the banner is greppable and visually one block.
expect(out.filter((l) => l !== "").every((l) => l.startsWith("!!! "))).toBe(true);
});
it("wraps the hint without dropping any of it", async () => {
const { out, log } = lines();
await reportHostChannelAtBoot({ health: async () => UNREACHABLE, log });
const body = out
.filter((l) => l.startsWith("!!! "))
.map((l) => l.slice(4))
.join(" ");
expect(body).toContain(UNREACHABLE.hint);
expect(out.every((l) => l.length <= 96)).toBe(true);
});
it("never reflows a multi-line rule — each command stays pasteable", async () => {
const { out, log } = lines();
// In a container we can't tell ufw from firewalld, so the rule offers both and is
// long. Wrapping it would produce commands that look right and don't run.
const rule = [
"# ufw:",
"sudo ufw allow from 172.18.0.0/16 to any port 2222 proto tcp",
"# firewalld:",
`sudo firewall-cmd --permanent --add-rich-rule='rule family="ipv4" source address="172.18.0.0/16" port port="2222" protocol="tcp" accept'`,
"sudo firewall-cmd --reload",
].join("\n");
const health = async () => ({
ok: false,
code: "unreachable" as const,
target: "root@10.0.0.4:2222",
hint: "Nothing answered.",
rule,
});
await reportHostChannelAtBoot({ health, log });
const printed = out.map((l) => l.replace(/^!!!\s*/, ""));
for (const command of rule.split("\n")) expect(printed).toContain(command);
});
it("reports a container with no channel provisioned at all", async () => {
const { out, log } = lines();
// The hint as `hostChannelHealth` actually builds it (#509) — taken from @repo/core
// rather than retyped, or this fixture would keep passing while the real boot log
// said something else.
const health = async () => ({
ok: false,
code: "not_configured" as const,
hint: `${HOST_CHANNEL_UNPROVISIONED} ${HOST_CHANNEL_NOT_PROVISIONED}`,
});
expect(await reportHostChannelAtBoot({ health, log })).toBe("not_configured");
// Unwrapped: the hint is prose, so the banner wraps it and no single output line
// holds the whole sentence.
const text = out.map((l) => l.replace(/^!!!\s*/, "")).join(" ");
expect(out.join("\n")).toContain("HOST CONTROL NOT CONFIGURED");
expect(text).toContain("no host channel");
// Both halves of the remedy: what repairs it, and how to confirm the repair.
expect(text).toContain("openship up");
expect(text).toContain("openship doctor");
// No firewall rule was implicated, so none is suggested.
expect(text).not.toContain("Fix:");
expect(text).not.toContain("ufw");
});
it("reports an unreadable key against its endpoint", async () => {
const { out, log } = lines();
const health = async () => ({
ok: false,
code: "key_unreadable" as const,
target: "root@host.docker.internal:22",
hint: "Cannot read the host SSH key at /app/.ssh/host_key (ENOENT).",
});
expect(await reportHostChannelAtBoot({ health, log })).toBe("key_unreadable");
expect(out.join("\n")).toContain("HOST CONTROL KEY UNREADABLE — root@host.docker.internal:22");
});
});
describe("reportHostChannelAtBoot — a failed check is not a failed channel", () => {
it("swallows a probe that throws and blames nothing", async () => {
const { out, log } = lines();
const health = async () => {
throw new Error("no procfs");
};
expect(await reportHostChannelAtBoot({ health, log })).toBe("error");
expect(out).toEqual([]);
});
});
+88
View File
@@ -0,0 +1,88 @@
import { hostChannelHealth, type HostChannelHealth } from "@repo/adapters";
import {
HOST_CHANNEL_BLOCKED,
HOST_CHANNEL_OPT_OUT,
HOST_CHANNEL_RECHECK,
HOST_CHANNEL_SYMPTOM,
HOST_CHANNEL_UNAFFECTED,
wrapText,
} from "@repo/core";
import { env } from "../config/env";
/**
* Boot-time diagnosis of the container→host SSH channel (#490).
*
* The channel's address is host-LOCAL, so unlike a published container port it
* traverses the host's filter/INPUT chain — where a default-deny ufw silently DROPs
* it. `openship up` now probes during install; this covers boxes provisioned before
* that existed, and any box whose firewall changed under a running install.
*
* Never throws and never blocks boot.
*/
export interface HostChannelBannerDeps {
health: (timeoutMs?: number) => Promise<HostChannelHealth>;
log: (line: string) => void;
}
const TITLES: Partial<Record<HostChannelHealth["code"], string>> = {
unreachable: "HOST CONTROL UNREACHABLE",
not_configured: "HOST CONTROL NOT CONFIGURED",
key_unreadable: "HOST CONTROL KEY UNREADABLE",
};
/** One blocked item, wrapped with a hanging indent so a long one still reads as a
* single bullet rather than two. */
function bullet(item: string): string[] {
return wrapText(item, 84).map((line, i) => (i === 0 ? ` - ${line}` : ` ${line}`));
}
/** The shared #490 impact copy, laid out for a boot log: one reassurance, then the
* blocked list as its own lines so it survives being skimmed. */
const IMPACT = [
...wrapText(HOST_CHANNEL_UNAFFECTED),
"Blocked until the channel works:",
...HOST_CHANNEL_BLOCKED.flatMap(bullet),
...wrapText(HOST_CHANNEL_SYMPTOM),
];
function defaultDeps(): HostChannelBannerDeps {
return { health: hostChannelHealth, log: (line) => console.error(line) };
}
export async function reportHostChannelAtBoot(
overrides: Partial<HostChannelBannerDeps> = {},
): Promise<HostChannelHealth["code"] | "cloud" | "error"> {
// The multi-tenant SaaS drives no host of its own.
if (env.CLOUD_MODE || env.DEPLOY_MODE === "cloud") return "cloud";
const deps = { ...defaultDeps(), ...overrides };
// A diagnostic that can't run says nothing: "we failed to check" must never be
// dressed up as "your firewall is blocking this".
let health: HostChannelHealth;
try {
health = await deps.health();
} catch {
return "error";
}
// `ok` covers a bare install (not_applicable) as well as a working channel.
// `disabled` is a deliberate --no-host-control opt-out: don't nag every restart.
if (health.ok || health.code === "disabled") return health.code;
const title = TITLES[health.code] ?? "HOST CONTROL UNAVAILABLE";
const headline = health.target ? `${title} — ${health.target}` : title;
const body = [...(health.hint ? wrapText(health.hint) : []), ...IMPACT];
// Never wrapped: the rule is multi-line and each line is a command to paste.
if (health.rule) body.push("Fix:", ...health.rule.split("\n").map((line) => ` ${line}`));
// Fix first, then how to confirm it worked, and only then the opt-out — an operator
// reading top-down must reach the repair before the switch that hides the warning.
body.push(...wrapText(HOST_CHANNEL_RECHECK), ...wrapText(HOST_CHANNEL_OPT_OUT));
deps.log("");
deps.log(`!!! ${headline}`);
for (const line of body) deps.log(`!!! ${line}`);
deps.log("");
return health.code;
}
@@ -0,0 +1,195 @@
import { beforeEach, describe, expect, it, vi } from "vitest";
/**
* A local server row's WORKLOAD is reached through the mounted Docker socket, so a
* box that cannot drive its host must still resolve a deploy target (#490).
*
* The old behaviour threw at RESOLVE time, because `createHostExecutor()` throws at
* construction. That made switching host control off — the very remedy printed for a
* firewall-blocked channel — kill every deploy to "This Server", and the two failures
* looked nothing alike. The property pinned here: resolve succeeds, and the host is
* refused at USE, with the reason intact.
*/
const h = vi.hoisted(() => ({
acquire: vi.fn(async (_id: string) => ({ tag: "real-host-channel" }) as unknown),
}));
vi.mock("@repo/db", () => ({
repos: {
server: {
getInOrganization: async (id: string) => ({
id,
isLocal: true,
sshHost: "127.0.0.1",
sshPort: 22,
sshUser: "root",
}),
update: async () => {},
},
},
}));
// The row IS this box; keyed off the flag so the test doesn't depend on loopback
// resolution or env.
vi.mock("./box-org", () => ({
isLocalHostRow: async (row: { isLocal?: boolean }) => Boolean(row?.isLocal),
}));
vi.mock("./ssh-manager", () => ({
sshManager: { acquire: h.acquire },
buildSshConfig: async () => ({ host: "127.0.0.1", port: 22, username: "root" }),
}));
vi.mock("./provision-lock", () => ({
createProvisionLock: () => ({ run: (f: () => unknown) => f() }),
}));
const { resolveServerExecutor, hostChannelDeployNotice } = await import("./deployment-runtime");
const { HostChannelUnavailableError } = await import("@repo/adapters");
const resolve = () => resolveServerExecutor("srv-local", "org1");
beforeEach(() => {
h.acquire.mockReset();
h.acquire.mockImplementation(async () => ({ tag: "real-host-channel" }));
});
describe("resolveServerExecutor — local row with no host channel", () => {
it("still resolves when host control is switched off", async () => {
h.acquire.mockRejectedValue(
new HostChannelUnavailableError("disabled", "Host control is disabled on this instance."),
);
const resolved = await resolve();
expect(resolved.isLocal).toBe(true);
expect(resolved.ssh).toBeNull();
});
it("refuses host operations at use, carrying the remedy", async () => {
h.acquire.mockRejectedValue(
new HostChannelUnavailableError(
"disabled",
"Host control is disabled on this instance (OPENSHIP_HOST_CONTROL=false).",
),
);
const { executor } = await resolve();
await expect(executor.exec("uname -a")).rejects.toThrow(/OPENSHIP_HOST_CONTROL=false/);
// Not a silent `false`: an unproven answer that reads as fact is how a blocked
// channel produced wrong work instead of a visible failure.
await expect(executor.exists("/opt/openship")).rejects.toThrow(/Host control is disabled/);
});
it("still resolves when containerized with no channel provisioned", async () => {
h.acquire.mockRejectedValue(
new HostChannelUnavailableError("not_configured", "no host channel is configured"),
);
const { executor } = await resolve();
await expect(executor.exec("ls")).rejects.toThrow(/no host channel is configured/);
});
it("still resolves when the channel is provisioned and the host firewall drops it", async () => {
// The state #490 actually reported. The workload here is reached through the
// mounted Docker socket, so failing the resolve would take down every container
// deploy to "This Server" over a channel those deploys never touch.
h.acquire.mockRejectedValue(
new HostChannelUnavailableError(
"unreachable",
"Nothing came back from root@host.docker.internal:22 — allow it with: " +
"sudo ufw allow from 172.18.0.0/16 to any port 22 proto tcp",
),
);
const resolved = await resolve();
expect(resolved.isLocal).toBe(true);
// A host-side step still refuses — with the rule, immediately, instead of a bare
// handshake timeout per operation.
await expect(resolved.executor.exec("uname -a")).rejects.toThrow(/sudo ufw allow from/);
});
it("an AUTH failure is still a real failure, not a degrade", async () => {
// The degrade above is specific to "this box can't drive its host". A key the host
// rejects is a different fault with a different fix, and swallowing it would
// recreate #490's actual problem: the deploy proceeds and fails somewhere unrelated.
h.acquire.mockRejectedValue(new Error("All configured authentication methods failed"));
await expect(resolve()).rejects.toThrow(/authentication methods failed/);
});
it("passes the pooled host channel through untouched when it works", async () => {
const { executor } = await resolve();
expect(executor).toEqual({ tag: "real-host-channel" });
});
});
/**
* The demotion above is the moment we decide this box cannot drive its host, and it
* used to leave no trace at all: the typed error was swallowed and an executor that
* refuses everything was handed back in silence. #509 is what that costs — install
* reported success, the box looked healthy, and the operator's first evidence arrived
* as an unrelated-looking failure inside a deploy.
*
* So the decision is announced twice, at the two altitudes that have a reader: the
* process log (here) and the deploy log (below).
*/
describe("the demotion leaves a trace", () => {
it("logs the decision with the remedy, ONCE per outage", async () => {
const warn = vi.spyOn(console, "warn").mockImplementation(() => {});
h.acquire.mockRejectedValue(
new HostChannelUnavailableError(
"not_configured",
"no host channel is configured. Re-run `openship up` to provision the host channel.",
),
);
await resolve();
await resolve();
await resolve();
// Once — this sits on every deploy AND on read paths (logs, status polls), so a
// line per resolve would bury the log it is meant to be found in.
expect(warn).toHaveBeenCalledTimes(1);
expect(warn.mock.calls[0]?.[0]).toContain("openship up");
expect(warn.mock.calls[0]?.[0]).toContain("not_configured");
warn.mockRestore();
});
it("re-arms after the channel recovers, so the next outage is not deduped away", async () => {
const reason = "the firewall dropped it — sudo ufw allow from 172.18.0.0/16 to any port 22 proto tcp";
h.acquire.mockRejectedValue(new HostChannelUnavailableError("unreachable", reason));
await resolve();
const warn = vi.spyOn(console, "warn").mockImplementation(() => {});
h.acquire.mockImplementation(async () => ({ tag: "real-host-channel" }));
await resolve();
h.acquire.mockRejectedValue(new HostChannelUnavailableError("unreachable", reason));
await resolve();
expect(warn).toHaveBeenCalledTimes(1);
warn.mockRestore();
});
});
describe("hostChannelDeployNotice", () => {
it("gives a deploy log one line naming the skip, the reason and the reassurance", async () => {
h.acquire.mockRejectedValue(
new HostChannelUnavailableError(
"not_configured",
"no host channel is configured. Re-run `openship up` to provision the host channel.",
),
);
const { executor } = await resolve();
const notice = hostChannelDeployNotice(executor);
// Named, so the operator connects it to the "couldn't read occupancy" and
// "deploy continues" lines that follow — those name neither the channel nor a fix.
expect(notice).toContain("port-occupancy");
expect(notice).toContain("edge/routing");
expect(notice).toContain("openship up");
// Straight from @repo/core: this is a degraded install, not a broken one, and the
// deploy in progress is going to succeed.
expect(notice).toContain("Ordinary deploys to this box still work");
});
it("says nothing about a channel nothing has decided anything about", async () => {
const { executor } = await resolve();
// A working pooled channel, and the no-executor case (cloud) — neither is
// evidence of a demotion, and a notice printed on either is a lie about the box.
expect(hostChannelDeployNotice(executor)).toBeNull();
expect(hostChannelDeployNotice(null)).toBeNull();
expect(hostChannelDeployNotice(undefined)).toBeNull();
});
});
+20 -4
View File
@@ -24,10 +24,26 @@ export function pinnedMailImage(): string {
}
/**
* apps/email/ holds the engine's Dockerfile; the image it produces is published as
* `openship-mail`. That rename is the one place the image↔directory mapping isn't
* 1:1, and .github/workflows/docker-images.yml encodes the same exception in its
* build matrix — keep the two in step if either side ever moves.
* apps/email/ holds TWO Dockerfiles, published under two names: this one — the
* engine, `openship-mail` — and `Dockerfile.webmail`, the Zero client published as
* `openship-webmail`. So the directory neither names its image nor owns just one;
* .github/workflows/docker-images.yml carries the image→Dockerfile map explicitly
* for the same reason. Keep the two in step if either side moves.
*
* Only the engine is a MANAGED image (pinned to APP_VERSION, built from source in a
* checkout, reconciled by this API). Webmail is a catalog app: it is installed and
* deployed like any other template, so it has no pin and no build spec here — see
* packages/core/src/apps/catalog/webmail.json.
*
* That catalog entry names `openship-webmail:latest`, deliberately unpinned, unlike
* every third-party app in the catalog. Two reasons. Skew is harmless: the engine is
* pinned because this API execs into it and assumes its layout, whereas webmail is a
* standalone IMAP/SMTP client nothing here reaches into. And a pinned tag would have
* to be rewritten in generated, drift-tested JSON on every release (scripts/release.ts
* syncs package.json versions only), so the pin would rot into a tag that was never
* published. Floating matches how openship ships its own other images (the compose
* install defaults to :latest, overridable by --image-version); an app update is a
* redeploy with trigger "update", which is the one trigger that force-pulls.
*/
const MAIL_DOCKERFILE = join("apps", "email", "Dockerfile");
+14 -1
View File
@@ -1,6 +1,7 @@
import {
NginxProvider,
detectOpenRestyPaths,
rootOrDegrade,
type OpenRestyPaths,
} from "@repo/adapters";
import type { CommandExecutor } from "@repo/adapters";
@@ -51,9 +52,21 @@ export async function withOpenRestyRouting<T>(
}
return sshManager.withExecutor(serverId, async (executor) => {
// `applyRateLimit` — the one write behind this helper — edits the root-owned
// nginx.conf, and this path never asked for privilege, so a non-root login got an
// EACCES rendered as "Failed to update OpenResty rate limit config". Elevated when
// the box allows it; degraded (not refused) when it doesn't, because the READS here
// work unelevated today on a world-readable nginx.conf and refusing would take the
// Security tab away from boxes where it currently works.
const edgeExecutor = await rootOrDegrade(executor, {
purpose: "Editing OpenResty configuration",
consequence: "Reads still work; a write will fail with the permission error it earns.",
report: (message) => console.error(`[openresty] ${message}`),
});
const run = async (forceRefresh = false) => {
const paths = await getOpenRestyPaths(serverId, executor, forceRefresh);
const routing = new NginxProvider({ paths, executor });
const routing = new NginxProvider({ paths, executor: edgeExecutor });
return fn(routing);
};
+8 -1
View File
@@ -16,7 +16,11 @@
import type { CommandExecutor } from "@repo/adapters";
import { repos, dumpSubgraph, type Project, type Deployment } from "@repo/db";
import { safeErrorMessage } from "@repo/core";
import { resolveDeploymentPlatform, type DeploymentMeta } from "./deployment-runtime";
import {
disposePlatform,
resolveDeploymentPlatform,
type DeploymentMeta,
} from "./deployment-runtime";
import {
readManifest,
writeManifest,
@@ -158,6 +162,9 @@ export async function removeProjectFromServerManifests(project: Project): Promis
const resolved = await resolveDeploymentPlatform(meta, {
organizationId: dep.organizationId,
});
// Only the executor is wanted (which sshManager owns), so the docker
// transport this eagerly bound can go straight back.
disposePlatform(resolved);
const exec = resolved.platform.executor;
if (exec) {
await removeProjectFromManifest(exec, project.id);
@@ -0,0 +1,116 @@
import { describe, it, expect } from "vitest";
import type { CommandExecutor } from "@repo/adapters";
// The real fixture rather than a hand-rolled profile literal: this exercises the actual
// resolver + privilege gate, so a probe shape change must break it rather than pass.
import { probeOutput } from "../../../../packages/adapters/src/system/environment.fixtures";
import {
OPENSHIP_DIR,
openshipFileExists,
readOpenshipFile,
writeOpenshipFile,
} from "./openship-server-store";
/**
* Privilege for the server state store.
*
* `.openship` is 0700 root-owned by design, so on a host we log in to as a non-root user
* EVERY operation here needs elevation — including the reads, which is the half that hid:
* a `cat` of a file the login user could never have written returns "" and is
* indistinguishable from "this server has no Openship state", which is exactly what
* `scan` reads to re-import our own projects.
*
* A fresh executor object per case on purpose: the profile is cached per executor
* identity, so sharing one would leak the first case's host into the rest.
*/
function fakeHost(probe: string, fileBody = "PAYLOAD") {
const commands: string[] = [];
const writes: string[] = [];
const executor = {
exec: async (command: string) => {
commands.push(command);
if (command.includes("opsh_begin")) return probe;
if (command.includes("cat ")) return fileBody;
if (command.includes("test -f")) return "yes";
return "";
},
writeFile: async (path: string) => {
writes.push(path);
},
} as unknown as CommandExecutor;
// The store's own commands. The probe is filtered out because it carries a `sudo -n
// true` of its own — a bare "no sudo anywhere" assertion would always fail.
const stored = () => commands.filter((c) => !c.includes("opsh_begin"));
return { executor, commands, writes, stored };
}
const NON_ROOT_SUDO = probeOutput({
uid: "1000",
user: "deploy",
home: "/home/deploy",
sudo: "y",
});
const NON_ROOT_NO_SUDO = probeOutput({
uid: "1000",
user: "deploy",
home: "/home/deploy",
sudo: "n",
});
describe("openship-server-store privilege", () => {
it("elevates every operation on a non-root login, without moving the path", async () => {
const write = fakeHost(NON_ROOT_SUDO);
await writeOpenshipFile(write.executor, "manifest.json", "{}");
// Staged through a user-writable temp, then moved into the root-owned dir as root —
// SFTP has no shell and cannot be elevated, so a direct write would EACCES.
expect(write.writes).toHaveLength(1);
expect(write.writes[0]!.startsWith("/tmp/")).toBe(true);
// Two hops, both as root: the staging mv out of /tmp, then the atomic promote.
const mv = write.stored().filter((c) => c.includes("mv -f"));
expect(mv).toHaveLength(2);
expect(mv.every((c) => c.includes("sudo -n sh -c"))).toBe(true);
expect(mv[0]).toContain(`${OPENSHIP_DIR}/manifest.json.tmp`);
expect(mv[1]).toContain(`${OPENSHIP_DIR}/manifest.json`);
const read = fakeHost(NON_ROOT_SUDO);
expect(await readOpenshipFile(read.executor, "manifest.json")).toBe("PAYLOAD");
expect(read.stored().find((c) => c.includes("cat "))).toContain("sudo -n sh -c");
const stat = fakeHost(NON_ROOT_SUDO);
expect(await openshipFileExists(stat.executor, "manifest.json")).toBe(true);
expect(stat.stored().find((c) => c.includes("test -f"))).toContain("sudo -n sh -c");
});
it("leaves a root login exactly as it was — no sudo, same path", async () => {
const { executor, stored, writes } = fakeHost(probeOutput());
await writeOpenshipFile(executor, "manifest.json", "{}");
expect(writes).toEqual([`${OPENSHIP_DIR}/manifest.json.tmp`]);
expect(stored().some((c) => c.includes("sudo"))).toBe(false);
});
it("still reads state on a host it could not measure", async () => {
// A banner or forced command means no profile. Refusing here would turn a readable
// manifest into a missing one — a regression dressed as a safety check.
const { executor, stored } = fakeHost("Please login as the user \"ec2-user\"\n");
expect(await readOpenshipFile(executor, "manifest.json")).toBe("PAYLOAD");
expect(stored().some((c) => c.includes("sudo"))).toBe(false);
});
it("names the fix when the login is measurably unprivileged", async () => {
const write = fakeHost(NON_ROOT_NO_SUDO);
await expect(writeOpenshipFile(write.executor, "manifest.json", "{}")).rejects.toThrow(
/passwordless sudo/,
);
expect(write.writes).toEqual([]);
// Reads stay non-throwing by contract; absent state is still absent state.
const read = fakeHost(NON_ROOT_NO_SUDO);
expect(await readOpenshipFile(read.executor, "manifest.json")).toBe("");
});
});
+53 -10
View File
@@ -11,10 +11,21 @@
* NOT re-implement the folder/mkdir/atomic-write logic — call these helpers.
*/
import type { CommandExecutor } from "@repo/adapters";
import {
HOST_STATE_DIR,
privilegedExecutor,
type CommandExecutor,
} from "@repo/adapters";
/** The one folder. Nothing else hard-codes this path. */
export const OPENSHIP_DIR = "/root/.openship";
/**
* The one folder — now spelled once, in the resolver's ops layer.
*
* Kept as a named re-export because every consumer here composes paths *under* it
* (`MANIFEST_PATH`, `STATE_FILE_PATH`) at module scope, where there is no target host
* to ask. The constant is right for that: these files are root-owned by contract on
* every managed Linux host, which is exactly the case `stateDir()` returns unchanged.
*/
export const OPENSHIP_DIR = HOST_STATE_DIR;
/**
* Single-quote wrap for safe interpolation into a remote LOGIN SHELL. The file
@@ -27,12 +38,38 @@ function sq(v: string): string {
return `'${v.replace(/'/g, "'\\''")}'`;
}
/**
* The executor these helpers actually run through.
*
* The directory is 0700 root-owned by design, so on a host we log in to as a non-root
* user EVERY operation here needs elevation — not just the writes. It used to need none
* because the login was assumed to be root: `mkdir -p /root/.openship` is the first
* thing such a server hits, and a `cat` of a file it could never have written returns
* "" — indistinguishable from "this server has no Openship state", which is what `scan`
* reads to re-import our own projects.
*
* The PATH does not move for a non-root login, unlike the remote journal's: elevation is
* available here (`elevatedExecutor`'s `writeFile` stages through `/tmp` then sudo-mv's,
* so the atomic write survives it), and this state is read back later — sometimes by a
* different login user — so a host provisioned before this fix must still be found.
*/
async function storeExecutor(exec: CommandExecutor, purpose: string): Promise<CommandExecutor> {
const grant = await privilegedExecutor(exec, purpose, { onRefusedHost: "proceed" });
if (!grant.supported) throw new Error(grant.reason);
return grant.value.executor;
}
/** The dir, created through an executor a caller has already gated. */
async function mkdirOpenship(e: CommandExecutor): Promise<void> {
await e.exec(`mkdir -p ${sq(OPENSHIP_DIR)} && chmod 0700 ${sq(OPENSHIP_DIR)}`);
}
/**
* Ensure the `.openship` dir exists, root-only (0700). Idempotent. THE single
* place the folder is created — callers never `mkdir` it themselves.
*/
export async function ensureOpenshipDir(exec: CommandExecutor): Promise<void> {
await exec.exec(`mkdir -p ${sq(OPENSHIP_DIR)} && chmod 0700 ${sq(OPENSHIP_DIR)}`);
await mkdirOpenship(await storeExecutor(exec, "Writing Openship server state"));
}
/**
@@ -42,7 +79,8 @@ export async function ensureOpenshipDir(exec: CommandExecutor): Promise<void> {
export async function readOpenshipFile(exec: CommandExecutor, name: string): Promise<string> {
const path = `${OPENSHIP_DIR}/${name}`;
try {
return (await exec.exec(`cat ${sq(path)} 2>/dev/null || echo ""`)).trim();
const e = await storeExecutor(exec, "Reading Openship server state");
return (await e.exec(`cat ${sq(path)} 2>/dev/null || echo ""`)).trim();
} catch {
return "";
}
@@ -59,22 +97,27 @@ export async function writeOpenshipFile(
): Promise<void> {
const path = `${OPENSHIP_DIR}/${name}`;
const tmp = `${path}.tmp`;
await ensureOpenshipDir(exec);
await exec.writeFile(tmp, content);
await exec.exec(`mv -f ${sq(tmp)} ${sq(path)} && chmod 0600 ${sq(path)}`);
// One grant for all three steps: the staged write must land under the dir this same
// grant just created, and re-gating per step would re-probe for nothing.
const e = await storeExecutor(exec, "Writing Openship server state");
await mkdirOpenship(e);
await e.writeFile(tmp, content);
await e.exec(`mv -f ${sq(tmp)} ${sq(path)} && chmod 0600 ${sq(path)}`);
}
/** Remove a file (and any stale temp) from `.openship`. Idempotent. */
export async function removeOpenshipFile(exec: CommandExecutor, name: string): Promise<void> {
const path = `${OPENSHIP_DIR}/${name}`;
await exec.exec(`rm -f ${sq(path)} ${sq(`${path}.tmp`)}`);
const e = await storeExecutor(exec, "Removing Openship server state");
await e.exec(`rm -f ${sq(path)} ${sq(`${path}.tmp`)}`);
}
/** Cheap existence check (no read) — `true` iff `.openship/<name>` is a file. */
export async function openshipFileExists(exec: CommandExecutor, name: string): Promise<boolean> {
const path = `${OPENSHIP_DIR}/${name}`;
try {
return (await exec.exec(`test -f ${sq(path)} && echo yes || echo no`)).trim() === "yes";
const e = await storeExecutor(exec, "Reading Openship server state");
return (await e.exec(`test -f ${sq(path)} && echo yes || echo no`)).trim() === "yes";
} catch {
return false;
}
+41 -19
View File
@@ -278,7 +278,7 @@ export const PROJECT_ROOTED: ReadonlySet<CheckedResourceType> = new Set([
/**
* Cloud fallback for the org lookup in `assert`: when a project-rooted resource
* has no local row, it may live on the SaaS. Return the request-scope org IFF
* has no local row, it may live on the SaaS. Return the caller's scope org IFF
* that org has a cloud link to proxy through; otherwise null (→ 404, IDOR-safe).
*
* The role check in `checkPermission` then runs against this org: owner/admin/
@@ -287,12 +287,11 @@ export const PROJECT_ROOTED: ReadonlySet<CheckedResourceType> = new Set([
* remains the authoritative per-project gate; a bogus id still 404s once proxied.
*/
async function resolveCloudFallbackOrg(
c: Context,
resourceType: CheckedResourceType,
scopeOrg: string | null,
): Promise<string | null> {
if (env.CLOUD_MODE) return null; // the SaaS IS canonical — no upstream to fall back to
if (!PROJECT_ROOTED.has(resourceType)) return null;
const scopeOrg = resolveRequestScopeOrg(c);
if (!scopeOrg) return null;
const ownerUserId = await resolveOrgCloudUserId(scopeOrg).catch(() => null);
return ownerUserId ? scopeOrg : null;
@@ -460,21 +459,24 @@ function permitsAction(permissions: readonly Permission[], action: Permission):
}
/**
* Resolve the org an authz check for `input` runs against: the request-scope org
* for list scope / org-singletons (resourceId "*"), else the resource's OWN org
* (with the cloud-project fallback — a project with no local row may be a CLOUD
* project canonical on the SaaS). Shared by `assert` + `checkPermissionOnResource`
* so use-time and mint-time org resolution can never drift.
* Resolve the org an authz check for `input` runs against: `scopeOrg` for the arms
* that have no resource to resolve (list scope / org-singletons at resourceId "*"),
* else the resource's OWN org — with the cloud-project fallback, since a project
* with no local row may be a CLOUD project canonical on the SaaS.
*
* `scopeOrg` is supplied by the caller, and the difference between the two callers
* is the point: `assert` ESTABLISHES request scope (from `X-Organization-Id`), while
* `checkPermissionOnResource` runs afterwards and CONSUMES the scope `assert`
* already resolved (`ctx.organizationId`). Everything downstream of that one choice
* is shared, so use-time and mint-time org resolution can never drift.
*/
async function resolveInputOrg(
ctx: RequestContext,
input: PermissionInput,
scopeOrg: string | null,
): Promise<string | null> {
if (input.scope === "list" || input.resourceId === "*") {
return resolveRequestScopeOrg(ctx.hono);
}
if (input.scope === "list" || input.resourceId === "*") return scopeOrg;
const resource = await resolveResourceOrg(input.resourceType, input.resourceId);
return resource?.orgId ?? (await resolveCloudFallbackOrg(ctx.hono, input.resourceType));
return resource?.orgId ?? (await resolveCloudFallbackOrg(input.resourceType, scopeOrg));
}
/**
@@ -499,12 +501,30 @@ function permissionOpts(ctx: RequestContext) {
* verifying the granted resource belongs to that org — so a grant naming another
* org's resource id would be accepted at mint (privilege escalation, SaaS audit).
* This makes mint-time acceptance consistent with `assert`'s use-time check.
*
* The arms with no resource to resolve — list scope and org-singletons
* (`resourceId: "*"`) — take their authority from the caller's ROLE in an org, so
* WHICH org is the whole decision. It is `ctx.organizationId`, never the raw
* `X-Organization-Id` header, for two reasons:
*
* - Every caller runs AFTER `assert` (via routePermission) has resolved the
* request's authoritative org and rebound ctx to it, and then reads its actual
* DATA from `ctx.organizationId`. Re-deriving from the header would gate on one
* org what the handler goes on to do in another — e.g. `canRunJob` on
* `/projects/:id/…` checked `{job,"*",write}` against the header while the
* project resolved to a different org.
* - A mint path writes the binding to `ctx.organizationId` (MCP consent picks the
* org explicitly — see `mintContextFor`), so a caller-chosen header could name a
* different org: a member of the target org gets a `billing`/`audit` grant
* validated against an org they happen to own. GHSA-qv27-39pc-qw9f finding 1.
*
* Consequence: this never touches `ctx.hono`, so it also holds for a background ctx.
*/
export async function checkPermissionOnResource(
ctx: RequestContext,
input: PermissionInput,
): Promise<boolean> {
const organizationId = await resolveInputOrg(ctx, input);
const organizationId = await resolveInputOrg(input, ctx.organizationId);
if (!organizationId) return false;
return checkPermission(ctx.userId, organizationId, input, permissionOpts(ctx));
}
@@ -535,11 +555,13 @@ export async function checkPermissionOnResource(
export async function assert(ctx: RequestContext, input: PermissionInput): Promise<void> {
const c = ctx.hono;
// Resolve the resource's OWN org (list/singleton → request scope) + gate on
// role. Shared with checkPermissionOnResource so mint-time acceptance and
// use-time enforcement can't drift. On deny we throw NotFoundError (not 403)
// so out-of-permission resources don't leak existence — the IDOR-safe pattern.
const organizationId = await resolveInputOrg(ctx, input);
// Resolve the resource's OWN org + gate on role. This is where request scope is
// ESTABLISHED, so the list/singleton arms read the header here (and only here) —
// every later check in the request consumes the org this rebinds ctx to. Shared
// with checkPermissionOnResource so mint-time acceptance and use-time enforcement
// can't drift. On deny we throw NotFoundError (not 403) so out-of-permission
// resources don't leak existence — the IDOR-safe pattern.
const organizationId = await resolveInputOrg(input, resolveRequestScopeOrg(c));
if (!organizationId) {
throw new NotFoundError(input.resourceType, input.resourceId);
}
+119
View File
@@ -0,0 +1,119 @@
import { describe, expect, it, vi, beforeEach } from "vitest";
/**
* What's pinned here is the PRECEDENCE, because each rung of it was a deliberate
* choice with a failure behind it:
*
* • CLOUD_MODE outranks everything — /api/mail is localOnly, so a mail rail on
* the SaaS is a screen of links that all 403.
* • the stored setting outranks env — the toggle has to work without an operator
* editing the launcher's environment and restarting the API.
* • a broken settings read falls back to env instead of failing closed — this
* value picks which nav to draw; there is nothing to fail closed about, and a
* mail-only operator dropped into a platform UI can't get to the mail pages.
*/
const h = vi.hoisted(() => ({
env: { CLOUD_MODE: false, OPENSHIP_PRODUCT: "platform" as string },
/** What `instanceSettings.get()` answers — or throws, when it's an Error. */
settings: null as unknown,
}));
vi.mock("../config/env", () => ({ env: h.env }));
vi.mock("@repo/db", () => ({
repos: {
instanceSettings: {
get: async () => {
if (h.settings instanceof Error) throw h.settings;
return h.settings;
},
},
},
}));
import { clearProductModeCache, isProductMode, resolveProductMode } from "./product-mode";
beforeEach(() => {
h.env.CLOUD_MODE = false;
h.env.OPENSHIP_PRODUCT = "platform";
h.settings = null;
clearProductModeCache();
});
describe("resolveProductMode", () => {
it("defaults to the platform with nothing declared and nothing stored", async () => {
expect(await resolveProductMode()).toBe("platform");
});
it("honours the launcher's declaration when nothing is stored", async () => {
// `openship up --mail` → OPENSHIP_PRODUCT=mail in the compose .env / unit argv.
h.env.OPENSHIP_PRODUCT = "mail";
expect(await resolveProductMode()).toBe("mail");
});
it("lets the stored setting override the declaration, in both directions", async () => {
h.settings = { productMode: "mail" };
expect(await resolveProductMode()).toBe("mail");
// And the case the dashboard toggle depends on: turning mail OFF on a box whose
// env declares it. The toggle writes an explicit "platform" for exactly this —
// clearing the row would hand the decision back to OPENSHIP_PRODUCT.
clearProductModeCache();
h.env.OPENSHIP_PRODUCT = "mail";
h.settings = { productMode: "platform" };
expect(await resolveProductMode()).toBe("platform");
});
it("ignores a stored value that isn't a mode", async () => {
h.env.OPENSHIP_PRODUCT = "mail";
h.settings = { productMode: "webmail" };
expect(await resolveProductMode()).toBe("mail");
});
it("forces the platform in cloud mode, whatever env or the DB say", async () => {
h.env.CLOUD_MODE = true;
h.env.OPENSHIP_PRODUCT = "mail";
h.settings = { productMode: "mail" };
expect(await resolveProductMode()).toBe("platform");
});
it("never caches the cloud answer over a self-hosted one", async () => {
// The cloud short-circuit returns BEFORE the cache is consulted or written, so a
// cached "platform" from a cloud call can't outlive it. Pinned because the check
// sits above the cache read and reordering the two would look harmless.
h.env.CLOUD_MODE = true;
expect(await resolveProductMode()).toBe("platform");
h.env.CLOUD_MODE = false;
h.settings = { productMode: "mail" };
expect(await resolveProductMode()).toBe("mail");
});
it("falls back to the declaration when the settings read fails", async () => {
h.env.OPENSHIP_PRODUCT = "mail";
h.settings = new Error("no database");
expect(await resolveProductMode()).toBe("mail");
});
it("caches until a write clears it", async () => {
h.settings = { productMode: "mail" };
expect(await resolveProductMode()).toBe("mail");
// /health/env is unauthenticated and polled by every dashboard load, so the
// second read must not hit the DB — a changed row is only picked up once the
// settings write clears the cache.
h.settings = { productMode: "platform" };
expect(await resolveProductMode()).toBe("mail");
clearProductModeCache();
expect(await resolveProductMode()).toBe("platform");
});
});
describe("isProductMode", () => {
it("accepts the two modes and nothing else", () => {
expect(isProductMode("platform")).toBe(true);
expect(isProductMode("mail")).toBe(true);
for (const bad of ["Mail", "", null, undefined, 1, {}]) {
expect(isProductMode(bad)).toBe(false);
}
});
});
+70
View File
@@ -0,0 +1,70 @@
/**
* THE product-mode resolver — which product this instance presents itself as.
*
* "platform" → the full deploy platform (default)
* "mail" → Openship Mail: the dashboard's left rail becomes the mail
* control plane and the platform nav is hidden
*
* Precedence:
* 1. CLOUD_MODE → always "platform", unconditionally. The entire /api/mail
* mount is localOnly (modules/mail/mail.routes.ts), so a mail-mode SaaS
* shell would render a rail of links that all 403. Not a preference.
* 2. instance_settings.product_mode — the operator's dashboard toggle.
* 3. OPENSHIP_PRODUCT — the launcher-declared instance default.
*
* NOTE the precedence is the OPPOSITE of getAuthMode() (lib/auth-mode.ts), which
* sits next door and lets env PIN the value against the DB. That asymmetry is
* deliberate: authMode is a security decision, so a declaration must not be
* overridable by a database write. Product mode is presentation, and the whole
* point of the settings row is that flipping the toggle takes effect without
* editing env and restarting the API. If this ever starts gating a route, that
* reasoning no longer holds and the precedence has to be revisited.
*
* Mail mode is a nav + branding SCOPE, never an authorization boundary. Nothing
* downstream may refuse a request because of it: the mail installer is the deploy
* machinery, and webmail ships as an ordinary catalog app through the standard
* pipeline (modules/mail/webmail/webmail-install.service.ts), so gating the
* platform endpoints on mail mode would break mail itself.
*/
import { env } from "../config/env";
/** The canonical mode set. Shared with the env parser and the write validator so
* there is exactly one list. */
export const PRODUCT_MODES = ["platform", "mail"] as const;
export type ProductMode = (typeof PRODUCT_MODES)[number];
export function isProductMode(value: unknown): value is ProductMode {
return PRODUCT_MODES.includes(value as ProductMode);
}
/** Cached because /health/env is unauthenticated and polled by every dashboard
* load. Cleared by every instance-settings write. */
let cached: ProductMode | null = null;
export async function resolveProductMode(): Promise<ProductMode> {
// Multi-tenant SaaS: mail is self-hosted-only infrastructure. Checked first so
// no stored value or env var can produce a dead rail on the cloud.
if (env.CLOUD_MODE) return "platform";
if (cached !== null) return cached;
// Unreadable settings fall back to the env default rather than failing closed.
// This value picks which nav to draw; there is nothing to fail closed ABOUT,
// and refusing to honour a declared OPENSHIP_PRODUCT=mail because one query
// failed would show a mail-only operator a platform UI they can't use.
try {
const { repos } = await import("@repo/db");
const stored = (await repos.instanceSettings.get())?.productMode;
cached = isProductMode(stored) ? stored : env.OPENSHIP_PRODUCT;
} catch {
cached = env.OPENSHIP_PRODUCT;
}
return cached;
}
/** Clear the cached value — called after setup.controller writes. */
export function clearProductModeCache() {
cached = null;
}
+19 -4
View File
@@ -94,7 +94,9 @@ async function resolveProjectTrackedDomains(project: Project): Promise<string[]>
*
* Server resolution order:
* 1. Active deployment's `meta.serverId`
* 2. First configured server (single-server setups)
* 2. This box's own canonical (`isLocal`) row, scoped to the project's organization
*
* There is no third step: a project whose org has neither is not observable from here.
*/
export async function resolveProjectTracking(projectId: string): Promise<ProjectTracking | null> {
const source = await resolveProjectTrafficSource(projectId);
@@ -128,10 +130,23 @@ async function buildTrafficSourcesForDomains(
}));
}
// Server: deployment meta first, then first configured server.
// Server: the deployment's own meta, else THIS box's canonical row — the same rule
// `postEdgeMgmt` uses below, so read and write resolve one machine.
//
// NOT `server.list()[0]`: that query is unscoped by organization AND by `isLocal`
// (server.repo.ts), so it returns the oldest row in the whole instance. On a box
// whose oldest row is a remote or foreign-org server — desktop, host-control-off,
// or an install that added remotes before its own row existed — a derived-local
// project (no `meta.serverId`, which is every one of them: deployment-runtime
// records the id only for the "server" target) had its traffic read from, and its
// tracked hostnames written to, a machine it never deployed to. `sshManager.acquire`
// looks the id up with no organizationId, and the caller holds only `{project, read}`.
//
// No local row means this box is not a deploy target, so there is no edge of ours to
// read and returning nothing is the correct answer, not a guess at another host.
if (!serverId) {
const servers = await repos.server.list();
serverId = servers[0]?.id ?? null;
const local = await repos.server.findLocal(project.organizationId).catch(() => undefined);
serverId = local?.id ?? null;
}
if (!serverId) return [];
@@ -0,0 +1,59 @@
/**
* The one place a project's vercel.json is compiled into the fields a `registerRoute`
* call must carry.
*
* It exists because `registerRoute` REPLACES the whole vhost file: a caller that omits
* these does not leave them alone, it DELETES them. That is what made the feature
* invisible in practice — the fields were built inline in only two places (the per-domain
* live path and the composite monorepo path), so a plain deploy neither applied a
* project's `cleanUrls`/`redirects`/`headers` nor preserved them, and an unrelated
* redeploy silently wiped whatever a "Retry routing" had installed.
*
* Deliberately a PURE module with no platform/runtime imports: `route-apply.service` is
* routinely `vi.mock`ed to keep side effects out of tests, and a pure compile step living
* there would be mocked away with it.
*/
import { compileVercelRouting } from "@repo/adapters";
import type { RoutingConfig } from "@repo/core";
import type { RouteRegister } from "./route-apply.service";
/** The compiled vercel.json fields a `RouteRegister` carries, and nothing else. */
export type CompiledRoutingFields = Pick<
RouteRegister,
"proxyLocations" | "redirects" | "headerRules" | "cleanUrls" | "trailingSlash"
>;
/**
* Compile a project's routing config into spreadable `RouteRegister` fields.
*
* Deliberately NO `backendTargetUrl`: which upstream a path rewrite (`/api/(.*)` →
* `/api/index.js`, a function on Vercel) belongs to is a TOPOLOGY question a single
* register site cannot answer — passing the domain's own upstream would, on a composite
* monorepo, point `/api/` at the FRONTEND. The composite path compiles its own with the
* backend it resolved; everything else gets the topology-free subset, which is why a
* full-URL rewrite still works here and a path rewrite is reported as skipped.
*
* `onSkipped` reports the rules that did NOT make it into the returned fields, so a
* refused rule is visible instead of vanishing. Wire it ONLY where no topology-aware
* pass follows: because we compile with no backend, the list includes "no backend to
* proxy to" for every path rewrite, which is a FALSE report anywhere the composite
* planner runs afterwards and resolves those same rewrites against a real backend
* (`buildCompositeRegistration` logs its own, accurate, list).
*/
export function compileProjectRoutingFields(
routingConfig: RoutingConfig | null | undefined,
onSkipped?: (note: string) => void,
): CompiledRoutingFields {
if (!routingConfig) return {};
const compiled = compileVercelRouting(routingConfig);
if (onSkipped) for (const note of compiled.skipped) onSkipped(note);
return {
...(compiled.proxyLocations.length ? { proxyLocations: compiled.proxyLocations } : {}),
...(compiled.redirects.length ? { redirects: compiled.redirects } : {}),
...(compiled.headerRules.length ? { headerRules: compiled.headerRules } : {}),
...(compiled.cleanUrls ? { cleanUrls: true } : {}),
...(compiled.trailingSlash === undefined ? {} : { trailingSlash: compiled.trailingSlash }),
};
}
+2 -7
View File
@@ -18,7 +18,7 @@
*/
import { env, runtimeTarget, localDashboardUrl } from "../config/env";
import { repos, db, schema, eq } from "@repo/db";
import { repos } from "@repo/db";
/**
* Dashboard same-origin proxy mount. Fixed contract with the dashboard route at
@@ -65,12 +65,7 @@ async function locateSelfAppProjectId(): Promise<string | null> {
const p = await repos.project.findBySlugInOrg(org, SELF_APP_SLUG);
if (p && p.appTemplateId === SELF_APP_SLUG) return p.id;
}
const [admin] = await db
.select({ id: schema.user.id })
.from(schema.user)
.where(eq(schema.user.autoProvisioned, false))
.orderBy(schema.user.createdAt)
.limit(1);
const admin = await repos.user.findFoundingAdmin();
if (admin) {
const p = await repos.project.findBySlugInOrg(`org_${admin.id}`, SELF_APP_SLUG);
if (p && p.appTemplateId === SELF_APP_SLUG) return p.id;
+8 -12
View File
@@ -37,9 +37,8 @@
* operations 5min — keeps a hung CDN from wedging the API.
*
* 6. **Operator escape hatch.** Every throw mentions the env var
* (`OPENSHIP_RELEASE_DIST_PATH` or `MAIL_WEBMAIL_SOURCE_DIR`)
* the operator can point at a local directory to bypass the
* download entirely.
* (`OPENSHIP_RELEASE_DIST_PATH`) the operator can point at a local
* directory to bypass the download entirely.
*
* Layout produced inside cacheDir:
*
@@ -85,7 +84,7 @@ export interface FetchAndExtractReleaseInput {
shaUrl?: string;
/** OR a pinned inline sha256 hex (64 chars) for the external tarball. */
sha256?: string;
/** Error-message escape-hatch hint (defaults derived from the asset name). */
/** Error-message escape-hatch hint (defaults to `OPENSHIP_RELEASE_DIST_PATH`). */
envOverride?: string;
}
@@ -97,14 +96,11 @@ export interface FetchAndExtractReleaseResult {
}
/**
* Map an asset filename to the env-override the operator can use to
* bypass the download. Surfaced in every error message so a stuck
* download isn't a dead end.
* The env-override an operator can point at a local dist to bypass the download.
* Surfaced in every error message so a stuck download isn't a dead end; callers
* with their own escape hatch pass `envOverride` instead.
*/
function envOverrideFor(asset: string): string {
if (asset.startsWith("openship-email-")) return "MAIL_WEBMAIL_SOURCE_DIR";
return "OPENSHIP_RELEASE_DIST_PATH";
}
const DEFAULT_ENV_OVERRIDE = "OPENSHIP_RELEASE_DIST_PATH";
export class ReleaseDownloadError extends Error {
readonly code = "RELEASE_DOWNLOAD_FAILED" as const;
@@ -128,7 +124,7 @@ export async function fetchAndExtractRelease(
): Promise<FetchAndExtractReleaseResult> {
const { tag, cacheDir } = input;
const external = Boolean(input.assetUrl);
const envOverride = input.envOverride ?? (input.asset ? envOverrideFor(input.asset) : "OPENSHIP_RELEASE_DIST_PATH");
const envOverride = input.envOverride ?? DEFAULT_ENV_OVERRIDE;
const targetDir = resolve(cacheDir, tag);
+4 -4
View File
@@ -1,9 +1,9 @@
/**
* ONE resolver for a prebuilt release/dist directory, generalizing the two
* near-identical copies that used to live in migration/openship-dist.ts and
* webmail-project.service.ts. A release-source project (or the openship-instance
* / webmail apps) deploys the directory this returns as `localPath` with no
* build.
* near-identical copies that used to live in migration/openship-dist.ts and the
* retired webmail dist shipper. A release-source project (or the
* openship-instance app) deploys the directory this returns as `localPath` with
* no build.
*
* Three-slot resolution (unchanged from the originals):
* 1. env override → point at a local dist (Docker/CI/air-gapped)
+25
View File
@@ -33,6 +33,13 @@ export function isConnectionLoss(err: unknown): boolean {
/timed out|timeout|etimedout|econnreset|econnrefused|ehostunreach|enetunreach/.test(msg) ||
msg.includes("channel open failure") ||
msg.includes("open failed") ||
// Deliberately here and NOT in the adapters' retryable set, because the two lists
// answer different questions. "Did we reach the host?" — no, so a 503 naming the
// target beats a 500 blaming the request. "Should we run the whole callback again?" —
// no: an HTTP client says this for any far end that vanished mid-request, including a
// permission-denied Docker socket, and those callbacks install mail and ensure the
// edge, so replaying one turns a non-retryable cause into a repeated mutation.
msg.includes("socket hang up") ||
msg.includes("connection lost") ||
msg.includes("not connected") ||
msg.includes("connection closed before ready")
@@ -43,3 +50,21 @@ export function isConnectionLoss(err: unknown): boolean {
export function isAbsent(err: unknown): boolean {
return isRuntimeNotFoundError(err);
}
/**
* True when the runtime REFUSED a state change because the resource is already in
* that state — Docker answers 304 to stopping a stopped container or starting a
* running one.
*
* Belongs with the other two classifiers because it is the fourth answer a state
* change can give, and it is an idempotent SUCCESS: a pause that stops four of a
* project's five containers and finds the fifth already down has done its job.
* Callers that swallowed every error to get this behaviour also swallowed
* "unreachable", which is how a failed pause reported success.
*/
export function isAlreadyInState(err: unknown): boolean {
if (err && typeof err === "object" && (err as { statusCode?: number }).statusCode === 304) {
return true;
}
return /already (stopped|started|running|paused|in progress)/i.test(safeErrorMessage(err));
}
+112 -87
View File
@@ -33,7 +33,12 @@ import type {
} from "@repo/adapters";
import { safeErrorMessage, sanitizeProxySettings, type RoutingConfig } from "@repo/core";
import { platform } from "./controller-helpers";
import { resolveDeploymentRuntime } from "./deployment-runtime";
import {
disposePlatform,
resolveDeploymentPlatform,
type DeploymentMeta,
type ResolvedDeploymentPlatform,
} from "./deployment-runtime";
import {
reapplyCloudProjectRoute,
removeCloudProjectRoute,
@@ -85,6 +90,10 @@ export interface RouteRegister {
proxyLocations?: RouteProxyLocation[];
redirects?: RouteRedirect[];
headerRules?: RouteHeaderRule[];
/** vercel.json `cleanUrls` / `trailingSlash`. Honoured for a `staticRoot` route only
* (registerRoute enforces that) — a proxied app's framework owns its URL shape. */
cleanUrls?: boolean;
trailingSlash?: boolean;
/**
* Canonical redirect to another host instead of serving (see
* RouteConfig.redirectHost). Carried on the LIVE path too, so turning a
@@ -129,96 +138,112 @@ export async function reconcileProjectRoutes(
}
// Self-hosted: the deployment's own routing provider, resolved once.
const routing =
opts.routing ??
(opts.deployment ? (await resolveDeploymentRuntime(opts.deployment)).routing : null);
if (!routing) {
// No deployment routing to resolve (e.g. the active deployment was already
// destroyed, clearing activeDeploymentId). We can't safely REGISTER to an
// unknown host, but a stray vhost from a prior deploy lives on the local
// orchestrator, so run REMOVES there — restoring the opportunistic teardown
// the pre-consolidation code did unconditionally (otherwise the vhost is
// orphaned → stale 502). On remote-server deploys the route isn't local, so
// this is a harmless no-op.
if (removes.length > 0) {
// Teardown runs against the API's OWN routing context. For a single-box
// install that IS the edge; for a containerized API or a remote/takeover'd
// target it isn't, so the stray vhost may actually live on another host and
// these removes are a no-op there. Non-fatal either way, but log it so an
// orphaned vhost that survives isn't mistaken for a completed teardown.
console.warn(
`[route-apply] no deployment routing resolved — running ${removes.length} route removal(s) ` +
`against the API's own edge context; a remote/takeover'd edge may retain the vhost until redeploy`,
);
const local = platform().routing;
for (const r of removes) {
await local
.removeRoute(r.hostname)
.catch((err) =>
console.warn(`[route-apply] fallback removeRoute ${r.hostname} failed (non-fatal): ${safeErrorMessage(err)}`),
);
//
// `resolved` is held so its transport can be released in the `finally` below:
// resolving a platform for a REMOTE server eagerly binds a Docker-over-SSH
// loopback bridge, and this path only ever wanted `.routing` — so it was
// binding a listener per route apply and never closing it.
let resolved: ResolvedDeploymentPlatform | null = null;
if (!opts.routing && opts.deployment) {
resolved = await resolveDeploymentPlatform((opts.deployment.meta ?? {}) as DeploymentMeta, {
organizationId: opts.deployment.organizationId,
});
}
const routing = opts.routing ?? resolved?.platform.routing ?? null;
try {
if (!routing) {
// No deployment routing to resolve (e.g. the active deployment was already
// destroyed, clearing activeDeploymentId). We can't safely REGISTER to an
// unknown host, but a stray vhost from a prior deploy lives on the local
// orchestrator, so run REMOVES there — restoring the opportunistic teardown
// the pre-consolidation code did unconditionally (otherwise the vhost is
// orphaned → stale 502). On remote-server deploys the route isn't local, so
// this is a harmless no-op.
if (removes.length > 0) {
// Teardown runs against the API's OWN routing context. For a single-box
// install that IS the edge; for a containerized API or a remote/takeover'd
// target it isn't, so the stray vhost may actually live on another host and
// these removes are a no-op there. Non-fatal either way, but log it so an
// orphaned vhost that survives isn't mistaken for a completed teardown.
console.warn(
`[route-apply] no deployment routing resolved — running ${removes.length} route removal(s) ` +
`against the API's own edge context; a remote/takeover'd edge may retain the vhost until redeploy`,
);
const local = platform().routing;
for (const r of removes) {
await local
.removeRoute(r.hostname)
.catch((err) =>
console.warn(`[route-apply] fallback removeRoute ${r.hostname} failed (non-fatal): ${safeErrorMessage(err)}`),
);
}
}
if (registers.length > 0) {
console.warn(
`[route-apply] no deployment routing resolved — ${registers.length} route(s) not applied (redeploy to re-sync)`,
);
}
return;
}
if (registers.length > 0) {
console.warn(
`[route-apply] no deployment routing resolved — ${registers.length} route(s) not applied (redeploy to re-sync)`,
);
const webhookHost = project.webhookDomain?.trim().toLowerCase() || null;
// Sanitized here, not trusted from the row: the API validates on write, but a
// value could also have been seeded from a repo config or an older schema, and
// this string is interpolated into generated nginx config.
const proxy = sanitizeProxySettings(project.routingConfig?.proxy);
for (const r of removes) {
await routing
.removeRoute(r.hostname)
.catch((err) =>
console.warn(`[route-apply] removeRoute ${r.hostname} failed (non-fatal): ${safeErrorMessage(err)}`),
);
}
return;
}
const webhookHost = project.webhookDomain?.trim().toLowerCase() || null;
// Sanitized here, not trusted from the row: the API validates on write, but a
// value could also have been seeded from a repo config or an older schema, and
// this string is interpolated into generated nginx config.
const proxy = sanitizeProxySettings(project.routingConfig?.proxy);
for (const r of removes) {
await routing
.removeRoute(r.hostname)
.catch((err) =>
console.warn(`[route-apply] removeRoute ${r.hostname} failed (non-fatal): ${safeErrorMessage(err)}`),
);
}
for (const r of registers) {
// A route serves `/` from ONE of two things: a host directory (static, files
// on disk) or an upstream. Neither → nothing to serve.
if (!r.staticRoot && !r.targetUrl) {
console.warn(
`[route-apply] no upstream or static root resolved for ${r.hostname} — route not applied (redeploy to re-sync)`,
);
continue;
for (const r of registers) {
// A route serves `/` from ONE of two things: a host directory (static, files
// on disk) or an upstream. Neither → nothing to serve.
if (!r.staticRoot && !r.targetUrl) {
console.warn(
`[route-apply] no upstream or static root resolved for ${r.hostname} — route not applied (redeploy to re-sync)`,
);
continue;
}
const isWebhook = r.webhook ?? (!!webhookHost && r.hostname.toLowerCase() === webhookHost);
await routing
.registerRoute({
domain: r.hostname,
tls: true,
// A custom domain's TLS is ours to terminate, so the edge must keep a :443
// listener up for it even before its cert exists — otherwise HTTPS for it
// falls through to the edge's 443 catch-all, which answers with a
// domain-less placeholder cert and the branded not-found page, i.e. the
// domain reads as unconfigured rather than pending (#308).
// A free *.opsh.io host is fronted by Cloud's edge; not ours.
terminatesTlsLocally: r.isCustomDomain,
// staticRoot wins when present: it is the more specific instruction, and a
// caller that resolved a doc root has already decided this domain serves
// files. registerRoute keys off which one is set.
...(r.staticRoot ? { staticRoot: r.staticRoot } : { targetUrl: r.targetUrl! }),
// Project-wide tunables, applied on the LIVE path too so raising an upload
// limit takes effect on save rather than waiting for a redeploy — the same
// treatment a domain/port edit already gets.
...(proxy ? { proxy } : {}),
...(isWebhook ? { webhookProxy: webhookProxyTarget } : {}),
...(r.proxyLocations?.length ? { proxyLocations: r.proxyLocations } : {}),
...(r.redirects?.length ? { redirects: r.redirects } : {}),
...(r.headerRules?.length ? { headerRules: r.headerRules } : {}),
...(r.cleanUrls ? { cleanUrls: true } : {}),
...(r.trailingSlash === undefined ? {} : { trailingSlash: r.trailingSlash }),
...(r.redirectHost ? { redirectHost: r.redirectHost } : {}),
})
.catch((err) =>
console.warn(`[route-apply] registerRoute ${r.hostname} failed (non-fatal): ${safeErrorMessage(err)}`),
);
}
const isWebhook = r.webhook ?? (!!webhookHost && r.hostname.toLowerCase() === webhookHost);
await routing
.registerRoute({
domain: r.hostname,
tls: true,
// A custom domain's TLS is ours to terminate, so the edge must keep a :443
// listener up for it even before its cert exists — otherwise HTTPS for it
// falls through to the edge's 443 catch-all, which answers with a
// domain-less placeholder cert and the branded not-found page, i.e. the
// domain reads as unconfigured rather than pending (#308).
// A free *.opsh.io host is fronted by Cloud's edge; not ours.
terminatesTlsLocally: r.isCustomDomain,
// staticRoot wins when present: it is the more specific instruction, and a
// caller that resolved a doc root has already decided this domain serves
// files. registerRoute keys off which one is set.
...(r.staticRoot ? { staticRoot: r.staticRoot } : { targetUrl: r.targetUrl! }),
// Project-wide tunables, applied on the LIVE path too so raising an upload
// limit takes effect on save rather than waiting for a redeploy — the same
// treatment a domain/port edit already gets.
...(proxy ? { proxy } : {}),
...(isWebhook ? { webhookProxy: webhookProxyTarget } : {}),
...(r.proxyLocations?.length ? { proxyLocations: r.proxyLocations } : {}),
...(r.redirects?.length ? { redirects: r.redirects } : {}),
...(r.headerRules?.length ? { headerRules: r.headerRules } : {}),
...(r.redirectHost ? { redirectHost: r.redirectHost } : {}),
})
.catch((err) =>
console.warn(`[route-apply] registerRoute ${r.hostname} failed (non-fatal): ${safeErrorMessage(err)}`),
);
} finally {
// Only ours to release when we resolved it — a caller-supplied `opts.routing`
// belongs to whoever built it and may still be in use after we return.
disposePlatform(resolved);
}
}
+13 -2
View File
@@ -444,8 +444,19 @@ export function isPublicSpec(spec: RouteSpec): spec is PublicSpec {
* tell-tale `github '*' not found`.
*
* Deliberately limited to read/list. Write/admin GitHub routes (create or
* delete repo, disconnect, instance-token) keep the strict org-wide check —
* owner role or an all-GitHub grant — and MCP exposes no GitHub mutations.
* delete repo, disconnect, instance-token) keep the org-wide check on
* `{github,"*"}`, and MCP exposes no GitHub mutations.
*
* Be precise about what that org-wide check buys, because it is easy to misread as
* a defense it is not: it is strict only for a RESTRICTED principal (a scoped
* token), which needs an all-GitHub grant to pass. For a session-authenticated
* owner/admin/**member** it is satisfied by bare membership —
* `roleAllowsResourceType` (permission.ts) ignores the action level for those
* roles, so `github:admin` is no higher a bar than `github:read`. The per-repo,
* read-vs-write authority for those principals is enforced ONLY by
* `github-access.ts` at the token funnel and at the mutating service helpers.
* Reading this comment as "the route layer already gates writes" is what let
* GHSA-hp2g-hw7g-f3vm sit behind a `github:admin` tag.
* Paramless GitHub routes (/home, /status, /repos) also stay on the org-wide
* path: `filterToolsForPrincipal` already hides them from a scoped token via
* `perm.wildcard`, and their handlers narrow results through
+4 -2
View File
@@ -27,8 +27,10 @@ function isLoopbackAddress(host: string): boolean {
* socket (DooD), exactly like the isLocal row — never SSH.
*
* Scope: server-host only (DEPLOY_MODE docker|bare, not CLOUD_MODE). The SaaS
* never treats a tenant's server row as the control plane, and the desktop app
* targets its own machine through an explicit isLocal "This Machine" row.
* never treats a tenant's server row as the control plane, and the desktop app has
* no server row for this rule to apply to — `ensureLocalServer` returns null off
* server-host, so the ONE isLocal row exists only there; desktop reaches its own
* machine through the `local` deploy target instead (resolveTargetPlatform).
*
* A row that is a deliberate SSH TUNNEL — a jump/bastion host set, or a loopback
* host on a non-default port (the classic `ssh -L` local-forward-to-elsewhere
@@ -16,7 +16,8 @@
import { randomBytes } from "node:crypto";
import { env } from "../config/env";
import type { ShellSession } from "@repo/adapters";
import type { RuntimeAdapter, ShellSession } from "@repo/adapters";
import { disposeRuntime } from "./deployment-runtime";
import type { TerminalExitReason } from "@repo/db";
import type { RequestContext } from "./request-context";
@@ -89,6 +90,18 @@ export interface ActiveServiceSession {
serviceId: string;
startedAt: number;
shell: ShellSession;
/**
* The runtime whose transport this session's SHELL rides on, released by
* `unregisterServiceSession`.
*
* Owned by the SESSION and not by the WebSocket connection, because a session
* outlives its connection: a parked session keeps this shell across a drop, and
* the connection that later RESUMES it resolves a different runtime it never
* uses. Hanging the lifetime off the connection therefore released the wrong
* handle and stranded the one actually carrying the shell — on a remote server
* that is a Docker-over-SSH loopback bridge per terminal opened.
*/
runtime: RuntimeAdapter | null;
onTimeout: (sessionId: string, reason: TerminalExitReason) => void;
lastActivityAt: number;
idleTimer: ReturnType<typeof setTimeout>;
@@ -117,6 +130,8 @@ export function registerServiceSession(args: {
userId: string;
serviceId: string;
shell: ShellSession;
/** The runtime that opened `shell`; disposed when the session ends. */
runtime?: RuntimeAdapter | null;
onTimeout: (sessionId: string, reason: TerminalExitReason) => void;
}): ActiveServiceSession {
const now = Date.now();
@@ -129,6 +144,7 @@ export function registerServiceSession(args: {
serviceId: args.serviceId,
startedAt: now,
shell: args.shell,
runtime: args.runtime ?? null,
onTimeout: args.onTimeout,
lastActivityAt: now,
closed: false,
@@ -257,6 +273,10 @@ export function unregisterServiceSession(sessionId: string): boolean {
clearTimeout(session.idleTimer);
clearTimeout(session.hardCapTimer);
// The one place every ending path converges (user close, remote exit, idle and
// hard-cap timeouts all land here), so the shell's transport is released once.
disposeRuntime(session.runtime);
session.runtime = null;
session.scrollback = [];
session.scrollbackSize = 0;
sessions.delete(sessionId);
+85 -16
View File
@@ -18,14 +18,28 @@
* - deny common system paths even when nested under an env-configured or
* caller-supplied root (the built-in DEFAULT_ROOTS are exempt — one of
* them, /etc/openship/ssh-keys, deliberately sits under /etc)
* - the operator's own `~/.ssh` is exempt from the denylist too, so a
* root-run API (homedir() === /root) can still use ~/.ssh/openship even
* though /root/.ssh is on the denylist — see the note at the check.
*
* Tests live in test/lib/ssh-key-path.test.ts.
*/
import { accessSync, constants, statSync } from "node:fs";
import { homedir } from "node:os";
import { isAbsolute, resolve, sep } from "node:path";
import { env } from "../config/env";
/** Hard-coded denylist of system paths regardless of root config. */
/**
* Hard-coded denylist of system paths regardless of root config.
*
* Deliberately a STATIC SUPERSET of every host family, not derived from the resolved
* host profile. A denylist that narrowed with detection would grant an attacker the
* paths of whichever family we mis-detected — and this list guards the API's own
* filesystem, which is one box, not the many targets the profile describes. So the
* Debian data dir (`/var/lib/postgresql`) and the RHEL one (`/var/lib/pgsql`) are
* both listed unconditionally.
*/
const SYSTEM_DENY = [
"/etc",
"/proc",
@@ -33,6 +47,7 @@ const SYSTEM_DENY = [
"/dev",
"/boot",
"/var/lib/postgresql",
"/var/lib/pgsql",
"/var/lib/docker",
"/root/.ssh",
];
@@ -73,20 +88,32 @@ export function resolveSafeSshKeyPath(
}
const resolved = resolve(trimmed);
const isUnder = (root: string) =>
resolved === root || resolved.startsWith(root + sep);
// SYSTEM_DENY is deliberately broad (`/etc`), and one of the built-in
// DEFAULT_ROOTS sits inside it (`/etc/openship/ssh-keys`). Those roots are
// hardcoded here — not operator- or caller-supplied — so a path under one is
// an explicit carve-out and skips the denylist. Env-configured and caller
// supplied roots deliberately do NOT get this exemption, so an `extraRoots`
// of `/` (or a `$HOME` of `/`) still can't unlock `/etc/shadow`.
const underDefaultRoot = DEFAULT_ROOTS.map((r) => resolve(r)).some(
(root) => resolved === root || resolved.startsWith(root + sep),
);
// SYSTEM_DENY is deliberately broad (`/etc`, `/root/.ssh`). Two kinds of path
// are exempt from it:
// 1. the built-in DEFAULT_ROOTS — hardcoded, not caller-supplied, so one
// like `/etc/openship/ssh-keys` is an explicit carve-out.
// 2. the operator's own `~/.ssh` — i.e. `<extraRoot>/.ssh`. extraRoots is
// always the home of the user the API runs as (never attacker input),
// and `~/.ssh` is where SSH keys naturally live. Without this a root-run
// API (homedir() === "/root") could never use `~/.ssh/openship`, because
// `/root/.ssh` is on the denylist — so the convenience buildSshConfig and
// hydrate-server advertise would silently fail only when running as root.
// The carve-out is narrow (only `<home>/.ssh`), so an `extraRoots` of `/`
// still can't unlock `/etc/shadow`. Tradeoff: on a root-run install this lets
// a servers row read `/root/.ssh/id_rsa` — the same exposure a non-root
// install already has for `/home/<op>/.ssh/*` (never on the denylist), so it
// makes the policy consistent rather than adding a new class of exposure.
const underDefaultRoot = DEFAULT_ROOTS.map((r) => resolve(r)).some(isUnder);
const underOwnSshDir = (opts.extraRoots ?? [])
.map((home) => resolve(home, ".ssh"))
.some(isUnder);
if (!underDefaultRoot) {
if (!underDefaultRoot && !underOwnSshDir) {
for (const denied of SYSTEM_DENY) {
if (resolved === denied || resolved.startsWith(denied + sep)) {
if (isUnder(denied)) {
throw new Error(
`sshKeyPath is inside a protected system directory (${denied}): ${trimmed}`,
);
@@ -104,10 +131,7 @@ export function resolveSafeSshKeyPath(
...(opts.extraRoots ?? []),
].map((r) => resolve(r));
const insideRoot = allowedRoots.some(
(root) => resolved === root || resolved.startsWith(root + sep),
);
if (!insideRoot) {
if (!allowedRoots.some(isUnder)) {
throw new Error(
`sshKeyPath must sit under one of: ${allowedRoots.join(", ")}`,
);
@@ -115,3 +139,48 @@ export function resolveSafeSshKeyPath(
return resolved;
}
/**
* The convenience root every runtime caller adds: the operator's home, so
* `$HOME/.ssh/openship` works with no explicit configuration. Defined once so
* `buildSshConfig` and the test-connection precheck can't disagree about which
* paths are acceptable — a mismatch would mean "Test Connection" passing on a key
* a real deploy then refuses (or vice versa).
*/
export function operatorSshKeyRoots(): string[] {
return [homedir()];
}
/**
* Why `raw` can't be used as a key path, or null when it can.
*
* `buildSshConfig` deliberately collapses every key-path failure into a null
* return, which callers render as "Invalid auth configuration" — accurate but
* useless when the real cause is a typo, a key the operator never copied to this
* host, or a path outside the allowlist. Diagnostic surfaces (test-connection)
* call this first so they can say which. Messages are operator-safe: they only
* name the path the caller just supplied and the roots policy.
*/
export function sshKeyPathProblem(raw: string): string | null {
let resolved: string;
try {
resolved = resolveSafeSshKeyPath(raw, { extraRoots: operatorSshKeyRoots() });
} catch (err) {
return err instanceof Error ? err.message : `sshKeyPath is not usable: ${raw}`;
}
try {
accessSync(resolved, constants.R_OK);
} catch {
return `sshKeyPath is not readable by Openship — no such file, or no permission: ${resolved}`;
}
// A directory passes the readability probe but blows up in readFileSync later,
// so name it here — picking `~/.ssh` instead of `~/.ssh/id_ed25519` is an easy
// slip in a file dialog.
if (!statSync(resolved, { throwIfNoEntry: false })?.isFile()) {
return `sshKeyPath is not a file: ${resolved}`;
}
return null;
}
@@ -0,0 +1,100 @@
import { describe, expect, it, vi, beforeEach } from "vitest";
/**
* buildSshConfig is the single choke point every SSH connection funnels
* through. These tests pin the paste/upload key path added for a remote
* instance whose key lives in the browser, not on the API host:
*
* - stored `enc1:` material decrypts to config.privateKey (never a file read),
* - pasted content wins over a host path,
* - the ephemeral test-connection path (RAW, unencrypted material) passes
* through verbatim, and
* - the host-path fallback still reads the file when no content is present.
*/
// Keep ssh-manager's heavy transitive graph out of the way: this suite only
// exercises buildSshConfig, which touches none of these at call time.
vi.mock("@repo/db", () => ({ repos: { server: { get: vi.fn() } } }));
vi.mock("@repo/adapters", async () => ({
...(await import("../../../../packages/adapters/src/system/errors")),
HOST_STATE_DIR: "/root/.openship",
createHostExecutor: vi.fn(),
hostChannelHealth: vi.fn(),
probeTcp: vi.fn(),
}));
vi.mock("./box-org", () => ({ isLocalHostRow: vi.fn() }));
// The path allowlist is tested on its own (ssh-key-path); here we only need to
// know WHETHER the path branch runs, so make it an identity + assert on the read.
vi.mock("./ssh-key-path", () => ({
resolveSafeSshKeyPath: vi.fn((p: string) => p),
operatorSshKeyRoots: vi.fn(() => []),
}));
// Control the file read without touching the real filesystem; leave the rest of
// node:fs real so nothing else that imports it breaks.
const readFileSync = vi.fn((..._args: unknown[]) => "FILE-ON-HOST-KEY");
vi.mock("node:fs", async (importOriginal) => ({
...(await importOriginal<typeof import("node:fs")>()),
readFileSync: (...args: unknown[]) => readFileSync(...args),
}));
import { buildSshConfig } from "./ssh-manager";
// REAL encryption — the whole point is that a stored enc1: value round-trips.
import { encryptSecretField } from "./credential-encryption";
const base = { sshHost: "10.0.0.1", sshAuthMethod: "key" as const };
beforeEach(() => {
readFileSync.mockClear();
});
describe("buildSshConfig — pasted/uploaded key material", () => {
it("decrypts stored material into privateKey and never reads a file", async () => {
const config = await buildSshConfig({
...base,
sshPrivateKey: encryptSecretField("STORED-PRIVATE-KEY"),
});
expect(config?.privateKey).toBe("STORED-PRIVATE-KEY");
expect(readFileSync).not.toHaveBeenCalled();
});
it("prefers pasted content over a host path", async () => {
const config = await buildSshConfig({
...base,
sshPrivateKey: encryptSecretField("STORED-PRIVATE-KEY"),
sshKeyPath: "/root/.ssh/id_ed25519",
});
expect(config?.privateKey).toBe("STORED-PRIVATE-KEY");
expect(readFileSync).not.toHaveBeenCalled();
});
it("passes a raw (unencrypted) ephemeral key through verbatim", async () => {
// The test-connection path sends the key the operator just pasted, before
// anything is persisted — decryptSecretField returns a non-enc1: value as-is.
const raw = "-----BEGIN OPENSSH PRIVATE KEY-----\nabc\n-----END OPENSSH PRIVATE KEY-----";
const config = await buildSshConfig({ ...base, sshPrivateKey: raw });
expect(config?.privateKey).toBe(raw);
expect(readFileSync).not.toHaveBeenCalled();
});
it("applies the passphrase alongside pasted content", async () => {
const config = await buildSshConfig({
...base,
sshPrivateKey: encryptSecretField("STORED-PRIVATE-KEY"),
sshKeyPassphrase: encryptSecretField("s3cret-pass"),
});
expect(config?.privateKey).toBe("STORED-PRIVATE-KEY");
expect(config?.privateKeyPassphrase).toBe("s3cret-pass");
});
it("still reads the host file when only a path is set", async () => {
const config = await buildSshConfig({ ...base, sshKeyPath: "/root/.ssh/id_ed25519" });
expect(config?.privateKey).toBe("FILE-ON-HOST-KEY");
expect(readFileSync).toHaveBeenCalledTimes(1);
});
it("returns null for key auth with neither content nor path", async () => {
expect(await buildSshConfig({ ...base })).toBeNull();
});
});
@@ -23,7 +23,17 @@ vi.mock("@repo/db", () => ({
repos: { server: { get: vi.fn(async (id: string) => h.rows[id]) } },
}));
vi.mock("@repo/adapters", () => ({
vi.mock("@repo/adapters", async () => ({
// The REAL error/predicate module, not stand-ins. The gate's contract is that
// `isHostChannelUnavailableError` recognises what it throws and that
// `isRetryableRemoteConnectionError` agrees with the executors about which
// failures are transport failures — local copies would keep passing after either
// stopped being true. Imported by path because that module is dependency-free;
// pulling in the package entry point would defeat mocking it.
...(await import("../../../../packages/adapters/src/system/errors")),
// Read at module scope by openship-server-store (imported transitively), so the
// mock has to carry it or nothing under test loads. Its value is irrelevant here.
HOST_STATE_DIR: "/root/.openship",
createHostExecutor: vi.fn(() => {
h.created += 1;
return {
@@ -33,6 +43,8 @@ vi.mock("@repo/adapters", () => ({
}),
};
}),
hostChannelHealth: vi.fn(async () => ({ ok: true, code: "ok" })),
probeTcp: vi.fn(async () => true),
}));
// isLocalHostRow decides "is this row THIS box". Keyed off the fixture flag so the
@@ -47,6 +59,9 @@ import { sshManager } from "./ssh-manager";
* assert the cache state that the leak was a symptom of. */
const pool = () => (sshManager as unknown as { servers: Map<string, unknown> }).servers;
/** Mirrors the module-private pool key for the shared host channel. */
const HOST_KEY = "__openship_host__";
beforeEach(() => {
h.created = 0;
h.disposed = 0;
@@ -60,6 +75,9 @@ beforeEach(() => {
true,
);
}
// Also clears the circuit breaker: probes here deliberately fail, and a cooldown
// left over from the previous case would answer the next one before it ran.
sshManager.invalidate();
h.created = 0;
h.disposed = 0;
vi.clearAllMocks();
@@ -140,9 +158,37 @@ describe("local server rows share the one host channel", () => {
expect(a).toBe(b);
});
it("a local row is cached so probeReachable can short-circuit", async () => {
it("caches the local row WITHOUT claiming it is reachable", async () => {
// #490: acquiring a local row seeds a borrow marker over an executor that has
// not dialed anything (ssh2 connects lazily). Treating that entry as proof made
// probeReachable answer Online for a host whose SSH port was firewalled — the
// one place an operator would have looked to discover the channel was dead.
await sshManager.acquire("local-1");
expect(pool().has("local-1")).toBe(true);
expect((pool().get("local-1") as { proven?: boolean }).proven).toBeFalsy();
});
it("answers probeReachable from the host channel's health, not from the cache", async () => {
const { hostChannelHealth } = await import("@repo/adapters");
await sshManager.acquire("local-1");
vi.mocked(hostChannelHealth).mockResolvedValueOnce({
ok: false,
code: "unreachable",
hint: "blocked",
});
expect(await sshManager.probeReachable("local-1")).toBe(false);
vi.mocked(hostChannelHealth).mockResolvedValueOnce({ ok: true, code: "ok" });
expect(await sshManager.probeReachable("local-1")).toBe(true);
});
it("does not report a deliberately disabled host channel as offline", async () => {
// `--no-host-control` switches off host-OS operations by choice; the row still
// deploys through the mounted Docker socket, so Offline would be a lie.
const { hostChannelHealth } = await import("@repo/adapters");
vi.mocked(hostChannelHealth).mockResolvedValueOnce({ ok: false, code: "disabled" });
expect(await sshManager.probeReachable("local-1")).toBe(true);
});
it("dropping a BORROWED row never closes the shared connection", async () => {
@@ -210,3 +256,309 @@ describe("host channel survives while a borrower is live (the owner/borrower lif
vi.useRealTimers();
});
});
describe("diagnoseReachability carries the reason, not just the verdict", () => {
it("names the address ops DIAL, not the row's display sshHost", async () => {
// The whole point: the row says 127.0.0.1 (display-only), so a UI keyed off it
// told operators to check a port that is supposed to be closed, on the wrong
// machine. The diagnosis names host.docker.internal and the real fix (#490).
const { hostChannelHealth } = await import("@repo/adapters");
vi.mocked(hostChannelHealth).mockResolvedValueOnce({
ok: false,
code: "unreachable",
host: "host.docker.internal",
port: 22,
target: "root@host.docker.internal:22",
rule: "sudo ufw allow from 172.18.0.0/16 to any port 22 proto tcp",
hint: "Nothing answered at root@host.docker.internal:22 …",
});
const d = await sshManager.diagnoseReachability("local-1");
expect(d).toMatchObject({
reachable: false,
code: "host_channel_blocked",
target: "root@host.docker.internal:22",
port: 22,
rule: "sudo ufw allow from 172.18.0.0/16 to any port 22 proto tcp",
});
expect(d.target).not.toContain("127.0.0.1");
});
it("still explains itself once the breaker has opened", async () => {
// The UI asks WHY at exactly the moment a host op just failed — i.e. when the
// breaker is open. A bare "cooldown" there would drop the diagnosis on the floor
// precisely when it's needed.
const { hostChannelHealth } = await import("@repo/adapters");
const blocked = {
ok: false,
code: "unreachable" as const,
target: "root@host.docker.internal:22",
rule: "sudo ufw allow from 172.18.0.0/16 to any port 22 proto tcp",
hint: "blocked",
};
vi.mocked(hostChannelHealth).mockResolvedValueOnce(blocked).mockResolvedValueOnce(blocked);
// Two failures trip the breaker (FAIL_THRESHOLD = 2).
await sshManager.diagnoseReachability("local-1");
await sshManager.diagnoseReachability("local-1");
vi.mocked(hostChannelHealth).mockClear();
const d = await sshManager.diagnoseReachability("local-1");
// No further probe was paid…
expect(hostChannelHealth).not.toHaveBeenCalled();
// …and the answer still says what to do about it.
expect(d).toMatchObject({
reachable: false,
code: "host_channel_blocked",
target: "root@host.docker.internal:22",
rule: "sudo ufw allow from 172.18.0.0/16 to any port 22 proto tcp",
});
});
it("distinguishes a deliberate opt-out from a broken channel", async () => {
const { hostChannelHealth } = await import("@repo/adapters");
vi.mocked(hostChannelHealth).mockResolvedValueOnce({ ok: false, code: "disabled" });
expect(await sshManager.diagnoseReachability("local-1")).toMatchObject({
reachable: true,
code: "host_control_disabled",
});
});
/**
* #509. `host_channel_blocked` is ONE verdict over several states, and the remedy
* differs per state — a dial that was dropped wants a firewall rule, a channel that was
* never provisioned wants `openship up`. So the state has to travel, and the fields
* that mean "we contacted this" must not be filled in for a channel nothing contacted:
* the dashboard reads a present `target` as evidence, and reported that nothing answered
* on an address nothing had ever dialed.
*/
it("carries the host-channel state, with no endpoint or rule invented for it", async () => {
const { hostChannelHealth } = await import("@repo/adapters");
vi.mocked(hostChannelHealth).mockResolvedValueOnce({
ok: false,
code: "not_configured",
hint: "…no host channel…",
});
const d = await sshManager.diagnoseReachability("local-1");
expect(d).toMatchObject({ reachable: false, code: "host_channel_blocked", channel: "not_configured" });
expect(d.target).toBeUndefined();
expect(d.host).toBeUndefined();
expect(d.port).toBeUndefined();
expect(d.rule).toBeUndefined();
});
it("distinguishes an unreachable channel from an unprovisioned one by that state", async () => {
const { hostChannelHealth } = await import("@repo/adapters");
vi.mocked(hostChannelHealth).mockResolvedValueOnce({
ok: false,
code: "unreachable",
target: "root@host.docker.internal:22",
hint: "dropped",
});
expect(await sshManager.diagnoseReachability("local-1")).toMatchObject({
code: "host_channel_blocked",
channel: "unreachable",
target: "root@host.docker.internal:22",
});
});
it("does not invent a host-channel story for a remote row", async () => {
const { hostChannelHealth, probeTcp } = await import("@repo/adapters");
h.rows["remote-1"] = { id: "remote-1", isLocal: false, sshHost: "203.0.113.9" };
vi.mocked(probeTcp).mockResolvedValueOnce(false);
const d = await sshManager.diagnoseReachability("remote-1");
expect(hostChannelHealth).not.toHaveBeenCalled();
expect(d).toMatchObject({ reachable: false, code: "unreachable", host: "203.0.113.9", port: 22 });
expect(d.rule).toBeUndefined();
});
});
/**
* #490's third clause: "mark host control as unavailable rather than letting
* operations fail one at a time".
*
* `createHostExecutor()` only CONSTRUCTS — ssh2 dials lazily — so a channel the host
* firewall drops hands back a healthy-looking executor, and every op that takes it
* pays its own handshake timeout before failing with a message that names neither the
* channel nor the fix. What the gate must NOT become is a cache of that verdict: a
* remembered failure that outlives the firewall rule is its own bug report.
*/
describe("a channel observed to fail is refused, not re-dialed forever", () => {
/** What a firewalled channel actually throws: ssh2 waiting out a handshake whose
* SYN was dropped. Classified by the REAL isRetryableRemoteConnectionError. */
const handshakeTimeout = () => new Error("Timed out while waiting for handshake");
const BLOCKED = {
ok: false as const,
code: "unreachable" as const,
host: "host.docker.internal",
port: 22,
target: "root@host.docker.internal:22",
cause: "timeout" as const,
// Shaped like the real thing: hint is prose, rule is a command, and they are
// separate fields. So the assertions below pin that the gate COMPOSES them into the
// error — without that, the rule reaches the dashboard banner and never the deploy
// log, which is where an operator is standing when a host op dies.
rule: "sudo ufw allow from 172.18.0.0/16 to any port 22 proto tcp",
hint: "root@host.docker.internal:22 — no response, not even a refusal (ETIMEDOUT).",
};
/** Leave the manager in the state a firewalled host leaves it in: the channel
* built, one host op attempted, one transport failure observed. */
async function observeTransportFailure() {
await expect(
sshManager.withHostExecutor(async () => {
throw handshakeTimeout();
}),
).rejects.toThrow(/Timed out/);
}
it("costs nothing while the channel is healthy", async () => {
// The gate must be free in the normal case, or every host op on every box pays a
// TCP probe to protect the one box whose firewall is shut.
const { hostChannelHealth } = await import("@repo/adapters");
await sshManager.withHostExecutor(async () => true);
await sshManager.withHostExecutor(async () => true);
expect(hostChannelHealth).not.toHaveBeenCalled();
});
it("drops the pooled channel on a transport failure, so the gate can run at all", async () => {
// Load-bearing: `acquireHostChannel`'s cache fast-path returns BEFORE the gate,
// so a channel left cached is handed to every later op and the gate never gets a
// say — exactly the failure-at-a-time behaviour it exists to end.
await sshManager.withHostExecutor(async () => true);
expect(pool().has(HOST_KEY)).toBe(true);
await observeTransportFailure();
expect(pool().has(HOST_KEY)).toBe(false);
});
it("refuses the next acquire with the remedy instead of another handshake timeout", async () => {
const { hostChannelHealth, isHostChannelUnavailableError } = await import("@repo/adapters");
await observeTransportFailure();
vi.mocked(hostChannelHealth).mockResolvedValueOnce(BLOCKED);
const err = await sshManager.withHostExecutor(async () => "never runs").catch((e) => e);
expect(isHostChannelUnavailableError(err)).toBe(true);
expect(err.code).toBe("unreachable");
// Both halves travel with the refusal — this is the only place an operator sees them.
expect(err.message).toContain(BLOCKED.hint);
expect(err.message).toContain("sudo ufw allow from 172.18.0.0/16 to any port 22 proto tcp");
// And no second executor was built to be timed out against.
expect(h.created).toBe(1);
});
it("re-probes every time — a channel that came back is used on the very next call", async () => {
// The version of this that shipped as a TTL cache kept answering "unavailable"
// for a host whose firewall had already been fixed, which is a worse lie than
// the one it replaced.
const { hostChannelHealth } = await import("@repo/adapters");
await observeTransportFailure();
vi.mocked(hostChannelHealth).mockResolvedValueOnce({ ok: true, code: "ok" });
await expect(sshManager.withHostExecutor(async () => "ok")).resolves.toBe("ok");
expect(hostChannelHealth).toHaveBeenCalledTimes(1);
expect(h.created).toBe(2);
// …and the suspicion is cleared, not merely bypassed by the fresh cache entry:
// reclaim the channel and the next acquire pays no probe.
(sshManager as unknown as { dropServer: (k: string, f?: boolean) => void }).dropServer(
HOST_KEY,
true,
);
await expect(sshManager.withHostExecutor(async () => "ok")).resolves.toBe("ok");
expect(hostChannelHealth).toHaveBeenCalledTimes(1);
});
it("does not blame the channel for a command that reached the host and failed", async () => {
// A non-zero exit or a missing file dialed successfully. Marking the channel
// suspect here would refuse host control over an `rm` of something already gone.
const { hostChannelHealth } = await import("@repo/adapters");
await sshManager.withHostExecutor(async () => true);
await expect(
sshManager.withHostExecutor(async () => {
throw new Error("exit 2: /etc/nope: No such file or directory");
}),
).rejects.toThrow(/No such file/);
expect(pool().has(HOST_KEY)).toBe(true);
await expect(sshManager.withHostExecutor(async () => "still fine")).resolves.toBe("still fine");
expect(hostChannelHealth).not.toHaveBeenCalled();
expect(h.created).toBe(1);
});
it("does not convert a deliberate opt-out into an unreachable channel", async () => {
// `disabled` / `not_configured` are decided by env and a local file read, and
// `createHostExecutor` raises them itself — fast, with their own remedy. The gate
// pre-empts only the state that costs a timeout to discover.
await observeTransportFailure();
const { hostChannelHealth } = await import("@repo/adapters");
vi.mocked(hostChannelHealth).mockResolvedValueOnce({ ok: false, code: "disabled" });
await expect(sshManager.withHostExecutor(async () => "ok")).resolves.toBe("ok");
});
it("believes an operator who just fixed the firewall and hit retry", async () => {
// `invalidate()` is that retry (and a settings save). A remembered failure must
// not answer for a channel nobody has re-tested.
const { hostChannelHealth } = await import("@repo/adapters");
await observeTransportFailure();
sshManager.invalidate();
await expect(sshManager.withHostExecutor(async () => "ok")).resolves.toBe("ok");
expect(hostChannelHealth).not.toHaveBeenCalled();
});
it("still builds exactly one executor when 8 callers hit the armed gate at once", async () => {
// The gate awaits, and an await between the pending-check and the pending-set is
// the window that orphaned 8,000+ sshd sessions (#291). It has to live INSIDE the
// deduped promise, and a slow probe is what proves it does.
const { hostChannelHealth } = await import("@repo/adapters");
await observeTransportFailure();
vi.mocked(hostChannelHealth).mockImplementationOnce(async () => {
await new Promise((r) => setTimeout(r, 5));
return { ok: true, code: "ok" };
});
const results = await Promise.all(
Array.from({ length: 8 }, () => sshManager.withHostExecutor(async (exec) => exec)),
);
expect(hostChannelHealth).toHaveBeenCalledTimes(1);
expect(h.created).toBe(2); // the pre-gate one, plus one for all 8 callers
expect(new Set(results).size).toBe(1);
});
it("arms from a LOCAL ROW's failure too — the row IS the channel", async () => {
// A deploy to "This Server" runs host steps through the row id, not the channel
// key. If only `withHostExecutor` armed the gate, the most common way to discover
// a blocked channel would be the one way that never recorded it.
const { hostChannelHealth, isHostChannelUnavailableError } = await import("@repo/adapters");
await sshManager.acquire("local-1");
await expect(
sshManager.withExecutor("local-1", async () => {
throw handshakeTimeout();
}),
).rejects.toThrow(/Timed out/);
vi.mocked(hostChannelHealth).mockResolvedValueOnce(BLOCKED);
const err = await sshManager.withHostExecutor(async () => "never runs").catch((e) => e);
expect(isHostChannelUnavailableError(err)).toBe(true);
expect(err.message).toContain("sudo ufw allow from");
});
it("keeps the breaker and the gate reading the same evidence", async () => {
// A transport failure that survives a retry on a FRESH connection used to escape
// both: the retry threw from inside the outer catch, past `recordFailure`. So a
// firewalled channel failed indefinitely, two handshake timeouts at a time.
await sshManager.acquire("local-1");
await expect(
sshManager.withExecutor("local-1", async () => {
throw handshakeTimeout();
}),
).rejects.toThrow(/Timed out/);
const health = (sshManager as unknown as { health: Map<string, { fails: number }> }).health;
expect(health.get("local-1")?.fails).toBe(1);
});
});
+360 -54
View File
@@ -25,7 +25,6 @@
* - Timers use unref() so they don't prevent graceful shutdown.
*/
import { homedir } from "node:os";
import { readFileSync } from "node:fs";
import { execFile } from "node:child_process";
import { promisify } from "node:util";
@@ -33,17 +32,21 @@ import { repos } from "@repo/db";
import {
createExecutor,
createHostExecutor,
hostChannelHealth,
HostChannelUnavailableError,
invalidateEnvironment,
isRetryableRemoteConnectionError,
probeTcp,
runReliable,
type CommandExecutor,
type HostChannelCode,
type HostChannelHealth,
type SshConfig,
} from "@repo/adapters";
import { formatDuration, systemDebug } from "@/lib/system-debug";
import { decryptSecretField } from "@/lib/credential-encryption";
import { resolveSafeSshKeyPath } from "@/lib/ssh-key-path";
import { operatorSshKeyRoots, resolveSafeSshKeyPath } from "@/lib/ssh-key-path";
import { isLocalHostRow } from "@/lib/box-org";
import { OPENSHIP_DIR } from "@/lib/openship-server-store";
import { safeErrorMessage } from "@repo/core";
const execFileAsync = promisify(execFile);
@@ -104,6 +107,9 @@ export interface SshSettingsInput {
sshAuthMethod?: string | null;
sshPassword?: string | null;
sshKeyPath?: string | null;
/** Pasted/uploaded key material. Stored `enc1:`-encrypted on a DB row; sent
* raw by the ephemeral test-connection path. Takes precedence over sshKeyPath. */
sshPrivateKey?: string | null;
sshKeyPassphrase?: string | null;
sshJumpHost?: string | null;
sshArgs?: string | null;
@@ -134,23 +140,35 @@ export async function buildSshConfig(
// Stored encrypted on insert; decrypted only here at the moment we
// hand it to the ssh2 client.
config.password = decryptSecretField(settings.sshPassword);
} else if (settings.sshAuthMethod === "key" && settings.sshKeyPath) {
// Centralised allowlist + traversal check — see lib/ssh-key-path.ts.
// homedir() is the operator's home, used as the default convenient
// root so `~/.ssh/openship` works without explicit env config.
let keyPath: string;
try {
keyPath = resolveSafeSshKeyPath(settings.sshKeyPath, {
extraRoots: [homedir()],
});
} catch {
return null;
}
} else if (
settings.sshAuthMethod === "key" &&
(settings.sshPrivateKey || settings.sshKeyPath)
) {
if (settings.sshPrivateKey) {
// Pasted/uploaded material — the key lives in the DB, not on this host's
// filesystem, so there is no path to allowlist. decryptSecretField returns
// the stored `enc1:` value as plaintext, and passes a raw (unprefixed) key
// through verbatim — which is exactly what the ephemeral test-connection
// path sends before anything is persisted.
config.privateKey = decryptSecretField(settings.sshPrivateKey);
} else {
// Path on the API host. Centralised allowlist + traversal check — see
// lib/ssh-key-path.ts. operatorSshKeyRoots() is the operator's home, used
// as the default convenient root so `~/.ssh/openship` works without config.
let keyPath: string;
try {
keyPath = resolveSafeSshKeyPath(settings.sshKeyPath!, {
extraRoots: operatorSshKeyRoots(),
});
} catch {
return null;
}
try {
config.privateKey = readFileSync(keyPath, "utf-8");
} catch {
return null;
try {
config.privateKey = readFileSync(keyPath, "utf-8");
} catch {
return null;
}
}
if (settings.sshKeyPassphrase) {
config.privateKeyPassphrase = decryptSecretField(settings.sshKeyPassphrase);
@@ -197,6 +215,93 @@ const DEFAULTS = {
const FAIL_THRESHOLD = 2;
const COOLDOWN_MS = 30_000;
export type ReachabilityCode =
| "ok"
/** Breaker open — reported unreachable without attempting anything. */
| "cooldown"
/** The TCP probe to the server's SSH port didn't complete. */
| "unreachable"
/** A remote row with no host to dial. */
| "no_address"
/**
* THIS box, containerized, and the container→host SSH channel is dead —
* almost always a host firewall dropping bridge→host traffic (#490). Deploys
* still work (Docker socket); host-OS operations don't.
*/
| "host_channel_blocked"
/** THIS box with `--no-host-control`: reachable, host-OS operations off. */
| "host_control_disabled"
/** No such row, or the manager is shut down. */
| "unknown";
export interface ReachabilityDiagnosis {
reachable: boolean;
code: ReachabilityCode;
/**
* `user@host:port` — the address ops DIAL, not the row's display sshHost.
*
* EVIDENCE, not intent: a consumer is entitled to read a present `target` as "we
* contacted this and it didn't answer". So it stays absent for a channel that was
* never provisioned — filling it with the endpoint such a channel WOULD use is how
* the dashboard came to report nothing answering on an address nothing had dialed
* (#509). A would-be endpoint belongs in a differently-named field.
*/
target?: string;
host?: string;
port?: number;
/**
* Which host-channel state produced `host_channel_blocked`.
*
* Refines the code rather than splitting it: to the server list and the breaker
* this is one state ("can't drive its host"), but the REMEDY differs — a channel
* that was dialed and dropped wants a firewall rule, one that was never
* provisioned wants `openship up` and no rule at all. Absent on a remote row,
* which has no channel.
*/
channel?: HostChannelCode;
/** Operator-facing remedy, when we have one. */
hint?: string;
/**
* A ready-to-paste firewall rule, when a packet filter is the likely cause. Only
* ever set for a dial that was DROPPED — see `firewallShaped` in @repo/core.
*/
rule?: string;
}
/**
* Did this failure come from the TRANSPORT rather than the command?
*
* A dropped/refused connection or a wait that never completed says the box (or the
* channel to it) is sick; a non-zero exit or a missing file reached it fine. One
* predicate because both readers — the circuit breaker and the host-channel gate —
* have to agree on the answer, and the copy that drifts is the one that mislabels a
* healthy box as unreachable. Exported for the HTTP layer, which has the same
* question to answer before it offers a connection diagnosis (see server-check).
*/
export function isTransportFailure(err: unknown): boolean {
if (isRetryableRemoteConnectionError(err)) return true;
return /timed out|timeout|ETIMEDOUT/i.test(safeErrorMessage(err));
}
/**
* The endpoint + remedy fields of a host-channel health, omitting the empty ones.
*
* `code` is carried across as `channel` — the one field a consumer can't reconstruct,
* because `host_channel_blocked` collapses every non-ok code into one verdict and the
* remedy doesn't (#509). Nothing is synthesized here: a health with no endpoint
* produces a diagnosis with no endpoint.
*/
function hostChannelFields(h: HostChannelHealth): Partial<ReachabilityDiagnosis> {
return {
channel: h.code,
...(h.target ? { target: h.target } : {}),
...(h.host ? { host: h.host } : {}),
...(h.port ? { port: h.port } : {}),
...(h.hint ? { hint: h.hint } : {}),
...(h.rule ? { rule: h.rule } : {}),
};
}
// ─── Reliable-run (journaled, exactly-once) tuning ─────────────────────────────
/** Max queued journaled ops per server before run() rejects (backpressure). */
@@ -265,6 +370,17 @@ interface ServerConnection {
* executor — the owner and the other borrowers are still using it.
*/
shared?: boolean;
/**
* Something has actually flowed over this executor (see {@link
* SshConnectionManager.recordSuccess}).
*
* Cached ≠ connected: `connect()` and `createHostExecutor()` only CONSTRUCT an
* executor — ssh2 dials lazily on the first exec. So a cache entry alone proves
* nothing, and `probeReachable` treating it as proof was reporting a host whose
* SSH port was firewalled as Online (#490): the local row's borrow marker is
* seeded by `acquireLocalHost` without a single packet sent.
*/
proven?: boolean;
}
/**
@@ -297,6 +413,16 @@ export class SshConnectionManager {
private journalReady = new WeakSet<CommandExecutor>();
/** Circuit-breaker state per server (consecutive fails + cooldown deadline). */
private health = new Map<string, { fails: number; unhealthyUntil: number }>();
/**
* The last host-channel probe result (instance-wide — there is one host). Kept
* only to explain a breaker cooldown, never as the reachability answer itself.
*/
private lastHostHealth: HostChannelHealth | null = null;
/**
* A host-channel failure we have actually observed. Never an answer on its own —
* {@link assertHostChannelUsable} re-probes before refusing anything.
*/
private hostChannelSuspect = false;
private destroyed = false;
private readonly opts: Required<SshManagerOptions>;
@@ -418,14 +544,24 @@ export class SshConnectionManager {
}
this.servers.set(serverId, conn);
this.touchIdleTimer(serverId);
this.recordSuccess(serverId);
}
/**
* The pooled HOST channel — one connection to the machine Openship runs on,
* reused by every caller and reclaimed by the same idle timer as a server.
*
* Public, and that is the point: this is the ONLY way to obtain this box's
* executor. Prefer {@link withHostExecutor} — it also reports the outcome back to
* the channel gate. Take the handle directly only when it must OUTLIVE the call
* (a deploy holds its executor across steps, so it cannot be scoped), which is the
* one case `withHostExecutor` cannot serve.
*
* It used to be private, so a caller holding no server row had nothing to ask and
* called `createHostExecutor()` itself — landing outside the cache, outside the
* concurrent-acquire dedup and outside {@link assertHostChannelUsable}. That is the
* unpooled idiom from #291, and it was reachable on the one door that has no row.
*/
private async acquireHostChannel(): Promise<CommandExecutor> {
async acquireHostChannel(): Promise<CommandExecutor> {
const cached = this.servers.get(HOST_CHANNEL_KEY);
if (cached) {
this.touchIdleTimer(HOST_CHANNEL_KEY);
@@ -439,7 +575,15 @@ export class SshConnectionManager {
// createHostExecutor throws (deliberately) when the API is containerized with
// no host channel provisioned. Let that propagate: callers must NOT silently
// treat "cannot reach the host at all" as success.
const promise = (async () => createHostExecutor())();
//
// The gate runs INSIDE the deduped promise, not before it: it awaits, and an
// await between the pending-check and the pending-set is a window for two
// callers to each build their own SshExecutor — the leak that reached 8,000+
// orphaned sshd sessions (#291).
const promise = (async () => {
await this.assertHostChannelUsable();
return createHostExecutor();
})();
this.connecting.set(HOST_CHANNEL_KEY, promise);
try {
const exec = await promise;
@@ -451,6 +595,70 @@ export class SshConnectionManager {
}
}
/**
* Refuse a host channel we have already watched fail, once a fresh probe agrees.
*
* `createHostExecutor()` only CONSTRUCTS — ssh2 dials on first use — so a channel
* the host firewall drops hands back a perfectly good-looking executor, and every
* operation that takes it pays its own SSH handshake timeout before failing with a
* bare timeout that names neither the channel nor the fix. Replacing that with one
* refusal carrying the remedy is #490's "mark host control as unavailable rather
* than letting operations fail one at a time".
*
* Gated on an OBSERVED failure and re-probed every time, so it is not a cache with
* a TTL (an earlier version was, and it kept answering "unavailable" for a channel
* that had come back): a working channel is used on the very next call, and a dead
* one costs a 2.5s TCP probe instead of a handshake timeout.
*/
private async assertHostChannelUsable(): Promise<void> {
if (!this.hostChannelSuspect) return;
const health = await hostChannelHealth();
this.lastHostHealth = health;
if (health.ok) {
this.hostChannelSuspect = false;
return;
}
// Every other code is decided by env or a local file read — createHostExecutor
// raises those itself, fast, with their own remedy. Only `unreachable` is the
// state that costs a timeout to discover, so only it is worth pre-empting.
if (health.code !== "unreachable") return;
// hint and rule are separate fields (prose is wrapped, a command must not be), so
// the one place that flattens them into a message does it here — otherwise the
// firewall rule reaches the dashboard banner and never the deploy log, which is
// exactly where an operator is standing when a host op dies.
const hint =
health.hint ?? `Openship can't reach the host channel at ${health.target ?? "this machine"}.`;
throw new HostChannelUnavailableError(
"unreachable",
health.rule ? `${hint}\n${health.rule}` : hint,
);
}
/**
* Record what a host operation just revealed about the channel.
*
* Dropping the pooled channel on failure is the load-bearing half: the cache
* fast-path in `acquireHostChannel` returns before the gate can run, so a channel
* left cached is re-dialed by every later op — precisely the failure-at-a-time
* behaviour the gate exists to end. A dead SSH executor is worth nothing anyway.
* (`dropServer` still declines to yank a channel a live terminal or stream is
* holding; that one keeps its executor until release, which is the right trade.)
*/
private noteHostChannel(ok: boolean): void {
if (ok) {
this.hostChannelSuspect = false;
return;
}
this.hostChannelSuspect = true;
this.dropServer(HOST_CHANNEL_KEY);
}
/** This key IS the host channel: the channel itself, or a local row borrowing it.
* Their outcomes are the same outcome, so they feed the same gate. */
private isHostChannelKey(serverId: string): boolean {
return serverId === HOST_CHANNEL_KEY || this.localHostRows.has(serverId);
}
/**
* Resolve a LOCAL server row to the shared host channel, keeping a borrow marker
* under the row id so `isConnected`/`probeReachable` short-circuit and the row
@@ -462,7 +670,6 @@ export class SshConnectionManager {
this.localHostRows.add(serverId);
const exec = await this.acquireHostChannel();
this.cacheSharedMarker(serverId, exec);
this.recordSuccess(serverId);
debugSsh(`acquire:local-host server=${serverId} (${formatDuration(startedAt)})`);
return exec;
}
@@ -495,7 +702,15 @@ export class SshConnectionManager {
async withHostExecutor<T>(fn: (executor: CommandExecutor) => Promise<T>): Promise<T> {
const exec = await this.acquireHostChannel();
try {
return await fn(exec);
const result = await fn(exec);
this.noteHostChannel(true);
return result;
} catch (err) {
// Only a TRANSPORT failure says anything about the channel. A command that
// exits non-zero, or a file that isn't there, reached the host perfectly well
// and must not mark it suspect.
if (isTransportFailure(err)) this.noteHostChannel(false);
throw err;
} finally {
// Extend the idle window from LAST USE, not from acquisition, or a long op
// can have the connection dropped from under its own tail.
@@ -519,18 +734,26 @@ export class SshConnectionManager {
const msg = safeErrorMessage(err);
debugSsh(`withExecutor:retry-after-connection-error server=${serverId} ${msg}`);
this.dropServer(serverId);
const freshExecutor = await this.acquire(serverId);
const result = await fn(freshExecutor);
this.recordSuccess(serverId);
debugSsh(`withExecutor:retry-done server=${serverId} (${formatDuration(startedAt)})`);
return result;
try {
const freshExecutor = await this.acquire(serverId);
const result = await fn(freshExecutor);
this.recordSuccess(serverId);
debugSsh(`withExecutor:retry-done server=${serverId} (${formatDuration(startedAt)})`);
return result;
} catch (retryErr) {
// A transport failure that survives a FRESH connection is evidence, not a
// blip — it has to reach the breaker and the host-channel gate. Rethrowing
// from inside the outer catch skipped both, so a local row whose host
// channel was firewalled failed this way indefinitely, paying two handshake
// timeouts per op and teaching us nothing (#490).
if (isTransportFailure(retryErr)) this.recordFailure(serverId);
throw retryErr;
}
}
const msg = safeErrorMessage(err);
// Connection errors and command timeouts count toward the breaker — a
// sick/unreachable box shouldn't be re-hit every poll tick.
if (isRetryableRemoteConnectionError(err) || /timed out|timeout|ETIMEDOUT/i.test(msg)) {
this.recordFailure(serverId);
}
if (isTransportFailure(err)) this.recordFailure(serverId);
debugSsh(`withExecutor:failed server=${serverId} (${formatDuration(startedAt)}) ${msg}`);
throw err;
}
@@ -612,7 +835,8 @@ export class SshConnectionManager {
* now" — delete/reconcile use it to fast-fail an unreachable host in ~2.5s
* instead of paying the 15-20s SSH connect timeout per resource.
*
* - live cached connection → reachable (don't disturb it).
* - PROVEN cached connection → reachable (don't disturb it). A merely-cached
* one isn't evidence — see ServerConnection.proven.
* - breaker in cooldown → unreachable, WITHOUT any connection attempt
* (the "no avoidable connection" fast path).
* - otherwise → a bounded TCP probe to the SSH port; the
@@ -621,15 +845,35 @@ export class SshConnectionManager {
* cleanup execs bypass `withExecutor`.
*
* Reuses `repos.server.get` — the same config source `connect()` uses — so
* there is no second notion of server connectivity.
* there is no second notion of server connectivity. The yes/no answer here and
* the reason in {@link diagnoseReachability} come from the same single pass.
*/
async probeReachable(serverId: string, timeoutMs = 2500): Promise<boolean> {
if (this.destroyed) return false;
if (this.servers.has(serverId)) return true;
if (this.cooldownRemaining(serverId) > 0) return false;
return (await this.diagnoseReachability(serverId, timeoutMs)).reachable;
}
/**
* The same check, with the REASON attached.
*
* Exists because "unreachable" alone sent operators after the wrong thing: the
* only address the UI had was the row's display `sshHost`, so a local row whose
* container→host SSH channel was firewalled off rendered "Can't reach 127.0.0.1"
* and told them to run `nc -zv 127.0.0.1 22` — a port that is *supposed* to be
* closed, on the wrong machine entirely (#490). `target` here is the address ops
* actually dial, and `hint`/`rule` carry the real remedy.
*/
async diagnoseReachability(serverId: string, timeoutMs = 2500): Promise<ReachabilityDiagnosis> {
if (this.destroyed) return { reachable: false, code: "unknown" };
if (this.servers.get(serverId)?.proven) return { reachable: true, code: "ok" };
// Read the cooldown BEFORE the row so a remote box in cooldown still pays no
// probe; the row read that follows is a local DB hit, and it's what decides
// whether a dial was ever the right question — a local row is answered by the
// host-channel health, whose cooldown path still has to carry the REASON.
const inCooldown = this.cooldownRemaining(serverId) > 0;
const server = await repos.server.get(serverId).catch(() => undefined);
if (!server) return false;
if (!server) return { reachable: false, code: "unknown" };
// An isLocal row is THIS box (the auto-registered "This Server"). Its ssh*
// fields are display-only — self-server.ts writes `127.0.0.1`/SERVER_IP purely
@@ -645,27 +889,47 @@ export class SshConnectionManager {
// So answer with the channel ops actually use: the host SSH bridge when we're
// containerized (host.docker.internal), else the same machine we're running on.
if (await isLocalHostRow(server)) {
const hostSsh = process.env.OPENSHIP_HOST_SSH_HOST?.trim();
if (!hostSsh) {
this.recordSuccess(serverId);
return true;
// The breaker still short-circuits — a blocked channel costs a full probe
// timeout per call — but a cooldown must not erase the REASON, because the UI
// asks precisely when a host operation has just failed, i.e. exactly when the
// breaker is open. A remembered failure is what opened it, so it still explains.
if (inCooldown) {
const last = this.lastHostHealth;
return last && !last.ok && last.code !== "disabled"
? { reachable: false, code: "host_channel_blocked", ...hostChannelFields(last) }
: { reachable: false, code: "cooldown" };
}
const hostOk = await probeTcp(
hostSsh,
Number(process.env.OPENSHIP_HOST_SSH_PORT || 22),
timeoutMs,
);
const health = await hostChannelHealth(timeoutMs);
this.lastHostHealth = health;
// `disabled` is an operator choice (`--no-host-control`), not a fault: host-OS
// operations are off, but the row still deploys through the mounted Docker
// socket, so it must not read as Offline. Every other non-ok code IS a fault.
const hostOk = health.ok || health.code === "disabled";
if (hostOk) this.recordSuccess(serverId);
else this.recordFailure(serverId);
return hostOk;
return {
reachable: hostOk,
code: health.ok
? "ok"
: health.code === "disabled"
? "host_control_disabled"
: "host_channel_blocked",
...hostChannelFields(health),
};
}
if (!server.sshHost) return false;
if (!server.sshHost) return { reachable: false, code: "no_address" };
const ok = await probeTcp(server.sshHost, server.sshPort ?? 22, timeoutMs);
const host = server.sshHost;
const port = server.sshPort ?? 22;
const endpoint = { target: `${server.sshUser ?? "root"}@${host}:${port}`, host, port };
if (inCooldown) return { reachable: false, code: "cooldown", ...endpoint };
const ok = await probeTcp(host, port, timeoutMs);
if (ok) this.recordSuccess(serverId);
else this.recordFailure(serverId);
return ok;
return { reachable: ok, code: ok ? "ok" : "unreachable", ...endpoint };
}
/**
@@ -679,18 +943,46 @@ export class SshConnectionManager {
// apply even to a retained connection — force the drop.
if (serverId) {
debugSsh(`invalidate server=${serverId}`);
this.forgetProfile(serverId);
this.dropServer(serverId, true);
// Config changed / explicit reset → give the breaker a fresh start.
// Config changed / explicit reset → give the breaker a fresh start, and the
// host-channel gate too: an operator who just fixed their firewall and hit
// retry has told us more than a remembered failure can.
this.health.delete(serverId);
this.lastHostHealth = null;
this.hostChannelSuspect = false;
} else {
debugSsh("invalidate:all");
for (const id of [...this.servers.keys()]) {
this.forgetProfile(id);
this.dropServer(id, true);
}
this.health.clear();
this.lastHostHealth = null;
this.hostChannelSuspect = false;
}
}
/**
* Drop the measured host profile along with the connection.
*
* The profile is cached per executor OBJECT, so dropping the connection already
* loses it — eventually, whenever the object is collected. This makes it immediate,
* and only on the explicit path, which is the one that carries the information: the
* operator changed this server's settings or deleted the row. Same reasoning as
* `health.delete` two lines up — a remembered answer about a box they just told us
* they changed is worth less than a fresh look.
*
* Must run BEFORE `dropServer`, which is what removes the entry we read the executor
* from. A borrowed entry (a local row pointing at the shared host channel) forgets the
* channel's profile too, and that is right rather than merely tolerable: it is the same
* machine, so a change to the row is a change to the box.
*/
private forgetProfile(serverId: string): void {
const conn = this.servers.get(serverId);
if (conn) invalidateEnvironment(conn.executor);
}
/**
* Mark a connection as actively in use by a long-lived operation
* (streaming, Docker tunnels, etc.).
@@ -834,8 +1126,9 @@ export class SshConnectionManager {
* layers its pool + circuit breaker around it via the acquire/hooks.
*/
private executeJournaledOp(serverId: string, op: QueuedOp): Promise<RunResult> {
// No `baseDir`: it used to pin every server in the fleet to `/root/.openship`, which
// is unwritable over SFTP on a non-root login. `runReliable` asks each host instead.
return runReliable(() => this.acquire(serverId), op.opId, op.command, {
baseDir: OPENSHIP_DIR,
timeoutMs: op.opts.timeoutMs ?? DEFAULT_RUN_TIMEOUT_MS,
waitSecs: op.opts.waitSecs,
envPrefix: op.opts.envPrefix,
@@ -859,14 +1152,27 @@ export class SshConnectionManager {
return Math.max(0, h.unhealthyUntil - Date.now());
}
/** One success clears the breaker entirely. */
/**
* One success clears the breaker entirely.
*
* Also the single place a cached connection earns `proven`: every caller here has
* observed the box actually respond (a completed op, or a TCP probe). Callers that
* merely built an executor deliberately do NOT call this.
*/
private recordSuccess(serverId: string): void {
if (this.health.has(serverId)) this.health.delete(serverId);
const conn = this.servers.get(serverId);
if (conn) conn.proven = true;
if (this.isHostChannelKey(serverId)) this.noteHostChannel(true);
}
/** Count a connect/command failure; trip the breaker at the threshold and
* drop any cached (now-suspect) connection so the cooldown actually bites. */
private recordFailure(serverId: string): void {
// A local row IS the host channel, so its failures are the channel's. Recorded
// here rather than at each call site: this is where `probeReachable`'s fresh
// health and a local-row `withExecutor` already converge.
if (this.isHostChannelKey(serverId)) this.noteHostChannel(false);
const h = this.health.get(serverId) ?? { fails: 0, unhealthyUntil: 0 };
h.fails += 1;
if (h.fails >= FAIL_THRESHOLD) {
@@ -0,0 +1,164 @@
import { describe, it, expect, beforeEach, vi } from "vitest";
/**
* The wizard's structured failure has to carry the CAUSE, not just the code.
*
* `provisionSelfAppEdge` forwarded only `reason` — a code callers branch on — so
* `SelfEdgeInfraResult.detail` (the resolver naming the distro it refuses, a takeover
* saying the stop was refused unelevated) reached the operator through the live log
* and nowhere else. A reattaching client, a headless run, or anything reading the
* finished session got "migrate_failed" with no cause: exactly the cause-less failure
* #490/#509 were about.
*/
const h = vi.hoisted(() => ({
infra: { ok: false, reason: "unsupported_host", detail: "" } as {
ok: boolean;
reason?: string;
detail?: string;
},
foreignProxy: { blocked: false, owner: "" } as { blocked: boolean; owner?: string },
reapply: vi.fn(async () => {}),
project: { id: "p1" } as unknown,
}));
vi.mock("./self-edge", () => ({ ensureSelfEdgeInfra: async () => h.infra }));
vi.mock("./self-services", () => ({ linkSelfAppServices: vi.fn(async () => {}) }));
vi.mock("./index", () => ({ registerStartupHook: vi.fn() }));
vi.mock("../../modules/deployments/build.service", () => ({ createQueuedDeployment: vi.fn() }));
vi.mock("../../modules/deployments/deployment-lifecycle", () => ({ onSuccess: vi.fn() }));
vi.mock("../../modules/domains/project-route.service", () => ({
reapplyProjectLiveRoutes: h.reapply,
}));
vi.mock("../domain-ssl", () => ({
manageDomainSsl: vi.fn(async () => ({ verified: false, reason: "challenge failed" })),
tlsIssuedElsewhere: () => null,
describeTlsIssuedElsewhere: () => "",
}));
vi.mock("../public-url", () => ({ refreshSelfAppPublicUrl: vi.fn(async () => {}) }));
vi.mock("@repo/adapters", () => ({
BareRuntime: class {},
foreignProxyOnEdge: async () => h.foreignProxy,
}));
vi.mock("../ssh-manager", () => ({
sshManager: { withHostExecutor: async (fn: (e: unknown) => unknown) => fn({}) },
}));
vi.mock("@repo/db", () => ({
repos: {
project: { findById: async () => h.project },
domain: { findByHostname: async () => null },
},
db: {},
schema: {},
eq: vi.fn(),
}));
import { provisionSelfAppEdge, type SelfEdgeStepProgress } from "./self-deploy";
import {
createSetupSession,
updateComponentProgress,
subscribeSetupSession,
} from "../../modules/system/setup-session";
/** Records every step event so the returned payload and the wizard's stream can be
* compared — the bug was one of them carrying the diagnosis and the other not. */
function recorder() {
const steps: { step: string; status: string; detail?: string }[] = [];
const progress: SelfEdgeStepProgress = {
onLog: () => {},
onStep: (step, status, detail) => steps.push({ step, status, detail }),
backoffs: [],
};
return { steps, progress, failed: () => steps.find((s) => s.status === "failed") };
}
beforeEach(() => {
vi.clearAllMocks();
h.infra = { ok: false, reason: "unsupported_host", detail: "" };
h.foreignProxy = { blocked: false };
h.reapply.mockResolvedValue(undefined);
});
const SUSE = "Openship has no package manager for openSUSE (distro family suse).";
describe("provisionSelfAppEdge failure detail", () => {
it("returns the infra layer's diagnosis alongside the code, and only when there is one", async () => {
h.infra = { ok: false, reason: "unsupported_host", detail: SUSE };
expect(await provisionSelfAppEdge("p1", "app.example.com", 3001, {})).toEqual({
verified: false,
reason: "unsupported_host",
detail: SUSE,
});
// The other half of the rule: a layer that diagnosed nothing must not have a
// `detail` invented for it — an empty key reads as "the cause is blank".
h.infra = { ok: false, reason: "not_linux" };
const bare = await provisionSelfAppEdge("p1", "app.example.com", 3001, {});
expect(bare).toEqual({ verified: false, reason: "not_linux" });
expect("detail" in bare).toBe(false);
});
it("puts the same diagnosis on the failed step the wizard renders", async () => {
h.infra = { ok: false, reason: "migrate_failed", detail: "port 80 still bound after stop" };
const r = recorder();
const res = await provisionSelfAppEdge("p1", "app.example.com", 3001, r.progress);
expect(r.failed()).toEqual({
step: "edge",
status: "failed",
detail: "port 80 still bound after stop",
});
expect(res.detail).toBe(r.failed()?.detail);
});
it("reports WHICH proxy still owns 80/443, in the words the log used", async () => {
h.infra = { ok: true };
h.foreignProxy = { blocked: true, owner: "nginx" };
const r = recorder();
const res = await provisionSelfAppEdge("p1", "app.example.com", 3001, r.progress);
expect(res.reason).toBe("edge_not_owned");
expect(res.detail).toContain("nginx");
expect(r.failed()).toMatchObject({ step: "route", detail: res.detail });
});
it("carries the routing pipeline's own error message", async () => {
h.infra = { ok: true };
h.reapply.mockRejectedValue(new Error("nginx -t: invalid vhost for app.example.com"));
const r = recorder();
const res = await provisionSelfAppEdge("p1", "app.example.com", 3001, r.progress);
expect(res.reason).toBe("route_failed");
expect(res.detail).toContain("invalid vhost");
expect(r.failed()).toMatchObject({ step: "route", detail: res.detail });
});
it("carries the last real cert failure, not just cert_pending", async () => {
h.infra = { ok: true };
const r = recorder();
const res = await provisionSelfAppEdge("p1", "app.example.com", 3001, r.progress);
expect(res.reason).toBe("cert_pending");
expect(res.detail).toBe("challenge failed");
expect(r.failed()).toMatchObject({ step: "ssl", detail: "challenge failed" });
});
});
/**
* The other end of the plumb: a step's detail has to leave the process. Wired the
* way self-app.controller.ts wires it, because "the cause is in the live log" was
* exactly the gap — a client that reattaches after the failure replays progress,
* not the log lines it missed.
*/
describe("a failed step's detail reaches the wizard's progress stream", () => {
it("rides the SSE progress frame and the replayed component state", async () => {
h.infra = { ok: false, reason: "migrate_failed", detail: "vhost write refused: EACCES" };
const session = createSetupSession([{ name: "edge", label: "Install the edge" }], "self");
await provisionSelfAppEdge("p1", "app.example.com", 3001, {
onStep: (step, status, detail) => updateComponentProgress(session.id, step, status, detail),
});
const frames: { event: string; data: string }[] = [];
subscribeSetupSession(session.id, (event, data) => {
frames.push({ event, data });
return true;
});
const replay = frames.find((f) => f.event === "progress");
expect(replay?.data).toContain("vhost write refused: EACCES");
});
});
+49 -26
View File
@@ -26,7 +26,7 @@
* a "public" signal into them.
*/
import { repos, db, schema, eq, type Project, type Deployment } from "@repo/db";
import { repos, type Project, type Deployment } from "@repo/db";
import { BareRuntime } from "@repo/adapters";
import { safeErrorMessage, UNLIMITED_RESOURCES } from "@repo/core";
import { env } from "../../config/env";
@@ -181,6 +181,8 @@ export interface SelfEdgeStepProgress {
onStep?: (
step: "edge" | "route" | "ssl",
status: "installing" | "installed" | "failed",
/** The cause, when `status` is "failed" — see `SelfEdgeInfraResult.detail`. */
detail?: string,
) => void;
backoffs?: number[];
}
@@ -195,7 +197,7 @@ export interface SelfEdgeStepProgress {
*/
async function foreignProxyBlocksEdge(
log?: (message: string, level?: "info" | "warn" | "error") => void,
): Promise<{ blocked: boolean; owner?: string }> {
): Promise<{ blocked: boolean; owner?: string; detail?: string }> {
try {
const { foreignProxyOnEdge } = await import("@repo/adapters");
const { sshManager } = await import("../ssh-manager");
@@ -206,17 +208,46 @@ async function foreignProxyBlocksEdge(
foreignProxyOnEdge(exec),
);
if (!blocked) return { blocked: false };
log?.(
// Returned as well as logged so the caller's structured failure carries the SAME
// sentence the live log shows — not a second wording of it.
const detail =
`Not issuing TLS: ${owner} still owns ports 80/443, so Openship isn't the reverse proxy yet — ` +
`an ACME challenge would hit it, not us. Re-run setup (or Domains → migrate) to take over.`,
"error",
);
return { blocked: true, owner };
`an ACME challenge would hit it, not us. Re-run setup (or Domains → migrate) to take over.`;
log?.(detail, "error");
return { blocked: true, owner, detail };
} catch {
return { blocked: false };
}
}
export interface SelfAppEdgeResult {
verified: boolean;
expiresAt?: string;
reason?: string;
/**
* The CAUSE behind `reason`, forwarded from whichever layer diagnosed it (see
* `SelfEdgeInfraResult.detail`). `reason` is a code callers branch on, so it can't
* carry the diagnosis; dropping `detail` here left the wizard's structured failure
* with only the code and put the cause exclusively in the live log — which a
* reattaching client, a headless CLI run, or anything reading the finished session
* never sees.
*/
detail?: string;
}
/** One failure shape for every exit: the code, plus the cause on BOTH surfaces —
* the returned payload and the step event the wizard renders — so the two can't
* disagree about how much of the diagnosis they carry. */
function edgeStepFailed(
progress: SelfEdgeStepProgress,
step: "edge" | "route" | "ssl",
reason: string | undefined,
detail?: string,
): SelfAppEdgeResult {
progress.onStep?.(step, "failed", detail);
return { verified: false, reason, ...(detail ? { detail } : {}) };
}
/**
* Custom-domain edge for the self-app: install the toolchain + take over
* 80/443, then hand routing + cert to the NORMAL pipeline (route via
@@ -234,23 +265,22 @@ export async function provisionSelfAppEdge(
// type describes the INFRA install (takeover/migrate) and is forwarded verbatim to
// `ensureSelfEdgeInfra`, which has no business knowing about Cloud's edge.
options?: SelfEdgeOptions & { managedEdgeSyncedByCaller?: boolean },
): Promise<{ verified: boolean; expiresAt?: string; reason?: string }> {
): Promise<SelfAppEdgeResult> {
const log = progress.onLog;
// 1. Toolchain install + optional 80/443 takeover/migrate (no route/cert).
progress.onStep?.("edge", "installing");
const infra = await ensureSelfEdgeInfra({ onLog: log }, options);
if (!infra.ok) {
progress.onStep?.("edge", "failed");
return { verified: false, reason: infra.reason };
return edgeStepFailed(progress, "edge", infra.reason, infra.detail);
}
progress.onStep?.("edge", "installed");
// Hard gate: never touch routing/cert unless OUR OpenResty owns 80/443 (takeover
// skipped / partial / respawned would otherwise 404 the ACME challenge opaquely).
if ((await foreignProxyBlocksEdge(log)).blocked) {
progress.onStep?.("route", "failed");
return { verified: false, reason: "edge_not_owned" };
const foreign = await foreignProxyBlocksEdge(log);
if (foreign.blocked) {
return edgeStepFailed(progress, "route", "edge_not_owned", foreign.detail);
}
// 2. Route hostname → 127.0.0.1:dashPort via the pipeline (owns the vhost +
@@ -258,8 +288,7 @@ export async function provisionSelfAppEdge(
progress.onStep?.("route", "installing");
const project = await repos.project.findById(projectId);
if (!project) {
progress.onStep?.("route", "failed");
return { verified: false, reason: "no_project" };
return edgeStepFailed(progress, "route", "no_project");
}
try {
await reapplyProjectLiveRoutes(project, [], {
@@ -270,9 +299,9 @@ export async function provisionSelfAppEdge(
managedEdgeSyncedByCaller: options?.managedEdgeSyncedByCaller,
});
} catch (err) {
log?.(safeErrorMessage(err), "error");
progress.onStep?.("route", "failed");
return { verified: false, reason: "route_failed" };
const detail = safeErrorMessage(err);
log?.(detail, "error");
return edgeStepFailed(progress, "route", "route_failed", detail);
}
progress.onStep?.("route", "installed");
log?.(`routing ${hostname} → http://127.0.0.1:${dashPort}`);
@@ -321,14 +350,13 @@ export async function provisionSelfAppEdge(
}
if (attempt < backoffs.length) await sleep(backoffs[attempt]);
}
progress.onStep?.("ssl", "failed");
log?.(
lastError
? `Couldn't issue TLS for ${hostname}: ${lastError} — it serves over HTTP and retries on next boot.`
: `could not issue TLS for ${hostname} yet — will retry on next boot (site still serves over HTTP).`,
"warn",
);
return { verified: false, reason: "cert_pending" };
return edgeStepFailed(progress, "ssl", "cert_pending", lastError);
}
/** Locate the self-app project across the cloud-linked / founding-admin org.
@@ -339,12 +367,7 @@ async function findSelfAppProject(): Promise<Project | null> {
const p = await repos.project.findBySlugInOrg(org, APP_SLUG);
if (p && p.appTemplateId === APP_TEMPLATE_ID) return p;
}
const [admin] = await db
.select({ id: schema.user.id })
.from(schema.user)
.where(eq(schema.user.autoProvisioned, false))
.orderBy(schema.user.createdAt)
.limit(1);
const admin = await repos.user.findFoundingAdmin();
if (admin) {
const p = await repos.project.findBySlugInOrg(`org_${admin.id}`, APP_SLUG);
if (p && p.appTemplateId === APP_TEMPLATE_ID) return p;
+106 -15
View File
@@ -1,4 +1,5 @@
import { describe, it, expect, beforeEach, afterEach, vi } from "vitest";
import type { EnvironmentProfile } from "@repo/adapters";
/**
* Halt-and-report contract for the non-interactive self-install edge step.
@@ -8,28 +9,41 @@ import { describe, it, expect, beforeEach, afterEach, vi } from "vitest";
* RESOLVE `{ ok:false, reason:"edge_conflict" }` — not throw, not fall through to a
* bare cert failure downstream. With a clear edge, it installs as normal.
*
* @repo/adapters is mocked so this runs with no real box; process.platform/getuid
* are stubbed to satisfy the Linux+root guard that gates the whole path.
* @repo/adapters is mocked so this runs with no real box; process.platform is stubbed
* and the mocked resolver returns a root Ubuntu profile, which together satisfy the
* Linux + can-elevate + supported-family guard that gates the whole path.
*/
const h = vi.hoisted(() => ({
canProceedClean: false,
sites: [{}, {}] as unknown[],
/** What the resolver reports for this box; per-case overrides on the base fixture. */
host: {} as Partial<EnvironmentProfile>,
ensureFeature: vi.fn(async () => {}),
foreignProxyOnEdge: vi.fn(),
importSites: vi.fn(),
}));
vi.mock("@repo/adapters", () => ({
createExecutor: () => ({}),
SystemManager: class {
ensureFeature = h.ensureFeature;
},
foreignProxyOnEdge: h.foreignProxyOnEdge,
importSites: h.importSites,
runEdgeTakeover: vi.fn(),
}));
vi.mock("@repo/adapters", async () => {
// The real fixture, not a hand-rolled literal: `resolveEnvironment` gates the whole
// path on isRoot/canSudo/supported, and a literal here would keep passing after the
// profile grows a field the product starts gating on. `profileFixture` also recomputes
// the support verdict from the facts, so `distroFamily: "suse"` below carries openSUSE's
// real reason rather than one this test invented.
const fixtures = await import("../../../../../packages/adapters/src/system/environment.fixtures");
return {
createExecutor: () => ({}),
resolveEnvironment: async () => fixtures.profileFixture(h.host),
SystemManager: class {
ensureFeature = h.ensureFeature;
},
foreignProxyOnEdge: h.foreignProxyOnEdge,
importSites: h.importSites,
runEdgeTakeover: h.runEdgeTakeover,
};
});
// The build-only APPLY (build the edge from source onto the local daemon before
// bring-up) has its own unit tests — here it's a no-op so these cases stay about
// the halt-and-report contract, not the deliver pipeline.
@@ -40,13 +54,13 @@ vi.mock("../deliver-managed-image", () => ({
import { ensureSelfEdgeInfra } from "./self-edge";
const origPlatform = Object.getOwnPropertyDescriptor(process, "platform");
const origGetuid = process.getuid;
beforeEach(() => {
vi.clearAllMocks();
Object.defineProperty(process, "platform", { value: "linux", configurable: true });
// stub root so the not-root guard passes
process.getuid = () => 0;
// No `process.getuid` stub: privilege comes from the resolver now, so root-ness is a
// property of `h.host` (the base fixture is a root box).
h.host = {};
h.canProceedClean = false;
h.foreignProxyOnEdge.mockImplementation(async () => {
const occupants = h.canProceedClean ? [] : [{ port: 80, command: "nginx", proxy: "nginx" }];
@@ -65,7 +79,6 @@ beforeEach(() => {
afterEach(() => {
if (origPlatform) Object.defineProperty(process, "platform", origPlatform);
process.getuid = origGetuid;
});
describe("ensureSelfEdgeInfra — halt + report", () => {
@@ -92,3 +105,81 @@ describe("ensureSelfEdgeInfra — halt + report", () => {
expect(h.ensureFeature).toHaveBeenCalledWith("ssl", expect.any(Function));
});
});
describe("ensureSelfEdgeInfra — a failed migrate reports WHY", () => {
/** The takeover's own words for a refused, unelevated stop — the #490/#509 shape. */
const refusal =
"The port 80 is still in use, so the edge can't bind — nothing was installed. " +
"Openship could not run the stop as root on this host, so stopping the existing " +
"proxy was almost certainly refused — retrying as this user will fail the same way.";
it("relays the takeover's classified warnings to the operator log and the result", async () => {
h.runEdgeTakeover.mockResolvedValue({
ok: false,
rolledBack: true,
registered: [],
warnings: [refusal],
});
const logs: { message: string; level?: string }[] = [];
const res = await ensureSelfEdgeInfra(
{ onLog: (message, level) => logs.push({ message, level }) },
{ edgeMigrate: true },
);
expect(res.ok).toBe(false);
expect(res.reason).toBe("migrate_failed");
// The CAUSE, not just the fact: "migrate_failed" alone sends the operator to retry
// an operation that can never succeed as this user.
expect(res.detail).toContain("could not run the stop as root");
expect(logs).toContainEqual({ message: refusal, level: "error" });
});
it("a migrate that SUCCEEDS with warnings still surfaces them (as warnings)", async () => {
h.runEdgeTakeover.mockResolvedValue({
ok: true,
rolledBack: false,
registered: ["a.example.com"],
warnings: ["a.example.com: existing cert unreadable — issuing a fresh certificate"],
});
const logs: { message: string; level?: string }[] = [];
const res = await ensureSelfEdgeInfra(
{ onLog: (message, level) => logs.push({ message, level }) },
{ edgeMigrate: true },
);
expect(res.ok).toBe(true);
expect(res.detail).toBeUndefined();
expect(logs.some((l) => l.level === "warn" && l.message.includes("existing cert unreadable"))).toBe(true);
});
});
describe("ensureSelfEdgeInfra — the host gate reads the resolver", () => {
it("non-root WITH passwordless sudo → installs (the installer elevates)", async () => {
// The regression this pins: a bare `getuid() !== 0` refused this box, even though
// `prepareExecutor` would have wrapped the executor in `elevatedExecutor`.
h.canProceedClean = true;
h.host = { isRoot: false, canSudo: true, loginUser: "ubuntu", home: "/home/ubuntu" };
const res = await ensureSelfEdgeInfra();
expect(res.ok).toBe(true);
expect(h.ensureFeature).toHaveBeenCalledWith("ssl", expect.any(Function));
});
it("non-root with NO sudo → { reason:'not_root' } and never touches the box", async () => {
h.canProceedClean = true;
h.host = { isRoot: false, canSudo: false, loginUser: "deploy", home: "/home/deploy" };
const res = await ensureSelfEdgeInfra();
expect(res.ok).toBe(false);
expect(res.reason).toBe("not_root");
expect(h.ensureFeature).not.toHaveBeenCalled();
});
it("unsupported family → { reason:'unsupported_host' } and never touches the box", async () => {
h.canProceedClean = true;
h.host = { distro: "opensuse", distroFamily: "suse", distroId: "opensuse-leap" };
const res = await ensureSelfEdgeInfra();
expect(res.ok).toBe(false);
expect(res.reason).toBe("unsupported_host");
// The resolver's verdict, not just the code: which distro was observed is the
// whole reason this box was refused.
expect(res.detail).toContain("openSUSE");
expect(h.ensureFeature).not.toHaveBeenCalled();
});
});
+58 -11
View File
@@ -9,7 +9,9 @@
* the same routing/SSL path as every other app — no duplication.
*
* Single-flight so the boot reconcile + the wizard endpoint never install twice
* at once. Root Linux only (apt/dnf + certbot + systemd); a no-op elsewhere.
* at once. Linux only, and only where the resolver says we can elevate (root or
* passwordless sudo) on a host family Openship has a package manager for; a no-op
* elsewhere.
*/
import { env } from "../../config/env";
@@ -23,6 +25,19 @@ export interface SelfEdgeInfraProgress {
export interface SelfEdgeInfraResult {
ok: boolean;
reason?: string;
/**
* The classified CAUSE behind `reason`, in the words of whoever diagnosed it.
*
* `reason` is a code callers branch on, so it cannot carry the diagnosis — and
* every failure below already knows one: the resolver names the distro it refuses,
* and a failed takeover knows whether the stop was refused unelevated, :80 stayed
* bound, the vhost writes hit EACCES, or the previous proxy never came back.
* Collapsing that to "migrate_failed" is the cause-less failure #490 and #509 were
* both about — it sends an operator to retry an operation that cannot succeed as
* this user. Never re-worded here: the text is whatever the diagnosing layer
* produced, so one fault reads the same on every surface.
*/
detail?: string;
/** When reason === "edge_conflict": what holds 80/443 and how many sites it serves. */
occupants?: string;
siteCount?: number;
@@ -45,9 +60,9 @@ let inFlight: Promise<SelfEdgeInfraResult> | null = null;
/**
* Ensure OpenResty + certbot are installed (and optionally take over/migrate an
* existing proxy). Single-flight. Returns `{ok:false, reason}` on a non-Linux /
* non-root host or a failed migrate — the caller treats that as "skip the local
* edge" (free/byo domains don't need it).
* existing proxy). Single-flight. Returns `{ok:false, reason}` on a non-Linux host,
* one we can't elevate on, one whose family we can't provision, or a failed migrate
* — the caller treats that as "skip the local edge" (free/byo domains don't need it).
*/
export function ensureSelfEdgeInfra(
progress?: SelfEdgeInfraProgress,
@@ -84,22 +99,45 @@ async function runEnsure(
log("managed edge needs a Linux host — skipping (use a reverse proxy in front).", "warn");
return { ok: false, reason: "not_linux" };
}
if (typeof process.getuid === "function" && process.getuid() !== 0) {
log("managed edge needs root (to install OpenResty/certbot) — skipping.", "warn");
return { ok: false, reason: "not_root" };
}
const {
createExecutor,
SystemManager,
foreignProxyOnEdge,
importSites,
resolveEnvironment,
runEdgeTakeover,
} = await import("@repo/adapters");
const executor = createExecutor(); // LocalExecutor — this same machine
// Privilege and host support come from the resolver, on the SAME executor the
// installer will use — so this gate and `prepareExecutor` (installer.ts) share one
// cached measurement and cannot disagree about the box.
//
// The `getuid() !== 0` test this replaces refused every non-root operator, including
// the ones with passwordless sudo whom the installer would have elevated: the managed
// edge was declared impossible before the one component that knows how to elevate was
// ever asked. Still ahead of the `deliver-managed-image` import below, so a box that
// skips here also skips pulling the deploy runtime onto the boot path.
const profile = await resolveEnvironment(executor);
if (!profile.isRoot && !profile.canSudo) {
const detail =
`managed edge needs root to install OpenResty/certbot — this API runs as ` +
`${profile.loginUser} with no passwordless sudo.`;
log(`${detail} Skipping.`, "warn");
return { ok: false, reason: "not_root", detail };
}
if (!profile.supported) {
// Worth saying here rather than letting it surface as an OpenResty install failure
// three steps later: the reason names the observed distro, and nothing downstream
// adds information.
const detail = profile.unsupportedReason;
log(`managed edge can't be installed on this host — skipping. ${detail}`, "warn");
return { ok: false, reason: "unsupported_host", ...(detail ? { detail } : {}) };
}
// Lazy, like @repo/adapters above: deliver pulls in the deploy runtime (db, ssh,
// dockerode), which must stay off the boot path on the topologies that skip early.
const { deliverManagedImage } = await import("../deliver-managed-image");
const executor = createExecutor(); // LocalExecutor — this same machine
// Stage-B APPLY, build-only: this host IS the target, so build the edge from our
// source onto the local daemon before either bring-up path pulls the pinned tag.
@@ -138,7 +176,16 @@ async function runEnsure(
},
(entry) => log(entry.message, entry.level),
);
if (!res.ok) return { ok: false, reason: "migrate_failed" };
// `runEdgeTakeover` reports its diagnosis in `warnings` — the refused stop, the
// port that stayed bound, the vhosts it couldn't write, the proxy that didn't come
// back — and only `ok` was ever read. Relay them verbatim rather than summarizing:
// these are the sentences the CLI preflight and the deploy log already print, and a
// paraphrase here is a second copy that drifts.
for (const warning of res.warnings) log(warning, res.ok ? "warn" : "error");
if (!res.ok) {
const detail = res.warnings.join(" ");
return { ok: false, reason: "migrate_failed", ...(detail ? { detail } : {}) };
}
return { ok: true };
}
+130 -11
View File
@@ -4,10 +4,11 @@
*
* When OpenShip runs ON a server, the host is itself a deployable target, so it
* gets exactly ONE `isLocal` row, owned by the box org (`boxOwningOrgId`). Deploys
* to it resolve to the LOCAL host executor (createHostExecutor), not SSH — see
* `deployment-runtime.resolveServerExecutor`.
* to it resolve to the pooled HOST channel (`sshManager.acquireHostChannel`), not to
* a per-row SSH dial — see `deployment-runtime.acquireLocalHostExecutor`.
*
* {@link ensureLocalServer} is the ONLY place that row is created. It states an
* {@link ensureLocalServer} is the ONLY place that row is created, and
* {@link findLocalServer} is how a hot path asks whether it exists. It states an
* invariant about the MACHINE — never about a domain, an install method, or which
* branch of the admin bootstrap happened to run. That coupling is exactly what
* broke: the row used to be written only in the tail of a SUCCESSFUL
@@ -25,8 +26,19 @@
* Atomic in both senses: concurrent callers cannot produce two rows (single-flight
* below), and there is no half-registered state — a caller gets the one canonical
* row, or null with a logged reason.
*
* The row's EXISTENCE and its host-channel HEALTH are separate questions, and this
* module answers both without conflating them (#509). Only the explicit
* `--no-host-control` opt-out withholds the row: that is a policy saying this box is
* not a target. A channel that was never provisioned, or one a firewall drops, is not
* — the workload is reached through the mounted Docker socket, so ordinary deploys to
* this box work and hiding the row would break them. What was missing is that nothing
* WARNED either: the wizard offered "This Server" and the first host-side step failed
* mid-deploy. So {@link localServerHostChannel} reports the channel for the row, for
* the target list and the deploy wizard to surface — an annotation, never a gate.
*/
import { env } from "../../config/env";
import type { HostChannelCode } from "@repo/adapters";
import { repos, type Server } from "@repo/db";
import { boxOwningOrgId } from "../box-org";
import { resolvePlatformConfig } from "../controller-helpers";
@@ -66,11 +78,23 @@ export async function ensureLocalServer(opts?: EnsureLocalServerOptions): Promis
return inFlight;
}
async function register(opts?: EnsureLocalServerOptions): Promise<Server | null> {
// Not a deploy target at all: the SaaS control plane never is, and desktop
// targets its own machine through its own "This Machine" row. Gated HERE rather
// than only on the boot hook's `modes`, because the read/write paths below call
// this directly.
/**
* Who may own this box's row, or null when nobody can yet.
*
* Shared by the reader and the writer so they cannot disagree about whether this box
* has a row: a reader with a laxer gate would answer for a machine that is not a
* deploy target, and one with a stricter gate would report "no row" for a row that
* plainly exists.
*/
async function localServerOwnerOrg(): Promise<string | null> {
// No row where this box can't be a picked deploy target: the SaaS control plane
// never is, and desktop deploys to its own machine by DERIVING it (no cloud
// workspace + no serverId = "here", see project.ts) rather than through a row.
// Callers must handle the null: `resolveTargetPlatform`'s derived-local branch
// takes the pooled host channel, which is the same box either way.
//
// Gated HERE rather than only on the boot hook's `modes`, because the read/write
// paths below call this directly.
if (resolvePlatformConfig().target !== "selfhosted") return null;
// Host control off → this box is NOT a deploy target, so don't advertise it as
@@ -83,11 +107,57 @@ async function register(opts?: EnsureLocalServerOptions): Promise<Server | null>
// The row is owned by the box org (the founding admin's personal org), so it
// can't exist before that admin does. Every call site is a retry of this gate:
// on a CLI install the API boots — and the hook runs — before the admin exists.
const organizationId = await boxOwningOrgId();
return boxOwningOrgId();
}
/**
* READ this box's canonical row. Never creates one, never dials anything.
*
* For callers that need the row's identity on a hot path — `resolveTargetPlatform`
* runs per deploy and per runtime read, and only wants the id. Those must not call
* {@link ensureLocalServer}: creating on a resolve makes a deploy responsible for
* registering a server, and drags the creation path's public-IP detection (an
* outbound request) onto every resolve on a box whose row is missing. Registration
* belongs to the paths that ARE preparation — the boot hook, the admin-establishing
* endpoints, and the servers read path that self-heals.
*/
export async function findLocalServer(): Promise<Server | null> {
const organizationId = await localServerOwnerOrg();
if (!organizationId) return null;
return (await repos.server.findLocal(organizationId)) ?? null;
}
async function register(opts?: EnsureLocalServerOptions): Promise<Server | null> {
const organizationId = await localServerOwnerOrg();
if (!organizationId) return null;
// The account the host channel actually logs in as. The CLI writes
// OPENSHIP_HOST_SSH_USER when it provisions the container→host channel; the default
// matches `hostChannelUser()`, so a bare install with no channel agrees too. A row
// claiming `root` while the channel dials someone else is issue #489 — and since
// these fields are display-only (below), nothing else would ever correct it.
const desiredSshUser = process.env.OPENSHIP_HOST_SSH_USER?.trim() || "root";
const existing = await repos.server.findLocal(organizationId);
if (existing) return existing;
if (existing) {
// Reconciled on every path that calls this, not only at boot — the `GET /servers`
// self-heal included. `?? "root"` normalizes a legacy null so an old row doesn't
// re-issue the write on every read. Best-effort (a display value never blocks the
// caller), but never silent: a swallowed failure would leave us reporting a value
// we never persisted, re-attempting forever with nothing in the log.
if ((existing.sshUser ?? "root") !== desiredSshUser) {
repos.server
.update(existing.id, { sshUser: desiredSshUser })
.then(() =>
console.log(
`[self-server] reconciled ssh_user → ${desiredSshUser} (matches host channel)`,
),
)
.catch((err: unknown) => console.warn("[self-server] ssh_user reconcile failed:", err));
return { ...existing, sshUser: desiredSshUser };
}
return existing;
}
// ssh* fields are display-only for an isLocal row (never dialed). Prefer a real
// address so the servers list reads truthfully AND the DNS A record for a domain
@@ -103,12 +173,61 @@ async function register(opts?: EnsureLocalServerOptions): Promise<Server | null>
organizationId,
name: opts?.name?.trim() || "This Server",
sshHost,
sshUser: desiredSshUser,
isLocal: true,
});
console.log(`[self-server] registered this host as a deploy target (${sshHost})`);
console.log(
`[self-server] registered this host as a deploy target (${sshHost}, ssh_user=${desiredSshUser})`,
);
return row;
}
/** The container→host channel behind a local row's host-side steps. */
export interface LocalServerHostChannel {
/**
* Host ("this machine") operations work.
*
* Derived from `channel`, NOT from reachability: a `disabled` row is deliberately
* REACHABLE (its containers deploy over the Docker socket) while every host-side
* step refuses, and those are different questions. This one is "will the host-side
* steps of a deploy to this row run".
*/
ok: boolean;
/** Which host-channel state, in `hostChannelHealth`'s own vocabulary. */
channel: HostChannelCode;
/** Operator-facing remedy, when there is one — the same string this row's own
* reachability endpoint returns. One wording, one source. */
hint: string | null;
}
/**
* The host-channel state to show ON a server row, or null when the row has no channel
* to report (a remote box, or a diagnosis that couldn't run).
*
* Goes through `sshManager.diagnoseReachability` rather than probing here, so a row's
* badge in the list and its own reachability endpoint cannot disagree — that kind of
* divergence is what had the dashboard reporting the wrong machine (#490). It also
* inherits the circuit breaker, so a dead channel costs a remembered reason instead of
* a probe timeout per list render.
*
* Safe to call for ANY row: `channel` is only ever set for a row that resolves to this
* box, so a remote row answers null instead of being labelled with our channel.
*/
export async function localServerHostChannel(
serverId: string,
): Promise<LocalServerHostChannel | null> {
// Dynamic, like `hostControlDisabled` in `register()`: this file runs on the boot
// path and must not drag the SSH stack in before anything asks it a question.
const { sshManager } = await import("../ssh-manager");
const d = await sshManager.diagnoseReachability(serverId).catch(() => null);
if (!d?.channel) return null;
return {
ok: d.channel === "ok" || d.channel === "not_applicable",
channel: d.channel,
hint: d.hint ?? null,
};
}
export function registerSelfServerReconcile(): void {
registerStartupHook({
id: "self-server:reconcile",
+2 -2
View File
@@ -144,8 +144,8 @@ export async function linkSelfAppServices(
// PUBLIC service on the dashboard port that matches no container (shows
// "Stopped") and carries a stray {slug}-{slug} free-subdomain route. The
// self-app's only real units are the compose rows linked above, so drop any
// monorepo leftover. (The dashboard can't: assertNotControlPlaneService
// blocks deleting control-plane services.) Prevention lives in
// monorepo leftover. (The dashboard can't: assertNotControlPlane blocks
// mutating control-plane services.) Prevention lives in
// materializeAppServiceRow; this clears instances that predate that guard.
const stale = await repos.service.listByProjectKind(projectId, "monorepo").catch(() => []);
for (const row of stale) {
+236
View File
@@ -0,0 +1,236 @@
import { describe, expect, it, vi } from "vitest";
import {
buildUpstreamUrl,
resolveLiveUpstreamUrl,
resolveRouteStrategy,
resolveUpstreamUrl,
} from "./upstream-url";
/** A docker-shaped runtime whose live inspect we control. */
function dockerRuntime(opts: {
/** undefined → the inspect throws (unreachable daemon). */
info?: { status?: string; ip?: string; hostPort?: number };
ip?: string | null;
}) {
return {
name: "docker" as const,
supports: (cap: string) => cap === "containerIp" || cap === "containerInfo",
getContainerInfo: vi.fn(async () => {
if (!opts.info) throw new Error("daemon unreachable");
return { containerId: "c1", status: "running", ...opts.info } as never;
}),
getContainerIp: vi.fn(async () => (opts.ip === undefined ? "172.19.0.2" : opts.ip)),
};
}
describe("buildUpstreamUrl", () => {
it("dials the loopback host port when the workload publishes one", () => {
expect(
buildUpstreamUrl({ strategy: "loopback-port", ip: "172.19.0.2", hostPort: 4000, containerPort: 3001 }),
).toBe("http://127.0.0.1:4000");
});
it("falls back to the container IP when nothing is published", () => {
expect(buildUpstreamUrl({ strategy: "loopback-port", ip: "172.19.0.2", containerPort: 3001 })).toBe(
"http://172.19.0.2:3001",
);
});
it("returns null when neither a host port nor an ip is known", () => {
expect(buildUpstreamUrl({ strategy: "loopback-port", containerPort: 3001 })).toBeNull();
});
it("ignores a published host port under the container-ip strategy", () => {
expect(
buildUpstreamUrl({ strategy: "container-ip", ip: "172.19.0.2", hostPort: 4000, containerPort: 3001 }),
).toBe("http://172.19.0.2:3001");
});
});
describe("resolveRouteStrategy", () => {
it.each([
["auto", "loopback-port"],
[null, "loopback-port"],
["nonsense", "loopback-port"],
["container-ip", "container-ip"],
])("%s → %s", (setting, expected) => {
expect(resolveRouteStrategy(setting)).toBe(expected);
});
});
describe("resolveUpstreamUrl", () => {
it("uses the passed host port without touching the runtime", async () => {
const runtime = dockerRuntime({});
await expect(
resolveUpstreamUrl({
strategy: "loopback-port",
runtime,
containerId: "c1",
containerPort: 3001,
hostPort: 4000,
}),
).resolves.toBe("http://127.0.0.1:4000");
expect(runtime.getContainerIp).not.toHaveBeenCalled();
});
it("resolves the container IP when no host port is passed", async () => {
await expect(
resolveUpstreamUrl({
strategy: "loopback-port",
runtime: dockerRuntime({}),
containerId: "c1",
containerPort: 3001,
}),
).resolves.toBe("http://172.19.0.2:3001");
});
});
describe("resolveLiveUpstreamUrl", () => {
/**
* #506: a same-server migration attaches a container that was never published
* to 127.0.0.1, while `service_deployment.hostPort` still carries a port from
* an earlier deploy. A live inspect that ANSWERED "nothing published" must win
* over that stored port — otherwise the edge keeps dialing a dead loopback
* port behind a domain the dashboard reports as Verified.
*/
it("ignores a stored host port the container no longer publishes", async () => {
const runtime = dockerRuntime({ info: { ip: "172.19.0.2" } });
await expect(
resolveLiveUpstreamUrl({
strategy: "loopback-port",
runtime,
containerId: "c1",
containerPort: 3001,
stored: { ip: "172.19.0.2", hostPort: 3001 },
}),
).resolves.toBe("http://172.19.0.2:3001");
});
it("uses the live host port when the container does publish one", async () => {
await expect(
resolveLiveUpstreamUrl({
strategy: "loopback-port",
runtime: dockerRuntime({ info: { ip: "172.19.0.2", hostPort: 4000 } }),
containerId: "c1",
containerPort: 3001,
stored: { ip: "172.19.0.2", hostPort: 3999 },
}),
).resolves.toBe("http://127.0.0.1:4000");
});
it("prefers the LIVE host port over a stale stored one", async () => {
await expect(
resolveLiveUpstreamUrl({
strategy: "loopback-port",
runtime: dockerRuntime({ info: { hostPort: 4100 } }),
containerId: "c1",
containerPort: 3001,
stored: { hostPort: 4000 },
}),
).resolves.toBe("http://127.0.0.1:4100");
});
it("keeps the last-known host port when the container CANNOT be inspected", async () => {
// Unreachable daemon is not evidence that nothing is published — a re-apply
// must not silently repoint a working vhost.
await expect(
resolveLiveUpstreamUrl({
strategy: "loopback-port",
runtime: dockerRuntime({ ip: null }),
containerId: "c1",
containerPort: 3001,
stored: { ip: "172.19.0.2", hostPort: 4000 },
}),
).resolves.toBe("http://127.0.0.1:4000");
});
it("keeps the last-known ip when the container cannot be inspected and nothing was published", async () => {
await expect(
resolveLiveUpstreamUrl({
strategy: "loopback-port",
runtime: dockerRuntime({ ip: null }),
containerId: "c1",
containerPort: 3001,
stored: { ip: "172.19.0.2" },
}),
).resolves.toBe("http://172.19.0.2:3001");
});
it("treats a MISSING container as publishing nothing, not as unknown", async () => {
// The container is gone: a stored host port must not resurrect it. The last
// known ip is still offered rather than blanking the route outright.
await expect(
resolveLiveUpstreamUrl({
strategy: "loopback-port",
runtime: dockerRuntime({ info: { status: "missing" }, ip: null }),
containerId: "c1",
containerPort: 3001,
stored: { ip: "172.19.0.2", hostPort: 4000 },
}),
).resolves.toBe("http://172.19.0.2:3001");
});
it("returns null when nothing — live or stored — resolves", async () => {
await expect(
resolveLiveUpstreamUrl({
strategy: "loopback-port",
runtime: dockerRuntime({ info: {}, ip: null }),
containerId: "c1",
containerPort: 3001,
}),
).resolves.toBeNull();
});
it("never inspects for the container-ip strategy", async () => {
const runtime = dockerRuntime({ info: { ip: "172.19.0.2", hostPort: 4000 } });
await expect(
resolveLiveUpstreamUrl({
strategy: "container-ip",
runtime,
containerId: "c1",
containerPort: 3001,
stored: { hostPort: 4000 },
}),
).resolves.toBe("http://172.19.0.2:3001");
expect(runtime.getContainerInfo).not.toHaveBeenCalled();
});
it("routes a bare workload at its own loopback port without inspecting", async () => {
const runtime = {
name: "bare" as const,
supports: () => true,
getContainerInfo: vi.fn(),
getContainerIp: vi.fn(async () => "127.0.0.1"),
};
await expect(
resolveLiveUpstreamUrl({
strategy: "loopback-port",
runtime,
containerId: "app",
containerPort: 3001,
}),
).resolves.toBe("http://127.0.0.1:3001");
expect(runtime.getContainerInfo).not.toHaveBeenCalled();
});
it("falls back to the stored row when the runtime cannot inspect at all", async () => {
const runtime = {
name: "docker" as const,
supports: (cap: string) => cap === "containerIp",
getContainerIp: vi.fn(async () => null),
};
await expect(
resolveLiveUpstreamUrl({
strategy: "loopback-port",
runtime,
containerId: "c1",
containerPort: 3001,
stored: { hostPort: 4000 },
}),
).resolves.toBe("http://127.0.0.1:4000");
});
});
+74
View File
@@ -21,6 +21,19 @@ export type RouteStrategySetting = RouteStrategy | "auto";
type UpstreamRuntime = Pick<RuntimeAdapter, "supports" | "getContainerIp">;
/** Runtime surface needed to read a container's CURRENT publishing. */
type LiveUpstreamRuntime = UpstreamRuntime &
Pick<RuntimeAdapter, "name"> & { getContainerInfo?: RuntimeAdapter["getContainerInfo"] };
/**
* Last-known upstream for a container, as persisted on `service_deployment`.
* A CACHE of a past live read — never authoritative on its own.
*/
export interface StoredUpstream {
ip?: string | null;
hostPort?: number | null;
}
export interface ResolveUpstreamArgs {
strategy: RouteStrategy;
runtime: UpstreamRuntime;
@@ -65,6 +78,67 @@ export async function resolveUpstreamUrl(args: ResolveUpstreamArgs): Promise<str
return buildUpstreamUrl({ strategy, ip, hostPort, containerPort });
}
/**
* Does the container publish a host port RIGHT NOW?
*
* `known:false` means "we could not ask" — an unreachable daemon, or a runtime
* that can't inspect. That is NOT the same as "publishes nothing", and every
* call site used to conflate the two by writing `info?.hostPort ?? row.hostPort`:
* a live read that answered "no binding" fell straight through to a stored port
* the container no longer had, so the edge kept dialing a dead
* `127.0.0.1:<port>`. That is how a same-server migration left a healthy app
* unreachable behind a Verified domain (#506).
*/
async function readLiveHostPort(
runtime: LiveUpstreamRuntime,
containerId: string,
strategy: RouteStrategy,
): Promise<{ known: boolean; hostPort?: number }> {
// container-ip never dials a host port, and a bare workload owns
// `127.0.0.1:<appPort>` outright — neither has a publish to read.
if (strategy !== "loopback-port" || runtime.name === "bare") return { known: true };
if (!runtime.getContainerInfo || !runtime.supports("containerInfo")) return { known: false };
try {
const info = await runtime.getContainerInfo(containerId);
// A `missing` container answers too: it is gone, so it publishes nothing and
// a stored port must not resurrect it.
return { known: true, hostPort: info.hostPort };
} catch {
return { known: false };
}
}
/**
* The ONE live upstream resolver — what every route-registration site outside a
* deploy should call.
*
* Live container state decides the upstream: a routed workload with no loopback
* publish (migrated, adopted in place, or an internal compose service) resolves
* to its container IP instead of a port nothing listens on. `stored` — the
* persisted `service_deployment` row — is consulted ONLY when the live read
* could not be performed, so one failed inspect keeps the last-known route
* instead of blanking a working vhost.
*/
export async function resolveLiveUpstreamUrl(args: {
strategy: RouteStrategy;
runtime: LiveUpstreamRuntime;
containerId: string;
containerPort: number;
stored?: StoredUpstream;
}): Promise<string | null> {
const { strategy, runtime, containerId, containerPort, stored } = args;
const live = await readLiveHostPort(runtime, containerId, strategy);
const hostPort = live.known ? live.hostPort : (stored?.hostPort ?? undefined);
const url = await resolveUpstreamUrl({
strategy,
runtime,
containerId,
containerPort,
hostPort,
}).catch(() => null);
return url ?? buildUpstreamUrl({ strategy, ip: stored?.ip, hostPort, containerPort });
}
/**
* Resolve a stored/selected strategy setting to a concrete strategy. "auto" (and
* any unknown/legacy value) → "loopback-port", the safe default for bare + docker
@@ -0,0 +1,145 @@
import { beforeEach, describe, expect, it, vi } from "vitest";
/**
* `withDeploymentRuntime` — the one entry point every container action (pause,
* resume, restart, logs, info, usage) now goes through.
*
* Its three guarantees are exactly the three things the hand-rolled call sites
* each got wrong in a different combination, so they're pinned here once instead
* of per caller:
*
* - the transport is ALWAYS disposed, including when the action throws. The SSH
* branch mints a new loopback bridge per runtime, so a missed dispose leaks a
* listening socket plus an ssh client per click.
* - "couldn't reach the host" becomes a 503 `HOST_UNREACHABLE` carrying the
* transport's own reason (a refused key names the target and what to check).
* - anything else propagates untouched — a real bug must not be relabelled as
* an infrastructure problem.
*/
const h = vi.hoisted(() => ({
disposed: 0,
created: 0,
}));
// Only the transport is faked. The error CLASSIFIERS stay real on purpose: they
// are the thing under test here — a stubbed `isRemoteConnectionError` would let
// this suite pass while the mapping it claims to verify did nothing.
vi.mock("@repo/adapters", async () => {
const actual = await vi.importActual<typeof import("@repo/adapters")>("@repo/adapters");
return {
...actual,
DockerRuntime: {
create: async () => {
h.created += 1;
return {
dispose: async () => {
h.disposed += 1;
},
};
},
},
};
});
// Self-hosted base + a snapshot with no serverId → effective target "local",
// which resolves through DockerRuntime.create above with no SSH anywhere.
vi.mock("./controller-helpers", () => ({ platform: () => ({ target: "selfhosted" }) }));
vi.mock("@repo/db", () => ({ repos: { service: { listByDeployment: async () => [] } } }));
vi.mock("./cloud/client", () => ({ cloudClient: {}, getOrgCloudToken: async () => null }));
vi.mock("./cloud/transport", () => ({ resolveOrgCloudUserId: async () => null }));
vi.mock("./ssh-manager", () => ({ buildSshConfig: async () => null, sshManager: {} }));
vi.mock("./provision-lock", () => ({ createProvisionLock: () => ({}) }));
vi.mock("./box-org", () => ({ isLocalHostRow: async () => true }));
vi.mock("./acme-config", () => ({ resolveAcmeProviderOptions: () => ({}) }));
const dep = { meta: {}, organizationId: "org_1" };
describe("withDeploymentRuntime", () => {
beforeEach(() => {
h.disposed = 0;
h.created = 0;
});
it("returns the action's value and disposes the transport", async () => {
const { withDeploymentRuntime } = await import("./deployment-runtime");
await expect(withDeploymentRuntime(dep, async () => "logs")).resolves.toBe("logs");
expect(h.created).toBe(1);
expect(h.disposed).toBe(1);
});
it("disposes the transport when the action throws", async () => {
const { withDeploymentRuntime } = await import("./deployment-runtime");
await expect(
withDeploymentRuntime(dep, async () => {
throw new Error("docker said no");
}),
).rejects.toThrow("docker said no");
expect(h.disposed).toBe(1);
});
it("maps a refused SSH key to 503 HOST_UNREACHABLE, keeping the reason", async () => {
const { withDeploymentRuntime } = await import("./deployment-runtime");
const reason =
"SSH key authentication failed for root@65.109.55.23. Check the username, private key, " +
"passphrase, or whether the server accepts this key. (All configured authentication methods failed)";
const err = await withDeploymentRuntime(dep, async () => {
throw new Error(reason);
}).catch((e: unknown) => e);
expect((err as { statusCode?: number }).statusCode).toBe(503);
expect((err as { code?: string }).code).toBe("HOST_UNREACHABLE");
expect((err as Error).message).toContain("65.109.55.23");
expect(h.disposed).toBe(1);
});
it.each([
["connect ETIMEDOUT 10.0.0.9:22"],
["socket hang up"],
["Channel open failure: open failed"],
["Command timed out after 30000ms"],
])("maps transport failure %j to 503", async (message) => {
const { withDeploymentRuntime } = await import("./deployment-runtime");
const err = await withDeploymentRuntime(dep, async () => {
throw new Error(message);
}).catch((e: unknown) => e);
expect((err as { statusCode?: number }).statusCode).toBe(503);
});
it("leaves an ordinary failure alone — no invented 503", async () => {
const { withDeploymentRuntime } = await import("./deployment-runtime");
const err = await withDeploymentRuntime(dep, async () => {
throw new Error("(HTTP code 404) no such container: abc123");
}).catch((e: unknown) => e);
expect((err as { statusCode?: number }).statusCode).toBeUndefined();
expect((err as Error).message).toContain("no such container");
});
});
describe("deploymentContainerIds", () => {
it("prefers the service containers, and falls back to the deployment's own", async () => {
vi.resetModules();
vi.doMock("@repo/db", () => ({
repos: { service: { listByDeployment: async () => [{ containerId: "svc-a" }, { containerId: null }] } },
}));
const { deploymentContainerIds } = await import("./deployment-runtime");
expect(await deploymentContainerIds({ id: "dep_1", containerId: "app" })).toEqual(["svc-a"]);
vi.resetModules();
vi.doMock("@repo/db", () => ({ repos: { service: { listByDeployment: async () => [] } } }));
const fresh = await import("./deployment-runtime");
expect(await fresh.deploymentContainerIds({ id: "dep_1", containerId: "app" })).toEqual(["app"]);
expect(await fresh.deploymentContainerIds({ id: "dep_1", containerId: null })).toEqual([]);
});
});
@@ -0,0 +1,203 @@
import { beforeEach, describe, expect, it, vi } from "vitest";
/**
* The resolver that decides which tenant a request is scoped to.
*
* It had no tests, which is the wrong shape for the one function in the codebase that is
* ALLOWED to guess an organization. Everywhere else reads `ctx.organizationId`; this is
* where that value comes from, so its two properties are worth pinning permanently:
*
* 1. **It can only ever return an org the user is a member of.** Every candidate is drawn
* from `memberOrgIds`, so the blast radius of a wrong guess is "scoped to the wrong org
* you already belong to" and never a cross-tenant read. A stale
* `session.activeOrganizationId` — the user was removed from that org after the session
* was issued — must fall through rather than be trusted.
* 2. **It is deterministic.** Two API nodes resolving the same user must agree, or the
* active org flickers between requests.
*/
const h = vi.hoisted(() => ({
members: [] as Array<{ organizationId: string; createdAt: Date | null }>,
orgs: [] as Array<{ id: string; isTeam: boolean } | null>,
/** Set to an Error to make the corresponding repo call reject. */
membersError: null as Error | null,
orgsError: null as Error | null,
}));
vi.mock("@repo/db", () => ({
repos: {
member: {
listByUser: async () => {
if (h.membersError) throw h.membersError;
// The real repo orders by createdAt; the resolver must not depend on more than
// that, so this returns rows in the order the fixture declares them.
return h.members;
},
},
organization: {
findManyById: async () => {
if (h.orgsError) throw h.orgsError;
return h.orgs;
},
},
},
}));
import { resolveActiveOrganizationId } from "./active-organization";
const at = (iso: string) => new Date(iso);
beforeEach(() => {
h.members = [];
h.orgs = [];
h.membersError = null;
h.orgsError = null;
});
/** Declare memberships plus which of those orgs are team orgs. */
function host(
members: Array<{ organizationId: string; createdAt: Date | null }>,
teamOrgIds: string[] = [],
) {
h.members = members;
h.orgs = members.map((m) => ({
id: m.organizationId,
isTeam: teamOrgIds.includes(m.organizationId),
}));
}
describe("resolveActiveOrganizationId", () => {
it("honours a session org the user is still a member of", async () => {
host([
{ organizationId: "org_a", createdAt: at("2026-01-01") },
{ organizationId: "org_b", createdAt: at("2026-01-02") },
]);
expect(await resolveActiveOrganizationId("u1", "org_b")).toBe("org_b");
});
it("falls through a session org the user has since been removed from", async () => {
// The security property: a session outlives a membership. Trusting the pointer here
// would scope the request to an org this user can no longer see.
host([{ organizationId: "org_a", createdAt: at("2026-01-01") }]);
expect(await resolveActiveOrganizationId("u1", "org_gone")).toBe("org_a");
});
it("never returns an org outside the user's memberships", async () => {
// `findManyById` answering with rows the membership set does not contain — a stale
// read, or an id collision — must not widen the candidate set.
h.members = [{ organizationId: "org_a", createdAt: at("2026-01-01") }];
h.orgs = [
{ id: "org_a", isTeam: false },
{ id: "org_foreign", isTeam: true },
];
expect(await resolveActiveOrganizationId("u1", null)).toBe("org_a");
});
it("prefers a team org over the personal workspace", async () => {
// The personal workspace is created first and is empty by default, so ordering alone
// would land every team member in the wrong place.
host(
[
{ organizationId: "org_personal", createdAt: at("2026-01-01") },
{ organizationId: "org_team", createdAt: at("2026-01-02") },
],
["org_team"],
);
expect(await resolveActiveOrganizationId("u1", null)).toBe("org_team");
});
it("picks the OLDEST team org when there are several", async () => {
host(
[
{ organizationId: "org_new", createdAt: at("2026-06-01") },
{ organizationId: "org_old", createdAt: at("2026-01-01") },
],
["org_new", "org_old"],
);
expect(await resolveActiveOrganizationId("u1", null)).toBe("org_old");
});
describe("determinism when timestamps tie", () => {
/**
* Onboarding writes the personal workspace and a team org in one transaction, so equal
* `createdAt` values are the normal case rather than a curiosity. `ORDER BY createdAt`
* leaves the order among equals up to the planner, so without a second key two nodes
* can disagree — and the tiebreaker used to be on the team-org branch only, leaving
* the plain fallback as the unstable one.
*/
const SAME = at("2026-01-01T00:00:00.000Z");
it("breaks a tie by org id on the fallback path", async () => {
host([
{ organizationId: "org_zzz", createdAt: SAME },
{ organizationId: "org_aaa", createdAt: SAME },
]);
expect(await resolveActiveOrganizationId("u1", null)).toBe("org_aaa");
});
it("and gives the same answer when the rows arrive reversed", async () => {
host([
{ organizationId: "org_aaa", createdAt: SAME },
{ organizationId: "org_zzz", createdAt: SAME },
]);
expect(await resolveActiveOrganizationId("u1", null)).toBe("org_aaa");
});
it("breaks a tie by org id among team orgs too", async () => {
host(
[
{ organizationId: "org_team_z", createdAt: SAME },
{ organizationId: "org_team_a", createdAt: SAME },
],
["org_team_z", "org_team_a"],
);
expect(await resolveActiveOrganizationId("u1", null)).toBe("org_team_a");
});
});
it("returns null for a user with no memberships", async () => {
host([]);
expect(await resolveActiveOrganizationId("u1", "org_anything")).toBeNull();
});
it("returns null rather than throwing when memberships cannot be read", async () => {
// Callers decide whether that is a 403 or a pass-through (org-free routes), so a
// transport failure must not become a 500 from the middleware.
h.membersError = new Error("connection terminated");
expect(await resolveActiveOrganizationId("u1", null)).toBeNull();
});
it("still resolves when the org lookup fails, treating none as a team org", async () => {
// Losing `isTeam` costs the team-org PREFERENCE, not the answer. Failing here would
// log every user out of a working session over a read that only ranks candidates.
h.members = [
{ organizationId: "org_a", createdAt: at("2026-01-01") },
{ organizationId: "org_b", createdAt: at("2026-01-02") },
];
h.orgsError = new Error("connection terminated");
expect(await resolveActiveOrganizationId("u1", null)).toBe("org_a");
});
it("tolerates a null createdAt without reordering the rest", async () => {
// `member.createdAt` is nullable in the schema; `new Date(null ?? 0)` is the epoch,
// so such a row sorts first rather than turning the comparison into NaN — which would
// make the sort's result depend on the engine's pivot choice.
host([
{ organizationId: "org_dated", createdAt: at("2026-01-01") },
{ organizationId: "org_undated", createdAt: null },
]);
expect(await resolveActiveOrganizationId("u1", null)).toBe("org_undated");
});
});
+41 -30
View File
@@ -49,43 +49,50 @@ export async function resolveActiveOrganizationId(
return sessionOrgId;
}
// TODO: is this clean way ? isnt resolve active org id shouldn't have fallbacks or what
// Prefer a team org over an empty personal workspace. Batch lookup —
// every authenticated request hits this resolver, an N+1 per
// membership would be unacceptable. Sort the team-org candidates
// deterministically by member.createdAt so the "first team org"
// pick is stable across nodes (MEDIUM cleanup).
// membership would be unacceptable.
const orgs = await repos.organization
.findManyById(Array.from(memberOrgIds))
.catch(() => []);
const teamOrgIds = new Set(
orgs.filter((o) => o?.isTeam === true).map((o) => o!.id),
);
if (teamOrgIds.size > 0) {
const teamMemberships = memberships
.filter((m) => teamOrgIds.has(m.organizationId))
.sort((a, b) => {
const ta =
a.createdAt instanceof Date
? a.createdAt.getTime()
: new Date(a.createdAt ?? 0).getTime();
const tb =
b.createdAt instanceof Date
? b.createdAt.getTime()
: new Date(b.createdAt ?? 0).getTime();
if (ta !== tb) return ta - tb;
return a.organizationId.localeCompare(b.organizationId);
});
if (teamMemberships.length > 0) return teamMemberships[0].organizationId;
}
// not ctx-scoped: middleware boundary. This IS the canonical resolver
// that BUILDS the per-request active org. The "memberships[0]"
// fallback is acceptable HERE because no ctx exists yet — it's the
// source from which ctx.organizationId gets populated. Foreground
// services downstream must read ctx.organizationId rather than re-
// running this fallback.
return memberships[0].organizationId;
// Sorted ONCE, and both picks below read from it. `listByUser` already orders by
// createdAt, but that is not a total order: memberships written in one transaction
// (onboarding creates the personal workspace and a team org together) share a
// timestamp, and row order among equals is then whatever the planner returns. The
// org id breaks the tie so two API nodes resolve the same user to the same org.
// This used to be applied to the team-org pick only, which left the branch WITHOUT
// a tiebreaker as the unstable one — the reverse of what you'd want.
const ordered = [...memberships].sort((a, b) => {
const at = new Date(a.createdAt ?? 0).getTime();
const bt = new Date(b.createdAt ?? 0).getTime();
if (at !== bt) return at - bt;
return a.organizationId.localeCompare(b.organizationId);
});
const team = ordered.find((m) => teamOrgIds.has(m.organizationId));
if (team) return team.organizationId;
/**
* The fallbacks are the point of this function, not a shortcut in it.
*
* "Shouldn't the resolver have no fallbacks?" — the opposite: it is BECAUSE this is
* the one place allowed to guess that everywhere else is forbidden to. This resolver
* BUILDS `ctx.organizationId`; there is no ctx yet to read, so someone has to decide,
* and centralizing that decision is what stops each controller inventing its own
* `memberships[0]`. Downstream services read `ctx.organizationId` — see
* `controller-helpers.ts`, which names this function as the sole exception.
*
* What makes the guessing safe is not the absence of fallbacks but the invariant that
* every candidate comes from `memberOrgIds`, which is built from this user's own
* memberships. No arm of this function can return an org the user is not a member of,
* so the worst outcome is landing in the wrong org the user already belongs to — a
* scoping annoyance, never a cross-tenant read.
*/
return ordered[0].organizationId;
}
/**
@@ -169,8 +176,12 @@ export function requireRole(
if (!m) {
return c.json({ error: "Not a member of this organization" }, 403);
}
const role = (m.role as "member" | "admin" | "owner") ?? "member";
if (RANK[role] < RANK[min]) {
// Deny any role not in RANK rather than comparing `undefined`. A member row
// with role "restricted" used to make `RANK[role] < RANK[min]` evaluate
// `undefined < n` → false, so requireRole FAILED OPEN for exactly the
// least-trusted role.
const rank = RANK[m.role as keyof typeof RANK];
if (rank === undefined || rank < RANK[min]) {
return c.json(
{ error: `Requires ${min} role`, code: "INSUFFICIENT_ROLE" },
403,
+5 -1
View File
@@ -34,7 +34,11 @@ export function handleApiError(err: unknown, c: Context) {
const { message, code, statusCode } = err;
return c.json(
{ error: message, code },
statusCode as 400 | 401 | 403 | 404 | 409 | 500,
// 502/503 included: an AppError can legitimately mean "an upstream we
// depend on failed" (HostUnreachableError, ManagedEdgeError), and casting
// those away made this the one place that couldn't express the status the
// error itself already carried.
statusCode as 400 | 401 | 403 | 404 | 409 | 500 | 502 | 503,
);
}
+1
View File
@@ -6,6 +6,7 @@ export { betterAuthShield } from "./better-auth-shield";
export { originGuard } from "./origin-guard";
export { clientIpMiddleware } from "./client-ip";
export { requireRole } from "./active-organization";
export { assertInstanceAdmin, requireInstanceAdmin } from "./instance-admin";
export { migrationGuard } from "./migration-guard";
export {
isLoopbackPeer,
+86
View File
@@ -0,0 +1,86 @@
/**
* Instance-level authorization — the gate for WHOLE-INSTANCE operations.
*
* Why this exists (GHSA-rwq6-r63g-3c8h): an org-scoped check cannot gate an
* instance-wide operation, so `requireRole("owner")` never separated privilege
* on these routes. Two facts compose into a bypass:
*
* 1. The org that check resolves is CALLER-SELECTED. `permission.assert`
* stashes `scopedOrganizationId` (lib/permission.ts) from
* `resolveRequestScopeOrg`, which reads `X-Organization-Id` ahead of the
* session org; `requireRole` reads that stash first and unconditionally
* (middleware/active-organization.ts).
* 2. EVERY user is `owner` of an auto-provisioned personal org
* `org_<userId>` (lib/provision-user.ts), which `lib/auth.ts` also sets as
* their default active org.
*
* So any authenticated user satisfied `requireRole("owner")` — no privilege
* escalation step required. The instance role is a DIFFERENT axis and is not
* caller-selectable: `user.role === "admin"` holds only for the zero-auth local
* user, the founding admin promoted from it by `bootstrapAdmin`, and
* cloud-mirrored users. Better Auth signups and `invite-signup` teammates are
* role `"user"` (lib/provision-user.ts defaults `role` to `"user"`).
*
* INVARIANT — do not break this: these helpers derive NO organization. They must
* never read `X-Organization-Id`, `scopedOrganizationId`, or
* `activeOrganizationId`. Reading any of them reintroduces the bypass.
*/
import type { Context, Next } from "hono";
import { ForbiddenError } from "@repo/core";
import { db, schema, eq } from "@repo/db";
import { getRequestContext, type RequestContext } from "../lib/request-context";
const DENIED = "Requires an instance administrator";
/** True when this principal is an admin OF THE INSTANCE (not of any org). */
async function isInstanceAdmin(ctx: RequestContext): Promise<boolean> {
// A scoped token must never carry instance-takeover capability, whoever owns
// it — a narrowly-granted PAT reaching a whole-instance export would defeat
// the point of scoping. Unscoped PATs (how the CLI authenticates) still pass.
if (ctx.tokenScope) return false;
const [row] = await db
.select({ role: schema.user.role })
.from(schema.user)
.where(eq(schema.user.id, ctx.userId))
.limit(1);
return row?.role === "admin";
}
/**
* Route middleware: 403 unless the caller is an instance administrator.
*
* Mount this INSTEAD OF `requireRole("owner")` on any route whose handler acts
* across the whole instance (every org's data, the host, the control plane).
*/
export function requireInstanceAdmin() {
return async (c: Context, next: Next) => {
let ctx: RequestContext;
try {
ctx = getRequestContext(c);
} catch {
// Route mounted without authMiddleware — fail loud rather than open.
return c.json({ error: "Unauthorized" }, 401);
}
if (!(await isInstanceAdmin(ctx))) {
return c.json({ error: DENIED, code: "INSUFFICIENT_INSTANCE_ROLE" }, 403);
}
await next();
};
}
/**
* Handler-level form, for defense in depth inside controllers and services so a
* future route (or an internal caller) can't re-open the hole by forgetting the
* middleware. Throws ForbiddenError (403).
*/
export async function assertInstanceAdmin(ctx: RequestContext): Promise<void> {
if (!(await isInstanceAdmin(ctx))) {
throw new ForbiddenError(DENIED);
}
}
@@ -0,0 +1,83 @@
/**
* What the app grid is allowed to show.
*
* `getAppCatalog` is the ONE place an app can be hidden from the browsable
* catalog, and hiding is deliberately not the same thing as refusing: `available:
* false` makes `installApp` throw `app-not-available`, while `unlisted` only drops
* the card. Webmail depends on that difference — the mail wizard installs it by id,
* so an unlisting that leaked into installability would break "connect existing".
*
* The mirror-image failure is a custom app that unlists ITSELF: the grid is the only
* entry point an org-uploaded app has, so `listOrgCustomApps` forces the flag off.
*/
import { describe, it, expect, vi } from "vitest";
import type { RequestContext } from "../../lib/request-context";
const h = vi.hoisted(() => ({
runtime: [] as unknown[],
custom: [] as unknown[],
}));
vi.mock("./catalog-source", () => ({
getRuntimeCatalog: () => h.runtime,
listOrgCustomApps: async () => h.custom,
getTemplateForOrg: async () => undefined,
}));
vi.mock("@repo/db", () => ({ repos: {} }));
vi.mock("../projects/project-crud.service", () => ({ createProject: vi.fn() }));
vi.mock("../services/service.service", () => ({
createService: vi.fn(),
updateService: vi.fn(),
setServiceEnvVars: vi.fn(),
}));
const { getAppCatalog } = await import("./app-install.service");
const ctx = { organizationId: "org1" } as RequestContext;
function app(id: string, extra: Record<string, unknown> = {}) {
return {
id,
name: id,
description: "d",
kind: "template",
logo: id,
category: "mail",
available: true,
...extra,
};
}
describe("getAppCatalog", () => {
it("drops an unlisted app from the grid", async () => {
h.runtime = [app("mail", { kind: "flow" }), app("webmail", { unlisted: true })];
h.custom = [];
const ids = (await getAppCatalog(ctx)).map((a) => a.id);
expect(ids).toEqual(["mail"]);
});
it("keeps a coming-soon app listed — unavailable is not unlisted", async () => {
h.runtime = [app("neon", { available: false })];
h.custom = [];
const listed = await getAppCatalog(ctx);
expect(listed.map((a) => a.id)).toEqual(["neon"]);
expect(listed[0].comingSoon).toBe(true);
});
it("lists a custom app even when its JSON asks to be unlisted", async () => {
// Matches what listOrgCustomApps hands back (it forces the flag), so this test
// fails if that normalization is ever dropped from either side.
h.runtime = [];
h.custom = [app("acme", { unlisted: false, custom: true })];
const ids = (await getAppCatalog(ctx)).map((a) => a.id);
expect(ids).toEqual(["acme"]);
});
});
@@ -0,0 +1,101 @@
import { beforeEach, describe, expect, it, vi } from "vitest";
/**
* `GET /apps/catalog/:id/host-fit` — WHICH machine the app's declared minimum is
* matched against.
*
* `isLocalTarget` is what makes a `source: "local"` capacity probe trusted, and that
* probe reads the API host's OWN `os.*`. So it has to be derived from the server row,
* not read off the query string: a caller naming "local" for a remote server would
* have had the app sized against the orchestrator's RAM and been told about a
* shortfall on the wrong box. Same `isLocalHostRow` test the deploy path uses, so the
* advisory notice here and the preflight refusal describe one machine.
*/
const h = vi.hoisted(() => ({
/** Every getTrustedHostCapacity call: the server asked about, and the claim. */
probes: [] as Array<{ serverId?: string; isLocalTarget: boolean }>,
rows: {} as Record<string, { id: string; isLocal: boolean } | undefined>,
}));
vi.mock("./catalog-source", () => ({
getTemplateForOrg: async (_org: string, id: string) => ({
id,
minResources: { memoryMb: 2048 },
}),
getRuntimeCatalog: async () => [],
listOrgCustomApps: async () => [],
}));
// Partial: the installer's module graph reaches auth/provision-user, which
// destructures `schema` at import time — a repos-only stub fails to LOAD.
vi.mock("@repo/db", async (importOriginal) => ({
...(await importOriginal<Record<string, unknown>>()),
repos: {
server: {
getInOrganization: async (id: string) => h.rows[id] ?? null,
},
project: { findDraftByAppTemplate: async () => null },
},
}));
// The same predicate the deploy path uses; keyed off the flag here so the test
// doesn't depend on loopback resolution or env.
vi.mock("../../lib/box-org", () => ({
isLocalHostRow: async (row: { isLocal?: boolean }) => Boolean(row?.isLocal),
}));
vi.mock("../../lib/host-capacity", () => ({
getTrustedHostCapacity: async (
serverId: string | undefined,
_org: string,
opts: { isLocalTarget?: boolean },
) => {
h.probes.push({ serverId, isLocalTarget: Boolean(opts?.isLocalTarget) });
return { memoryMb: 4096, cpus: 2, diskGb: 40, source: "docker" as const };
},
}));
const { getAppHostFit } = await import("./app-install.service");
const ctx = { userId: "u1", organizationId: "org1" } as never;
const fit = (target: { deployTarget?: string; serverId?: string }) =>
getAppHostFit(ctx, "some-app", target);
beforeEach(() => {
h.probes = [];
h.rows = {
"srv-local": { id: "srv-local", isLocal: true },
"srv-remote": { id: "srv-remote", isLocal: false },
};
});
describe("getAppHostFit — the destination is derived, not claimed", () => {
it("ignores a caller claiming 'local' for a REMOTE server", async () => {
await fit({ deployTarget: "local", serverId: "srv-remote" });
expect(h.probes).toEqual([{ serverId: "srv-remote", isLocalTarget: false }]);
});
it("trusts the local probe for this box's own row", async () => {
await fit({ deployTarget: "server", serverId: "srv-local" });
expect(h.probes).toEqual([{ serverId: "srv-local", isLocalTarget: true }]);
});
it("treats no server at all as this box — an unbound project derives exactly that", async () => {
await fit({ deployTarget: "server" });
expect(h.probes).toEqual([{ serverId: undefined, isLocalTarget: true }]);
});
it("reports unknown for a server this org doesn't own, without probing anything", async () => {
const res = await fit({ deployTarget: "server", serverId: "srv-other-tenant" });
expect(h.probes).toEqual([]);
expect(res.capacity.source).toBe("unknown");
expect(res.fit.ok).toBe(true);
});
it("skips the match for cloud, which is sized from the tier table", async () => {
const res = await fit({ deployTarget: "cloud" });
expect(h.probes).toEqual([]);
expect(res.fit.ok).toBe(true);
});
});
+123 -10
View File
@@ -14,20 +14,30 @@ import {
getAppEndpoints,
declaredServiceRoutes,
defaultAppRouteLabel,
fitsCapacity,
hasMinResources,
normalizeCustomHostname,
isValidCustomHostname,
resolveServiceHostnameLabel,
slugify,
ConflictError,
UNKNOWN_CAPACITY,
type AppConfigField,
type AppMinResources,
type AppTemplate,
type HostCapacity,
type ResourceFit,
type TemplateServiceSpec,
type TemplateServiceBuild,
} from "@repo/core";
import { getRuntimeCatalog, getTemplateForOrg, listOrgCustomApps } from "./catalog-source";
import { repos } from "@repo/db";
import { env } from "../../config";
import type { RequestContext } from "../../lib/request-context";
import { isLocalHostRow } from "../../lib/box-org";
import { parseServicePort } from "../../lib/deployable-service";
import { requireCloud } from "../../lib/cloud/require-cloud";
import { getTrustedHostCapacity } from "../../lib/host-capacity";
import { createProject } from "../projects/project-crud.service";
import { createService, updateService, setServiceEnvVars } from "../services/service.service";
@@ -64,10 +74,15 @@ function signHs256Jwt(secret: string, role: string): string {
* Catalog for the Create-App UI. Only operator-supplied config fields are
* returned as form inputs — `generate:"secret"` fields are filled server-side and
* never surfaced.
*
* `unlisted` apps are dropped here and only here: they stay installable by id and
* their wizard still resolves (`catalogEntry` → `getTemplateForOrg`), they just
* don't get a card. That's how webmail rides along under Openship Mail instead of
* sitting beside it as a second, near-identical tile.
*/
export async function getAppCatalog(ctx: RequestContext) {
const custom = await listOrgCustomApps(ctx.organizationId);
return [...getRuntimeCatalog(), ...custom].map((t) => ({
return [...getRuntimeCatalog(), ...custom].filter((t) => !t.unlisted).map((t) => ({
id: t.id,
name: t.name,
description: t.description,
@@ -80,6 +95,11 @@ export async function getAppCatalog(ctx: RequestContext) {
management: getAppManagement(t),
// Verified trust mark (official open-source image + reviewed pipeline).
verified: !!t.verified,
// Hosting model for the catalog badge + wizard notice (self-hosted default).
hosting: t.hosting ?? "self-hosted",
// What the app needs from the machine — the wizard shows it against the
// chosen destination's real capacity, and deploy preflight enforces it.
minResources: t.minResources,
// A per-org user-uploaded app — always unverified; dashboard shows the warning.
custom: !!t.custom,
// Not installable this version → dashboard dims it + blocks the click.
@@ -105,6 +125,71 @@ export async function getAppCatalog(ctx: RequestContext) {
}));
}
/** One app's declared minimum vs. what a chosen destination actually has. */
export interface AppHostFitView {
/** What the app says it needs. Null when it declares nothing (most apps). */
minResources: AppMinResources | null;
/** What the machine reported. `source: "unknown"` = we couldn't ask, which is
* never a refusal — see `fitsCapacity`. */
capacity: HostCapacity;
fit: ResourceFit;
}
/**
* Match an app's declared `minResources` against a destination BEFORE anything is
* created, so the install wizard can say "PostHog wants 8 GB; this server has 4"
* next to the picker instead of letting the operator find out from a failed
* deploy.
*
* ADVISORY ONLY. The gate is deploy preflight's `host-capacity` check, reading the
* same declaration through the same `fitsCapacity` verdict on the same probed
* numbers — so the notice and the refusal cannot disagree. Same split as the
* free-domain cloud requirement: the wizard pre-checks, the API enforces.
*/
export async function getAppHostFit(
ctx: RequestContext,
templateId: string,
/** The destination as the wizard has it: cloud, or a server row (none = this box,
* which is what an unbound project derives). "This machine" is NOT taken from
* here — see below. */
target: { deployTarget?: string; serverId?: string },
): Promise<AppHostFitView> {
const template = await getTemplateForOrg(ctx.organizationId, templateId);
const minResources = template?.minResources ?? null;
const unchecked: AppHostFitView = {
minResources,
capacity: { ...UNKNOWN_CAPACITY },
fit: { ok: true },
};
// Nothing declared → nothing to match. Cloud is sized from the tier table, not
// from host hardware, so there is no machine to compare against either.
if (!hasMinResources(minResources) || target.deployTarget === "cloud") return unchecked;
// A serverId off a query string is read ORG-SCOPED, and an id this org doesn't
// own reports "unknown" rather than probing another tenant's box.
//
// Whether the destination is THIS machine is then DERIVED from that row, never
// read off the query: `isLocalTarget` is what makes a `source: "local"` probe —
// the API host's own `os.*` — trusted, so a caller claiming it for a remote
// server would have matched the app against the orchestrator's RAM and reported
// a shortfall about the wrong machine. `isLocalHostRow` is the same test the
// deploy path uses, so the notice and the refusal describe one box.
let isLocalTarget = !target.serverId && !env.CLOUD_MODE;
if (target.serverId) {
const server = await repos.server
.getInOrganization(target.serverId, ctx.organizationId)
.catch(() => null);
if (!server) return unchecked;
isLocalTarget = await isLocalHostRow(server);
}
const capacity = await getTrustedHostCapacity(target.serverId, ctx.organizationId, {
isLocalTarget,
});
return { minResources, capacity, fit: fitsCapacity(minResources, capacity) };
}
/**
* This org's not-yet-deployed draft of an app, if it has one.
*
@@ -272,6 +357,27 @@ export function planInstallRouting(
return plan;
}
/**
* One plan entry as an `updateService` patch.
*
* Exists because the array is not optional. A patch carrying only `domainType` loses to
* the stored `publicEndpoints`, which is how a corrected custom domain came back as the
* old free route — so every writer has to send the FULL array, and a rule that every
* writer has to remember belongs in one place instead. Both the install path and the
* webmail re-apply path spelled this out separately.
*/
export function serviceRoutingPatch(routing: {
exposed: boolean;
publicEndpoints: PlannedEndpoint[];
}): { exposed: boolean; publicEndpoints: PlannedEndpoint[]; domainType?: "free" | "custom" } {
return {
exposed: routing.exposed,
publicEndpoints: routing.publicEndpoints,
// The scalar column mirrors entry[0] — the template's primary route.
...(routing.publicEndpoints[0] ? { domainType: routing.publicEndpoints[0].domainType } : {}),
};
}
export type InstallAppResult =
| { kind: "flow"; flowHref: string }
| { kind: "template"; projectId: string; slug: string };
@@ -413,6 +519,20 @@ export async function installApp(
filesByService.set(f.service, list);
}
// Resolve a service's inline build context, if any — `{{config:KEY}}` is
// inlined in the Dockerfile and every context file's content (the same
// generated-key surface as env/files). Carried onto `advanced.build`; the
// deploy pipeline materializes it and builds on the host.
const resolveBuild = (b: TemplateServiceBuild | undefined) =>
b
? {
dockerfile: inlineConfig(b.dockerfile),
...(b.files?.length
? { files: b.files.map((f) => ({ path: f.path, content: inlineConfig(f.content) })) }
: {}),
}
: undefined;
// `secretEnv` declares which of a service's env keys are secrets — stored
// encrypted, never written as plaintext compose env. Wired here (the field was
// previously inert): a listed key sourced from `environment` is re-routed into
@@ -445,15 +565,7 @@ export async function installApp(
// No `routes` in the request = no decision expressed; leave the draft's
// stored routing alone rather than silently unrouting it.
if ((input.routes ?? []).length > 0) {
await updateService(ctx, project.id, existingRow.id, {
exposed: routing.exposed,
// Always the FULL array: a scalar-only patch loses to the stored array,
// which is how a corrected custom domain came back as the old free route.
publicEndpoints: routing.publicEndpoints,
...(routing.publicEndpoints[0]
? { domainType: routing.publicEndpoints[0].domainType }
: {}),
});
await updateService(ctx, project.id, existingRow.id, serviceRoutingPatch(routing));
}
continue;
}
@@ -487,6 +599,7 @@ export async function installApp(
...(filesByService.get(svc.name)?.length
? { files: filesByService.get(svc.name) }
: {}),
...(svc.build ? { build: resolveBuild(svc.build) } : {}),
},
// Routing is exactly what the operator chose — never the template's
// `exposed` flag turned into a hostname.
@@ -8,6 +8,7 @@ import { getRequestContext } from "../../lib/request-context";
import { param } from "../../lib/controller-helpers";
import {
getAppCatalog,
getAppHostFit,
installApp,
findOpenAppDraft,
type InstallAppRoute,
@@ -45,6 +46,23 @@ export async function catalogEntry(c: Context) {
return c.json({ data: template, draft: await findOpenAppDraft(ctx, template.id) });
}
/**
* GET /api/apps/catalog/:id/host-fit — does the chosen destination meet what this
* app declares it needs? Advisory: the wizard shows the shortfall next to the
* destination picker, and deploy preflight is what actually refuses. Query:
* `deployTarget` (server|cloud) and `serverId`. There is no "local": whether the
* destination is this box is derived from the server row, not claimed by the caller.
*/
export async function hostFit(c: Context) {
const ctx = getRequestContext(c);
return c.json({
data: await getAppHostFit(ctx, param(c, "id"), {
deployTarget: c.req.query("deployTarget") || undefined,
serverId: c.req.query("serverId") || undefined,
}),
});
}
/** POST /api/apps/custom — validate + store an uploaded app JSON as a per-org
* (unverified) custom app. Returns its id; then it appears in the catalog. */
export async function addCustom(c: Context) {
+11
View File
@@ -25,6 +25,17 @@ r.get(
{ tag: "project:list", mcp: { description: "Get one app's full template (services, config, endpoints) by id." } },
ctrl.catalogEntry,
);
r.get(
"/catalog/:id/host-fit",
{
tag: "project:list",
mcp: {
description:
"Check whether a destination meets an app's declared minimum resources, before installing. Query: deployTarget, serverId.",
},
},
ctrl.hostFit,
);
r.get(
"/custom",
{ tag: "project:list", mcp: { description: "List this org's custom (user-uploaded, unverified) apps." } },
+5 -5
View File
@@ -45,8 +45,6 @@ export type ResolvedAppTemplate = AppTemplate & {
requiresUpdate?: { minVersion?: string };
/** A newer (engine-gated) version exists in the overlay; the bundled copy is served. */
updateAvailable?: boolean;
/** A per-org user-uploaded app — always unverified (provenance-based trust). */
custom?: boolean;
};
/** This instance's Openship version, for the `minEngine` gate.
@@ -220,10 +218,12 @@ export function getRuntimeTemplate(id: string | null | undefined): ResolvedAppTe
}
/** An org's custom (user-uploaded) apps as catalog entries — always unverified
* (provenance-based trust; the stored `verified` is ignored). */
* (provenance-based trust; the stored `verified` is ignored), and always listed:
* the grid is a custom app's only entry point, so an authored `unlisted` would
* make it unreachable. */
export async function listOrgCustomApps(organizationId: string): Promise<ResolvedAppTemplate[]> {
const rows = await repos.customAppTemplate.listByOrg(organizationId);
return rows.map((r) => ({ ...r.template, verified: false, custom: true }));
return rows.map((r) => ({ ...r.template, verified: false, unlisted: false, custom: true }));
}
/** Resolve a template by id FOR AN ORG: the curated catalog first, else the
@@ -237,7 +237,7 @@ export async function getTemplateForOrg(
const curated = getRuntimeTemplate(id);
if (curated) return curated;
const custom = await repos.customAppTemplate.findByAppId(organizationId, id);
return custom ? { ...custom.template, verified: false, custom: true } : undefined;
return custom ? { ...custom.template, verified: false, unlisted: false, custom: true } : undefined;
}
// Warm the overlay at boot so instances pick up repo changes promptly. Skipped
@@ -13,32 +13,28 @@
import { repos, type BackupDestination } from "@repo/db";
import { type DestinationKind, type BackupDestinationRow } from "@repo/adapters";
import crypto from "node:crypto";
import { assertLocalEndpointInRoot } from "./local-path";
import { encryptSecretField } from "../../lib/credential-encryption";
import { assertResourceInOrg } from "../../lib/controller-helpers";
import type { RequestContext } from "../../lib/request-context";
import { env } from "../../config/env";
import { assertPublicUrl, assertPublicHost } from "../../lib/ssrf-guard";
import { toAdapterRow, hydrateServerAdapterRow } from "./hydrate-server";
import { assertLocalDestinationAllowed } from "./local-gate";
import { safeErrorMessage, type ConnectivityCode } from "@repo/core";
import { runConnectivityCheck } from "../../lib/connectivity";
import "../../lib/connectivity-checks"; // registers the backup-destination check
/**
* Gate + sandbox a local destination endpoint. The env gates live here; the
* path rules live in local-path.ts (see it for why the deny list applies to
* the root rather than the endpoint).
* Gate + sandbox a local destination endpoint at WRITE time, so the operator gets
* the refusal while they're still editing rather than at the next backup run.
*
* The policy itself lives in ./local-gate.ts and is enforced again on every path
* that USES a destination (`toAdapterRow`) — this call is the early, friendly copy
* of that check, not the authority. Keeping one implementation is the point: the
* two used to be able to disagree, and only this one existed.
*/
async function validateLocalEndpoint(endpoint: string): Promise<void> {
if (env.CLOUD_MODE) {
throw new Error("Local destinations are disabled in cloud mode");
}
if (!env.BACKUP_ALLOW_LOCAL_DESTINATION) {
throw new Error(
"Local destinations are disabled. Set BACKUP_ALLOW_LOCAL_DESTINATION=true and BACKUP_LOCAL_ROOT to enable.",
);
}
await assertLocalEndpointInRoot(endpoint, env.BACKUP_LOCAL_ROOT);
await assertLocalDestinationAllowed(endpoint);
}
// ─── Public shapes ───────────────────────────────────────────────────────────
@@ -26,6 +26,7 @@ import type { BackupDestinationRow } from "@repo/adapters";
import { encryptSecretField } from "../../lib/credential-encryption";
import { resolveSafeSshKeyPath } from "../../lib/ssh-key-path";
import { safeErrorMessage } from "@repo/core";
import { assertLocalDestinationAllowed } from "./local-gate";
/**
* Take a raw backup_destination DB row and produce a BackupDestinationRow
@@ -33,6 +34,13 @@ import { safeErrorMessage } from "@repo/core";
* `openship_server` is the special case that needs server-table lookup.
*/
export async function toAdapterRow(row: BackupDestination): Promise<BackupDestinationRow> {
// The consumer-side gate. Every path that USES a destination funnels through here
// (a backup run, the retention prune, both restore phases), so this is the one
// place that can speak for all of them — the create/update gate only ever
// constrained rows that took the write path. See ./local-gate.ts.
if (row.kind === "local") {
await assertLocalDestinationAllowed(row.endpoint);
}
if (row.kind === "openship_server") {
if (!row.serverId) {
throw new Error(
@@ -83,7 +91,13 @@ export async function hydrateServerAdapterRow(params: {
serverId: string;
}): Promise<BackupDestinationRow> {
const { id, organizationId, name, pathPrefix, serverId } = params;
const server = await repos.server.get(serverId);
// ORG-SCOPED: this function goes on to read the server's SSH password or private
// KEY MATERIAL and hand it to the SFTP adapter. An unscoped lookup would let a
// destination row whose serverId names another org's box borrow that box's
// credentials — and `backup_destination` is organization-scope dumpable, so a row
// can arrive by ingest without ever passing the create-time checks. A foreign id
// must read as "no longer exists", which is also what a deleted one reads as.
const server = await repos.server.getInOrganization(serverId, organizationId);
if (!server) {
throw new Error(
`Server ${serverId} referenced by destination "${name}" no longer exists`,
@@ -96,33 +110,40 @@ export async function hydrateServerAdapterRow(params: {
if (server.sshAuthMethod === "password" && server.sshPassword) {
sftpPasswordEnc = server.sshPassword;
} else if (server.sshAuthMethod === "key" && server.sshKeyPath) {
// Centralised allowlist + traversal check — see lib/ssh-key-path.ts.
// homedir() is added as an extra root so an operator's
// ~/.ssh/openship key works without explicit env configuration.
let keyPath: string;
try {
keyPath = resolveSafeSshKeyPath(server.sshKeyPath, {
extraRoots: [homedir()],
});
} catch (err) {
throw new Error(
`Server ${server.id} sshKeyPath rejected: ${
safeErrorMessage(err)
}`,
);
} else if (server.sshAuthMethod === "key" && (server.sshPrivateKey || server.sshKeyPath)) {
if (server.sshPrivateKey) {
// Pasted/uploaded material — already `enc1:`-encrypted at rest, so hand it
// straight to the SFTP adapter (same passthrough as the password branch).
// No host filesystem to read, so buildSshConfig's path allowlist is moot.
sftpPrivateKeyEnc = server.sshPrivateKey;
} else {
// Key on THIS host's filesystem. Centralised allowlist + traversal check —
// see lib/ssh-key-path.ts. homedir() is added as an extra root so an
// operator's ~/.ssh/openship key works without explicit env configuration.
let keyPath: string;
try {
keyPath = resolveSafeSshKeyPath(server.sshKeyPath!, {
extraRoots: [homedir()],
});
} catch (err) {
throw new Error(
`Server ${server.id} sshKeyPath rejected: ${
safeErrorMessage(err)
}`,
);
}
let keyMaterial: string;
try {
keyMaterial = await readFile(keyPath, "utf-8");
} catch (err) {
throw new Error(
`Failed to read SSH key at ${keyPath} for server ${server.id}: ${
safeErrorMessage(err)
}`,
);
}
sftpPrivateKeyEnc = encryptSecretField(keyMaterial);
}
let keyMaterial: string;
try {
keyMaterial = await readFile(keyPath, "utf-8");
} catch (err) {
throw new Error(
`Failed to read SSH key at ${keyPath} for server ${server.id}: ${
safeErrorMessage(err)
}`,
);
}
sftpPrivateKeyEnc = encryptSecretField(keyMaterial);
if (server.sshKeyPassphrase) {
sftpKeyPassphraseEnc = server.sshKeyPassphrase;
}
@@ -0,0 +1,56 @@
/**
* THE gate for `kind: 'local'` backup destinations — asked on every path that is
* about to USE one, not just the paths that create one.
*
* Why it lives here rather than in the service: the policy used to be enforced
* only where a destination row is WRITTEN (create / update / draft preflight),
* while the three paths that CONSUME a row — a backup run, the retention prune,
* and a restore — went straight to `toAdapterRow`. The adapter is deliberately
* env-blind (see the docblock on `packages/adapters/src/backup/destinations/local.ts`,
* which states the gate is "enforced at the apps/api layer"), so nothing between
* a row and `fs.unlink`/`fs.rename` re-asked the question.
*
* That mattered because a row can arrive without passing the write path at all:
* `backup_destination` is organization-scope dumpable, so an ingested row lands
* complete, and `retention-prune-daily` runs on every instance. A local row plus
* attacker-chosen artifact keys is then a filesystem write/delete on the control
* plane.
*
* The rule this encodes: gate the CONSUMER, not the writer. A gate on the write
* path only constrains rows that took it.
*
* Kept out of `local-path.ts` on purpose — that module's contract is "is this path
* allowed", with the env gates explicitly excluded — and out of `destination.service.ts`
* to avoid an import cycle with `hydrate-server.ts`.
*/
import { env } from "../../config";
import { assertLocalEndpointInRoot } from "./local-path";
/**
* Throw unless a local destination may be used on this instance, and unless its
* endpoint is inside the configured sandbox root.
*
* Both halves are checked every time. The endpoint is re-validated (not just the
* policy flags) because a row that never went through the write path never had its
* path checked either, and containment is what keeps a key from resolving outside
* `BACKUP_LOCAL_ROOT`.
*/
export async function assertLocalDestinationAllowed(
endpoint: string | null | undefined,
): Promise<void> {
if (env.CLOUD_MODE) {
throw new Error("Local destinations are disabled in cloud mode");
}
if (!env.BACKUP_ALLOW_LOCAL_DESTINATION) {
throw new Error(
"Local destinations are disabled. Set BACKUP_ALLOW_LOCAL_DESTINATION=true and BACKUP_LOCAL_ROOT to enable.",
);
}
if (!endpoint) {
// A local row with no endpoint would resolve against the process CWD in the
// adapter. Refuse rather than guess.
throw new Error("Local destination has no endpoint — refusing to resolve a backup path");
}
await assertLocalEndpointInRoot(endpoint, env.BACKUP_LOCAL_ROOT);
}
+14 -4
View File
@@ -94,10 +94,20 @@ one of them came to be the one that didn't.
A run whose policy was deleted (`SET NULL`, so history outlives the schedule) is
unrecoverable; those ids are logged every boot rather than swallowed, because the
alternative is the operator discovering it at the one moment it matters.
- **Mail policy retention still stores null-on-omission** (`mail.controller.ts`,
~1339, comment-flagged there). It's inert today because `prunePolicy` skips
mail-server policies outright, but omitted and explicit-null have to be told apart
before mail retention can run.
- ~~**Mail policy retention still stores null-on-omission**~~ — **done.**
`saveMailBackupPolicy` now draws the same omitted-vs-explicit-null distinction
`createPolicy` does (omitted keeps the stored value, or defaults ON when creating),
and `prunePolicy` prunes mail policies for real: it pages runs by `mailServerId`
and reads the org off the mail server's row (`policyOrganizationId`), instead of
bailing out on `!projectId`. Also from that pass: a non-positive `retainCount`
normalizes to "unset" inside `prunePolicy` rather than putting every run outside
the keep-set, and `saveMailBackupPolicy` calls `syncPolicySchedule` so a schedule
change registers its job immediately instead of at the next boot reconcile.
Existing installs still carry mail policies written before this, whose retention
columns are both NULL and therefore read as "unlimited" — the same ambiguity 0096
fixed for project policies. **Not** backfilled here: unlike 0096, that would be a
behavior change (start deleting backups) on rows an operator may have left alone
deliberately, so it wants an explicit call rather than a migration.
## Retention, and what NULL means now
@@ -19,6 +19,7 @@
import {
repos,
type Project,
type Service,
type BackupRunStatus,
type BackupPolicy,
@@ -43,12 +44,17 @@ import {
type BackupTrigger,
type PayloadKind,
type ProducerOpts,
type RuntimeAdapter,
type ServiceHandle,
} from "@repo/adapters";
import { Readable } from "node:stream";
import { resolveDeploymentPlatform, resolveTargetPlatform } from "../../lib/deployment-runtime";
import { decryptEnvMap } from "../../lib/encryption";
import {
disposeRuntime,
resolveDeploymentPlatform,
resolveTargetPlatform,
} from "../../lib/deployment-runtime";
import { notification } from "../../lib/notification-dispatcher";
import { serviceHandleFor } from "./service-handle";
import crypto from "node:crypto";
import { safeErrorMessage } from "@repo/core";
import {
@@ -245,6 +251,11 @@ export class BackupOrchestrator {
let policy = null as Awaited<ReturnType<typeof repos.backupPolicy.findById>> | null;
let executor: BackupExecutor | null = null;
let serviceHandle: ServiceHandle | null = null;
// The runtime the BackupExecutor wraps. Held for the whole run (it shells into
// the container to produce the dump) and released in the `finally` — on a
// remote server it carries a Docker-over-SSH loopback bridge that only
// `dispose()` closes, so a scheduled policy would otherwise strand one per run.
let sourceRuntime: RuntimeAdapter | null = null;
try {
await this.transition(runId, "preparing");
@@ -316,6 +327,7 @@ export class BackupOrchestrator {
(activeDeployment?.meta ?? {}) as Parameters<typeof resolveDeploymentPlatform>[0],
{ organizationId: destinationRow.organizationId },
);
sourceRuntime = platform.platform.runtime;
executor = resolveExecutor(platform.platform.runtime.name, platform.platform.runtime);
ctx = {
@@ -502,6 +514,8 @@ export class BackupOrchestrator {
});
}
}
} finally {
disposeRuntime(sourceRuntime);
}
}
@@ -718,43 +732,20 @@ export class BackupOrchestrator {
const project = await repos.project.findById(serviceRow.projectId);
if (!project) throw new Error(`Project ${serviceRow.projectId} not found`);
// Decrypt env vars at the boundary so producers can use them
// (pg_dump -U $POSTGRES_USER etc.). Two sources:
// service.environment — plaintext defaults from compose
// env_var rows — encrypted per-key (user-set)
// Project env wins over service defaults.
const envFromService =
(serviceRow.environment as Record<string, string> | null) ?? {};
const envFromProjectEncrypted = await repos.project
.listEnvVars(serviceRow.projectId)
.then((vars) => {
const out: Record<string, string> = {};
for (const v of vars) out[v.key] = v.value;
return out;
})
.catch(() => ({}));
const projectEnv = decryptEnvMap(envFromProjectEncrypted);
const decrypted = { ...envFromService, ...projectEnv };
return {
id: serviceRow.id,
projectId: serviceRow.projectId,
name: serviceRow.name,
image: serviceRow.image,
env: decrypted,
volumes: (serviceRow.volumes as string[] | null) ?? [],
containerId: await this.resolveServiceContainerId(serviceRow),
return serviceHandleFor(serviceRow, {
projectSlug: project.slug,
namespaceVolumes: serviceRow.namespaceVolumes,
};
containerId: await this.resolveServiceContainerId(project, serviceRow),
});
}
/** Find the live container id for a service, via the shared resolver —
* verified against the host, so a backup never targets a container a
* redeploy already replaced. */
private async resolveServiceContainerId(serviceRow: Service): Promise<string | null> {
const project = await repos.project.findById(serviceRow.projectId);
if (!project?.activeDeploymentId) return null;
private async resolveServiceContainerId(
project: Project,
serviceRow: Service,
): Promise<string | null> {
if (!project.activeDeploymentId) return null;
const dep = await repos.deployment.findById(project.activeDeploymentId);
if (!dep) return null;
return liveContainerIdForService(project, dep, serviceRow, { projectId: project.id });
@@ -50,11 +50,16 @@ import {
type BackupExecutor,
type BackupTrigger,
type PayloadKind,
type RuntimeAdapter,
type ServiceHandle,
} from "@repo/adapters";
import { staticReleaseDir, usableRef } from "../deployments/rollback/restore-plan";
import { decryptEnvMap } from "../../lib/encryption";
import { resolveDeploymentPlatform, resolveTargetPlatform } from "../../lib/deployment-runtime";
import { serviceHandleFor } from "./service-handle";
import {
disposeRuntime,
resolveDeploymentPlatform,
resolveTargetPlatform,
} from "../../lib/deployment-runtime";
import { safeErrorMessage } from "@repo/core";
import { assertResourceInOrg } from "../../lib/controller-helpers";
import type { RequestContext } from "../../lib/request-context";
@@ -540,11 +545,14 @@ export class RestoreOrchestrator {
destinationRow: { organizationId: string },
artifacts: RecordedArtifact[],
): Promise<{ meta: Record<string, unknown> }> {
const { executor, serviceHandle } = await this.resolveTarget(
const { executor, serviceHandle, runtime } = await this.resolveTarget(
restore,
sourceRun,
destinationRow,
);
// A prepare-time PROBE only — nothing here outlives the method, so the
// transport goes back when it returns (including on a throw).
try {
// A probe that FAILED is not the same fact as a target with nothing in it,
// and conflating them would report an unreachable host as a misconfigured
@@ -620,6 +628,9 @@ export class RestoreOrchestrator {
targetSources: restorable.length,
},
};
} finally {
disposeRuntime(runtime);
}
}
/**
@@ -823,6 +834,8 @@ export class RestoreOrchestrator {
// first millisecond of apply still finds a handle to abort.
const controller = new AbortController();
this.inFlight.set(restoreId, controller);
// Released alongside the in-flight handle at the bottom — see resolveTarget.
let targetRuntime: RuntimeAdapter | null = null;
try {
const restore = await repos.backupRestore.findById(restoreId);
if (!restore) return;
@@ -838,11 +851,14 @@ export class RestoreOrchestrator {
const adapterRow = await toAdapterRow(destinationRow);
const destination = resolveDestination(adapterRow);
const { executor, serviceHandle } = await this.resolveTarget(
const { executor, serviceHandle, runtime } = await this.resolveTarget(
restore,
sourceRun,
destinationRow,
);
// Used through the whole restore (the executor shells into the target), so
// it is released in this method's `finally`, not here.
targetRuntime = runtime;
// Checkpoint 1 — before anything is stopped. Free.
await this.throwIfCancelRequested(restoreId, null);
@@ -997,6 +1013,7 @@ export class RestoreOrchestrator {
});
} finally {
this.inFlight.delete(restoreId);
disposeRuntime(targetRuntime);
}
}
@@ -1131,11 +1148,22 @@ export class RestoreOrchestrator {
* disagree about the target: a preflight that passed against a different
* resolution than apply uses would be worse than no preflight.
*/
/**
* `runtime` is returned so the CALLER can release it: the BackupExecutor wraps
* it and shells into the target for the whole restore, and on a remote server it
* carries a Docker-over-SSH loopback bridge that only `dispose()` closes. Both
* call sites hand it to `disposeRuntime` when they're done (the mail branch's
* bare runtime has nothing to release — a deliberate no-op).
*/
private async resolveTarget(
restore: BackupRestore,
sourceRun: BackupRun,
destinationRow: { organizationId: string },
): Promise<{ executor: BackupExecutor; serviceHandle: ServiceHandle }> {
): Promise<{
executor: BackupExecutor;
serviceHandle: ServiceHandle;
runtime: RuntimeAdapter | null;
}> {
if (sourceRun.sourceKind === "mail_server") {
const targetMailServerId =
restore.mode === "to_fork" ? restore.forkMailServerId : sourceRun.mailServerId;
@@ -1143,7 +1171,7 @@ export class RestoreOrchestrator {
throw new Error("Mail restore has no target mail server");
}
const built = await this.buildMailTarget(targetMailServerId, destinationRow.organizationId);
return { executor: built.executor, serviceHandle: built.handle };
return { executor: built.executor, serviceHandle: built.handle, runtime: null };
}
if (!sourceRun.serviceId) throw new Error("Source run has no serviceId");
@@ -1160,6 +1188,7 @@ export class RestoreOrchestrator {
return {
executor: resolveExecutor(platform.platform.runtime.name, platform.platform.runtime),
serviceHandle: await this.buildServiceHandle(serviceRow),
runtime: platform.platform.runtime,
};
}
@@ -1248,19 +1277,6 @@ export class RestoreOrchestrator {
const project = await repos.project.findById(serviceRow.projectId);
if (!project) throw new Error(`Project ${serviceRow.projectId} not found`);
const envFromService =
(serviceRow.environment as Record<string, string> | null) ?? {};
const envFromProjectEncrypted = await repos.project
.listEnvVars(serviceRow.projectId)
.then((vars) => {
const out: Record<string, string> = {};
for (const v of vars) out[v.key] = v.value;
return out;
})
.catch(() => ({}));
const projectEnv = decryptEnvMap(envFromProjectEncrypted);
const decrypted = { ...envFromService, ...projectEnv };
let containerId: string | null = null;
if (project.activeDeploymentId) {
const dep = await repos.deployment.findById(project.activeDeploymentId);
@@ -1274,17 +1290,7 @@ export class RestoreOrchestrator {
}
}
return {
id: serviceRow.id,
projectId: serviceRow.projectId,
name: serviceRow.name,
image: serviceRow.image,
env: decrypted,
volumes: (serviceRow.volumes as string[] | null) ?? [],
containerId,
projectSlug: project.slug,
namespaceVolumes: serviceRow.namespaceVolumes,
};
return serviceHandleFor(serviceRow, { projectSlug: project.slug, containerId });
}
}
@@ -0,0 +1,203 @@
import { describe, it, expect, beforeEach, vi } from "vitest";
/**
* Retention for mail-server policies.
*
* `prunePolicy` used to return early on any policy without a `projectId`, so the
* one backup source that can carry every stored message on a server was also the
* one whose runs accumulated forever. These pin the contract now that it doesn't:
*
* • a mail policy's runs page by `mailServerId`, never by project (a project
* filter on a null projectId is what made this un-implementable before)
* • the org comes off the mail server's row, so a deleted server is a named
* skip rather than an unscoped read across the whole table
* • artifacts + manifest go to the destination's deleteMany, same as a project
* • `retentionLockedUntil` ("Protect this backup") still wins
* • a negative retainCount is treated as unset, not as "delete everything"
*/
const h = vi.hoisted(() => ({
runs: [] as Array<Record<string, unknown>>,
listCalls: [] as Array<Record<string, unknown>>,
softDeleted: [] as string[],
deletedKeys: [] as string[][],
mailServer: { organizationId: "org1" } as { organizationId: string } | undefined,
project: undefined as { organizationId: string } | undefined,
destination: { id: "dest1", kind: "s3" } as Record<string, unknown> | undefined,
}));
vi.mock("@repo/db", () => ({
repos: {
project: { findById: vi.fn(async () => h.project) },
server: { get: vi.fn(async () => h.mailServer) },
backupPolicy: {
iterateEnabledForRetention: async function* () {},
},
backupRun: {
listByOrganization: vi.fn(
async (organizationId: string, opts: Record<string, unknown>) => {
h.listCalls.push({ organizationId, ...opts });
// One page, then empty — the caller pages until short.
return (opts.offset as number) === 0 ? h.runs : [];
},
),
softDelete: vi.fn(async (id: string) => {
h.softDeleted.push(id);
}),
},
backupDestination: { findById: vi.fn(async () => h.destination) },
},
}));
vi.mock("@repo/adapters", () => ({
// Read at module scope by openship-server-store (imported transitively), so the
// mock has to carry it or nothing under test loads. Its value is irrelevant here.
HOST_STATE_DIR: "/root/.openship",
resolveDestination: () => ({
deleteMany: vi.fn(async (keys: string[]) => {
h.deletedKeys.push(keys);
}),
}),
}));
vi.mock("../backup-destinations/hydrate-server", () => ({
toAdapterRow: vi.fn(async (row: unknown) => row),
}));
const { prunePolicy } = await import("./retention-prune");
type PolicyArg = Parameters<typeof prunePolicy>[0];
const mailPolicy = (over: Record<string, unknown> = {}): PolicyArg =>
({
id: "bkp_mail",
sourceKind: "mail_server",
projectId: null,
serviceId: null,
mailServerId: "srv_mail",
destinationId: "dest1",
enabled: true,
retainCount: 2,
retainDays: null,
...over,
}) as unknown as PolicyArg;
/** A succeeded mail run, `ageDays` old. */
const run = (id: string, ageDays: number, over: Record<string, unknown> = {}) => ({
id,
policyId: "bkp_mail",
destinationId: "dest1",
projectId: null,
serviceId: null,
mailServerId: "srv_mail",
status: "succeeded",
deletedAt: null,
retentionLockedUntil: null,
finishedAt: new Date(Date.now() - ageDays * 86_400_000),
artifacts: [{ key: `mail/${id}.tar.zst` }],
manifestKey: `mail/${id}.json`,
...over,
});
beforeEach(() => {
h.runs = [];
h.listCalls = [];
h.softDeleted = [];
h.deletedKeys = [];
h.mailServer = { organizationId: "org1" };
h.project = undefined;
h.destination = { id: "dest1", kind: "s3" };
});
describe("prunePolicy — mail-server policies", () => {
it("pages runs by mailServerId under the mail server's org", async () => {
h.runs = [run("r1", 1)];
await prunePolicy(mailPolicy({ retainCount: 5 }));
expect(h.listCalls[0]).toMatchObject({
organizationId: "org1",
mailServerId: "srv_mail",
});
// A project filter here would match nothing (projectId is null) — that read
// returning zero rows is exactly what "not implemented" looked like.
expect(h.listCalls[0]).not.toHaveProperty("projectId");
});
it("drops the runs outside retainCount, newest kept", async () => {
h.runs = [run("newest", 1), run("middle", 5), run("oldest", 9)];
const result = await prunePolicy(mailPolicy({ retainCount: 2 }));
expect(result).toEqual({ dropped: 1, skipped: null });
expect(h.softDeleted).toEqual(["oldest"]);
// Artifacts AND the manifest leave the destination, or the bytes outlive the row.
expect(h.deletedKeys).toEqual([["mail/oldest.tar.zst", "mail/oldest.json"]]);
});
it("drops runs past retainDays", async () => {
h.runs = [run("fresh", 2), run("stale", 40)];
const result = await prunePolicy(mailPolicy({ retainCount: null, retainDays: 30 }));
expect(result.dropped).toBe(1);
expect(h.softDeleted).toEqual(["stale"]);
});
it("never touches a protected run", async () => {
h.runs = [
run("newest", 1),
run("locked", 40, { retentionLockedUntil: new Date(Date.now() + 86_400_000) }),
];
const result = await prunePolicy(mailPolicy({ retainCount: 1, retainDays: 30 }));
expect(result).toEqual({ dropped: 0, skipped: null });
expect(h.softDeleted).toEqual([]);
});
it("ignores other policies' runs on the same destination", async () => {
h.runs = [run("mine", 9), run("theirs", 40, { policyId: "bkp_other" })];
await prunePolicy(mailPolicy({ retainCount: 1 }));
expect(h.softDeleted).toEqual([]);
});
it("skips by name when the mail server row is gone", async () => {
h.mailServer = undefined;
h.runs = [run("r1", 40)];
const result = await prunePolicy(mailPolicy({ retainCount: 1 }));
// No org means no scoped read is even possible; deleting on a guess would be
// a cross-tenant delete.
expect(result).toEqual({ dropped: 0, skipped: "mail server row is gone" });
expect(h.listCalls).toEqual([]);
expect(h.softDeleted).toEqual([]);
});
it("treats a non-positive retainCount as unlimited, not as delete-everything", async () => {
h.runs = [run("a", 1), run("b", 2), run("c", 3)];
const result = await prunePolicy(mailPolicy({ retainCount: -1 }));
expect(result).toEqual({ dropped: 0, skipped: "retention set to unlimited" });
expect(h.softDeleted).toEqual([]);
});
it("keeps everything when both dimensions are null", async () => {
h.runs = [run("a", 400)];
const result = await prunePolicy(mailPolicy({ retainCount: null, retainDays: null }));
expect(result).toEqual({ dropped: 0, skipped: "retention set to unlimited" });
});
it("still prunes project policies by projectId", async () => {
h.project = { organizationId: "org2" };
h.runs = [
run("p_new", 1, { projectId: "prj1", mailServerId: null, policyId: "bkp_prj" }),
run("p_old", 9, { projectId: "prj1", mailServerId: null, policyId: "bkp_prj" }),
];
const result = await prunePolicy(
mailPolicy({ id: "bkp_prj", projectId: "prj1", mailServerId: null, retainCount: 1 }),
);
expect(h.listCalls[0]).toMatchObject({ organizationId: "org2", projectId: "prj1" });
expect(result.dropped).toBe(1);
expect(h.softDeleted).toEqual(["p_old"]);
});
});
+43 -17
View File
@@ -1,6 +1,11 @@
/**
* Retention prune — runs daily, applies each policy's retention rules.
*
* Source-agnostic: a project/service policy and a mail-server policy differ only
* in which column scopes the run list and where the owning org is read from. Mail
* used to be skipped outright, which meant the one source that backs up entire
* maildirs was also the one with no ceiling.
*
* Two retention dimensions, evaluated independently per policy:
* - `retainCount` — keep at most N most-recent succeeded runs.
* - `retainDays` — drop succeeded runs older than N days.
@@ -23,6 +28,7 @@
import { repos, type BackupPolicy, type BackupRun } from "@repo/db";
import { resolveDestination } from "@repo/adapters";
import { toAdapterRow } from "../backup-destinations/hydrate-server";
import { policyOrganizationId } from "./backup.service";
import { safeErrorMessage } from "@repo/core";
export async function runRetentionSweep(): Promise<{
@@ -66,40 +72,60 @@ type PruneOutcome = { dropped: number; skipped: string | null };
const PRUNE_PAGE_SIZE = 500;
export async function prunePolicy(policy: BackupPolicy): Promise<PruneOutcome> {
const retainCount = policy.retainCount;
// Non-positive retention is not a tighter window, it's a loaded gun:
// `retainCount: -1` puts every run outside the keep-set and deletes the lot.
// Nothing validates the number on the way in, so it's normalized here, where
// the deletes happen. Zero already behaved as unset; negatives now do too.
const retainCount =
policy.retainCount && policy.retainCount > 0 ? policy.retainCount : null;
const retainDays = policy.retainDays && policy.retainDays > 0 ? policy.retainDays : null;
// Both null now means "keep everything", asked for deliberately: omitting
// retention yields `DEFAULT_RETAIN_COUNT` from the column default, and
// migration 0096 backfilled the rows written before it. There is no fallback
// here on purpose — one would override the explicit choice.
if (!retainCount && !policy.retainDays) {
if (!retainCount && !retainDays) {
return { dropped: 0, skipped: "retention set to unlimited" };
}
// Mail-server policies aren't project-scoped; their runs list by project,
// so retention pruning for them is a follow-up. Skip cleanly for now.
if (!policy.projectId) {
return { dropped: 0, skipped: "mail-server policy — retention not implemented" };
}
const destinationId = policy.destinationId;
const project = await repos.project.findById(policy.projectId);
if (!project) {
return { dropped: 0, skipped: "project soft-deleted" };
// Both backup sources page the same run table and delete through the same
// destination adapter; they differ only in which column scopes the page and
// where the owning org is read from. Mail used to bail out here — its runs
// grew without a ceiling on a source whose whole point is message data, which
// is the largest thing openship backs up.
const scope = policy.projectId
? ({ projectId: policy.projectId } as const)
: policy.mailServerId
? ({ mailServerId: policy.mailServerId } as const)
: null;
if (!scope) {
return { dropped: 0, skipped: "policy has neither a project nor a mail server" };
}
const organizationId = await policyOrganizationId(policy);
if (!organizationId) {
// The source row is gone, so the org that owns the runs is unknowable and
// the paged read can't even be issued. Named, not silent — these artifacts
// are now unprunable and someone has to go delete them by hand.
return {
dropped: 0,
skipped: policy.projectId ? "project soft-deleted" : "mail server row is gone",
};
}
// Page through every run for this project. The 1000-run cap was a
// Page through every run for this source. The 1000-run cap was a
// silent data leak: projects past it never had older runs pruned and
// accumulated forever. The page-then-filter pattern below has the
// same memory footprint as the old code in practice (candidates are
// a subset of total) but never silently truncates.
// Candidates = THIS policy's succeeded, unlocked runs. Filtering by policyId
// (not just project+destination) keeps two policies sharing a destination from
// (not just source+destination) keeps two policies sharing a destination from
// co-mingling their retention windows.
const now = new Date();
const candidates: BackupRun[] = [];
for (let offset = 0; ; offset += PRUNE_PAGE_SIZE) {
const page = await repos.backupRun.listByOrganization(project.organizationId, {
projectId: policy.projectId,
const page = await repos.backupRun.listByOrganization(organizationId, {
...scope,
limit: PRUNE_PAGE_SIZE,
offset,
});
@@ -115,8 +141,8 @@ export async function prunePolicy(policy: BackupPolicy): Promise<PruneOutcome> {
if (page.length < PRUNE_PAGE_SIZE) break;
}
const cutoffDate = policy.retainDays
? new Date(Date.now() - policy.retainDays * 24 * 60 * 60 * 1000)
const cutoffDate = retainDays
? new Date(Date.now() - retainDays * 24 * 60 * 60 * 1000)
: null;
// Apply retention PER SERVICE. A project-default policy fans out to N services
@@ -0,0 +1,65 @@
/**
* @module service-handle
*
* The `ServiceHandle` a backup or a restore runs against.
*
* One definition because the two directions have to agree on the env a producer
* sees, and not loosely: `pg_dump -U $POSTGRES_USER` writes the dump and
* `psql -U $POSTGRES_USER` reads it back, so if either side resolved
* `POSTGRES_USER` from a different source — or merged the sources in a different
* order — a restore would authenticate as someone the dump was never taken as.
* Both orchestrators spelled the resolution out separately, with the comment
* explaining the precedence on the backup copy only.
*
* The container id stays with the caller: a backup targets the live container or
* nothing, while a restore is allowed one narrow fallback
* (`deploymentManagedContainerId`). That asymmetry is deliberate, so it is a
* parameter rather than a branch in here.
*/
import { repos, type Service } from "@repo/db";
import type { ServiceHandle } from "@repo/adapters";
import { decryptEnvMap } from "../../lib/encryption";
/**
* Env as a producer will see it, decrypted at this boundary so no producer has to
* hold a key. Two sources:
* service.environment — plaintext defaults from compose
* env_var rows — encrypted per-key (user-set)
* Project env wins over service defaults.
*
* A failed env-var read degrades to the compose defaults rather than aborting —
* the behaviour both call sites already had. It does mean "no user-set variables"
* and "could not read them" produce the same map.
*/
async function resolveServiceEnv(serviceRow: Service): Promise<Record<string, string>> {
const envFromService = (serviceRow.environment as Record<string, string> | null) ?? {};
const envFromProjectEncrypted = await repos.project
.listEnvVars(serviceRow.projectId)
.then((vars) => {
const out: Record<string, string> = {};
for (const v of vars) out[v.key] = v.value;
return out;
})
.catch(() => ({}));
return { ...envFromService, ...decryptEnvMap(envFromProjectEncrypted) };
}
/** The handle for a real service row, given its project slug and an
* already-resolved container id. */
export async function serviceHandleFor(
serviceRow: Service,
target: { projectSlug: string; containerId: string | null },
): Promise<ServiceHandle> {
return {
id: serviceRow.id,
projectId: serviceRow.projectId,
name: serviceRow.name,
image: serviceRow.image,
env: await resolveServiceEnv(serviceRow),
volumes: (serviceRow.volumes as string[] | null) ?? [],
containerId: target.containerId,
projectSlug: target.projectSlug,
namespaceVolumes: serviceRow.namespaceVolumes,
};
}
@@ -402,7 +402,6 @@ export async function mintOrgInstallationToken(
| { kind: "ok"; token: string; expiresAt: string }
| { kind: "not-found"; owner: string }
> {
void repos_;
const { ownerUserId } = await resolveCloudOwnerById(organizationId);
// Resolve installationId from the ORG OWNER's row — the org's GitHub
@@ -422,6 +421,13 @@ export async function mintOrgInstallationToken(
}),
owner,
installation.installationId,
// Honor the caller's repo narrowing. Dropping it here silently widened every
// narrowed mint that proxies through the cloud (`cloud-app` mode, which is the
// canonical self-hosted path once an org is cloud-connected): the caller asked
// for a token scoped to one repo and got an installation-wide one back — the
// "authorized for repo A, credential reaches repo B" shape of
// GHSA-hp2g-hw7g-f3vm, one layer down in the proxy.
{ repositories: repos_ },
)
.catch(() => null);
if (!token) {
@@ -33,6 +33,7 @@ import {
import { runCloudPreflight } from "../../lib/cloud-preflight";
import { cloudRuntimeTarget } from "../../config/env";
import * as githubAuth from "../github/github.auth";
import { canMintInstallationToken } from "../github/github-access";
import {
proxyCloudAnalytics,
CloudAnalyticsForbiddenError,
@@ -1155,13 +1156,37 @@ export async function githubInstallations(c: Context) {
* against github.com for the actual git clone — cloud never sees the
* source code.
*
* SECURITY: `installationId` is intentionally NOT accepted from the
* request body — see service comments.
* SECURITY, two halves:
* - `installationId` is intentionally NOT accepted from the request body, so a
* caller cannot name another org's installation — see service comments.
* - the caller must hold a GitHub grant for what they are asking for. The route
* tag alone does NOT establish this: "cloud" is an org-singleton resource, so a
* plain `member` satisfies `cloud:write`. See the gate in the handler.
*/
export async function githubInstallationToken(c: Context) {
const ctx = getRequestContext(c);
const body = await c.req.json<{ owner?: string; repos?: string[] }>();
if (!body.owner) return c.json({ error: "owner is required" }, 400);
// The repo-grant gate. Without it the route's `cloud:write` tag was the only check,
// and "cloud" is an org SINGLETON resource — so the assert runs with resourceId "*"
// and roleAllowsResourceType lets plain `member` through. Any member could mint a
// live GitHub App installation token, and the grant system that decides WHICH repos
// a member may touch was never consulted. The repos-vs-no-repos distinction is the
// security-relevant part; it lives in canMintInstallationToken.
const requested = (body.repos ?? []).map((r) => r.trim()).filter(Boolean);
if (!(await canMintInstallationToken(ctx, body.owner, body.repos))) {
return c.json(
{
error: requested.length
? `You don't have access to all of the requested repositories under ${body.owner}. Ask an organization owner to grant you access.`
: `You don't have access to every repository under ${body.owner}. Ask an organization owner to grant you access, or request specific repositories instead.`,
code: "GITHUB_ACCESS_DENIED",
},
403,
);
}
const result = await mintOrgInstallationToken(ctx.organizationId, body.owner, body.repos);
if (result.kind === "not-found") {
return c.json({ error: `No GitHub App installation found for ${result.owner}` }, 404);
@@ -1,7 +1,15 @@
import type { LogEntry } from "@repo/adapters";
/**
* Make captured build output STORABLE before it becomes a jsonb value.
* Make captured build output STORABLE — and safe to store — before it becomes a
* jsonb value. Two jobs, both funnelled through `sanitizeLogText`:
*
* 1. STORABILITY (below): Postgres refuses NULs and unpaired surrogates.
* 2. REDACTION (`redactCredentials`): the pipeline embeds a git credential in a
* command it runs, and that command's output is persisted.
*
* Same reason for both: this is the one place the persisted array is built, so no
* call site can forget either.
*
* `build_session.logs` is jsonb, and Postgres refuses a jsonb value that
* contains a NUL ("unsupported Unicode escape sequence") or an unpaired
@@ -31,6 +39,53 @@ export const MAX_ENTRY_CHARS = 32_768;
/** Whole-payload cap. jsonb tops out at 1 GB; this keeps a single row sane. */
export const MAX_TOTAL_CHARS = 4_000_000;
/**
* Credentials embedded in a URL's userinfo. The shape that matters is the one
* `injectGitToken` builds for a private clone:
*
* https://x-access-token:<installation-token>@github.com/owner/repo.git
*
* That URL is interpolated into a shell command whose git output is streamed into
* the persisted build log, so without this the token lands in `build_session.logs`
* — readable by anyone with `deployment:read` long after it was minted, and by
* anyone restoring a backup of that table.
*
* `[^\s/@]` cannot run past the authority, so a path segment containing `@` is
* untouched. The SSH form (`git@github.com:owner/repo`) has no `://` and carries no
* secret, so it is deliberately not matched.
*
* BOTH quantifiers are BOUNDED, and that is not cosmetic. Written as `[a-z0-9+.-]*`
* the scheme prefix backtracks across any long run of scheme-legal characters at
* every start position — quadratic. Build output is exactly where a multi-megabyte
* line of such characters shows up, and this function runs on the persist path, so
* the unbounded form turned a big log into a hang (the oversized-payload case in
* build-log-sanitize.test.ts went from milliseconds to >280s). Real schemes are
* under 16 characters and real userinfo is far under 512.
*/
const URL_USERINFO = /([a-z][a-z0-9+.-]{0,15}:\/\/)[^\s/@]{1,512}@/gi;
/**
* Bare GitHub tokens, for the paths that log a token without a URL around it
* (a `git config` echo, a curl -H line, an error body quoting the header).
* Belt-and-braces on top of URL_USERINFO, not a replacement for it.
*/
const GITHUB_TOKEN = /\b(?:gh[pousr]_[A-Za-z0-9]{16,}|github_pat_[A-Za-z0-9_]{20,})\b/g;
const REDACTED = "***";
/**
* Strip credentials from text on its way into storage.
*
* Deliberately NOT trying to be a general secret scanner — env values are already
* masked upstream by `maskServicesEnv`/`secret-env`, and a greedy matcher here would
* corrupt legitimate build output. This covers the one class the deploy path
* manufactures itself: a credential the pipeline embedded in a command it ran.
*/
export function redactCredentials(text: string): string {
if (!text) return text;
return text.replace(URL_USERINFO, `$1${REDACTED}@`).replace(GITHUB_TOKEN, REDACTED);
}
/**
* Cut to at most `max` UTF-16 code units WITHOUT splitting a surrogate pair.
* `slice` indexes code UNITS, so a cut that lands between the halves of an
@@ -101,7 +156,10 @@ function stripAndRepair(text: string): string {
* what the input was.
*/
export function sanitizeLogText(text: string): string {
const cleaned = stripAndRepair(text);
// Redaction runs FIRST, before the cap. A cut that lands mid-token would leave a
// prefix behind and, worse, would stop the pattern matching at all — so redacting
// after capping would silently do nothing for exactly the longest lines.
const cleaned = stripAndRepair(redactCredentials(text));
const capped = capEntryLength(cleaned);
return capped === cleaned ? capped : stripAndRepair(capped);
}
@@ -19,6 +19,7 @@ import type {
LogEntry,
PromptUserFn,
ResourceConfig,
RuntimeAdapter,
} from "@repo/adapters";
import {
BareRuntime,
@@ -26,9 +27,12 @@ import {
CloudRuntime,
DockerRuntime,
STATIC_RELEASE_BASE,
sharedMountExecutor,
resolveStaticOutputPath,
ensurePortAvailable,
allocateHostPort,
pickHostPort,
isHostChannelUnavailableError,
runDeployPipeline,
isMultiServiceRuntime,
waitForReady,
@@ -36,11 +40,14 @@ import {
} from "@repo/adapters";
import { platform } from "../../lib/controller-helpers";
import { resolveUpstreamUrl, resolveRouteStrategy } from "../../lib/upstream-url";
import { compileProjectRoutingFields } from "../../lib/project-routing-fields";
import { webhookProxyTarget } from "../../config";
import {
disposeRuntime,
resolveDeploymentRuntime,
resolveDeploymentPlatform,
resolveEffectiveTarget,
hostChannelDeployNotice,
} from "../../lib/deployment-runtime";
import { ensureRoutingReady } from "../../lib/edge-reconcile";
import { sshManager } from "../../lib/ssh-manager";
@@ -310,6 +317,8 @@ async function reuseRetainedArtifact(opts: {
runtime: { name: string };
buildSessionId: string;
targetExecutor?: CommandExecutor | null;
/** Reaches the static release tree; see `sharedMountExecutor`. */
staticExecutor?: CommandExecutor | null;
logger: BuildLogger;
}): Promise<BuildResult | null> {
const { snapshot, runtime, buildSessionId, targetExecutor, logger } = opts;
@@ -334,7 +343,11 @@ async function reuseRetainedArtifact(opts: {
const staticDir = pinnedStaticDir(snapshot);
if (staticDir) {
const exists = await (targetExecutor?.exists(staticDir) ?? Promise.resolve(false));
// A pin we cannot VERIFY is a pin we don't trust: rebuild rather than fail the
// deploy, which is what an executor that can't reach the tree at all would do.
const exists = await (opts.staticExecutor ?? targetExecutor)
?.exists(staticDir)
.catch(() => false);
return exists ? reuse(staticDir) : gone(staticDir);
}
@@ -442,6 +455,10 @@ export async function finalizeComposeDeploy(opts: {
async function executeBuildAndDeploy(project: Project, dep: Deployment, buildSessionId: string) {
const plat = platform();
let { runtime, routing, ssl, system } = plat;
// Every transport THIS deploy opens, released in the `finally` at the bottom.
// `plat`'s own runtime is the process-wide singleton and is deliberately never
// added: disposing it would close the control plane's own Docker transport.
const transports = new Set<RuntimeAdapter>();
const snapshot = dep.meta as DeploymentConfigSnapshot | null;
if (!snapshot) {
@@ -518,6 +535,7 @@ async function executeBuildAndDeploy(project: Project, dep: Deployment, buildSes
ssl = resolved.platform.ssl;
system = resolved.platform.system;
ctx.runtime = runtime;
if (runtime !== plat.runtime) transports.add(runtime);
// Persist the serve/lifecycle identity ONCE (no undo): bare for static
// file-serve, docker for services, unchanged otherwise.
if (runtimeModes.serveRuntimeMode !== undefined) {
@@ -534,6 +552,9 @@ async function executeBuildAndDeploy(project: Project, dep: Deployment, buildSes
const usesManagedRouting = resolved.usesManagedRouting;
const targetExecutor: CommandExecutor | null = resolved.platform.executor;
// Static releases live on a mount the api container shares 1:1 with its host,
// so on the local box they need no host channel — see `sharedMountExecutor`.
const staticExecutor = await sharedMountExecutor(resolved.platform);
// Surface the resolved deploy path so the operator can SEE where it lands —
// in particular the self-hosted sandbox-vs-direct runtime, the choice that
@@ -550,6 +571,13 @@ async function executeBuildAndDeploy(project: Project, dep: Deployment, buildSes
}\n`,
);
// Say ONCE, next to the target it describes, that this box can't drive its host
// (#509). Deliberately outside any `routeStrategy` branch: the port-scan hint it
// replaces only appeared under loopback-port, which is half of why a demoted
// channel reached the first crashed container unannounced. Never gates the deploy.
const hostNotice = hostChannelDeployNotice(targetExecutor);
if (hostNotice) logger.log(`${hostNotice}\n`, "warn");
await repos.deployment.updateBuildSession(buildSessionId, {
status: "building",
startedAt: new Date(),
@@ -620,6 +648,12 @@ async function executeBuildAndDeploy(project: Project, dep: Deployment, buildSes
snapshot.buildStrategy,
{ deployTarget: snapshot.deployTarget },
);
// Written BACK, not just held locally. Without this the resolved value reached
// only the readers in this function (clone planning), while `createBuildConfig`
// further down copies `snapshot.buildStrategy` verbatim into the adapter's
// BuildConfig — so the clone plan saw "server" and the runtime saw the frozen
// "local". One snapshot, one answer.
snapshot.buildStrategy = buildStrategy;
const buildEnv = buildScopedEnvVars(envMap);
// Resolve a fresh GitHub token for cloning private repos.
@@ -879,6 +913,7 @@ async function executeBuildAndDeploy(project: Project, dep: Deployment, buildSes
ssl,
system,
executor: targetExecutor,
localHost: resolved.platform.localHost,
usesManagedRouting,
logger,
ctx,
@@ -952,7 +987,14 @@ async function executeBuildAndDeploy(project: Project, dep: Deployment, buildSes
// clone or relay a credential for (see reuseRetainedArtifact). A pin is a
// hint, never a guarantee — if the artifact is gone we build from source.
const buildResult =
(await reuseRetainedArtifact({ snapshot, runtime, buildSessionId, targetExecutor, logger })) ??
(await reuseRetainedArtifact({
snapshot,
runtime,
buildSessionId,
targetExecutor,
staticExecutor,
logger,
})) ??
(await buildFromSource());
provisioned.imageRef = buildResult.imageRef;
@@ -989,6 +1031,7 @@ async function executeBuildAndDeploy(project: Project, dep: Deployment, buildSes
ssl,
system,
targetExecutor,
staticExecutor,
baseTarget: plat.target,
effectiveTarget: resolved.effectiveTarget,
serverId: resolved.serverId,
@@ -999,6 +1042,7 @@ async function executeBuildAndDeploy(project: Project, dep: Deployment, buildSes
prodResources,
logger,
deployRouting,
transports,
};
// deployMode is derived from runtime.name === "cloud", so the cast is sound.
@@ -1014,6 +1058,12 @@ async function executeBuildAndDeploy(project: Project, dep: Deployment, buildSes
// outcome was recorded is bookkeeping: reporting it as a failure would
// invert a working deploy and tear its containers down.
await reportPipelineError(ctx, message, logger);
} finally {
// The deploy is over either way — release the loopback bridges it opened.
// Safe here and not earlier: the readiness/stabilization gate runs INLINE as
// the pipeline's healthCheck hook, so nothing still needs a transport once
// this function settles.
for (const rt of transports) disposeRuntime(rt);
}
}
@@ -1028,6 +1078,8 @@ interface DeployPhaseInputs {
ssl: Awaited<ReturnType<typeof platform>>["ssl"];
system: Awaited<ReturnType<typeof platform>>["system"];
targetExecutor: CommandExecutor | null;
/** Reaches the static release tree; see `sharedMountExecutor`. */
staticExecutor: CommandExecutor | null;
/** Base platform target ("desktop" | "selfhosted" | "cloud") + the resolved
* per-deployment target/server — used to gate the `.openship` manifest write
* to desktop-mode server deploys only. */
@@ -1043,6 +1095,17 @@ interface DeployPhaseInputs {
/** Build/deploy routing decided once from the resolved runtime (static-sandbox /
* static-file-serve / server / static-edge) — replaces scattered `instanceof`. */
deployRouting: DeployRouting;
/**
* Transports this deploy opened, released together when it ends.
*
* A deploy resolves MORE than one runtime — its own target platform, plus the
* PREVIOUS deployment's runtime to deactivate the old containers — and each one
* for a remote server binds its own Docker-over-SSH loopback bridge that only
* `dispose()` closes. A phase can't release its own (a throw would skip it, and
* `executeBuildAndDeploy`'s catch can't see phase locals), so they're collected
* here and released in that function's `finally`.
*/
transports: Set<RuntimeAdapter>;
}
/** Static edge deploy via CloudRuntime (Oblien Pages). */
@@ -1359,15 +1422,15 @@ async function executeServerDeploy(phase: DeployPhaseInputs): Promise<void> {
// of how they were BUILT — a Docker sandbox (server/self-hosted, the common
// case) or bare (Docker-less desktop "This Machine"). A dedicated bare
// file-serve runtime rooted at the edge-shared STATIC_RELEASE_BASE promotes
// the built dir into a release and hands the edge a `root`. Its executor is
// the platform executor, which is exactly the FS the build wrote to (SSH for a
// remote server; local for a local / docker-edge host where the extract landed
// on the shared /opt/openship/static mount).
// the built dir into a release and hands the edge a `root`. Its executor is the
// one that reaches that tree — SSH for a remote server, LOCAL for the local box,
// where /opt/openship/static is the one bind mount the api container, the edge and
// the host all see, so no host channel is in the path (`sharedMountExecutor`).
const isStaticFileServe = phase.deployRouting.deployMode === "static-file-serve";
const staticServeRuntime = isStaticFileServe
? new BareRuntime({
workDir: STATIC_RELEASE_BASE,
executor: phase.targetExecutor ?? undefined,
executor: phase.staticExecutor ?? undefined,
})
: null;
// Where the static doc-root lives: "" when a Docker sandbox build already
@@ -1575,9 +1638,32 @@ async function executeServerDeploy(phase: DeployPhaseInputs): Promise<void> {
// The deploy's own executor when it has one; otherwise the POOLED host
// channel — never a bare `createHostExecutor()`, which builds a fresh
// SSH connection per call and leaked one sshd session each time (#291).
pinnedHostPort = phase.targetExecutor
const allocation = phase.targetExecutor
? await allocateHostPort(phase.targetExecutor, { avoid })
: await sshManager.withHostExecutor((exec) => allocateHostPort(exec, { avoid }));
: // No executor for this target, so the pooled host channel is the only way to
// read the box's live ports. When that channel is unusable, degrade to the
// SAME answer scanPorts gives an executor it can't reach — "couldn't scan",
// pick from `avoid` alone — instead of failing the deploy. Docker publishes
// the port either way; refusing to deploy over an unreadable scan would make
// a blocked channel take down container deploys it never touches (#490).
await sshManager
.withHostExecutor((exec) => allocateHostPort(exec, { avoid }))
.catch((e) => {
if (!isHostChannelUnavailableError(e)) throw e;
return { port: pickHostPort(new Set(), { avoid }), scanned: false };
});
pinnedHostPort = allocation.port;
// A scan that couldn't RUN is not "nothing is listening". Say so here or the
// bind failure that follows blames Docker for a host we simply couldn't read
// — the #490 pattern, where the cause never appears in the log.
if (!allocation.scanned) {
logger.log(
`Couldn't read live port occupancy on the target, so ${pinnedHostPort} avoids only ` +
`the ports pinned to other projects. If publishing it fails as "already allocated", ` +
`this is why — check that Openship can reach this host (Servers → this box).\n`,
"warn",
);
}
await repos.project
.update(project.id, { hostPort: pinnedHostPort })
.catch((err) => logger.log(`Couldn't persist host port ${pinnedHostPort}: ${safeErrorMessage(err)}\n`, "warn"));
@@ -1633,11 +1719,17 @@ async function executeServerDeploy(phase: DeployPhaseInputs): Promise<void> {
const prevDep = project.activeDeploymentId
? await repos.deployment.findById(project.activeDeploymentId)
: null;
// A DISTINCT platform from this deploy's, so on a remote server it binds its own
// bridge. Registered rather than released here: `deployEnv` closes over it to
// deactivate/destroy the old containers, so it has to outlive this line — and the
// `=== runtime` fallback must never be disposed, since that IS the live deploy's
// transport (the Set dedupes it away).
const previousRuntime = prevDep?.containerId
? await resolveDeploymentRuntime(prevDep)
.then((r) => r.runtime)
.catch(() => runtime)
: runtime;
if (previousRuntime !== runtime) phase.transports.add(previousRuntime);
// buildProjectRouteDomains turns the project's public endpoints (and
// existing domain rows) into concrete routes. We persist a domain
@@ -1795,6 +1887,19 @@ async function executeServerDeploy(phase: DeployPhaseInputs): Promise<void> {
? { webhookDomain: project.webhookDomain, webhookProxy: webhookProxyTarget }
: {}),
...(proxySettings ? { proxy: proxySettings } : {}),
// The project's vercel.json rules. Required, not optional: registerRoute REPLACES
// the vhost, so without them this deploy DELETES any redirects/headers/cleanUrls a
// domain edit or "Retry routing" had installed — a rule that worked would stop
// working on the next push, silently.
//
// Skips are logged here (and nowhere else on this path): this is the single-app /
// static shape, so no composite planner runs afterwards to resolve a path rewrite
// against a real backend — what the compiler refused is genuinely not live, and the
// deploy log is where someone looks after editing vercel.json. Same wording the
// compose composite path uses, so there is one phrase to recognise.
...compileProjectRoutingFields(project.routingConfig, (note) =>
logger.log(`vercel.json rule not applied — ${note}\n`, "warn"),
),
},
promptUser: (prompt) => sessionManager.promptUser(dep.id, prompt),
},
@@ -62,8 +62,19 @@ export async function getBuildSessionStatus(deploymentId: string) {
// Resolve the target server's display name (when this deployed to a server),
// so the detail UI can show "Server · <name>" rather than a raw id.
//
// ORG-SCOPED, and it must be: `snapshot.serverId` is CLIENT-SUPPLIED (it arrives in
// the deploy body, and resolveSnapshotTarget gives that override top priority with no
// org check), and the deployment row is persisted with it BEFORE the target is
// validated — so a deploy that fails on "not in this organization" still leaves a row
// whose meta names a foreign server. An unscoped read here hands that server's name or
// sshHost back to anyone who can post a deploy: the cross-tenant name oracle the
// docblock on resolveOrgServer (lib/deployment-runtime.ts) exists to prevent. A
// foreign id must read as absent, exactly like a deleted one.
const targetServer = snapshot?.serverId
? await repos.server.get(snapshot.serverId).catch(() => null)
? await repos.server
.getInOrganization(snapshot.serverId, dep.organizationId)
.catch(() => null)
: null;
// Derive step progress from persisted log entries when no active session
@@ -127,6 +127,10 @@ export async function runDeploymentPreflight(
/** Project id — passed to the remote-clone-token preflight check so
* project-scoped clone tokens are considered. */
projectId?: string;
/** Catalog app this project instantiates + whether it has ever been live, so
* the app's declared host minimum is matched against the target machine. */
appTemplateId?: string | null;
firstDeploy?: boolean;
},
): Promise<void> {
const preflight = await runPreflightChecks(snapshot, {
@@ -141,6 +145,8 @@ export async function runDeploymentPreflight(
...(opts.multiService !== undefined ? { multiService: opts.multiService } : {}),
...(opts.gitOwner !== undefined ? { gitOwner: opts.gitOwner } : {}),
...(opts.projectId !== undefined ? { projectId: opts.projectId } : {}),
...(opts.appTemplateId !== undefined ? { appTemplateId: opts.appTemplateId } : {}),
...(opts.firstDeploy !== undefined ? { firstDeploy: opts.firstDeploy } : {}),
buildStrategy: snapshot.buildStrategy as "local" | "server" | undefined,
});
if (!preflight.ok) {
@@ -1193,6 +1199,10 @@ export async function requestBuildAccess(ctx: RequestContext, input: BuildAccess
multiService: useServicePipeline,
gitOwner: project.gitOwner,
projectId: project.id,
// An app project carries its catalog id; a never-deployed one is the only
// deploy a host-capacity shortfall is allowed to refuse.
appTemplateId: project.appTemplateId,
firstDeploy: !project.activeDeploymentId,
});
const env = environment || "production";
@@ -1397,11 +1407,24 @@ export async function redeployBuildSession(
meta.deployTarget = t.deployTarget;
meta.serverId = t.serverId;
meta.runtimeMode = t.runtimeMode;
meta.buildStrategy = await settingsService.resolveStrategy(meta.framework, meta.buildStrategy, {
deployTarget: meta.deployTarget,
});
}
// buildStrategy is re-resolved on EVERY redeploy, frozen snapshot or not — it is a
// policy answer about the instance, not part of the build identity, so it belongs
// with `resources` above rather than with the frozen source.
//
// Passing the frozen value through as `explicit` keeps it: resolveStrategy returns
// an explicit choice unchanged. The one thing it does NOT keep is a "local" that is
// no longer permitted, because its CLOUD_MODE branch answers "server" before it
// looks at `explicit`. That is the whole point — a project promoted from a
// self-hosted install arrives with a frozen "local" and, while this sat inside the
// `!frozenMeta` branch, every redeploy of it asked the cloud runtime to build on
// the SaaS host. The runtime now refuses that too (HostBuildForbiddenError); this
// is the half that keeps a legitimate deploy working instead of failing.
meta.buildStrategy = await settingsService.resolveStrategy(meta.framework, meta.buildStrategy, {
deployTarget: meta.deployTarget,
});
// Release/dist source: refresh the resolved dist dir. useExistingCommit →
// redeploy the SAME version; default → newest advertised (parity with the
// "redeploy latest commit" semantics below). Re-resolving also guards against
@@ -1734,13 +1757,19 @@ export async function triggerDeployment(
await applyReleaseSourceToSnapshot(project, snapshot, { version: data.releaseVersion });
}
if (!reuse) {
{
// Non-UI callers (CI, webhook, manual API) don't pass buildStrategy, so the
// snapshot inherits `undefined` from buildConfigSnapshot and the later
// fallback at resolveBuildGitToken collapses everything to "server". Run
// it through resolveStrategy so a non-cloud stack with a "local" default
// gets the same answer the UI would give — single source of truth. A reused
// snapshot already froze its resolved strategy, so leave it untouched.
// gets the same answer the UI would give — single source of truth.
//
// Runs for a REUSED (rollback) snapshot too. That looks like it contradicts
// "restore the exact prior state", and for the source it would — but a frozen
// explicit value is returned unchanged here, so the only thing a rollback loses
// is a "local" the instance no longer permits, which it could not have honoured
// anyway (the cloud runtime refuses it at the sink). Rolling back to a build that
// cannot run is not a restored state.
snapshot.buildStrategy = await settingsService.resolveStrategy(
snapshot.framework,
snapshot.buildStrategy,
@@ -1760,6 +1789,10 @@ export async function triggerDeployment(
multiService: useServicePipeline,
gitOwner: project.gitOwner,
projectId: project.id,
// An app project carries its catalog id; a never-deployed one is the only
// deploy a host-capacity shortfall is allowed to refuse.
appTemplateId: project.appTemplateId,
firstDeploy: !project.activeDeploymentId,
});
// Env: a reused snapshot ships the EXACT encrypted env captured with the
@@ -0,0 +1,128 @@
import { beforeEach, describe, expect, it, vi } from "vitest";
import type { CommandExecutor } from "@repo/adapters";
/**
* An app template's generated config files are bind-mounted by HOST path, so the one
* thing that must never happen is writing them with an executor that reaches a
* different machine than the Docker daemon. On a Compose install `platform.executor`
* for a plain local target is a LocalExecutor — the api CONTAINER — and a write there
* SUCCEEDS, after which Docker mounts an empty directory over the file the service
* needs. That failure mode has no error to follow, which is how it hid inside #490.
*/
const h = vi.hoisted(() => ({
withHostExecutor: vi.fn(async (fn: (e: CommandExecutor) => Promise<void>) =>
fn({ tag: "host-channel" } as unknown as CommandExecutor),
),
}));
vi.mock("../../../lib/ssh-manager", () => ({
sshManager: { withHostExecutor: h.withHostExecutor },
}));
const { APP_CONFIG_HOST_ROOT, appConfigHostPath, withAppConfigHost, writeAppConfigFile } =
await import("./app-config-host");
const LOCAL = { tag: "local-executor" } as unknown as CommandExecutor;
const REMOTE = { tag: "remote-ssh" } as unknown as CommandExecutor;
describe("withAppConfigHost", () => {
// Block body on purpose: `mockClear()` returns the mock, and a function returned
// from beforeEach is taken as its teardown and called with no arguments.
beforeEach(() => {
h.withHostExecutor.mockClear();
});
it("writes through the host channel when the target is this box", async () => {
const seen: CommandExecutor[] = [];
const { ran } = await withAppConfigHost(
// The executor a local target carries is the container's own filesystem; the
// point of the branch is that it is NOT the one used.
{ executor: LOCAL, localHost: true, isCloud: false },
async (host) => void seen.push(host),
);
expect(ran).toBe(true);
expect(h.withHostExecutor).toHaveBeenCalledOnce();
expect(seen).toEqual([{ tag: "host-channel" }]);
});
it("uses the server's own executor for a remote target", async () => {
const seen: CommandExecutor[] = [];
const { ran } = await withAppConfigHost(
{ executor: REMOTE, localHost: false, isCloud: false },
async (host) => void seen.push(host),
);
expect(ran).toBe(true);
expect(seen).toEqual([REMOTE]);
// A remote box's files must never be written over the LOCAL host channel.
expect(h.withHostExecutor).not.toHaveBeenCalled();
});
it("propagates an unusable host channel instead of writing somewhere else", async () => {
h.withHostExecutor.mockRejectedValueOnce(new Error("host channel unreachable"));
await expect(
withAppConfigHost({ executor: LOCAL, localHost: true, isCloud: false }, async () => {}),
).rejects.toThrow(/host channel unreachable/);
});
it("runs nothing on cloud, which mounts no host paths", async () => {
const fn = vi.fn();
expect(await withAppConfigHost({ executor: LOCAL, localHost: true, isCloud: true }, fn)).toEqual(
{ ran: false },
);
expect(fn).not.toHaveBeenCalled();
});
it("runs nothing for a remote target with no executor", async () => {
const fn = vi.fn();
expect(
await withAppConfigHost({ executor: null, localHost: false, isCloud: false }, fn),
).toEqual({ ran: false });
expect(fn).not.toHaveBeenCalled();
});
});
describe("appConfigHostPath", () => {
it("nests by project and service under the host root", () => {
expect(appConfigHostPath("proj1", "kong", "/etc/kong/kong.yml")).toBe(
`${APP_CONFIG_HOST_ROOT}/proj1/kong/etc/kong/kong.yml`,
);
});
it("sanitizes a service name that could reshape the path", () => {
expect(appConfigHostPath("proj1", "a/b c", "/x.yml")).toBe(
`${APP_CONFIG_HOST_ROOT}/proj1/a_b_c/x.yml`,
);
});
it("refuses a traversal in the template-supplied container path", () => {
// Template-authored and written on the HOST — escaping the root would write
// anywhere root can.
expect(() => appConfigHostPath("proj1", "kong", "/../../etc/cron.d/x")).toThrow(/traversal/);
});
});
describe("writeAppConfigFile", () => {
it("names the requirement, the file and the underlying error on failure", async () => {
const writer = {
writeFile: async () => {
throw new Error("Timed out while waiting for handshake");
},
} as unknown as CommandExecutor;
await expect(
writeAppConfigFile(writer, "/var/lib/openship/app-config/p/kong/kong.yml", "x", "kong", "/kong.yml"),
).rejects.toThrow(
/Service "kong".*\/kong\.yml.*HOST path.*Timed out while waiting for handshake/s,
);
});
it("passes the content straight through on success", async () => {
const calls: Array<[string, string]> = [];
const writer = {
writeFile: async (p: string, c: string) => void calls.push([p, c]),
} as unknown as CommandExecutor;
await writeAppConfigFile(writer, "/host/kong.yml", "_format_version: '3.0'", "kong", "/kong.yml");
expect(calls).toEqual([["/host/kong.yml", "_format_version: '3.0'"]]);
});
});
@@ -0,0 +1,104 @@
/**
* Where an app template's generated config files (`advanced.files`) live, and which
* executor is allowed to write them.
*
* These files are bind-mounted read-only into the service container (Kong's
* `kong.yml`, a Postgres init `.sql`), and Docker resolves bind sources on the
* machine running the DAEMON. So they are host state, unavoidably: a file written to
* the api container's own filesystem looks written, and then Docker mounts an empty
* directory over the path the service expects a file at — the container starts and
* dies with a config error that names neither the file nor the reason (#490).
*
* Unlike the static release tree, this root is deliberately NOT bind-mounted into the
* api container (see EDGE_CONTAINER_MOUNTS): a mount would only ever be right for the
* local box, and would quietly do the wrong thing the moment the target is a remote
* server. Choosing the executor per target is the honest version of that, which is
* what this module is for.
*/
import type { CommandExecutor } from "@repo/adapters";
import { sshManager } from "../../../lib/ssh-manager";
/**
* Persistent on-host root for app template config files. Sibling of the other
* openship host state (`/var/lib/openship/ssh-keys`); overridable for hosts that keep
* openship state elsewhere. Files land at
* `<root>/<projectId>/<service>/<container-path>` — the executor creates parent dirs —
* and the container-absolute path is appended verbatim so binds are unique and
* self-describing.
*/
export const APP_CONFIG_HOST_ROOT =
process.env.OPENSHIP_APP_CONFIG_DIR || "/var/lib/openship/app-config";
export function appConfigHostPath(
projectId: string,
serviceName: string,
containerPath: string,
): string {
const safeSvc = serviceName.replace(/[^a-zA-Z0-9._-]/g, "_");
const rel = containerPath.replace(/^\/+/, "");
// Reject `..` traversal: `rel` is a template-supplied path written on the HOST
// (and bind-mounted), so a crafted `../../etc/...` must not escape the root.
if (rel.split("/").some((seg) => seg === "..")) {
throw new Error(`Unsafe app-config path (directory traversal): ${containerPath}`);
}
return `${APP_CONFIG_HOST_ROOT}/${projectId}/${safeSvc}/${rel}`;
}
/**
* Run `fn` with an executor that reaches the machine whose Docker daemon will mount
* these files, or report that nothing can.
*
* The local box goes through the host channel rather than `platform.executor`: for a
* plain local target that executor is a `LocalExecutor`, which on a Compose install is
* the api CONTAINER's filesystem — see the module note. `withHostExecutor` resolves to
* that same local executor on a bare install, so this is a no-op there, and it throws
* (rather than writing to the wrong machine) when the api is containerized with no
* usable channel.
*
* Cloud mounts no host paths at all, and a remote target with no executor is nothing
* we can reach — both report `ran: false` so the caller can say so per service.
*/
export async function withAppConfigHost(
opts: { executor?: CommandExecutor | null; localHost?: boolean; isCloud: boolean },
fn: (host: CommandExecutor) => Promise<void>,
): Promise<{ ran: boolean }> {
if (opts.isCloud) return { ran: false };
if (opts.localHost) {
await sshManager.withHostExecutor(fn);
return { ran: true };
}
if (!opts.executor) return { ran: false };
await fn(opts.executor);
return { ran: true };
}
/**
* Write one generated config file, failing with the REQUIREMENT rather than the
* symptom.
*
* When a host firewall drops the container→host channel (#490) the raw failure is an
* SSH timeout, or a "No such file" naming a path that looks like it should exist —
* neither of which points at the channel. The deploy MUST still fail (a service whose
* `kong.yml` never landed starts and then dies); it just has to fail legibly.
*/
export async function writeAppConfigFile(
writer: CommandExecutor,
hostPath: string,
content: string,
serviceName: string,
containerPath: string,
): Promise<void> {
try {
await writer.writeFile(hostPath, content);
} catch (err) {
const detail = err instanceof Error ? err.message : String(err);
throw new Error(
`Service "${serviceName}": couldn't write its generated config file ${containerPath} to ` +
`${hostPath} on the deploy target. Docker mounts this file by HOST path, so it has to ` +
`be written on the machine running the containers — the usual causes are that Openship ` +
`can't reach that machine, or can't write that path. Underlying error: ${detail}`,
);
}
}
@@ -0,0 +1,273 @@
import { existsSync, readdirSync } from "node:fs";
import { readFile } from "node:fs/promises";
import { tmpdir } from "node:os";
import { isAbsolute, join } from "node:path";
import { BuildLogger, DEFAULT_RESOURCE_CONFIG, type BuildConfig } from "@repo/adapters";
import { describe, expect, it, vi } from "vitest";
// buildComposeImages reads the project's services from the DB and broadcasts SSE
// status. Both are side effects the inline-build materialization contract does not
// depend on — stub them so the test exercises ONLY the materialize/COPY path.
// (Importing the real @repo/db would also risk the PGlite single-instance lock.)
const { listByProjectMock } = vi.hoisted(() => ({ listByProjectMock: vi.fn() }));
vi.mock("@repo/db", () => ({
repos: { service: { listByProject: listByProjectMock } },
}));
vi.mock("../session-manager", () => ({
broadcastServiceStatus: vi.fn(),
broadcastInstallPhase: vi.fn(),
}));
import { buildComposeImages } from "./build.service";
/**
* These pin the AUTHOR-FACING contract of an inline catalog build (`advanced.build`):
* the Dockerfile + `files[]` are materialized on the orchestrator under a per-service
* subdir whose name is `sanitizeComposeImageName(service.name)`, inside ONE shared
* context root, and `docker build` runs with that shared root as context. So a
* template author writes `COPY <service-name>/<file>` — the subdir IS the public COPY
* prefix. If the subdir scheme ever changes, every authored inline-build Dockerfile
* breaks at `docker build` (COPY source not found) while the catalog string-match
* tests stay green; this file is the regression guard against that silent break.
*
* The subdir rides in `dockerfilePath`, NOT `rootDirectory`, and that split is load
* bearing: docker builds with the whole context dir, while cloud tars up
* `rootDirectory` alone (cloud.ts resolveDockerfileBuildSource). Moving the subdir
* into `rootDirectory` keeps docker working and silently breaks every COPY on cloud.
*/
const SNAPSHOT = {
repoUrl: "",
branch: "main",
framework: "docker",
buildImage: "",
runtimeImage: "",
packageManager: "",
installCommand: "",
buildCommand: "",
outputDirectory: "",
productionPaths: [] as string[],
rootDirectory: "",
port: 3000,
startCommand: "",
hasServer: true,
hasBuild: false,
};
type InlineBuild = { dockerfile: string; files?: { path: string; content: string }[] };
// Only the fields buildComposeImages actually reads off a service row.
function inlineService(name: string, build: InlineBuild, id = name) {
return {
id,
name,
enabled: true,
image: null,
build: null,
advanced: { build },
framework: null,
buildCommand: null,
startCommand: null,
rootDirectory: null,
dockerfile: null,
};
}
/** Materialized context roots currently sitting in os.tmpdir() (leak detector). */
function catalogTempRoots(): string[] {
return readdirSync(tmpdir()).filter((n) => n.startsWith("openship-catalog-build-"));
}
type Captured = { config: BuildConfig; serviceName: string };
/**
* Drive buildComposeImages with a FAKE runtime.buildImages that (a) captures each
* service's resolved BuildConfig and (b) hands the materialized context root to
* `onContext` while it still exists on disk — the `finally` rm's the temp root the
* moment buildComposeImages returns, so any disk assertion must run inside here.
*/
async function run(
services: unknown[],
onContext?: (root: string, item: Captured) => void | Promise<void>,
) {
listByProjectMock.mockResolvedValue(services);
const captured: Captured[] = [];
const runtime = {
name: "docker" as const,
build: vi.fn(),
buildImages: vi.fn(async (items: Array<Record<string, unknown>>) => {
for (const item of items) {
const entry: Captured = {
config: item.config as BuildConfig,
serviceName: item.serviceName as string,
};
captured.push(entry);
const root = entry.config.localPath;
if (root && onContext) await onContext(root, entry);
(item.onResult as (r: unknown) => void)({
sessionId: "s",
status: "running",
imageRef: `img/${entry.serviceName}`,
});
}
}),
};
const result = await buildComposeImages({
// Only the fields the function reads; the rest of the graph is stubbed.
project: {
id: "p1",
slug: "app",
name: "app",
localPath: null,
workspacePrepareCommand: null,
} as never,
dep: { id: "d1", branch: "main", commitSha: null, trigger: "deploy", meta: null } as never,
runtime: runtime as never,
logger: new BuildLogger(() => {}),
snapshot: SNAPSHOT as never,
buildSessionId: "sess",
buildEnvVars: {},
buildResources: DEFAULT_RESOURCE_CONFIG,
});
return { result, captured };
}
describe("buildComposeImages — inline catalog build materialization", () => {
it("materializes the Dockerfile + files under the sanitized service-name subdir and points the build config at the shared root", async () => {
const dockerfile =
'FROM alpine\nCOPY compute/hello.sh /hello.sh\nRUN chmod +x /hello.sh\nENTRYPOINT ["/hello.sh"]\n';
let rootSeen = "";
const disk: Record<string, string> = {};
const { result, captured } = await run(
[
inlineService("compute", {
dockerfile,
files: [
{ path: "hello.sh", content: "echo hi\n" },
{ path: "conf/app.toml", content: "x=1\n" }, // nested path → dirname mkdir
],
}),
],
async (root) => {
rootSeen = root;
disk.dockerfile = await readFile(join(root, "compute", "Dockerfile"), "utf-8");
disk.hello = await readFile(join(root, "compute", "hello.sh"), "utf-8");
disk.nested = await readFile(join(root, "compute", "conf", "app.toml"), "utf-8");
},
);
expect(captured).toHaveLength(1);
const cfg = captured[0].config;
// Source is the shared orchestrator temp root, not a repo clone.
expect(cfg.localPath).toBe(rootSeen);
expect(isAbsolute(cfg.localPath!)).toBe(true);
expect(cfg.localPath).toContain("openship-catalog-build-");
// The context stays the SHARED ROOT; the subdir — the author-facing COPY prefix
// — rides in the Dockerfile path, so `COPY compute/…` resolves on every runtime.
expect(cfg.rootDirectory).toBe("");
expect(cfg.dockerfilePath).toBe("compute/Dockerfile");
// Materialized exactly where `COPY compute/<file>` expects it.
expect(disk.dockerfile).toBe(dockerfile);
expect(disk.hello).toBe("echo hi\n");
expect(disk.nested).toBe("x=1\n");
expect(result.imageRefs.get("compute")).toBe("img/compute");
// The temp root is removed once the build phase completes (no leak).
expect(existsSync(rootSeen)).toBe(false);
});
it("sanitizes a service name with caps/spaces into the COPY-prefix subdir", async () => {
let existedAtSubdir = false;
const { captured } = await run(
[
inlineService(
"My Compute",
{ dockerfile: "FROM alpine\n", files: [{ path: "x", content: "y" }] },
"svc-1",
),
],
(root) => {
existedAtSubdir = existsSync(join(root, "my-compute", "x"));
},
);
expect(captured[0].config.dockerfilePath).toBe("my-compute/Dockerfile");
expect(existedAtSubdir).toBe(true);
});
it("rejects a service name that sanitizes to a parent ref, leaving no temp root behind", async () => {
// sanitizeComposeImageName preserves dots, so ".." survives it — and the subdir is
// joined onto the shared root, so an unguarded ".." writes the Dockerfile and every
// context file into os.tmpdir(), one level ABOVE the dir cleanup() removes.
const before = catalogTempRoots();
await expect(
run([
inlineService("..", { dockerfile: "FROM alpine\n", files: [{ path: "x", content: "y" }] }),
]),
).rejects.toThrow(/invalid build-context subdir "\.\."/);
expect(catalogTempRoots()).toEqual(before);
});
it("rejects two inline services whose names sanitize to the same subdir", async () => {
await expect(
run([
inlineService("compute", { dockerfile: "FROM alpine\n" }, "a"),
inlineService("Compute", { dockerfile: "FROM alpine\n" }, "b"),
]),
).rejects.toThrow(/both map to build-context subdir "compute"/);
});
it("rejects a context file path that escapes its service dir, leaving no temp root behind", async () => {
// The throw lands AFTER mkdtemp, so this also pins the try/finally placement:
// a materialize failure must not leak one temp dir per failed deploy.
const before = catalogTempRoots();
await expect(
run([
inlineService("compute", {
dockerfile: "FROM alpine\n",
files: [{ path: "../evil.sh", content: "x" }],
}),
]),
).rejects.toThrow(/Invalid build file path "\.\.\/evil\.sh"/);
expect(catalogTempRoots()).toEqual(before);
});
it("rejects an inline build blob with no Dockerfile contents", async () => {
await expect(run([inlineService("compute", { dockerfile: "" })])).rejects.toThrow(
/no Dockerfile contents/,
);
});
it("rejects an inline build mixed with a repo-built service in the same batch", async () => {
// Both would build from ONE shared tree (specs[0]'s), so the inline service
// would fall through to the repo's root Dockerfile — reject instead.
const repoService = {
...inlineService("api", { dockerfile: "" }, "b"),
advanced: {},
build: "./api",
};
await expect(
run([inlineService("compute", { dockerfile: "FROM alpine\n" }, "a"), repoService]),
).rejects.toThrow(/can't be built alongside repo-built services \(api\)/);
});
it("allows a context filename that merely starts with dots (not a parent ref)", async () => {
let existed = false;
await run(
[
inlineService("compute", {
dockerfile: "FROM alpine\n",
files: [{ path: "..keep", content: "x" }],
}),
],
(root) => {
existed = existsSync(join(root, "compute", "..keep"));
},
);
expect(existed).toBe(true);
});
});
@@ -6,6 +6,10 @@
* resolved directly without a build step.
*/
import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
import { tmpdir } from "node:os";
import { dirname, isAbsolute, join, relative, sep } from "node:path";
import type {
AmbientGitVia,
MultiServiceRuntimeAdapter,
@@ -13,6 +17,7 @@ import type {
BuildResult,
} from "@repo/adapters";
import { BuildLogger, STATIC_RELEASE_BASE } from "@repo/adapters";
import type { ComposeAdvanced } from "@repo/core";
import { repos, type Deployment, type Project, type Service } from "@repo/db";
import {
@@ -36,6 +41,122 @@ function sanitizeComposeImageName(value: string): string {
);
}
/** A catalog template ships this service's Docker build context INLINE
* (`advanced.build`) — no repo, no pullable image. */
function hasInlineBuild(service: { advanced: unknown }): boolean {
return !!(service.advanced as ComposeAdvanced | null)?.build;
}
interface InlineBuildContexts {
/** serviceId → the shared context root + this service's Dockerfile inside it. */
byServiceId: Map<string, { root: string; dockerfile: string }>;
cleanup(): Promise<void>;
}
/**
* Materialize every inline build context to a subdir under ONE shared temp root
* on the orchestrator, and hand back the Dockerfile each service should build
* (relative to that root, which `prepareSourceTree` consumes as `localPath` —
* no git clone).
*
* One shared root rather than one per service for two reasons: `runtime.buildImages`
* builds a whole batch against a single tree (specs[0]'s), and the subdir IS the
* COPY prefix template authors write against — `COPY <service-name>/<file>`, which
* only resolves when the Docker context is the root that subdir sits in.
*
* Callers own `cleanup()` once the build phase is done; a throw in here cleans up
* after itself, so a failed materialize never leaks a temp dir per deploy.
*/
async function materializeInlineBuildContexts(buildable: Service[]): Promise<InlineBuildContexts> {
const inline = buildable.filter(hasInlineBuild);
const byServiceId = new Map<string, { root: string; dockerfile: string }>();
if (inline.length === 0) return { byServiceId, cleanup: async () => {} };
// One batch = one tree, so an inline context and a repo checkout can't coexist in
// it: the inline service would fall through to the repo's ROOT Dockerfile and
// build the wrong image under its own tag. Reachable only by hand — `advanced` is
// an open object on the compose-sync route — since a catalog install produces
// inline builds plus image-only sidecars, and those aren't buildable.
if (inline.length !== buildable.length) {
const repoBuilt = buildable
.filter((service) => !hasInlineBuild(service))
.map((service) => service.name)
.join(", ");
throw new Error(
`Services with an inline build context can't be built alongside repo-built services (${repoBuilt}) — they need different build sources. Split them into separate projects.`,
);
}
const root = await mkdtemp(join(tmpdir(), "openship-catalog-build-"));
const cleanup = () => rm(root, { recursive: true, force: true }).catch(() => {});
try {
const subdirOwner = new Map<string, string>();
for (const service of inline) {
const build = (service.advanced as ComposeAdvanced).build!;
// `advanced` is stored unvalidated, so a malformed blob has to be named here —
// writeFile would only ever report ERR_INVALID_ARG_TYPE.
if (typeof build.dockerfile !== "string" || !build.dockerfile.trim()) {
throw new Error(
`Service "${service.name}" has an inline build context with no Dockerfile contents.`,
);
}
// Must be the sanitized service NAME, since that is what an author can write in
// a COPY path (service.id is unknowable to them) — so two names that sanitize
// alike would share a context dir AND an ambiguous prefix. Reject rather than
// clobber one's Dockerfile and build the wrong image under the other's tag.
const subdir = sanitizeComposeImageName(service.name);
const clash = subdirOwner.get(subdir);
if (clash) {
throw new Error(
`Inline build services "${clash}" and "${service.name}" both map to build-context subdir "${subdir}" — rename one so their Docker build contexts don't collide.`,
);
}
subdirOwner.set(subdir, service.name);
// Exactly one segment BELOW the root: sanitizing preserves dots, so a service
// named ".." would otherwise write its context into os.tmpdir() — outside the
// directory cleanup() removes.
const serviceDir = join(root, subdir);
if (dirname(serviceDir) !== root) {
throw new Error(
`Service "${service.name}" maps to an invalid build-context subdir "${subdir}".`,
);
}
await mkdir(serviceDir, { recursive: true });
await writeFile(join(serviceDir, "Dockerfile"), build.dockerfile, "utf-8");
for (const file of build.files ?? []) {
// Same unvalidated-blob reason as the Dockerfile check above.
if (typeof file?.path !== "string" || typeof file?.content !== "string") {
throw new Error(
`Service "${service.name}" has an inline build file with a non-string path or content.`,
);
}
// A context file must stay inside its service dir, and must not BE it (that
// would writeFile over a directory). Matches a real parent ref only, not a
// filename that merely starts with two dots ("..keep", "..dockerignore").
const dest = join(serviceDir, file.path);
const rel = relative(serviceDir, dest);
if (!rel || rel === ".." || rel.startsWith(`..${sep}`) || isAbsolute(rel)) {
throw new Error(`Invalid build file path "${file.path}" for service "${service.name}"`);
}
await mkdir(dirname(dest), { recursive: true });
await writeFile(dest, file.content, "utf-8");
}
byServiceId.set(service.id, { root, dockerfile: `${subdir}/Dockerfile` });
}
} catch (error) {
await cleanup();
throw error;
}
return { byServiceId, cleanup };
}
/**
* Resolve a compose service's `build.context` to a path relative to the CLONE
* ROOT, which is what BuildConfig.rootDirectory means.
@@ -112,7 +233,13 @@ interface SubAppOverrideInputs {
};
snapshot: Pick<
BuildConfigSnapshotLike,
"framework" | "buildImage" | "packageManager" | "installCommand" | "buildCommand" | "startCommand" | "outputDirectory"
| "framework"
| "buildImage"
| "packageManager"
| "installCommand"
| "buildCommand"
| "startCommand"
| "outputDirectory"
>;
logger: Pick<BuildLogger, "log">;
}
@@ -246,6 +373,9 @@ export async function buildComposeImages(opts: {
(!opts.targetServiceIds || opts.targetServiceIds.has(service.id)) &&
!opts.refreshServiceIds?.has(service.id) &&
(!!service.build ||
// Catalog template shipping an inline Docker build context: no repo and no
// `build` column, so it needs the materialization below to build at all.
hasInlineBuild(service) ||
(serviceKind(service) === "monorepo" &&
!service.image &&
// Apply the SAME project-snapshot fallback that the build-spec resolver
@@ -259,287 +389,320 @@ export async function buildComposeImages(opts: {
!!(service.startCommand ?? opts.snapshot.startCommand)))),
);
const external = enabled.filter(
(service) => !isHandedOver(service) && !service.build && !!service.image,
(service) =>
!isHandedOver(service) && !service.build && !hasInlineBuild(service) && !!service.image,
);
// This seeds the UI check-list immediately so users see every service.
for (const service of enabled) {
sessionManager.broadcastServiceStatus(opts.dep.id, {
serviceName: service.name,
serviceId: service.id,
status: "pending",
});
}
// Repo-less catalog builds need their context on disk before any spec is built,
// and gone once the build phase ends (the images live on the deploy host).
const inlineBuilds = await materializeInlineBuildContexts(buildable);
for (const service of enabled) {
const pinned = pinnedImageForService(opts.snapshot, service.name);
if (!pinned) continue;
imageRefs.set(service.id, pinned);
sessionManager.broadcastServiceStatus(opts.dep.id, {
serviceName: service.name,
serviceId: service.id,
status: "built",
});
}
try {
// This seeds the UI check-list immediately so users see every service.
for (const service of enabled) {
sessionManager.broadcastServiceStatus(opts.dep.id, {
serviceName: service.name,
serviceId: service.id,
status: "pending",
});
}
for (const service of external) {
if (service.image) {
imageRefs.set(service.id, service.image);
for (const service of enabled) {
const pinned = pinnedImageForService(opts.snapshot, service.name);
if (!pinned) continue;
imageRefs.set(service.id, pinned);
sessionManager.broadcastServiceStatus(opts.dep.id, {
serviceName: service.name,
serviceId: service.id,
status: "built",
});
}
}
if (buildable.length > 0) {
opts.logger.step(
"build",
"running",
`Building ${buildable.length} compose service image${buildable.length === 1 ? "" : "s"}...`,
);
sessionManager.broadcastInstallPhase(opts.dep.id, { id: "images", status: "active" });
} else {
opts.logger.step(
"build",
"completed",
"Compose services use pre-built images - skipping build phase",
);
// Pull-only app (the common catalog case): no image to build → show the phase
// complete instantly rather than briefly "active".
sessionManager.broadcastInstallPhase(opts.dep.id, { id: "images", status: "skipped" });
}
for (const service of external) {
if (service.image) {
imageRefs.set(service.id, service.image);
sessionManager.broadcastServiceStatus(opts.dep.id, {
serviceName: service.name,
serviceId: service.id,
status: "built",
});
}
}
// Prepare a build spec per service: constructs the per-service BuildConfig +
// logger and broadcasts "building". Kept SEPARATE from the build itself so a
// runtime that builds every service on one daemon (Docker) can clone/transfer
// the shared source ONCE and build N images from it (see runtime.buildImages)
// instead of re-cloning per service.
const buildSpecFor = (service: Service) => {
const isMonorepo = serviceKind(service) === "monorepo";
// Build context resolution:
// - Compose service with Dockerfile → service.build, resolved against the
// compose file's directory (compose semantics — see
// resolveComposeBuildContext)
// - Monorepo sub-app → service.rootDirectory
// - Fallback → snapshot.rootDirectory
const context =
service.build != null
? resolveComposeBuildContext(opts.snapshot.rootDirectory ?? "", service.build)
: service.rootDirectory ?? opts.snapshot.rootDirectory;
const dockerfileLabel = service.dockerfile ? ` using ${service.dockerfile}` : "";
opts.logger.log(
`Building ${isMonorepo ? "monorepo app" : "compose service"} "${service.name}" from ${context || "."}${dockerfileLabel}...\n`,
"info",
{ serviceName: service.name },
);
if (buildable.length > 0) {
opts.logger.step(
"build",
"running",
`Building ${buildable.length} compose service image${buildable.length === 1 ? "" : "s"}...`,
);
sessionManager.broadcastInstallPhase(opts.dep.id, { id: "images", status: "active" });
} else {
opts.logger.step(
"build",
"completed",
"Compose services use pre-built images - skipping build phase",
);
// Pull-only app (the common catalog case): no image to build → show the phase
// complete instantly rather than briefly "active".
sessionManager.broadcastInstallPhase(opts.dep.id, { id: "images", status: "skipped" });
}
// NOTE: "building" is broadcast when this service's build actually STARTS
// (after the shared clone), not here — otherwise every service shows as
// building while we're still in the shared clone/prepare phase.
// Per-service logger keeps native terminal bytes intact and routes by
// serviceName. Inner step events are forwarded as plain service logs;
// the outer orchestrator owns the top-level step lifecycle.
const serviceLogger = new BuildLogger((entry) => {
opts.logger.callback({
timestamp: entry.timestamp,
message: entry.message,
level: entry.level,
serviceName: service.name,
serviceId: service.id,
rawData: entry.rawData,
});
});
if (opts.runtime.name === "cloud" && !opts.project.localPath) {
// Prepare a build spec per service: constructs the per-service BuildConfig +
// logger and broadcasts "building". Kept SEPARATE from the build itself so a
// runtime that builds every service on one daemon (Docker) can clone/transfer
// the shared source ONCE and build N images from it (see runtime.buildImages)
// instead of re-cloning per service.
const buildSpecFor = (service: Service) => {
const inlineBuild = inlineBuilds.byServiceId.get(service.id);
// An inline context is a Dockerfile build by definition, so a monorepo row
// carrying one must not reach the source-build factory below — that one
// ignores localPath and would build the project's repo instead.
const isMonorepo = !inlineBuild && serviceKind(service) === "monorepo";
// Build context resolution:
// - Inline catalog build → the shared materialized ROOT. The
// subdir rides in the Dockerfile path instead, which is what makes the
// author's `COPY <service-name>/<file>` resolve on every runtime (docker
// builds with the context dir; cloud tars up `rootDirectory`).
// - Compose service with Dockerfile → service.build, resolved against the
// compose file's directory (compose semantics — see
// resolveComposeBuildContext)
// - Monorepo sub-app → service.rootDirectory
// - Fallback → snapshot.rootDirectory
const context = inlineBuild
? ""
: service.build != null
? resolveComposeBuildContext(opts.snapshot.rootDirectory ?? "", service.build)
: (service.rootDirectory ?? opts.snapshot.rootDirectory);
const dockerfile = inlineBuild?.dockerfile ?? service.dockerfile;
opts.logger.log(
`Resolving Dockerfile for compose service "${service.name}" from the build source checkout.\n`,
`Building ${isMonorepo ? "monorepo app" : "compose service"} "${service.name}" from ${context || "."}${dockerfile ? ` using ${dockerfile}` : ""}...\n`,
"info",
{ serviceName: service.name },
);
}
// BuildConfig differs by kind:
// - Compose service (Dockerfile in repo) → createDockerfileBuildConfig
// forces stack="docker", clears install/build/start so the runtime
// defers everything to the repo Dockerfile.
// - Monorepo sub-app (source build) → createMonorepoSourceBuildConfig
// keeps the sub-app's stack/installCommand/buildCommand/startCommand/
// outputDirectory so the runtime synthesizes a Dockerfile from them.
const buildSlug = `${sanitizeComposeImageName(opts.project.slug ?? opts.project.name)}-${sanitizeComposeImageName(service.name)}`;
const buildConfig = isMonorepo
? createMonorepoSourceBuildConfig({
project: opts.project,
dep: opts.dep,
snapshot: opts.snapshot,
sessionId: `${opts.buildSessionId}-${service.id}`,
envVars: opts.buildEnvVars,
resources: opts.buildResources,
gitToken: opts.gitToken,
overrides: {
slug: buildSlug,
...resolveSubAppOverrides({ service, snapshot: opts.snapshot, logger: serviceLogger }),
rootDirectory: context,
port: resolveServicePort(service, opts.snapshot.port) ?? opts.snapshot.port,
// A static sub-app (no start command) serves FILES; a server sub-app
// runs its start command. Derived from framework + start command.
//
// On self-hosted the files are moved to the host and served by the edge
// — no container, no port, no second web server. That is why
// `staticExtractOnly` is gated on the runtime: on CLOUD there is no host
// directory to serve (Oblien runs the workload), so those keep the
// generated nginx image and stay a proxied container.
...(isStaticService(service)
? {
isStatic: true,
hasServer: false,
...(opts.runtime.name === "cloud"
? {}
: {
staticExtractOnly: true,
staticOutDir: `${STATIC_RELEASE_BASE}/.builds/${opts.buildSessionId}-${service.id}`,
}),
}
: { hasServer: true }),
},
})
: createDockerfileBuildConfig({
project: opts.project,
dep: opts.dep,
snapshot: opts.snapshot,
sessionId: `${opts.buildSessionId}-${service.id}`,
envVars: opts.buildEnvVars,
resources: opts.buildResources,
gitToken: opts.gitToken,
overrides: {
slug: buildSlug,
rootDirectory: context,
dockerfilePath: service.dockerfile ?? undefined,
hasServer: true,
},
// NOTE: "building" is broadcast when this service's build actually STARTS
// (after the shared clone), not here — otherwise every service shows as
// building while we're still in the shared clone/prepare phase.
// Per-service logger keeps native terminal bytes intact and routes by
// serviceName. Inner step events are forwarded as plain service logs;
// the outer orchestrator owns the top-level step lifecycle.
const serviceLogger = new BuildLogger((entry) => {
opts.logger.callback({
timestamp: entry.timestamp,
message: entry.message,
level: entry.level,
serviceName: service.name,
serviceId: service.id,
rawData: entry.rawData,
});
});
// Clone-on-server credential, shared across the fan-out (all services share
// one repo): the relay helper (desktop), a per-server ssh key, the server's
// own ambient credentials, or the token already on buildConfig.gitToken.
if (opts.cloneOnServer) buildConfig.cloneOnServer = true;
if (opts.gitCredentialHelperPath) {
buildConfig.gitCredentialHelperPath = opts.gitCredentialHelperPath;
}
if (opts.gitSsh) buildConfig.gitSsh = opts.gitSsh;
if (opts.gitAmbient) buildConfig.gitAmbient = opts.gitAmbient;
if (opts.runtime.name === "cloud" && !opts.project.localPath) {
opts.logger.log(
`Resolving Dockerfile for compose service "${service.name}" from the build source checkout.\n`,
"info",
{ serviceName: service.name },
);
}
return { service, buildConfig, serviceLogger };
};
// BuildConfig differs by kind:
// - Compose service (Dockerfile in repo) → createDockerfileBuildConfig
// forces stack="docker", clears install/build/start so the runtime
// defers everything to the repo Dockerfile.
// - Monorepo sub-app (source build) → createMonorepoSourceBuildConfig
// keeps the sub-app's stack/installCommand/buildCommand/startCommand/
// outputDirectory so the runtime synthesizes a Dockerfile from them.
const buildSlug = `${sanitizeComposeImageName(opts.project.slug ?? opts.project.name)}-${sanitizeComposeImageName(service.name)}`;
const buildConfig = isMonorepo
? createMonorepoSourceBuildConfig({
project: opts.project,
dep: opts.dep,
snapshot: opts.snapshot,
sessionId: `${opts.buildSessionId}-${service.id}`,
envVars: opts.buildEnvVars,
resources: opts.buildResources,
gitToken: opts.gitToken,
overrides: {
slug: buildSlug,
...resolveSubAppOverrides({
service,
snapshot: opts.snapshot,
logger: serviceLogger,
}),
rootDirectory: context,
port: resolveServicePort(service, opts.snapshot.port) ?? opts.snapshot.port,
// A static sub-app (no start command) serves FILES; a server sub-app
// runs its start command. Derived from framework + start command.
//
// On self-hosted the files are moved to the host and served by the edge
// — no container, no port, no second web server. That is why
// `staticExtractOnly` is gated on the runtime: on CLOUD there is no host
// directory to serve (Oblien runs the workload), so those keep the
// generated nginx image and stay a proxied container.
...(isStaticService(service)
? {
isStatic: true,
hasServer: false,
...(opts.runtime.name === "cloud"
? {}
: {
staticExtractOnly: true,
staticOutDir: `${STATIC_RELEASE_BASE}/.builds/${opts.buildSessionId}-${service.id}`,
}),
}
: { hasServer: true }),
},
})
: createDockerfileBuildConfig({
project: opts.project,
dep: opts.dep,
snapshot: opts.snapshot,
sessionId: `${opts.buildSessionId}-${service.id}`,
envVars: opts.buildEnvVars,
resources: opts.buildResources,
gitToken: opts.gitToken,
overrides: {
slug: buildSlug,
rootDirectory: context,
dockerfilePath: dockerfile ?? undefined,
// Inline catalog build: the source IS the materialized root, so
// prepareSourceTree copies it instead of cloning a repo.
...(inlineBuild ? { localPath: inlineBuild.root } : {}),
hasServer: true,
},
});
// Record a per-service build result: image refs / failures + status broadcast.
const applyBuildResult = (service: Service, buildResult: BuildResult) => {
if (buildResult.status === "failed" || !buildResult.imageRef) {
const failureMessage =
buildResult.errorMessage ?? `Failed to build service "${service.name}"`;
buildFailures.set(service.id, failureMessage);
// Clone-on-server credential, shared across the fan-out (all services share
// one repo): the relay helper (desktop), a per-server ssh key, the server's
// own ambient credentials, or the token already on buildConfig.gitToken.
// Never for an inline build — cloning on the host replaces the shared tree,
// so the materialized context would never reach the build.
if (opts.cloneOnServer && !inlineBuild) buildConfig.cloneOnServer = true;
if (opts.gitCredentialHelperPath) {
buildConfig.gitCredentialHelperPath = opts.gitCredentialHelperPath;
}
if (opts.gitSsh) buildConfig.gitSsh = opts.gitSsh;
if (opts.gitAmbient) buildConfig.gitAmbient = opts.gitAmbient;
return { service, buildConfig, serviceLogger };
};
// Record a per-service build result: image refs / failures + status broadcast.
const applyBuildResult = (service: Service, buildResult: BuildResult) => {
if (buildResult.status === "failed" || !buildResult.imageRef) {
const failureMessage =
buildResult.errorMessage ?? `Failed to build service "${service.name}"`;
buildFailures.set(service.id, failureMessage);
opts.logger.log(
`Compose service "${service.name}" build failed: ${failureMessage}\n`,
"error",
{ serviceName: service.name },
);
sessionManager.broadcastServiceStatus(opts.dep.id, {
serviceName: service.name,
serviceId: service.id,
status: "failed",
error: failureMessage,
});
return;
}
imageRefs.set(service.id, buildResult.imageRef);
builtImageRefs.set(service.id, buildResult.imageRef);
opts.logger.log(
`Compose service "${service.name}" build failed: ${failureMessage}\n`,
"error",
`Compose service "${service.name}" image ready: ${buildResult.imageRef}\n`,
"info",
{ serviceName: service.name },
);
sessionManager.broadcastServiceStatus(opts.dep.id, {
serviceName: service.name,
serviceId: service.id,
status: "failed",
error: failureMessage,
status: "built",
});
return;
}
};
imageRefs.set(service.id, buildResult.imageRef);
builtImageRefs.set(service.id, buildResult.imageRef);
opts.logger.log(
`Compose service "${service.name}" image ready: ${buildResult.imageRef}\n`,
"info",
{ serviceName: service.name },
);
sessionManager.broadcastServiceStatus(opts.dep.id, {
serviceName: service.name,
serviceId: service.id,
status: "built",
});
};
const specs = buildable.map(buildSpecFor);
const specs = buildable.map(buildSpecFor);
if (typeof opts.runtime.buildImages === "function") {
// Docker: clone + prune the shared repo ONCE (transfer once for SSH), then
// build every image from that single tree — sequentially — no per-service
// re-clone. Per-service status/imageRef is applied via onResult the moment
// each build settles (not after the whole batch), so the UI follows the one
// service that's currently building and each tab streams its own output.
await opts.runtime.buildImages(
specs.map((spec) => ({
config: spec.buildConfig,
serviceName: spec.service.name,
logger: spec.serviceLogger,
// Flip to "building" only when THIS image's build starts (post-clone),
// so services stay "pending" through the shared clone/prepare phase.
onStart: () =>
if (typeof opts.runtime.buildImages === "function") {
// Docker: clone + prune the shared repo ONCE (transfer once for SSH), then
// build every image from that single tree — sequentially — no per-service
// re-clone. Per-service status/imageRef is applied via onResult the moment
// each build settles (not after the whole batch), so the UI follows the one
// service that's currently building and each tab streams its own output.
await opts.runtime.buildImages(
specs.map((spec) => ({
config: spec.buildConfig,
serviceName: spec.service.name,
logger: spec.serviceLogger,
// Flip to "building" only when THIS image's build starts (post-clone),
// so services stay "pending" through the shared clone/prepare phase.
onStart: () =>
sessionManager.broadcastServiceStatus(opts.dep.id, {
serviceName: spec.service.name,
serviceId: spec.service.id,
status: "building",
}),
onResult: (result) => applyBuildResult(spec.service, result),
})),
opts.logger,
);
} else {
// Runtimes without a batch build (e.g. cloud — each service is a separate
// instance): build per service, as before.
await Promise.all(
specs.map(async (spec) => {
sessionManager.broadcastServiceStatus(opts.dep.id, {
serviceName: spec.service.name,
serviceId: spec.service.id,
status: "building",
}),
onResult: (result) => applyBuildResult(spec.service, result),
})),
opts.logger,
);
} else {
// Runtimes without a batch build (e.g. cloud — each service is a separate
// instance): build per service, as before.
await Promise.all(
specs.map(async (spec) => {
sessionManager.broadcastServiceStatus(opts.dep.id, {
serviceName: spec.service.name,
serviceId: spec.service.id,
status: "building",
});
applyBuildResult(spec.service, await opts.runtime.build(spec.buildConfig, spec.serviceLogger));
}),
);
}
if (buildable.length > 0) {
// Count of images actually BUILT (builtImageRefs is set only on a build) —
// not imageRefs.size, which also holds external/pull + handed-over images.
const succeeded = builtImageRefs.size;
if (buildFailures.size === 0) {
opts.logger.step(
"build",
"completed",
`All ${succeeded} service image${succeeded === 1 ? "" : "s"} built successfully`,
});
applyBuildResult(
spec.service,
await opts.runtime.build(spec.buildConfig, spec.serviceLogger),
);
}),
);
opts.logger.log("Compose image build phase complete. Preparing deployment phase...\n");
sessionManager.broadcastInstallPhase(opts.dep.id, { id: "images", status: "done" });
} else if (succeeded > 0) {
opts.logger.step(
"build",
"failed",
`Built ${succeeded}/${buildable.length} images, but ${buildFailures.size} failed`,
);
opts.logger.log(
"Compose image build phase failed. Deployment will not continue.\n",
"error",
);
sessionManager.broadcastInstallPhase(opts.dep.id, { id: "images", status: "failed" });
} else {
opts.logger.step(
"build",
"failed",
`All ${buildFailures.size} service image builds failed`,
);
opts.logger.log("Compose image build phase failed. Deployment will not continue.\n", "error");
sessionManager.broadcastInstallPhase(opts.dep.id, { id: "images", status: "failed" });
}
if (buildable.length > 0) {
// Count of images actually BUILT (builtImageRefs is set only on a build) —
// not imageRefs.size, which also holds external/pull + handed-over images.
const succeeded = builtImageRefs.size;
if (buildFailures.size === 0) {
opts.logger.step(
"build",
"completed",
`All ${succeeded} service image${succeeded === 1 ? "" : "s"} built successfully`,
);
opts.logger.log("Compose image build phase complete. Preparing deployment phase...\n");
sessionManager.broadcastInstallPhase(opts.dep.id, { id: "images", status: "done" });
} else if (succeeded > 0) {
opts.logger.step(
"build",
"failed",
`Built ${succeeded}/${buildable.length} images, but ${buildFailures.size} failed`,
);
opts.logger.log(
"Compose image build phase failed. Deployment will not continue.\n",
"error",
);
sessionManager.broadcastInstallPhase(opts.dep.id, { id: "images", status: "failed" });
} else {
opts.logger.step(
"build",
"failed",
`All ${buildFailures.size} service image builds failed`,
);
opts.logger.log(
"Compose image build phase failed. Deployment will not continue.\n",
"error",
);
sessionManager.broadcastInstallPhase(opts.dep.id, { id: "images", status: "failed" });
}
}
} finally {
await inlineBuilds.cleanup();
}
return {
@@ -92,6 +92,9 @@ export function buildCompositeProxyLocations(
}
export interface CompositeRegistration {
/** vercel.json rules the compiler could not reproduce, for the caller to LOG. Nothing
* read `compiled.skipped` before, so a dropped rule vanished without a word. */
skipped: string[];
register: RouteRegister;
frontendServiceId: string;
backendServiceId: string;
@@ -147,6 +150,7 @@ export function buildCompositeRegistration(input: {
return {
frontendServiceId: plan.frontendServiceId,
backendServiceId: plan.backendServiceId,
skipped: compiled?.skipped ?? [],
register: {
hostname: domain.hostname,
isCustomDomain: domain.isCustomDomain,
@@ -155,6 +159,8 @@ export function buildCompositeRegistration(input: {
proxyLocations,
...(compiled?.redirects.length ? { redirects: compiled.redirects } : {}),
...(compiled?.headerRules.length ? { headerRules: compiled.headerRules } : {}),
...(compiled?.cleanUrls ? { cleanUrls: true } : {}),
...(compiled?.trailingSlash === undefined ? {} : { trailingSlash: compiled.trailingSlash }),
},
};
}
@@ -51,6 +51,11 @@ import { isLoopbackHost, resolveServerHost } from "../../../lib/server-target";
import { resolveEdgeTargetHost } from "../../../lib/edge-target";
import { containerIdForService } from "../../services/service-container";
import { isConnectionLoss } from "../../../lib/remote-state";
import {
appConfigHostPath,
withAppConfigHost,
writeAppConfigFile,
} from "./app-config-host";
import {
buildServiceRouteDomains,
createTrackedSslProvider,
@@ -63,7 +68,7 @@ import { resolveServiceEndpointUrls, resolveServicePublicEndpoints } from "../..
import { ensureManagedEdgeProxy } from "../../../lib/managed-edge-proxy";
import { ensureRoutingReady } from "../../../lib/edge-reconcile";
import * as sessionManager from "../session-manager";
import { isStaticService, parseServicePort } from "../../../lib/deployable-service";
import { isStaticService, parseServicePort, serviceAliasExtras } from "../../../lib/deployable-service";
import { computeKeepSet } from "../image-gc";
import { auditPorts } from "../port-audit.service";
import {
@@ -72,8 +77,12 @@ import {
type StabilityTarget,
} from "../stability-audit.service";
import { resolveReadinessGate, type ResolvedReadinessGate } from "../readiness-gate";
import type { PortCheckResult } from "../../../lib/deployment-runtime";
import {
hostChannelDeployNotice,
type PortCheckResult,
} from "../../../lib/deployment-runtime";
import { resolveServicePort } from "./domain-helpers";
import { compileProjectRoutingFields } from "../../../lib/project-routing-fields";
import { buildCompositeRegistration, buildDomainFanoutRegistrations } from "./composite-route";
import { newerThanRestoredRelease, serviceKind } from "./project-services";
import { buildUpstreamUrl, resolveRouteStrategy } from "../../../lib/upstream-url";
@@ -321,17 +330,6 @@ export async function resolvePortOnlyEnvHost(
};
}
/**
* Persistent on-host root for app template config files bind-mounted into
* service containers (Kong's `kong.yml`, Postgres init `.sql`). Sibling of the
* other openship host state (`/var/lib/openship/ssh-keys`); overridable for
* hosts that keep openship state elsewhere. Files land at
* `<root>/<projectId>/<service>/<container-path>` — the executor creates parent
* dirs — and the container-absolute path is appended verbatim so binds are
* unique and self-describing.
*/
const APP_CONFIG_HOST_ROOT = process.env.OPENSHIP_APP_CONFIG_DIR || "/var/lib/openship/app-config";
/**
* Wall-clock ceiling for the whole advisory port audit of a stack.
*
@@ -343,17 +341,6 @@ const APP_CONFIG_HOST_ROOT = process.env.OPENSHIP_APP_CONFIG_DIR || "/var/lib/op
*/
const PORT_AUDIT_BUDGET_MS = 8000;
function appConfigHostPath(projectId: string, serviceName: string, containerPath: string): string {
const safeSvc = serviceName.replace(/[^a-zA-Z0-9._-]/g, "_");
const rel = containerPath.replace(/^\/+/, "");
// Reject `..` traversal: `rel` is a template-supplied path written on the HOST
// (and bind-mounted), so a crafted `../../etc/...` must not escape the root.
if (rel.split("/").some((seg) => seg === "..")) {
throw new Error(`Unsafe app-config path (directory traversal): ${containerPath}`);
}
return `${APP_CONFIG_HOST_ROOT}/${projectId}/${safeSvc}/${rel}`;
}
/** A service's public endpoints as DeployConfig entries (free slug resolved via
* the hostname-label default, custom hostname passed through). */
function serviceDeployPublicEndpoints(
@@ -457,16 +444,6 @@ function resolveServiceResources(
};
}
/** Normalized custom alias(es) for a compose service, drawn from
* `service.advanced.alias`. Returns undefined when unset or when it collapses
* to the service's own name (the default alias already covers that). */
function aliasExtras(service: Service): string[] | undefined {
const raw = (service.advanced as ComposeAdvanced | null | undefined)?.alias;
if (!raw) return undefined;
const alias = normalizeServiceLabel(raw);
if (!alias || alias === normalizeServiceLabel(service.name)) return undefined;
return [alias];
}
function createServiceRuntimeConfig(opts: {
project: Project;
@@ -511,7 +488,7 @@ function createServiceRuntimeConfig(opts: {
// Operator-chosen east-west alias (service.advanced.alias) resolving
// alongside the default service name. Normalized here; skipped when it
// collapses to the service name (no extra alias needed).
extraAliases: aliasExtras(service),
extraAliases: serviceAliasExtras(service),
resources,
expose: service.exposed,
publicPort: resolveServicePublicPort(service),
@@ -672,6 +649,11 @@ export async function deployComposeServices(
* onto the Docker host so they can be bind-mounted read-only into the
* service. Null on cloud (no host bind-mount) → file services are skipped. */
executor?: CommandExecutor | null;
/** The deploy target is the machine this process runs on (`platform.localHost`).
* Host-path writes then go through the host channel instead of `executor`,
* which for a plain local target is a LocalExecutor — the CONTAINER's own
* filesystem on a Compose install. */
localHost?: boolean;
},
): Promise<ComposeDeployResult> {
const services = await repos.service.listByProject(project.id);
@@ -698,6 +680,15 @@ export async function deployComposeServices(
const ordered = topoSort(enabled);
logger.step("deploy", "running", `Deploying ${ordered.length} services...`);
// Say ONCE, up front, that this box can't drive its host (#509). Every host
// touchpoint below absorbs the refusal on its own terms — the port scan reports
// "couldn't read occupancy" (and only under loopback-port routing), the routing
// preflight logs "deploy continues" — so without this the operator's first legible
// symptom is a service that dies later over a config file that never landed.
const hostNotice = hostChannelDeployNotice(opts?.executor);
if (hostNotice) logger.log(`${hostNotice}\n`, "warn");
logger.log("Preparing shared service group for project services...\n");
const group = await runtime.ensureServiceGroup({
@@ -866,6 +857,29 @@ export async function deployComposeServices(
// routeOptions at all.
const proxySettings = sanitizeProxySettings(project.routingConfig?.proxy);
// Compiled ONCE per deploy, for the same reason `proxySettings` is: it is a property of
// the project, so every registration below wants the identical object. The static loop
// recompiled it per route, per service.
const routingFields = compileProjectRoutingFields(project.routingConfig);
// Everything a per-service registration carries that belongs to the PROJECT rather than
// to one upstream. Assembled HERE for the same reason `proxySettings` is, and it is what
// finally gives a PROXIED compose service its vercel.json rules: the static branch below
// spreads `routingFields` onto its own registerRoute call, and the composite path compiles
// its own, but a containerized service routes through `runDeployPipeline` — which only
// ever saw webhook + proxy here, so a redirect declared for the project applied on every
// deploy mode EXCEPT this one.
//
// Safe against the composite: `registerRoute` is last-writer-wins per hostname and the
// composite registers AFTER this loop, so a composite domain still ends up with the
// topology-aware compile (the one that resolves `/api/` to the real backend) rather than
// the backend-free subset here.
const serviceRouteOptions: RouteRegistrationOptions = {
...opts?.routeOptions,
...(proxySettings ? { proxy: proxySettings } : {}),
...routingFields,
};
let routeContext: ServiceRouteContext | undefined;
if (opts?.routing && opts.ssl && typeof opts.usesManagedRouting === "boolean") {
// Reuses the map built above (needsDomainMap covers this branch).
@@ -875,14 +889,10 @@ export async function deployComposeServices(
usesManagedRouting: opts.usesManagedRouting,
organizationId: dep.organizationId,
serverId: opts.serverId,
...(opts.routeOptions || proxySettings
? {
routeOptions: {
...opts.routeOptions,
...(proxySettings ? { proxy: proxySettings } : {}),
},
}
: {}),
// Omitted entirely when there is nothing to carry, so a project with no webhook, no
// proxy tunables and no vercel.json keeps passing `undefined` rather than an empty
// object into the pipeline.
...(Object.keys(serviceRouteOptions).length ? { routeOptions: serviceRouteOptions } : {}),
domainByHostname,
...(proxySettings ? { proxy: proxySettings } : {}),
};
@@ -1011,8 +1021,20 @@ export async function deployComposeServices(
if (alive) {
// A network reconnect may have re-assigned the container's IP, so
// prefer the live values over the stored row when we have them.
const carriedIp = live?.ip ?? carried.ip ?? null;
const carriedHostPort = live?.hostPort ?? carried.hostPort ?? null;
//
// An inspect that ANSWERED is authoritative about whether the container
// publishes anything at all: no binding → CLEAR the row rather than carry
// a port nothing listens on, which is what left migrated workloads routed
// at a dead 127.0.0.1:<port> (#506). Which port it is stays the carried
// (pinned) value — docker's first-binding is ambiguous once the operator
// has declared extra ports. `live === undefined | null` = couldn't ask, so
// the row stays as the last-known value.
const carriedIp = live ? (live.ip ?? null) : (carried.ip ?? null);
const carriedHostPort = live
? live.hostPort === undefined
? null
: (carried.hostPort ?? live.hostPort)
: (carried.hostPort ?? null);
await repos.service.upsertServiceDeployment({
deploymentId: dep.id,
serviceId: svc.id,
@@ -1247,6 +1269,11 @@ export async function deployComposeServices(
route.hostname,
routeContext.domainByHostname.get(routeKey),
),
// registerRoute REPLACES the vhost, so omitting these would not leave the
// project's vercel.json rules alone — it would delete whatever a live
// re-apply installed. This is also the only place a plain static deploy
// ever gets cleanUrls/trailingSlash, so without it the flags never applied.
...routingFields,
...(routeContext.proxy ? { proxy: routeContext.proxy } : {}),
})
.catch((err) => {
@@ -1326,14 +1353,8 @@ export async function deployComposeServices(
// bind-mounts host paths — cloud has neither, so warn-and-skip there.
const advancedFiles = (svc.advanced as ComposeAdvanced | null)?.files ?? [];
if (advancedFiles.length > 0) {
if (runtime.name === "cloud" || !opts?.executor) {
logger.log(
`Service "${svc.name}": ${advancedFiles.length} config file(s) require a self-hosted host mount — skipping on the ${runtime.name} runtime.\n`,
"warn",
{ serviceName: svc.name },
);
} else {
const writer = await resolveHostConfigWriter(opts.executor);
const writeConfigFiles = async (host: CommandExecutor) => {
const writer = await resolveHostConfigWriter(host);
for (const file of advancedFiles) {
// Token-local on purpose: one unresolved URL must not blank the whole
// file (the mount is required — a missing kong.yml is a dead service),
@@ -1348,18 +1369,30 @@ export async function deployComposeServices(
);
}
const hostPath = appConfigHostPath(project.id, svc.name, file.path);
await writer.writeFile(hostPath, resolved.value);
await writeAppConfigFile(writer, hostPath, resolved.value, svc.name, file.path);
serviceRuntimeConfig.volumes = [
...serviceRuntimeConfig.volumes,
`${hostPath}:${file.path}:ro`,
];
}
logger.log(
`Service "${svc.name}": mounted ${advancedFiles.length} generated config file(s).\n`,
"info",
{ serviceName: svc.name },
);
}
};
// Which executor may write them is the whole question — see `withAppConfigHost`.
const { ran } = await withAppConfigHost(
{
executor: opts?.executor,
localHost: opts?.localHost,
isCloud: runtime.name === "cloud",
},
writeConfigFiles,
);
logger.log(
ran
? `Service "${svc.name}": mounted ${advancedFiles.length} generated config file(s).\n`
: `Service "${svc.name}": ${advancedFiles.length} config file(s) require a self-hosted host mount — skipping on the ${runtime.name} runtime.\n`,
ran ? "info" : "warn",
{ serviceName: svc.name },
);
}
const serviceDeployConfig = createServiceDeployConfig({
@@ -1423,9 +1456,24 @@ export async function deployComposeServices(
routedContainerPort !== undefined &&
opts?.executor
) {
servicePinnedHostPort =
previousByServiceId.get(svc.id)?.hostPort ??
(await allocateHostPort(opts.executor, { avoid: usedHostPorts }));
const carried = previousByServiceId.get(svc.id)?.hostPort;
if (carried) {
servicePinnedHostPort = carried;
} else {
const allocation = await allocateHostPort(opts.executor, { avoid: usedHostPorts });
servicePinnedHostPort = allocation.port;
// "Couldn't read occupancy" is not "nothing is listening" — without this the
// bind failure that follows blames Docker for an unreachable host (#490).
if (!allocation.scanned) {
logger.log(
`Couldn't read live port occupancy on the target, so ${allocation.port} for ` +
`${svc.name} avoids only ports this deploy already took. If publishing it fails ` +
`as "already allocated", check that Openship can reach this host ` +
`(Servers → this box).\n`,
"warn",
);
}
}
usedHostPorts.add(servicePinnedHostPort);
serviceRuntimeConfig.ports = withLoopbackPublish(
serviceRuntimeConfig.ports,
@@ -2096,24 +2144,45 @@ export async function deployComposeServices(
r.hostname,
routeContext.domainByHostname.get(r.hostname.toLowerCase()),
),
targetUrl: r.targetUrl!,
// staticRoot wins when present — the same rule route-apply.service applies.
// `buildCompositeRegistration` returns one OR the other, and hardcoding
// targetUrl threw `Invalid proxy target: undefined` for every static-frontend
// composite (which is all of them self-hosted, since build.service sets
// staticExtractOnly whenever the runtime isn't cloud) — so the flagship
// monorepo shape never got a composite vhost at all.
...(r.staticRoot ? { staticRoot: r.staticRoot } : { targetUrl: r.targetUrl! }),
...(r.proxyLocations?.length ? { proxyLocations: r.proxyLocations } : {}),
...(r.redirects?.length ? { redirects: r.redirects } : {}),
...(r.headerRules?.length ? { headerRules: r.headerRules } : {}),
...(r.cleanUrls ? { cleanUrls: true } : {}),
...(r.trailingSlash === undefined ? {} : { trailingSlash: r.trailingSlash }),
...(routeContext.proxy ? { proxy: routeContext.proxy } : {}),
});
logger.log(
`Composed single domain ${r.hostname}: frontend at "/", backend proxied per vercel.json.\n`,
);
// A rule we could not translate is NOT live. Say so here rather than letting it
// disappear — the deploy log is where someone looks after changing vercel.json.
for (const note of composite.skipped) {
logger.log(`vercel.json rule not applied — ${note}\n`, "warn");
}
}
// Re-emit any migration path-fan-out domains (a domain whose paths route to
// DIFFERENT services) from this deploy's live upstreams — persisted on the
// project so a redeploy reproduces `/v3 → api` instead of dropping it.
// These hostnames ARE project domains, so the live path (project-route.service) puts
// the project's vercel.json rules on them. Spread them here too or the deploy would
// strip what a live re-apply installed.
for (const reg of buildDomainFanoutRegistrations({
routes: project.compositeRoutes,
resolveTargetUrl,
})) {
// CONCATENATED, not overwritten: spreading the fan-out's locations after the
// compiled ones would ASSIGN over them, silently dropping a vercel.json external
// rewrite on a path-routed domain. Fan-out first, so its explicit per-path
// upstreams are matched ahead of a broader compiled rule.
const proxyLocations = [...(reg.proxyLocations ?? []), ...(routingFields.proxyLocations ?? [])];
await routeContext.routing.registerRoute({
domain: reg.hostname,
tls: true,
@@ -2122,7 +2191,8 @@ export async function deployComposeServices(
routeContext.domainByHostname.get(reg.hostname.toLowerCase()),
),
targetUrl: reg.targetUrl!,
...(reg.proxyLocations?.length ? { proxyLocations: reg.proxyLocations } : {}),
...routingFields,
...(proxyLocations.length ? { proxyLocations } : {}),
...(routeContext.proxy ? { proxy: routeContext.proxy } : {}),
});
logger.log(
@@ -56,6 +56,9 @@ export interface ComposePipelineOpts {
/** Target host executor (SSH/local) — writes app template config files onto
* the Docker host for read-only bind-mounts. Null on cloud. */
executor: CommandExecutor | null;
/** The target IS this machine (`platform.localHost`) — host-path writes go
* through the host channel, not `executor`. */
localHost?: boolean;
usesManagedRouting: boolean;
logger: BuildLogger;
ctx: LifecycleContext;
@@ -92,6 +95,7 @@ export async function executeComposePipeline(opts: ComposePipelineOpts): Promise
ssl,
system,
executor,
localHost,
usesManagedRouting,
logger,
ctx,
@@ -159,6 +163,7 @@ export async function executeComposePipeline(opts: ComposePipelineOpts): Promise
ssl,
system,
executor,
localHost,
usesManagedRouting,
serverId: snapshot.serverId,
targetServiceIds,
@@ -26,10 +26,7 @@ import type { BuildSessionState } from "./session-manager";
import { failureStatusFor } from "./blocking-errors";
import { sanitizeStorableStrings, sliceWithoutSplittingPair } from "./build-log-sanitize";
import { detectAndStoreFavicon } from "../../lib/favicon-detector";
import {
markWebmailInstalled,
mailServerIdFromWebmailSlug,
} from "../mail/webmail/webmail-project.service";
import { onWebmailDeployed } from "../mail/webmail/webmail-install.service";
/**
* The "your domains didn't route" line for a deploy that otherwise succeeded.
@@ -699,14 +696,8 @@ export async function onSuccess(
void detectAndStoreFavicon(project.id, result.url);
}
// Webmail: flip mail-state `installed=true` so the /emails Open-webmail
// CTA can finally surface. Slug is the only carrier of mailServerId
// through the generic lifecycle - preserved by `ensureWebmailProject`.
// For cloud deploys we also pass `result.url` so the success hook can
// register an OpenResty proxy on the mail VPS pointing mail.<install>
// → opsh.io (when that's the chosen hostname).
if (project.framework === "webmail") {
const mailServerId = mailServerIdFromWebmailSlug(project.slug);
if (mailServerId) void markWebmailInstalled(mailServerId, project.organizationId, result.url);
}
// Webmail on Openship Cloud, routed on the mail server's own `mail.<domain>`:
// the mail VPS proxies that hostname to the cloud URL, which can only be
// registered once the URL exists. Returns immediately for every other deploy.
void onWebmailDeployed(project, result.url);
}
@@ -3,7 +3,7 @@
*/
import { Type, type Static } from "@sinclair/typebox";
import { CloudResourceTierEnum } from "../projects/project.schema";
import { CloudResourceTierEnum, NO_TRAVERSAL_PATTERN } from "../projects/project.schema";
// ─── Route params ────────────────────────────────────────────────────────────
@@ -79,7 +79,7 @@ const BuildServiceInput = Type.Object({
// Source-built (monorepo) sub-app fields — optional, mirror MonorepoSubAppFields.
kind: Type.Optional(Type.Union([Type.Literal("compose"), Type.Literal("monorepo")])),
enabled: Type.Optional(Type.Boolean()),
rootDirectory: Type.Optional(Type.String()),
rootDirectory: Type.Optional(Type.String({ pattern: NO_TRAVERSAL_PATTERN })),
installCommand: Type.Optional(Type.String()),
buildCommand: Type.Optional(Type.String()),
startCommand: Type.Optional(Type.String()),
@@ -11,11 +11,15 @@ import { NotFoundError, ForbiddenError } from "@repo/core";
import type { LogEntry } from "@repo/adapters";
import type { RequestContext } from "../../lib/request-context";
import {
resolveDeploymentRuntime,
resolveDeploymentRuntimeForRead,
deploymentContainerIds,
withDeploymentRuntime,
type DeploymentMeta,
} from "../../lib/deployment-runtime";
import { assertResourceInOrg } from "../../lib/controller-helpers";
import {
assertNotControlPlane,
assertNotControlPlaneById,
assertResourceInOrg,
} from "../../lib/controller-helpers";
import { collectDeploymentManifest, executeCleanup } from "../projects/project-cleanup.service";
import { assertGitHubRepoAccess } from "../github/github-access";
import { maskDeploymentEnv } from "../../lib/secret-env";
@@ -65,17 +69,6 @@ export async function assertGitHubAccessForDeployment(
});
}
async function listServiceContainerIds(deploymentId: string): Promise<string[]> {
const rows = await repos.service.listByDeployment(deploymentId);
return [...new Set(rows.map((row) => row.containerId).filter((id): id is string => !!id))];
}
async function listDeploymentContainerIds(dep: { id: string; containerId?: string | null }) {
const serviceContainerIds = await listServiceContainerIds(dep.id);
if (serviceContainerIds.length > 0) return serviceContainerIds;
return dep.containerId ? [dep.containerId] : [];
}
export async function listDeployments(
organizationId: string,
opts: {
@@ -161,35 +154,25 @@ export async function getDeployment(
return dep;
}
/**
* Refuse a mutating action on the self-deployed control plane's deployment. Its
* row is an ADOPT deployment over the already-running host process (supervised
* by the CLI, not this pipeline) — rollback/restart/delete/pin would be
* meaningless or would detach the live app (clearing activeDeploymentId).
* Build/redeploy is already blocked in build.service (triggerDeployment).
*/
async function assertNotControlPlaneDeployment(dep: { projectId: string }): Promise<void> {
const project = await repos.project.findById(dep.projectId);
if (project?.appTemplateId === "openship") {
throw new ForbiddenError(
"The Openship control plane manages its own runtime — this action isn't available here. Use the CLI.",
);
}
}
// Mutating actions on the self-deployed control plane's deployment are refused by
// the shared policy (`assertNotControlPlane*`, controller-helpers): its row is an
// ADOPT deployment over a host process the CLI supervises, so rollback / restart /
// delete / pin would either be meaningless or detach the live app. Build and
// redeploy are blocked separately, in build.service's triggerDeployment.
export async function deleteDeployment(
deploymentId: string,
organizationId: string,
) {
const dep = await getDeployment(deploymentId, organizationId);
await assertNotControlPlaneDeployment(dep);
const project = await repos.project.findById(dep.projectId);
assertNotControlPlane(project);
if (["queued", "building", "deploying"].includes(dep.status)) {
throw new ForbiddenError("Cannot delete a deployment that is in progress. Cancel it first.");
}
const project = await repos.project.findById(dep.projectId);
// protectRetained: a compose service that didn't change carries its container
// and image onto later releases, so this release's rows can name artifacts a
// retained (or the live) release still needs.
@@ -223,7 +206,7 @@ export async function rollbackDeployment(
) {
// Existence + org-scope check (throws if deployment isn't in this org).
const dep = await getDeployment(deploymentId, organizationId);
await assertNotControlPlaneDeployment(dep);
await assertNotControlPlaneById(dep.projectId);
await rollback(deploymentId);
// Return the post-rollback deployment row (now with any updated container id).
return (await repos.deployment.findById(dep.id)) ?? dep;
@@ -238,7 +221,7 @@ export async function rollbackDeployment(
*/
export async function previewRestore(deploymentId: string, organizationId: string) {
const dep = await getDeployment(deploymentId, organizationId);
await assertNotControlPlaneDeployment(dep);
await assertNotControlPlaneById(dep.projectId);
const { target, project, plan } = await resolveRestorePlan(deploymentId);
const consequences =
plan.mode === "ineligible"
@@ -346,7 +329,7 @@ export async function setDeploymentPin(
pinned: boolean,
) {
const dep = await getDeployment(deploymentId, organizationId);
await assertNotControlPlaneDeployment(dep);
await assertNotControlPlaneById(dep.projectId);
await setPin(deploymentId, pinned);
return (await repos.deployment.findById(dep.id)) ?? dep;
}
@@ -539,9 +522,9 @@ export async function getDeploymentLogs(
return buildSessions.logs as LogEntry[];
}
if (dep.containerId) {
const { runtime } = await resolveDeploymentRuntime(dep);
return runtime.getRuntimeLogs(dep.containerId, tail);
const containerId = dep.containerId;
if (containerId) {
return withDeploymentRuntime(dep, (runtime) => runtime.getRuntimeLogs(containerId, tail));
}
return [];
@@ -552,20 +535,21 @@ export async function restartDeployment(
organizationId: string,
) {
const dep = await getDeployment(deploymentId, organizationId);
await assertNotControlPlaneDeployment(dep);
await assertNotControlPlaneById(dep.projectId);
if (dep.status !== "ready") {
throw new ForbiddenError("Can only restart a running deployment");
}
const containerIds = await listDeploymentContainerIds(dep);
const containerIds = await deploymentContainerIds(dep);
if (containerIds.length === 0) {
throw new ForbiddenError("Deployment has no container");
}
const { runtime } = await resolveDeploymentRuntime(dep);
for (const containerId of containerIds) {
await runtime.restart(containerId);
}
await withDeploymentRuntime(dep, async (runtime) => {
for (const containerId of containerIds) {
await runtime.restart(containerId);
}
});
return dep;
}
@@ -578,12 +562,8 @@ export async function getContainerInfo(
if (!dep.containerId) {
throw new ForbiddenError("Deployment has no container");
}
const { runtime } = await resolveDeploymentRuntimeForRead(dep);
try {
return await runtime.getContainerInfo(dep.containerId);
} finally {
void Promise.resolve(runtime.dispose?.()).catch(() => {});
}
const containerId = dep.containerId;
return withDeploymentRuntime(dep, (runtime) => runtime.getContainerInfo(containerId));
}
/**
@@ -607,12 +587,8 @@ export async function getContainerUsage(
if (!dep.containerId) {
throw new ForbiddenError("Deployment has no container");
}
const { runtime } = await resolveDeploymentRuntimeForRead(dep);
try {
return await runtime.getUsage(dep.containerId);
} finally {
void Promise.resolve(runtime.dispose?.()).catch(() => {});
}
const containerId = dep.containerId;
return withDeploymentRuntime(dep, (runtime) => runtime.getUsage(containerId));
}
export async function getBuildLogs(
@@ -23,6 +23,9 @@ import {
cloudRequiredCode,
CLOUD_UNREACHABLE_CODE,
stackExpectsBuildCommand,
describeResourceFit,
fitsCapacity,
hasMinResources,
} from "@repo/core";
import { cloudClient } from "../../lib/cloud/client";
import { isCloudConnectedForOrg } from "../../lib/cloud/session";
@@ -52,6 +55,8 @@ import { canResolveServerGitCredential } from "../github/server-github.service";
import { parseRepoUrl } from "../github/github.service";
import { resolveRecords, lookupAddresses } from "../../lib/dns-resolver";
import { type RequestContext } from "../../lib/request-context";
import { getTrustedHostCapacity } from "../../lib/host-capacity";
import { getTemplateForOrg } from "../apps/catalog-source";
import { repos } from "@repo/db";
/**
@@ -107,6 +112,10 @@ export const PREFLIGHT_ERROR_CODES = {
* uses at deploy time. Failing here surfaces the missing-credential
* modal up-front instead of letting the build pipeline fail later. */
GITHUB_REMOTE_TOKEN_REQUIRED: "GITHUB_REMOTE_TOKEN_REQUIRED",
/** The target machine is measurably smaller than the app's declared
* `minResources` (catalog). Only ever raised for a FIRST deploy — see
* `checkHostCapacity`. */
HOST_RESOURCES_INSUFFICIENT: "HOST_RESOURCES_INSUFFICIENT",
} as const;
export interface PreflightResult {
@@ -148,6 +157,13 @@ export interface PreflightOptions {
* target (`server`). For non-App auth modes, only `local` keeps the
* user's broad-scope token from leaving the API process. */
buildStrategy?: "local" | "server";
/** Catalog app this project is an instance of (`project.appTemplateId`), so the
* app's declared host minimum can be matched against the target machine. Null
* for an ordinary project — nothing is declared, nothing is checked. */
appTemplateId?: string | null;
/** True when the project has never had a live deployment. A shortfall FAILS a
* first deploy and only warns afterwards — see `checkHostCapacity`. */
firstDeploy?: boolean;
}
/** Resolve owner/repo for the public-ness probe: prefer the already-parsed
@@ -1335,6 +1351,57 @@ async function checkCloudRuntime(
};
}
/**
* Match a catalog app's declared `minResources` against the machine it is about
* to be installed on. Generic: any app that declares a minimum gets this, and an
* app that declares none (almost all of them) is never checked.
*
* Two rules keep it from being a footgun of its own:
*
* • It FAILS a first deploy and only WARNS afterwards. Refusing a redeploy
* would brick an app already running on a box that turned out to be
* undersized — the operator's way out of that is a deploy, not a refusal.
* • An unknown capacity never fails (`fitsCapacity`). A box we couldn't probe
* means we didn't look, not that the hardware is too small.
*
* Returns null when there is nothing to check, so no cosmetic row appears on the
* checklist of an ordinary project.
*/
async function checkHostCapacity(
organizationId: string,
appTemplateId: string,
serverId: string | undefined,
isLocalTarget: boolean,
firstDeploy: boolean,
): Promise<PreflightCheck | null> {
const template = await getTemplateForOrg(organizationId, appTemplateId).catch(() => undefined);
const min = template?.minResources;
if (!template || !hasMinResources(min)) return null;
const capacity = await getTrustedHostCapacity(serverId, organizationId, { isLocalTarget });
const fit = fitsCapacity(min, capacity);
const check: PreflightCheck = { id: "host-capacity", label: "Host capacity", status: "pass" };
if (fit.ok) return check;
const shortfall = describeResourceFit(fit);
if (firstDeploy) {
return {
...check,
status: "fail",
code: PREFLIGHT_ERROR_CODES.HOST_RESOURCES_INSUFFICIENT,
message: `${template.name} needs ${shortfall}. Install it on a bigger machine, or pick a different destination.`,
};
}
// Written for the day warns are surfaced: today `runDeploymentPreflight` acts
// only on `!ok`, so every preflight warn's message is dropped. The status is
// what matters here — it keeps this off the failure path.
return {
...check,
status: "warn",
message: `${template.name} needs ${shortfall}. It will deploy, but expect it to be slow or OOM-killed on this machine.`,
};
}
export async function runPreflightChecks(
snapshot: DeploymentConfigSnapshot,
opts?: PreflightOptions,
@@ -1377,6 +1444,20 @@ export async function runPreflightChecks(
: checkStack(snapshot),
];
// Does this machine meet what the app says it needs? Cloud is sized from the
// tier table, not from host hardware, so there is nothing to match there (and
// nothing to probe — a multi-tenant control plane must not dial a tenant's box).
if (opts?.appTemplateId && snapshot.organizationId && effectiveTarget !== "cloud") {
const hostCapacity = await checkHostCapacity(
snapshot.organizationId,
opts.appTemplateId,
snapshot.serverId,
effectiveTarget === "local",
opts.firstDeploy ?? false,
);
if (hostCapacity) checks.push(hostCapacity);
}
if (!hasEndpointRouting && opts?.slug && !opts?.customDomain) {
checks.push(checkSlugFormat(opts.slug));
const fqdn = `${opts.slug}.${getRoutingBaseDomain()}`.toLowerCase();
@@ -22,7 +22,7 @@
import { repos, type Deployment } from "@repo/db";
import { safeErrorMessage } from "@repo/core";
import { resolveDeploymentRuntime } from "../../lib/deployment-runtime";
import { disposeRuntime, resolveDeploymentRuntime } from "../../lib/deployment-runtime";
import { createReachabilityProbe } from "../../lib/server-reachability";
import { isConnectionLoss } from "../../lib/remote-state";
@@ -60,7 +60,16 @@ export async function reconcileDeployment(deploymentId: string): Promise<Reconci
// Server-backed: fast-fail if the host still isn't answering — leave the
// deployment `reconciling` for the next tick rather than guessing.
//
// Org-checked BEFORE the probe: `meta.serverId` is a client-supplied snapshot value
// and the probe resolves the row unscoped, so without this a deploy body could aim a
// TCP dial at another org's host. A foreign id takes the same conservative branch as
// a deleted one (below at the runtime resolve) — unreachable, never a guess.
if (!isCloud && serverId) {
const inOrg = await repos.server
.getInOrganization(serverId, dep.organizationId)
.catch(() => undefined);
if (!inOrg) return "unreachable";
const probe = createReachabilityProbe();
if (!(await probe.isReachable(serverId))) return "unreachable";
}
@@ -76,104 +85,111 @@ export async function reconcileDeployment(deploymentId: string): Promise<Reconci
return "unreachable";
}
// Bare runtime can't inspect containers by id — leave reconciling; the
// one-active-per-project index excludes reconciling so a redeploy can land.
if (!runtime.supports("containerInfo")) return "unsupported";
// The resolved runtime holds a Docker-over-SSH loopback bridge for a remote
// server, and this runs on a SCHEDULE for every reconciling deployment — so a
// missed release accumulates one listener per tick, forever.
try {
// Bare runtime can't inspect containers by id — leave reconciling; the
// one-active-per-project index excludes reconciling so a redeploy can land.
if (!runtime.supports("containerInfo")) return "unsupported";
// Inspect targets: every service container, or the single-app container.
const serviceDeps = await repos.serviceDeployment.listByDeployment(dep.id);
const targets = serviceDeps
.filter((sd) => sd.containerId)
.map((sd) => ({
rowId: sd.id,
containerId: sd.containerId as string,
name: sd.serviceName ?? undefined,
isService: true,
}));
if (targets.length === 0 && dep.containerId && dep.containerId !== "compose") {
targets.push({ rowId: dep.id, containerId: dep.containerId, name: undefined, isService: false });
}
// Services that terminally FAILED with no container (e.g. build error) count
// toward the verdict as "down" — otherwise a mix of build-failure +
// connection-loss could wrongly resolve to "ready" and mask the failure.
// `skipped` (unchanged, carried-forward) rows are NOT failures and are excluded.
const failedNoContainer = serviceDeps.filter((sd) => {
// `failed` vs `failure` — the compose catch writes "failed" while the repo
// union says "failure"; match both. Cast because "failed" isn't in the union.
const s = sd.status as string;
return !sd.containerId && (s === "failure" || s === "failed" || s === "cancelled");
}).length;
if (targets.length === 0) {
await repos.deployment.updateStatus(dep.id, "failed", {
errorMessage: "Reconcile found no containers to verify.",
});
return "finalized";
}
const missing: string[] = [];
let up = 0;
for (const t of targets) {
let state: "running" | "missing" | "down";
try {
const info = await runtime.getContainerInfo(t.containerId);
state = info.status === "running" ? "running" : info.status === "missing" ? "missing" : "down";
} catch (err) {
// A connection error mid-inspect means the host went away again — abort
// the whole reconcile and retry later rather than recording half-truths.
if (isConnectionLoss(err)) return "unreachable";
state = "down";
// Inspect targets: every service container, or the single-app container.
const serviceDeps = await repos.serviceDeployment.listByDeployment(dep.id);
const targets = serviceDeps
.filter((sd) => sd.containerId)
.map((sd) => ({
rowId: sd.id,
containerId: sd.containerId as string,
name: sd.serviceName ?? undefined,
isService: true,
}));
if (targets.length === 0 && dep.containerId && dep.containerId !== "compose") {
targets.push({ rowId: dep.id, containerId: dep.containerId, name: undefined, isService: false });
}
if (state === "running") up++;
if (state === "missing") missing.push(t.name ?? t.containerId.slice(0, 12));
// Services that terminally FAILED with no container (e.g. build error) count
// toward the verdict as "down" — otherwise a mix of build-failure +
// connection-loss could wrongly resolve to "ready" and mask the failure.
// `skipped` (unchanged, carried-forward) rows are NOT failures and are excluded.
const failedNoContainer = serviceDeps.filter((sd) => {
// `failed` vs `failure` — the compose catch writes "failed" while the repo
// union says "failure"; match both. Cast because "failed" isn't in the union.
const s = sd.status as string;
return !sd.containerId && (s === "failure" || s === "failed" || s === "cancelled");
}).length;
if (t.isService) {
await repos.serviceDeployment
.update(t.rowId, {
status: state === "running" ? "success" : state === "missing" ? "missing" : "failure",
})
.catch(() => {});
if (targets.length === 0) {
await repos.deployment.updateStatus(dep.id, "failed", {
errorMessage: "Reconcile found no containers to verify.",
});
return "finalized";
}
}
const total = targets.length + failedNoContainer;
const verdict = up === total ? "ready" : up > 0 ? "partial_failure" : "failed";
const missing: string[] = [];
let up = 0;
for (const t of targets) {
let state: "running" | "missing" | "down";
try {
const info = await runtime.getContainerInfo(t.containerId);
state = info.status === "running" ? "running" : info.status === "missing" ? "missing" : "down";
} catch (err) {
// A connection error mid-inspect means the host went away again — abort
// the whole reconcile and retry later rather than recording half-truths.
if (isConnectionLoss(err)) return "unreachable";
state = "down";
}
const nextMeta: Record<string, unknown> = { ...meta };
if (missing.length > 0) {
nextMeta.drift = {
missingContainers: missing,
detectedAt: new Date().toISOString(),
serverId: serverId ?? null,
} satisfies DeploymentDrift;
} else {
delete nextMeta.drift;
}
if (state === "running") up++;
if (state === "missing") missing.push(t.name ?? t.containerId.slice(0, 12));
if (verdict === "failed") {
// Forward-only: a failed reconcile NEVER advances the project pointer.
await repos.deployment.updateStatus(dep.id, "failed", { meta: nextMeta });
if (t.isService) {
await repos.serviceDeployment
.update(t.rowId, {
status: state === "running" ? "success" : state === "missing" ? "missing" : "failure",
})
.catch(() => {});
}
}
const total = targets.length + failedNoContainer;
const verdict = up === total ? "ready" : up > 0 ? "partial_failure" : "failed";
const nextMeta: Record<string, unknown> = { ...meta };
if (missing.length > 0) {
nextMeta.drift = {
missingContainers: missing,
detectedAt: new Date().toISOString(),
serverId: serverId ?? null,
} satisfies DeploymentDrift;
} else {
delete nextMeta.drift;
}
if (verdict === "failed") {
// Forward-only: a failed reconcile NEVER advances the project pointer.
await repos.deployment.updateStatus(dep.id, "failed", { meta: nextMeta });
return "finalized";
}
if (verdict === "partial_failure") {
// Hold for an explicit keep/reject decision — same as the normal compose
// finalize path — so a connection-loss deploy that reconciles to a partial
// failure can't silently read as a clean "Deployed".
const existingCompose =
(nextMeta.composeDeployment as Record<string, unknown> | undefined) ?? {};
nextMeta.composeDeployment = { ...existingCompose, decision: "pending" };
}
await repos.deployment.updateStatus(dep.id, verdict, { errorMessage: null, meta: nextMeta });
const project = await repos.project.findById(dep.projectId);
if (project && !(await isSuperseded(project.activeDeploymentId, dep))) {
await repos.project.setActiveDeployment(project.id, dep.id);
}
return "finalized";
} finally {
disposeRuntime(runtime);
}
if (verdict === "partial_failure") {
// Hold for an explicit keep/reject decision — same as the normal compose
// finalize path — so a connection-loss deploy that reconciles to a partial
// failure can't silently read as a clean "Deployed".
const existingCompose =
(nextMeta.composeDeployment as Record<string, unknown> | undefined) ?? {};
nextMeta.composeDeployment = { ...existingCompose, decision: "pending" };
}
await repos.deployment.updateStatus(dep.id, verdict, { errorMessage: null, meta: nextMeta });
const project = await repos.project.findById(dep.projectId);
if (project && !(await isSuperseded(project.activeDeploymentId, dep))) {
await repos.project.setActiveDeployment(project.id, dep.id);
}
return "finalized";
}
// De-dupe concurrent on-demand reconciles (e.g. rapid deployment-detail loads)
@@ -247,19 +247,22 @@ async function resolveEffectiveServiceImages(
*/
async function hostPathExists(target: Deployment, path: string): Promise<boolean> {
const serverId = (target.meta as { serverId?: string } | null)?.serverId;
const { createExecutor, sharedMountExecutor } = await import("@repo/adapters");
const { resolveServerExecutor } = await import("../../../lib/deployment-runtime");
try {
const { executor } = await resolveServerExecutor(serverId, target.organizationId);
return await executor.exists(path);
} catch (err) {
const { executor, isLocal } = await resolveServerExecutor(serverId, target.organizationId);
// The static tree is a mount this process shares 1:1 with its host, so on the local
// box read it directly — the same rule the promote half already uses, and asking the
// host channel instead let a firewall answer "reclaimed" about a release sitting
// right there (#490).
const exec = await sharedMountExecutor({ localHost: isLocal, executor });
return exec ? await exec.exists(path) : false;
} catch {
if (serverId) return false; // a real server we couldn't reach → assume gone, rebuild
try {
const { createHostExecutor } = await import("@repo/adapters");
return await createHostExecutor().exists(path);
} catch {
void err;
return false;
}
// No server recorded: a desktop/local release, same shared tree.
return await createExecutor()
.exists(path)
.catch(() => false);
}
}
@@ -8,9 +8,8 @@ vi.mock("../../lib/notification-dispatcher", () => ({ notification: {} }));
vi.mock("../../lib/audit", () => ({ audit: {} }));
vi.mock("../../lib/favicon-detector", () => ({ detectAndStoreFavicon: vi.fn() }));
vi.mock("./session-manager", () => ({}));
vi.mock("../mail/webmail/webmail-project.service", () => ({
markWebmailInstalled: vi.fn(),
mailServerIdFromWebmailSlug: vi.fn(),
vi.mock("../mail/webmail/webmail-install.service", () => ({
onWebmailDeployed: vi.fn(),
}));
import { routeIssuesWarning } from "./deployment-lifecycle";
@@ -2,6 +2,7 @@ import { repos, type Project, type Deployment } from "@repo/db";
import { isServiceSuccessStatus, isServiceFailureStatus } from "@repo/core";
import { runtimeTarget } from "../../config";
import { buildBackgroundContext } from "../../lib/request-context";
import { resolveOrgOwner } from "../../lib/org-actor";
import { createCheckRun, updateCheckRun } from "../github/github.service";
// Per-service GitHub-Checks + service_deployment fan-out for a multi-service
@@ -97,13 +98,15 @@ export async function emitServiceCheckRun(opts: {
const { project, dep, serviceDeploymentId, serviceName, phase, conclusion, output } = opts;
if (!project.gitOwner || !project.gitRepo || !dep.commitSha) return;
const orgMembers = await repos.member
.listByOrganization(dep.organizationId)
.catch(() => [] as Array<{ userId: string }>);
const actorUserId = orgMembers[0]?.userId;
if (!actorUserId) return;
// Act as the org OWNER, not `members[0]`. A check run is Openship reporting on a
// build it already ran — not a member action — but the GitHub authorization gate
// resolves the actor's real role from the DB, so an arbitrary first member
// (ordering is unspecified) made this feature work or silently vanish depending
// on who happened to sort first and what repos they were granted.
const actor = await resolveOrgOwner(dep.organizationId).catch(() => null);
if (!actor?.userId) return;
const actorCtx = buildBackgroundContext({
userId: actorUserId,
userId: actor.userId,
organizationId: dep.organizationId,
label: "build:check-run",
});
+16 -11
View File
@@ -33,7 +33,11 @@ import { generateToken } from "../../lib/domain-token";
import { untrackedSiteFor } from "../../lib/edge-orphans.service";
import { publicEndpointHostname, resolveServicePublicEndpoints } from "../../lib/public-endpoints";
import { sshManager } from "../../lib/ssh-manager";
import { resolveServerIdForProject, withServerHostExecutor } from "../../lib/edge-host-executor";
import {
resolveServerIdForProject,
withCertStoreExecutor,
withServerHostExecutor,
} from "../../lib/edge-host-executor";
import {
assertRedirectSupported,
assertRedirectTargets,
@@ -534,14 +538,13 @@ function isPathSafeHostname(hostname: string): boolean {
* box, or a foreign reverse proxy (nginx/caddy/apache/traefik, bare OR container)
* we're taking over — adopt the cert that's already there instead of re-issuing via
* ACME (which fails behind Cloudflare, or when the cert isn't at certbot's standard
* path). Sources, in order, all read on the HOST executor so it works when the API
* is containerized:
* path). Sources, in order, each read on the executor that can actually see it:
* 1. certbot's /etc/letsencrypt on the serving host, via the platform provider
* (verifyExistingCert).
* 2. the host's certbot lineage dir read directly on the HOST executor — the
* bare-edge case where the API container's own /etc/letsencrypt is a
* different volume. Includes the `-0001` re-issue lineages, which a bare
* `live/<host>` lookup misses entirely.
* 2. certbot's lineage dirs read as plain files. On the local box that store is a
* 1:1 bind mount in the api container, so this reads it THERE and needs no host
* channel (`withCertStoreExecutor`). Includes the `-0001` re-issue lineages,
* which a bare `live/<host>` lookup misses entirely.
* 3. whatever the edge proxy itself serves, via `edgeProxy().certFor()` — our
* OpenResty at a non-standard path, an nginx/apache declared path, caddy's
* own cert store, or traefik's acme.json.
@@ -609,11 +612,13 @@ export async function reuseServerCertForDomain(ctx: RequestContext, domainId: st
const rejections: string[] = [];
// 2. Read the host's certbot store directly on the HOST executor — covers a
// bare-metal edge whose certs live on the host while the API container's own
// /etc/letsencrypt is a separate, empty volume.
// 2. Read certbot's store directly — covers a bare-metal edge whose certs live on
// the host. On the local box that store is a 1:1 bind mount in the api
// container, so this must NOT go over the host channel: a firewalled or
// switched-off channel would make an adoptable cert unreadable and send the
// domain to ACME instead (#490). See `withCertStoreExecutor`.
if (isPathSafeHostname(domain.hostname)) {
const hostCert = await withServerHostExecutor(project, async (exec) => {
const hostCert = await withCertStoreExecutor(project, async (exec) => {
for (const base of await certbotLineageDirs(exec, domain.hostname)) {
const certPem = await readEdgeFile(exec, `${base}/fullchain.pem`);
const keyPem = await readEdgeFile(exec, `${base}/privkey.pem`);
@@ -24,7 +24,7 @@ import {
type CommandExecutor,
} from "@repo/adapters";
import { getRequestContext } from "../../lib/request-context";
import { resolveDeploymentRuntime } from "../../lib/deployment-runtime";
import { withDeploymentPlatform } from "../../lib/deployment-runtime";
import { ensureEdgeChallengeReady } from "../../lib/edge-challenge";
import { permission } from "../../lib/permission";
import { param } from "../../lib/controller-helpers";
@@ -256,11 +256,12 @@ export async function ensureEdgeStream(c: Context) {
await (async () => {
const dep = await repos.deployment.findById(resolved.project.activeDeploymentId!);
if (!dep) return;
const { routing } = await resolveDeploymentRuntime(dep);
await ensureEdgeChallengeReady(ctx.organizationId, routing, {
serverId,
onLog: (m) => appendEdgeLog(session.id, m.trim(), "warn"),
});
await withDeploymentPlatform(dep, ({ routing }) =>
ensureEdgeChallengeReady(ctx.organizationId, routing, {
serverId,
onLog: (m) => appendEdgeLog(session.id, m.trim(), "warn"),
}),
);
})().catch(() => {});
await applyProjectRouting(id).catch((e) =>
appendEdgeLog(session.id, `Route apply warning: ${safeErrorMessage(e)}`, "warn"),
@@ -1,6 +1,7 @@
import { repos, type Domain, type Project } from "@repo/db";
import { safeErrorMessage } from "@repo/core";
import { resolveServedStaticPath } from "@repo/adapters";
import { compileProjectRoutingFields } from "../../lib/project-routing-fields";
import {
isLoopbackHost,
isReservedLoopbackPort,
@@ -11,11 +12,16 @@ import {
type StoredPublicEndpoint,
} from "../../lib/public-endpoints";
import { assertValidCustomDomain, assertValidCustomDomains } from "../../lib/custom-domain-guard";
import { resolveUpstreamUrl, resolveRouteStrategy } from "../../lib/upstream-url";
import { resolveLiveUpstreamUrl, resolveRouteStrategy } from "../../lib/upstream-url";
import { deregisterManagedEdgeRoutes, syncManagedEdgeRoutes } from "../../lib/managed-edge-proxy";
import { syncProjectPublicRoutes } from "../../lib/project-route-store";
import { resolveRouteRedirect } from "../../lib/domain-redirect";
import { resolveDeploymentRuntime, resolveDeploymentStaticRoot } from "../../lib/deployment-runtime";
import {
disposePlatform,
resolveDeploymentPlatform,
resolveDeploymentStaticRoot,
type DeploymentMeta,
} from "../../lib/deployment-runtime";
import { pushProjectRules } from "../route-rules/route-rule.service";
import { pushProjectAnalyticsConfig } from "../analytics/analytics-config.service";
import {
@@ -391,164 +397,194 @@ export async function reapplyProjectLiveRoutes(
);
return;
}
const { routing, runtime, effectiveTarget, serverId } =
await resolveDeploymentRuntime(deployment);
// Register the managed (*.opsh.io) hostnames that are NEW in this edit on
// Openship Cloud's edge — the "add" half. Oblien's edge has NO route EDIT
// (only sync + deregister), so a slug change is drop-old (deregistered above)
// + add-new (here). PER-ROUTE by design: only hostnames absent from
// `previousHostnames` are synced — symmetric with the dropped-slug deregister
// above — so editing ONE route never re-hits Oblien (or re-resolves the target
// host) for the project's OTHER, unchanged routes. A target-host change on an
// UNCHANGED hostname (e.g. a server move) is re-synced by the deploy path, not
// here. Best-effort/fire-and-forget: the app is live locally; a failure only
// delays the free URL (same contract as the deploy path's sync).
const previouslyPresent = new Set(previousHostnames.map((h) => h.toLowerCase()));
const syncAddedManagedEdge = () => {
if (opts.managedEdgeSyncedByCaller) return;
// NOT filtered by target kind. The edge route is `<slug>.opsh.io` → this
// server's :80; what the vhost then does with the request — proxy to a
// container or serve files — is decided locally and is none of Cloud's
// business. Excluding path targets here meant a free domain on a STATIC
// project registered nothing on the edge, so the URL resolved to the
// wildcard with no origin: the free domain worked for proxied projects and
// silently did nothing for static ones.
const addedTargets = current
.filter((d) => !previouslyPresent.has(d.hostname.toLowerCase()))
.map((d) => ({ hostname: d.hostname, subdomain: managedHostnameToSlug(d.hostname) }))
.filter((t): t is { hostname: string; subdomain: string } => !!t.subdomain);
if (addedTargets.length === 0) return;
void syncManagedEdgeRoutes(addedTargets, {
organizationId: project.organizationId,
serverId: serverId ?? undefined,
})
.then(({ failures }) => {
if (failures.length > 0) {
console.warn(
`[project-route] ${project.slug}: managed edge sync failed for ${failures.join(", ")}`,
);
}
// Held for the `finally` below: a remote-server platform binds a
// Docker-over-SSH loopback bridge that only `dispose` closes, and this runs on
// every live route edit. Releasing it leaves `routing` fully usable — dispose
// touches the docker transport, while routing drives the box through the pooled
// SSH executor.
const resolved = await resolveDeploymentPlatform((deployment.meta ?? {}) as DeploymentMeta, {
organizationId: deployment.organizationId,
});
const { routing, runtime } = resolved.platform;
const { effectiveTarget, serverId } = resolved;
try {
// Register the managed (*.opsh.io) hostnames that are NEW in this edit on
// Openship Cloud's edge — the "add" half. Oblien's edge has NO route EDIT
// (only sync + deregister), so a slug change is drop-old (deregistered above)
// + add-new (here). PER-ROUTE by design: only hostnames absent from
// `previousHostnames` are synced — symmetric with the dropped-slug deregister
// above — so editing ONE route never re-hits Oblien (or re-resolves the target
// host) for the project's OTHER, unchanged routes. A target-host change on an
// UNCHANGED hostname (e.g. a server move) is re-synced by the deploy path, not
// here. Best-effort/fire-and-forget: the app is live locally; a failure only
// delays the free URL (same contract as the deploy path's sync).
const previouslyPresent = new Set(previousHostnames.map((h) => h.toLowerCase()));
const syncAddedManagedEdge = () => {
if (opts.managedEdgeSyncedByCaller) return;
// NOT filtered by target kind. The edge route is `<slug>.opsh.io` → this
// server's :80; what the vhost then does with the request — proxy to a
// container or serve files — is decided locally and is none of Cloud's
// business. Excluding path targets here meant a free domain on a STATIC
// project registered nothing on the edge, so the URL resolved to the
// wildcard with no origin: the free domain worked for proxied projects and
// silently did nothing for static ones.
const addedTargets = current
.filter((d) => !previouslyPresent.has(d.hostname.toLowerCase()))
.map((d) => ({ hostname: d.hostname, subdomain: managedHostnameToSlug(d.hostname) }))
.filter((t): t is { hostname: string; subdomain: string } => !!t.subdomain);
if (addedTargets.length === 0) return;
void syncManagedEdgeRoutes(addedTargets, {
organizationId: project.organizationId,
serverId: serverId ?? undefined,
})
.catch(() => {});
};
const containerId = deployment.containerId;
if (!containerId) {
// Compose/multi-service deployments track containers per-service, so the
// parent deployment row has no containerId — nothing to point a single-app
// route at (per-service routes are handled in updateService). Still tear
// down any dropped hostnames on the correct host.
console.warn(
`[project-route] ${project.slug}: deployment ${deployment.id} has no containerId (target=${effectiveTarget}) — skipping single-app route registration`,
);
await reconcileProjectRoutes(project, { routing, removes });
await pushProjectRules(project.id, serverId ?? null, previousHostnames).catch(() => {});
// Shared-dict state is RAM: the analytics collection switches have to be re-pushed
// whenever routing is applied, or an nginx restart silently reverts them to off.
await pushProjectAnalyticsConfig(project.id, serverId ?? null, previousHostnames).catch(() => {});
syncAddedManagedEdge();
return;
}
const resolveTargetUrl = async (port: number): Promise<string | null> => {
const strategy = resolveRouteStrategy(project.routeStrategy);
// loopback-port: dial the container's published loopback host port (read
// live). Bare / no-host-port fall back to container IP (or 127.0.0.1 bare).
let hostPort: number | undefined;
if (strategy === "loopback-port" && runtime.name !== "bare") {
hostPort = (await runtime.getContainerInfo?.(containerId).catch(() => null))?.hostPort ?? undefined;
}
const url = await resolveUpstreamUrl({ strategy, runtime, containerId, containerPort: port, hostPort });
if (!url) {
console.warn(
`[project-route] ${project.slug}: could not resolve upstream for ${containerId} (target=${effectiveTarget}, server=${serverId ?? "local"})`,
);
return null;
}
// Never proxy a public route at a reserved control-plane/mgmt port on the
// host loopback — that would expose the admin API (env.PORT) or the
// unauthenticated OpenResty mgmt port (9145). Only guards loopback: a
// container's own bridge IP:<port> is the app's, not ours. The self-app is
// exempt (see ReapplyProjectLiveRoutesOptions.isSelfApp).
const m = url.match(/^https?:\/\/([^:/]+):(\d+)$/);
if (m && shouldRefuseLoopbackRoute(m[1], Number(m[2]), opts)) {
console.warn(
`[project-route] ${project.slug}: refusing reserved loopback upstream port ${m[2]} for a public route`,
);
return null;
}
return url;
};
// Where a path-targeted (static) domain serves its files from — the SAME
// resolver the post-deploy output probe uses, so the vhost and the check that
// audits it can never disagree about the directory.
const staticRootBase = resolveDeploymentStaticRoot(deployment, project);
// A redirect only goes live when its target is one of the hostnames this
// project currently routes — see resolveRouteRedirect.
const liveHostnames = current.map((domain) => domain.hostname);
const registers: RouteRegister[] = [];
for (const domain of current) {
const redirectHost = resolveRouteRedirect(domain, liveHostnames);
const common = {
hostname: domain.hostname,
isCustomDomain: domain.domainType === "custom",
...(redirectHost ? { redirectHost } : {}),
.then(({ failures }) => {
if (failures.length > 0) {
console.warn(
`[project-route] ${project.slug}: managed edge sync failed for ${failures.join(", ")}`,
);
}
})
.catch(() => {});
};
// A domain targets a PORT (proxy to the app) or a PATH (serve files) —
// exactly one, same rule the deploy path enforces. `continue`-ing on
// targetPath is what left static projects unrouted here: adding a domain to
// one wrote no vhost at all, so the hostname fell through to
// default_server, while the deploy path (which does emit a static root)
// made the same domain work — so it only ever "broke" on edit.
if (domain.targetPath) {
if (!staticRootBase) {
const containerId = deployment.containerId;
if (!containerId) {
// Compose/multi-service deployments track containers per-service, so the
// parent deployment row has no containerId — nothing to point a single-app
// route at (per-service routes are handled in updateService). Still tear
// down any dropped hostnames on the correct host.
console.warn(
`[project-route] ${project.slug}: deployment ${deployment.id} has no containerId (target=${effectiveTarget}) — skipping single-app route registration`,
);
await reconcileProjectRoutes(project, { routing, removes });
await pushProjectRules(project.id, serverId ?? null, previousHostnames).catch(() => {});
// Shared-dict state is RAM: the analytics collection switches have to be re-pushed
// whenever routing is applied, or an nginx restart silently reverts them to off.
await pushProjectAnalyticsConfig(project.id, serverId ?? null, previousHostnames).catch(() => {});
syncAddedManagedEdge();
return;
}
const resolveTargetUrl = async (port: number): Promise<string | null> => {
const strategy = resolveRouteStrategy(project.routeStrategy);
// loopback-port: dial the container's published loopback host port (read
// live). Bare / no-host-port fall back to container IP (or 127.0.0.1 bare).
const url = await resolveLiveUpstreamUrl({ strategy, runtime, containerId, containerPort: port });
if (!url) {
console.warn(
`[project-route] ${project.slug}: no static root for ${domain.hostname} (path ${domain.targetPath}) — skipping`,
`[project-route] ${project.slug}: could not resolve upstream for ${containerId} (target=${effectiveTarget}, server=${serverId ?? "local"})`,
);
return null;
}
// Never proxy a public route at a reserved control-plane/mgmt port on the
// host loopback — that would expose the admin API (env.PORT) or the
// unauthenticated OpenResty mgmt port (9145). Only guards loopback: a
// container's own bridge IP:<port> is the app's, not ours. The self-app is
// exempt (see ReapplyProjectLiveRoutesOptions.isSelfApp).
const m = url.match(/^https?:\/\/([^:/]+):(\d+)$/);
if (m && shouldRefuseLoopbackRoute(m[1], Number(m[2]), opts)) {
console.warn(
`[project-route] ${project.slug}: refusing reserved loopback upstream port ${m[2]} for a public route`,
);
return null;
}
return url;
};
// Where a path-targeted (static) domain serves its files from — the SAME
// resolver the post-deploy output probe uses, so the vhost and the check that
// audits it can never disagree about the directory.
const staticRootBase = resolveDeploymentStaticRoot(deployment, project);
// A redirect only goes live when its target is one of the hostnames this
// project currently routes — see resolveRouteRedirect.
const liveHostnames = current.map((domain) => domain.hostname);
const registers: RouteRegister[] = [];
/**
* The project's vercel.json rules, for EVERY project shape.
*
* `applyProjectRouting` also compiles them, but only for the 1-static + 1-server
* monorepo `planCompositeRoute` recognises — so a lone static site or a single app
* had its redirects, headers and URL shape silently dropped, which is most projects.
* This is the per-domain surface, so it is where the general case belongs.
*
* Deliberately NO `backendTargetUrl`. Which upstream a path rewrite (`/api/(.*)` →
* `/api/index.js`, a function on Vercel) belongs to is a TOPOLOGY question this
* per-domain loop cannot answer: passing the domain's own upstream would, on a
* composite monorepo, point `/api/` at the FRONTEND. `applyProjectRouting` runs
* afterwards and would overwrite it — but it is best-effort, so a failure there
* would leave that wrong upstream live. Rewrites needing a backend are therefore
* left to the path that knows the topology; a full-URL rewrite needs no backend and
* still compiles here, and a single app already receives `/api/…` via `location /`.
*/
const routingFields = compileProjectRoutingFields(project.routingConfig);
for (const domain of current) {
const redirectHost = resolveRouteRedirect(domain, liveHostnames);
const common = {
hostname: domain.hostname,
isCustomDomain: domain.domainType === "custom",
...(redirectHost ? { redirectHost } : {}),
};
// A domain targets a PORT (proxy to the app) or a PATH (serve files) —
// exactly one, same rule the deploy path enforces. `continue`-ing on
// targetPath is what left static projects unrouted here: adding a domain to
// one wrote no vhost at all, so the hostname fell through to
// default_server, while the deploy path (which does emit a static root)
// made the same domain work — so it only ever "broke" on edit.
if (domain.targetPath) {
if (!staticRootBase) {
console.warn(
`[project-route] ${project.slug}: no static root for ${domain.hostname} (path ${domain.targetPath}) — skipping`,
);
continue;
}
try {
// Same call the deploy path's route registration and the output probe
// make — one rule for "which directory does this path serve".
registers.push({
...common,
...routingFields,
staticRoot: resolveServedStaticPath(staticRootBase, domain.targetPath),
});
} catch (err) {
// A `../` in the operator's route path. Refuse this ONE route; the rest of
// the re-apply (and the project's other domains) must still go through.
console.warn(
`[project-route] ${project.slug}: refusing ${domain.hostname} — ${safeErrorMessage(err)}`,
);
}
continue;
}
try {
// Same call the deploy path's route registration and the output probe
// make — one rule for "which directory does this path serve".
registers.push({ ...common, staticRoot: resolveServedStaticPath(staticRootBase, domain.targetPath) });
} catch (err) {
// A `../` in the operator's route path. Refuse this ONE route; the rest of
// the re-apply (and the project's other domains) must still go through.
console.warn(
`[project-route] ${project.slug}: refusing ${domain.hostname} — ${safeErrorMessage(err)}`,
);
const port = domain.targetPort ?? project.port;
if (!port) {
console.warn(`[project-route] ${project.slug}: no port for ${domain.hostname} — skipping`);
continue;
}
continue;
const targetUrl = await resolveTargetUrl(port);
if (!targetUrl) continue;
registers.push({ ...common, ...routingFields, targetUrl });
}
const port = domain.targetPort ?? project.port;
if (!port) {
console.warn(`[project-route] ${project.slug}: no port for ${domain.hostname} — skipping`);
continue;
}
const targetUrl = await resolveTargetUrl(port);
if (!targetUrl) continue;
registers.push({ ...common, targetUrl });
// The webhook-proxy location is re-attached automatically for the project's
// webhookDomain inside reconcileProjectRoutes.
await reconcileProjectRoutes(project, { routing, registers, removes });
// Re-sync per-route edge rules (rate-limit / ban / allow-deny) for the current
// hostnames. Best-effort — the DB is the source of truth; a failure defers to
// the next reconcile. previousHostnames clears rules for any dropped hostname.
await pushProjectRules(project.id, serverId ?? null, previousHostnames).catch(() => {});
// Shared-dict state is RAM: the analytics collection switches have to be re-pushed
// whenever routing is applied, or an nginx restart silently reverts them to off.
await pushProjectAnalyticsConfig(project.id, serverId ?? null, previousHostnames).catch(() => {});
// Register the newly-added managed slug(s) on the cloud edge (the "add" half
// of the edit; dropped slugs were deregistered above). Per-route — unchanged
// hostnames are not re-synced.
syncAddedManagedEdge();
} finally {
disposePlatform(resolved);
}
// The webhook-proxy location is re-attached automatically for the project's
// webhookDomain inside reconcileProjectRoutes.
await reconcileProjectRoutes(project, { routing, registers, removes });
// Re-sync per-route edge rules (rate-limit / ban / allow-deny) for the current
// hostnames. Best-effort — the DB is the source of truth; a failure defers to
// the next reconcile. previousHostnames clears rules for any dropped hostname.
await pushProjectRules(project.id, serverId ?? null, previousHostnames).catch(() => {});
// Shared-dict state is RAM: the analytics collection switches have to be re-pushed
// whenever routing is applied, or an nginx restart silently reverts them to off.
await pushProjectAnalyticsConfig(project.id, serverId ?? null, previousHostnames).catch(() => {});
// Register the newly-added managed slug(s) on the cloud edge (the "add" half
// of the edit; dropped slugs were deregistered above). Per-route — unchanged
// hostnames are not re-synced.
syncAddedManagedEdge();
}
@@ -0,0 +1,182 @@
import { beforeEach, describe, expect, it, vi } from "vitest";
const projectRepo = vi.hoisted(() => ({ findById: vi.fn() }));
const deploymentRepo = vi.hoisted(() => ({ findById: vi.fn() }));
const serviceRepo = vi.hoisted(() => ({ listByProject: vi.fn(), listByDeployment: vi.fn() }));
const resolveDeploymentRuntime = vi.hoisted(() => vi.fn());
const usesManagedRouting = vi.hoisted(() => vi.fn().mockReturnValue(false));
const reconcileProjectRoutes = vi.hoisted(() => vi.fn());
vi.mock("@repo/db", async (importOriginal) => {
const actual = await importOriginal<typeof import("@repo/db")>();
return {
...actual,
repos: {
...actual.repos,
project: projectRepo,
deployment: deploymentRepo,
service: serviceRepo,
},
};
});
// The service resolves a PLATFORM now (`{platform:{routing,runtime}}`) and releases
// its docker transport when done, so the flat stub above is adapted to that shape
// rather than re-written per test.
vi.mock("../../lib/deployment-runtime", () => ({
usesManagedRouting,
disposePlatform: () => {},
resolveDeploymentPlatform: async (...args: unknown[]) => {
const flat = (await resolveDeploymentRuntime(...args)) as Record<string, unknown>;
return { platform: flat, effectiveTarget: flat.effectiveTarget, serverId: flat.serverId };
},
}));
vi.mock("../../lib/route-apply.service", () => ({ reconcileProjectRoutes }));
vi.mock("../../lib/controller-helpers", () => ({ platform: () => ({ target: "selfhosted" }) }));
import { applyProjectRouting } from "./routing-apply.service";
/** The single register `applyProjectRouting` emitted for the fan-out domain. */
function emittedRegister() {
expect(reconcileProjectRoutes).toHaveBeenCalledTimes(1);
const [, opts] = reconcileProjectRoutes.mock.calls[0];
expect(opts.registers).toHaveLength(1);
return opts.registers[0];
}
function runtimeReturning(opts: {
/** undefined → the live inspect throws. */
info?: { status?: string; ip?: string; hostPort?: number };
ip?: string | null;
}) {
resolveDeploymentRuntime.mockResolvedValue({
runtime: {
name: "docker",
supports: (cap: string) => cap === "containerIp" || cap === "containerInfo",
getContainerInfo: vi.fn(async () => {
if (!opts.info) throw new Error("not found");
return { containerId: "container_1", status: "running", ...opts.info };
}),
getContainerIp: vi.fn(async () => (opts.ip === undefined ? "172.19.0.2" : opts.ip)),
},
routing: { registerRoute: vi.fn() },
effectiveTarget: "local",
});
}
/** `service_deployment` row for the migrated service. */
function storeRow(row: { containerId?: string | null; ip?: string | null; hostPort?: number | null }) {
serviceRepo.listByDeployment.mockResolvedValue([
{ id: "sd_1", serviceId: "svc_1", deploymentId: "dep_1", containerId: "container_1", ip: null, hostPort: null, ...row },
]);
}
const project = (routeStrategy = "auto") => ({
id: "proj_1",
activeDeploymentId: "dep_1",
routeStrategy,
routingConfig: null,
compositeRoutes: [
{ hostname: "app.example.com", isCustomDomain: true, rootServiceId: "svc_1", locations: [] },
],
slug: "app",
cloudWorkspaceId: null,
webhookDomain: null,
});
describe("applyProjectRouting — upstream resolution", () => {
beforeEach(() => {
vi.clearAllMocks();
projectRepo.findById.mockResolvedValue(project());
deploymentRepo.findById.mockResolvedValue({
id: "dep_1",
organizationId: "org_1",
meta: { deployTarget: "local", runtimeMode: "docker" },
});
serviceRepo.listByProject.mockResolvedValue([
{
id: "svc_1",
projectId: "proj_1",
name: "web",
enabled: true,
exposed: true,
exposedPort: "3001",
domain: null,
customDomain: "app.example.com",
domainType: "custom",
publicEndpoints: null,
ports: [],
kind: "compose",
},
]);
storeRow({});
runtimeReturning({ info: { ip: "172.19.0.2" } });
usesManagedRouting.mockReturnValue(false);
reconcileProjectRoutes.mockResolvedValue(undefined);
});
it("routes a migrated container with no loopback publish at its container IP", async () => {
await applyProjectRouting("proj_1");
expect(emittedRegister()).toMatchObject({
hostname: "app.example.com",
targetUrl: "http://172.19.0.2:3001",
});
});
/**
* #506 regression guard. The row still carries a host port from an earlier
* deploy, but the live container publishes nothing — the stored value must not
* win, or the edge dials a dead 127.0.0.1:3001 behind a Verified domain.
*/
it("ignores a stored host port the live container no longer publishes", async () => {
storeRow({ ip: "172.19.0.2", hostPort: 3001 });
await applyProjectRouting("proj_1");
expect(emittedRegister().targetUrl).toBe("http://172.19.0.2:3001");
});
it("uses the live loopback port when the container publishes one", async () => {
storeRow({ ip: "172.19.0.2", hostPort: 3999 });
runtimeReturning({ info: { ip: "172.19.0.2", hostPort: 4000 } });
await applyProjectRouting("proj_1");
expect(emittedRegister().targetUrl).toBe("http://127.0.0.1:4000");
});
it("keeps the last-known upstream when the container cannot be inspected", async () => {
// A failed inspect is not evidence that nothing is published, so the working
// vhost must survive the re-apply rather than being repointed or dropped.
storeRow({ ip: "172.19.0.2", hostPort: 4000 });
runtimeReturning({ ip: null });
await applyProjectRouting("proj_1");
expect(emittedRegister().targetUrl).toBe("http://127.0.0.1:4000");
});
it("falls back to the stored row for a service with no container yet", async () => {
storeRow({ containerId: null, ip: "172.19.0.2" });
await applyProjectRouting("proj_1");
expect(emittedRegister().targetUrl).toBe("http://172.19.0.2:3001");
});
it("honours an explicit container-ip strategy over a published host port", async () => {
projectRepo.findById.mockResolvedValue(project("container-ip"));
storeRow({ ip: "172.19.0.2", hostPort: 4000 });
runtimeReturning({ info: { ip: "172.19.0.2", hostPort: 4000 } });
await applyProjectRouting("proj_1");
expect(emittedRegister().targetUrl).toBe("http://172.19.0.2:3001");
});
});
@@ -23,8 +23,15 @@ import {
type OblienRoutingContext,
} from "@repo/adapters";
import { platform } from "../../lib/controller-helpers";
import { resolveDeploymentRuntime, usesManagedRouting } from "../../lib/deployment-runtime";
import {
disposePlatform,
resolveDeploymentPlatform,
usesManagedRouting,
type DeploymentMeta,
type ResolvedDeploymentPlatform,
} from "../../lib/deployment-runtime";
import { reconcileProjectRoutes } from "../../lib/route-apply.service";
import { compileProjectRoutingFields } from "../../lib/project-routing-fields";
import { resolveServicePort } from "../../lib/deployable-service";
import { buildServiceRouteDomain } from "../../lib/routing-domains";
import {
@@ -32,7 +39,11 @@ import {
buildDomainFanoutRegistrations,
planCompositeRoute,
} from "../deployments/compose/composite-route";
import { buildUpstreamUrl, resolveRouteStrategy } from "../../lib/upstream-url";
import {
buildUpstreamUrl,
resolveLiveUpstreamUrl,
resolveRouteStrategy,
} from "../../lib/upstream-url";
export async function applyProjectRouting(projectId: string): Promise<void> {
const project = await repos.project.findById(projectId);
@@ -41,12 +52,20 @@ export async function applyProjectRouting(projectId: string): Promise<void> {
// No active deployment → the persisted routingConfig applies on the next deploy.
if (!project.activeDeploymentId) return;
// Held for the `finally`: a remote-server platform binds a Docker-over-SSH
// loopback bridge that only `dispose` closes, and this ran on every live route
// edit. Releasing it leaves `routing` fully usable — the bridge is the docker
// transport, while routing drives the box through the pooled SSH executor.
let resolved: ResolvedDeploymentPlatform | null = null;
try {
const deployment = await repos.deployment.findById(project.activeDeploymentId);
if (!deployment) return;
const { routing, runtime, effectiveTarget } = await resolveDeploymentRuntime(deployment);
const managed = usesManagedRouting(platform().target, effectiveTarget);
resolved = await resolveDeploymentPlatform((deployment.meta ?? {}) as DeploymentMeta, {
organizationId: deployment.organizationId,
});
const { routing, runtime } = resolved.platform;
const managed = usesManagedRouting(platform().target, resolved.effectiveTarget);
const defs = await repos.service.listByProject(project.id);
const liveRows = await repos.service.listByDeployment(project.activeDeploymentId);
@@ -62,8 +81,34 @@ export async function applyProjectRouting(projectId: string): Promise<void> {
const routeStrategy = resolveRouteStrategy(project.routeStrategy);
// One live-upstream resolver, shared by the vercel composite AND the migration
// path-fan-out — each service's URL from its service_deployment row.
// path-fan-out. Resolved from the LIVE container (not the service_deployment
// row) so a workload with no loopback publish — migrated, adopted in place —
// routes at its container IP instead of a dead 127.0.0.1:<port>. Awaited up
// front because the composite/fan-out builders take a sync resolver.
const liveUpstreams = new Map<string, string | null>();
await Promise.all(
defs.map(async (def) => {
const row = rowByService.get(def.id);
const port = resolveServicePort(def, project.port);
if (!port || !row?.containerId) return;
liveUpstreams.set(
def.id,
await resolveLiveUpstreamUrl({
strategy: routeStrategy,
runtime,
containerId: row.containerId,
containerPort: port,
stored: { ip: row.ip, hostPort: row.hostPort },
}),
);
}),
);
const resolveTargetUrl = (serviceId: string) => {
// A service with no container to inspect (cloud peer, not yet deployed)
// still resolves from its persisted row.
const live = liveUpstreams.get(serviceId);
if (live !== undefined) return live;
const def = defs.find((s) => s.id === serviceId);
const row = rowByService.get(serviceId);
const port = def ? resolveServicePort(def, project.port) : null;
@@ -93,7 +138,25 @@ export async function applyProjectRouting(projectId: string): Promise<void> {
// Re-emit any migration path-fan-out domains from live upstreams (a domain
// whose paths route to different services) — persisted so it survives here.
const fanout = buildDomainFanoutRegistrations({ routes: project.compositeRoutes, resolveTargetUrl });
//
// They carry the project's compiled vercel.json rules because this is the LAST
// writer for those hostnames on the live path (callers run
// `reapplyProjectLiveRoutes` first, this second) and `registerRoute` REPLACES the
// vhost — so without them a routing save applied its redirects to every domain
// EXCEPT the fan-out one, and the deploy path (which does carry them) then
// disagreed with the live path about the same vhost. The composite is left alone:
// it compiles its own topology-aware superset with the backend it resolved.
const routingFields = compileProjectRoutingFields(project.routingConfig);
const fanout = buildDomainFanoutRegistrations({
routes: project.compositeRoutes,
resolveTargetUrl,
}).map((reg) => {
// CONCATENATED, not overwritten — same rule and same order as the deploy path:
// the fan-out's explicit per-path upstreams first, then the compiled rules, or
// the spread would ASSIGN over them and drop a vercel.json external rewrite.
const proxyLocations = [...(reg.proxyLocations ?? []), ...(routingFields.proxyLocations ?? [])];
return { ...reg, ...routingFields, ...(proxyLocations.length ? { proxyLocations } : {}) };
});
const registers = [...(composite ? [composite.register] : []), ...fanout];
if (registers.length > 0) {
@@ -103,6 +166,8 @@ export async function applyProjectRouting(projectId: string): Promise<void> {
console.warn(
`[routing-apply] ${project.slug}: live routing re-apply failed (non-fatal, applies next deploy): ${safeErrorMessage(err)}`,
);
} finally {
disposePlatform(resolved);
}
}
@@ -0,0 +1,118 @@
import { beforeEach, describe, expect, it, vi } from "vitest";
/**
* `POST /api/cloud/github/installation-token` mints a live GitHub App installation
* token. Its route tag (`cloud:write`) does NOT authorize it: "cloud" is an org
* SINGLETON resource, so the permission assert runs with resourceId "*" and
* `roleAllowsResourceType` lets a plain `member` through. Before this gate existed,
* any org member could mint a token and the grant system that decides which repos a
* member may touch was never consulted.
*
* What is pinned here is the MAPPING, because that is where the risk actually lives:
*
* named repos → the returned token is scoped to exactly those, so each is
* authorized at REPO level;
* no repos → the token covers every repo in the installation, so the question
* is owner-level AUTHORITY. Asking at the default "reach" level
* would let a read grant on one repo under `acme` mint a token over
* every repo under `acme` — the GHSA-qv27-39pc-qw9f finding-2 shape
* arriving through a different door.
*
* Tested here rather than through the handler because cloud-saas.controller.ts pulls
* in Better Auth and the whole SaaS graph, while the decision is this one function.
* Mock shape matches github-access.write-escalation.test.ts.
*/
const { memberFind, listByMember } = vi.hoisted(() => ({
memberFind: vi.fn(),
listByMember: vi.fn(),
}));
vi.mock("@repo/db", () => ({
repos: {
member: { find: memberFind },
resourceGrant: { listByMember },
},
}));
import { canMintInstallationToken } from "./github-access";
// A plain member, not a scoped token: owner auto-access does not apply, so grant
// matching actually runs.
// eslint-disable-next-line @typescript-eslint/no-explicit-any
const memberCtx = { userId: "m1", organizationId: "o1" } as any;
const grant = (resourceType: string, resourceId: string, permissions: string[]) => ({
resourceType,
resourceId,
permissions,
});
describe("canMintInstallationToken", () => {
beforeEach(() => {
memberFind.mockReset();
listByMember.mockReset();
memberFind.mockResolvedValue({ role: "member" });
listByMember.mockResolvedValue([]);
});
it("refuses an un-narrowed token when the member only has ONE repo under the owner", async () => {
// The core case. A read grant on acme/one is not authority over acme.
listByMember.mockResolvedValue([grant("github_repository", "acme/one", ["read"])]);
await expect(canMintInstallationToken(memberCtx, "acme")).resolves.toBe(false);
await expect(canMintInstallationToken(memberCtx, "acme", [])).resolves.toBe(false);
});
it("allows an un-narrowed token for an installation-level grant", async () => {
listByMember.mockResolvedValue([grant("github_installation", "acme", ["read"])]);
await expect(canMintInstallationToken(memberCtx, "acme")).resolves.toBe(true);
});
it("allows a narrowed token for exactly the granted repo", async () => {
listByMember.mockResolvedValue([grant("github_repository", "acme/one", ["read"])]);
await expect(canMintInstallationToken(memberCtx, "acme", ["one"])).resolves.toBe(true);
});
it("refuses when ANY requested repo is ungranted (all-or-nothing)", async () => {
// A token narrowed to [one, two] still reaches `two`, so partial access must not
// be enough — the caller would otherwise get a working token for a repo they
// cannot touch.
listByMember.mockResolvedValue([grant("github_repository", "acme/one", ["read"])]);
await expect(canMintInstallationToken(memberCtx, "acme", ["one", "two"])).resolves.toBe(false);
});
it("refuses a sibling repo under the same owner", async () => {
listByMember.mockResolvedValue([grant("github_repository", "acme/one", ["read"])]);
await expect(canMintInstallationToken(memberCtx, "acme", ["two"])).resolves.toBe(false);
});
it("treats whitespace-only repos as un-narrowed, not as nothing-to-check", async () => {
// `[" "]` must NOT read as "no repos requested, so vacuously allowed". It falls
// back to the stricter owner-authority question.
listByMember.mockResolvedValue([grant("github_repository", "acme/one", ["read"])]);
await expect(canMintInstallationToken(memberCtx, "acme", [" "])).resolves.toBe(false);
listByMember.mockResolvedValue([grant("github_installation", "acme", ["read"])]);
await expect(canMintInstallationToken(memberCtx, "acme", [" "])).resolves.toBe(true);
});
it("still allows the org owner (auto-access), narrowed or not", async () => {
memberFind.mockResolvedValue({ role: "owner" });
listByMember.mockResolvedValue([]);
await expect(canMintInstallationToken(memberCtx, "acme")).resolves.toBe(true);
await expect(canMintInstallationToken(memberCtx, "acme", ["anything"])).resolves.toBe(true);
});
it("refuses a non-member of the org", async () => {
memberFind.mockResolvedValue(null);
await expect(canMintInstallationToken(memberCtx, "acme")).resolves.toBe(false);
await expect(canMintInstallationToken(memberCtx, "acme", ["one"])).resolves.toBe(false);
});
it("fails closed when the grant lookup throws", async () => {
listByMember.mockRejectedValue(new Error("db down"));
await expect(canMintInstallationToken(memberCtx, "acme")).resolves.toBe(false);
await expect(canMintInstallationToken(memberCtx, "acme", ["one"])).resolves.toBe(false);
});
});
+67 -8
View File
@@ -75,6 +75,26 @@ export interface GitHubAccessTarget {
installationId?: string | number | null;
}
/**
* How to read an owner-level target — one where `target.repo` is omitted, so the
* question is about the account as a whole rather than a single repo.
*
* "reach" (default) any granted repo under the owner passes. Correct where
* the answer is later NARROWED — list filtering, the picker, the
* project binding — so a repo-only member still sees and builds
* their granted repo.
* "authority" the caller must hold the owner as a whole: an installation-level
* or all-GitHub grant. Required wherever the decision is FINAL and
* nothing downstream narrows it.
*
* The distinction exists because "reach" is only sound when something narrows it
* afterwards. At mint time nothing does: the grant that gets written IS the
* downstream authority, so one repo under `acme` would authorize a persisted
* owner-wide `github_installation` grant over every repo under `acme`
* (GHSA-qv27-39pc-qw9f finding 2).
*/
export type OwnerLevelMode = "reach" | "authority";
/**
* Authorize a GitHub action for the request's caller.
*
@@ -85,17 +105,52 @@ export interface GitHubAccessTarget {
* - Everyone else → allow ONLY if a matching grant exists at the repo,
* installation, or all-GitHub level with sufficient permission.
*
* When `target.repo` is omitted (owner-level list/token gating), a member
* who holds ANY repo grant under that owner passes — the actual repo set
* is narrowed downstream by list filtering / the project binding, so a
* repo-only member can still see and build their granted repo.
* When `target.repo` is omitted the check is owner-level; `opts.ownerLevel`
* decides whether a single granted repo under that owner suffices. See
* OwnerLevelMode — callers making a FINAL authority decision must pass
* "authority".
*
* Fails CLOSED on any lookup error.
*/
/**
* May this caller mint a GitHub App installation token for `owner`, narrowed to
* `repos` (or un-narrowed when `repos` is empty)?
*
* Separate from `canUseGitHubRepo` because the QUESTION differs with the narrowing,
* and getting that mapping wrong is the whole risk:
*
* - named repos → each one is authorized at REPO level. The token GitHub returns
* is scoped to exactly those, so a per-repo grant is the right currency.
* - no repos → the token covers EVERY repo in the installation, which is an
* owner-level AUTHORITY decision. "reach" would be wrong here: a read grant on
* one repo under `acme` would hand back a token over every repo under `acme` —
* the GHSA-qv27-39pc-qw9f finding-2 shape, arriving by a different door.
*
* Lives here rather than in the cloud controller so the rule sits beside the grant
* logic it depends on, and so it is testable without the SaaS controller's import
* graph. Fails CLOSED: `canUseGitHubRepo` already does, and an empty `repos` after
* trimming is treated as un-narrowed rather than as "nothing to check".
*/
export async function canMintInstallationToken(
ctx: RequestContext,
owner: string,
repos?: string[],
): Promise<boolean> {
const named = (repos ?? []).map((r) => r.trim()).filter(Boolean);
if (!named.length) {
return canUseGitHubRepo(ctx, { owner }, "read", { ownerLevel: "authority" });
}
const results = await Promise.all(
named.map((repo) => canUseGitHubRepo(ctx, { owner, repo }, "read")),
);
return results.every(Boolean);
}
export async function canUseGitHubRepo(
ctx: RequestContext,
target: GitHubAccessTarget,
op: GitHubAccessOp,
opts?: { ownerLevel?: OwnerLevelMode },
): Promise<boolean> {
const organizationId = ctx.organizationId || undefined;
if (!organizationId) return !isScoped(ctx);
@@ -108,6 +163,7 @@ export async function canUseGitHubRepo(
const grants = await grantSourceFor(ctx).listByMember(organizationId, ctx.userId);
const ownerLevel = opts?.ownerLevel ?? "reach";
const ownerLc = target.owner.toLowerCase();
const repoKey = target.repo ? `${ownerLc}/${target.repo.toLowerCase()}` : null;
@@ -127,10 +183,13 @@ export async function canUseGitHubRepo(
const grantedRepo = g.resourceId.toLowerCase();
// Exact repo match …
if (repoKey && grantedRepo === repoKey) return true;
// … or, for an owner-level check (no specific repo), any granted
// repo under this owner lets them through (downstream filtering /
// the project binding narrows to the actual repo).
if (!repoKey && grantedRepo.startsWith(`${ownerLc}/`)) return true;
// … or, for an owner-level check (no specific repo) in "reach" mode, any
// granted repo under this owner lets them through (downstream filtering /
// the project binding narrows to the actual repo). In "authority" mode it
// does NOT: one repo is not authority over the account.
if (!repoKey && ownerLevel === "reach" && grantedRepo.startsWith(`${ownerLc}/`)) {
return true;
}
}
}
@@ -0,0 +1,185 @@
import { beforeEach, describe, expect, it, vi } from "vitest";
/**
* Regression for GHSA-hp2g-hw7g-f3vm — GitHub repo management was authorized
* read-level and owner-wide, letting a non-owner member with a *read-only* grant
* on ONE repo perform write/admin operations (delete/create repo, register/delete
* webhooks) on ANY repo under the same owner via the org App installation token.
*
* Two defects fed the same token funnel:
* 1. the operation was hardcoded to "read" regardless of HTTP method, so a read
* grant satisfied a DELETE/POST; and
* 2. `repo` was dropped, so a grant on repo A satisfied an owner-wide check that
* then authorized repo B.
*
* These tests pin the authorization core (`canUseGitHubRepo`, github-access.ts)
* that both defects violated — a read grant must never satisfy a write, and a
* per-repo grant must never reach a sibling. The token funnel (github.token.ts /
* github.auth.ts) now threads the real op + repo INTO this function; see
* `test/modules/github/tokenFor-op-wiring.test.ts` for that wiring.
*/
const { memberFind, listByMember } = vi.hoisted(() => ({
memberFind: vi.fn(),
listByMember: vi.fn(),
}));
vi.mock("@repo/db", () => ({
repos: {
member: { find: memberFind },
// grantSourceFor(ctx) returns repos.resourceGrant for a non-scoped caller.
resourceGrant: { listByMember },
},
}));
import { canUseGitHubRepo } from "./github-access";
// A plain (non-owner) member on org o1, NOT a scoped token → owner-auto-access
// only applies to the owner role, so grant matching actually runs.
// eslint-disable-next-line @typescript-eslint/no-explicit-any
const memberCtx = { userId: "m1", organizationId: "o1" } as any;
beforeEach(() => {
vi.clearAllMocks();
memberFind.mockResolvedValue({ role: "member" });
// Read-only grant on exactly ONE repo under `acme`.
listByMember.mockResolvedValue([
{
resourceType: "github_repository",
resourceId: "acme/marketing-site",
permissions: ["read"],
},
]);
});
describe("canUseGitHubRepo — GHSA-hp2g-hw7g-f3vm write escalation", () => {
it("allows READ on the exact granted repo (the legitimate case)", async () => {
expect(
await canUseGitHubRepo(memberCtx, { owner: "acme", repo: "marketing-site" }, "read"),
).toBe(true);
});
it("DENIES write on the granted repo when the grant is read-only (defect 1: op level)", async () => {
expect(
await canUseGitHubRepo(memberCtx, { owner: "acme", repo: "marketing-site" }, "write"),
).toBe(false);
});
it("DENIES write on a SIBLING repo under the same owner (defect 2: per-repo)", async () => {
expect(
await canUseGitHubRepo(memberCtx, { owner: "acme", repo: "production-api" }, "write"),
).toBe(false);
});
it("DENIES an owner-level write (repo omitted) — the exact call the pre-fix funnel made for a DELETE", async () => {
// Pre-fix, the funnel called canUseGitHubRepo(ctx, {owner, repo: undefined}, "read")
// for EVERY method; the read grant satisfied it owner-wide and minted the App
// token. With the op corrected to "write", a read grant can no longer pass.
expect(await canUseGitHubRepo(memberCtx, { owner: "acme" }, "write")).toBe(false);
});
it("still ALLOWS write when the member holds a WRITE grant on that repo", async () => {
listByMember.mockResolvedValue([
{
resourceType: "github_repository",
resourceId: "acme/production-api",
permissions: ["write"],
},
]);
expect(
await canUseGitHubRepo(memberCtx, { owner: "acme", repo: "production-api" }, "write"),
).toBe(true);
});
it("still DENIES a write grant on repo A from reaching sibling repo B", async () => {
listByMember.mockResolvedValue([
{
resourceType: "github_repository",
resourceId: "acme/marketing-site",
permissions: ["write"],
},
]);
expect(
await canUseGitHubRepo(memberCtx, { owner: "acme", repo: "production-api" }, "write"),
).toBe(false);
});
it("still ALLOWS the org OWNER (auto-access) to write any repo", async () => {
memberFind.mockResolvedValue({ role: "owner" });
listByMember.mockResolvedValue([]);
expect(
await canUseGitHubRepo(memberCtx, { owner: "acme", repo: "production-api" }, "write"),
).toBe(true);
});
it("honors an installation-level (owner-wide) grant for writes across repos", async () => {
listByMember.mockResolvedValue([
{ resourceType: "github_installation", resourceId: "acme", permissions: ["write"] },
]);
expect(
await canUseGitHubRepo(memberCtx, { owner: "acme", repo: "production-api" }, "write"),
).toBe(true);
});
});
/**
* The `ownerLevel` mode is the only behavioural change inside this module, and it
* only bites when `repo` is omitted — so the cases above (all repo-specific, all
* 3-arg) would pass against the pre-fix module too. These are the ones that pin it.
*/
describe("canUseGitHubRepo — owner-level checks: reach vs authority", () => {
beforeEach(() => {
// One WRITE grant on one repo under `acme`. Enough to write THAT repo; the
// question here is what it implies about the ACCOUNT.
listByMember.mockResolvedValue([
{
resourceType: "github_repository",
resourceId: "acme/marketing-site",
permissions: ["write"],
},
]);
});
it("'reach' (default) lets one granted repo satisfy an owner-level check", async () => {
// Correct where the answer is narrowed downstream — list filtering, the picker.
expect(await canUseGitHubRepo(memberCtx, { owner: "acme" }, "read")).toBe(true);
});
it("'authority' does NOT — one repo is not authority over the account", async () => {
expect(
await canUseGitHubRepo(memberCtx, { owner: "acme" }, "write", {
ownerLevel: "authority",
}),
).toBe(false);
});
it("'authority' passes on an installation-level grant", async () => {
listByMember.mockResolvedValue([
{ resourceType: "github_installation", resourceId: "acme", permissions: ["write"] },
]);
expect(
await canUseGitHubRepo(memberCtx, { owner: "acme" }, "write", {
ownerLevel: "authority",
}),
).toBe(true);
});
it("'authority' passes on an all-GitHub grant", async () => {
listByMember.mockResolvedValue([
{ resourceType: "github", resourceId: "*", permissions: ["write"] },
]);
expect(
await canUseGitHubRepo(memberCtx, { owner: "acme" }, "write", {
ownerLevel: "authority",
}),
).toBe(true);
});
it("'authority' is irrelevant once a repo is named (the repo match decides)", async () => {
expect(
await canUseGitHubRepo(memberCtx, { owner: "acme", repo: "marketing-site" }, "write", {
ownerLevel: "authority",
}),
).toBe(true);
});
});

Some files were not shown because too many files have changed in this diff Show More