Files

211 lines
13 KiB
YAML

# Chooses the npm registry for this job, and is the single place that logic lives.
#
# It was previously duplicated verbatim in 66 jobs across 18 workflows. During PR #9993 the same
# lockstep edit had to be applied to all 66 copies on nine separate occasions, and on three of those
# the code was updated while the surrounding comment was not - which is exactly the drift CodeRabbit
# and cubic flagged when they asked for this extraction.
#
# Behaviour is unchanged from that reviewed implementation; only its location has moved.
name: Configure npm registry
description: >-
Point the job at the internal Verdaccio VIP when it is reachable, otherwise at the public npm
registry. The Cloudflare-fronted packages.ever.co route is gated behind an explicit opt-in.
inputs:
verdaccio-registry:
# NOTE: no GitHub expression syntax in this field. GitHub evaluates action metadata, and `vars` is
# not a valid context there - an expression here makes every CROSS-REPO use fail at 'Set up job' with
# TemplateValidationException: Unrecognized named-value: 'vars'. Name the variable in prose instead.
description: Internal Verdaccio VIP, normally the org variable VERDACCIO_REGISTRY. Empty disables the probe.
required: false
default: ''
verdaccio-token:
description: Auth token for the public packages.ever.co route. Only used when force-public is true.
required: false
default: ''
force-public:
description: Set to 'true' to force the public packages.ever.co route even when the VIP answers.
required: false
default: ''
expect-vip:
description: >-
'true' for in-network runners. Controls whether an unreachable VIP is retried and warned about.
Derive it from the runner, e.g. contains(matrix.os, 'self-hosted') || contains(matrix.os, 'ever-k8s').
required: false
default: 'false'
runs:
using: composite
steps:
- name: Configure Registry
shell: bash
env:
VERDACCIO_REGISTRY: ${{ inputs.verdaccio-registry }}
VERDACCIO_TOKEN: ${{ inputs.verdaccio-token }}
VERDACCIO_FORCE_PUBLIC: ${{ inputs.force-public }}
EXPECT_VIP: ${{ inputs.expect-vip }}
run: |
# Registry selection is CONDITIONAL on where this job runs.
#
# The public Cloudflare-fronted route https://packages.ever.co intermittently TRUNCATES
# large tarballs mid-stream. Measured 2026-08-16 on
# @rspack/binding-win32-arm64-msvc@1.7.6: 9,525,323 bytes from packages.ever.co against
# 14,590,565 bytes (byte-exact) from BOTH the internal VIP and npmjs. The route sustains
# only ~280 KB/s, and a single release-windows-arm64 job logged 12 "unexpected end of
# file" errors against it.
#
# The truncation is UNDETECTABLE by the client: Cloudflare always sends
# "Accept-Encoding: gzip" to the origin, Verdaccio then re-gzips the already-gzipped
# .tgz, which drops Content-Length in favour of chunked encoding. A cut stream is
# therefore not a short read but a complete-looking HTTP 200 whose body fails gunzip,
# and Range requests are ignored so there is no resume and no length cross-check.
#
# Jobs in this repo run on a MIX of runners:
# in-network - [self-hosted, Windows, X64] and the ever-k8s ARC pools
# external - windows-11-arm, ubicloud-*-arm, cirruslabs macOS, ubuntu-latest
# Only in-network runners can reach the VIP, and the internal cache buys an external
# runner nothing while adding a failure mode. The choice is therefore made by PROBING
# the VIP rather than by hardcoding a runner list, because release-linux runs on
# "vars.RUNNER_LINUX_APPS_X64 or ubuntu-latest" and so is not statically known.
#
# NO-REMOVAL RULE: the packages.ever.co path below is GATED, not deleted. Set the
# repo/org variable VERDACCIO_FORCE_PUBLIC=true to bring it back.
# Start from a known state. The self-hosted Windows jobs check out with clean: false and
# their cleanup step removes only dist/ and node_modules/, so registry edits from a
# PREVIOUS run can still be on disk: a stale "registry" line from an earlier run would
# silently win over this run's decision.
#
# .yarnrc goes through the per-path loop below rather than the rm -f list. It is untracked
# in ever-gauzy, but ever-co/ever-teams TRACKS it and pins the yarn binary there
# (yarn-path ".yarn/releases/yarn-1.22.22.cjs"). Deleting it outright dropped that pin, so
# `yarn install --frozen-lockfile` ran against whatever yarn the runner resolved. The loop
# restores it when tracked and still removes it when it is a stale untracked leftover.
#
# This reset is NOT best-effort -- if we cannot establish
# a known starting state we must not go on to pick a registry, because the install would
# then quietly run against whatever the last job left behind.
rm -f .npmrc.bak .npmrc.scrubbed yarn.lock.bak yarn.lock.rewritten package-lock.json.rewritten
# Reset per-path, not as one pathspec list. `git checkout -- a b` exits 1 when EITHER path is
# untracked, so the original form hard-failed on every pnpm repo (no .npmrc, no yarn.lock) and
# turned 'this repo has no cache configured' into 'this repo's CI is red'. Tracked files are
# restored; untracked leftovers from a previous run on a clean:false runner are removed.
reset_failed=""
reset_err=""
for f in .npmrc .yarnrc yarn.lock package-lock.json; do
# Exit status matters: 0 = tracked, 1 = genuinely absent, anything else (128 = git
# metadata unavailable) means we CANNOT tell. Treating every nonzero as absent would
# delete the files and continue, which is exactly the unknown-state hazard this reset
# exists to prevent - so only 1 is trusted, and everything else fails closed.
# The probe MUST sit in an `if` guard: composite bash runs with -e, so the bare
# `git ls-files ...; tracked=$?` form never reaches the assignment on any repo
# where $f is untracked - the step dies with exit 1 and no output at all
# (measured on the first pnpm consumer repo of this action, 2026-08-21).
# Stderr is CAPTURED, not discarded (`2>&1 >/dev/null` = stderr into the
# substitution, git's stdout dropped). On 2026-08-27 three self-hosted Windows
# runners failed here with a bare "(git-exit-128)" and finding the real error -
# "fatal: detected dubious ownership", after the runner service account changed
# out from under a NETWORK SERVICE-owned workspace - took a remote login to the
# machines, because this log never showed it. The first stderr line now rides
# along in the ::error below so the annotation self-diagnoses. The exit-status
# handling and fail-closed behavior are unchanged.
if probe_err=$(git ls-files --error-unmatch "$f" 2>&1 >/dev/null); then
tracked=0
else
tracked=$?
fi
if [ "$tracked" -eq 0 ]; then
if ! restore_err=$(git checkout -- "$f" 2>&1); then
# Re-emit so the raw log keeps the full message it has always shown.
if [ -n "$restore_err" ]; then printf '%s\n' "$restore_err" >&2; fi
reset_failed="$reset_failed $f"
if [ -z "$reset_err" ]; then reset_err="${restore_err%%$'\n'*}"; fi
fi
elif [ "$tracked" -eq 1 ]; then
rm -f "$f"
else
reset_failed="$reset_failed $f(git-exit-$tracked)"
if [ -z "$reset_err" ]; then reset_err="${probe_err%%$'\n'*}"; fi
fi
done
if [ -n "$reset_failed" ]; then
# The runner percent-DECODES workflow-command data (%25 -> %, %0D -> CR,
# %0A -> LF), so encode literal percent and line controls before emission.
# Percent must go first so the generated escape sequences stay intact.
reset_err="${reset_err//%/%25}"
reset_err="${reset_err//$'\r'/%0D}"
reset_err="${reset_err//$'\n'/%0A}"
echo "::error title=Could not reset registry configuration::git checkout -- failed for:$reset_failed.${reset_err:+ First git error: $reset_err.} The previous run's registry settings may still be in place. Refusing to select a registry from an unknown state."
exit 1
fi
REGISTRY=""
if [ "$VERDACCIO_FORCE_PUBLIC" = "true" ] && [ -n "$VERDACCIO_TOKEN" ]; then
# Evaluated BEFORE the probe so the override actually overrides: an operator setting
# this wants the public route even when the VIP is up (e.g. the internal cache is
# serving bad data). Gating it on a failed probe would make the flag a no-op.
REGISTRY="https://packages.ever.co/"
echo "::warning title=Verdaccio public route in use::VERDACCIO_FORCE_PUBLIC=true - installing through packages.ever.co, which is known to truncate large tarballs mid-stream."
echo "//packages.ever.co/:_authToken=$VERDACCIO_TOKEN" >> .npmrc
echo "always-auth=true" >> .npmrc
else
if [ "$VERDACCIO_FORCE_PUBLIC" = "true" ]; then
echo "::warning title=VERDACCIO_FORCE_PUBLIC set without a token::VERDACCIO_TOKEN is empty, so packages.ever.co cannot be used; probing the internal VIP instead."
fi
# Retry only where the VIP is expected to answer. An external runner can never reach
# it, so a second attempt would just add dead time to every release job.
if [ "$EXPECT_VIP" = "true" ]; then ATTEMPTS=2; else ATTEMPTS=1; fi
if [ -n "$VERDACCIO_REGISTRY" ]; then
attempt=1
while [ "$attempt" -le "$ATTEMPTS" ]; do
if curl -fsS --connect-timeout 5 --max-time 15 -o /dev/null "${VERDACCIO_REGISTRY%/}/-/ping"; then
REGISTRY="${VERDACCIO_REGISTRY%/}/"
echo "Verdaccio VIP is reachable (attempt $attempt) - using the internal npm cache."
break
fi
attempt=$((attempt + 1))
if [ "$attempt" -le "$ATTEMPTS" ]; then sleep 3; fi
done
fi
if [ -z "$REGISTRY" ] && [ -z "$VERDACCIO_REGISTRY" ]; then
echo "No internal Verdaccio registry is configured (VERDACCIO_REGISTRY is empty) - resolving from the public npm registry."
elif [ -z "$REGISTRY" ]; then
echo "Verdaccio VIP did not answer - resolving from the public npm registry."
# An EXTERNAL runner not reaching the VIP is the designed outcome and stays quiet.
# An IN-NETWORK runner means the internal cache is silently unavailable (this fleet
# has a recurring MetalLB speaker flap that presents exactly this way), so surface
# it. Deliberately a warning, not a failure: a cache blip must never redden a
# release pipeline that npmjs can serve correctly.
if [ "$EXPECT_VIP" = "true" ]; then
echo "::warning title=Verdaccio VIP unreachable from an in-network runner::${RUNNER_NAME:-unknown} fell back to the public npm registry; the internal cache was expected to be reachable here."
fi
fi
fi
if [ -n "$REGISTRY" ]; then
echo "registry=$REGISTRY" >> .npmrc
echo "registry \"$REGISTRY\"" >> .yarnrc
# One pass, no -i: an -i.bak backup survives on a clean: false runner if the step
# fails between the rewrite and its cleanup, and bare -i is not portable to the BSD
# sed on the macOS runners. The temp file is only ever the finished rewrite.
#
# BOTH rewrites are gated on the lockfile existing. This action is consumed cross-repo
# (pnpm and npm repos included): a pnpm repo has NO root yarn.lock, and under the
# composite `bash -eo pipefail` shell an unguarded `sed missing-file` exits 2 and kills
# the job the moment the VIP answers - config-only is all a pnpm repo needs, because
# pnpm-lock.yaml stores no registry host. yarn/npm lock files DO store the resolved
# tarball host, so where those files exist the rewrite is load-bearing: without it the
# install follows the lockfile's npmjs URLs and the registry config is a measured no-op.
if [ -f yarn.lock ]; then
sed -e "s|https://registry.yarnpkg.com|${REGISTRY%/}|g" \
-e "s|https://registry.npmjs.org|${REGISTRY%/}|g" yarn.lock > yarn.lock.rewritten
mv -f yarn.lock.rewritten yarn.lock
fi
# npm follows package-lock.json's `resolved` URLs the same way (integrity hashes are
# host-independent sha512 sums, so `npm ci` still verifies every tarball byte-for-byte).
if [ -f package-lock.json ]; then
sed -e "s|https://registry.npmjs.org|${REGISTRY%/}|g" package-lock.json > package-lock.json.rewritten
mv -f package-lock.json.rewritten package-lock.json
fi
fi
echo "Using npm registry: ${REGISTRY:-https://registry.yarnpkg.com/ (repo default)}"