From d16ca44f0b1fdf2c5eff2630bc7c718320bfe9ff Mon Sep 17 00:00:00 2001 From: Rohit Ghumare Date: Fri, 24 Apr 2026 12:25:33 +0100 Subject: [PATCH] feat(phase-17/20): shadow, canary, and progressive LLM rollouts --- .../assets/rollout.svg | 60 ++++++++ .../20-shadow-canary-progressive/code/main.py | 99 +++++++++++++ .../20-shadow-canary-progressive/docs/en.md | 130 ++++++++++++++++++ .../notebook/.gitkeep | 0 .../outputs/skill-rollout-runbook.md | 31 +++++ 5 files changed, 320 insertions(+) create mode 100644 phases/17-infrastructure-and-production/20-shadow-canary-progressive/assets/rollout.svg create mode 100644 phases/17-infrastructure-and-production/20-shadow-canary-progressive/code/main.py create mode 100644 phases/17-infrastructure-and-production/20-shadow-canary-progressive/docs/en.md create mode 100644 phases/17-infrastructure-and-production/20-shadow-canary-progressive/notebook/.gitkeep create mode 100644 phases/17-infrastructure-and-production/20-shadow-canary-progressive/outputs/skill-rollout-runbook.md diff --git a/phases/17-infrastructure-and-production/20-shadow-canary-progressive/assets/rollout.svg b/phases/17-infrastructure-and-production/20-shadow-canary-progressive/assets/rollout.svg new file mode 100644 index 000000000..588f55a9b --- /dev/null +++ b/phases/17-infrastructure-and-production/20-shadow-canary-progressive/assets/rollout.svg @@ -0,0 +1,60 @@ + + + + + LLM rollout sequence — shadow → canary → A/B → 100% + + + 1. shadow mode + zero user impact + · duplicate prod requests to candidate + · log outputs, token counts, latency + · diff vs production output + · catches: cost spikes, length shifts, + obvious refusal changes, hard errors + not a quality test — a smoke test + + + 2. canary rollout + 1% → 10% → 25% → 50% → 75% → 100% + · five gates at each step: + latency P99 > 1.5x baseline + cost/request > 1.2x baseline + error/refusal > 2x baseline + output-length shift > 1.4x + thumbs-down > 1.5x baseline + + + 3. A/B (optional) + only for distinct alternatives + · 50/50 split + · run until stats significance + · CUPED / sequential / Benjamini-H + · skip if just improved variant + · Phase 17 · 21 covers GrowthBook + + Statsig semantics + + + non-determinism sets the noise floor + up to 15% run-to-run variance on identical inputs + causes: GPU FP non-associativity, batch-size variance, sampling + gates must sit above the noise floor, not at identity with baseline + + + rollback in seconds + policy flag (feature flags) + model pin (registry digest) + rollback = flip flag + revert digest + if rollback requires redeploy you are too slow — fix the stack first + tooling: Argo Rollouts, Flagger, Istio weighted, KServe, feature flag system + diff --git a/phases/17-infrastructure-and-production/20-shadow-canary-progressive/code/main.py b/phases/17-infrastructure-and-production/20-shadow-canary-progressive/code/main.py new file mode 100644 index 000000000..96adad0e8 --- /dev/null +++ b/phases/17-infrastructure-and-production/20-shadow-canary-progressive/code/main.py @@ -0,0 +1,99 @@ +"""Canary rollout simulator — stdlib Python. + +Progressively increases candidate traffic share and checks five gates at each +step. Halts when any gate breaches. Supports injected regressions. +""" + +from __future__ import annotations + +from dataclasses import dataclass +import random + + +STAGES = [0.01, 0.10, 0.25, 0.50, 0.75, 1.00] + +BASELINE = { + "latency_p99_ms": 900, + "cost_per_req": 0.02, + "error_rate": 0.02, + "output_len_p99": 450, + "thumbs_down_rate": 0.03, +} + +GATES = { + "latency_p99_ms": 1.5, + "cost_per_req": 1.2, + "error_rate": 2.0, + "output_len_p99": 1.4, + "thumbs_down_rate": 1.5, +} + + +@dataclass +class Regression: + latency_mult: float = 1.0 + cost_mult: float = 1.0 + error_mult: float = 1.0 + output_len_mult: float = 1.0 + thumbs_down_mult: float = 1.0 + + +def measure_stage(stage: float, reg: Regression, seed: int) -> dict: + rng = random.Random(seed) + noise = lambda v: v * rng.uniform(0.92, 1.08) + return { + "latency_p99_ms": noise(BASELINE["latency_p99_ms"] * reg.latency_mult), + "cost_per_req": noise(BASELINE["cost_per_req"] * reg.cost_mult), + "error_rate": noise(BASELINE["error_rate"] * reg.error_mult), + "output_len_p99": noise(BASELINE["output_len_p99"] * reg.output_len_mult), + "thumbs_down_rate": noise(BASELINE["thumbs_down_rate"] * reg.thumbs_down_mult), + } + + +def check_gates(metrics: dict) -> list[str]: + breaches = [] + for k, mult in GATES.items(): + if metrics[k] > BASELINE[k] * mult: + breaches.append(k) + return breaches + + +def rollout(name: str, reg: Regression) -> None: + print(f"\n{name}") + print(f"Regression: latency={reg.latency_mult}, cost={reg.cost_mult}, error={reg.error_mult}, len={reg.output_len_mult}, thumbs={reg.thumbs_down_mult}") + for i, stage in enumerate(STAGES): + metrics = measure_stage(stage, reg, seed=stage_seed(i)) + breaches = check_gates(metrics) + status = "PASS" if not breaches else f"HALT ({','.join(breaches)})" + pct = int(stage * 100) + print(f" stage {pct:3}% " + f"lat_p99={metrics['latency_p99_ms']:5.0f} " + f"cost=${metrics['cost_per_req']:.4f} " + f"err={metrics['error_rate']*100:4.1f}% " + f"thumbs_dn={metrics['thumbs_down_rate']*100:4.1f}% " + f"{status}") + if breaches: + print(f" → ROLLBACK (policy flip, pinned model reverted)") + return + print(" → PROMOTED to 100%") + + +def stage_seed(i: int) -> int: + return 11 + i * 3 + + +def main() -> None: + print("=" * 95) + print("CANARY ROLLOUT — six stages, five gates, injected regressions") + print("=" * 95) + + rollout("Clean promotion", Regression()) + rollout("Small cost regression (10%) — within gate", Regression(cost_mult=1.10)) + rollout("Cost regression 25%", Regression(cost_mult=1.25)) + rollout("Latency regression 80%", Regression(latency_mult=1.80)) + rollout("Thumbs-down regression 60%", Regression(thumbs_down_mult=1.60)) + rollout("Quality silent + cost creep", Regression(cost_mult=1.15, thumbs_down_mult=1.45)) + + +if __name__ == "__main__": + main() diff --git a/phases/17-infrastructure-and-production/20-shadow-canary-progressive/docs/en.md b/phases/17-infrastructure-and-production/20-shadow-canary-progressive/docs/en.md new file mode 100644 index 000000000..6f4d32680 --- /dev/null +++ b/phases/17-infrastructure-and-production/20-shadow-canary-progressive/docs/en.md @@ -0,0 +1,130 @@ +# Shadow Traffic, Canary Rollout, and Progressive Deployment for LLMs + +> LLM rollouts combine the hardest parts of software deployment: no unit tests, diffuse failure modes, delayed signals. The sequence is (1) shadow mode — duplicate prod requests to candidate model, log, compare with zero user impact; catches obvious distribution issues but is not a quality guarantee; (2) canary rollout — progressive traffic shift 10% → 25% → 50% → 75% → 100% with gates at each step; track latency percentiles, cost/request, error/refusal rate, output length distribution, user-feedback rate; (3) A/B testing for distinct alternatives after stability confirmed. Non-determinism is irreducible — up to 15% accuracy variation across runs with identical inputs due to GPU FP non-associativity plus batch-size variance. Cost is a variable, not constant — a 20% better model can be 3x more expensive per call. Rollback speed is decisive: if rollback requires redeploy, you are too slow. Policy lives in config/flags; model lives in registry with pinned digests; rollback = flip policy + revert threshold + pin old model in seconds. + +**Type:** Learn +**Languages:** Python (stdlib, toy canary-progression simulator) +**Prerequisites:** Phase 17 · 13 (Observability), Phase 17 · 21 (A/B Testing) +**Time:** ~60 minutes + +## Learning Objectives + +- Distinguish shadow mode (zero-impact compare), canary (live traffic progressive), and A/B (stability-confirmed comparison). +- Enumerate five LLM-specific canary metrics (latency, cost/request, error/refusal, output-length distribution, user feedback). +- Explain why LLM non-determinism (up to 15%) changes what "stable" means in a rollout. +- Design a rollback path that takes seconds (policy flip) not hours (redeploy). + +## The Problem + +You ship a new model. Offline evals show 3% accuracy gain. You flip it on in production. Within 24 hours, cost is up 40%, user thumbs-down is up 8%, three customer tickets report "weird answers." You roll back. Redeploy takes 3 hours. Your weekend is ruined. + +Every piece of that was avoidable. Shadow mode would have caught the 40% cost spike before any user saw it. Canary would have stopped at 10% when thumbs-down moved. Policy-flag rollback would have taken 30 seconds. The discipline is what fills in the gap between "offline evals look good" and "real users are happy." + +## The Concept + +### Shadow mode + +Candidate receives the same requests as production; outputs are logged, not returned to users. Zero user impact. Log: + +- Output content (diff against production). +- Token counts (cost delta). +- Latency. +- Refusal and error. + +Catches: cost blow-ups, length regressions, obvious refusal changes, hard errors. Does NOT catch: quality delta users would perceive. Shadow is a smoke test, not a quality test. + +### Canary rollout + +Progressive traffic shift with gates. Typical progression: 1% → 10% → 25% → 50% → 75% → 100%. Gate on 5 metrics at each step: + +1. **Latency percentiles** — P50, P95, P99. Breach: canary has P99 > 1.5x baseline. +2. **Cost per request** — blended $. Breach: >20% above baseline. +3. **Error / refusal rate** — 5xx plus explicit refusals. Breach: 2x baseline. +4. **Output length distribution** — mean + P99. Breach: distributional shift. +5. **User-feedback rate** — thumbs-down / ticket filings. Breach: 1.5x baseline. + +### Non-determinism is the new variance + +Identical inputs produce non-identical outputs. Reasons: + +- GPU FP non-associativity (floating-point reduction order varies by batch). +- Batch-size variance (same prompt in a batch of 128 vs batch of 16). +- Sampling (temperature > 0). + +Measured: up to 15% accuracy variation run-to-run on identical eval sets. "Stable" in a rollout means metrics are within expected variance, not identical to baseline. Set gates above the noise floor. + +### Cost is a variable + +A 20% better model can be 3x more expensive per call. Cost/request is one of the five gates. Shipping a "better" model that breaks unit economics is a rollback case. + +### Rollback is the weapon + +- Policy flag (feature flag system): flip percentage in config; takes seconds. +- Model pinning (registry digest): pinned model does not auto-upgrade. +- Rollback = revert flag + set pinned digest to previous. Seconds, not hours. + +If your stack requires redeploy to rollback, fix that before rolling. + +### Tooling + +**Argo Rollouts** / **Flagger** — Kubernetes progressive delivery controllers. Integrate with Istio/Linkerd weighted routing. + +**Istio weighted routing** — service-mesh-level traffic split. + +**KServe / Seldon Core** — model serving with built-in canary. + +**Feature flags** — LaunchDarkly, Flagsmith, Unleash. Policy-level flip, no redeploy. + +### Metrics cadence + +Canary gates check every 5-15 minutes depending on traffic volume. 1% traffic with 10 req/min gives 50-150 data points per window — enough for latency but noisy for user feedback. 10% gives ~10x more. Progressions should pause long enough to accumulate enough samples at each step. + +### The A/B step is optional + +If the new model is distinctly different (different behavior, different cost curve, different tone), A/B test it at 50% after canary passes. If it's just an improved version, skip to 100% when canary gates pass. + +### Numbers you should remember + +- Canary progression: 1% → 10% → 25% → 50% → 75% → 100%. +- Non-determinism ceiling: up to 15% run-to-run variance on identical inputs. +- Five canary metrics: latency, cost, error/refusal, output length, user feedback. +- Cost gate: >20% above baseline is a breach. +- Rollback: seconds, not hours. + +## Use It + +`code/main.py` simulates a canary rollout with injected regressions. Reports which stage the rollout halts at and which gate triggered. + +## Ship It + +This lesson produces `outputs/skill-rollout-runbook.md`. Given candidate model, baseline, and risk tolerance, designs shadow→canary→100% plan. + +## Exercises + +1. Run `code/main.py`. Inject a 25% cost regression. At which stage does the canary halt? +2. Your new model has 3% accuracy gain offline but cost/request is +18%. Is it a ship? Depends on the policy — write both paths. +3. Design a rollback that takes under 60 seconds end-to-end. List the required infrastructure. +4. Non-determinism shows ±7% on your eval. Set canary gates so you don't false-alarm. What multipliers do you use? +5. Shadow mode catches a 40% cost spike before canary. Write the alert rule that fires in shadow. + +## Key Terms + +| Term | What people say | What it actually means | +|------|----------------|------------------------| +| Shadow mode | "duplicate to new" | Zero-impact send-to-candidate for logging | +| Canary | "progressive traffic" | Gradual user-exposed rollout with gates | +| Gates | "rollout checks" | Metric thresholds that block progression | +| Non-determinism | "LLM variance" | Irreducible run-to-run differences | +| Policy flag | "flag flip rollback" | Config-level rollback, seconds not hours | +| Model pin | "registry digest" | Immutable reference to a model version | +| Argo Rollouts | "K8s progressive" | Kubernetes-native canary/rollback controller | +| KServe | "inference K8s" | Model serving with canary primitives | +| Istio weighted | "mesh split" | Service-mesh traffic splitter | + +## Further Reading + +- [TianPan — Releasing AI Features Without Breaking Production](https://tianpan.co/blog/2026-04-09-llm-gradual-rollout-shadow-canary-ab-testing) +- [MarkTechPost — Safely Deploying ML Models](https://www.marktechpost.com/2026/03/21/safely-deploying-ml-models-to-production-four-controlled-strategies-a-b-canary-interleaved-shadow-testing/) +- [APXML — Advanced LLM Deployment Patterns](https://apxml.com/courses/mlops-for-large-models-llmops/chapter-4-llm-deployment-serving-optimization/advanced-llm-deployment-patterns) +- [Argo Rollouts docs](https://argo-rollouts.readthedocs.io/) +- [Flagger docs](https://docs.flagger.app/) diff --git a/phases/17-infrastructure-and-production/20-shadow-canary-progressive/notebook/.gitkeep b/phases/17-infrastructure-and-production/20-shadow-canary-progressive/notebook/.gitkeep new file mode 100644 index 000000000..e69de29bb diff --git a/phases/17-infrastructure-and-production/20-shadow-canary-progressive/outputs/skill-rollout-runbook.md b/phases/17-infrastructure-and-production/20-shadow-canary-progressive/outputs/skill-rollout-runbook.md new file mode 100644 index 000000000..d889d4151 --- /dev/null +++ b/phases/17-infrastructure-and-production/20-shadow-canary-progressive/outputs/skill-rollout-runbook.md @@ -0,0 +1,31 @@ +--- +name: rollout-runbook +description: Design a shadow → canary → A/B → 100% rollout plan for a new LLM model or prompt template, with five canary gates, noise-floor-aware thresholds, and a seconds-fast rollback path. +version: 1.0.0 +phase: 17 +lesson: 20 +tags: [rollout, canary, shadow, progressive-delivery, feature-flags, argo-rollouts, flagger, kserve] +--- + +Given a candidate change (new model, new prompt template, new router policy), baseline production metrics, and risk tolerance, produce a rollout runbook. + +Produce: + +1. Shadow plan. Duration (24-72 hours). Metrics logged: outputs, token counts, latency, refusal, error. Alert on: >20% cost shift, >30% output length shift, any schema violation. +2. Canary progression. Stages (1% → 10% → 25% → 50% → 75% → 100%). Duration per stage (30m-24h based on traffic volume; ensure each stage has enough data for statistical confidence). +3. Five gates. Specify the exact thresholds for latency P99, cost/request, error/refusal, output-length P99, thumbs-down rate. Set above noise floor (expect 15% irreducible variance). +4. Tooling. Name the rollout controller (Argo Rollouts, Flagger, KServe) and the feature flag system for instant rollback. +5. Rollback path. Document the three actions: flip flag → revert pinned digest → verify. Target time: under 60 seconds end to end. +6. Skip A/B? Justify. Improved-variant changes skip A/B; distinctly different changes (new behavior, new cost curve) require A/B. + +Hard rejects: +- Skipping shadow mode. Refuse — cost spikes and length regressions slip past offline eval. +- Gates tighter than 15% variance. Refuse — false alarms will halt legitimate rollouts. +- Rollback that requires redeploy. Refuse — it is not a rollback, it is a damage report. + +Refusal rules: +- If the change is safety-critical (e.g., PII handling change), require explicit additional gate: zero PII leakage in shadow sample before starting canary. +- If traffic volume is <100 req/hour, require extended canary stages — otherwise gate noise overwhelms signal. +- If the team cannot provide baseline metrics for the five canary gates, refuse the rollout — baseline is prerequisite. + +Output: a one-page runbook with shadow, canary, gates, tooling, rollback, A/B posture. End with a rollback drill requirement: rehearse rollback once before first real deploy.