mirror of
https://github.com/rohitg00/ai-engineering-from-scratch.git
synced 2026-10-02 10:04:49 +08:00
feat(phase-17/22): load testing - k6, LLMPerf, GenAI-Perf, GIL and uniformity traps
This commit is contained in:
@@ -0,0 +1,56 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 960 500" font-family="Georgia, 'Times New Roman', serif">
|
||||
<defs>
|
||||
<style>
|
||||
.box { fill: #faf6ef; stroke: #1a1a1a; stroke-width: 1.5; }
|
||||
.bad { fill: #ffe1e1; stroke: #b71c1c; stroke-width: 1.5; }
|
||||
.tool { fill: #dfe9ff; stroke: #2c5ea9; stroke-width: 1.5; }
|
||||
.pattern { fill: #e6f4ea; stroke: #2e7d32; stroke-width: 1.5; }
|
||||
.title { font-size: 16px; font-weight: 700; fill: #1a1a1a; }
|
||||
.head { font-size: 12px; font-weight: 700; fill: #1a1a1a; }
|
||||
.step { font-size: 12px; font-family: 'Menlo', monospace; fill: #222; }
|
||||
.small { font-size: 10px; font-family: 'Menlo', monospace; fill: #555; }
|
||||
.caption { font-size: 11px; fill: #555; font-style: italic; }
|
||||
</style>
|
||||
</defs>
|
||||
<text x="480" y="24" text-anchor="middle" class="title">load testing LLM APIs — two traps, four patterns, five tools</text>
|
||||
|
||||
<rect x="40" y="50" width="440" height="110" class="bad"/>
|
||||
<text x="260" y="72" text-anchor="middle" class="head">GIL trap (Locust stock)</text>
|
||||
<text x="60" y="96" class="small">· client tokenizes under Python GIL</text>
|
||||
<text x="60" y="114" class="small">· competes with request generation</text>
|
||||
<text x="60" y="132" class="small">· tokenization backlog inflates reported ITL</text>
|
||||
<text x="260" y="152" text-anchor="middle" class="caption">your client is the bottleneck, not the server</text>
|
||||
|
||||
<rect x="500" y="50" width="420" height="110" class="bad"/>
|
||||
<text x="710" y="72" text-anchor="middle" class="head">prompt-uniformity trap</text>
|
||||
<text x="520" y="96" class="small">· loop with one prompt = 100% prefix cache</text>
|
||||
<text x="520" y="114" class="small">· request coalescing serves "N concurrent" as 1</text>
|
||||
<text x="520" y="132" class="small">· throughput looks great, production falls over</text>
|
||||
<text x="710" y="152" text-anchor="middle" class="caption">fix: LLMPerf --mean + --stddev input tokens</text>
|
||||
|
||||
<rect x="40" y="180" width="440" height="160" class="tool"/>
|
||||
<text x="260" y="202" text-anchor="middle" class="head">2026 tools</text>
|
||||
<text x="60" y="226" class="step">LLMPerf — Anyscale, Rust tokenizers, streaming</text>
|
||||
<text x="60" y="244" class="step">NVIDIA GenAI-Perf — Triton-backed reference</text>
|
||||
<text x="60" y="262" class="step">LLM-Locust — Locust + GIL fix</text>
|
||||
<text x="60" y="280" class="step">guidellm — large-scale synthetic</text>
|
||||
<text x="60" y="298" class="step">k6 v2026.1.0 + Operator 1.0 GA</text>
|
||||
<text x="60" y="316" class="small"> streaming-aware, CRD-native, best CI gate</text>
|
||||
|
||||
<rect x="500" y="180" width="420" height="160" class="pattern"/>
|
||||
<text x="710" y="202" text-anchor="middle" class="head">four load patterns</text>
|
||||
<text x="520" y="226" class="step">steady — 30-60 min constant RPS</text>
|
||||
<text x="520" y="244" class="small"> catches baseline regressions</text>
|
||||
<text x="520" y="264" class="step">ramp — 0 to target over 15 min</text>
|
||||
<text x="520" y="280" class="small"> catches capacity breakpoint + warm-up</text>
|
||||
<text x="520" y="300" class="step">spike — 3-10x sudden burst</text>
|
||||
<text x="520" y="316" class="small"> catches autoscaling + cold-start impact</text>
|
||||
<text x="520" y="328" class="step">soak — 4-8h steady</text>
|
||||
|
||||
<rect x="40" y="360" width="880" height="130" class="box"/>
|
||||
<text x="480" y="382" text-anchor="middle" class="head">CI gate recipe</text>
|
||||
<text x="480" y="406" text-anchor="middle" class="step">k6 on PR with 30-50 iterations at baseline RPS</text>
|
||||
<text x="480" y="426" text-anchor="middle" class="step">gate: P50 / P95 TTFT, 5xx < 5%, TPOT threshold</text>
|
||||
<text x="480" y="446" text-anchor="middle" class="caption">break the build on breach — treat performance as a compile error</text>
|
||||
<text x="480" y="468" text-anchor="middle" class="small">GenAI-Perf ITL excludes TTFT · LLMPerf includes it — same server, different TPOT</text>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 3.9 KiB |
@@ -0,0 +1,89 @@
|
||||
"""Load-test anti-pattern demonstrator — stdlib Python.
|
||||
|
||||
Simulates how uniform prompts inflate reported throughput via prefix-cache
|
||||
and request-coalescing, while realistic distribution reveals the true ceiling.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
import random
|
||||
import statistics
|
||||
|
||||
|
||||
PREFIX_CACHE_HIT_TTFT_MS = 80
|
||||
PREFIX_CACHE_MISS_TTFT_MS = 800
|
||||
TPOT_MS = 15
|
||||
BATCH_EFFICIENCY_SHARED_PREFIX = 0.8 # batch serves 1/0.8 = 1.25x fewer slots
|
||||
|
||||
|
||||
@dataclass
|
||||
class Request:
|
||||
prompt_tokens: int
|
||||
prefix_hash: str
|
||||
|
||||
|
||||
def make_uniform_workload(n: int = 500) -> list[Request]:
|
||||
return [Request(2000, "single_prefix") for _ in range(n)]
|
||||
|
||||
|
||||
def make_realistic_workload(n: int = 500, seed: int = 7) -> list[Request]:
|
||||
rng = random.Random(seed)
|
||||
reqs = []
|
||||
prefixes = [f"prefix_{i}" for i in range(80)]
|
||||
for _ in range(n):
|
||||
prompt = max(50, int(rng.gauss(500, 180)))
|
||||
reqs.append(Request(prompt, rng.choice(prefixes)))
|
||||
return reqs
|
||||
|
||||
|
||||
def simulate(reqs: list[Request], concurrency: int) -> dict:
|
||||
cache: set[str] = set()
|
||||
ttft_samples: list[float] = []
|
||||
# serialize in groups of "concurrency"
|
||||
for i in range(0, len(reqs), concurrency):
|
||||
batch = reqs[i:i + concurrency]
|
||||
unique_prefixes = len({r.prefix_hash for r in batch})
|
||||
for r in batch:
|
||||
hit = r.prefix_hash in cache
|
||||
ttft = PREFIX_CACHE_HIT_TTFT_MS if hit else PREFIX_CACHE_MISS_TTFT_MS
|
||||
if not hit:
|
||||
cache.add(r.prefix_hash)
|
||||
ttft_samples.append(ttft)
|
||||
ttft_samples.sort()
|
||||
p50 = ttft_samples[len(ttft_samples) // 2]
|
||||
p99 = ttft_samples[int(len(ttft_samples) * 0.99) - 1]
|
||||
return {
|
||||
"n": len(reqs),
|
||||
"p50": p50,
|
||||
"p99": p99,
|
||||
"mean": statistics.mean(ttft_samples),
|
||||
"cache_hits": sum(1 for t in ttft_samples if t == PREFIX_CACHE_HIT_TTFT_MS),
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
print("=" * 95)
|
||||
print("PROMPT-UNIFORMITY TRAP — same test harness, different prompt distributions")
|
||||
print("=" * 95)
|
||||
|
||||
for concurrency in (10, 50, 200):
|
||||
print(f"\nConcurrency = {concurrency}")
|
||||
header = f"{'Workload':22} {'n':>5} {'TTFT_P50':>9} {'TTFT_P99':>9} {'mean':>7} cache_hits"
|
||||
print(header)
|
||||
print("-" * len(header))
|
||||
|
||||
uniform = make_uniform_workload(500)
|
||||
u = simulate(uniform, concurrency)
|
||||
print(f"{'UNIFORM':22} {u['n']:5} {u['p50']:8.0f}ms {u['p99']:8.0f}ms {u['mean']:6.0f}ms {u['cache_hits']:4}")
|
||||
|
||||
realistic = make_realistic_workload(500)
|
||||
r = simulate(realistic, concurrency)
|
||||
print(f"{'REALISTIC':22} {r['n']:5} {r['p50']:8.0f}ms {r['p99']:8.0f}ms {r['mean']:6.0f}ms {r['cache_hits']:4}")
|
||||
|
||||
print("\nRead: uniform prompts make your endpoint look fast. Realistic prompts tell the truth.")
|
||||
print("LLMPerf: --mean-input-tokens + --stddev-input-tokens. Always.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,124 @@
|
||||
# Load Testing LLM APIs — Why k6 and Locust Lie
|
||||
|
||||
> Traditional load testers were not designed for streaming responses, variable output lengths, token-level metrics, or GPU saturation. Two traps bite most teams. The GIL trap: Locust's token-level measurement runs tokenization under the Python GIL, which competes with request generation under heavy concurrency; tokenization backlog then inflates reported inter-token latency — your client is the bottleneck, not the server. The prompt-uniformity trap: identical prompts in a loop test one point on the token distribution; real traffic has variable length and diverse prefix matches. LLMPerf fixes this with `--mean-input-tokens` + `--stddev-input-tokens`. Tool mapping in 2026: LLM-specialized (GenAI-Perf, LLMPerf, LLM-Locust, guidellm) for token-level accuracy; **k6 v2026.1.0** + **k6 Operator 1.0 GA (Sept 2025)** — streaming-aware, Kubernetes-native distributed via TestRun/PrivateLoadZone CRDs, best for CI/CD gates; Vegeta for Go constant-rate saturation; Locust 2.43.3 only with LLM-Locust extension for streaming. Load patterns: steady-state, ramp, spike (autoscaling test), soak (memory leaks).
|
||||
|
||||
**Type:** Build
|
||||
**Languages:** Python (stdlib, toy realistic-prompt generator + latency collector)
|
||||
**Prerequisites:** Phase 17 · 08 (Inference Metrics), Phase 17 · 03 (GPU Autoscaling)
|
||||
**Time:** ~75 minutes
|
||||
|
||||
## Learning Objectives
|
||||
|
||||
- Explain the two anti-patterns (GIL trap, prompt-uniformity trap) that make generic load testers lie for LLM APIs.
|
||||
- Pick a tool for a given purpose: LLMPerf (benchmark run), k6 + streaming extension (CI gate), guidellm (large-scale synthetic), GenAI-Perf (NVIDIA reference).
|
||||
- Design four load patterns (steady, ramp, spike, soak) and name the failure mode each catches.
|
||||
- Build a realistic prompt distribution using mean + stddev of input tokens rather than fixed length.
|
||||
|
||||
## The Problem
|
||||
|
||||
You k6-tested your LLM endpoint at 500 concurrent users. It held. You shipped. In production at 200 actual users the service fell over — P99 TTFT exploded, GPUs pinned.
|
||||
|
||||
Two things happened. First, k6 sent 500 identical prompts — your request-coalescing and prefix caching made it look like you were handling 500 concurrent decodes when you were actually handling one. Second, k6 doesn't track inter-token latency on streaming responses the way the eye experiences it; it sees one HTTP connection, not 500 tokens arriving at varying intervals.
|
||||
|
||||
Load testing for LLMs is its own discipline.
|
||||
|
||||
## The Concept
|
||||
|
||||
### The GIL trap (Locust)
|
||||
|
||||
Locust uses Python and runs tokenization client-side under the GIL. Under high concurrency the tokenizer queues behind request generation. Reported inter-token latency includes client-side tokenization backlog. You think the server is slow; it's the test harness.
|
||||
|
||||
Fix: LLM-Locust extension moves tokenization to separate processes, or use a compiled-language harness (k6, LLMPerf using tokenizers.rs).
|
||||
|
||||
### The prompt-uniformity trap
|
||||
|
||||
All known load testers let you configure one prompt. In a loop test of 10,000 iterations the exact same prompt sends each time. Server sees the same prefix every time — prefix cache hits approach 100%, throughput looks great.
|
||||
|
||||
Fix: sample from a prompt distribution. LLMPerf uses `--mean-input-tokens 500 --stddev-input-tokens 150` — diverse lengths, diverse content.
|
||||
|
||||
### Four load patterns
|
||||
|
||||
1. **Steady-state** — constant RPS for 30-60 min. Catches: baseline performance regressions.
|
||||
2. **Ramp** — linearly increase RPS from 0 to target over 15 min. Catches: capacity breakpoint, warm-up anomalies.
|
||||
3. **Spike** — sudden 3-10x RPS for 2 min then back. Catches: autoscaling latency, queue saturation, cold-start impact.
|
||||
4. **Soak** — steady-state for 4-8 hours. Catches: memory leaks, connection-pool drift, observability overflow.
|
||||
|
||||
### 2026 tool mapping
|
||||
|
||||
**LLMPerf** (Anyscale) — Python but Rust-backed tokenization. Mean/stddev prompts. Streaming-aware. Best default for performance runs.
|
||||
|
||||
**NVIDIA GenAI-Perf** — NVIDIA's reference. Uses Triton client; comprehensive metric coverage. Note its ITL excludes TTFT; LLMPerf's includes it. Two tools produce different TPOT for the same server.
|
||||
|
||||
**LLM-Locust** (TrueFoundry) — Locust extension that fixes the GIL trap. Familiar Locust DSL + streaming metrics.
|
||||
|
||||
**guidellm** — large-scale synthetic benchmarking.
|
||||
|
||||
**k6 v2026.1.0** + **k6 Operator 1.0 GA (Sept 2025)**:
|
||||
- k6 itself (Go, compiled, no GIL) added streaming-aware metrics.
|
||||
- k6 Operator uses TestRun / PrivateLoadZone CRDs for Kubernetes-native distributed testing.
|
||||
- Best for CI/CD gates and SLA testing.
|
||||
|
||||
**Vegeta** — Go, simpler than k6. Constant-rate HTTP saturation. Not LLM-aware but good for gateway / rate-limit testing.
|
||||
|
||||
**Locust 2.43.3 stock** — has the GIL trap for LLM. Only with LLM-Locust extension.
|
||||
|
||||
### SLA gate in CI
|
||||
|
||||
Run k6 on the PR with:
|
||||
|
||||
- 30-50 iterations each at baseline RPS.
|
||||
- Gate: P50/P95 TTFT, 5xx < 5%, TPOT under threshold.
|
||||
- Break the build on breach.
|
||||
|
||||
### Realistic prompt distribution
|
||||
|
||||
Build from real traffic samples (if you have them) or from published distributions (e.g., ShareGPT prompts for chat, HumanEval for code). Feed the mean + stddev to LLMPerf. Avoid loop-with-one-prompt at all costs.
|
||||
|
||||
### Numbers you should remember
|
||||
|
||||
- k6 Operator 1.0 GA: September 2025.
|
||||
- k6 v2026.1.0: streaming-aware metrics.
|
||||
- Typical LLMPerf run: 100-1000 requests at concurrency X.
|
||||
- Typical CI gate: 30-50 iterations per PR.
|
||||
- Four patterns: steady, ramp, spike, soak.
|
||||
|
||||
## Use It
|
||||
|
||||
`code/main.py` simulates a load test with realistic prompt distribution, measures effective TPOT, and demonstrates the uniform-prompt trap.
|
||||
|
||||
## Ship It
|
||||
|
||||
This lesson produces `outputs/skill-load-test-plan.md`. Given workload and SLA, picks tool and designs the four load patterns.
|
||||
|
||||
## Exercises
|
||||
|
||||
1. Run `code/main.py`. Compare uniform vs realistic distribution — where is the gap?
|
||||
2. Write the k6 script for a CI gate: TTFT P95 < 800 ms at 100 concurrent, runtime 5 minutes.
|
||||
3. Your soak test shows memory growing 50 MB/hour. Name three causes and the instrumentation to pick between them.
|
||||
4. Spike test from 10 RPS to 100 RPS. What's the expected recovery time if Karpenter + vLLM production-stack are in place (Phase 17 · 03 + 18)?
|
||||
5. GenAI-Perf reports TPOT=6ms; LLMPerf reports TPOT=11ms on the same server. Explain.
|
||||
|
||||
## Key Terms
|
||||
|
||||
| Term | What people say | What it actually means |
|
||||
|------|----------------|------------------------|
|
||||
| LLMPerf | "the LLM harness" | Anyscale benchmark tool, streaming-aware |
|
||||
| GenAI-Perf | "NVIDIA tool" | NVIDIA reference harness |
|
||||
| LLM-Locust | "Locust for LLMs" | Locust extension fixing GIL trap |
|
||||
| guidellm | "synthetic benchmark" | Large-scale synthetic tool |
|
||||
| k6 Operator | "K8s k6" | CRD-based distributed k6 |
|
||||
| GIL trap | "Python client overhead" | Tokenization backlog inflates reported latency |
|
||||
| Prompt-uniformity trap | "single-prompt lie" | Loop with same prompt hits cache, inflates throughput |
|
||||
| Steady-state | "constant load" | Flat RPS for N minutes |
|
||||
| Ramp | "linear up" | 0 to target over duration |
|
||||
| Spike | "burst test" | Sudden multiplier then revert |
|
||||
| Soak | "long test" | Hours for leak detection |
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [TianPan — Load Testing LLM Applications](https://tianpan.co/blog/2026-03-19-load-testing-llm-applications)
|
||||
- [PremAI — Load Testing LLMs 2026](https://blog.premai.io/load-testing-llms-tools-metrics-realistic-traffic-simulation-2026/)
|
||||
- [NVIDIA NIM — Introduction to LLM Inference Benchmarking](https://docs.nvidia.com/nim/large-language-models/1.0.0/benchmarking.html)
|
||||
- [TrueFoundry — LLM-Locust](https://www.truefoundry.com/blog/llm-locust-a-tool-for-benchmarking-llm-performance)
|
||||
- [LLMPerf](https://github.com/ray-project/llmperf)
|
||||
- [k6 Operator](https://github.com/grafana/k6-operator)
|
||||
+31
@@ -0,0 +1,31 @@
|
||||
---
|
||||
name: load-test-plan
|
||||
description: Design a realistic LLM load test — pick tool (LLMPerf, k6, GenAI-Perf, guidellm), build four patterns (steady, ramp, spike, soak), and gate in CI.
|
||||
version: 1.0.0
|
||||
phase: 17
|
||||
lesson: 22
|
||||
tags: [load-testing, llmperf, k6, genai-perf, guidellm, llm-locust, ci-gate]
|
||||
---
|
||||
|
||||
Given workload (endpoint, SLA for TTFT/TPOT/error), target scale (concurrency, RPS), and CI posture (PR gate or release-only), produce a load test plan.
|
||||
|
||||
Produce:
|
||||
|
||||
1. Tool. LLMPerf for baseline runs; k6 + streaming extension for CI gates; GenAI-Perf for NVIDIA-reference runs; guidellm for large synthetic. LLM-Locust only if already on Locust.
|
||||
2. Prompt distribution. Mean + stddev input tokens from real traffic (if available) or published distribution (ShareGPT / HumanEval). Forbid loop-with-one-prompt.
|
||||
3. Four patterns. Steady, ramp, spike, soak. For each: target RPS, duration, expected failure mode.
|
||||
4. CI gate. Specific thresholds: TTFT P95 < X, 5xx < 5%, TPOT < Y. Runtime per PR: 3-5 min.
|
||||
5. Metric alignment. Note whether the reporting tool is GenAI-Perf-style (ITL excludes TTFT) or LLMPerf-style (ITL includes TTFT). Pick one and stay consistent.
|
||||
6. Output. A script file (k6 JS, LLMPerf CLI) committed to the repo.
|
||||
|
||||
Hard rejects:
|
||||
- Load test with uniform prompts. Refuse — the numbers lie.
|
||||
- Load test without streaming support. Refuse — LLM endpoints are streaming by default.
|
||||
- Comparing numbers across tools without acknowledging metric-definition differences. Refuse.
|
||||
|
||||
Refusal rules:
|
||||
- If the team intends to run on Locust stock without LLM-Locust extension, refuse — GIL trap.
|
||||
- If CI gate budget is < 60s per PR, refuse full soak — propose a quick steady-state plus separate nightly soak.
|
||||
- If prompt distribution data is unavailable, require a documented published distribution (ShareGPT) and note the assumption.
|
||||
|
||||
Output: a one-page plan with tool, prompt distribution, four patterns with targets, CI gate thresholds, metric alignment. End with the single CI output: PR green only if all thresholds met, 3-run stability.
|
||||
Reference in New Issue
Block a user