bench(cuda): measure paired resident projection batching

This commit is contained in:
ZacharyZcR
2026-09-22 20:29:24 +08:00
parent 0fc47ccdba
commit aee54ce112
4 changed files with 130 additions and 0 deletions
+4
View File
@@ -956,6 +956,10 @@ rans: $(RANSLIB)
$(RANSLIB): tools/rans_ctypes.c rans.h
$(CC) $(CFLAGS) -fPIC -shared $< -o $@ $(LDFLAGS)
# Matched resident int8 GPU projection timings; run explicitly, not in CI.
tests/bench_cuda_resident_batch$(EXE): tests/bench_cuda_resident_batch.cu backend_cuda.cu backend_cuda.h backend_gpu_compat.h
"$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/bench_cuda_resident_batch.cu -o $@ $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS)
cuda-test: backend_cuda.cu backend_cuda.h backend_gpu_compat.h tests/test_backend_cuda.cu tests/test_ragged_attention.cu tests/test_absorb_determinism.cu tests/test_mxfp4_cuda.cu tests/mxfp4_ref.c tests/test_fp8_warp_cuda.cu tests/test_fp8_cuda.cu tests/test_weights_owned_cuda.cu tests/test_cuda_fmt_trap_cuda.cu tests/test_alloc_footprint_cuda.cu
@command -v "$(GPUCC)" >/dev/null 2>&1 || { echo "$(GPUCC_NAME) not found" >&2; exit 1; }
"$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/test_backend_cuda.cu -o backend_cuda_test$(EXE) $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS)
+80
View File
@@ -0,0 +1,80 @@
/* Bounded resident-int8 projection microbenchmark; no model files required.
* Times synchronous host API calls (copies + compute), excluding upload.
* Compare S one-row calls with one S-row call on identical resident weights.
* This is not full-model prefill throughput or a CPU/GPU comparison. */
#include "../backend_cuda.h"
#include <algorithm>
#include <chrono>
#include <cmath>
#include <cstdio>
#include <vector>
static double median(std::vector<double> values) {
std::sort(values.begin(),values.end());
return values[values.size()/2];
}
static bool measure(int I,int O,int S,int device) {
std::vector<int8_t> q((size_t)I*O);
std::vector<float> scale(O), x((size_t)S*I), serial((size_t)S*O), batch(serial.size());
for(size_t i=0;i<q.size();i++) q[i]=(int)((i*17+i/I*13)%31)-15;
for(int o=0;o<O;o++) scale[o]=(o%3+1)/128.f;
for(size_t i=0;i<x.size();i++) x[i]=(int(i%17)-8)/64.f;
ColiCudaTensor *tensor=nullptr;
if(!coli_cuda_tensor_upload(&tensor,q.data(),scale.data(),1,I,O,device)) return false;
auto run = [&](bool batched) {
float *out=batched?batch.data():serial.data();
if(batched) return coli_cuda_matmul(&tensor,out,x.data(),nullptr,nullptr,1,S,I,O,device,0)!=0;
for(int row=0;row<S;row++)
if(!coli_cuda_matmul(&tensor,out+(size_t)row*O,x.data()+(size_t)row*I,
nullptr,nullptr,1,1,I,O,device,0)) return false;
return true;
};
bool ok=true;
for(int i=0;i<2 && ok;i++) ok=run(false)&&run(true);
std::vector<double> times[2], ratios;
for(int rep=0;rep<9 && ok;rep++) {
double pair[2]={};
for(int arm=0;arm<2 && ok;arm++) {
int mode=(rep+arm)%2;
auto start=std::chrono::steady_clock::now();
ok=run(mode!=0);
pair[mode]=std::chrono::duration<double,std::milli>(std::chrono::steady_clock::now()-start).count();
times[mode].push_back(pair[mode]);
}
if(ok) ratios.push_back(pair[0]/pair[1]);
for(size_t i=0;i<serial.size() && ok;i++)
ok=std::isfinite(serial[i]) && std::isfinite(batch[i]) &&
std::fabs(serial[i]-batch[i])<=1e-5f*(1.f+std::fabs(serial[i]));
}
// Independent sampled CPU reference, outside timed regions. Full outputs
// are compared between arms after every pair, not only the final sample.
double max_error=0;
for(int row=0;row<S && ok;row+=std::max(1,S/7)) for(int o=0;o<O;o+=std::max(1,O/7)) {
double reference=0;
for(int i=0;i<I;i++) reference+=double(x[(size_t)row*I+i])*q[(size_t)o*I+i];
reference*=scale[o];
double error=std::fabs(batch[(size_t)row*O+o]-reference);
max_error=std::max(max_error,error);
if(error>1e-5*(1+std::fabs(reference))) ok=false;
}
coli_cuda_tensor_free(tensor);
if(!ok){std::fprintf(stderr,"FAIL: resident batch I=%d O=%d S=%d\n",I,O,S);return false;}
std::printf("{\"input\":%d,\"output\":%d,\"rows\":%d,\"pairs\":9,"
"\"serial_median_ms\":%.6f,\"batch_median_ms\":%.6f,"
"\"paired_speedup_median\":%.6f,\"cpu_sample_max_abs_error\":%.9g,"
"\"serial_ms\":[",I,O,S,median(times[0]),median(times[1]),median(ratios),max_error);
for(size_t i=0;i<times[0].size();i++) std::printf("%s%.6f",i?",":"",times[0][i]);
std::printf("],\"batch_ms\":[");
for(size_t i=0;i<times[1].size();i++) std::printf("%s%.6f",i?",":"",times[1][i]);
std::puts("]}");
return true;
}
int main() {
int device=0;
if(!coli_cuda_init(&device,1)) return 77;
bool ok=measure(512,1024,32,device) && measure(2048,2048,32,device) && measure(2048,2048,128,device);
coli_cuda_shutdown();
return ok?0:1;
}
@@ -0,0 +1,3 @@
{"input":512,"output":1024,"rows":32,"pairs":9,"serial_median_ms":2.974592,"batch_median_ms":0.293992,"paired_speedup_median":10.018000,"cpu_sample_max_abs_error":0,"serial_ms":[2.879376,3.728909,2.975827,2.811392,2.950584,2.995621,2.974592,2.966765,3.317602],"batch_ms":[0.291110,0.293992,0.297048,0.313669,0.287397,0.288322,0.299809,0.318531,0.286065]}
{"input":2048,"output":2048,"rows":32,"pairs":9,"serial_median_ms":3.588027,"batch_median_ms":0.812655,"paired_speedup_median":4.314296,"cpu_sample_max_abs_error":0,"serial_ms":[2.770069,4.095639,2.697552,3.588027,3.637298,4.933534,4.105374,2.849211,2.814900],"batch_ms":[0.793348,0.812655,0.815726,0.831660,0.812434,0.848347,0.814705,0.811320,0.785599]}
{"input":2048,"output":2048,"rows":128,"pairs":9,"serial_median_ms":12.715961,"batch_median_ms":2.912792,"paired_speedup_median":4.365558,"cpu_sample_max_abs_error":0,"serial_ms":[12.715961,13.312398,13.366993,13.219180,15.129256,11.415760,11.160518,10.967418,11.883601],"batch_ms":[2.912792,2.887985,2.912167,2.882372,2.944258,2.923530,2.899905,3.928513,2.944615]}
+43
View File
@@ -197,3 +197,46 @@ particular has no int8 analogue (see **Memory**).
CPU-only baseline of this engine before the tier: 0.35 tok/s.
Numerics: logits cosine vs the f32 CPU reference 0.9992 (dense int8 on),
bit-identical GPU-vs-CPU on the same container (cosine 1.0000001).
## Resident projection batching microbenchmark
Build and run a bounded, model-free comparison of `S` resident-int8 one-row GPU
calls with one `S`-row GPU call:
```sh
make -C c tests/bench_cuda_resident_batch CUDA=1 CUDA_ARCH=sm_89
c/tests/bench_cuda_resident_batch > resident-batch.jsonl
```
Set `CUDA_HOME` if the toolkit is not on PATH and select the architecture for
your device. The harness uses CUDA device 0, uploads each matrix once, warms
both paths twice, then alternates their order over nine paired measurements.
The host wall time includes synchronous input/output copies and compute, but
excludes upload and validation. All output elements are checked between arms
after every pair; sampled elements also have an independent CPU reference.
JSONL retains all nine timings per arm and the median of the paired ratios.
An execution/validation failure exits nonzero; absent CUDA initialization exits
77. The target is explicitly run and is not part of ordinary CI.
A single run on 2026-09-22 (RTX 4070 12 GB, driver 591.86, CUDA 12.9, `sm_89`,
Core Ultra 9 285K, WSL2 Linux 6.6.114.1, `-O3 -ftz=false`) produced:
| Input × output | Rows | Serial median ms | Batch median ms | Median paired ratio |
|---|---:|---:|---:|---:|
| 512 × 1024 | 32 | 2.975 | 0.294 | 10.02 |
| 2048 × 2048 | 32 | 3.588 | 0.813 | 4.31 |
| 2048 × 2048 | 128 | 12.716 | 2.913 | 4.37 |
[Raw paired timings](experiments/qwen36-resident-batch-2026-09-22.jsonl)
include zero sampled CPU-reference error for these deterministic synthetic
inputs. The device was not isolated: telemetry before/after showed 4% GPU
utilization, approximately 3 GB occupied VRAM and 2550 MHz SM clock. Timings show
visible variation; one nine-pair run does not establish reproducibility across
sessions, devices or real activation distributions.
The serial GPU baseline reflects the projection-call pattern replaced by the
DeltaNet input batching change. These are synthetic shapes, not complete
DeltaNet layers: convolution, recurrent updates, normalization, other projections,
model loading and HTTP queueing are excluded. This is **not** an end-to-end model
speedup, nor an attention-prefill speedup: the old attention prefill path used
CPU batch matmul, which is not an arm in this benchmark.