mirror of
https://github.com/JustVugg/colibri.git
synced 2026-10-02 02:54:37 +08:00
bench(cuda): measure paired resident projection batching
This commit is contained in:
@@ -956,6 +956,10 @@ rans: $(RANSLIB)
|
||||
$(RANSLIB): tools/rans_ctypes.c rans.h
|
||||
$(CC) $(CFLAGS) -fPIC -shared $< -o $@ $(LDFLAGS)
|
||||
|
||||
# Matched resident int8 GPU projection timings; run explicitly, not in CI.
|
||||
tests/bench_cuda_resident_batch$(EXE): tests/bench_cuda_resident_batch.cu backend_cuda.cu backend_cuda.h backend_gpu_compat.h
|
||||
"$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/bench_cuda_resident_batch.cu -o $@ $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS)
|
||||
|
||||
cuda-test: backend_cuda.cu backend_cuda.h backend_gpu_compat.h tests/test_backend_cuda.cu tests/test_ragged_attention.cu tests/test_absorb_determinism.cu tests/test_mxfp4_cuda.cu tests/mxfp4_ref.c tests/test_fp8_warp_cuda.cu tests/test_fp8_cuda.cu tests/test_weights_owned_cuda.cu tests/test_cuda_fmt_trap_cuda.cu tests/test_alloc_footprint_cuda.cu
|
||||
@command -v "$(GPUCC)" >/dev/null 2>&1 || { echo "$(GPUCC_NAME) not found" >&2; exit 1; }
|
||||
"$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/test_backend_cuda.cu -o backend_cuda_test$(EXE) $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS)
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
/* Bounded resident-int8 projection microbenchmark; no model files required.
|
||||
* Times synchronous host API calls (copies + compute), excluding upload.
|
||||
* Compare S one-row calls with one S-row call on identical resident weights.
|
||||
* This is not full-model prefill throughput or a CPU/GPU comparison. */
|
||||
#include "../backend_cuda.h"
|
||||
#include <algorithm>
|
||||
#include <chrono>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <vector>
|
||||
|
||||
static double median(std::vector<double> values) {
|
||||
std::sort(values.begin(),values.end());
|
||||
return values[values.size()/2];
|
||||
}
|
||||
|
||||
static bool measure(int I,int O,int S,int device) {
|
||||
std::vector<int8_t> q((size_t)I*O);
|
||||
std::vector<float> scale(O), x((size_t)S*I), serial((size_t)S*O), batch(serial.size());
|
||||
for(size_t i=0;i<q.size();i++) q[i]=(int)((i*17+i/I*13)%31)-15;
|
||||
for(int o=0;o<O;o++) scale[o]=(o%3+1)/128.f;
|
||||
for(size_t i=0;i<x.size();i++) x[i]=(int(i%17)-8)/64.f;
|
||||
ColiCudaTensor *tensor=nullptr;
|
||||
if(!coli_cuda_tensor_upload(&tensor,q.data(),scale.data(),1,I,O,device)) return false;
|
||||
auto run = [&](bool batched) {
|
||||
float *out=batched?batch.data():serial.data();
|
||||
if(batched) return coli_cuda_matmul(&tensor,out,x.data(),nullptr,nullptr,1,S,I,O,device,0)!=0;
|
||||
for(int row=0;row<S;row++)
|
||||
if(!coli_cuda_matmul(&tensor,out+(size_t)row*O,x.data()+(size_t)row*I,
|
||||
nullptr,nullptr,1,1,I,O,device,0)) return false;
|
||||
return true;
|
||||
};
|
||||
bool ok=true;
|
||||
for(int i=0;i<2 && ok;i++) ok=run(false)&&run(true);
|
||||
std::vector<double> times[2], ratios;
|
||||
for(int rep=0;rep<9 && ok;rep++) {
|
||||
double pair[2]={};
|
||||
for(int arm=0;arm<2 && ok;arm++) {
|
||||
int mode=(rep+arm)%2;
|
||||
auto start=std::chrono::steady_clock::now();
|
||||
ok=run(mode!=0);
|
||||
pair[mode]=std::chrono::duration<double,std::milli>(std::chrono::steady_clock::now()-start).count();
|
||||
times[mode].push_back(pair[mode]);
|
||||
}
|
||||
if(ok) ratios.push_back(pair[0]/pair[1]);
|
||||
for(size_t i=0;i<serial.size() && ok;i++)
|
||||
ok=std::isfinite(serial[i]) && std::isfinite(batch[i]) &&
|
||||
std::fabs(serial[i]-batch[i])<=1e-5f*(1.f+std::fabs(serial[i]));
|
||||
}
|
||||
// Independent sampled CPU reference, outside timed regions. Full outputs
|
||||
// are compared between arms after every pair, not only the final sample.
|
||||
double max_error=0;
|
||||
for(int row=0;row<S && ok;row+=std::max(1,S/7)) for(int o=0;o<O;o+=std::max(1,O/7)) {
|
||||
double reference=0;
|
||||
for(int i=0;i<I;i++) reference+=double(x[(size_t)row*I+i])*q[(size_t)o*I+i];
|
||||
reference*=scale[o];
|
||||
double error=std::fabs(batch[(size_t)row*O+o]-reference);
|
||||
max_error=std::max(max_error,error);
|
||||
if(error>1e-5*(1+std::fabs(reference))) ok=false;
|
||||
}
|
||||
coli_cuda_tensor_free(tensor);
|
||||
if(!ok){std::fprintf(stderr,"FAIL: resident batch I=%d O=%d S=%d\n",I,O,S);return false;}
|
||||
std::printf("{\"input\":%d,\"output\":%d,\"rows\":%d,\"pairs\":9,"
|
||||
"\"serial_median_ms\":%.6f,\"batch_median_ms\":%.6f,"
|
||||
"\"paired_speedup_median\":%.6f,\"cpu_sample_max_abs_error\":%.9g,"
|
||||
"\"serial_ms\":[",I,O,S,median(times[0]),median(times[1]),median(ratios),max_error);
|
||||
for(size_t i=0;i<times[0].size();i++) std::printf("%s%.6f",i?",":"",times[0][i]);
|
||||
std::printf("],\"batch_ms\":[");
|
||||
for(size_t i=0;i<times[1].size();i++) std::printf("%s%.6f",i?",":"",times[1][i]);
|
||||
std::puts("]}");
|
||||
return true;
|
||||
}
|
||||
|
||||
int main() {
|
||||
int device=0;
|
||||
if(!coli_cuda_init(&device,1)) return 77;
|
||||
bool ok=measure(512,1024,32,device) && measure(2048,2048,32,device) && measure(2048,2048,128,device);
|
||||
coli_cuda_shutdown();
|
||||
return ok?0:1;
|
||||
}
|
||||
@@ -0,0 +1,3 @@
|
||||
{"input":512,"output":1024,"rows":32,"pairs":9,"serial_median_ms":2.974592,"batch_median_ms":0.293992,"paired_speedup_median":10.018000,"cpu_sample_max_abs_error":0,"serial_ms":[2.879376,3.728909,2.975827,2.811392,2.950584,2.995621,2.974592,2.966765,3.317602],"batch_ms":[0.291110,0.293992,0.297048,0.313669,0.287397,0.288322,0.299809,0.318531,0.286065]}
|
||||
{"input":2048,"output":2048,"rows":32,"pairs":9,"serial_median_ms":3.588027,"batch_median_ms":0.812655,"paired_speedup_median":4.314296,"cpu_sample_max_abs_error":0,"serial_ms":[2.770069,4.095639,2.697552,3.588027,3.637298,4.933534,4.105374,2.849211,2.814900],"batch_ms":[0.793348,0.812655,0.815726,0.831660,0.812434,0.848347,0.814705,0.811320,0.785599]}
|
||||
{"input":2048,"output":2048,"rows":128,"pairs":9,"serial_median_ms":12.715961,"batch_median_ms":2.912792,"paired_speedup_median":4.365558,"cpu_sample_max_abs_error":0,"serial_ms":[12.715961,13.312398,13.366993,13.219180,15.129256,11.415760,11.160518,10.967418,11.883601],"batch_ms":[2.912792,2.887985,2.912167,2.882372,2.944258,2.923530,2.899905,3.928513,2.944615]}
|
||||
@@ -197,3 +197,46 @@ particular has no int8 analogue (see **Memory**).
|
||||
CPU-only baseline of this engine before the tier: 0.35 tok/s.
|
||||
Numerics: logits cosine vs the f32 CPU reference 0.9992 (dense int8 on),
|
||||
bit-identical GPU-vs-CPU on the same container (cosine 1.0000001).
|
||||
|
||||
## Resident projection batching microbenchmark
|
||||
|
||||
Build and run a bounded, model-free comparison of `S` resident-int8 one-row GPU
|
||||
calls with one `S`-row GPU call:
|
||||
|
||||
```sh
|
||||
make -C c tests/bench_cuda_resident_batch CUDA=1 CUDA_ARCH=sm_89
|
||||
c/tests/bench_cuda_resident_batch > resident-batch.jsonl
|
||||
```
|
||||
|
||||
Set `CUDA_HOME` if the toolkit is not on PATH and select the architecture for
|
||||
your device. The harness uses CUDA device 0, uploads each matrix once, warms
|
||||
both paths twice, then alternates their order over nine paired measurements.
|
||||
The host wall time includes synchronous input/output copies and compute, but
|
||||
excludes upload and validation. All output elements are checked between arms
|
||||
after every pair; sampled elements also have an independent CPU reference.
|
||||
JSONL retains all nine timings per arm and the median of the paired ratios.
|
||||
An execution/validation failure exits nonzero; absent CUDA initialization exits
|
||||
77. The target is explicitly run and is not part of ordinary CI.
|
||||
|
||||
A single run on 2026-09-22 (RTX 4070 12 GB, driver 591.86, CUDA 12.9, `sm_89`,
|
||||
Core Ultra 9 285K, WSL2 Linux 6.6.114.1, `-O3 -ftz=false`) produced:
|
||||
|
||||
| Input × output | Rows | Serial median ms | Batch median ms | Median paired ratio |
|
||||
|---|---:|---:|---:|---:|
|
||||
| 512 × 1024 | 32 | 2.975 | 0.294 | 10.02 |
|
||||
| 2048 × 2048 | 32 | 3.588 | 0.813 | 4.31 |
|
||||
| 2048 × 2048 | 128 | 12.716 | 2.913 | 4.37 |
|
||||
|
||||
[Raw paired timings](experiments/qwen36-resident-batch-2026-09-22.jsonl)
|
||||
include zero sampled CPU-reference error for these deterministic synthetic
|
||||
inputs. The device was not isolated: telemetry before/after showed 4% GPU
|
||||
utilization, approximately 3 GB occupied VRAM and 2550 MHz SM clock. Timings show
|
||||
visible variation; one nine-pair run does not establish reproducibility across
|
||||
sessions, devices or real activation distributions.
|
||||
|
||||
The serial GPU baseline reflects the projection-call pattern replaced by the
|
||||
DeltaNet input batching change. These are synthetic shapes, not complete
|
||||
DeltaNet layers: convolution, recurrent updates, normalization, other projections,
|
||||
model loading and HTTP queueing are excluded. This is **not** an end-to-end model
|
||||
speedup, nor an attention-prefill speedup: the old attention prefill path used
|
||||
CPU batch matmul, which is not an arm in this benchmark.
|
||||
|
||||
Reference in New Issue
Block a user