Merge dev into fix/windows-qwen36-cuda-dll: keep dev's qwen36 prerequisites plus the alias and .build-config

This commit is contained in:
JustVugg
2026-09-22 22:28:50 +02:00
3 changed files with 104 additions and 1 deletions
+9 -1
View File
@@ -1117,7 +1117,15 @@ QWEN36_TIER_SRC =
QWEN36_CFLAGS = $(NOCUDA_CFLAGS)
QWEN36_LDFLAGS = $(NOCUDA_LDFLAGS)
endif
qwen36$(EXE): qwen36.c decode_batch.h serve_poll.h cli_args.h qwen36_tier.h expert_ffn.h idot.h st.h json.h compat.h omp_tune.h kv_prefix.h pin_pool.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h gsgemv.h qgemv.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
# On Windows, EXE=.exe. Keep a bare qwen36 target so GNU make does not
# fall through to its implicit %: %.c rule and compile qwen36.c alone.
ifneq ($(EXE),)
.PHONY: qwen36
qwen36: qwen36$(EXE)
endif
# Rebuild when CUDA_DLL changes; otherwise an existing CPU-only executable can
# be reported as up to date despite selecting the Windows CUDA DLL tier.
qwen36$(EXE): qwen36.c decode_batch.h serve_poll.h cli_args.h qwen36_tier.h expert_ffn.h idot.h st.h json.h compat.h omp_tune.h kv_prefix.h pin_pool.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h gsgemv.h qgemv.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) .build-config
$(CC) $(QWEN36_CFLAGS) qwen36.c $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o qwen36$(EXE) $(QWEN36_LDFLAGS)
# DeepSeek V4.1 Flash: one file, like every other portable engine. The fp4 experts
@@ -0,0 +1,73 @@
"""Every coli_cuda_* the header exports must have a Windows loader forwarder.
On Linux a host links backend_cuda.o directly, so a new COLI_CUDA_DLLEXPORT
prototype in backend_cuda.h is callable the moment backend_cuda.cu defines it.
On Windows the host only sees what backend_loader.c resolves and forwards.
A symbol added to the header but not to the loader therefore builds and runs
everywhere except a CUDA_DLL=1 host, where it surfaces as an undefined
reference -- which is how coli_cuda_available_device_count broke qwen36.exe.
test_backend_loader.py derives its ABI FROM the loader, so it cannot see this
gap. This test compares the loader against the header instead. Pure text, no
compiler: it runs on every CI host, not only on Windows.
"""
import re
import unittest
from pathlib import Path
HERE = Path(__file__).resolve().parent.parent
# Header exports deliberately absent from the Windows loader, with the reason.
# Adding a name here should be as conscious as adding a RESOLVE.
NOT_FORWARDED = {
# COLI_ANS (DietGPU) is wired only for the Linux CUDA=1 path.
"coli_cuda_tensor_upload_compressed",
}
def header_exports():
src = (HERE / "backend_cuda.h").read_text(encoding="utf-8")
return set(re.findall(
r"COLI_CUDA_DLLEXPORT[^;(]*?\b(coli_cuda_\w+)\s*\(", src))
def loader_resolved():
src = (HERE / "backend_loader.c").read_text(encoding="utf-8")
names = re.findall(r"^\s+RESOLVE(?:_OPT)?\((\w+),", src, re.M)
return {"coli_cuda_" + n for n in names}
def loader_defined():
src = (HERE / "backend_loader.c").read_text(encoding="utf-8")
return set(re.findall(
r"^[A-Za-z_][\w \t\*]*?\b(coli_cuda_\w+)\s*\([^;]*?\)\s*\{", src, re.M))
class LoaderHeaderParityTest(unittest.TestCase):
def test_parsers_found_the_abi(self):
# Guard against a regex that silently matches nothing.
self.assertGreater(len(header_exports()), 40)
self.assertGreater(len(loader_resolved()), 40)
self.assertIn("coli_cuda_init", header_exports())
def test_every_header_export_is_resolved_by_the_loader(self):
missing = sorted(header_exports() - loader_resolved() - NOT_FORWARDED)
self.assertEqual(missing, [],
"declared COLI_CUDA_DLLEXPORT in backend_cuda.h but never "
"RESOLVE/RESOLVE_OPT'd in backend_loader.c; a CUDA_DLL=1 "
"host calling them fails to link: %s" % missing)
def test_every_resolved_symbol_has_a_forwarder(self):
undefined = sorted(loader_resolved() - loader_defined())
self.assertEqual(undefined, [],
"resolved from the DLL but no public wrapper is defined "
"in backend_loader.c: %s" % undefined)
def test_exemptions_are_still_real(self):
stale = sorted(NOT_FORWARDED - header_exports())
self.assertEqual(stale, [], "NOT_FORWARDED names no longer in header: %s"
% stale)
if __name__ == "__main__":
unittest.main()
+22
View File
@@ -46,6 +46,28 @@ OMP_NUM_THREADS=<physical cores> OMP_WAIT_POLICY=ACTIVE OMP_PROC_BIND=close \
SNAP=<container> N_NEW=200 ./c/qwen36 256 4 prompt.txt
```
### Windows (CUDA_DLL=1)
MinGW cannot link CUDA directly, so the backend is built into `coli_cuda.dll`
with nvcc + MSVC and `qwen36.exe` reaches it through `backend_loader.c`.
`CUDA=1` is rejected on Windows by design. From an *x64 Native Tools* prompt
with MSYS2's `mingw64\bin` and `usr\bin` on `PATH`:
```cmd
cd c
make cuda-dll CUDA_ARCH=sm_89
make qwen36.exe CUDA_DLL=1 ARCH=native
set COLI_CUDA=1
set COLI_GPUS=0
set CUDA_EXPERT_GB=auto
qwen36.exe <same arguments as the CPU build>
```
Keep `coli_cuda.dll` next to `qwen36.exe`, built from the same checkout, and
the CUDA toolkit's `bin` directory on `PATH` for `cudart`. A startup line
`[gpu] MoE experts -> CUDA VRAM tier` confirms the tier is active; without it
the run is CPU-only.
`cap` (argv[1]) must equal `n_experts` (full RAM residency). int4 containers
only (the int8 container keeps the CPU path). `COLI_TIMERS=1` prints
per-phase timings and tier telemetry.