mirror of
https://github.com/JustVugg/colibri.git
synced 2026-10-02 02:54:37 +08:00
Merge dev into fix/windows-qwen36-cuda-dll: keep dev's qwen36 prerequisites plus the alias and .build-config
This commit is contained in:
+9
-1
@@ -1117,7 +1117,15 @@ QWEN36_TIER_SRC =
|
||||
QWEN36_CFLAGS = $(NOCUDA_CFLAGS)
|
||||
QWEN36_LDFLAGS = $(NOCUDA_LDFLAGS)
|
||||
endif
|
||||
qwen36$(EXE): qwen36.c decode_batch.h serve_poll.h cli_args.h qwen36_tier.h expert_ffn.h idot.h st.h json.h compat.h omp_tune.h kv_prefix.h pin_pool.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h gsgemv.h qgemv.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
|
||||
# On Windows, EXE=.exe. Keep a bare qwen36 target so GNU make does not
|
||||
# fall through to its implicit %: %.c rule and compile qwen36.c alone.
|
||||
ifneq ($(EXE),)
|
||||
.PHONY: qwen36
|
||||
qwen36: qwen36$(EXE)
|
||||
endif
|
||||
# Rebuild when CUDA_DLL changes; otherwise an existing CPU-only executable can
|
||||
# be reported as up to date despite selecting the Windows CUDA DLL tier.
|
||||
qwen36$(EXE): qwen36.c decode_batch.h serve_poll.h cli_args.h qwen36_tier.h expert_ffn.h idot.h st.h json.h compat.h omp_tune.h kv_prefix.h pin_pool.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h gsgemv.h qgemv.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) .build-config
|
||||
$(CC) $(QWEN36_CFLAGS) qwen36.c $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o qwen36$(EXE) $(QWEN36_LDFLAGS)
|
||||
|
||||
# DeepSeek V4.1 Flash: one file, like every other portable engine. The fp4 experts
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
"""Every coli_cuda_* the header exports must have a Windows loader forwarder.
|
||||
|
||||
On Linux a host links backend_cuda.o directly, so a new COLI_CUDA_DLLEXPORT
|
||||
prototype in backend_cuda.h is callable the moment backend_cuda.cu defines it.
|
||||
On Windows the host only sees what backend_loader.c resolves and forwards.
|
||||
A symbol added to the header but not to the loader therefore builds and runs
|
||||
everywhere except a CUDA_DLL=1 host, where it surfaces as an undefined
|
||||
reference -- which is how coli_cuda_available_device_count broke qwen36.exe.
|
||||
|
||||
test_backend_loader.py derives its ABI FROM the loader, so it cannot see this
|
||||
gap. This test compares the loader against the header instead. Pure text, no
|
||||
compiler: it runs on every CI host, not only on Windows.
|
||||
"""
|
||||
import re
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
HERE = Path(__file__).resolve().parent.parent
|
||||
|
||||
# Header exports deliberately absent from the Windows loader, with the reason.
|
||||
# Adding a name here should be as conscious as adding a RESOLVE.
|
||||
NOT_FORWARDED = {
|
||||
# COLI_ANS (DietGPU) is wired only for the Linux CUDA=1 path.
|
||||
"coli_cuda_tensor_upload_compressed",
|
||||
}
|
||||
|
||||
|
||||
def header_exports():
|
||||
src = (HERE / "backend_cuda.h").read_text(encoding="utf-8")
|
||||
return set(re.findall(
|
||||
r"COLI_CUDA_DLLEXPORT[^;(]*?\b(coli_cuda_\w+)\s*\(", src))
|
||||
|
||||
|
||||
def loader_resolved():
|
||||
src = (HERE / "backend_loader.c").read_text(encoding="utf-8")
|
||||
names = re.findall(r"^\s+RESOLVE(?:_OPT)?\((\w+),", src, re.M)
|
||||
return {"coli_cuda_" + n for n in names}
|
||||
|
||||
|
||||
def loader_defined():
|
||||
src = (HERE / "backend_loader.c").read_text(encoding="utf-8")
|
||||
return set(re.findall(
|
||||
r"^[A-Za-z_][\w \t\*]*?\b(coli_cuda_\w+)\s*\([^;]*?\)\s*\{", src, re.M))
|
||||
|
||||
|
||||
class LoaderHeaderParityTest(unittest.TestCase):
|
||||
def test_parsers_found_the_abi(self):
|
||||
# Guard against a regex that silently matches nothing.
|
||||
self.assertGreater(len(header_exports()), 40)
|
||||
self.assertGreater(len(loader_resolved()), 40)
|
||||
self.assertIn("coli_cuda_init", header_exports())
|
||||
|
||||
def test_every_header_export_is_resolved_by_the_loader(self):
|
||||
missing = sorted(header_exports() - loader_resolved() - NOT_FORWARDED)
|
||||
self.assertEqual(missing, [],
|
||||
"declared COLI_CUDA_DLLEXPORT in backend_cuda.h but never "
|
||||
"RESOLVE/RESOLVE_OPT'd in backend_loader.c; a CUDA_DLL=1 "
|
||||
"host calling them fails to link: %s" % missing)
|
||||
|
||||
def test_every_resolved_symbol_has_a_forwarder(self):
|
||||
undefined = sorted(loader_resolved() - loader_defined())
|
||||
self.assertEqual(undefined, [],
|
||||
"resolved from the DLL but no public wrapper is defined "
|
||||
"in backend_loader.c: %s" % undefined)
|
||||
|
||||
def test_exemptions_are_still_real(self):
|
||||
stale = sorted(NOT_FORWARDED - header_exports())
|
||||
self.assertEqual(stale, [], "NOT_FORWARDED names no longer in header: %s"
|
||||
% stale)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -46,6 +46,28 @@ OMP_NUM_THREADS=<physical cores> OMP_WAIT_POLICY=ACTIVE OMP_PROC_BIND=close \
|
||||
SNAP=<container> N_NEW=200 ./c/qwen36 256 4 prompt.txt
|
||||
```
|
||||
|
||||
### Windows (CUDA_DLL=1)
|
||||
|
||||
MinGW cannot link CUDA directly, so the backend is built into `coli_cuda.dll`
|
||||
with nvcc + MSVC and `qwen36.exe` reaches it through `backend_loader.c`.
|
||||
`CUDA=1` is rejected on Windows by design. From an *x64 Native Tools* prompt
|
||||
with MSYS2's `mingw64\bin` and `usr\bin` on `PATH`:
|
||||
|
||||
```cmd
|
||||
cd c
|
||||
make cuda-dll CUDA_ARCH=sm_89
|
||||
make qwen36.exe CUDA_DLL=1 ARCH=native
|
||||
set COLI_CUDA=1
|
||||
set COLI_GPUS=0
|
||||
set CUDA_EXPERT_GB=auto
|
||||
qwen36.exe <same arguments as the CPU build>
|
||||
```
|
||||
|
||||
Keep `coli_cuda.dll` next to `qwen36.exe`, built from the same checkout, and
|
||||
the CUDA toolkit's `bin` directory on `PATH` for `cudart`. A startup line
|
||||
`[gpu] MoE experts -> CUDA VRAM tier` confirms the tier is active; without it
|
||||
the run is CPU-only.
|
||||
|
||||
`cap` (argv[1]) must equal `n_experts` (full RAM residency). int4 containers
|
||||
only (the int8 container keeps the CPU path). `COLI_TIMERS=1` prints
|
||||
per-phase timings and tier telemetry.
|
||||
|
||||
Reference in New Issue
Block a user