diff --git a/packages/kyojin/package.nix b/packages/kyojin/package.nix index f7ee3d3..f6f9b7d 100644 --- a/packages/kyojin/package.nix +++ b/packages/kyojin/package.nix @@ -145,6 +145,14 @@ let sed -i '1i import sys as _sys, types as _types; _exl3_pkg = _types.ModuleType("exllamav3"); _exl3_pkg.__path__ = ["exllamav3"]; _sys.modules.setdefault("exllamav3", _exl3_pkg)' setup.py sed -i 's|^from exllamav3\.exllamav3_ext\.build_config import get_sources as _get_sources$|import sys as _exl3_sys; _exl3_sys.path.insert(0, "exllamav3/exllamav3_ext"); from build_config import get_sources as _get_sources; _exl3_sys.path.pop(0)|' setup.py grep -q 'from build_config import' setup.py + + # Upstream serves these from the checkout (pip install -e .); the wheel only + # packages what pyproject lists, and three things are read out of the + # installed package at run time: the .hip sources the JIT kernels compile, + # the qsa_proof marker gating the QSA prefill kernel, and the dense-GEMM + # tuning seed (without it every start re-tunes for ~7 minutes). + sed -i 's|^ "exllamav3_ext/\*\*/\*",$|&\n "**/*.hip",\n "**/*.ok",\n "model/dense_gemm_tune_seed.txt",|' pyproject.toml + grep -q '"\*\*/\*.hip"' pyproject.toml ''; # tests/ needs a real gfx1151 GPU and the published model packs. @@ -169,6 +177,19 @@ stdenv.mkDerivation { # Only the wrappers are installed here: the serve scripts are plain scripts, # they import their siblings by path, and the module itself is in `library`. + # + # Upstream `tools/strix_halo/env.sh` is sourced before serving, and on a + # distro box that brings a system `cc` (the AMD Triton backend compiles a + # CPython glue module at runtime, and again per kernel launcher) plus a hipcc + # for the JIT HIP kernels (qsa prefill, gdn, kda, ple). Nix gives the wrapper + # neither, so export them here instead: Triton takes the compiler from $CC, + # exllamav3 finds hipcc under $EXL3_ROCM_SDK, and hipclang only finds the + # amdgcn bitcode through $HIP_DEVICE_LIB_PATH (hipcc's own DEVICE_LIB_PATH is + # not enough for a --genco call). The ROCm paths stay out of LD_LIBRARY_PATH + # (upstream Trap 1/5: a second libhsa segfaults torch), and PYTHONPATH gets + # `gr/`: upstream env.sh puts the repo root there so `import gr_mix_hip` (the + # hand-written gated-residual WMMA kernel the server asks for with + # EXL3_GR_HIP=1) resolves, and it is not part of the wheel either. dontConfigure = true; dontBuild = true; @@ -181,6 +202,11 @@ stdenv.mkDerivation { for model in glm mimo qwen; do makeWrapper ${python}/bin/python $out/bin/kyojin-serve-$model \ --set PYTORCH_ROCM_ARCH ${rocmArch} \ + --set CC ${stdenv.cc}/bin/cc \ + --set EXL3_ROCM_SDK ${rocmToolkit} \ + --set HIP_DEVICE_LIB_PATH ${rocmPackages."rocm-device-libs"}/amdgcn/bitcode \ + --prefix PATH : ${lib.makeBinPath [ stdenv.cc ]} \ + --prefix PYTHONPATH : ${src}/gr \ --add-flags "${src}/tools/$model/serve.py" done