Files
millerson-overlay.nix/packages/freetoken/package.nix
T
alex 6013b17aad chore(packages): bump pins to current upstream releases
Refresh every package pin so the overlay stops carrying versions upstream
has moved past, and repair the pins where an automated bump had produced a
version that no longer resolves:

- graphify 0.9.61 -> 1.0.0 and back to tag tracking: the branch-tracking
  update script generated v0.9.77-unstable-<date> for a rev = "v${version}"
  src, a tag that never existed
- radar 1.13.1 -> 1.16.2 (upstream retagged its releases as k8s-ui-v*)
- llama-cpp b9645 -> b11439
- hipengine 0.5.0 -> 0.6.1, traycer 1.3.0 -> 1.4.2, freetoken 0.1.2 -> 0.1.3
- awg-tool 0.4.0, freebuff 0.2.19, kubernetes-mcp-server 0.0.67,
  marmel 0-unstable-2026-10-05, open-code-review 1.12.12,
  ds4 0-unstable-2026-09-20
- version-string normalisation for the branch-tracked packages already at
  their newest commit (g3, haivemind, shardr, skillsmcp)

Every changed src hash was re-fetched and verified. omniroute stays at
3.8.50: 3.8.51 changes its npm lockfile, so its npmDepsHash would have to
be recomputed before the bump is usable.

🤖 Generated with Codebuff
Co-Authored-By: Codebuff <noreply@codebuff.com>
2026-10-06 16:46:26 +03:00

192 lines
6.6 KiB
Nix

{
lib,
pkgs,
fetchFromGitHub,
}:
let
# FreeToken is an NVIDIA-GPU inference engine: setup.py builds two C++
# extensions that link the CUDA runtime, and the runtime needs a
# CUDA-enabled torch. CUDA libraries carry the unfree "CUDA EULA" license,
# so this package re-imports the flake's nixpkgs with allowUnfree enabled —
# scoped to freetoken only, leaving the rest of the overlay unchanged.
nixpkgsUnfree = import pkgs.path {
inherit (pkgs.stdenv.hostPlatform) system;
config.allowUnfree = true;
};
py = nixpkgsUnfree.python3Packages;
# torch>=2.11,<2.12: torch-bin 2.11 is this nixpkgs' CUDA build (it links
# the cuda12.9-* libraries). The extensions must link the same libcudart
# instance torch-bin links, so CUDA_HOME below is built from the matching
# cudaPackages.cuda_cudart.
cudart = nixpkgsUnfree.cudaPackages.cuda_cudart;
# cuda_runtime_api.h includes crt/host_defines.h & co., which ship with
# nvcc's header set, not with cudart.
nvccHeaders = nixpkgsUnfree.cudaPackages.cuda_nvcc;
# setup.py requires CUDA_HOME containing CUDA headers and libcudart.
# Single-output cudart provides ${cudart}/include and ${cudart}/lib; the
# crt/ headers come from nvcc; lib64/ is added for torch's cpp_extension,
# lib/ is picked up by setup.py's cuda-home/lib fallback.
cudaHome = nixpkgsUnfree.runCommand "freetoken-cuda-home" { } ''
mkdir -p $out/include $out/lib64
ln -s ${cudart}/include/* $out/include/
ln -s ${nvccHeaders}/include/crt $out/include/crt
ln -s ${cudart}/lib $out/lib
for f in ${cudart}/lib/*.so*; do
ln -s "$f" "$out/lib64/$(basename "$f")"
done
'';
# flashlib==0.3.0, pinned by freetoken, is not in nixpkgs. torch and triton
# come from the prebuilt wheel builds (torch-bin / triton-bin), the same
# instances torch-bin propagates — the scope defaults (source-built torch
# and triton) would put two tritons/torches into the closure.
flashlib = py.callPackage ./flashlib.nix {
torch = py.torch-bin;
triton = py.triton-bin;
};
# freetoken pins apache-tvm-ffi==0.1.13.post3; nixpkgs has 0.1.10.
# 0.1.13 needs cython>=3.2.8 (nixpkgs: 3.2.4), so build it with a newer
# cython, scoped to this package only.
apache-tvm-ffi =
(py.apache-tvm-ffi.override {
cython = py.cython.overridePythonAttrs (old: rec {
version = "3.3.0";
src = fetchFromGitHub {
owner = "cython";
repo = "cython";
tag = version;
hash = "sha256-gIEqq8DAJF197gH1cMuq5JxI4rV/VHS70vg8hywXDqw=";
};
});
}).overridePythonAttrs
(old: {
version = "0.1.13.post3";
src = fetchFromGitHub {
owner = "apache";
repo = "tvm-ffi";
tag = "v0.1.13-post3";
fetchSubmodules = true;
hash = "sha256-AN7AqBl62T8DqnRt7KRvGjqo/c0SJ66QZrCQQ5yicHw=";
};
# 0.1.13 runs its suite with pytest-xdist (pyproject addopts = [ "-n"
# "auto" ]) and touches GPUs; the nixpkgs check inputs have neither
# xdist nor a GPU. The imports check still runs (and passes).
dontUsePytestCheck = true;
});
in
py.buildPythonApplication (finalAttrs: {
pname = "freetoken";
version = "0.1.3";
pyproject = true;
src = fetchFromGitHub {
owner = "FlashML-org";
repo = "FreeToken";
tag = "v${finalAttrs.version}";
hash = "sha256-59itjnwDr4P7I/ZLIbkNw9/8CCPvTP7yNWl9ogBnZ3o=";
};
# setup.py imports torch.utils.cpp_extension at build time and compiles two
# C++ extensions (freetoken.kernel._pinned_tensor and _cpu_moe) that link
# libcudart from CUDA_HOME. ninja is required by BuildExtension.
build-system = with py; [
ninja
setuptools
torch-bin
wheel
];
nativeBuildInputs = [ nixpkgsUnfree.autoPatchelfHook ];
# cudart: RPATH target for the built extensions (autoPatchelf finds
# libcudart.so.12 here); libtorch comes from torch-bin via dependencies.
buildInputs = [ cudart ];
env.CUDA_HOME = cudaHome;
dependencies = with py; [
apache-tvm-ffi
einops
fastapi
flashlib
gguf
huggingface-hub
# freetoken's floor is modelscope>=1.37 (nixpkgs has 1.36.2); the only
# API used is modelscope.snapshot_download (server/args.py), which is
# stable across both.
modelscope
msgpack
numpy
openai
partial-json-parser
prompt-toolkit
pydantic
pyzmq
safetensors
torch-bin
# triton==3.6.0 pin, satisfied with the wheel build (triton-bin) — the
# same instance torch-bin propagates, keeping one triton in the closure.
triton-bin
tqdm
transformers
uvicorn
];
# Tests need an NVIDIA GPU and real model checkpoints (pytest markers
# slow/needs_weights) — impossible in the build sandbox.
doCheck = false;
pythonImportsCheck = [ "freetoken" ];
# Wheel metadata constraints that nixpkgs packages do not match:
# - gguf is versioned by upstream git rev in nixpkgs ("8951") rather than
# PyPI semver; the reader API freetoken uses is stable.
# - modelscope floor is 1.37 (nixpkgs: 1.36.2); only the stable
# snapshot_download API is used.
pythonRelaxDeps = [
"gguf"
"modelscope"
];
passthru = {
category = "AI Inference";
updateScript = [
"nix-update"
"--flake"
".#freetoken"
];
};
meta = {
description = "Local MoE-offload LLM inference runtime with OpenAI- and Anthropic-compatible APIs";
longDescription = ''
FreeToken is an edge-native MoE serving engine for running
frontier-scale open-weight models on consumer NVIDIA GPUs (RTX
30/40/50 series). It treats GPUs, CPUs and host memory as one elastic
inference platform, with semantic-aware KV/expert caching and runtime
VRAM re-allocation between caches. It serves OpenAI- and
Anthropic-compatible APIs.
Nix packaging notes:
- Built against this nixpkgs' CUDA 12.9 stack (torch-bin 2.11 +
cudaPackages.cuda_cudart) instead of upstream's cu130 wheels; CUDA
12.9 supports the same GPU range, including RTX 50 (sm_120).
- The optional native accel extras (flashinfer, sglang-kernel) and the
freetoken-kernel-cache companion wheel are not packaged; the runtime
falls back to its pure-Triton kernels where those would be used.
- Requires an NVIDIA GPU at runtime.
'';
homepage = "https://github.com/FlashML-org/FreeToken";
changelog = "https://github.com/FlashML-org/FreeToken/releases/tag/v${finalAttrs.version}";
# Upstream code is Apache-2.0, but this derivation links CUDA libraries
# under the unfree CUDA EULA (same as nixpkgs' torch-bin).
license = lib.licenses.unfree;
platforms = [ "x86_64-linux" ];
mainProgram = "ft";
};
})