diff --git a/pkgs/development/python-modules/flashinfer/default.nix b/pkgs/development/python-modules/flashinfer/default.nix index 5c4c653ed981..905dc5d47a04 100644 --- a/pkgs/development/python-modules/flashinfer/default.nix +++ b/pkgs/development/python-modules/flashinfer/default.nix @@ -15,30 +15,21 @@ ninja, numpy, torch, + pynvml, + einops, }: -let +buildPythonPackage rec { pname = "flashinfer"; - version = "0.2.9"; - - src_cutlass = fetchFromGitHub { - owner = "NVIDIA"; - repo = "cutlass"; - # Using the revision obtained in submodule inside flashinfer's `3rdparty`. - tag = "v${version}"; - hash = "sha256-d4czDoEv0Focf1bJHOVGX4BDS/h5O7RPoM/RrujhgFQ="; - }; - -in -buildPythonPackage { - format = "setuptools"; - inherit pname version; + version = "0.2.14"; + pyproject = true; src = fetchFromGitHub { owner = "flashinfer-ai"; repo = "flashinfer"; tag = "v${version}"; - hash = "sha256-M0q6d+EpuTehbw68AQ73Fhwmw2tzjymYjSXaol9QC7Y="; + hash = "sha256-MZiZwdedz+Vxa1+VBfHDKf4NVSiOAytGboIJ0DvCXmk="; + fetchSubmodules = true; }; build-system = [ setuptools ]; @@ -48,20 +39,16 @@ buildPythonPackage { ninja (lib.getBin cudaPackages.cuda_nvcc) ]; + dontUseCmakeConfigure = true; - buildInputs = [ - cudaPackages.cuda_cudart - cudaPackages.libcublas - cudaPackages.cuda_cccl - cudaPackages.libcurand + buildInputs = with cudaPackages; [ + cuda_cccl + cuda_cudart + libcublas + libcurand ]; - postPatch = '' - rmdir 3rdparty/cutlass - ln -s ${src_cutlass} 3rdparty/cutlass - ''; - # FlashInfer offers two installation modes: # # JIT mode: CUDA kernels are compiled at runtime using PyTorch’s JIT, with @@ -86,6 +73,8 @@ buildPythonPackage { dependencies = [ numpy torch + pynvml + einops ]; meta = with lib; { diff --git a/pkgs/development/python-modules/kserve/default.nix b/pkgs/development/python-modules/kserve/default.nix index 56c1382c95bd..10d86bf89974 100644 --- a/pkgs/development/python-modules/kserve/default.nix +++ b/pkgs/development/python-modules/kserve/default.nix @@ -158,6 +158,14 @@ buildPythonPackage rec { ]; disabledTests = [ + # AssertionError: assert CompletionReq...lm_xargs=None) == CompletionReq...lm_xargs=None) + "test_convert_params" + + # Flaky: ray.exceptions.ActorDiedError: The actor died unexpectedly before finishing this task. + "test_explain" + "test_infer" + "test_predict" + # Require network access "test_infer_graph_endpoint" "test_infer_path_based_routing" diff --git a/pkgs/development/python-modules/mistral-common/default.nix b/pkgs/development/python-modules/mistral-common/default.nix index e245f5b7e524..305581a4ef91 100644 --- a/pkgs/development/python-modules/mistral-common/default.nix +++ b/pkgs/development/python-modules/mistral-common/default.nix @@ -1,57 +1,96 @@ { lib, buildPythonPackage, - fetchPypi, - poetry-core, - numpy, - pydantic, + fetchFromGitHub, + + # build-system + setuptools, + + # dependencies jsonschema, + numpy, opencv-python-headless, - sentencepiece, - typing-extensions, - tiktoken, pillow, + pydantic, + pydantic-extra-types, requests, + sentencepiece, + tiktoken, + typing-extensions, + + # tests + click, + fastapi, + huggingface-hub, + openai, + pycountry, + pydantic-settings, + pytestCheckHook, + soundfile, + soxr, + uvicorn, }: buildPythonPackage rec { pname = "mistral-common"; - version = "1.8.3"; + version = "1.8.4"; pyproject = true; - src = fetchPypi { - pname = "mistral_common"; - inherit version; - hash = "sha256-DRl52CIntiX21xs8goF28FnajQ9aMwfN9TtIQJo5cKQ="; + src = fetchFromGitHub { + owner = "mistralai"; + repo = "mistral-common"; + tag = "v${version}"; + hash = "sha256-HB6dsqiDSLhjyANk7ZT/cU98mjJamegAF0uKH8GfgM8="; }; - pythonRelaxDeps = [ - "pillow" - "tiktoken" - ]; - - build-system = [ poetry-core ]; + build-system = [ setuptools ]; dependencies = [ - numpy - pydantic jsonschema + numpy opencv-python-headless - sentencepiece - typing-extensions - tiktoken pillow + pydantic + pydantic-extra-types requests + sentencepiece + tiktoken + typing-extensions ]; - doCheck = true; - pythonImportsCheck = [ "mistral_common" ]; - meta = with lib; { + nativeCheckInputs = [ + click + fastapi + huggingface-hub + openai + pycountry + pydantic-settings + pytestCheckHook + soundfile + soxr + uvicorn + ]; + + disabledTests = [ + # Require internet + "test_download_gated_image" + "test_image_encoder_formats" + "test_image_processing" + + # AssertionError: Regex pattern did not match. + "test_from_url" + + # AssertionError, Extra items in the right set + "test_openai_chat_fields" + ]; + + meta = { description = "Tools to help you work with Mistral models"; homepage = "https://github.com/mistralai/mistral-common"; - license = licenses.asl20; - maintainers = with maintainers; [ bgamari ]; + changelog = "https://github.com/mistralai/mistral-common/releases/tag/v${version}"; + license = lib.licenses.asl20; + maintainers = with lib.maintainers; [ bgamari ]; }; } diff --git a/pkgs/development/python-modules/openai-harmony/default.nix b/pkgs/development/python-modules/openai-harmony/default.nix new file mode 100644 index 000000000000..74251b28ad4e --- /dev/null +++ b/pkgs/development/python-modules/openai-harmony/default.nix @@ -0,0 +1,61 @@ +{ + lib, + buildPythonPackage, + fetchFromGitHub, + + # build-system + rustPlatform, + + # dependencies + pydantic, + + # optional-dependencies + fastapi, + uvicorn, +}: + +buildPythonPackage rec { + pname = "openai-harmony"; + version = "0.0.4"; + pyproject = true; + + src = fetchFromGitHub { + owner = "openai"; + repo = "harmony"; + rev = "v${version}"; + hash = "sha256-2LOrMLrNR1D3isbjiv5w+1Ni9IMwMEPPTOnG1rm//ag="; + }; + + cargoDeps = rustPlatform.fetchCargoVendor { + inherit pname version src; + hash = "sha256-tQq6PFMYghIfJu8CddbXFKXxr41GJaElbCCQuSpnaqk="; + }; + + nativeBuildInputs = [ + rustPlatform.cargoSetupHook + rustPlatform.maturinBuildHook + ]; + + dependencies = [ + pydantic + ]; + + optional-dependencies = { + demo = [ + fastapi + uvicorn + ]; + }; + + pythonImportsCheck = [ "openai_harmony" ]; + + # Tests require internet access + doCheck = false; + + meta = { + description = "Renderer for the harmony response format to be used with gpt-oss"; + homepage = "https://github.com/openai/harmony"; + license = lib.licenses.asl20; + maintainers = with lib.maintainers; [ GaetanLepage ]; + }; +} diff --git a/pkgs/development/python-modules/vllm/0004-drop-lsmod.patch b/pkgs/development/python-modules/vllm/0004-drop-lsmod.patch deleted file mode 100644 index 3c80f8e24e65..000000000000 --- a/pkgs/development/python-modules/vllm/0004-drop-lsmod.patch +++ /dev/null @@ -1,18 +0,0 @@ ---- a/setup.py -+++ b/setup.py -@@ -340,14 +340,7 @@ def _is_hpu() -> bool: - out = subprocess.run(["hl-smi"], capture_output=True, check=True) - is_hpu_available = out.returncode == 0 - except (FileNotFoundError, PermissionError, subprocess.CalledProcessError): -- if sys.platform.startswith("linux"): -- try: -- output = subprocess.check_output( -- 'lsmod | grep habanalabs | wc -l', shell=True) -- is_hpu_available = int(output) > 0 -- except (ValueError, FileNotFoundError, PermissionError, -- subprocess.CalledProcessError): -- pass -+ is_hpu_available = False - return is_hpu_available - - diff --git a/pkgs/development/python-modules/vllm/0005-drop-intel-reqs.patch b/pkgs/development/python-modules/vllm/0005-drop-intel-reqs.patch index 9bf7d3cdf4d8..f16982ece4a7 100644 --- a/pkgs/development/python-modules/vllm/0005-drop-intel-reqs.patch +++ b/pkgs/development/python-modules/vllm/0005-drop-intel-reqs.patch @@ -1,25 +1,12 @@ -From 7511784ceb9252091a9d63ac6b54dcc67dd2b262 Mon Sep 17 00:00:00 2001 -From: Conroy Cheers -Date: Fri, 13 Jun 2025 17:42:10 +1000 -Subject: [PATCH] drop intel reqs - ---- - requirements/cpu.txt | 3 --- - 1 file changed, 3 deletions(-) - diff --git a/requirements/cpu.txt b/requirements/cpu.txt -index d7b0fc6d8..be2df751b 100644 +index d80354342..7434f32f0 100644 --- a/requirements/cpu.txt +++ b/requirements/cpu.txt -@@ -24,8 +24,5 @@ datasets # for benchmark scripts - # cpu cannot use triton 3.3.0 - triton==3.2.0; platform_machine == "x86_64" +@@ -21,7 +21,4 @@ torchvision; platform_machine != "ppc64le" and platform_machine != "s390x" + torchvision==0.22.0; platform_machine == "ppc64le" + datasets # for benchmark scripts -# Intel Extension for PyTorch, only for x86_64 CPUs -intel-openmp==2024.2.1; platform_machine == "x86_64" --intel_extension_for_pytorch==2.7.0; platform_machine == "x86_64" - py-libnuma; platform_system != "Darwin" - psutil; platform_system != "Darwin" --- -2.49.0 - +-intel_extension_for_pytorch==2.6.0; platform_machine == "x86_64" # torch>2.6.0+cpu has performance regression on x86 platform, see https://github.com/pytorch/pytorch/pull/151218 + triton==3.2.0; platform_machine == "x86_64" # Triton is required for torch 2.6+cpu, as it is imported in torch.compile. diff --git a/pkgs/development/python-modules/vllm/default.nix b/pkgs/development/python-modules/vllm/default.nix index cec9c00eadcc..b5e4edd75edc 100644 --- a/pkgs/development/python-modules/vllm/default.nix +++ b/pkgs/development/python-modules/vllm/default.nix @@ -5,7 +5,6 @@ buildPythonPackage, pythonAtLeast, fetchFromGitHub, - fetchpatch, symlinkJoin, autoAddDriverRunpath, @@ -45,6 +44,8 @@ lm-format-enforcer, prometheus-fastapi-instrumentator, cupy, + cbor2, + pybase64, gguf, einops, importlib-metadata, @@ -69,6 +70,8 @@ bitsandbytes, flashinfer, py-libnuma, + setproctitle, + openai-harmony, # internal dependency - for overriding in overlays vllm-flash-attn ? null, @@ -92,13 +95,13 @@ let shouldUsePkg = pkg: if pkg != null && lib.meta.availableOn stdenv.hostPlatform pkg then pkg else null; - # see CMakeLists.txt, grepping for GIT_TAG near cutlass + # see CMakeLists.txt, grepping for CUTLASS_REVISION # https://github.com/vllm-project/vllm/blob/v${version}/CMakeLists.txt cutlass = fetchFromGitHub { owner = "NVIDIA"; repo = "cutlass"; - tag = "v3.9.2"; - hash = "sha256-teziPNA9csYvhkG5t2ht8W8x5+1YGGbHm8VKx4JoxgI="; + tag = "v4.0.0"; + hash = "sha256-HJY+Go1viPkSVZPEs/NyMtYJzas4mMLiIZF3kNX+WgA="; }; flashmla = stdenv.mkDerivation { @@ -111,8 +114,8 @@ let src = fetchFromGitHub { owner = "vllm-project"; repo = "FlashMLA"; - rev = "575f7724b9762f265bbee5889df9c7d630801845"; - hash = "sha256-8WrKMl0olr0nYV4FRJfwSaJ0F5gWQpssoFMjr9tbHBk="; + rev = "0e43e774597682284358ff2c54530757b654b8d1"; + hash = "sha256-wxL/jtq/lsLg1o+4392KNgfw5TYlW6lqEVbmR3Jl4/Q="; }; dontConfigure = true; @@ -138,8 +141,8 @@ let src = fetchFromGitHub { owner = "vllm-project"; repo = "flash-attention"; - rev = "8798f27777fb57f447070301bf33a9f9c607f491"; - hash = "sha256-UTUvATGN1NU/Bc8qo078q6bEgILLmlrjL7Yk2iAJhg4="; + rev = "57b4e68b9f9d94750b46de8f8dbd2bfcc86edd4f"; + hash = "sha256-c7L7WZVVEnXMOTPBoSp7jhkl9d4TA4sj11QvOSWTDIE="; }; dontConfigure = true; @@ -161,7 +164,7 @@ let cpuSupport = !cudaSupport && !rocmSupport; - # https://github.com/pytorch/pytorch/blob/v2.7.0/torch/utils/cpp_extension.py#L2343-L2345 + # https://github.com/pytorch/pytorch/blob/v2.7.1/torch/utils/cpp_extension.py#L2343-L2345 supportedTorchCudaCapabilities = let real = [ @@ -231,6 +234,7 @@ let libcusolver # cusolverDn.h cuda_nvtx cuda_nvrtc + # cusparselt # cusparseLt.h libcublas ]; @@ -247,7 +251,7 @@ in buildPythonPackage rec { pname = "vllm"; - version = "0.9.1"; + version = "0.10.1.1"; pyproject = true; # https://github.com/vllm-project/vllm/issues/12083 @@ -259,23 +263,12 @@ buildPythonPackage rec { owner = "vllm-project"; repo = "vllm"; tag = "v${version}"; - hash = "sha256-sp7rDpewTPXTVRBJHJMj+8pJDS6wAu0/OTJZwbPPqKc="; + hash = "sha256-lLNjBv5baER0AArX3IV4HWjDZ2jTGXyGIvnHupR8MGM="; }; patches = [ - (fetchpatch { - name = "remove-unused-opentelemetry-semantic-conventions-ai-dep.patch"; - url = "https://github.com/vllm-project/vllm/commit/6a5d7e45f52c3a13de43b8b4fa9033e3b342ebd2.patch"; - hash = "sha256-KYthqu+6XwsYYd80PtfrMMjuRV9+ionccr7EbjE4jJE="; - }) - (fetchpatch { - name = "fall-back-to-gloo-when-nccl-unavailable.patch"; - url = "https://github.com/vllm-project/vllm/commit/aa131a94410683b0a02e74fed2ce95e6c2b6b030.patch"; - hash = "sha256-jNlQZQ8xiW85JWyBjsPZ6FoRQsiG1J8bwzmQjnaWFBg="; - }) ./0002-setup.py-nix-support-respect-cmakeFlags.patch ./0003-propagate-pythonpath.patch - ./0004-drop-lsmod.patch ./0005-drop-intel-reqs.patch ]; @@ -354,6 +347,7 @@ buildPythonPackage rec { aioprometheus blake3 cachetools + cbor2 depyf fastapi llguidance @@ -366,6 +360,7 @@ buildPythonPackage rec { prometheus-fastapi-instrumentator py-cpuinfo pyarrow + pybase64 pydantic python-json-logger python-multipart @@ -393,6 +388,8 @@ buildPythonPackage rec { opentelemetry-api opentelemetry-exporter-otlp bitsandbytes + setproctitle + openai-harmony # vLLM needs Torch's compiler to be present in order to use torch.compile torch.stdenv.cc ] @@ -476,8 +473,5 @@ buildPythonPackage rec { # find_isa "x86_64-darwin" ]; - # ValueError: 'aimv2' is already used by a Transformers config, pick another name. - # Version bump ongoing in https://github.com/NixOS/nixpkgs/pull/429117 - broken = true; }; } diff --git a/pkgs/top-level/python-packages.nix b/pkgs/top-level/python-packages.nix index 1f45c475922e..365d496bd202 100644 --- a/pkgs/top-level/python-packages.nix +++ b/pkgs/top-level/python-packages.nix @@ -10797,6 +10797,8 @@ self: super: with self; { openai-agents = callPackage ../development/python-modules/openai-agents { }; + openai-harmony = callPackage ../development/python-modules/openai-harmony { }; + openai-whisper = callPackage ../development/python-modules/openai-whisper { }; openaiauth = callPackage ../development/python-modules/openaiauth { };