KBaba7/llama.cpp
0
1{2 lib,3 glibc,4 config,5 stdenv,6 runCommand,7 cmake,8 ninja,9 pkg-config,10 git,11 mpi,12 blas,13 cudaPackages,14 autoAddDriverRunpath,15 darwin,16 rocmPackages,17 vulkan-headers,18 vulkan-loader,19 curl,20 shaderc,21 useBlas ?22 builtins.all (x: !x) [23 useCuda24 useMetalKit25 useRocm26 useVulkan27 ]28 && blas.meta.available,29 useCuda ? config.cudaSupport,30 useMetalKit ? stdenv.isAarch64 && stdenv.isDarwin,31 # Increases the runtime closure size by ~700M32 useMpi ? false,33 useRocm ? config.rocmSupport,34 rocmGpuTargets ? builtins.concatStringsSep ";" rocmPackages.clr.gpuTargets,35 enableCurl ? true,36 useVulkan ? false,37 llamaVersion ? "0.0.0", # Arbitrary version, substituted by the flake38 39 # It's necessary to consistently use backendStdenv when building with CUDA support,40 # otherwise we get libstdc++ errors downstream.41 effectiveStdenv ? if useCuda then cudaPackages.backendStdenv else stdenv,42 enableStatic ? effectiveStdenv.hostPlatform.isStatic,43 precompileMetalShaders ? false,44}:45 46let47 inherit (lib)48 cmakeBool49 cmakeFeature50 optionals51 strings52 ;53 54 stdenv = throw "Use effectiveStdenv instead";55 56 suffices =57 lib.optionals useBlas [ "BLAS" ]58 ++ lib.optionals useCuda [ "CUDA" ]59 ++ lib.optionals useMetalKit [ "MetalKit" ]60 ++ lib.optionals useMpi [ "MPI" ]61 ++ lib.optionals useRocm [ "ROCm" ]62 ++ lib.optionals useVulkan [ "Vulkan" ];63 64 pnameSuffix =65 strings.optionalString (suffices != [ ])66 "-${strings.concatMapStringsSep "-" strings.toLower suffices}";67 descriptionSuffix = strings.optionalString (68 suffices != [ ]69 ) ", accelerated with ${strings.concatStringsSep ", " suffices}";70 71 xcrunHost = runCommand "xcrunHost" { } ''72 mkdir -p $out/bin73 ln -s /usr/bin/xcrun $out/bin74 '';75 76 # apple_sdk is supposed to choose sane defaults, no need to handle isAarch6477 # separately78 darwinBuildInputs =79 with darwin.apple_sdk.frameworks;80 [81 Accelerate82 CoreVideo83 CoreGraphics84 ]85 ++ optionals useMetalKit [ MetalKit ];86 87 cudaBuildInputs = with cudaPackages; [88 cuda_cudart89 cuda_cccl # <nv/target>90 libcublas91 ];92 93 rocmBuildInputs = with rocmPackages; [94 clr95 hipblas96 rocblas97 ];98 99 vulkanBuildInputs = [100 vulkan-headers101 vulkan-loader102 shaderc103 ];104in105 106effectiveStdenv.mkDerivation (finalAttrs: {107 pname = "llama-cpp${pnameSuffix}";108 version = llamaVersion;109 110 # Note: none of the files discarded here are visible in the sandbox or111 # affect the output hash. This also means they can be modified without112 # triggering a rebuild.113 src = lib.cleanSourceWith {114 filter =115 name: type:116 let117 noneOf = builtins.all (x: !x);118 baseName = baseNameOf name;119 in120 noneOf [121 (lib.hasSuffix ".nix" name) # Ignore *.nix files when computing outPaths122 (lib.hasSuffix ".md" name) # Ignore *.md changes whe computing outPaths123 (lib.hasPrefix "." baseName) # Skip hidden files and directories124 (baseName == "flake.lock")125 ];126 src = lib.cleanSource ../../.;127 };128 129 postPatch = ''130 substituteInPlace ./ggml/src/ggml-metal/ggml-metal.m \131 --replace '[bundle pathForResource:@"ggml-metal" ofType:@"metal"];' "@\"$out/bin/ggml-metal.metal\";"132 substituteInPlace ./ggml/src/ggml-metal/ggml-metal.m \133 --replace '[bundle pathForResource:@"default" ofType:@"metallib"];' "@\"$out/bin/default.metallib\";"134 '';135 136 # With PR#6015 https://github.com/ggerganov/llama.cpp/pull/6015,137 # `default.metallib` may be compiled with Metal compiler from XCode138 # and we need to escape sandbox on MacOS to access Metal compiler.139 # `xcrun` is used find the path of the Metal compiler, which is varible140 # and not on $PATH141 # see https://github.com/ggerganov/llama.cpp/pull/6118 for discussion142 __noChroot = effectiveStdenv.isDarwin && useMetalKit && precompileMetalShaders;143 144 nativeBuildInputs =145 [146 cmake147 ninja148 pkg-config149 git150 ]151 ++ optionals useCuda [152 cudaPackages.cuda_nvcc153 154 autoAddDriverRunpath155 ]156 ++ optionals (effectiveStdenv.hostPlatform.isGnu && enableStatic) [ glibc.static ]157 ++ optionals (effectiveStdenv.isDarwin && useMetalKit && precompileMetalShaders) [ xcrunHost ];158 159 buildInputs =160 optionals effectiveStdenv.isDarwin darwinBuildInputs161 ++ optionals useCuda cudaBuildInputs162 ++ optionals useMpi [ mpi ]163 ++ optionals useRocm rocmBuildInputs164 ++ optionals useBlas [ blas ]165 ++ optionals useVulkan vulkanBuildInputs166 ++ optionals enableCurl [ curl ];167 168 cmakeFlags =169 [170 (cmakeBool "LLAMA_BUILD_SERVER" true)171 (cmakeBool "BUILD_SHARED_LIBS" (!enableStatic))172 (cmakeBool "CMAKE_SKIP_BUILD_RPATH" true)173 (cmakeBool "LLAMA_CURL" enableCurl)174 (cmakeBool "GGML_NATIVE" false)175 (cmakeBool "GGML_BLAS" useBlas)176 (cmakeBool "GGML_CUDA" useCuda)177 (cmakeBool "GGML_HIP" useRocm)178 (cmakeBool "GGML_METAL" useMetalKit)179 (cmakeBool "GGML_VULKAN" useVulkan)180 (cmakeBool "GGML_STATIC" enableStatic)181 ]182 ++ optionals useCuda [183 (184 with cudaPackages.flags;185 cmakeFeature "CMAKE_CUDA_ARCHITECTURES" (186 builtins.concatStringsSep ";" (map dropDot cudaCapabilities)187 )188 )189 ]190 ++ optionals useRocm [191 (cmakeFeature "CMAKE_HIP_COMPILER" "${rocmPackages.llvm.clang}/bin/clang")192 (cmakeFeature "CMAKE_HIP_ARCHITECTURES" rocmGpuTargets)193 ]194 ++ optionals useMetalKit [195 (lib.cmakeFeature "CMAKE_C_FLAGS" "-D__ARM_FEATURE_DOTPROD=1")196 (cmakeBool "GGML_METAL_EMBED_LIBRARY" (!precompileMetalShaders))197 ];198 199 # Environment variables needed for ROCm200 env = optionals useRocm {201 ROCM_PATH = "${rocmPackages.clr}";202 HIP_DEVICE_LIB_PATH = "${rocmPackages.rocm-device-libs}/amdgcn/bitcode";203 };204 205 # TODO(SomeoneSerge): It's better to add proper install targets at the CMake level,206 # if they haven't been added yet.207 postInstall = ''208 mkdir -p $out/include209 cp $src/include/llama.h $out/include/210 '';211 212 meta = {213 # Configurations we don't want even the CI to evaluate. Results in the214 # "unsupported platform" messages. This is mostly a no-op, because215 # cudaPackages would've refused to evaluate anyway.216 badPlatforms = optionals useCuda lib.platforms.darwin;217 218 # Configurations that are known to result in build failures. Can be219 # overridden by importing Nixpkgs with `allowBroken = true`.220 broken = (useMetalKit && !effectiveStdenv.isDarwin);221 222 description = "Inference of LLaMA model in pure C/C++${descriptionSuffix}";223 homepage = "https://github.com/ggerganov/llama.cpp/";224 license = lib.licenses.mit;225 226 # Accommodates `nix run` and `lib.getExe`227 mainProgram = "llama-cli";228 229 # These people might respond, on the best effort basis, if you ping them230 # in case of Nix-specific regressions or for reviewing Nix-specific PRs.231 # Consider adding yourself to this list if you want to ensure this flake232 # stays maintained and you're willing to invest your time. Do not add233 # other people without their consent. Consider removing people after234 # they've been unreachable for long periods of time.235 236 # Note that lib.maintainers is defined in Nixpkgs, but you may just add237 # an attrset following the same format as in238 # https://github.com/NixOS/nixpkgs/blob/f36a80e54da29775c78d7eff0e628c2b4e34d1d7/maintainers/maintainer-list.nix239 maintainers = with lib.maintainers; [240 philiptaron241 SomeoneSerge242 ];243 244 # Extend `badPlatforms` instead245 platforms = lib.platforms.all;246 };247})248 