diff --git a/.github/workflows/release-prism.yml b/.github/workflows/release-prism.yml index 04b7638e95e0..5e4e1e921d5e 100644 --- a/.github/workflows/release-prism.yml +++ b/.github/workflows/release-prism.yml @@ -412,12 +412,14 @@ jobs: - name: Build run: | - cmake -S . -B build -DGGML_VULKAN=ON -DGGML_NATIVE=OFF -DGGML_CPU=OFF -DGGML_BACKEND_DL=ON -DLLAMA_BUILD_BORINGSSL=ON - cmake --build build --config Release --target ggml-vulkan + # Full self-contained bundle (CPU backends + Vulkan), not just the ggml-vulkan.dll + # add-on, so users can download this one zip and run without merging with the CPU zip. + cmake -S . -B build -DGGML_VULKAN=ON -DGGML_NATIVE=OFF -DGGML_BACKEND_DL=ON -DGGML_CPU_ALL_VARIANTS=ON -DGGML_OPENMP=OFF -DLLAMA_BUILD_BORINGSSL=ON + cmake --build build --config Release - name: Pack artifacts run: | - 7z a -snl llama-bin-win-vulkan-x64.zip .\build\bin\Release\ggml-vulkan.dll + 7z a -snl llama-bin-win-vulkan-x64.zip .\build\bin\Release\* - name: Upload artifacts uses: actions/upload-artifact@v6 @@ -581,12 +583,13 @@ jobs: -DCMAKE_BUILD_TYPE=Release ` -DGGML_BACKEND_DL=ON ` -DGGML_NATIVE=OFF ` - -DGGML_CPU=OFF ` + -DGGML_OPENMP=OFF ` -DGPU_TARGETS="${{ matrix.gpu_targets }}" ` -DGGML_HIP_ROCWMMA_FATTN=ON ` -DGGML_HIP=ON ` -DLLAMA_BUILD_BORINGSSL=ON - cmake --build build --target ggml-hip -j ${env:NUMBER_OF_PROCESSORS} + # Full self-contained bundle (CPU backend + HIP), not just the ggml-hip.dll add-on. + cmake --build build -j ${env:NUMBER_OF_PROCESSORS} md "build\bin\rocblas\library\" md "build\bin\hipblaslt\library" cp "${env:HIP_PATH}\bin\libhipblas.dll" "build\bin\" diff --git a/README.md b/README.md index d9f2e18231c7..bcd822e3667d 100644 --- a/README.md +++ b/README.md @@ -1,5 +1,22 @@ # llama.cpp +> [!IMPORTANT] +> **This is the PrismML fork of llama.cpp.** It adds the `Q2_0` 2-bit quantization used by the [Bonsai](https://huggingface.co/collections/prism-ml/bonsai) models. +> +> **New here? Start with the [Bonsai-demo](https://github.com/PrismML-Eng/Bonsai-demo) repo.** It downloads the right models and the correct prebuilt binaries for your hardware/backend automatically. +> +> Ternary (`Q2_0`) support is migrating into mainline llama.cpp backend-by-backend, so which build + model file to use depends on where you run: +> +> - `*-Q2_0.gguf` (group size 128): the format **this fork** uses. Run it with this fork's builds / [releases](https://github.com/PrismML-Eng/llama.cpp/releases). Does not load on mainline llama.cpp. +> - `*-Q2_0_g64.gguf` (group size 64): the **official mainline** llama.cpp format (currently CPU and Metal). Use a recent `ggml-org/llama.cpp` build for these, not this fork. +> - `*-PQ2_0.gguf`: planned future fork format, **not supported anywhere yet**. +> +> Use a complete matching build. Do NOT drop this fork's `ggml-*` libraries into a stock llama.cpp build (ABI/format mismatch, models fail to load). +> +> **For the latest backend-by-backend migration status, see [Upstream Status for Ternary](https://github.com/PrismML-Eng/Bonsai-demo#upstream-status-for-ternary) in the Bonsai-demo README.** + +--- + ![llama](https://user-images.githubusercontent.com/1991296/230134379-7181e485-c521-4d23-a0d6-f7b3b61ba524.png) [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://opensource.org/licenses/MIT)