release : fix ubuntu-cuda tag step (rename in release job) #25
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: Release | |
| on: | |
| workflow_dispatch: | |
| inputs: | |
| create_release: | |
| description: 'Create new release' | |
| required: true | |
| type: boolean | |
| push: | |
| branches: | |
| - master | |
| paths: [ | |
| '.github/workflows/release.yml', | |
| '**/CMakeLists.txt', | |
| '**/.cmake', | |
| '**/*.h', | |
| '**/*.hpp', | |
| '**/*.c', | |
| '**/*.cpp', | |
| '**/*.cu', | |
| '**/*.cuh' | |
| ] | |
| env: | |
| GH_TOKEN: ${{ github.token }} | |
| BRANCH_NAME: ${{ github.head_ref || github.ref_name }} | |
| CMAKE_ARGS: "-DLLAMA_BUILD_EXAMPLES=OFF -DLLAMA_BUILD_TESTS=OFF -DLLAMA_BUILD_TOOLS=ON -DLLAMA_BUILD_SERVER=ON -DGGML_RPC=ON" | |
| # note: run this workflow one at a time for better cache reuse | |
| concurrency: | |
| group: release | |
| queue: max | |
| jobs: | |
| check-release: | |
| runs-on: ubuntu-slim | |
| outputs: | |
| should_release: ${{ steps.check.outputs.should_release }} | |
| steps: | |
| - id: check | |
| env: | |
| COMMIT_MESSAGE: ${{ github.event.head_commit.message }} | |
| run: | | |
| if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then | |
| echo "should_release=true" >> $GITHUB_OUTPUT | |
| elif [[ "${{ github.event_name }}" == "push" && "${{ github.ref }}" == "refs/heads/master" ]]; then | |
| if echo "$COMMIT_MESSAGE" | grep -q '\[no release\]'; then | |
| echo "should_release=false" >> $GITHUB_OUTPUT | |
| else | |
| echo "should_release=true" >> $GITHUB_OUTPUT | |
| fi | |
| else | |
| echo "should_release=false" >> $GITHUB_OUTPUT | |
| fi | |
| get-version: | |
| runs-on: ubuntu-slim | |
| outputs: | |
| ui_version: ${{ steps.version.outputs.ui_version }} | |
| steps: | |
| - uses: actions/checkout@v6 | |
| with: | |
| fetch-depth: 0 | |
| - id: version | |
| run: | | |
| # Resolve UI version: BUILD_NUMBER from cmake/build-info.cmake > git hash + epoch > fallback | |
| version="" | |
| if grep -q "BUILD_NUMBER" cmake/build-info.cmake; then | |
| build_number=$(grep "set(BUILD_NUMBER" cmake/build-info.cmake | grep -oP '\d+') | |
| if [ -n "$build_number" ] && [ "$build_number" -gt 0 ]; then | |
| version="b${build_number}" | |
| fi | |
| fi | |
| if [ -z "$version" ]; then | |
| version=$(git rev-parse --short HEAD)-$(date +%s) | |
| fi | |
| echo "ui_version=${version}" >> $GITHUB_OUTPUT | |
| windows-cpu: | |
| needs: [check-release] | |
| if: ${{ needs.check-release.outputs.should_release == 'true' }} | |
| runs-on: windows-2025-vs2026 | |
| permissions: | |
| actions: write | |
| strategy: | |
| matrix: | |
| include: | |
| - arch: 'x64' | |
| #- arch: 'arm64' | |
| steps: | |
| - name: Clone | |
| uses: actions/checkout@v6 | |
| with: | |
| fetch-depth: 0 | |
| - name: Setup Node.js | |
| uses: actions/setup-node@v6 | |
| with: | |
| node-version: "24" | |
| cache: "npm" | |
| cache-dependency-path: "tools/ui/package-lock.json" | |
| - name: Install Ninja | |
| run: | | |
| choco install ninja | |
| - name: ccache | |
| uses: ggml-org/ccache-action@v1.2.21 | |
| with: | |
| key: release-windows-2025-vs2026-${{ matrix.arch }}-cpu | |
| - name: Build | |
| shell: cmd | |
| run: | | |
| rem Original (arch is always x64 now): | |
| rem call "C:\Program Files\Microsoft Visual Studio\18\Enterprise\VC\Auxiliary\Build\vcvarsall.bat" ${{ matrix.arch == 'x64' && 'x64' || 'amd64_arm64' }} | |
| call "C:\Program Files\Microsoft Visual Studio\18\Enterprise\VC\Auxiliary\Build\vcvarsall.bat" x64 | |
| cmake -S . -B build -G "Ninja Multi-Config" ^ | |
| rem -D CMAKE_TOOLCHAIN_FILE=cmake/${{ matrix.arch }}-windows-llvm.cmake ^ | |
| -D CMAKE_TOOLCHAIN_FILE=cmake/x64-windows-llvm.cmake ^ | |
| -DLLAMA_BUILD_BORINGSSL=ON ^ | |
| -DGGML_NATIVE=OFF ^ | |
| -DGGML_BACKEND_DL=ON ^ | |
| rem -DGGML_CPU_ALL_VARIANTS=${{ matrix.arch == 'x64' && 'ON' || 'OFF' }} ^ | |
| -DGGML_CPU_ALL_VARIANTS=ON ^ | |
| -DGGML_OPENMP=ON ^ | |
| ${{ env.CMAKE_ARGS }} | |
| cmake --build build --config Release | |
| - name: ccache-clear | |
| uses: ./.github/actions/ccache-clear | |
| with: | |
| key: release-windows-2025-vs2026-${{ matrix.arch }}-cpu | |
| - name: Pack artifacts | |
| id: pack_artifacts | |
| run: | | |
| Copy-Item "C:\Program Files\Microsoft Visual Studio\18\Enterprise\VC\Redist\MSVC\14.51.36231\debug_nonredist\${{ matrix.arch }}\Microsoft.VC145.OpenMP.LLVM\libomp140.${{ matrix.arch == 'x64' && 'x86_64' || 'aarch64' }}.dll" .\build\bin\Release\ | |
| 7z a -snl llama-bin-win-cpu-${{ matrix.arch }}.zip .\build\bin\Release\* | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v6 | |
| with: | |
| path: llama-bin-win-cpu-${{ matrix.arch }}.zip | |
| name: llama-bin-win-cpu-${{ matrix.arch }}.zip | |
| windows-cuda: | |
| needs: [check-release] | |
| if: ${{ needs.check-release.outputs.should_release == 'true' }} | |
| runs-on: windows-2022 | |
| permissions: | |
| actions: write | |
| strategy: | |
| matrix: | |
| cuda: ['12.8'] # '13.3' | |
| steps: | |
| - name: Clone | |
| id: checkout | |
| uses: actions/checkout@v6 | |
| - name: Setup Node.js | |
| uses: actions/setup-node@v6 | |
| with: | |
| node-version: "24" | |
| cache: "npm" | |
| cache-dependency-path: "tools/ui/package-lock.json" | |
| - name: Install Cuda Toolkit | |
| uses: ./.github/actions/windows-setup-cuda | |
| with: | |
| cuda_version: ${{ matrix.cuda }} | |
| - name: Install Ninja | |
| id: install_ninja | |
| run: | | |
| choco install ninja | |
| - name: ccache | |
| uses: ggml-org/ccache-action@v1.2.21 | |
| with: | |
| key: release-windows-2022-x64-cuda-${{ matrix.cuda }} | |
| - name: Build | |
| id: cmake_build | |
| shell: cmd | |
| # TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project | |
| run: | | |
| call "C:\Program Files\Microsoft Visual Studio\2022\Enterprise\VC\Auxiliary\Build\vcvarsall.bat" x64 | |
| cmake -S . -B build -G "Ninja Multi-Config" ^ | |
| -DGGML_BACKEND_DL=ON ^ | |
| -DCMAKE_CUDA_ARCHITECTURES=89 ^ | |
| -DGGML_CUDA_FA_ALL_QUANTS=ON ^ | |
| -DGGML_CUDA_F16=ON -DGGML_CUDA_IQK_FORCE_BF16=1 ^ | |
| -DGGML_SCHED_MAX_COPIES=1 ^ | |
| -DGGML_NATIVE=OFF ^ | |
| -DGGML_CPU=OFF ^ | |
| -DGGML_CUDA=ON ^ | |
| -DLLAMA_BUILD_BORINGSSL=ON ^ | |
| -DGGML_CUDA_CUB_3DOT2=ON | |
| set /A NINJA_JOBS=%NUMBER_OF_PROCESSORS%-1 | |
| cmake --build build --config Release -j %NINJA_JOBS% --target ggml-cuda | |
| - name: ccache-clear | |
| uses: ./.github/actions/ccache-clear | |
| with: | |
| key: release-windows-2022-x64-cuda-${{ matrix.cuda }} | |
| - name: Pack artifacts | |
| id: pack_artifacts | |
| run: | | |
| 7z a -snl llama-bin-win-cuda-${{ matrix.cuda }}-x64.zip .\build\bin\Release\ggml-cuda.dll | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v6 | |
| with: | |
| path: llama-bin-win-cuda-${{ matrix.cuda }}-x64.zip | |
| name: llama-bin-win-cuda-${{ matrix.cuda }}-x64.zip | |
| - name: Copy and pack Cuda runtime | |
| run: | | |
| echo "Cuda install location: ${{ env.CUDA_PATH }}" | |
| $dst='.\build\bin\cudart\' | |
| robocopy "${{env.CUDA_PATH}}\bin" $dst cudart64_*.dll cublas64_*.dll cublasLt64_*.dll | |
| robocopy "${{env.CUDA_PATH}}\lib" $dst cudart64_*.dll cublas64_*.dll cublasLt64_*.dll | |
| robocopy "${{env.CUDA_PATH}}\bin\x64" $dst cudart64_*.dll cublas64_*.dll cublasLt64_*.dll | |
| 7z a cudart-llama-bin-win-cuda-${{ matrix.cuda }}-x64.zip $dst\* | |
| - name: Upload Cuda runtime | |
| uses: actions/upload-artifact@v6 | |
| with: | |
| path: cudart-llama-bin-win-cuda-${{ matrix.cuda }}-x64.zip | |
| name: cudart-llama-bin-win-cuda-${{ matrix.cuda }}-x64.zip | |
| ubuntu-cuda: | |
| needs: [check-release] | |
| if: ${{ needs.check-release.outputs.should_release == 'true' }} | |
| runs-on: ubuntu-24.04 | |
| container: nvidia/cuda:12.6.2-devel-ubuntu24.04 | |
| permissions: | |
| actions: write | |
| steps: | |
| - name: Clone | |
| id: checkout | |
| uses: actions/checkout@v6 | |
| with: | |
| fetch-depth: 0 | |
| - name: Install dependencies | |
| env: | |
| DEBIAN_FRONTEND: noninteractive | |
| run: | | |
| apt update | |
| apt install -y cmake build-essential ninja-build libgomp1 git libssl-dev | |
| - name: ccache | |
| uses: ggml-org/ccache-action@v1.2.21 | |
| with: | |
| key: release-cuda-ubuntu-24.04 | |
| evict-old-files: 1d | |
| - name: Build with CMake | |
| # TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project | |
| run: | | |
| cmake -S . -B build -G Ninja \ | |
| -DLLAMA_FATAL_WARNINGS=ON \ | |
| -DCMAKE_BUILD_TYPE=Release \ | |
| -DCMAKE_CUDA_ARCHITECTURES=89-real \ | |
| -DGGML_CUDA_FA_ALL_QUANTS=ON \ | |
| -DGGML_CUDA_F16=ON -DGGML_CUDA_IQK_FORCE_BF16=1 \ | |
| -DGGML_SCHED_MAX_COPIES=1 \ | |
| -DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined \ | |
| -DGGML_NATIVE=OFF \ | |
| -DGGML_CUDA=ON \ | |
| -DGGML_CUDA_CUB_3DOT2=ON | |
| cmake --build build | |
| - name: Pack artifacts | |
| id: pack_artifacts | |
| run: | | |
| cp LICENSE ./build/bin/ | |
| tar -czvf llama-bin-ubuntu-cuda-x64.tar.gz --transform "s,^\.,llama-bin-ubuntu-cuda-x64," -C ./build/bin . | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v6 | |
| with: | |
| path: llama-bin-ubuntu-cuda-x64.tar.gz | |
| name: llama-bin-ubuntu-cuda-x64.tar.gz | |
| release: | |
| if: ${{ ( github.event_name == 'push' && github.ref == 'refs/heads/master' ) || github.event.inputs.create_release == 'true' }} | |
| # Fine-grant permission | |
| # https://docs.github.com/en/actions/security-for-github-actions/security-guides/automatic-token-authentication#modifying-the-permissions-for-the-github_token | |
| permissions: | |
| contents: write # for creating release | |
| runs-on: ubuntu-slim | |
| needs: | |
| - get-version | |
| - windows-cpu | |
| - windows-cuda | |
| - ubuntu-cuda | |
| #- windows-hip | |
| #- windows | |
| #- windows-openvino | |
| #- ubuntu-22-rocm | |
| # Removed jobs: | |
| # - ubuntu-cpu | |
| #- ubuntu-vulkan | |
| #- ubuntu-24-openvino | |
| #- android-arm64 | |
| #- macos-cpu | |
| #- ios-xcode | |
| #- ui-build | |
| outputs: | |
| tag_name: ${{ steps.tag.outputs.name }} | |
| steps: | |
| - name: Clone | |
| id: checkout | |
| uses: actions/checkout@v6 | |
| with: | |
| fetch-depth: 0 | |
| - name: Determine tag name | |
| id: tag | |
| uses: ./.github/actions/get-tag-name | |
| - name: Delete existing release if present | |
| run: | | |
| tag="${{ steps.tag.outputs.name }}" | |
| if gh release view "$tag" >/dev/null 2>&1; then | |
| echo "Release $tag already exists, deleting before recreate..." | |
| gh release delete "$tag" --yes | |
| else | |
| echo "No existing release for $tag" | |
| fi | |
| - name: Download artifacts | |
| id: download-artifact | |
| uses: actions/download-artifact@v7 | |
| with: | |
| path: ./artifact | |
| merge-multiple: true | |
| - name: Move artifacts | |
| id: move_artifacts | |
| run: | | |
| mkdir -p release | |
| shopt -s nullglob | |
| echo "Adding CPU backend files to existing zips..." | |
| for arch in x64; do | |
| cpu_zip="artifact/llama-bin-win-cpu-${arch}.zip" | |
| temp_dir=$(mktemp -d) | |
| echo "Extracting CPU backend for $arch..." | |
| unzip "$cpu_zip" -d "$temp_dir" | |
| echo "Adding CPU files to $arch zips..." | |
| for target_zip in artifact/llama-bin-win-*-${arch}.zip; do | |
| if [[ "$target_zip" == "$cpu_zip" ]]; then | |
| continue | |
| fi | |
| echo "Adding CPU backend to $(basename "$target_zip")" | |
| realpath_target_zip=$(realpath "$target_zip") | |
| (cd "$temp_dir" && zip -r "$realpath_target_zip" .) | |
| done | |
| rm -rf "$temp_dir" | |
| done | |
| echo "Renaming and moving zips to release..." | |
| for zip_file in artifact/llama-bin-win-*.zip artifact/llama-bin-ubuntu-cuda-x64.tar.gz; do | |
| base_name=$(basename "$zip_file") | |
| if [[ "$base_name" == *.tar.gz ]]; then | |
| bare="${base_name%.tar.gz}" | |
| ext=".tar.gz" | |
| else | |
| bare="${base_name%.zip}" | |
| ext=".zip" | |
| fi | |
| zip_name="llama-${{ steps.tag.outputs.name }}-${bare#llama-}${ext}" | |
| echo "Moving $zip_file to release/$zip_name" | |
| mv "$zip_file" "release/$zip_name" | |
| done | |
| echo "Moving other artifacts..." | |
| shopt -s nullglob | |
| for f in artifact/*.zip artifact/*.tar.gz; do | |
| mv -v "$f" release/ | |
| done | |
| - name: Create release | |
| id: create_release | |
| uses: ggml-org/action-create-release@v1 | |
| env: | |
| GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} | |
| with: | |
| tag_name: ${{ steps.tag.outputs.name }} | |
| body: | | |
| <details open> | |
| ${{ github.event.head_commit.message }} | |
| </details> | |
| **Linux:** | |
| - [Ubuntu x64 (CUDA)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-x64.tar.gz) | |
| # - [Ubuntu x64 (CPU)](.../llama-${{ steps.tag.outputs.name }}-bin-ubuntu-x64.tar.gz) | |
| # - [Ubuntu arm64 (CPU)](.../llama-${{ steps.tag.outputs.name }}-bin-ubuntu-arm64.tar.gz) | |
| # - [Ubuntu arm64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-arm64.tar.gz) | |
| # - [Ubuntu x64 (ROCm 7.2)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-rocm-7.2-x64.tar.gz) | |
| **Windows:** | |
| - [Windows x64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cpu-x64.zip) | |
| # - [Windows arm64 (CPU)](.../llama-${{ steps.tag.outputs.name }}-bin-win-cpu-arm64.zip) | |
| - [Windows x64 (CUDA 12.8)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-12.8-x64.zip) - [CUDA 12.8 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-12.8-x64.zip) | |
| # - [Windows x64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-13.3-x64.zip) - [CUDA 13.3 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-13.3-x64.zip) | |
| # - [Windows x64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-vulkan-x64.zip) | |
| # - [Windows x64 (HIP)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-hip-radeon-x64.zip) | |
| - name: Upload release | |
| id: upload_release | |
| uses: actions/github-script@v8 | |
| with: | |
| github-token: ${{secrets.GITHUB_TOKEN}} | |
| script: | | |
| const path = require('path'); | |
| const fs = require('fs'); | |
| const release_id = '${{ steps.create_release.outputs.id }}'; | |
| for (let file of await fs.readdirSync('./release')) { | |
| if (path.extname(file) === '.zip' || file.endsWith('.tar.gz')) { | |
| console.log('uploadReleaseAsset', file); | |
| await github.rest.repos.uploadReleaseAsset({ | |
| owner: context.repo.owner, | |
| repo: context.repo.repo, | |
| release_id: release_id, | |
| name: file, | |
| data: await fs.readFileSync(`./release/${file}`) | |
| }); | |
| } | |
| } |