CoolFace
Datasetpublic

echodict/llama.cpp

version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786

sourceHugging Faceupdated 5mo agoView on Hugging Face
0likes773downloads
bench.yml.disabled305 linesDownload Raw Back to workflows
1# TODO: there have been some issues with the workflow, so disabling for now2#       https://github.com/ggml-org/llama.cpp/issues/78933#4# Benchmark5name: Benchmark6 7on:8  workflow_dispatch:9    inputs:10      gpu-series:11        description: 'Azure GPU series to run with'12        required: true13        type: choice14        options:15          - Standard_NC4as_T4_v316          - Standard_NC24ads_A100_v417          - Standard_NC80adis_H100_v518      sha:19        description: 'Commit SHA1 to build'20        required: false21        type: string22      duration:23        description: 'Duration of the bench'24        type: string25        default: 10m26 27  push:28    branches:29      - master30    paths: ['llama.cpp', 'ggml.c', 'ggml-backend.cpp', 'ggml-quants.c', '**/*.cu', 'tools/server/*.h*', 'tools/server/*.cpp']31  pull_request_target:32    types: [opened, synchronize, reopened]33    paths: ['llama.cpp', 'ggml.c', 'ggml-backend.cpp', 'ggml-quants.c', '**/*.cu', 'tools/server/*.h*', 'tools/server/*.cpp']34  schedule:35    -  cron: '04 2 * * *'36 37concurrency:38  group: ${{ github.workflow }}-${{ github.ref }}-${{ github.head_ref || github.run_id }}-${{ github.event.inputs.sha }}39  cancel-in-progress: true40 41jobs:42  bench-server-baseline:43    runs-on: Standard_NC4as_T4_v344    env:45      RUNNER_LABEL: Standard_NC4as_T4_v3 # FIXME Do not find a way to not duplicate it46      N_USERS: 847      DURATION: 10m48 49    strategy:50      matrix:51        model: [phi-2]52        ftype: [q4_0, q8_0, f16]53        include:54          - model: phi-255            ftype: q4_056            pr_comment_enabled: "true"57 58    if: |59      inputs.gpu-series == 'Standard_NC4as_T4_v3'60      || github.event_name == 'pull_request_target'61    steps:62      - name: Clone63        id: checkout64        uses: actions/checkout@v465        with:66          fetch-depth: 067          ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}68 69      - name: Install python env70        id: pipenv71        run: |72          cd tools/server/bench73          python3 -m venv venv74          source venv/bin/activate75          pip install -r requirements.txt76 77      - name: Prometheus78        id: install_prometheus79        run: |80          wget --quiet https://github.com/prometheus/prometheus/releases/download/v2.51.0/prometheus-2.51.0.linux-amd64.tar.gz81          tar xzf prometheus*.tar.gz --strip-components=182          ./prometheus --config.file=tools/server/bench/prometheus.yml &83          while ! nc -z localhost 9090; do84            sleep 0.185          done86 87      - name: Set up Go88        uses: actions/setup-go@v589        with:90          go-version: '1.21'91 92      - name: Install k6 and xk6-sse93        id: k6_installation94        run: |95          cd tools/server/bench96          go install go.k6.io/xk6/cmd/xk6@latest97          xk6 build master \98              --with github.com/phymbert/xk6-sse99 100      - name: Build101        id: cmake_build102        run: |103          set -eux104          cmake -B build \105              -DGGML_NATIVE=OFF \106              -DLLAMA_BUILD_SERVER=ON \107              -DLLAMA_CUBLAS=ON \108              -DCUDAToolkit_ROOT=/usr/local/cuda \109              -DCMAKE_CUDA_COMPILER=/usr/local/cuda/bin/nvcc \110              -DCMAKE_CUDA_ARCHITECTURES=75 \111              -DLLAMA_FATAL_WARNINGS=OFF \112              -DLLAMA_ALL_WARNINGS=OFF \113              -DCMAKE_BUILD_TYPE=Release;114          cmake --build build --config Release -j $(nproc) --target llama-server115 116      - name: Download the dataset117        id: download_dataset118        run: |119          cd tools/server/bench120          wget --quiet https://huggingface.co/datasets/anon8231489123/ShareGPT_Vicuna_unfiltered/resolve/main/ShareGPT_V3_unfiltered_cleaned_split.json121 122      - name: Server bench123        id: server_bench124        env:125            HEAD_REF: ${{ github.head_ref || github.ref_name }}126        run: |127          set -eux128 129          cd tools/server/bench130          source venv/bin/activate131          python bench.py \132              --runner-label ${{ env.RUNNER_LABEL }} \133              --name ${{ github.job }} \134              --branch $HEAD_REF \135              --commit ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha }} \136              --scenario script.js \137              --duration ${{ github.event.inputs.duration || env.DURATION }} \138              --hf-repo ggml-org/models	 \139              --hf-file ${{ matrix.model }}/ggml-model-${{ matrix.ftype }}.gguf \140              --model-path-prefix /models \141              --parallel ${{ env.N_USERS }} \142              -ngl 33 \143              --batch-size 2048 \144              --ubatch-size	256 \145              --ctx-size 16384 \146              --n-prompts 1000 \147              --max-prompt-tokens 1024 \148              --max-tokens 2048149 150          cat results.github.env >> $GITHUB_ENV151 152          # Remove dataset as we do not want it in the artefact153          rm ShareGPT_V3_unfiltered_cleaned_split.json154 155      - uses: actions/upload-artifact@v4156        with:157          name: bench-server-${{ github.job }}-${{ env.RUNNER_LABEL }}-${{ matrix.model }}-${{ matrix.ftype }}158          compression-level: 9159          path: |160            tools/server/bench/*.jpg161            tools/server/bench/*.json162            tools/server/bench/*.log163 164      - name: Commit status165        uses: Sibz/github-status-action@v1166        with:167          authToken: ${{secrets.GITHUB_TOKEN}}168          sha: ${{ inputs.sha || github.event.pull_request.head.sha || github.sha }}169          context: bench-server-${{ github.job }}-${{ env.RUNNER_LABEL }}-${{ matrix.model }}-${{ matrix.ftype }}170          description: |171            ${{ env.BENCH_RESULTS }}172          state: 'success'173 174      - name: Upload benchmark images175        uses: devicons/public-upload-to-imgur@v2.2.2176        continue-on-error: true # Important as it looks unstable: 503177        id: imgur_step178        with:179          client_id: ${{secrets.IMGUR_CLIENT_ID}}180          path: |181            tools/server/bench/prompt_tokens_seconds.jpg182            tools/server/bench/predicted_tokens_seconds.jpg183            tools/server/bench/kv_cache_usage_ratio.jpg184            tools/server/bench/requests_processing.jpg185 186      - name: Extract mermaid187        id: set_mermaid188        run: |189          set -eux190 191          cd tools/server/bench192          PROMPT_TOKENS_SECONDS=$(cat prompt_tokens_seconds.mermaid)193          echo "PROMPT_TOKENS_SECONDS<<EOF" >> $GITHUB_ENV194          echo "$PROMPT_TOKENS_SECONDS" >> $GITHUB_ENV195          echo "EOF" >> $GITHUB_ENV196 197          PREDICTED_TOKENS_SECONDS=$(cat predicted_tokens_seconds.mermaid)198          echo "PREDICTED_TOKENS_SECONDS<<EOF" >> $GITHUB_ENV199          echo "$PREDICTED_TOKENS_SECONDS" >> $GITHUB_ENV200          echo "EOF" >> $GITHUB_ENV201 202          KV_CACHE_USAGE_RATIO=$(cat kv_cache_usage_ratio.mermaid)203          echo "KV_CACHE_USAGE_RATIO<<EOF" >> $GITHUB_ENV204          echo "$KV_CACHE_USAGE_RATIO" >> $GITHUB_ENV205          echo "EOF" >> $GITHUB_ENV206 207          REQUESTS_PROCESSING=$(cat requests_processing.mermaid)208          echo "REQUESTS_PROCESSING<<EOF" >> $GITHUB_ENV209          echo "$REQUESTS_PROCESSING" >> $GITHUB_ENV210          echo "EOF" >> $GITHUB_ENV211 212      - name: Extract image url213        id: extract_image_url214        continue-on-error: true215        run: |216          set -eux217 218          echo "IMAGE_O=${{ fromJSON(steps.imgur_step.outputs.imgur_urls)[0] }}" >> $GITHUB_ENV219          echo "IMAGE_1=${{ fromJSON(steps.imgur_step.outputs.imgur_urls)[1] }}" >> $GITHUB_ENV220          echo "IMAGE_2=${{ fromJSON(steps.imgur_step.outputs.imgur_urls)[2] }}" >> $GITHUB_ENV221          echo "IMAGE_3=${{ fromJSON(steps.imgur_step.outputs.imgur_urls)[3] }}" >> $GITHUB_ENV222 223      - name: Comment PR224        uses: mshick/add-pr-comment@v2225        id: comment_pr226        if: ${{ github.event.pull_request != '' && matrix.pr_comment_enabled == 'true' }}227        with:228          message-id: bench-server-${{ github.job }}-${{ env.RUNNER_LABEL }}-${{ matrix.model }}-${{ matrix.ftype }}229          message: |230            <p align="center">231 232            ๐Ÿ“ˆ **llama.cpp server** for _${{ github.job }}_ on _${{ env.RUNNER_LABEL }}_ for `${{ matrix.model }}`-`${{ matrix.ftype }}`: **${{ env.BENCH_ITERATIONS}} iterations** ๐Ÿš€233 234            </p>235 236            <details>237 238            <summary>Expand details for performance related PR only</summary>239 240            - Concurrent users: ${{ env.N_USERS }}, duration: ${{ github.event.inputs.duration || env.DURATION }}241            - HTTP request          : avg=${{ env.HTTP_REQ_DURATION_AVG }}ms        p(95)=${{ env.HTTP_REQ_DURATION_P_95_ }}ms fails=${{ env.HTTP_REQ_FAILED_PASSES }}, finish reason: stop=${{ env.LLAMACPP_COMPLETIONS_STOP_RATE_PASSES }} truncated=${{ env.LLAMACPP_COMPLETIONS_TRUNCATED_RATE_PASSES }}242            - Prompt processing (pp): avg=${{ env.LLAMACPP_PROMPT_PROCESSING_SECOND_AVG }}tk/s p(95)=${{ env.LLAMACPP_PROMPT_PROCESSING_SECOND_P_95_ }}tk/s243            - Token generation  (tg): avg=${{ env.LLAMACPP_TOKENS_SECOND_AVG }}tk/s p(95)=${{ env.LLAMACPP_TOKENS_SECOND_P_95_ }}tk/s244            - ${{ env.BENCH_GRAPH_XLABEL }}245 246 247            <p align="center">248 249            <img width="100%" height="100%" src="${{ env.IMAGE_O }}" alt="prompt_tokens_seconds" />250 251            <details>252 253            <summary>More</summary>254 255            ```mermaid256            ${{ env.PROMPT_TOKENS_SECONDS }}257            ```258 259            </details>260 261            <img width="100%" height="100%" src="${{ env.IMAGE_1 }}" alt="predicted_tokens_seconds"/>262 263            <details>264                <summary>More</summary>265 266            ```mermaid267            ${{ env.PREDICTED_TOKENS_SECONDS }}268            ```269 270            </details>271 272            </p>273 274            <details>275 276            <summary>Details</summary>277 278            <p align="center">279 280            <img width="100%" height="100%" src="${{ env.IMAGE_2 }}" alt="kv_cache_usage_ratio" />281 282            <details>283                <summary>More</summary>284 285            ```mermaid286            ${{ env.KV_CACHE_USAGE_RATIO }}287            ```288 289            </details>290 291            <img width="100%" height="100%" src="${{ env.IMAGE_3 }}" alt="requests_processing"/>292 293            <details>294                <summary>More</summary>295 296            ```mermaid297            ${{ env.REQUESTS_PROCESSING }}298            ```299 300            </details>301 302            </p>303            </details>304            </details>305