MeshChatX/.github/workflows/bench.yml

86 lines
3.1 KiB
YAML

# Benchmarks and integrity checks (push to main branches and workflow_dispatch).
# Results are stored in a runner cache and compared on every run; the job fails
# when any metric regresses beyond 150% of the stored baseline, and a commit
# comment is posted with the offending numbers.
#
# Pinned first-party actions (bump tag and SHA together when upgrading):
# actions/checkout@v6.0.1 8e8c483db84b4bee98b60c0593521ed34d9990e8
# actions/cache@v4.2.0 1bd1e32a3bdc45362d1e726936510720a7c30a57
# benchmark-action/github-action-benchmark@v1.22.0
# a60cea5bc7b49e15c1f58f411161f99e0df48372
name: Benchmarks
on:
workflow_dispatch:
push:
branches:
- master
- dev
concurrency:
group: bench-${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
permissions:
contents: write
pull-requests: write
env:
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true
PYTHON_VERSION: "3.14"
NODE_VERSION: "24"
UV_VERSION: "0.11.15"
PNPM_VERSION: "11.1.2"
jobs:
bench:
runs-on: ubuntu-latest
timeout-minutes: 60
steps:
- name: Checkout
uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8
- name: Set up development environment
uses: ./.github/actions/setup-dev-environment
with:
python-version: ${{ env.PYTHON_VERSION }}
uv-version: ${{ env.UV_VERSION }}
node-version: ${{ env.NODE_VERSION }}
pnpm-version: ${{ env.PNPM_VERSION }}
- name: Setup Task
run: sh scripts/ci/setup-task.sh
- name: Restore benchmark baseline cache
uses: actions/cache@1bd1e32a3bdc45362d1e726936510720a7c30a57
with:
path: ./cache
key: ${{ runner.os }}-bench-baseline-${{ github.ref_name }}
restore-keys: |
${{ runner.os }}-bench-baseline-
- name: Run benchmarks
run: |
set -euo pipefail
uv run python tests/backend/run_comprehensive_benchmarks.py \
--json-output bench_results.json 2>&1 | tee bench_results.txt
- name: Run integrity tests
run: |
set -euo pipefail
task test:integrity 2>&1 | tee -a bench_results.txt
- name: Store and compare benchmark results
uses: benchmark-action/github-action-benchmark@a60cea5bc7b49e15c1f58f411161f99e0df48372
with:
name: MeshChatX Backend Benchmarks
tool: customSmallerIsBetter
output-file-path: bench_results.json
external-data-json-path: ./cache/benchmark-data.json
github-token: ${{ secrets.GITHUB_TOKEN }}
alert-threshold: "200%"
fail-threshold: "300%"
fail-on-alert: true
comment-on-alert: true
summary-always: true