Skip to content

Nightly Stress Test #130

Nightly Stress Test

Nightly Stress Test #130

name: Nightly Stress Test
on:
schedule:
- cron: '0 2 * * *' # 2 AM UTC
workflow_dispatch:
inputs:
duration:
description: Stress test duration (GitHub-hosted runner caps at 6h)
required: true
default: '0.5h'
type: choice
options:
- '10m'
- '0.5h'
- '1h'
- '1.5h'
- '2h'
- '2.5h'
- '3h'
- '3.5h'
- '4h'
- '4.5h'
- '5h'
- '5.5h'
enable_metrics:
description: Enable OpenTelemetry metrics (metrics.enabled)
required: false
default: false
type: boolean
worker_threads:
description: Salt master worker_threads
required: false
default: '5'
type: string
jobs:
stress-test:
runs-on: ubuntu-latest
# Repo-level opt-out. Set `SKIP_NIGHTLY_STRESS_TEST=true` on repos where
# this shouldn't run (e.g. saltstack/salt — the stress test's runner cost
# belongs on the salt-nightlies fork alongside the other nightlies infra).
if: vars.SKIP_NIGHTLY_STRESS_TEST != 'true'
# ``contents: write`` lets the ``Publish panels to stress-snapshots
# branch`` step push the rendered PNGs to an orphan branch so the
# step summary can embed them via raw.githubusercontent.com URLs.
permissions:
contents: write
steps:
- uses: actions/checkout@v4
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
- name: Cache Docker layers
uses: actions/cache@v4
with:
path: /tmp/.buildx-cache
key: ${{ runner.os }}-buildx-${{ github.sha }}
restore-keys: |
${{ runner.os }}-buildx-
- name: Build and Start Environment
run: |
cd tests/monitoring
# The prometheus container runs as ``nobody`` (uid 65534) and
# writes its TSDB to ``/prometheus``. Without pre-creating the
# bind-mount source with that ownership, Docker auto-creates
# it owned by root and prometheus fails to start with
# "permission denied" on /prometheus -- which then surfaces
# downstream as a bare ConnectionRefusedError when
# analyze_stats.py tries to query http://localhost:19090.
mkdir -p prometheus_data
sudo chown -R 65534:65534 prometheus_data
docker compose build
docker compose up -d
sleep 30 # Wait for initialization
- name: Configure salt-master
# Apply the workflow_dispatch overrides to ``master.conf`` and,
# if anything actually changed, restart salt-master so it picks
# them up. ``metrics.enabled`` gates the entire
# ``salt.utils.metrics`` stack (including the ``_load_otel``
# deferred OpenTelemetry import); the pip package on disk has
# zero runtime cost when this gate is off, so a config toggle
# is all that is needed to measure the OTel-on vs OTel-off
# profile. ``worker_threads`` sizes the MWorker pool and lets
# a run sweep the parallelism / RSS trade-off.
env:
ENABLE_METRICS: ${{ github.event.inputs.enable_metrics || 'false' }}
WORKER_THREADS: ${{ github.event.inputs.worker_threads || '5' }}
run: |
cd tests/monitoring
need_restart=0
if [ "$ENABLE_METRICS" = "true" ]; then
echo "Enabling OpenTelemetry metrics on salt-master"
# Append the metrics block only if not already present so
# re-runs are idempotent. ``printf`` (not a heredoc) keeps
# the YAML block-scalar indentation intact.
if ! grep -q '^metrics:' master.conf; then
printf '\nmetrics:\n enabled: true\n' >> master.conf
need_restart=1
fi
else
echo "Leaving metrics.enabled at default (false); OTel package is"
echo "shipped but never imported by the lazy loader."
fi
# Update ``worker_threads`` only when it differs from the value
# already in the file, so unchanged defaults skip the restart.
current_workers=$(awk '/^worker_threads:/ {print $2}' master.conf)
if [ -n "$current_workers" ] && [ "$current_workers" != "$WORKER_THREADS" ]; then
echo "Setting worker_threads: $current_workers -> $WORKER_THREADS"
sed -i "s/^worker_threads:.*/worker_threads: $WORKER_THREADS/" master.conf
need_restart=1
fi
if [ "$need_restart" = "1" ]; then
docker restart salt-master
sleep 20
fi
if [ "$ENABLE_METRICS" = "true" ]; then
# Trigger a code path that calls ``metrics.configure`` so
# the OpenTelemetry import fires and any misconfiguration
# surfaces here rather than mid-stress.
docker exec salt-master python3 -c \
"import salt.utils.metrics as m; m._load_otel(); assert m._OTEL_AVAILABLE, 'OTel import failed'"
fi
- name: Verify Connections
# The salt CLI returns exit 0 even when the master returns an
# error string (the legacy ``'str' object has no attribute
# 'pop'`` failure path surfaces this way), so we have to inspect
# the JSON output ourselves to fail the step.
run: |
out=$(docker exec salt-master salt --out=json '*' test.ping)
echo "$out"
python3 - "$out" <<'PY'
import json, sys
payload = sys.argv[1].strip()
decoder = json.JSONDecoder()
idx = 0
bad = []
while idx < len(payload):
while idx < len(payload) and payload[idx].isspace():
idx += 1
if idx >= len(payload):
break
try:
obj, end = decoder.raw_decode(payload, idx)
except ValueError as exc:
bad.append(f" non-JSON at offset {idx}: {exc}")
break
idx = end
if isinstance(obj, dict):
for mid, val in obj.items():
if val is not True:
bad.append(f" minion {mid!r} returned {val!r}")
else:
bad.append(f" unexpected payload type: {type(obj).__name__}={obj!r}")
if bad:
print("Verify Connections failed:", file=sys.stderr)
for b in bad:
print(b, file=sys.stderr)
sys.exit(1)
PY
- name: Run Aggressive Stress Test
run: |
cd tests/monitoring
chmod +x stress_test.sh stress_api.sh
# Run in background and wait for defined duration
./stress_test.sh &
STRESS_PID=$!
# Default to 30m if not workflow_dispatch
DURATION="${{ github.event.inputs.duration || '0.5h' }}"
echo "Running stress test for $DURATION..."
# Use sleep with suffix support (m, h)
sleep $DURATION
echo "Stopping stress test..."
pkill -P $STRESS_PID || true
kill $STRESS_PID || true
- name: Analyze Results
run: |
cd tests/monitoring
# Give Prometheus a moment to finish scraping the final points
sleep 30
python3 analyze_stats.py
- name: Render Dashboard Panels
# Render BEFORE Snapshot Metrics stops prometheus. We *only*
# produce the PNGs here; the markdown step-summary that links
# to them runs after the upload so it can embed a real URL.
# Runs even if Analyze Results failed -- the graphs are usually
# the most useful diagnostic for a failed run.
if: always()
run: |
cd tests/monitoring
python3 -m pip install --quiet matplotlib
PANELS_DIR="${GITHUB_WORKSPACE}/artifacts/panels" \
python3 render_panels.py
- name: Snapshot Metrics
if: always()
run: |
# Stop containers to ensure data is flushed to disk
cd tests/monitoring
docker compose stop prometheus
sudo tar -czf ../../prometheus-data.tar.gz ./prometheus_data
- name: Collect Logs on Failure
if: failure()
run: |
mkdir -p artifacts
docker logs salt-master > artifacts/salt-master.log
docker logs salt-minion-1 > artifacts/salt-minion-1.log
cp tests/monitoring/event_log.txt artifacts/ || true
# Always grab prometheus' own log so we can tell whether the
# connection refused was a startup issue (bind mount permissions
# etc.) vs. a runtime crash.
docker logs prometheus > artifacts/prometheus.log 2>&1 || true
- name: Upload Artifacts
id: upload-artifacts
if: always()
uses: actions/upload-artifact@v4
with:
name: stress-test-results
path: |
artifacts/
prometheus-data.tar.gz
- name: Publish panels to stress-snapshots branch
# GitHub's step-summary sanitizer strips ``data:`` URIs but
# allows real ``https:`` image URLs, so we push the rendered
# PNGs to a dedicated ``stress-snapshots`` branch and let the
# summary embed them via ``raw.githubusercontent.com``. The
# branch is orphan-style: it never merges anywhere and is
# auto-pruned 14 days back so it stays small. Skipped silently
# when no PNGs exist (e.g. early-stage failures).
id: publish-snapshots
if: always() && hashFiles('artifacts/panels/*.png') != ''
env:
GH_TOKEN: ${{ github.token }}
run: |
set -e
REPO_URL="https://x-access-token:${GH_TOKEN}@github.com/${{ github.repository }}.git"
BRANCH=stress-snapshots
RUN_DIR="runs/${{ github.run_id }}"
if git ls-remote --exit-code --heads "$REPO_URL" "$BRANCH" >/dev/null 2>&1; then
git clone --depth=1 --branch="$BRANCH" "$REPO_URL" snaps
else
git clone --depth=1 "$REPO_URL" snaps
cd snaps
git switch --orphan "$BRANCH"
git rm -rf . >/dev/null 2>&1 || true
cd ..
fi
mkdir -p "snaps/$RUN_DIR"
cp artifacts/panels/*.png "snaps/$RUN_DIR/"
cd snaps
# Drop any run directory older than 14 days so the branch
# never accumulates unbounded.
find runs -mindepth 1 -maxdepth 1 -type d -mtime +14 -exec rm -rf {} + 2>/dev/null || true
git config user.email "actions@github.com"
git config user.name "github-actions[bot]"
git add -A
if ! git commit -m "Stress run ${{ github.run_id }}"; then
echo "nothing to commit"
else
# Tiny retry loop in case two runs raced. Cron is nightly
# so this is essentially defensive.
for attempt in 1 2 3; do
if git push origin "$BRANCH"; then
break
fi
echo "push attempt $attempt failed; rebasing"
git fetch origin "$BRANCH"
git rebase "origin/$BRANCH"
done
fi
URL_PREFIX="https://raw.githubusercontent.com/${{ github.repository }}/${BRANCH}/${RUN_DIR}/"
echo "url-prefix=${URL_PREFIX}" >> "$GITHUB_OUTPUT"
echo "Published panels under ${URL_PREFIX}"
- name: Panel Summary
# Build the workflow step summary from the PNGs the Render
# step already produced, NOT by re-querying prometheus -- by
# this point ``Snapshot Metrics`` has stopped the prometheus
# container, so every range query would return no data.
# ``--image-url-prefix`` points at the stress-snapshots branch
# so each panel renders inline; without it the summary falls
# back to listing the artifact bundle.
if: always()
run: |
cd tests/monitoring
PANELS_DIR="${GITHUB_WORKSPACE}/artifacts/panels" \
python3 render_panels.py --summary --from-existing \
--artifact-name stress-test-results \
--artifact-url "${{ steps.upload-artifacts.outputs.artifact-url }}" \
--image-url-prefix "${{ steps.publish-snapshots.outputs.url-prefix }}"