Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
64 changes: 40 additions & 24 deletions .github/workflows/arm-wedge-hunt.yml
Original file line number Diff line number Diff line change
Expand Up @@ -23,10 +23,10 @@ name: arm-wedge-hunt
on:
workflow_dispatch:
inputs:
iterations:
description: "How many times to run the fixture"
rounds:
description: "Rounds of 4 concurrent runs (4 samples per round)"
required: false
default: "20"
default: "6"

jobs:
hunt:
Expand Down Expand Up @@ -62,28 +62,44 @@ jobs:
set +e
TEST=../out_orly/debug/orly/server/import_replication.test
test -x "$TEST" || { echo "no test binary at $TEST"; exit 1; }
# Oversubscribe: `make test` runs four binaries at once, and both
# real failures looked load-related.
for c in 1 2 3 4; do bash -c 'while :; do :; done' & done
LOAD=$(jobs -p | tr '\n' ' ')
rc=0
for i in $(seq 1 ${{ github.event.inputs.iterations || '20' }}); do
echo "==================== iteration $i ===================="
timeout -s KILL 600 "$TEST" --le --log_info > "/tmp/hunt.$i.out" 2>&1
rc=$?
tail -5 "/tmp/hunt.$i.out"
if [ $rc -ne 0 ]; then
echo "CAUGHT on iteration $i (rc=$rc)"
echo "caught=$i" >> "$GITHUB_OUTPUT"
echo "--- the wedge dump, if the fixture produced one ---"
grep -A 200 "interrogating before it dies" "/tmp/hunt.$i.out" || \
echo "(no dump -- the failure was not a Reap timeout)"
break
fi
pkill -9 -x orlyi 2>/dev/null
# Run COPIES CONCURRENTLY rather than one at a time behind CPU
# spinners. The first version of this did the latter and took 35
# samples without once reproducing #564, while CI reproduces it about
# one run in sixteen -- the difference being that CI's `make test`
# runs four real binaries at once, so the contention is for disk,
# memory and the scheduler, not just for CPU. Concurrent copies buy
# the right kind of load and four samples per round at the same time.
ROUNDS=${{ github.event.inputs.rounds || '6' }}
COPIES=4
caught=""
for r in $(seq 1 "$ROUNDS"); do
echo "==================== round $r ($COPIES concurrent) ===================="
pids=""
for c in $(seq 1 "$COPIES"); do
timeout -s KILL 900 "$TEST" --le --log_info > "/tmp/hunt.r$r.c$c.out" 2>&1 &
pids="$pids $!"
done
for p in $pids; do wait "$p"; done
for c in $(seq 1 "$COPIES"); do
f="/tmp/hunt.r$r.c$c.out"
g=$(grep -c "end GracefulShutdownUnresponsiveSlave; fail" "$f")
i=$(grep -c "end ImportReplication; fail" "$f")
echo " round $r copy $c: graceful_fail=$g import_fail=$i"
# Only the graceful fixture is what this hunt is for. An
# ImportReplication failure is a different arm flake and must not
# end the batch -- that is what cut the first hunt short at 15 of
# 20 samples.
if [ "$g" != "0" ]; then
caught="round $r copy $c"
echo "CAUGHT the shutdown wedge: $caught"
echo "--- wedge dump ---"
grep -A 200 "interrogating before it dies" "$f" || \
echo "(the graceful fixture failed WITHOUT a Reap timeout -- different failure mode, read the artifact)"
fi
done
[ -n "$caught" ] && break
done
kill $LOAD 2>/dev/null
echo "final rc=$rc"
[ -n "$caught" ] && echo "caught=$caught" >> "$GITHUB_OUTPUT"
# Report the hunt itself as successful; a caught wedge is the payload,
# not a build failure.
exit 0
Expand Down
Loading