Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
60 changes: 59 additions & 1 deletion .github/workflows/workflow.yml
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,12 @@ jobs:
spock_ref: main

runs-on: ${{ matrix.os }}
# Without this a hung tester burns the 6 hour job ceiling on every matrix
# entry, and the run is cancelled rather than failed, which skips the
# diagnostic steps below. Leave room for image build and container startup
# ahead of the 25 minute test wait, so the wait's own deadline is what fires
# and the diagnostics actually run.
timeout-minutes: 60

steps:
- name: Checkout lolor
Expand Down Expand Up @@ -115,14 +121,66 @@ jobs:
id: test_step
continue-on-error: true
run: |
# Cap the wait so a test blocked on a lock or on sub_wait_for_sync()
# fails the step while the cluster is still up to be inspected.
deadline=$(( SECONDS + 25 * 60 ))
while [ "$(docker inspect -f '{{.State.Running}}' tester)" == "true" ]; do
if [ "$SECONDS" -ge "$deadline" ]; then
echo "::error::Tests did not finish within 25 minutes; dumping cluster state"
# These run precisely because the cluster is wedged, so bound each
# one: "|| true" catches a bad exit status but not a hang, and a
# blocked query here would burn the job timeout and skip the
# failure-log step below.
PGV=${{ matrix.pgver }}
echo "===== tester output so far ====="
tail -n 200 tests/out.txt || true
for n in n1 n2 n3; do
echo "===== $n: sessions and what they are waiting on ====="
timeout --kill-after=5s 20s docker exec "$n" bash -c "source /home/pgedge/pgedge/pg${PGV}/pg${PGV}.env && psql -U admin -d demo -x -c \"
SELECT pid, state, wait_event_type, wait_event, now() - xact_start AS xact_age,
left(query, 400) AS query
FROM pg_stat_activity
WHERE datname = 'demo' AND pid <> pg_backend_pid()
ORDER BY xact_start\"" || true
echo "===== $n: ungranted locks ====="
timeout --kill-after=5s 20s docker exec "$n" bash -c "source /home/pgedge/pgedge/pg${PGV}/pg${PGV}.env && psql -U admin -d demo -c \"
SELECT l.pid, l.locktype, l.mode, l.granted,
coalesce(c.relname, l.classid::text) AS object
FROM pg_locks l LEFT JOIN pg_class c ON c.oid = l.relation
WHERE NOT l.granted OR l.mode LIKE 'Share%' OR l.mode LIKE '%Exclusive%'
ORDER BY l.granted, l.pid\"" || true
echo "===== $n: subscription status ====="
timeout --kill-after=5s 20s docker exec "$n" bash -c "source /home/pgedge/pgedge/pg${PGV}/pg${PGV}.env && psql -U admin -d demo -c 'SELECT * FROM spock.sub_show_status()'" || true
done
exit 1
fi
echo "Waiting for tests to complete..."
sleep 1
done
docker logs n1
docker logs n2
docker logs n3
grep -Eq "FAIL|ERROR" tests/out.txt && exit 1 || exit 0
# grep exits 2 when the file is missing, which "|| exit 0" would turn
# into a pass: a tester that died before writing anything would look
# green. Check for output first, then for failures in it.
if [ ! -s tests/out.txt ]; then
echo "::error::tests/out.txt is missing or empty; the tester produced no output"
exit 1
fi
if grep -Eq "FAIL|ERROR" tests/out.txt; then
exit 1
fi

- name: Dump container logs when the tests failed
if: steps.test_step.outcome == 'failure'
run: |
cd docker
for c in n1 n2 n3 tester; do
echo "===== $c ====="
docker logs "$c" > /tmp/$c.log 2>&1 || true
echo "--- first 120 lines (setup) ---"; head -n 120 /tmp/$c.log || true
echo "--- last 120 lines ---"; tail -n 120 /tmp/$c.log || true
done

- name: Upload Log File as Artifact
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
Expand Down
7 changes: 7 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
@@ -1,3 +1,10 @@
*.so
*.bc
*.o
*.dylib
results/
regression.diffs
regression.out
tmp_check/
log/
delete_old_cluster.sh
48 changes: 38 additions & 10 deletions docker/entrypoint.sh
Original file line number Diff line number Diff line change
Expand Up @@ -9,8 +9,18 @@ echo ". /home/pgedge/pgedge/pg${PG_VER}/pg${PG_VER}.env" >> /home/pgedge/.bashrc
# deprecated pgedge CLI.
initdb -D "$PGDATA" -U admin --encoding=UTF8 --locale=C

# Recent minors gate logical decoding on output_plugin_libraries, which does
# not list spock_output; without it every apply worker fails to create its
# slot and no subscription ever syncs. Older minors reject the parameter, so
# ask the binary whether it is supported.
OUTPUT_PLUGIN_LIBRARIES=""
if postgres --describe-config 2>/dev/null | grep -q '^output_plugin_libraries'; then
OUTPUT_PLUGIN_LIBRARIES="output_plugin_libraries = 'pgoutput, test_decoding, spock_output'"
fi

cat >> "$PGDATA/postgresql.conf" <<_EOF_
listen_addresses = '*'
$OUTPUT_PLUGIN_LIBRARIES
wal_level = logical
track_commit_timestamp = on
max_worker_processes = 32
Expand Down Expand Up @@ -51,20 +61,38 @@ _EOF_

IFS=',' read -r -a peer_names <<< "$PEER_NAMES"

# Bounded: an unbounded wait leaves the container up with the temporary
# postgres still listening, so the health check keeps passing while setup
# never finishes and the test harness blocks with no indication of why.
#
# The bound is wall clock rather than a loop count, and psql gets its own
# connect and statement timeouts: an unresponsive peer can block psql itself,
# which would let a counted loop run far past its nominal limit.
# One budget for all peers, not 300s each: a node waits for two peers and the
# tester for three, so a per-peer deadline would let startup run for fifteen
# minutes while still claiming a 300s bound.
deadline=$(( SECONDS + 300 ))
for PEER_HOSTNAME in "${peer_names[@]}";
do
while :
peer_ready=0
while [ "$SECONDS" -lt "$deadline" ]; do
mapfile -t node_array < <(PGCONNECT_TIMEOUT=5 PGOPTIONS='-c statement_timeout=10s' \
psql -A -t demo -h "$PEER_HOSTNAME" -c "SELECT node_name FROM spock.node;")
for element in "${node_array[@]}";
do
mapfile -t node_array < <(psql -A -t demo -h $PEER_HOSTNAME -c "SELECT node_name FROM spock.node;")
for element in "${node_array[@]}";
do
if [[ "$element" == "$PEER_HOSTNAME" ]]; then
break 2
fi
done
sleep 1
echo "Waiting for $PEER_HOSTNAME..."
if [[ "$element" == "$PEER_HOSTNAME" ]]; then
peer_ready=1
break
fi
done
[ "$peer_ready" = "1" ] && break
sleep 1
echo "Waiting for $PEER_HOSTNAME..."
done
if [ "$peer_ready" != "1" ]; then
echo "ERROR: peer $PEER_HOSTNAME did not register a spock node within the 300s startup budget" >&2
exit 1
fi
done

# spock.sub_create connects to the provider synchronously, and the peer
Expand Down
Loading
Loading