mirror of
https://github.com/calibrain/shelfmark.git
synced 2026-09-28 17:55:03 +01:00
PR #1169 (python:3.14.6-slim -> python:3.15.0b3-slim) ran for 6h before GitHub's max job limit killed it, then did it again on re-run. Two independent defects. Dependabot proposed a beta at all: the config already excluded python from the docker digest group for dependabot-core#9496, but the comment claimed ungrouped python updates get their pre-release filtered. They don't. dependabot-core#13815 rewrote the Docker pre-release heuristic to catch PEP 440 tags (its tests cover 3.15.0a2 and 3.5.0b3), yet the suffixed real tag still got through seven months later. CPython spells pre-releases without a separator, so 3.15.0b3 parses as an ordinary version sorting above 3.14.6. Ignore python semver-minor/major instead of trusting the heuristic; patch and digest updates still flow. The run took hours rather than failing: the health wait looked bounded at 60 iterations x 2s, but bare `curl` has no timeout. The 3.15 image booted a container that bound 8084 without ever serving (greenlet has no 3.15 wheel, so the gevent gunicorn worker was wedged), so curl blocked on read forever and the loop never reached iteration 2. Every job's orphan process at cancellation was that curl. Bound each probe and switch to a wall-clock deadline, and add timeout-minutes so a hang can never reach 6h again. Verified against a socket that accepts and never responds: the old loop was still hung at 30s, the new one exits at 120s with HEALTHY=0 into the existing log-dump path, and a responsive endpoint is still detected immediately.
104 lines
3.8 KiB
Bash
Executable File
104 lines
3.8 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Run the Shelfmark e2e platform for a single config profile.
|
|
#
|
|
# ./run-e2e.sh [env/<profile>.env] [extra pytest args...]
|
|
#
|
|
# Boots the stack defined by the profile env file, waits for health, runs the
|
|
# matching cluster tests (the suite skips tests not applicable to the profile),
|
|
# then tears down. Set KEEP_UP=1 to leave the stack running for debugging.
|
|
set -euo pipefail
|
|
|
|
PLATFORM_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
cd "$PLATFORM_DIR"
|
|
|
|
ENV_FILE="${1:-env/baseline.env}"
|
|
shift || true
|
|
PYTEST_ARGS=("$@")
|
|
|
|
if [[ ! -f "$ENV_FILE" ]]; then
|
|
echo "error: env file not found: $ENV_FILE" >&2
|
|
echo "available profiles:" >&2
|
|
ls env/*.env >&2
|
|
exit 2
|
|
fi
|
|
|
|
# shellcheck disable=SC1090
|
|
set -a; source "$ENV_FILE"; set +a # export SM_*, COMPOSE_PROFILES, E2E_PROFILE
|
|
PROFILE="${E2E_PROFILE:-baseline}"
|
|
if [[ "$PROFILE" == "client-qbittorrent-delayed" ]]; then
|
|
E2E_RUN_ID="${E2E_RUN_ID:-$(date +%s)-$$}"
|
|
export SM_DOWNLOADS_HOST_DIR="${SM_DOWNLOADS_HOST_DIR:-./.state/$PROFILE/$E2E_RUN_ID/shelfmark-downloads}"
|
|
export SM_QBITTORRENT_DOWNLOADS_HOST_DIR="${SM_QBITTORRENT_DOWNLOADS_HOST_DIR:-./.state/$PROFILE/$E2E_RUN_ID/client-downloads}"
|
|
mkdir -p "$SM_DOWNLOADS_HOST_DIR" "$SM_QBITTORRENT_DOWNLOADS_HOST_DIR"
|
|
fi
|
|
COMPOSE=(docker compose --env-file "$ENV_FILE" -f docker-compose.e2e.yml)
|
|
|
|
STATE_DIR="$PLATFORM_DIR/.state"
|
|
LOG_FILE="$STATE_DIR/shelfmark.$PROFILE.log"
|
|
mkdir -p "$STATE_DIR/config" "$STATE_DIR/books" "$STATE_DIR/downloads" "$STATE_DIR/tmp"
|
|
|
|
cleanup() {
|
|
if [[ "${KEEP_UP:-0}" != "1" ]]; then
|
|
echo "==> tearing down ($PROFILE)"
|
|
"${COMPOSE[@]}" down -v --remove-orphans >/dev/null 2>&1 || true
|
|
else
|
|
echo "==> KEEP_UP=1: leaving stack running ($PROFILE)"
|
|
fi
|
|
}
|
|
trap cleanup EXIT
|
|
|
|
# E2E_NO_BUILD=1 reuses already-built images (see `make e2e-platform-build` /
|
|
# run-matrix.sh) so a matrix run builds the heavy shelfmark image only once.
|
|
if [[ "${E2E_NO_BUILD:-0}" == "1" ]]; then
|
|
echo "==> [$PROFILE] starting stack, reusing built images (profiles='${COMPOSE_PROFILES:-<none>}')"
|
|
"${COMPOSE[@]}" up -d --no-build
|
|
else
|
|
echo "==> [$PROFILE] building + starting stack (profiles='${COMPOSE_PROFILES:-<none>}')"
|
|
"${COMPOSE[@]}" up -d --build
|
|
fi
|
|
|
|
echo "==> [$PROFILE] waiting for shelfmark health"
|
|
HEALTHY=0
|
|
# The curl timeouts are load-bearing, not belt-and-braces. A container that
|
|
# binds 8084 but never answers (e.g. a broken C-extension wheel wedging the
|
|
# gunicorn worker) blocks a bare `curl` forever on read, so an iteration-counted
|
|
# loop never reaches iteration 2 and the wait becomes unbounded — that hung CI
|
|
# for the full 6h job limit on PR #1169. Bound each probe AND the whole wait.
|
|
HEALTH_DEADLINE=$((SECONDS + 120))
|
|
while ((SECONDS < HEALTH_DEADLINE)); do
|
|
if curl -fsS --connect-timeout 3 --max-time 5 http://localhost:8084/api/health >/dev/null 2>&1; then
|
|
HEALTHY=1
|
|
break
|
|
fi
|
|
sleep 2
|
|
done
|
|
|
|
# Capture boot diagnostics for the entrypoint/permission tests.
|
|
"${COMPOSE[@]}" logs shelfmark > "$LOG_FILE" 2>&1 || true
|
|
RESTARTS="$(docker inspect -f '{{.RestartCount}}' e2e-shelfmark 2>/dev/null || echo 0)"
|
|
echo "==> [$PROFILE] healthy=$HEALTHY restarts=$RESTARTS log=$LOG_FILE"
|
|
|
|
if [[ "$HEALTHY" != "1" && "$PROFILE" != "tor" ]]; then
|
|
echo "error: shelfmark never became healthy under profile '$PROFILE'" >&2
|
|
"${COMPOSE[@]}" logs --tail 40 shelfmark >&2 || true
|
|
exit 1
|
|
fi
|
|
|
|
# Hand context to the pytest suite.
|
|
export E2E_PROFILE="$PROFILE"
|
|
export E2E_BASE_URL="http://localhost:8084"
|
|
export E2E_BOOKS_DIR="$STATE_DIR/books"
|
|
export E2E_TMP_DIR="$STATE_DIR/tmp"
|
|
export E2E_SHELFMARK_LOG="$LOG_FILE"
|
|
export E2E_SHELFMARK_RESTARTS="$RESTARTS"
|
|
|
|
echo "==> [$PROFILE] running suite"
|
|
set +e
|
|
( cd "$PLATFORM_DIR/../../.." && \
|
|
uv run pytest tests/e2e/platform/suite -m platform -o addopts="--tb=short" "${PYTEST_ARGS[@]}" )
|
|
RC=$?
|
|
set -e
|
|
|
|
echo "==> [$PROFILE] pytest exit=$RC"
|
|
exit $RC
|