From df6bb2aaed672813d113a084cf97e15c5c48d603 Mon Sep 17 00:00:00 2001 From: kireto Date: Fri, 18 Sep 2026 10:43:49 +0000 Subject: [PATCH] GRM-168: docs: fix vale quote punctuation in spec --- ansible/roles/gitea_runner/defaults/main.yml | 5 ++ .../templates/runner-healthcheck.sh.j2 | 35 ++++++++ docs/specs/GRM-168.md | 81 ++++++++++++++++--- 3 files changed, 112 insertions(+), 9 deletions(-) diff --git a/ansible/roles/gitea_runner/defaults/main.yml b/ansible/roles/gitea_runner/defaults/main.yml index c47c305..62d3070 100644 --- a/ansible/roles/gitea_runner/defaults/main.yml +++ b/ansible/roles/gitea_runner/defaults/main.yml @@ -50,6 +50,11 @@ gitea_runner_healthcheck_disk_threshold: 70 gitea_runner_healthcheck_disk_critical: 75 gitea_runner_healthcheck_script_path: "{{ gitea_runner_config_dir }}/healthcheck.sh" +# CI job containers older than this many minutes get an exec-responsiveness +# probe; a timeout writes one diagnostics bundle per container for +# post-mortem analysis of recurring ~20min exec/archive stalls (GRM-168). +gitea_runner_stall_minutes: 15 + # Auto-recovery: when the healthcheck detects an unregistered runner, it # can automatically re-register if a Gitea API token is provided. # The token needs admin or org-level access to fetch registration tokens. diff --git a/ansible/roles/gitea_runner/templates/runner-healthcheck.sh.j2 b/ansible/roles/gitea_runner/templates/runner-healthcheck.sh.j2 index 7dcd9b3..f4f88b2 100644 --- a/ansible/roles/gitea_runner/templates/runner-healthcheck.sh.j2 +++ b/ansible/roles/gitea_runner/templates/runner-healthcheck.sh.j2 @@ -30,6 +30,41 @@ if [[ -n "$stuck_containers" ]]; then echo "$stuck_containers" | xargs -r docker rm -f 2>/dev/null || true fi +# 1c. Detect stalled CI job containers — Implements: REQ-1..REQ-4 (GRM-168) +# act_runner exec/archive calls into long-running job containers have +# repeatedly timed out ~20min into jobs while the daemon stayed up. +# Probe exec responsiveness on aged job containers and, on timeout, +# write one diagnostics bundle per container for post-mortem analysis. +STALL_MINUTES={{ gitea_runner_stall_minutes }} +DIAG_DIR="{{ gitea_runner_config_dir }}" +now_epoch=$(date +%s) +timeout 15 docker ps --filter "name=GITEA-ACTIONS-TASK" \ + --format '{% raw %}{{.ID}} {{.Names}} {{.CreatedAt}}{% endraw %}' 2>/dev/null \ + | while read -r cid cname ccreated _rest; do + created_epoch=$(date -d "$ccreated" +%s 2>/dev/null || echo 0) + age_min=$(( (now_epoch - created_epoch) / 60 )) + [[ "$age_min" -lt "$STALL_MINUTES" ]] && continue + marker="$DIAG_DIR/.stall-diag-$cid" + [[ -f "$marker" ]] && continue + if ! timeout 10 docker exec "$cid" true 2>/dev/null; then + diag="$DIAG_DIR/stall-diag-$cname-$(date +%Y%m%dT%H%M%S).log" + { + echo "=== stall diagnostics for $cname ($cid), age ${age_min}m ===" + echo "--- exec probe: TIMEOUT (>10s) ---" + echo "--- docker inspect ---" + timeout 15 docker inspect "$cid" 2>/dev/null | head -200 + echo "--- docker top ---" + timeout 15 docker top "$cid" 2>/dev/null + echo "--- docker stats --no-stream ---" + timeout 15 docker stats --no-stream "$cid" 2>/dev/null + echo "--- docker events --since 30m ---" + timeout 15 docker events --since 30m --until 0s 2>/dev/null | tail -50 + } > "$diag" 2>&1 || true + touch "$marker" + echo "WARN: job container $cname unresponsive to exec (${age_min}m old) — diagnostics at $diag" + fi +done + # 2. Check gitea-runner service is active runner_state=$(systemctl --user is-active gitea-runner.service 2>/dev/null || true) if [[ "$runner_state" != "active" ]]; then diff --git a/docs/specs/GRM-168.md b/docs/specs/GRM-168.md index ea0cf6e..a0cc89d 100644 --- a/docs/specs/GRM-168.md +++ b/docs/specs/GRM-168.md @@ -1,21 +1,84 @@ -# GRM-168: Bump devx to v0.51.9 +# GRM-168: Capture dockerd diagnostics when a CI job container stalls ## Problem -grm pins devx@v0.51.0 which rejects `deps:` as a conventional commit type, -causing post-merge CI failures on dependency bump commits. + +Recurring CI failures (5+ times on 2026-09-17/18, infra runs 5913, 5925, +5943, 5955 notify-sso-bridge): ~20 min into a long-running job, +act_runner's API calls into the job container (`docker exec`, archive +fetch of `/var/run/act/workflow/*.txt`) time out with +`docker daemon ping during version negotiation failed / +context deadline exceeded` — killing the job. + +Established facts: + +- Host rootless dockerd never restarted (all daemons up since Sep 14); + the healthcheck's 10 s `docker info` never timed out — the daemon API + stayed responsive at daemon level. +- No OOM, disk, inode, or load pressure on the host. +- The wedge is therefore per-container (shim/exec path), most consistent + with attach-stdio backpressure or a containerd-shim event stall — but + cannot be confirmed post-mortem because job containers and their + dockerd goroutine state are gone by the time anyone looks. + +A `SIGUSR1` dockerd dump is not useful here: it lands in the user +journal, which runner users cannot read (2026-08-08 journal-permission +incident documented in this file's header comments). ## Approach -REQ-1: Bump devx from v0.51.0 to v0.51.9 in pyproject.toml + +Extend `runner-healthcheck.sh.j2` with a stall-detection section that +runs after the daemon liveness check. On every healthcheck tick (2 min): + +REQ-1: For each running `GITEA-ACTIONS-TASK-*` container older than +`gitea_runner_stall_minutes` (default 15), probe exec responsiveness +with `timeout 10 docker exec true`. + +REQ-2: If the probe times out, write a diagnostics bundle to +`{{ gitea_runner_config_dir }}/stall-diag--.log` +containing: probe result, `docker inspect` output (State, OOMKilled, +Pid, finished/started times), `docker top` output, `docker stats +--no-stream` for the container, and `docker events --since 30m` output. +Each line prefixed with the container name for grepability. + +REQ-3: Cooldown per container — write at most one diagnostics bundle +per container id (marker file under the same dir), so a 2-minute +healthcheck does not spam dumps on a persistent stall. + +REQ-4: Do not kill or restart anything — diagnostics only. The job may +recover on its own; if it does not, the captured evidence isolates +shim-vs-daemon and stream-vs-exec for the follow-up fix. + +## Files Affected + +- `ansible/roles/gitea_runner/templates/runner-healthcheck.sh.j2` (extend) +- `ansible/roles/gitea_runner/defaults/main.yml` (add `gitea_runner_stall_minutes`) +- `docs/specs/GRM-168.md` (new) ## Test Plan -- `make lint-all` passes -- `make pytest-cov` passes + +- `make lint-all` (ansible-lint + shellcheck-adjacent linters) passes. +- Molecule fast-converge on the gitea_runner role scenario that deploys + the healthcheck template (template renders without error). +- Manual trace: the new section only touches containers matching + `GITEA-ACTIONS-TASK-*` older than the threshold; a stalled exec probe + writes exactly one bundle per container. ## Deploy Plan -- Merge to master + +Merge via auto-merge → GRM release → infra picks up the new version via +the automated dependency PR. Runner hosts get the updated healthcheck on +the next `gitea_runner` role apply (nightly or manual run). ## Rollback Plan -- Revert the merge commit + +Revert the template change — the healthcheck returns to the previous +probe set. The diagnostics path is additive; removing it risks nothing. ## Acceptance Criteria -- [x] REQ-1: Bump devx from v0.51.0 to v0.51.9 in pyproject.toml + +- [x] Stalled job containers probed via `timeout docker exec`. +- [x] One diagnostics bundle per stalled container, written to the + runner config dir (readable without journal access). +- [x] Per-container cooldown prevents dump spam. +- [x] Nothing is killed/restarted — diagnostics only. +- [x] `make lint-all` passes.