feat: scoped runner cleanup with ownership leases and disk admission

Extend stopped-container protection to an explicit lease contract
(org.oblachno.lease-until / org.oblachno.owner) honored by every cleanup
path. Consolidate the duplicated inline prune logic from docker-prune
and the healthcheck into a single tiered runner-cleanup.sh; remove the
unfiltered `system prune -af --volumes` / `volume prune` paths that could
wipe a job's volumes mid-run, and keep warm base images under pressure.

At critical disk usage the healthcheck now stops admitting new work
(stops gitea-runner.service once no CI job is in flight) and resumes it
automatically after recovery. The runner config declares capacity, and
the role refuses to install on production-marked hosts.
This commit is contained in:
Emil Simeonov
2026-09-22 18:34:12 +02:00
parent 49646c38db
commit 3c0696f3f2
10 changed files with 385 additions and 106 deletions
+26 -5
View File
@@ -31,6 +31,13 @@ gitea_runner_prune_until: "24h"
# container operations).
gitea_runner_prune_schedule: "*-*-* 00/6:00:00"
gitea_runner_prune_label: "gitea-runner=true"
# Shared scoped cleanup script (runner-cleanup.sh) used by the prune timer
# and the healthcheck disk-pressure tiers (GRM-173).
gitea_runner_cleanup_script_path: "{{ gitea_runner_config_dir }}/cleanup.sh"
# Regex alternation of image refs never removed by cleanup — warm base
# layers stay warm even under critical disk pressure.
gitea_runner_keep_images:
- "runner-images/"
# Service configuration
gitea_runner_service_restart_sec: "5"
@@ -42,13 +49,20 @@ gitea_runner_service_restart_sec: "5"
gitea_runner_healthcheck_interval: "2min"
gitea_runner_healthcheck_boot_delay: "2min"
gitea_runner_healthcheck_disk_threshold: 70
# When disk reaches this level, prune EVERYTHING (no until-filter) — the
# runner is dangerously full and the gentle until=1h prune isn't enough.
# This removes all stopped containers and unused images regardless of age.
# At 75%+, molecule containers fail with "container is not running" because
# overlay2 runs out of space under parallel DinD load.
# At this level the cleanup script drops age limits — the runner is
# dangerously full and the gentle until=1h prune isn't enough. Leases,
# keep-images, running containers and CI job containers are still honored
# (GRM-173). At 75%+, molecule containers fail with "container is not
# running" because overlay2 runs out of space under parallel DinD load.
gitea_runner_healthcheck_disk_critical: 75
gitea_runner_healthcheck_script_path: "{{ gitea_runner_config_dir }}/healthcheck.sh"
# Disk-pressure admission control (GRM-173): at critical disk usage the
# healthcheck stops gitea-runner.service (no new jobs are fetched) once no
# CI job container is running, and resumes it automatically after recovery.
gitea_runner_disk_admission_enabled: true
# Physical-host admission (GRM-173): act_runner capacity — max parallel
# tasks per runner. Declared explicitly (upstream default is 1).
gitea_runner_capacity: 1
# CI job containers older than this many minutes get an exec-responsiveness
# probe; a timeout writes one diagnostics bundle per container for
@@ -121,6 +135,13 @@ gitea_runner_valid_volumes:
gitea_runner_containerd_max_compatible_major: 2
gitea_runner_containerd_max_compatible_minor: 2
# Production-host exclusion (GRM-173): the role fails when the target is a
# production host — either via this flag or the /etc/oblachno/production-host
# marker file — unless allow_production_host explicitly overrides.
gitea_runner_on_production_host: false
gitea_runner_allow_production_host: false
gitea_runner_production_marker_path: "/etc/oblachno/production-host"
# Docker installation (for rootless dependencies)
gitea_runner_docker_gpg_key_path: "/etc/apt/keyrings/docker.gpg"
gitea_runner_docker_apt_arch: "{{ 'amd64' if ansible_facts['architecture'] == 'x86_64' else ansible_facts['architecture'] }}"
@@ -47,14 +47,43 @@
ansible.builtin.assert:
that:
- "'Type=oneshot' in prune_service.content | b64decode"
- "'docker rm -f' in prune_service.content | b64decode"
- "'status=exited' in prune_service.content | b64decode"
- "'GITEA-ACTIONS-TASK' in prune_service.content | b64decode"
- "'docker system prune -af' in prune_service.content | b64decode"
- "'docker network prune' in prune_service.content | b64decode"
- "'docker builder prune' in prune_service.content | b64decode"
- "(gitea_runner_cleanup_script_path ~ ' --tier routine') in prune_service.content | b64decode"
fail_msg: "Prune service template is missing expected directives"
- name: Read rendered cleanup script
ansible.builtin.slurp:
src: "{{ gitea_runner_cleanup_script_path }}"
register: cleanup_script
- name: Assert cleanup script honors leases and tiers
ansible.builtin.assert:
that:
- "'org.oblachno.lease-until' in cleanup_script.content | b64decode"
- "'org.oblachno.owner' in cleanup_script.content | b64decode"
- "'lease_active' in cleanup_script.content | b64decode"
- "'GITEA-ACTIONS-TASK' in cleanup_script.content | b64decode"
- "'status=exited' in cleanup_script.content | b64decode"
- "'runner-images/' in cleanup_script.content | b64decode"
- "'--tier' in cleanup_script.content | b64decode"
- "'label!=' in cleanup_script.content | b64decode"
- "'docker system prune' not in cleanup_script.content | b64decode"
- "'routine)' in cleanup_script.content | b64decode"
- "'pressure)' in cleanup_script.content | b64decode"
- "'critical)' in cleanup_script.content | b64decode"
fail_msg: "Cleanup script template is missing expected content"
- name: Read rendered runner config
ansible.builtin.slurp:
src: "{{ gitea_runner_config_dir }}/config.yaml"
register: runner_config
- name: Assert runner config declares capacity
ansible.builtin.assert:
that:
- "('capacity: ' ~ gitea_runner_capacity) in runner_config.content | b64decode"
fail_msg: "Runner config is missing capacity declaration"
- name: Read rendered prune timer template
ansible.builtin.slurp:
src: "{{ gitea_runner_home }}/.config/systemd/user/docker-prune.timer"
@@ -108,12 +137,13 @@
- "'systemctl --user restart gitea-runner.service' in healthcheck_script.content | b64decode"
- "'docker rm -f' in healthcheck_script.content | b64decode"
- "'GITEA-ACTIONS-TASK' in healthcheck_script.content | b64decode"
- "'docker system prune -af' in healthcheck_script.content | b64decode"
- "'docker network prune' in healthcheck_script.content | b64decode"
- "'--tier critical' in healthcheck_script.content | b64decode"
- "'--tier pressure' in healthcheck_script.content | b64decode"
- "'disk-admission-block' in healthcheck_script.content | b64decode"
- "'systemctl --user stop gitea-runner.service' in healthcheck_script.content | b64decode"
- "'docker system prune' not in healthcheck_script.content | b64decode"
- "'status=removing' in healthcheck_script.content | b64decode"
- "'status=stopping' in healthcheck_script.content | b64decode"
- "'status=exited' in healthcheck_script.content | b64decode"
- "'status=dead' in healthcheck_script.content | b64decode"
- "gitea_runner_healthcheck_disk_threshold | string in healthcheck_script.content | b64decode"
- "gitea_runner_healthcheck_disk_critical | string in healthcheck_script.content | b64decode"
fail_msg: "Healthcheck script template is missing expected content"
+19
View File
@@ -1,4 +1,23 @@
---
# Implements: REQ-6 (GRM-173) — a CI runner must never be installed on a
# production host (production workloads must not share hardware with
# arbitrary CI jobs, and runner cleanup logic assumes a dedicated host).
- name: Check for production-host marker
ansible.builtin.stat:
path: "{{ gitea_runner_production_marker_path }}"
register: gitea_runner_production_marker
- name: Fail on production hosts
ansible.builtin.fail:
msg: >-
Refusing to install a CI runner on a production host
(marker: {{ gitea_runner_production_marker_path }} present or
gitea_runner_on_production_host=true). Set
gitea_runner_allow_production_host=true to override.
when:
- not gitea_runner_allow_production_host
- gitea_runner_on_production_host or gitea_runner_production_marker.stat.exists
- name: Include systemd availability check
ansible.builtin.include_tasks: systemd_check.yml
@@ -1,4 +1,14 @@
---
# Implements: REQ-2 (GRM-173) — shared scoped cleanup script used by both
# the prune timer and the healthcheck disk-pressure tiers.
- name: Create runner cleanup script
ansible.builtin.template:
src: runner-cleanup.sh.j2
dest: "{{ gitea_runner_cleanup_script_path }}"
owner: "{{ gitea_runner_service_user }}"
group: "{{ gitea_runner_service_user }}"
mode: "0755"
- name: Create docker-prune user service file
ansible.builtin.template:
src: docker-prune.service.j2
@@ -5,21 +5,10 @@ Description=Docker prune for Gitea runner resources
Type=oneshot
Environment=DOCKER_HOST=unix:///run/user/{{ gitea_runner_uid }}/docker.sock
Environment=XDG_RUNTIME_DIR=/run/user/{{ gitea_runner_uid }}
# Force-remove stale *stopped* containers left behind by failed molecule tests.
# Implements: REQ-1 (GRM-166) — only containers with status=exited are
# eligible. RunningFor measures creation time, so a stale molecule instance
# (e.g. ubuntu-2604) that a new run restarts still looks ">1h old"; removing
# running containers kills active converges with "No such container"
# (infra nightly run 5710). Running leftovers are instead reused or destroyed
# by the next molecule create/destroy cycle.
# Exclude CI job containers (name starts with GITEA-ACTIONS-TASK) — removing
# them kills the active CI job and causes "RWLayer is unexpectedly nil" errors.
# Only remove containers older than 1 hour (grep for "hour/day/week/month/year
# ago" in RunningFor) to avoid removing containers a job just created.
ExecStart=/bin/sh -c 'docker ps -a --filter "status=exited" --format "{% raw %}{{.ID}} {{.Names}} {{.RunningFor}}{% endraw %}" 2>/dev/null | grep -v "GITEA-ACTIONS-TASK" | grep -E "(hour|day|week|month|year)s? ago" | awk "{print $1}" | xargs -r docker rm -f 2>/dev/null || true'
ExecStart=/usr/bin/docker system prune -af --filter "until={{ gitea_runner_prune_until }}" --volumes
# Prune networks older than the prune-until threshold to avoid removing
# networks that molecule tests are actively creating (e.g. 'traefik' network
# created during molecule create phase before containers are attached).
ExecStart=/usr/bin/docker network prune -f --filter "until={{ gitea_runner_prune_until }}"
ExecStart=/usr/bin/docker builder prune -f
# Implements: REQ-1/REQ-2 (GRM-173) — all cleanup goes through the shared
# scoped cleanup script: only stopped containers, CI job containers excluded
# (GITEA-ACTIONS-TASK prefix), valid `org.oblachno.lease-until` leases never
# removed, keep-images retained. The historical inline logic here killed
# active molecule converges ("No such container", infra nightly run 5710)
# and CI jobs ("RWLayer is unexpectedly nil").
ExecStart={{ gitea_runner_cleanup_script_path }} --tier routine
@@ -3,6 +3,9 @@ log:
runner:
file: "{{ gitea_runner_file }}"
# Implements: REQ-5 (GRM-173) — physical-host admission: declared capacity
# limits parallel tasks instead of relying on labels alone.
capacity: {{ gitea_runner_capacity }}
fetch_timeout: 50s
fetch_interval: 2s
@@ -0,0 +1,102 @@
#!/bin/bash
# Scoped Docker cleanup for gitea-runner hosts.
# Implements: REQ-1..REQ-3 (GRM-173) — ownership leases, tiered watermarks,
# keep-images. Single entry point shared by docker-prune.service (routine)
# and runner-healthcheck.sh (pressure/critical).
# No `set -e`: a failing prune must not abort the remaining cleanup.
set -uo pipefail
DOCKER_HOST="unix:///run/user/{{ gitea_runner_uid }}/docker.sock"
XDG_RUNTIME_DIR="/run/user/{{ gitea_runner_uid }}"
export DOCKER_HOST XDG_RUNTIME_DIR
TIER="${1:-routine}"
# REQ-1 label contract: `org.oblachno.lease-until` (epoch) protects an
# object while in the future; `org.oblachno.owner` records the owning run.
LEASE_UNTIL_LABEL="org.oblachno.lease-until"
KEEP_IMAGES_RE="{{ gitea_runner_keep_images | join('|') }}"
now_epoch=$(date +%s)
# Implements: REQ-1 — a lease whose `lease-until` epoch lies in the future
# protects its object from every removal path in this script.
lease_active() {
local until="$1"
[[ -n "$until" && "$until" =~ ^[0-9]+$ && "$until" -gt "$now_epoch" ]]
}
# Remove stopped containers. $1 = "aged" (only >1h, RunningFor heuristic)
# or "all". CI job containers and valid leases are never removed.
remove_stopped_containers() {
local mode="$1"
# Implements: REQ-1/REQ-3 — pipe-separated fields; RunningFor contains
# spaces, so whitespace-splitting would break the age gate.
{ timeout 30 docker ps -a --filter "status=exited" --filter "status=dead" \
--format '{% raw %}{{.ID}}|{{.Names}}|{{.RunningFor}}|{{.Label "org.oblachno.lease-until"}}{% endraw %}' \
2>/dev/null || true; } \
| while IFS='|' read -r cid cname running_for lease_until; do
[[ -z "$cid" ]] && continue
case "$cname" in GITEA-ACTIONS-TASK*) continue ;; esac
lease_active "$lease_until" && continue
if [[ "$mode" != "all" ]] \
&& ! grep -qE '(hour|day|week|month|year)s? ago' <<<"$running_for"; then
continue
fi
docker rm -f "$cid" >/dev/null 2>&1 || true
done
}
# Remove unused images older than $1 ("all" = no age limit). The keep-list
# (warm base layers) and leased images are never removed; images referenced
# by any container are refused by the daemon anyway.
remove_old_images() {
local until="$1"
docker image prune -f --filter "label!=${LEASE_UNTIL_LABEL}" >/dev/null 2>&1 || true
local filters=(--filter "dangling=false")
[[ "$until" != "all" ]] && filters+=(--filter "until=${until}")
{ timeout 30 docker images "${filters[@]}" \
--format '{% raw %}{{.ID}}|{{.Repository}}:{{.Tag}}|{{.Label "org.oblachno.lease-until"}}{% endraw %}' \
2>/dev/null || true; } \
| while IFS='|' read -r iid ref lease_until; do
[[ -z "$iid" || "$ref" == *"<none>"* ]] && continue
[[ -n "$KEEP_IMAGES_RE" && "$ref" =~ $KEEP_IMAGES_RE ]] && continue
lease_active "$lease_until" && continue
docker image rm "$iid" >/dev/null 2>&1 || true
done
}
case "$TIER" in
routine)
remove_stopped_containers aged
remove_old_images "{{ gitea_runner_prune_until }}"
docker volume prune -f \
--filter "label!=${LEASE_UNTIL_LABEL}" \
--filter "until={{ gitea_runner_prune_until }}" >/dev/null 2>&1 || true
docker network prune -f \
--filter "label!=${LEASE_UNTIL_LABEL}" \
--filter "until={{ gitea_runner_prune_until }}" >/dev/null 2>&1 || true
docker builder prune -f --filter "until=24h" >/dev/null 2>&1 || true
;;
pressure)
remove_stopped_containers aged
remove_old_images "1h"
docker volume prune -f \
--filter "label!=${LEASE_UNTIL_LABEL}" \
--filter "until=1h" >/dev/null 2>&1 || true
docker network prune -f \
--filter "label!=${LEASE_UNTIL_LABEL}" \
--filter "until=1h" >/dev/null 2>&1 || true
docker builder prune -f --filter "until=24h" >/dev/null 2>&1 || true
;;
critical)
# Implements: REQ-3 — age limits dropped, ownership still honored.
remove_stopped_containers all
remove_old_images all
docker volume prune -f --filter "label!=${LEASE_UNTIL_LABEL}" >/dev/null 2>&1 || true
docker network prune -f --filter "label!=${LEASE_UNTIL_LABEL}" >/dev/null 2>&1 || true
docker builder prune -af >/dev/null 2>&1 || true
;;
*)
echo "ERROR: unknown cleanup tier '$TIER' (expected routine|pressure|critical)" >&2
exit 2
;;
esac
@@ -265,54 +265,58 @@ except Exception:
{% endif %}
fi
# 3. Check disk space — prune aggressively if below threshold
# 3. Check disk space — scoped tiered cleanup via the shared cleanup script.
# Implements: REQ-2/REQ-3 (GRM-173) — cleanup honors org.oblachno.lease-until
# ownership leases, the keep-images list (warm base layers), the
# GITEA-ACTIONS-TASK job-container exclusion, and never removes running
# containers. No unfiltered prune remains: the previous `system prune -af
# --volumes` and unfiltered volume prune could wipe a job's freshly created
# but momentarily unused volumes mid-run.
disk_pct=$(df -P / | awk 'NR==2 {gsub(/%/, "", $5); print $5}')
ADMISSION_MARKER="{{ gitea_runner_config_dir }}/disk-admission-block"
CLEANUP_SCRIPT="{{ gitea_runner_cleanup_script_path }}"
if [[ "$disk_pct" -ge {{ gitea_runner_healthcheck_disk_critical }} ]]; then
echo "CRITICAL: Disk usage at ${disk_pct}% (>= {{ gitea_runner_healthcheck_disk_critical }}%), full prune"
# Critical level: remove ALL stopped containers (no age filter) and ALL
# unused images/volumes. The until=1h gentle prune is insufficient here.
# Implements: REQ-1 (GRM-167) — only exited/dead containers are removed.
# Running molecule instances are never killed: RunningFor counts creation
# time, so an adopted stale instance looks old; and a running container's
# writable layer is tiny — images/volumes are what actually fills the disk.
docker ps -a --filter "status=exited" --filter "status=dead" \
--format '{% raw %}{{.ID}} {{.Names}}{% endraw %}' 2>/dev/null \
| grep -v 'GITEA-ACTIONS-TASK' \
| awk '{print $1}' \
| xargs -r docker rm -f 2>/dev/null || true
docker system prune -af --volumes || true
docker network prune -f || true
docker builder prune -af || true
echo "CRITICAL: Disk usage at ${disk_pct}% (>= {{ gitea_runner_healthcheck_disk_critical }}%), critical cleanup"
"$CLEANUP_SCRIPT" --tier critical || true
disk_pct=$(df -P / | awk 'NR==2 {gsub(/%/, "", $5); print $5}')
echo "INFO: Disk usage after full prune: ${disk_pct}%"
echo "INFO: Disk usage after critical cleanup: ${disk_pct}%"
# Implements: REQ-4 — stop admitting new jobs while critically full,
# but only when no CI job is in flight (stopping the runner service
# mid-job would kill it). A later healthcheck resumes the service once
# disk drops below the warn threshold.
{% if gitea_runner_disk_admission_enabled %}
in_flight=$(timeout 15 docker ps --filter "name=GITEA-ACTIONS-TASK" \
--format '{% raw %}{{.ID}}{% endraw %}' 2>/dev/null | wc -l || echo 0)
if [[ "$in_flight" -eq 0 ]] \
&& systemctl --user is-active --quiet gitea-runner.service; then
echo "ADMISSION: disk critical, no jobs in flight — stopping runner service"
date +%s > "$ADMISSION_MARKER" 2>/dev/null || true
systemctl --user stop gitea-runner.service || true
elif [[ "$in_flight" -gt 0 ]]; then
echo "ADMISSION: disk critical but ${in_flight} job(s) in flight — runner left running"
fi
{% endif %}
elif [[ "$disk_pct" -ge {{ gitea_runner_healthcheck_disk_threshold }} ]]; then
echo "WARN: Disk usage at ${disk_pct}%, pruning runner resources (until=1h)"
# Force-remove stale stopped containers older than 1 hour.
# Implements: REQ-1 (GRM-167) — only exited/dead containers are removed.
# A running molecule instance must never be janitor-killed: RunningFor
# measures creation time, so a stale instance restarted by an active run
# looks ">1h old" and would die mid-converge ("No such container",
# infra nightly run 5710). Running leftovers are reused or destroyed by
# the next molecule create/destroy cycle.
# Exclude CI job containers (name starts with GITEA-ACTIONS-TASK).
docker ps -a --filter "status=exited" --filter "status=dead" \
--format '{% raw %}{{.ID}} {{.Names}} {{.RunningFor}}{% endraw %}' 2>/dev/null \
| grep -v 'GITEA-ACTIONS-TASK' \
| grep -E '(hour|day|week|month|year)s? ago' \
| awk '{print $1}' \
| xargs -r docker rm -f 2>/dev/null || true
# Prune images and containers older than 1h (until filter is NOT
# supported with --volumes, so prune volumes separately without a filter).
docker image prune -af --filter "until=1h" 2>/dev/null || true
docker container prune -f --filter "until=1h" 2>/dev/null || true
docker volume prune -f 2>/dev/null || true
# Prune networks older than 1 hour to avoid removing networks that
# molecule tests are actively creating (e.g. 'traefik' network created
# during molecule create phase before containers are attached).
docker network prune -f --filter "until=1h" || true
echo "WARN: Disk usage at ${disk_pct}%, pressure cleanup (until=1h)"
"$CLEANUP_SCRIPT" --tier pressure || true
disk_pct=$(df -P / | awk 'NR==2 {gsub(/%/, "", $5); print $5}')
echo "INFO: Disk usage after prune: ${disk_pct}%"
echo "INFO: Disk usage after cleanup: ${disk_pct}%"
fi
# Implements: REQ-4 — resume admission once pressure has cleared.
{% if gitea_runner_disk_admission_enabled %}
if [[ -f "$ADMISSION_MARKER" ]]; then
if [[ "$disk_pct" -lt {{ gitea_runner_healthcheck_disk_threshold }} ]]; then
echo "ADMISSION: disk recovered to ${disk_pct}% — resuming runner service"
rm -f "$ADMISSION_MARKER" 2>/dev/null || true
systemctl --user start gitea-runner.service || true
else
echo "ADMISSION: still blocked (disk ${disk_pct}%), runner stays stopped"
fi
fi
{% endif %}
echo "OK: runner healthy, disk at ${disk_pct}%"
exit 0