feat(ci): add GC CronJob for runner cleanup, expand reg PVCs to 20Gi

- Add gc-cronjob.yaml: daily prune of DinD Docker images/volumes/build-cache
  and actcache across all forgejo-runner pods. Keeps :latest tagged images,
  deletes non-latest older than 72h.
- Expand runner reg PVCs from 1Gi to 20Gi (all three runners) to prevent
  action tool cache from filling disk.
- Rust runner gets explicit 20Gi persistence override.
- GC only renders from golang (default) values to avoid duplicate resources.
This commit is contained in:
2026-08-30 07:08:59 -07:00
parent bde6c740ec
commit 91729f65e9
4 changed files with 171 additions and 1 deletions
@@ -0,0 +1,146 @@
{{- if .Values.gc.enabled }}
# Garbage-collects DinD Docker images/volumes/build-cache and actcache across
# ALL forgejo-runner pods. Prevents PVC fill-up that breaks CI runs.
# Only rendered once (enable in default values.yaml, disable in per-runner overrides).
apiVersion: v1
kind: ServiceAccount
metadata:
name: runner-gc
namespace: {{ .Release.Namespace }}
---
apiVersion: rbac.authorization.k8s.io/v1
kind: Role
metadata:
name: runner-gc
namespace: {{ .Release.Namespace }}
rules:
- apiGroups: [""]
resources: ["pods"]
verbs: ["get", "list"]
- apiGroups: [""]
resources: ["pods/exec"]
verbs: ["create"]
---
apiVersion: rbac.authorization.k8s.io/v1
kind: RoleBinding
metadata:
name: runner-gc
namespace: {{ .Release.Namespace }}
subjects:
- kind: ServiceAccount
name: runner-gc
namespace: {{ .Release.Namespace }}
roleRef:
kind: Role
name: runner-gc
apiGroup: rbac.authorization.k8s.io
---
apiVersion: batch/v1
kind: CronJob
metadata:
name: forgejo-runner-gc
namespace: {{ .Release.Namespace }}
labels:
app: forgejo-runner-gc
spec:
schedule: {{ .Values.gc.schedule | quote }}
concurrencyPolicy: Forbid
successfulJobsHistoryLimit: 3
failedJobsHistoryLimit: 3
jobTemplate:
spec:
backoffLimit: 1
activeDeadlineSeconds: 900
template:
spec:
serviceAccountName: runner-gc
restartPolicy: Never
tolerations:
- key: node-role.kubernetes.io/control-plane
operator: Exists
effect: NoSchedule
containers:
- name: gc
image: {{ .Values.gc.image }}
command:
- sh
- -c
- |
set -e
# Iterate all forgejo-runner pods (golang, rust, node)
PODS=$(kubectl -n {{ .Release.Namespace }} get pod \
-l app -o jsonpath='{range .items[*]}{.metadata.name}{"\t"}{.metadata.labels.app}{"\n"}{end}' \
| grep 'forgejo-runner-' | awk '{print $1}')
if [ -z "$PODS" ]; then
echo "no forgejo-runner pods found, skipping"
exit 0
fi
for POD in $PODS; do
echo "===== $POD ====="
# 1. Docker image prune (DinD sidecar)
echo "[docker] before:"
kubectl -n {{ .Release.Namespace }} exec "$POD" -c dind -- docker system df 2>/dev/null || true
echo "[docker] pruning non-latest images older than {{ .Values.gc.pruneAge }}..."
# Keep :latest tagged images, delete all others older than pruneAge.
# docker image prune can't filter by tag, so we list and selectively rmi.
kubectl -n {{ .Release.Namespace }} exec "$POD" -c dind -- \
sh -c '
# Remove dangling (untagged) images older than {{ .Values.gc.pruneAge }}
docker image prune -f --filter "until={{ .Values.gc.pruneAge }}" 2>/dev/null
# Remove tagged non-latest images older than {{ .Values.gc.pruneAge }}
CUTOFF=$(date -d "-{{ .Values.gc.pruneAgeHours }} hours" +%s 2>/dev/null || date -v-{{ .Values.gc.pruneAgeHours }}H +%s)
docker images --format "{{"{{"}} .Repository {{"}}"}}:{{"{{"}} .Tag {{"}}"}} {{"{{"}} .CreatedAt {{"}}"}}" | while read -r IMAGE_TAG CREATED_REST; do
TAG=$(echo "$IMAGE_TAG" | rev | cut -d: -f1 | rev)
# Skip latest-tagged images
if [ "$TAG" = "latest" ]; then
echo "[keep] $IMAGE_TAG (latest)"
continue
fi
# Check image age via inspect
CREATED_TS=$(docker inspect --format="{{"{{"}} .Created {{"}}"}}" "$IMAGE_TAG" 2>/dev/null | head -1)
if [ -z "$CREATED_TS" ]; then continue; fi
IMAGE_EPOCH=$(date -d "$CREATED_TS" +%s 2>/dev/null || date -jf "%Y-%m-%dT%H:%M:%S" "$(echo $CREATED_TS | cut -dT -f1-2 | cut -d. -f1)" +%s 2>/dev/null || echo 0)
if [ "$IMAGE_EPOCH" -lt "$CUTOFF" ] 2>/dev/null; then
echo "[delete] $IMAGE_TAG (older than {{ .Values.gc.pruneAge }})"
docker rmi -f "$IMAGE_TAG" 2>/dev/null || true
else
echo "[keep] $IMAGE_TAG (recent)"
fi
done
' 2>/dev/null || true
echo "[docker] pruning build cache unused >{{ .Values.gc.pruneAge }}..."
kubectl -n {{ .Release.Namespace }} exec "$POD" -c dind -- \
docker builder prune -af --filter "until={{ .Values.gc.pruneAge }}" 2>/dev/null || true
echo "[docker] pruning dangling volumes..."
kubectl -n {{ .Release.Namespace }} exec "$POD" -c dind -- \
docker volume prune -af 2>/dev/null || true
echo "[docker] after:"
kubectl -n {{ .Release.Namespace }} exec "$POD" -c dind -- docker system df 2>/dev/null || true
# 2. Actcache cleanup (runner container) — remove cached artifacts older than pruneAge
echo "[actcache] cleaning stale cache entries..."
kubectl -n {{ .Release.Namespace }} exec "$POD" -c runner -- \
sh -c 'find /data/.cache/actcache/cache -type f -mtime +{{ .Values.gc.actcacheMaxAgeDays }} -delete 2>/dev/null; \
find /data/.cache/actcache/cache -type d -empty -delete 2>/dev/null; \
echo "actcache size: $(du -sh /data/.cache/actcache/cache 2>/dev/null | cut -f1)"' || true
echo ""
done
echo "GC complete"
resources:
requests:
cpu: 50m
memory: 64Mi
limits:
cpu: 250m
memory: 128Mi
{{- end }}
@@ -7,3 +7,7 @@
runner: runner:
name: node-runner name: node-runner
labels: "node:docker://node:22-bookworm" labels: "node:docker://node:22-bookworm"
# GC CronJob renders only from the default (golang) values to avoid duplicates
gc:
enabled: false
@@ -8,3 +8,12 @@
runner: runner:
name: rust-runner name: rust-runner
labels: "rust:docker://rust:1.83-bookworm" labels: "rust:docker://rust:1.83-bookworm"
persistence:
reg:
storageClass: longhorn
size: 20Gi
# GC CronJob renders only from the default (golang) values to avoid duplicates
gc:
enabled: false
+12 -1
View File
@@ -39,7 +39,7 @@ dind:
persistence: persistence:
reg: reg:
storageClass: longhorn # Unified StorageClass (3 replicas) storageClass: longhorn # Unified StorageClass (3 replicas)
size: 1Gi # .runner registration file + config — survives pod restarts size: 20Gi # .runner registration file + action tool cache + actcache artifacts
dind: dind:
storageClass: longhorn # Unified StorageClass (3 replicas) storageClass: longhorn # Unified StorageClass (3 replicas)
size: 30Gi # docker layer cache — keeps rebuilds fast across restarts size: 30Gi # docker layer cache — keeps rebuilds fast across restarts
@@ -53,3 +53,14 @@ tolerations:
# attach there. # attach there.
nodeSelector: nodeSelector:
topology.kubernetes.io/zone: az-a topology.kubernetes.io/zone: az-a
# GC CronJob — prunes Docker images/volumes/build-cache and actcache across
# ALL forgejo-runner pods. Only enable in default values (golang instance);
# disable in per-runner overrides so it renders once.
gc:
enabled: true
schedule: "0 3 * * *" # daily 03:00 UTC
image: alpine/k8s:1.31.0
pruneAge: "72h" # Docker artifacts unused longer than this get pruned
pruneAgeHours: 72 # Same as pruneAge but numeric for date arithmetic in shell
actcacheMaxAgeDays: 3 # actcache files older than N days get deleted