aboutsummaryrefslogtreecommitdiffstats
path: root/image-builder
diff options
context:
space:
mode:
authorDanilo M. <danix@danix.xyz>2026-09-22 11:21:10 +0200
committerDanilo M. <danix@danix.xyz>2026-09-22 11:21:10 +0200
commite2a9bfbc4775b48665dab40d60bb654eeeff5762 (patch)
treebf59e3502a99e62c8265fdadaee79ab36494fde1 /image-builder
parent48882edb25df1dabb6197edda8ffb092c32731fd (diff)
downloadsbo-dockerbuild-e2a9bfbc4775b48665dab40d60bb654eeeff5762.tar.gz
sbo-dockerbuild-e2a9bfbc4775b48665dab40d60bb654eeeff5762.zip
image-builder: alert to Gotify on failure and on staleness
The chain had no failure notification at all, which is why the September outage ran ten days: every failure was in the log from the first night and nobody read the log. notify.sh has two modes because the chain fails in two ways and only one of them has a non-zero exit status: run <label> <cmd...> runs the command, posts on non-zero, and passes the status through so cron still sees the truth. stale posts if any tag is older than its budget. The second exists because exit status alone would not have caught what happened. build-full-image.sh did report non-zero for ten nights, but build-sbo-testbuild.sh exited 0 every one of them: it saw an unchanged parent digest and skipped, which is correct behaviour. After the first alert the chain would have gone quiet while its tags aged six days. The staleness check asks the registry a different question, "is anything still current", and catches a skipped stage, a stopped cron or a wedged mirror alike. Budgets are split: -current rebuilds nightly (2 days), 15.0 is frozen and legitimately sits still for weeks (30). A single budget would either cry wolf on the stable tags or go blind on the rolling ones. The token is read from /root/.gotify-token (mode 600, not in the repo). Posting is best-effort throughout: a notifier that fails a build because the notifier is down would be worse than no notifier. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Diffstat (limited to 'image-builder')
-rw-r--r--image-builder/crontab.example33
-rwxr-xr-ximage-builder/notify.sh171
2 files changed, 197 insertions, 7 deletions
diff --git a/image-builder/crontab.example b/image-builder/crontab.example
index 174ace0..28f72d1 100644
--- a/image-builder/crontab.example
+++ b/image-builder/crontab.example
@@ -18,9 +18,14 @@
# 05:00 15.0 bootstrap -> full -> testbuild
# 07:00 reclaim dangling images (catches both variants)
# 08:00 registry blob GC, Sundays only
+# 09:00 staleness alert if any tag stopped moving
#
# Repos sync at 01:00/02:00, so the chain starts after that and the images are
# ready by 09:00.
+#
+# Every build runs under `notify.sh run`, which posts to Gotify on a non-zero
+# exit and passes the status through. That is necessary but not sufficient:
+# see the staleness check at the bottom for why.
# ---------------------------------------------------------------------------
# Pre-build reclaim
@@ -47,13 +52,13 @@
# no-op that exits in seconds.
#
# -current (moves daily):
-0 3 * * * /opt/sbo-testbuild/image-builder/bootstrap.sh --version current >> /var/log/sbo-testbuild.log 2>&1
-20 3 * * * /opt/sbo-testbuild/image-builder/build-full-image.sh --version current >> /var/log/sbo-testbuild.log 2>&1
-30 4 * * * /opt/sbo-testbuild/image-builder/build-sbo-testbuild.sh --version current >> /var/log/sbo-testbuild.log 2>&1
+0 3 * * * /opt/sbo-testbuild/image-builder/notify.sh run "bootstrap current" /opt/sbo-testbuild/image-builder/bootstrap.sh --version current >> /var/log/sbo-testbuild.log 2>&1
+20 3 * * * /opt/sbo-testbuild/image-builder/notify.sh run "full current" /opt/sbo-testbuild/image-builder/build-full-image.sh --version current >> /var/log/sbo-testbuild.log 2>&1
+30 4 * * * /opt/sbo-testbuild/image-builder/notify.sh run "testbuild current" /opt/sbo-testbuild/image-builder/build-sbo-testbuild.sh --version current >> /var/log/sbo-testbuild.log 2>&1
# 15.0 (frozen stable; rebuilds only on a real repo update):
-0 5 * * * /opt/sbo-testbuild/image-builder/bootstrap.sh --version 15.0 >> /var/log/sbo-testbuild.log 2>&1
-20 5 * * * /opt/sbo-testbuild/image-builder/build-full-image.sh --version 15.0 >> /var/log/sbo-testbuild.log 2>&1
-30 6 * * * /opt/sbo-testbuild/image-builder/build-sbo-testbuild.sh --version 15.0 >> /var/log/sbo-testbuild.log 2>&1
+0 5 * * * /opt/sbo-testbuild/image-builder/notify.sh run "bootstrap 15.0" /opt/sbo-testbuild/image-builder/bootstrap.sh --version 15.0 >> /var/log/sbo-testbuild.log 2>&1
+20 5 * * * /opt/sbo-testbuild/image-builder/notify.sh run "full 15.0" /opt/sbo-testbuild/image-builder/build-full-image.sh --version 15.0 >> /var/log/sbo-testbuild.log 2>&1
+30 6 * * * /opt/sbo-testbuild/image-builder/notify.sh run "testbuild 15.0" /opt/sbo-testbuild/image-builder/build-sbo-testbuild.sh --version 15.0 >> /var/log/sbo-testbuild.log 2>&1
# ---------------------------------------------------------------------------
# Post-build cleanup
@@ -65,4 +70,18 @@
# The registry never reclaims on its own: every push adds blobs and nothing
# removes them, so its store grows until the disk fills. Weekly is enough.
# registry-gc.sh has its own safety gates; see the script.
-0 8 * * 0 /opt/sbo-testbuild/image-builder/registry-gc.sh >> /var/log/sbo-testbuild.log 2>&1
+0 8 * * 0 /opt/sbo-testbuild/image-builder/notify.sh run "registry GC" /opt/sbo-testbuild/image-builder/registry-gc.sh >> /var/log/sbo-testbuild.log 2>&1
+
+# ---------------------------------------------------------------------------
+# Staleness check
+# ---------------------------------------------------------------------------
+# The exit-status alerts above would not have caught the September 2026
+# outage on their own. build-full-image.sh failed for ten nights and did
+# report non-zero, but build-sbo-testbuild.sh exited 0 every single night:
+# it saw an unchanged parent digest and skipped, which is correct. So after
+# the first alert the chain went quiet while its tags aged six days.
+#
+# This asks the registry a different question: not "did anything error" but
+# "is anything still current". It catches a skipped stage, a stopped cron and
+# a wedged mirror alike. Runs after the chain has had its chance.
+0 9 * * * /opt/sbo-testbuild/image-builder/notify.sh stale >> /var/log/sbo-testbuild.log 2>&1
diff --git a/image-builder/notify.sh b/image-builder/notify.sh
new file mode 100755
index 0000000..3f220e1
--- /dev/null
+++ b/image-builder/notify.sh
@@ -0,0 +1,171 @@
+#!/bin/bash
+#
+# Copyright (C) 2026 Danilo M. <danix@danix.xyz>
+#
+# This program is free software; you can redistribute it and/or modify
+# it under the terms of the GNU General Public License version 2 as
+# published by the Free Software Foundation.
+#
+# This program is distributed in the hope that it will be useful,
+# but WITHOUT ANY WARRANTY; without even the implied warranty of
+# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+# GNU General Public License for more details.
+#
+# notify.sh — push build-chain failures to Gotify.
+#
+# Two modes, because the chain fails in two ways and only one of them
+# has a non-zero exit status:
+#
+# run <label> <cmd...> run cmd; on failure post the label, the exit
+# code and the tail of the log.
+# stale post if any tag is older than its budget.
+#
+# The second mode exists because of how the chain broke in September 2026:
+# build-full-image.sh failed for ten nights, but build-sbo-testbuild.sh
+# exited 0 every one of them. It saw an unchanged parent digest and skipped,
+# which is correct behaviour. Nothing errored, nothing moved, and nothing
+# said so. An exit-status alert alone would have caught the first failure
+# but not the six days of silence that followed, so `stale` asks the
+# registry a different question: not "did it error" but "is it current".
+set -euo pipefail
+PROJECT_VERSION="1.1.2" # bump via sed across all scripts; see CLAUDE.md Releases
+HERE="$(cd "$(dirname "$0")" && pwd)"
+source "${HERE}/config"
+
+GOTIFY_URL="${GOTIFY_URL:-http://gotify.noland.dnx}"
+GOTIFY_TOKEN_FILE="${GOTIFY_TOKEN_FILE:-/root/.gotify-token}"
+LOGFILE="${LOGFILE:-/var/log/sbo-testbuild.log}"
+# -current rebuilds nightly; 15.0 is frozen and legitimately sits for weeks.
+STALE_DAYS="${STALE_DAYS:-2}"
+STALE_DAYS_STABLE="${STALE_DAYS_STABLE:-30}"
+STABLE_VARIANT="${STABLE_VARIANT:-15.0}"
+
+# post TITLE MESSAGE PRIORITY
+# Best-effort by design: a notifier that fails a build because the notifier
+# is down is worse than no notifier. Never exits non-zero.
+post() {
+ local title="$1" message="$2" priority="${3:-8}" token
+
+ if [[ ! -r "${GOTIFY_TOKEN_FILE}" ]]; then
+ echo "notify: no readable token at ${GOTIFY_TOKEN_FILE}; not posting" >&2
+ return 0
+ fi
+ token="$(tr -d '\n' < "${GOTIFY_TOKEN_FILE}")"
+ [[ -n "${token}" ]] || { echo "notify: empty token; not posting" >&2; return 0; }
+
+ curl -sf -m 10 -o /dev/null \
+ "${GOTIFY_URL}/message?token=${token}" \
+ -F "title=${title}" \
+ -F "message=${message}" \
+ -F "priority=${priority}" \
+ || echo "notify: POST to ${GOTIFY_URL} failed (ignored)" >&2
+ return 0
+}
+
+# run LABEL CMD...
+# Runs CMD and propagates its exit status, so cron and the caller still see
+# the real result; the notification is a side effect, not a substitute.
+cmd_run() {
+ local label="$1"; shift
+ [[ $# -gt 0 ]] || { echo "notify: run needs a command" >&2; exit 2; }
+
+ local rc=0
+ "$@" || rc=$?
+ [[ ${rc} -eq 0 ]] && return 0
+
+ # The scripts log their own diagnostics; the last lines are usually the
+ # actual error ("no space left on device" and friends).
+ local tail_txt=""
+ [[ -r "${LOGFILE}" ]] && tail_txt="$(tail -n 12 "${LOGFILE}" 2>/dev/null || true)"
+
+ post "sbo-testbuild: ${label} FAILED" \
+ "exit ${rc} at $(date '+%F %T') on $(hostname -s)
+
+${tail_txt}" \
+ 8
+ return "${rc}"
+}
+
+# stale
+# Ask the registry how old each tag is. Independent of the build scripts on
+# purpose: it catches a wedged chain, a stopped cron and a dead mirror alike,
+# none of which produce a failing exit status anywhere.
+cmd_stale() {
+ local now stale_list="" repo tag created age budget
+ now=$(date +%s)
+
+ local unreadable=""
+ for repo in sbo-base sbo-full sbo-testbuild; do
+ for tag in "${VARIANTS[@]}"; do
+ # A tag we cannot read is its own kind of bad news (registry down,
+ # tag never pushed), so say so rather than skipping quietly.
+ if ! created=$(tag_created "${repo}" "${tag}") || [[ -z "${created}" ]]; then
+ unreadable+="${repo}:${tag}"$'\n'
+ continue
+ fi
+
+ budget="${STALE_DAYS}"
+ [[ "${tag}" == "${STABLE_VARIANT}" ]] && budget="${STALE_DAYS_STABLE}"
+
+ age=$(( (now - created) / 86400 ))
+ if [[ ${age} -gt ${budget} ]]; then
+ stale_list+="${repo}:${tag} — ${age}d old (budget ${budget}d)"$'\n'
+ fi
+ done
+ done
+
+ if [[ -z "${stale_list}" && -z "${unreadable}" ]]; then
+ echo "all tags current"
+ return 0
+ fi
+
+ local body=""
+ [[ -n "${stale_list}" ]] && body+="Not refreshed:"$'\n'"${stale_list}"$'\n'
+ [[ -n "${unreadable}" ]] && body+="Could not read from the registry:"$'\n'"${unreadable}"$'\n'
+
+ echo "${body}"
+ post "sbo-testbuild: images going stale" \
+ "The chain has not refreshed these tags. A stage may be failing, or
+skipping because an earlier one did.
+
+${body}Check ${LOGFILE} on $(hostname -s)." \
+ 7
+}
+
+# tag_created REPO TAG -> unix timestamp on stdout, or non-zero if unreadable.
+# Two hops: the manifest names the config blob, the config blob has the date.
+#
+# Parsed with python3 rather than sed: the registry pretty-prints its JSON, so
+# the "config" object spans several lines and a line-oriented regex silently
+# matches nothing. python3 is already required here (build-full-image.sh serves
+# the mirror with http.server), so this adds no dependency.
+tag_created() {
+ local repo="$1" tag="$2" manifest cfg created
+ local accept='application/vnd.docker.distribution.manifest.v2+json'
+
+ manifest=$(curl -sf -m 10 -H "Accept: ${accept}" \
+ "http://${REGISTRY}/v2/${repo}/manifests/${tag}" 2>/dev/null) || return 1
+ cfg=$(printf '%s' "${manifest}" | python3 -c \
+ 'import sys,json; print(json.load(sys.stdin)["config"]["digest"])' 2>/dev/null) \
+ || return 1
+ [[ -n "${cfg}" ]] || return 1
+
+ created=$(curl -sf -m 10 "http://${REGISTRY}/v2/${repo}/blobs/${cfg}" 2>/dev/null \
+ | python3 -c 'import sys,json; print(json.load(sys.stdin).get("created",""))' \
+ 2>/dev/null) || return 1
+ [[ -n "${created}" ]] || return 1
+
+ # Registry timestamps carry nanoseconds, which `date -d` rejects; keep the
+ # seconds and the offset, drop the fraction.
+ created="${created/Z/+00:00}"
+ created="$(printf '%s' "${created}" | sed -E 's/\.[0-9]+([+-][0-9:]+)?$/\1/')"
+ date -d "${created}" +%s 2>/dev/null || return 1
+}
+
+case "${1:-}" in
+ -V|--version) echo "notify.sh ${PROJECT_VERSION}"; exit 0 ;;
+ run) shift; cmd_run "$@" ;;
+ stale) cmd_stale ;;
+ test) post "sbo-testbuild: test" "Notification test at $(date '+%F %T')." 2 ;;
+ *) echo "usage: $0 {run <label> <cmd...>|stale|test} [-V]" >&2; exit 2 ;;
+esac