#!/bin/sh # HAPX-UI privileged update helper. # # Installed root:root mode 0755 — NOT writable by the hapx-ui service user. # Invoked only through the pinned sudoers rule (UNCHANGED — three verbs): # hapx-ui ALL=(root) NOPASSWD: /usr/local/sbin/hapx-ui-update preflight, # /usr/local/sbin/hapx-ui-update build, # /usr/local/sbin/hapx-ui-update activate # # Keeping the privileged logic in THIS fixed, root-owned script (instead of a # script generated at runtime by the web process) is the security boundary: # the unprivileged web user can trigger exactly these three actions and nothing # else — it can never inject arbitrary root commands. # # Robustness model: # build → obtain /var/lib/hapx-ui/hapx-ui.staged by the BEST available # method: download a verified prebuilt release binary (primary), # else build from a writable source clone (fallback). No downtime. # activate → atomic swap + DB backup + health-checked restart + auto-rollback. # Everything (target binary, service, db path, health URL) is auto-detected so # it works regardless of where/how the operator installed HAPX-UI. set -u # ── Fixed locations (the data dir is always writable; see systemd ReadWritePaths) DATA=/var/lib/hapx-ui LOG="$DATA/last-update.log" STATE="$DATA/update-state.json" LOCK="$DATA/update.lock" BACKUP_DIR="$DATA/backups" # Source-build working copy lives under the ROOT-ONLY staging tree (a /var/lib # sibling), NOT under the service-writable $DATA — same reasoning as STAGED: root # runs git+go here, so the service user must not be able to plant/rename it (L36). WORK="/var/lib/hapx-ui-staging/src" KEEP=5 # how many binary/DB backups to retain ETC=/etc/hapx-ui # root-owned config dir for trust-anchor opt-outs # ── Staging (ROOT-ONLY) ─────────────────────────────────────────────────────── # The staged binary is produced by root (download+verify or gated source build) # and later consumed by root (activate execs/installs it as the system binary). # It MUST therefore live where the unprivileged service user cannot tamper with # it — otherwise the service user could drop an arbitrary binary here and have # `activate` install it over /usr/local/bin/hapx-ui (signing-anchor bypass) or # have `build` exec it as root. $DATA itself is owned+writable by the service # user, so staging lives in a SIBLING dir whose PARENT (/var/lib) is root-owned, # meaning the service user can neither write into it nor rename/substitute it. # ensure_staging_dir() (run as root at build/activate) creates+locks it 0700. STAGING_DIR=/var/lib/hapx-ui-staging STAGED="$STAGING_DIR/hapx-ui.staged" SHAFILE="$STAGING_DIR/hapx-ui.sha256" # root-owned, verified hash of $STAGED LEGACY_STAGED="$DATA/hapx-ui.staged" # pre-hardening location, cleaned up # ensure_staging_dir creates the root-only staging dir and hard-resets its owner # and mode every time (self-healing: repairs a dir a prior version left in $DATA # or with loose perms). Only ever called from root code paths. ensure_staging_dir() { mkdir -p "$STAGING_DIR" 2>/dev/null || true chown root:root "$STAGING_DIR" 2>/dev/null || true chmod 0700 "$STAGING_DIR" 2>/dev/null || true # Remove any legacy staged artifact from the service-user-writable data dir so # a stale/attacker-planted one can never be picked up. rm -f "$LEGACY_STAGED" "$LEGACY_STAGED.meta" 2>/dev/null || true } # ── Trust-anchor guard ─────────────────────────────────────────────────────── # Opt-outs that WEAKEN the release-signature trust anchor (accept unsigned # releases, or build unverified code from source) must live in a ROOT-OWNED # file. $DATA is writable by the (unprivileged) service user, so a marker there # could be forged by a compromised service user to escalate to root — this # helper rejects any marker not owned by uid 0. is_root_marker() { [ -f "$1" ] || return 1 _o="$(stat -c %u "$1" 2>/dev/null)" || return 1 [ "$_o" = 0 ] } # allow_unsigned_enabled: true only if a root-owned opt-out exists. Prefers the # new /etc path; still honours the legacy $DATA path but ONLY when root-owned # (i.e. created via `sudo touch`, not by the service user). allow_unsigned_enabled() { is_root_marker "$ETC/allow-unsigned" || is_root_marker "$DATA/update-allow-unsigned" } # ── Release source (the internal Gitea server). Asset names match # .gitea/workflows/release.yml. Defaults below; the web UI / installer can # override the full repo URL via $DATA/update-server (e.g. # https://gitea.itm-technologies.de/nepomuk.gail/hapx-ui-releases). # The release assets live in a PUBLIC personal repo (nepomuk.gail/hapx-ui-releases) # so anonymous, tokenless pulls work — the ITMGmbH org is "limited" visibility and # blocks anonymous access. Keyless by default; the ed25519 signature still gates trust. UPDATE_SERVER="https://gitea.itm-technologies.de" REPO_SLUG="nepomuk.gail/hapx-ui-releases" _cfg_url="$(head -1 "$DATA/update-server" 2>/dev/null | tr -d '[:space:]')" # $DATA/update-server is writable by the (untrusted) service user, and we run as # root and feed this value to curl. Reject anything that is not a clean http(s) # URL — in particular a value starting with '-' would be read by curl as an # OPTION (e.g. -K = read a curl config → arbitrary file write as root). # On any violation we ignore the file and fall back to the compiled default. case "$_cfg_url" in http://*|https://*) case "$_cfg_url" in *[!A-Za-z0-9.:/_-]*) echo "[build] WARNUNG: update-server enthält ungültige Zeichen — ignoriert, Standard verwendet." >&2 _cfg_url="" ;; esac ;; "") : ;; *) echo "[build] WARNUNG: update-server ist keine http(s)-URL — ignoriert, Standard verwendet." >&2 _cfg_url="" ;; esac if [ -n "$_cfg_url" ]; then _cfg_url="${_cfg_url%.git}"; _cfg_url="${_cfg_url%/}" _name="${_cfg_url##*/}"; _rest="${_cfg_url%/*}" _owner="${_rest##*/}" if [ -n "$_name" ] && [ -n "$_owner" ] && [ "$_rest" != "$_cfg_url" ]; then REPO_SLUG="$_owner/$_name" UPDATE_SERVER="${_cfg_url%/$REPO_SLUG}" fi fi REPO_URL_DEFAULT="$UPDATE_SERVER/$REPO_SLUG.git" RELEASE_API="$UPDATE_SERVER/api/v1/repos/$REPO_SLUG" # Access token for the private Gitea repo. Saved by the web UI (Update page) # as $DATA/update-token (0600, service user); we run as root and read it here. # Works on both the Gitea API and web routes (raw, releases/download). UPDATE_TOKEN="$(head -1 "$DATA/update-token" 2>/dev/null | tr -d '[:space:]')" # Transport note: the release signature (mandatory, see verify_release_signature) # guarantees binary INTEGRITY even over plaintext. But on http:// the Gitea token # travels in cleartext and is sniffable on the LAN — prefer HTTPS for the # update-server URL. We warn rather than abort so the documented internal-Gitea # (HTTP) setup keeps working. case "$UPDATE_SERVER" in http://*) [ -n "$UPDATE_TOKEN" ] && echo "[build] WARNUNG: Update-Server ist HTTP — Token wird im Klartext übertragen. HTTPS empfohlen." ;; esac # gcurl: curl against the update server, with the token attached when present. # The token is fed through a --config file on stdin, NOT on the argv, so it never # appears in `ps`/ /proc//cmdline. `--` guards against a URL that begins # with '-' being mis-read as an option. gcurl() { if [ -n "$UPDATE_TOKEN" ]; then printf 'header = "Authorization: token %s"\n' "$UPDATE_TOKEN" | curl --config - "$@" else curl "$@" fi } # resolve_release_tag prints the tag of the current "latest" release (e.g. # build-1997de1). Each build publishes a UNIQUE tag, so the asset download URLs # are immutable and never served stale — which is the failure mode a reused # "latest" tag suffers from. resolve_release_tag() { gcurl -fsSL --connect-timeout 10 -m 20 \ "$RELEASE_API/releases/latest" 2>/dev/null \ | grep -oE '"tag_name"[[:space:]]*:[[:space:]]*"[^"]*"' | head -1 \ | sed -E 's/.*"tag_name"[[:space:]]*:[[:space:]]*"([^"]+)".*/\1/' } # ── Defaults; overridden by auto-detection from the running systemd unit. SERVICE=hapx-ui TARGET=/usr/local/bin/hapx-ui DB_PATH="$DATA/hapx.db" HEALTH_URL="" # filled by detect_env # ── Build-fallback toolchain seeds (only used if download fails) GO_BIN=/home/itm/go-install/go/bin/go export GOPATH="$DATA/go" GOCACHE="$DATA/go-build" GOMODCACHE="$DATA/go/pkg/mod" export GOTOOLCHAIN=auto HOME="$DATA" state() { # $STATE lives in the service-writable data dir, so drop any symlink an # attacker may have planted before root writes THROUGH it (arbitrary-write). if [ -L "$STATE.tmp" ]; then rm -f "$STATE.tmp"; fi if [ -L "$STATE" ]; then rm -f "$STATE"; fi printf '{"phase":"%s","version":"%s","commit":"%s"}' "$1" "${2:-}" "${3:-}" >"$STATE.tmp" \ && mv "$STATE.tmp" "$STATE" && chmod 0644 "$STATE" 2>/dev/null || true # Owned by the service user (= the data dir owner) so the web UI, which runs # unprivileged, can reset log/state when it triggers the next update. chown --reference="$DATA" "$STATE" 2>/dev/null || true } # Serialize build/activate so two clicks can't race. Steals a stale lock (>30m). acquire_lock() { if ! mkdir "$LOCK" 2>/dev/null; then if [ -d "$LOCK" ]; then age=$(( $(date +%s) - $(stat -c %Y "$LOCK" 2>/dev/null || echo 0) )) if [ "$age" -gt 1800 ]; then rmdir "$LOCK" 2>/dev/null || true mkdir "$LOCK" 2>/dev/null || { echo "[lock] konnte Lock nicht übernehmen"; exit 3; } else echo "[lock] Ein anderes Update läuft bereits (seit ${age}s). Abbruch."; exit 3 fi fi fi trap 'rmdir "$LOCK" 2>/dev/null || true' EXIT INT TERM } resolve_go() { for g in "$GO_BIN" /usr/local/go/bin/go /opt/go/bin/go /usr/bin/go \ /home/itm/go-install/go/bin/go "$(command -v go 2>/dev/null || true)"; do [ -n "$g" ] && [ -x "$g" ] && { echo "$g"; return; } done echo "$GO_BIN" } # Updates come ALWAYS from the internal Gitea server — even when an existing # working copy still has an old (e.g. GitHub) origin, build_from_source # re-points it via `git remote set-url origin`. The URL is token-FREE — the # token is supplied per-invocation via `git -c http.extraHeader=...` (see # git_auth), so it never lands in the build log, in `.git/config`, or in a # git error message echoing the remote. resolve_url() { echo "$REPO_URL_DEFAULT" } # git_auth runs git with the access token attached as an HTTP header (when set), # so the secret never appears in the URL/remote/config or in any logged output. # The header is passed via GIT_CONFIG_* env vars rather than `-c http.extraHeader=` # on the argv, so the token is not visible in /proc//cmdline. git_auth() { if [ -n "$UPDATE_TOKEN" ]; then GIT_CONFIG_COUNT=1 \ GIT_CONFIG_KEY_0=http.extraHeader \ GIT_CONFIG_VALUE_0="Authorization: token $UPDATE_TOKEN" \ git "$@" else git "$@" fi } # Map uname -m → Go arch used in the release asset names. go_arch() { case "$(uname -m)" in x86_64|amd64) echo amd64 ;; aarch64|arm64) echo arm64 ;; *) echo "" ;; esac } # Auto-detect target binary, db path and health URL from the systemd unit, so # nothing is hard-coded to one install layout. detect_env() { unit="$(systemctl show -p FragmentPath --value "$SERVICE" 2>/dev/null || true)" exec_line="" if [ -n "$unit" ] && [ -r "$unit" ]; then # Join the (possibly backslash-continued) ExecStart= line into one string. exec_line="$(awk '/^ExecStart=/{f=1} f{print} f&&!/\\$/{exit}' "$unit" \ | sed -E 's/^ExecStart=//' | tr -d '\\' | tr '\n' ' ')" fi if [ -n "$exec_line" ]; then t="$(printf '%s' "$exec_line" | awk '{print $1}')" [ -x "$t" ] && TARGET="$t" d="$(printf '%s' "$exec_line" | sed -nE 's/.*--db[= ]+([^ ]+).*/\1/p')" [ -n "$d" ] && DB_PATH="$d" addr="$(printf '%s' "$exec_line" | sed -nE 's/.*--addr[= ]+([^ ]+).*/\1/p')" tlsc="$(printf '%s' "$exec_line" | sed -nE 's/.*--tls-cert[= ]+([^ ]+).*/\1/p')" # /healthz only reports version+commit to an authorised caller. With # --metrics-token configured, an unauthenticated probe gets a bare # {"status":"ok"} — the commit gate in wait_health would then never match and # would roll back a perfectly healthy update. Pick the token up here so the # probe can authenticate (see hcurl). mt="$(printf '%s' "$exec_line" | sed -nE 's/.*--metrics-token[= ]+([^ ]+).*/\1/p')" [ -n "$mt" ] && METRICS_TOKEN="$mt" port="$(printf '%s' "$addr" | sed -E 's/^.*://')" [ -z "$port" ] && port=8443 if [ -n "$tlsc" ]; then HEALTH_URL="https://127.0.0.1:$port/healthz" else HEALTH_URL="http://127.0.0.1:$port/healthz"; fi fi } # hcurl: curl against a LOCAL health endpoint, attaching the metrics token when the # unit configures one. Like gcurl, the token goes through a --config file on stdin # rather than argv, so it never shows up in `ps` / /proc//cmdline. hcurl() { if [ -n "${METRICS_TOKEN:-}" ]; then printf 'header = "Authorization: Bearer %s"\n' "$METRICS_TOKEN" | curl --config - "$@" else curl "$@" fi } # MainPID of the managed service ("0" when stopped). Used as a fallback proof that # a restart really replaced the process when /healthz withholds the commit. service_mainpid() { systemctl show -p MainPID --value "$SERVICE" 2>/dev/null; } # Returns 0 if the URL answers 2xx within the timeout (curl -k for self-signed). health_ok() { hcurl -ksS -m 3 -o /dev/null -w '%{http_code}' "$1" 2>/dev/null | grep -q '^2'; } # Poll the (detected + fallback) health URLs for up to ~30s. If we know which # commit we expect, require it in the body so a stale process can't fake "up". wait_health() { expect="${1:-}" # "unknown" is what the provenance sidecar carries when the staged binary's # version could not be read. It can never appear in a /healthz body, so treating # it as an expectation would burn the full 90s and then roll back a healthy # update. Treat it as "no expectation" instead. [ "$expect" = unknown ] && expect="" urls="$HEALTH_URL https://127.0.0.1:8443/healthz http://127.0.0.1:8080/healthz http://127.0.0.1:8080/login" i=0 # 90s, not 30: on a large prod DB the new binary runs migrations + an initial # HAProxy sync + cert checks before it serves /healthz. A 30s ceiling tripped # a *spurious* rollback (the binary was fine, just slow) which then swapped the # live DB file and corrupted it — see the rollback note below. while [ "$i" -lt 90 ]; do for u in $urls; do [ -z "$u" ] && continue body="$(hcurl -ksS -m 3 "$u" 2>/dev/null)" || continue code="$(hcurl -ksS -m 3 -o /dev/null -w '%{http_code}' "$u" 2>/dev/null)" case "$code" in 2*|3*) if [ -z "$expect" ] || printf '%s' "$body" | grep -q "$expect"; then return 0; fi # The expected commit is absent. If the body carries a commit field at all, # it belongs to a DIFFERENT build — the old process is still serving, so # keep waiting (this is exactly the staleness the gate exists for). If it # carries none (e.g. a metrics token is configured and we could not read it # from the unit), fall back to proving the process was actually replaced: # systemd reports a new MainPID and the service answers 2xx. case "$body" in *'"commit"'*) : ;; *) if [ -n "${OLD_MAINPID:-}" ] && [ "$(service_mainpid)" != "${OLD_MAINPID}" ]; then return 0 fi ;; esac esac done i=$((i+1)); sleep 1 done return 1 } # haproxy_ok: the update must not break the proxy. Require a valid config AND an # active haproxy. Missing haproxy binary/unit (non-appliance dev) is tolerated. haproxy_ok() { command -v haproxy >/dev/null 2>&1 || return 0 [ -f /etc/haproxy/haproxy.cfg ] || return 0 haproxy -c -f /etc/haproxy/haproxy.cfg >/dev/null 2>&1 || { echo "[activate] HAProxy-Config ungültig"; return 1; } systemctl is-active haproxy >/dev/null 2>&1 || { echo "[activate] HAProxy nicht aktiv"; return 1; } return 0 } # Keep only the newest $KEEP backups/restore-points of each kind. prune_backups() { ls -1dt "$TARGET".bak-* 2>/dev/null | tail -n +$((KEEP+1)) | while read -r f; do rm -f "$f"; done ls -1dt "$BACKUP_DIR"/db-* 2>/dev/null | tail -n +$((KEEP+1)) | while read -r d; do rm -rf "$d"; done ls -1dt "$STAGING_DIR"/restore-* 2>/dev/null | tail -n +$((KEEP+1)) | while read -r d; do rm -rf "$d"; done } # ── Primary build path: download a verified prebuilt binary ─────────────────── # Embedded ed25519 public key the release SHA256SUMS is signed with. The matching # private key is the CI secret RELEASE_SIGNING_KEY. Verifying here (in the trusted # helper, via system openssl) proves provenance before we trust the checksums — # without relying on the old/new binary supporting a verify subcommand. RELEASE_PUBKEY='-----BEGIN PUBLIC KEY----- MCowBQYDK2VwAyEAnbrMEw7Akn0JF5f+x8UlUnphkS+0JzFSMNzvo9W7mPE= -----END PUBLIC KEY-----' # verify_release_signature # 0 = signature valid · 1 = present but INVALID (abort) · 2 = cannot verify (skip) verify_release_signature() { command -v openssl >/dev/null 2>&1 || { echo "[build] openssl fehlt — Signatur nicht prüfbar"; return 2; } pub="$STAGED.pub" printf '%s\n' "$RELEASE_PUBKEY" > "$pub" out="$(openssl pkeyutl -verify -pubin -inkey "$pub" -rawin -in "$1" -sigfile "$2" 2>&1)" rm -f "$pub" case "$out" in *"Signature Verified"*) return 0 ;; *rawin*|*"nknown option"*|*"nrecognized"*) echo "[build] openssl zu alt für ed25519 (-rawin) — Signatur übersprungen"; return 2 ;; *) return 1 ;; esac } # fetch_verified # Downloads the named asset from the current release and verifies it against # the ed25519-SIGNED SHA256SUMS (same policy as the binary download). Returns 0 # only on a verified checksum match. Used to self-update the helper script from # the SIGNED release rather than an unverified raw/branch fetch. fetch_verified() { _name="$1"; _dest="$2" _tag="$(resolve_release_tag)"; [ -z "$_tag" ] && return 1 _base="$UPDATE_SERVER/$REPO_SLUG/releases/download/$_tag" _tmp="$_dest.dl.$$"; _sums="$STAGED.vsums.$$"; _sig="$STAGED.vsig.$$" gcurl -fsL --retry 2 --connect-timeout 10 -m 60 -o "$_tmp" "$_base/$_name" 2>/dev/null || { rm -f "$_tmp"; return 1; } gcurl -fsL --retry 2 --connect-timeout 10 -m 30 -o "$_sums" "$_base/SHA256SUMS" 2>/dev/null || { rm -f "$_tmp" "$_sums"; return 1; } _allow=0; allow_unsigned_enabled && _allow=1 if gcurl -fsL --retry 2 --connect-timeout 10 -m 30 -o "$_sig" "$_base/SHA256SUMS.sig" 2>/dev/null; then verify_release_signature "$_sums" "$_sig"; _vrc=$?; rm -f "$_sig" if [ "$_vrc" = 1 ]; then rm -f "$_tmp" "$_sums"; return 1; fi # tamper → abort if [ "$_vrc" != 0 ] && [ "$_allow" != 1 ]; then rm -f "$_tmp" "$_sums"; return 1; fi elif [ "$_allow" != 1 ]; then rm -f "$_tmp" "$_sums"; return 1 # no sig + not opted-in → abort fi _want="$(grep -E "[ *]$_name\$" "$_sums" | awk '{print $1}' | head -1)"; rm -f "$_sums" _got="$(sha256sum "$_tmp" 2>/dev/null | awk '{print $1}')" if [ -n "$_want" ] && [ "$_want" = "$_got" ]; then mv "$_tmp" "$_dest"; return 0; fi rm -f "$_tmp"; return 1 } download_staged() { arch="$(go_arch)" if [ -z "$arch" ]; then echo "[build] Architektur $(uname -m) hat kein Release-Asset"; return 1; fi asset="hapx-ui-linux-$arch" tag="$(resolve_release_tag)" if [ -z "$tag" ]; then echo "[build] Konnte aktuelles Release nicht auflösen (Gitea-API)"; return 1; fi base="$UPDATE_SERVER/$REPO_SLUG/releases/download/$tag" echo "[build] Aktuelles Release: $tag" # Unique per-build tag → immutable URLs. The retry loop is belt-and-suspenders # for any rare inconsistency. i=1 while [ "$i" -le 3 ]; do echo "[build] Lade vorkompiliertes Binary: $asset (Versuch $i)" if gcurl -fL --retry 3 --connect-timeout 10 -m 180 -o "$STAGED.dl" "$base/$asset" 2>&1 \ && gcurl -fL --retry 3 --connect-timeout 10 -m 30 -o "$STAGED.sums" "$base/SHA256SUMS" 2>&1; then # Provenance: the ed25519 signature of SHA256SUMS is verified against the # pinned public key (RELEASE_PUBKEY). A valid signature proves the release # came from CI even over plaintext transport, so a network MITM cannot forge # a binary + matching SHA256SUMS. # # Policy: # - present & VALID → proceed # - present & INVALID (rc=1) → ABORT always (tamper signal; no opt-out) # - MISSING or unverifiable → ABORT, UNLESS the operator has explicitly # opted into unsigned releases by creating $DATA/update-allow-unsigned # (transition switch for deployments whose CI does not sign yet — # RELEASE_SIGNING_KEY unset). Secure-by-default, but recoverable without # hand-editing this script. allow_unsigned=0 allow_unsigned_enabled && allow_unsigned=1 if gcurl -fsL --retry 2 --connect-timeout 10 -m 30 -o "$STAGED.sig" "$base/SHA256SUMS.sig" 2>/dev/null; then verify_release_signature "$STAGED.sums" "$STAGED.sig"; vrc=$? rm -f "$STAGED.sig" if [ "$vrc" = 0 ]; then echo "[build] Signatur verifiziert (ed25519)" elif [ "$vrc" = 1 ]; then echo "[build] SIGNATUR UNGÜLTIG — Abbruch (möglicher Manipulationsversuch)" rm -f "$STAGED.dl" "$STAGED.sums"; return 1 elif [ "$allow_unsigned" = 1 ]; then echo "[build] WARNUNG: Signatur nicht prüfbar (rc=$vrc) — durch update-allow-unsigned erlaubt." echo "[build] WARNUNG: OHNE Signatur bietet die Prüfsumme KEINEN Schutz — Binary UND SHA256SUMS" echo "[build] stammen beide vom (frei konfigurierbaren) Update-Server. Nur für vertrauenswürdige Quellen!" else echo "[build] SIGNATUR NICHT PRÜFBAR (rc=$vrc) — Abbruch (openssl zu alt für ed25519)" rm -f "$STAGED.dl" "$STAGED.sums"; return 1 fi elif [ "$allow_unsigned" = 1 ]; then echo "[build] WARNUNG: keine Signatur im Release — durch update-allow-unsigned erlaubt." echo "[build] WARNUNG: OHNE Signatur bietet die Prüfsumme KEINEN Schutz — Binary UND SHA256SUMS" echo "[build] stammen beide vom (frei konfigurierbaren) Update-Server. Nur für vertrauenswürdige Quellen!" else echo "[build] SIGNATUR FEHLT — Abbruch. Releases signieren (RELEASE_SIGNING_KEY im CI) ODER" echo "[build] zum Übergang (als root): 'sudo mkdir -p $ETC && sudo touch $ETC/allow-unsigned'." rm -f "$STAGED.dl" "$STAGED.sums"; return 1 fi want="$(grep -E "[ *]$asset\$" "$STAGED.sums" | awk '{print $1}' | head -1)" got="$(sha256sum "$STAGED.dl" | awk '{print $1}')" rm -f "$STAGED.sums" if [ -n "$want" ] && [ "$want" = "$got" ]; then echo "[build] SHA256 verifiziert: $got" chmod 0755 "$STAGED.dl"; mv "$STAGED.dl" "$STAGED"; return 0 fi echo "[build] Prüfsumme passt noch nicht (erwartet=$want erhalten=$got) — neuer Versuch ..." else echo "[build] Download fehlgeschlagen — neuer Versuch ..." fi rm -f "$STAGED.dl" "$STAGED.sums" i=$((i+1)); [ "$i" -le 3 ] && sleep 5 done echo "[build] Download nach mehreren Versuchen nicht konsistent" return 1 } # ── Fallback build path: clone + compile in a writable working copy ─────────── build_from_source() { # SECURITY: a source build produces an UNVERIFIED binary (no release signature # can cover locally-compiled code), and the clone URL is derived from the # service-user-writable $DATA/update-server. Executing/installing that as root # would let a compromised service user (or a malicious repo) escalate to root, # bypassing the ed25519 trust anchor entirely. So the fallback is DISABLED by # default and only runs when a root-owned marker explicitly opts in. if ! is_root_marker "$ETC/allow-source-build"; then echo "[build] Quelltext-Build ist deaktiviert — es wird nur ein signiertes Release installiert." echo "[build] Kein passendes signiertes Binary verfügbar (Download fehlgeschlagen)." echo "[build] Falls ein Quelltext-Build wirklich nötig ist, als root freischalten:" echo "[build] sudo mkdir -p $ETC && sudo touch $ETC/allow-source-build" return 1 fi GO_BIN="$(resolve_go)" RURL="$(resolve_url)" echo "[build] Fallback: Build aus Quelltext (per $ETC/allow-source-build freigeschaltet)" echo "[build] Arbeitskopie: $WORK (beschreibbar)" echo "[build] Repository : $RURL" if [ ! -x "$GO_BIN" ]; then echo "[build] FEHLER: kein Go-Compiler gefunden ($GO_BIN)"; return 1; fi echo "[build] Go : $("$GO_BIN" version 2>/dev/null)" mkdir -p "$WORK" 2>/dev/null || true if [ ! -d "$WORK/.git" ]; then echo "[build] Lege beschreibbare Arbeitskopie an (git clone) ..." rm -rf "$WORK" 2>/dev/null || true git_auth -c safe.directory='*' clone "$RURL" "$WORK" 2>&1 || { echo "[build] FEHLER: git clone fehlgeschlagen"; return 1; } fi cd "$WORK" || { echo "[build] FEHLER: Arbeitskopie fehlt"; return 1; } git -c safe.directory='*' remote set-url origin "$RURL" 2>/dev/null || true echo "[build] git fetch origin main ..." git_auth -c safe.directory='*' fetch --quiet origin main 2>&1 || { echo "[build] FEHLER: git fetch fehlgeschlagen"; return 1; } git -c safe.directory='*' reset --hard origin/main 2>&1 VERSION="$(git -c safe.directory='*' describe --tags --always 2>/dev/null || echo dev)" COMMIT="$(git -c safe.directory='*' rev-parse --short HEAD)" echo "[build] Version: $VERSION Commit: $COMMIT" echo "[build] Kompiliere (Dienst läuft weiter; erster Lauf lädt ggf. die Go-Toolchain) ..." if ! "$GO_BIN" build -buildvcs=false \ -ldflags "-s -w -X main.version=$VERSION -X main.commit=$COMMIT -X main.sourceDir=$WORK" \ -o "$STAGED.tmp" ./cmd/hapxui 2>&1; then echo "[build] FEHLER: go build fehlgeschlagen"; rm -f "$STAGED.tmp"; return 1 fi chmod 0755 "$STAGED.tmp"; mv "$STAGED.tmp" "$STAGED" return 0 } # Read "hapx-ui ()" from a binary; echoes "ver commit". # Hard-timeout the exec: a corrupt or misbehaving staged binary must never be # able to hang the updater (which would also hold the lock indefinitely). On # timeout/failure we just return empty — callers treat that as "unknown". binary_version() { if command -v timeout >/dev/null 2>&1; then timeout 5 "$1" --version 2>/dev/null else "$1" --version 2>/dev/null; fi | sed -nE 's/^hapx-ui (.+) \((.+)\)$/\1 \2/p' } # ── Auto-update gating (used only by the `auto` verb) ───────────────────────── AUTOCONF="$DATA/autoupdate.json" # {enabled, window_from, window_to} FAILED_VERSIONS="$STAGING_DIR/failed-versions" # root-only crashloop guard state_phase() { sed -n 's/.*"phase"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$STATE" 2>/dev/null | head -1; } # Default is ON — only an explicit false/0/off disables auto-update. autoupdate_enabled() { [ -f "$AUTOCONF" ] || return 0 _v="$(sed -n 's/.*"enabled"[[:space:]]*:[[:space:]]*\([A-Za-z0-9]*\).*/\1/p' "$AUTOCONF" 2>/dev/null | head -1)" case "$_v" in false|False|0|off) return 1 ;; *) return 0 ;; esac } # If a maintenance window "HH:MM"/"HH:MM" is configured, return 0 only inside it # (wrap-around past midnight supported). No window ⇒ always allowed. in_window() { [ -f "$AUTOCONF" ] || return 0 _f="$(sed -n 's/.*"window_from"[[:space:]]*:[[:space:]]*"\([0-9:]*\)".*/\1/p' "$AUTOCONF" 2>/dev/null | head -1)" _t="$(sed -n 's/.*"window_to"[[:space:]]*:[[:space:]]*"\([0-9:]*\)".*/\1/p' "$AUTOCONF" 2>/dev/null | head -1)" [ -n "$_f" ] && [ -n "$_t" ] || return 0 # Strip leading zeros so POSIX arithmetic doesn't read "0930" as octal. _now=$(date +%H%M | sed 's/^0*//'); _fn=$(printf '%s' "$_f" | tr -dc 0-9 | sed 's/^0*//'); _tn=$(printf '%s' "$_t" | tr -dc 0-9 | sed 's/^0*//') : "${_now:=0}"; : "${_fn:=0}"; : "${_tn:=0}" if [ "$_fn" -gt "$_tn" ]; then { [ "$_now" -ge "$_fn" ] || [ "$_now" -lt "$_tn" ]; }; else [ "$_now" -ge "$_fn" ] && [ "$_now" -lt "$_tn" ]; fi } # auto_status — record the outcome of an unattended run where the # SERVICE can read it. The crashloop marker and the staging tree are root-only # (0700), so without this the web UI could never explain why auto-update went # quiet — the operator would just see nothing happening. Best-effort: a failure to # write must never abort an update. # result: disabled | window | unreachable | current | skipped | updated | failed auto_status() { [ -d "$DATA" ] || return 0 printf '{"checked_at":"%s","running":"%s","available":"%s","result":"%s","detail":"%s"}\n' \ "$(date -u +%FT%TZ)" "${cur_ver:-}" "${tag:-}" "$1" "${2:-}" \ > "$DATA/autoupdate-status.json" 2>/dev/null || return 0 chmod 0644 "$DATA/autoupdate-status.json" 2>/dev/null || true chown --reference="$DATA" "$DATA/autoupdate-status.json" 2>/dev/null || true } is_failed_version() { [ -f "$FAILED_VERSIONS" ] && grep -qxF "$1" "$FAILED_VERSIONS"; } mark_failed_version(){ mkdir -p "$STAGING_DIR" 2>/dev/null; printf '%s\n' "$1" >> "$FAILED_VERSIONS"; chmod 0600 "$FAILED_VERSIONS" 2>/dev/null || true; } # running_version prints " " of the INSTALLED binary by reusing # binary_version (which parses `hapx-ui ()` and hard-timeouts the exec). # # The previous implementation grepped for a 'build-' prefix that only ever appears # in the release TAG, never in --version output (CI sets version=commit=, # see .gitea/workflows/release.yml). It therefore ALWAYS returned empty, which made # the "already current" guard in the auto verb unreachable: every timer tick # re-downloaded and re-activated the very same release — a permanent restart loop. running_version() { binary_version "$TARGET"; } # is_current # 0 = the installed build already IS the release named by . # Mirrors the Go poller (internal/api/update_webhook.go): strip the "build-" prefix # and prefix-match BOTH ways, so a short/long SHA difference can't fake an update. # A v* tag carries no SHA, so it is compared against the binary's VERSION field # instead (its commit is an unrelated SHA). is_current() { _tag="$1"; _ver="$2"; _cmt="$3" if [ -z "$_tag" ] || [ -z "$_cmt" ] || [ "$_cmt" = unknown ]; then return 1 # unknown state → never claim "current" fi if [ "$_tag" = "$_ver" ]; then return 0; fi # v1.2.3 == VERSION _sha="${_tag#build-}" case "$_sha" in "$_cmt"*) return 0 ;; esac case "$_cmt" in "$_sha"*) return 0 ;; esac return 1 } # Test hook: `HAPX_UPDATE_LIB=1 . hapx-ui-update.sh` defines the functions above and # returns instead of dispatching a verb, so test/update.bats can exercise the # version logic directly. Without the variable this is a no-op. [ -n "${HAPX_UPDATE_LIB:-}" ] && return 0 case "${1:-}" in preflight) arch="$(go_arch)" rel_ok=false if [ -n "$arch" ]; then tag="$(resolve_release_tag)" if [ -n "$tag" ] && gcurl -fsI --connect-timeout 8 -m 15 \ "$UPDATE_SERVER/$REPO_SLUG/releases/download/$tag/hapx-ui-linux-$arch" >/dev/null 2>&1; then rel_ok=true fi fi GO_BIN="$(resolve_go)"; go_ok=false; gv="" [ -x "$GO_BIN" ] && { go_ok=true; gv="$("$GO_BIN" version 2>/dev/null)"; } git_ok=false; command -v git >/dev/null 2>&1 && git_ok=true detect_env # ready if the prebuilt download is possible, OR a source build is possible # AND explicitly opted into via the root-owned marker (else the fallback is # disabled and would abort — reporting it as ready would be misleading). src_ok=false; { $go_ok && $git_ok && is_root_marker "$ETC/allow-source-build"; } && src_ok=true ready=false; { $rel_ok || $src_ok; } && ready=true commit=""; [ -x "$WORK/cmd/hapxui" ] && commit="$(git -C "$WORK" -c safe.directory='*' rev-parse --short HEAD 2>/dev/null || true)" printf '{"method":"%s","releaseOk":%s,"arch":"%s","sourceDir":"%s","goBin":"%s","goBinOk":%s,"goVersion":"%s","gitOk":%s,"target":"%s","dbPath":"%s","healthUrl":"%s","repoCommit":"%s","ready":%s}\n' \ "$([ "$rel_ok" = true ] && echo download || echo source)" \ "$rel_ok" "$arch" "$WORK" "$GO_BIN" "$go_ok" "$gv" "$git_ok" \ "$TARGET" "$DB_PATH" "$HEALTH_URL" "$commit" "$ready" ;; build) # Drop a planted symlink first — $LOG is in the service-writable data dir and # we are about to write to it as root (symlink-follow = arbitrary write). if [ -L "$LOG" ]; then rm -f "$LOG"; fi exec >"$LOG" 2>&1 chown --reference="$DATA" "$LOG" 2>/dev/null || true chmod 0644 "$LOG" 2>/dev/null || true acquire_lock state building mkdir -p "$DATA" 2>/dev/null || true ensure_staging_dir rm -f "$STAGED" "$STAGED.meta" "$SHAFILE" 2>/dev/null || true echo "[build] gestartet: $(date -u +%FT%TZ)" echo "[build] Architektur: $(uname -m) → $(go_arch)" method="download" if ! download_staged; then echo "[build] ------------------------------------------------------------" method="source" if ! build_from_source; then echo "[build] FEHLER: weder Download noch Quelltext-Build erfolgreich"; state build_failed; exit 1 fi fi # Provenance for version/commit. NEVER execute the staged binary as root to # read its version when it came from an unverified source build — that would # hand root to attacker-controlled code. The verified-download binary is # signature-checked CI output, so reading it is safe; the source build # reports the git-derived VERSION/COMMIT set inside build_from_source. if [ "$method" = "download" ]; then set -- $(binary_version "$STAGED") VERSION="${1:-dev}"; COMMIT="${2:-unknown}" fi VERSION="${VERSION:-dev}"; COMMIT="${COMMIT:-unknown}" # Persist provenance so `activate` never has to exec the staged binary. printf '%s %s\n' "$VERSION" "$COMMIT" > "$STAGED.meta" 2>/dev/null || true chmod 0644 "$STAGED.meta" 2>/dev/null || true # Record the staged binary's hash in a root-owned file (staging dir is 0700 # root:root). `activate` re-hashes $STAGED and compares against this before # installing — a second layer under the root-only staging dir that ties the # thing we verified/built to the exact bytes we install. sha256sum "$STAGED" 2>/dev/null | awk '{print $1}' > "$SHAFILE" 2>/dev/null || true chmod 0600 "$SHAFILE" 2>/dev/null || true state built "$VERSION" "$COMMIT" echo "[build] ============================================================" echo "[build] OK Bereit ($method) — $VERSION ($COMMIT)" echo "[build] Klicke »Neu starten & aktivieren«, um zu übernehmen." echo "[build] (Der Dienst ist bis dahin unverändert online.)" ;; activate) if [ -L "$LOG" ]; then rm -f "$LOG"; fi exec >>"$LOG" 2>&1 chown --reference="$DATA" "$LOG" 2>/dev/null || true acquire_lock echo "" echo "[activate] $(date -u +%FT%TZ) — aktiviere neues Binary ..." state activating ensure_staging_dir if [ ! -x "$STAGED" ]; then echo "[activate] FEHLER: kein gebautes/geladenes Binary vorhanden"; state activate_failed; exit 1; fi # Trust gate: re-hash the staged binary and compare against the root-owned # hash recorded at build time. The staging dir is root-only, so the service # user cannot have swapped $STAGED; this check ALSO fails closed if the file # was produced without a recorded hash (e.g. hand-dropped) — we never install # a staged binary we didn't verify/build ourselves. Closes the signing-anchor # bypass where a staged binary could be installed over the system binary # without re-verification. if [ ! -f "$SHAFILE" ]; then echo "[activate] FEHLER: kein Integritäts-Hash zum Staging-Binary (Abbruch — nicht über den Updater erzeugt)"; state activate_failed; exit 1 fi _want_sha="$(cat "$SHAFILE" 2>/dev/null)" _got_sha="$(sha256sum "$STAGED" 2>/dev/null | awk '{print $1}')" if [ -z "$_want_sha" ] || [ "$_want_sha" != "$_got_sha" ]; then echo "[activate] FEHLER: Staging-Binary stimmt nicht mit dem verifizierten Hash überein (erwartet=$_want_sha erhalten=$_got_sha) — Abbruch"; state activate_failed; exit 1 fi echo "[activate] Integrität ok (SHA256 $_got_sha)" detect_env echo "[activate] Ziel-Binary: $TARGET Dienst: $SERVICE" echo "[activate] DB: $DB_PATH Health: ${HEALTH_URL:-}" # Which commit must be live afterwards (defends against a stale process). # Read provenance from the sidecar written at build time — we must NOT exec # the staged binary as root here, since it may be an unverified source build. NEW_COMMIT="" if [ -f "$STAGED.meta" ]; then set -- $(cat "$STAGED.meta" 2>/dev/null); NEW_COMMIT="${2:-}"; fi # 1) COMPLETE, VERIFIED restore point BEFORE touching anything. Lives in the # root-only staging tree; holds the previous binary + a full backup (DB via # VACUUM INTO + secret.key + haproxy.cfg + crt-list + certs) produced by the # CURRENTLY-RUNNING binary while the service is still up (online, consistent). # A failure here is FATAL — we never swap the binary without a verified way back. ts="$(date +%Y%m%d-%H%M%S)" RP="$STAGING_DIR/restore-$ts" mkdir -p "$RP" 2>/dev/null || { echo "[activate] FEHLER: Restore-Punkt-Verzeichnis"; state activate_failed; exit 1; } BACKUP="$RP/binary" # previous binary (for the fast binary-first rollback) RPZIP="$RP/backup.zip" # full state if ! cp -a "$TARGET" "$BACKUP"; then echo "[activate] FEHLER: Binary-Backup fehlgeschlagen — Abbruch (kein Update)"; state activate_failed; exit 1 fi OLD_SCHEMA=0 # The current binary provides the backup subcommand from this version on. On the # ONE bootstrap update from an older binary that lacks it, fall back to a # best-effort DB+binary snapshot so the first update still goes through. if "$TARGET" backup --schema-version --db "$DB_PATH" >/dev/null 2>&1; then echo "[activate] Erstelle vollständigen Restore-Punkt: $RP" if ! "$TARGET" backup --out "$RPZIP" --db "$DB_PATH"; then echo "[activate] FEHLER: Backup fehlgeschlagen — Abbruch (kein Update)"; state activate_failed; exit 1 fi if ! "$TARGET" backup --verify "$RPZIP"; then echo "[activate] FEHLER: Backup nicht verifizierbar — Abbruch (kein Update)"; state activate_failed; exit 1 fi OLD_SCHEMA="$("$TARGET" backup --schema-version --db "$DB_PATH" 2>/dev/null | tr -dc '0-9')" [ -n "$OLD_SCHEMA" ] || OLD_SCHEMA=0 echo "[activate] Restore-Punkt verifiziert (Schema $OLD_SCHEMA)." else echo "[activate] Hinweis: aktuelles Binary kennt 'backup' noch nicht — Bootstrap-Snapshot (DB roh)." systemctl stop "$SERVICE" 2>/dev/null || true _dbname="$(basename "$DB_PATH")" if command -v sqlite3 >/dev/null 2>&1 && sqlite3 "$DB_PATH" ".backup '$RP/$_dbname'" 2>/dev/null && [ -s "$RP/$_dbname" ]; then echo "[activate] DB-Bootstrap-Backup (sqlite3 .backup)" else for ext in "" "-wal" "-shm"; do [ -f "$DB_PATH$ext" ] && cp -a "$DB_PATH$ext" "$RP/" 2>/dev/null || true; done fi fi # 2) Stop the service so the binary swap + restart are clean. Remember the PID # first: if /healthz withholds the commit (metrics token configured), a changed # MainPID is what proves the restart actually replaced the old process. OLD_MAINPID="$(service_mainpid)" echo "[activate] Stoppe Dienst ..." systemctl stop "$SERVICE" 2>/dev/null || true # 3) Atomic swap on the SAME filesystem as TARGET (rename is atomic). newbin="$(dirname "$TARGET")/.hapx-ui.new.$ts" if ! cp -a "$STAGED" "$newbin" 2>/dev/null; then echo "[activate] FEHLER: konnte Staging nicht neben $TARGET kopieren"; systemctl start "$SERVICE" 2>/dev/null || true; state activate_failed; exit 1 fi chmod 0755 "$newbin" mv -f "$newbin" "$TARGET" # atomic rename over the old binary rm -f "$STAGED" "$STAGED.meta" "$SHAFILE" 2>/dev/null || true # 3b) Ensure the service is in the groups it needs, WITHOUT touching the main # unit (systemd merges SupplementaryGroups across drop-ins additively): # * haproxy — read the admin stats socket /run/haproxy/admin.sock (mode 660, # group haproxy). WITHOUT this the live host status / `show stat` fails with # "connect: permission denied" and EVERY host is stuck on WAITING. Older # appliance units lacked this group; the updater used to add only 'adm', # so the socket stayed inaccessible — this line is the fix, self-healing on # the next update. # * adm — read /var/log/haproxy.log for the Live-Logs viewer. dropin_dir="/etc/systemd/system/${SERVICE}.service.d" dropin="$dropin_dir/10-hapx-loggroup.conf" want='SupplementaryGroups=haproxy adm' if [ "$(grep -hs '^SupplementaryGroups=' "$dropin" 2>/dev/null)" != "$want" ]; then mkdir -p "$dropin_dir" 2>/dev/null || true if printf '[Service]\n%s\n' "$want" > "$dropin" 2>/dev/null; then echo "[activate] Gruppen 'haproxy'+'adm' ergänzt (Socket-Zugriff + Live-Logs)" systemctl daemon-reload 2>/dev/null || true fi fi # Belt-and-suspenders: also add the service user to the haproxy group in the # user DB (some setups init supplementary groups from /etc/group at start too). svc_user="$(systemctl show -p User --value "$SERVICE" 2>/dev/null)"; svc_user="${svc_user:-hapx-ui}" if getent group haproxy >/dev/null 2>&1 && ! id -nG "$svc_user" 2>/dev/null | tr ' ' '\n' | grep -qx haproxy; then usermod -aG haproxy "$svc_user" 2>/dev/null && echo "[activate] $svc_user zur Gruppe 'haproxy' hinzugefügt" || true fi # 3b2) Keep CAP_NET_RAW in the BOUNDING set only (never ambient). A traceroute # binary with cap_net_raw=ep file caps only gets the cap if it's in the # bounding set; ping needs no capability (net.ipv4.ping_group_range). Forcing # it ambient onto every child is unnecessary attack surface. This drop-in adds # it to the bounding set additively without touching ambient — and REWRITES an # older ambient version left by a previous update. Idempotent, self-healing. cap_dropin="$dropin_dir/11-hapx-netraw.conf" if ! grep -qs '^CapabilityBoundingSet=CAP_NET_RAW$' "$cap_dropin" || grep -qs 'AmbientCapabilities' "$cap_dropin"; then mkdir -p "$dropin_dir" 2>/dev/null || true if printf '[Service]\nCapabilityBoundingSet=CAP_NET_RAW\n' > "$cap_dropin" 2>/dev/null; then echo "[activate] CAP_NET_RAW auf Bounding-Set beschränkt (systemd Drop-in)" systemctl daemon-reload 2>/dev/null || true fi fi # 3c) Ensure the rolling config-archive dir is writable by the service. # On older installs /etc/haproxy/backups was created root:root, so the # non-root service couldn't write it ("config archive write: permission # denied"). Make it group 'haproxy', setgid + group-writable. Idempotent, # best-effort, self-healing across updates. bkdir="/etc/haproxy/backups" if getent group haproxy >/dev/null 2>&1; then mkdir -p "$bkdir" 2>/dev/null || true if [ -d "$bkdir" ]; then chgrp haproxy "$bkdir" 2>/dev/null || true chmod 2775 "$bkdir" 2>/dev/null || true echo "[activate] Config-Backup-Verzeichnis beschreibbar gemacht ($bkdir, Gruppe haproxy)" fi fi # 3d) Ensure the decoupled web-update path is installed so the UI buttons # work. The flow is split into two operator-gated steps, each its own oneshot # run as root in its OWN cgroup (survives the hapx-ui restart): # STEP 1 hapx-ui-update.service → build (download/verify, no restart) # STEP 2 hapx-ui-activate.service → activate (swap + restart + rollback) # The web service triggers each via `systemctl start --no-block`. Installing # both here is idempotent and self-healing — existing installs that still # have the old combined unit get the split on the next activate. install_unit() { # _u="$1"; _tmp="${_u}.tmp.$$" cat > "$_tmp" if [ -f "$_tmp" ]; then if ! cmp -s "$_tmp" "$_u" 2>/dev/null; then mv -f "$_tmp" "$_u"; chmod 0644 "$_u" echo "[activate] Update-Dienst installiert/aktualisiert ($_u)" UNIT_CHANGED=1 else rm -f "$_tmp" fi fi } UNIT_CHANGED=0 install_unit /etc/systemd/system/hapx-ui-update.service <<'UNIT' [Unit] Description=HAPX-UI self-update — STEP 1: download & verify (no restart) After=network-online.target Wants=network-online.target [Service] Type=oneshot ExecStart=/usr/local/sbin/hapx-ui-update build UNIT install_unit /etc/systemd/system/hapx-ui-activate.service <<'UNIT' [Unit] Description=HAPX-UI self-update — STEP 2: activate staged binary (restart + rollback) After=network-online.target Wants=network-online.target [Service] Type=oneshot ExecStart=/usr/local/sbin/hapx-ui-update activate UNIT [ "$UNIT_CHANGED" = 1 ] && systemctl daemon-reload 2>/dev/null || true upd_sudo="/etc/sudoers.d/hapx-ui-update" upd_sudo_tmp="${upd_sudo}.tmp.$$" cat > "$upd_sudo_tmp" <<'SUDO' hapx-ui ALL=(root) NOPASSWD: /usr/local/sbin/hapx-ui-update preflight, /usr/bin/systemctl start --no-block hapx-ui-update.service, /bin/systemctl start --no-block hapx-ui-update.service, /usr/bin/systemctl start --no-block hapx-ui-activate.service, /bin/systemctl start --no-block hapx-ui-activate.service SUDO if visudo -cf "$upd_sudo_tmp" >/dev/null 2>&1; then if ! cmp -s "$upd_sudo_tmp" "$upd_sudo" 2>/dev/null; then chown root:root "$upd_sudo_tmp" 2>/dev/null || true chmod 0440 "$upd_sudo_tmp" mv -f "$upd_sudo_tmp" "$upd_sudo" echo "[activate] Update-Berechtigung installiert/aktualisiert ($upd_sudo)" else rm -f "$upd_sudo_tmp" fi else echo "[activate] WARN: Update-sudoers-Validierung fehlgeschlagen — übersprungen" rm -f "$upd_sudo_tmp" fi # 3d2) CrowdSec setup wizard files (root helper + oneshot unit + sudoers). # These make the Security → CrowdSec one-click wizard work, but — unlike the # binary — they are NOT part of the update; they come from install.sh. Fetch # them verified from the SIGNED release (same trust anchor as the binary) and # install if missing or changed, so the wizard self-heals on every update, # even a prebuilt-binary update that never refreshes the source clone. # Non-fatal: a fetch failure just leaves the wizard unavailable (falls back to # the manual "Erweitert" form) exactly as before. echo "[activate] Prüfe CrowdSec-Wizard-Dateien ..." ensure_staging_dir cs_stage="$STAGING_DIR/crowdsec" mkdir -p "$cs_stage" 2>/dev/null || true CS_UNIT_CHANGED=0 cs_install() { # _a="$1"; _d="$2"; _m="$3"; _k="$4"; _t="$cs_stage/$_a" if ! fetch_verified "$_a" "$_t"; then echo "[activate] CrowdSec: $_a nicht (verifiziert) ladbar — übersprungen"; return 1 fi if [ "$_k" = sudo ] && ! visudo -cf "$_t" >/dev/null 2>&1; then echo "[activate] CrowdSec: $_a visudo-Validierung fehlgeschlagen — übersprungen"; rm -f "$_t"; return 1 fi if cmp -s "$_t" "$_d" 2>/dev/null; then rm -f "$_t"; return 0; fi chown root:root "$_t" 2>/dev/null || true chmod "$_m" "$_t" mv -f "$_t" "$_d" echo "[activate] CrowdSec: $_d installiert/aktualisiert" [ "$_k" = unit ] && CS_UNIT_CHANGED=1 return 0 } cs_install hapx-ui-crowdsec.sh /usr/local/sbin/hapx-ui-crowdsec 0755 bin cs_install hapx-ui-crowdsec-setup.service /etc/systemd/system/hapx-ui-crowdsec-setup.service 0644 unit cs_install hapx-ui-crowdsec.sudoers /etc/sudoers.d/hapx-ui-crowdsec 0440 sudo # Network helper (Settings → Netzwerk: IPv6 / DNS / flush). Same verified-fetch # + self-heal path, so existing appliances gain the feature on update. cs_install hapx-ui-netcfg.sh /usr/local/sbin/hapx-ui-netcfg 0755 bin cs_install hapx-ui-netcfg.sudoers /etc/sudoers.d/hapx-ui-netcfg 0440 sudo # Speicher- und Protokollvorgaben einmalig richtigstellen. Zwei Werte, die # niemand bewusst gewaehlt hat und die jede Anfrage bremsen: vm.swappiness 60 # schiebt auf einer knapp bestueckten Maschine den Arbeitsspeicher von HAProxy # auf die Platte (auf prod gemessen: 22 MB des laufenden Arbeiters, dazu # durchgehendes Zurueckholen), und ein Journal ohne Deckel waechst in die # Gigabytes, obwohl fast alle Eintraege HAProxy-Anfragezeilen sind, die # ohnehin in /var/log/haproxy.log stehen. # # Nur wenn die verwaltete Datei noch fehlt. Wer sie geloescht hat, hat sich # dagegen entschieden, und diese Entscheidung wird nicht ueberstimmt. if [ -x /usr/local/sbin/hapx-ui-netcfg ] && [ ! -f /etc/sysctl.d/90-hapx-haproxy.conf ]; then if /usr/local/sbin/hapx-ui-netcfg tune apply >/dev/null 2>&1; then echo "[activate] Kern- und Protokollvorgaben gesetzt (BBR, Puffer, vm.swappiness, Journal-Deckel)" fi fi # Auto-update timer — self-heal onto existing boxes that predate it. Seed the # default config + enable the timer (idempotent, on by default). cs_install hapx-ui-autoupdate.service /etc/systemd/system/hapx-ui-autoupdate.service 0644 unit cs_install hapx-ui-autoupdate.timer /etc/systemd/system/hapx-ui-autoupdate.timer 0644 unit if [ ! -f "$DATA/autoupdate.json" ]; then printf '{"enabled": true}\n' > "$DATA/autoupdate.json" chown --reference="$DATA" "$DATA/autoupdate.json" 2>/dev/null || true chmod 0644 "$DATA/autoupdate.json" 2>/dev/null || true fi [ "$CS_UNIT_CHANGED" = 1 ] && systemctl daemon-reload 2>/dev/null || true systemctl enable --now hapx-ui-autoupdate.timer 2>/dev/null || true # 3e) certexport system user + SSH cert-export feature setup (idempotent) # Ensures the certexport user, directory layout, sshd config, and the # systemd drop-in that allows the service to write authorized_keys all exist # and are up-to-date. Safe to repeat on every activate. echo "[activate] Prüfe certexport-Setup (SSH-Zertifikat-Export) ..." _ce_svc_user="$(systemctl show -p User --value hapx-ui 2>/dev/null | tr -d '\n')" [ -z "$_ce_svc_user" ] && _ce_svc_user=hapx-ui # Create certexport user if missing if ! id certexport >/dev/null 2>&1; then useradd --system --create-home --home-dir /home/certexport \ --shell /usr/sbin/nologin certexport 2>/dev/null || true echo "[activate] System-User 'certexport' angelegt" fi # Add certexport to haproxy group (read /etc/haproxy/certs/*.pem) if getent group haproxy >/dev/null 2>&1; then usermod -aG haproxy certexport 2>/dev/null || true fi # grants.json store: hapx-ui SCHREIBT, die certexport-Gruppe LIEST. Die # Rechte werden bei JEDEM Aktivieren gesetzt, nicht nur beim Anlegen — # sonst bleibt ein zurückgesetztes Datenverzeichnis (z. B. nach einem # Neuaufsetzen) unrepariert und der Cert-Sync fällt still aus. mkdir -p "$DATA/certexport" 2>/dev/null || true chown "$_ce_svc_user:certexport" "$DATA/certexport" 2>/dev/null || true chmod 2750 "$DATA/certexport" 2>/dev/null || true # setgid: neue Dateien erben Gruppe certexport if [ -f "$DATA/certexport/grants.json" ]; then chgrp certexport "$DATA/certexport/grants.json" 2>/dev/null || true chmod 0640 "$DATA/certexport/grants.json" 2>/dev/null || true fi # certexport muss $DATA nur DURCHqueren (x), um an sein Unterverzeichnis zu # kommen — ohne Leserecht auf den übrigen UI-State. Genau dafür eine # execute-only-ACL statt einer Gruppenmitgliedschaft. Ohne acl-Werkzeug wird # es übersprungen und der Betreiber gewarnt (der SSH-Command läuft dann bis # zur nächsten Erneuerung, bricht aber danach ab). if command -v setfacl >/dev/null 2>&1; then setfacl -m u:certexport:--x "$DATA" 2>/dev/null && echo "[activate] certexport-Traverse-ACL auf $DATA gesetzt" || echo "[activate] WARN: setfacl auf $DATA fehlgeschlagen — Cert-Sync prüfen" else echo "[activate] WARN: setfacl fehlt — certexport kann $DATA evtl. nicht durchqueren (Paket 'acl' installieren)" fi # Audit log dir: certexport user writes here (runs outside systemd) if [ ! -d /var/log/hapx-ui ]; then mkdir -p /var/log/hapx-ui 2>/dev/null || true chown certexport:certexport /var/log/hapx-ui 2>/dev/null || true chmod 0750 /var/log/hapx-ui 2>/dev/null || true echo "[activate] /var/log/hapx-ui angelegt" fi # sshd-Konfiguration für certexport. # # Die Schlüssel kommen über AuthorizedKeysCommand statt AuthorizedKeysFile. # StrictModes — hier früher global ABGESCHALTET — verlangt, dass die # Schlüsseldatei und jedes Verzeichnis darüber root oder dem sich # anmeldenden Benutzer gehört und für die Gruppe nicht schreibbar ist. Die # Datei liegt im Datenverzeichnis des Dienstes und gehört dem Dienstbenutzer, # kann das also nie erfüllen. Und StrictModes ist in einem Match-Block nicht # erlaubt, das Abschalten galt deshalb für JEDES Konto der Maschine. Der # Umweg über ein Kommando macht die Ausnahme überflüssig. _ce_keycmd_dir=/usr/lib/hapx-ui _ce_keycmd="$_ce_keycmd_dir/certexport-authorized-keys" mkdir -p "$_ce_keycmd_dir" 2>/dev/null || true chown root:root "$_ce_keycmd_dir" 2>/dev/null || true chmod 0755 "$_ce_keycmd_dir" 2>/dev/null || true cat > "$_ce_keycmd" <<'KEYCMD' #!/bin/sh # Gibt die zugelassenen Schlüssel des certexport-Kontos aus. sshd ruft das bei # jedem Anmeldeversuch für diesen Benutzer auf; Argumente werden nicht benutzt, # es erreicht also nichts aus der SSH-Verbindung dieses Skript. KEYS=/var/lib/hapx-ui/certexport/authorized_keys [ -r "$KEYS" ] || exit 0 exec /bin/cat "$KEYS" KEYCMD chown root:root "$_ce_keycmd" 2>/dev/null || true chmod 0755 "$_ce_keycmd" 2>/dev/null || true _ce_sshd=/etc/ssh/sshd_config.d/60-certexport.conf mkdir -p "$(dirname "$_ce_sshd")" 2>/dev/null || true _ce_sshd_want="Match User certexport AuthorizedKeysFile none AuthorizedKeysCommand $_ce_keycmd AuthorizedKeysCommandUser root PermitTTY no X11Forwarding no AllowTcpForwarding no AllowAgentForwarding no PermitOpen none" _ce_cur="$(cat "$_ce_sshd" 2>/dev/null || true)" if [ "$_ce_cur" != "$_ce_sshd_want" ]; then printf '%s\n' "$_ce_sshd_want" > "$_ce_sshd" chmod 0644 "$_ce_sshd" if sshd -t >/dev/null 2>&1; then systemctl reload ssh 2>/dev/null || systemctl reload sshd 2>/dev/null || true echo "[activate] sshd-Konfiguration für certexport installiert und geladen" else # Eine von sshd abgelehnte Datei darf nicht liegen bleiben: der laufende # Dienst behält seine alte Konfiguration, aber der nächste Neustart — # Reboot, Paketaktualisierung — scheitert daran und nimmt den # Fernzugang mit. if [ -n "$_ce_cur" ]; then printf '%s\n' "$_ce_cur" > "$_ce_sshd" else rm -f "$_ce_sshd" fi echo "[activate] WARN: sshd -t fehlgeschlagen — vorheriger Stand wiederhergestellt" fi fi # authorized_keys lives in $DATA/certexport/ which is already covered by the # service unit's ReadWritePaths — no drop-in needed. Remove the old one if # it was written by a previous version of this script. _ce_dropin=/etc/systemd/system/hapx-ui.service.d/20-certexport.conf if [ -f "$_ce_dropin" ]; then rm -f "$_ce_dropin" systemctl daemon-reload 2>/dev/null || true echo "[activate] Altes certexport Drop-in entfernt (nicht mehr benötigt)" fi # 3f) Self-update this helper script — but ONLY from the SIGNED release # asset, verified against the ed25519-signed SHA256SUMS (same trust anchor as # the binary). The previous version fetched the script from raw/branch/main # over plaintext with NO verification and copied it over itself as root, which # turned a LAN MITM or a Gitea write into root code execution on every box. # A failed/unverifiable fetch is non-fatal: the current (already-trusted) # helper is kept. # Fetch into the ROOT-ONLY staging dir, not $DATA: fetch_verified checks the # signature, but if the verified file sat in a service-user-writable path the # user could swap it between verify and the `cp` below (same TOCTOU class as # the staged binary). In the 0700 root:root staging dir that window is closed. _self_new="$STAGING_DIR/hapx-ui-update.sh.$$" if fetch_verified "hapx-ui-update.sh" "$_self_new" && [ -s "$_self_new" ]; then chmod 0755 "$_self_new" if ! cmp -s "$_self_new" "$0" 2>/dev/null; then # Install via an atomic same-directory rename, NEVER cp-in-place: this # script is still being executed, and overwriting $0's bytes in place # corrupts the running interpreter (dash re-reads the file by offset, so a # longer replacement shifts every later line → "syntax error"). rename() # swaps the directory entry while the running process keeps the old inode; # the new helper takes effect on the next invocation. The temp sits in # $0's own (root-owned) dir so the rename stays on one filesystem. _self_dst="$(dirname "$0")/.hapx-ui-update.new.$$" if cp -f "$_self_new" "$_self_dst" 2>/dev/null && chmod 0755 "$_self_dst" 2>/dev/null && mv -f "$_self_dst" "$0" 2>/dev/null; then echo "[activate] Update-Helper aktualisiert (signiert verifiziert) — aktiv beim nächsten Lauf" else rm -f "$_self_dst" 2>/dev/null || true echo "[activate] WARNUNG: Helper-Self-Update nicht installiert (Helper unverändert)" fi fi rm -f "$_self_new" else rm -f "$_self_new" 2>/dev/null || true echo "[activate] Helper-Self-Update übersprungen (Release nicht verifizierbar)" fi # 4) Start and health-check the NEW version — service AND HAProxy. echo "[activate] Starte Dienst mit neuem Binary ..." systemctl start "$SERVICE" 2>/dev/null || true echo "[activate] Warte auf Health-Check (bis zu 90s) ..." if wait_health "$NEW_COMMIT" && haproxy_ok; then echo "[activate] OK — neue Version ist gesund und HAProxy bedient Anfragen." prune_backups state done else # 5) Auto-rollback. Binary-FIRST (fast, safe). The DB is only rolled back if # the new (failed) binary actually MIGRATED it — otherwise touching the DB is # itself a corruption risk (this is what damaged prod on 2026-06-11). The full # restore point ($RPZIP) makes a DB rollback safe when it is genuinely needed. echo "[activate] FEHLER: Health-Check/HAProxy fehlgeschlagen — ROLLBACK ..." systemctl stop "$SERVICE" 2>/dev/null || true _rbnew="$(dirname "$TARGET")/.hapx-ui.rb.$ts" if cp -a "$BACKUP" "$_rbnew" 2>/dev/null; then chmod 0755 "$_rbnew"; mv -f "$_rbnew" "$TARGET"; fi NEW_SCHEMA="$(tr -dc '0-9' < "$(dirname "$DB_PATH")/.schema-version" 2>/dev/null)" [ -n "$NEW_SCHEMA" ] || NEW_SCHEMA="$OLD_SCHEMA" if [ "$NEW_SCHEMA" != "$OLD_SCHEMA" ] && [ -f "$RPZIP" ] && "$TARGET" backup --verify "$RPZIP" >/dev/null 2>&1; then echo "[activate] DB wurde migriert ($OLD_SCHEMA→$NEW_SCHEMA) — stelle DB aus Restore-Punkt wieder her ..." "$TARGET" restore --db "$DB_PATH" "$RPZIP" 2>&1 || echo "[activate] WARNUNG: DB-Restore meldete einen Fehler" else echo "[activate] DB unverändert (Schema $OLD_SCHEMA) — nur Binary zurückgerollt." fi systemctl start "$SERVICE" 2>/dev/null || true if wait_health ""; then echo "[activate] Rollback OK — vorherige Version läuft wieder." else echo "[activate] WARNUNG: auch nach Rollback kein Health — bitte 'journalctl -u $SERVICE' prüfen." fi echo "[activate] Restore-Punkt bleibt unter $RP erhalten." prune_backups state activate_failed; exit 1 fi ;; auto) # Unattended update path — invoked by the hapx-ui-autoupdate.timer as ROOT # (no sudoers involved). It gates on the operator config + a maintenance # window + a crashloop guard, then reuses the EXACT same build+activate verbs # the manual UI path uses (so the hardened restore-point/rollback logic above # applies unchanged). Fleet-wide timing spread comes from the timer's # RandomizedDelaySec, not from here. AUTOLOG="$DATA/last-autoupdate.log" if [ -L "$AUTOLOG" ]; then rm -f "$AUTOLOG"; fi exec >"$AUTOLOG" 2>&1 chmod 0644 "$AUTOLOG" 2>/dev/null || true echo "[auto] $(date -u +%FT%TZ) — Auto-Update-Prüfung" autoupdate_enabled || { echo "[auto] deaktiviert (autoupdate.json) — Ende"; auto_status disabled ""; exit 0; } in_window || { echo "[auto] ausserhalb des Wartungsfensters — Ende"; auto_status window ""; exit 0; } tag="$(resolve_release_tag)" [ -n "$tag" ] || { echo "[auto] kein Release erreichbar — Ende"; auto_status unreachable ""; exit 0; } # detect_env resolves $TARGET/$SERVICE/$DB_PATH from the actual unit. Without # it the version check below would inspect the hard-coded default path, which # is only correct by luck on a stock install. detect_env set -- $(running_version) cur_ver="${1:-}"; cur_cmt="${2:-}" echo "[auto] laufend=${cur_ver:-?} (${cur_cmt:-?}) verfuegbar=$tag" if is_current "$tag" "$cur_ver" "$cur_cmt"; then echo "[auto] bereits aktuell — Ende"; auto_status current ""; exit 0 fi if is_failed_version "$tag"; then echo "[auto] $tag ist als fehlgeschlagen markiert — ueberspringe (warte auf neuere Version)" auto_status skipped "Diese Version ist bei einem fruehreren Versuch fehlgeschlagen und wird nicht erneut automatisch installiert." exit 0 fi echo "[auto] Neue Version $tag — starte build ..." if ! "$0" build; then echo "[auto] build fehlgeschlagen — Ende"; auto_status failed "Download oder Verifikation fehlgeschlagen."; exit 1; fi if [ "$(state_phase)" != built ]; then echo "[auto] build-Phase != built ($(state_phase)) — Ende"; auto_status failed "Build-Phase unerwartet beendet."; exit 1; fi # Belt-and-suspenders against a restart loop: if the freshly staged binary is # byte-identical to the installed one, activating would stop the service and # swap a file for its own copy — pure downtime, no change. The version compare # above should already have caught this; this second gate keeps a future # parsing regression from costing the fleet a restart every timer tick. # Deliberately only in the auto path — the manual "activate" button must still # run its self-healing steps (units, sudoers, groups, helper self-update). _staged_sha="$(cat "$SHAFILE" 2>/dev/null)" _target_sha="$(sha256sum "$TARGET" 2>/dev/null | awk '{print $1}')" if [ -n "$_staged_sha" ] && [ "$_staged_sha" = "$_target_sha" ]; then echo "[auto] Staging-Binary ist identisch mit dem installierten — kein Neustart noetig." auto_status current "Bereits installiert (identisches Binary)." rm -f "$STAGED" "$STAGED.meta" "$SHAFILE" 2>/dev/null || true state done exit 0 fi echo "[auto] Aktiviere $tag ..." if "$0" activate; then echo "[auto] Auto-Update auf $tag erfolgreich." auto_status updated "" else echo "[auto] Auto-Update auf $tag FEHLGESCHLAGEN — Rollback erfolgt, Version wird nicht erneut versucht." mark_failed_version "$tag" auto_status failed "Aktivierung fehlgeschlagen — es wurde automatisch zurueckgerollt. Diese Version wird nicht erneut automatisch versucht." exit 1 fi ;; *) echo "usage: $0 preflight|build|activate|auto" >&2 exit 2 ;; esac