Watch
1
0
Fork
You've already forked souveraine
0
souveraine/packaging/arch/publish-edge.sh
Fimeg 044c373cad ci: publish into edge additively
edge is a shared archive but ci.yml deleted and recreated it each push, erasing
other producers. publish-edge.sh merges into the live db under a flock and
removes only its own superseded builds. Verifies the merge kept every foreign
entry, and reads the db back before trusting it.
2026-07-25 17:50:30 -04:00

406 lines
17 KiB
Shell
Executable file

#!/usr/bin/env bash
# Additive publisher for the rolling `edge` pacman archive.
#
# The `edge` release on Fimeg/souveraine is a MULTI-PRODUCER archive: souveraine
# publishes its own binaries there, and so does every other repo that ships a
# package to the phone. Until 2026-07-25 souveraine's CI deleted and recreated
# the release on every push and uploaded only its own `pacman-repo`, so anything
# another job had published was erased — silently, with nothing erroring. This
# script is the additive replacement.
#
# Rules it enforces:
# * The release and its tag are NEVER deleted. Created once, patched after.
# * The per-arch database is read-modify-written: the live
# `souveraine-<arch>.db.tar.zst` is fetched, our packages are `repo-add`ed
# into it, and the result is uploaded back. Entries belonging to other
# producers survive because repo-add only touches the names it is given.
# * The ONLY assets deleted are ones this producer supersedes: for each
# package we publish, the `%FILENAME%` recorded in the live database under
# the same `%NAME%`. A producer can never delete another's package.
# * Everything runs under one exclusive flock on the archdev runner, so two
# jobs cannot interleave their read-modify-write of the same database.
#
# Usage:
# publish-edge.sh <producer> <repo-dir> [extra-asset ...]
#
# <producer> short name for this publisher (souveraine, souveraine-updater).
# Used for the manifest line in the release body and for the
# per-producer checksum asset. NOT used to decide deletions.
# <repo-dir> directory holding one subdirectory per architecture, each with
# `*.pkg.tar.zst` and matching `.sig` files. The databases are
# built here; do not pre-create them.
# extra-asset optional loose files to publish alongside (raw binaries,
# checksum manifests). Replaced in place by name.
#
# Environment:
# EDGE_TOKEN required. Gitea token with write access to the ARCHIVE
# repo — for a producer other than souveraine this must be
# a PAT, since a job's own GITHUB_TOKEN is scoped to its
# own repository and cannot upload here.
# ARCHIVE_KEY required. GPG key id used for `repo-add --sign`.
# GITHUB_SERVER_URL Gitea base URL. Default http://10.10.20.120:4455.
# ARCHIVE_REPO owner/name holding the release. Default Fimeg/souveraine.
# EDGE_TARGET_SHA commitish for the tag, used only when the release does
# not exist yet. Default: the archive repo's default branch.
# PRODUCER_VERSION version string recorded in the release body manifest.
# EDGE_LOCK lock file path. Default ~/.cache/souveraine-archive/edge.lock.
set -euo pipefail
PRODUCER="${1:?usage: publish-edge.sh <producer> <repo-dir> [extra-asset ...]}"
REPO_DIR="${2:?usage: publish-edge.sh <producer> <repo-dir> [extra-asset ...]}"
shift 2
EXTRA_ASSETS=("$@")
: "${EDGE_TOKEN:?EDGE_TOKEN must be set (Gitea token with write access to the archive repo)}"
: "${ARCHIVE_KEY:?ARCHIVE_KEY must be set (gpg key id for repo-add --sign)}"
SERVER="${GITHUB_SERVER_URL:-http://10.10.20.120:4455}"
ARCHIVE_REPO="${ARCHIVE_REPO:-Fimeg/souveraine}"
PRODUCER_VERSION="${PRODUCER_VERSION:-unknown}"
LOCK="${EDGE_LOCK:-$HOME/.cache/souveraine-archive/edge.lock}"
# Overridable so a change to this script can be rehearsed end-to-end against a
# throwaway tag before it is pointed at the archive the phone actually installs
# from. Leave it alone in CI.
EDGE_TAG="${EDGE_TAG:-edge}"
# Re-exec under an exclusive lock. The runner is a single host-mode box shared
# by every repo's jobs, so this one file serialises all producers. Without it
# two concurrent read-modify-writes of the same database both start from the
# same base and the second upload silently drops the first's entries — the same
# lost-update the delete-and-recreate shape caused, just narrower.
if [ "${EDGE_LOCK_HELD:-}" != 1 ]; then
mkdir -p "$(dirname "$LOCK")"
export EDGE_LOCK_HELD=1
exec flock "$LOCK" "$0" "$PRODUCER" "$REPO_DIR" "${EXTRA_ASSETS[@]+"${EXTRA_ASSETS[@]}"}"
fi
API="$SERVER/api/v1"
AUTH="Authorization: token $EDGE_TOKEN"
api() { curl -sf -H "$AUTH" "$@"; }
log() { echo "[publish-edge] $*"; }
# --- locate or create the release -------------------------------------------
# Note the asymmetry with the old code: there is no DELETE anywhere in this
# script for the release or the tag. `edge` outlives every individual push.
REL_JSON=$(curl -s -H "$AUTH" "$API/repos/$ARCHIVE_REPO/releases/tags/$EDGE_TAG" || true)
REL_ID=$(printf '%s' "$REL_JSON" | python3 -c "
import json,sys
try: print(json.load(sys.stdin).get('id',''))
except Exception: print('')")
if [ -z "$REL_ID" ]; then
TARGET="${EDGE_TARGET_SHA:-}"
if [ -z "$TARGET" ]; then
TARGET=$(api "$API/repos/$ARCHIVE_REPO" | python3 -c "
import json,sys; print(json.load(sys.stdin)['default_branch'])")
fi
log "no $EDGE_TAG release yet — creating it at $TARGET"
REL_ID=$(api -X POST -H "Content-Type: application/json" \
"$API/repos/$ARCHIVE_REPO/releases" \
-d "$(python3 -c "
import json,sys
print(json.dumps({
'tag_name': sys.argv[2],
'target_commitish': sys.argv[1],
'name': sys.argv[2],
'body': 'Rolling multi-producer pacman archive. Each repo publishes its own '
'signed packages into the shared souveraine-<arch> databases; the '
'release is never deleted. Producers are listed below.\n',
'prerelease': True,
}))" "$TARGET" "$EDGE_TAG")" | python3 -c "import json,sys; print(json.load(sys.stdin)['id'])")
else
log "publishing into existing $EDGE_TAG release $REL_ID"
fi
ASSETS_JSON=$(api "$API/repos/$ARCHIVE_REPO/releases/$REL_ID/assets")
asset_id() {
printf '%s' "$ASSETS_JSON" | python3 -c "
import json,sys
name = sys.argv[1]
for a in json.load(sys.stdin):
if a['name'] == name:
print(a['id']); break
" "$1"
}
delete_asset() {
local name="$1" id
id=$(asset_id "$name")
[ -n "$id" ] || return 0
curl -sf -X DELETE -H "$AUTH" \
"$API/repos/$ARCHIVE_REPO/releases/$REL_ID/assets/$id" -o /dev/null
log "deleted stale asset $name"
}
# Gitea happily stores two assets with the same name, which would leave pacman
# fetching whichever the API returned first. Always delete before uploading.
upload_asset() {
local path="$1" name
name="$(basename "$path")"
delete_asset "$name"
curl -sf -X POST -H "$AUTH" \
"$API/repos/$ARCHIVE_REPO/releases/$REL_ID/assets?name=$name" \
-F "attachment=@$path" -o /dev/null
log "uploaded $name"
}
# Fetch by the SAME URL shape pacman uses on the phone
# (`releases/download/<tag>/<name>`), not by asset id. The by-id API endpoint
# returns attachment metadata as JSON in some Gitea versions, and writing that
# into a .db.tar.zst would look like a corrupt database rather than an error.
# This path is proven: it is exactly what souveraine-pacman-fetch already pulls.
download_asset() {
local name="$1" dest="$2"
curl -sfL -H "$AUTH" \
"$SERVER/$ARCHIVE_REPO/releases/download/$EDGE_TAG/$name" -o "$dest"
}
# --- per-arch read-modify-write ---------------------------------------------
STAGE=$(mktemp -d)
trap 'rm -rf "$STAGE"' EXIT
# Dump `NAME<TAB>FILENAME` for every entry in a pacman database. Used three
# times: to find our own superseded builds, to snapshot what was in the archive
# before we touched it, and to prove after upload that the snapshot survived.
cat > "$STAGE/dbentries.py" <<'PY'
import subprocess, sys
# `bsdtar -xO` streams every member concatenated; entry `desc` files are the
# only ones we want. Each is a sequence of `%KEY%` header lines followed by
# their values, and a `%FILENAME%` header is what starts a new entry.
out = subprocess.run(
["bsdtar", "-xOf", sys.argv[1], "--include=*/desc"],
capture_output=True, text=True, check=True).stdout
entries, cur, key = [], {}, None
for line in out.splitlines():
line = line.strip()
if line.startswith("%") and line.endswith("%") and len(line) > 2:
if line == "%FILENAME%" and cur:
entries.append(cur)
cur = {}
key = line.strip("%")
elif line and key and key not in cur:
cur[key] = line
if cur:
entries.append(cur)
for e in entries:
if "NAME" in e and "FILENAME" in e:
print(f"{e['NAME']}\t{e['FILENAME']}")
PY
db_entries() { python3 "$STAGE/dbentries.py" "$1"; }
# Package name is everything before the trailing -pkgver-pkgrel-arch. pkgver and
# pkgrel may not contain a hyphen, so the last three fields are always known.
pkg_name_of() {
local base
base="$(basename "$1")"
base="${base%.pkg.tar.*}"
printf '%s\n' "${base%-*-*-*}"
}
for ARCH_DIR in "$REPO_DIR"/*/; do
[ -d "$ARCH_DIR" ] || continue
ARCH="$(basename "$ARCH_DIR")"
shopt -s nullglob
PKGS=("$ARCH_DIR"*.pkg.tar.zst)
shopt -u nullglob
if [ ${#PKGS[@]} -eq 0 ]; then
log "$ARCH: no packages, skipping"
continue
fi
DB="souveraine-$ARCH"
WORK="$STAGE/$ARCH"
mkdir -p "$WORK"
# Pull the LIVE database. Every other producer's entries live in here, and
# they survive only because we hand this file to repo-add rather than
# building a fresh one.
#
# The distinction below is the whole safety property of this script. Starting
# a fresh database is correct ONLY when the archive genuinely has none yet. If
# the asset is listed but we cannot fetch or read it — a transient 500, a
# truncated body, an HTML error page — falling through to a fresh database
# would erase every other producer on upload. That is precisely the clobber
# this script exists to remove, so it is a hard failure instead.
if [ -n "$(asset_id "$DB.db.tar.zst")" ]; then
download_asset "$DB.db.tar.zst" "$WORK/$DB.db.tar.zst" \
|| { echo "FATAL: $DB.db.tar.zst is published but could not be fetched;" \
"refusing to publish a database that would drop other producers" >&2
exit 1; }
bsdtar -tf "$WORK/$DB.db.tar.zst" >/dev/null 2>&1 \
|| { echo "FATAL: fetched $DB.db.tar.zst is not a readable archive;" \
"refusing to overwrite the live database" >&2
exit 1; }
log "$ARCH: fetched live database ($(bsdtar -tf "$WORK/$DB.db.tar.zst" | grep -c '/desc$') entries)"
else
log "$ARCH: archive has no $DB database yet — creating the first one"
fi
# The files database is published too (pacman -F). repo-add maintains it
# beside the .db automatically, but only merges into one that is already
# there — without this fetch it would be rebuilt from just our packages and
# every other producer would vanish from `pacman -F` while still installing
# fine, which is the kind of half-broken that goes unnoticed for months.
if [ -n "$(asset_id "$DB.files.tar.zst")" ]; then
download_asset "$DB.files.tar.zst" "$WORK/$DB.files.tar.zst" \
|| { echo "FATAL: $DB.files.tar.zst is published but could not be fetched" >&2; exit 1; }
bsdtar -tf "$WORK/$DB.files.tar.zst" >/dev/null 2>&1 \
|| { echo "FATAL: fetched $DB.files.tar.zst is not a readable archive" >&2; exit 1; }
fi
# Snapshot the archive as it stands BEFORE we touch it. Everything in here
# that we do not deliberately supersede must still be in the database we
# upload; the check at the end of this loop enforces exactly that.
: > "$WORK/before.tsv"
if [ -f "$WORK/$DB.db.tar.zst" ]; then
db_entries "$WORK/$DB.db.tar.zst" > "$WORK/before.tsv"
fi
# The package NAMES we are publishing. Any entry in the live database under
# one of these names is a previous build of OURS — those filenames are the
# only assets this script is ever entitled to delete.
: > "$WORK/ours.txt"
: > "$WORK/newfiles.txt"
for p in "${PKGS[@]}"; do
pkg_name_of "$p" >> "$WORK/ours.txt"
basename "$p" >> "$WORK/newfiles.txt"
done
SUPERSEDED=$(awk -F'\t' '
NR==FNR && FILENAME==ARGV[1] { ours[$0]=1; next }
FILENAME==ARGV[2] { newf[$0]=1; next }
($1 in ours) && !($2 in newf) { print $2 }
' "$WORK/ours.txt" "$WORK/newfiles.txt" "$WORK/before.tsv")
cp "${PKGS[@]}" "$WORK/"
for p in "${PKGS[@]}"; do
if [ -f "$p.sig" ]; then cp "$p.sig" "$WORK/"; fi
done
shopt -s nullglob
STAGED=("$WORK"/*.pkg.tar.zst)
shopt -u nullglob
repo-add --include-sigs --sign --key "$ARCHIVE_KEY" \
"$WORK/$DB.db.tar.zst" "${STAGED[@]}"
# Before anything is uploaded: the merged database must still carry every
# entry the archive had, except the ones we deliberately replaced. A failure
# here means repo-add did not merge the way this script assumes, and the
# upload is abandoned with the live archive untouched.
db_entries "$WORK/$DB.db.tar.zst" > "$WORK/after.tsv"
MISSING=$(awk -F'\t' '
NR==FNR && FILENAME==ARGV[1] { ours[$0]=1; next }
FILENAME==ARGV[2] { kept[$1]=1; next }
!($1 in ours) && !($1 in kept) { print $1 }
' "$WORK/ours.txt" "$WORK/after.tsv" "$WORK/before.tsv")
if [ -n "$MISSING" ]; then
echo "FATAL: merging into $DB dropped entries this producer does not own:" >&2
printf ' %s\n' $MISSING >&2
echo "nothing was uploaded; the live archive is unchanged" >&2
exit 1
fi
log "$ARCH: merged — $(wc -l < "$WORK/before.tsv") entries in, $(wc -l < "$WORK/after.tsv") out"
# repo-add leaves .db and .db.sig as symlinks; release assets cannot carry
# those, so publish real files.
rm -f "$WORK/$DB.db" "$WORK/$DB.db.sig"
cp "$WORK/$DB.db.tar.zst" "$WORK/$DB.db"
cp "$WORK/$DB.db.tar.zst.sig" "$WORK/$DB.db.sig"
# Producer-scoped checksums: a whole-archive manifest would go stale the
# moment another producer published, and claim coverage it does not have.
( cd "$WORK" && sha256sum ./*.pkg.tar.zst > "$PRODUCER-$ARCH.sha256" )
# Packages first, database last. A reader that races us then sees either the
# old database (pointing at packages that are all still present) or the new
# one (pointing at packages already uploaded) — never a database naming a
# package that has not landed yet.
for f in "${STAGED[@]}"; do
upload_asset "$f"
if [ -f "$f.sig" ]; then upload_asset "$f.sig"; fi
done
upload_asset "$WORK/$PRODUCER-$ARCH.sha256"
if [ -f "$WORK/$DB.files.tar.zst" ]; then
upload_asset "$WORK/$DB.files.tar.zst"
if [ -f "$WORK/$DB.files.tar.zst.sig" ]; then
upload_asset "$WORK/$DB.files.tar.zst.sig"
fi
fi
upload_asset "$WORK/$DB.db.tar.zst"
upload_asset "$WORK/$DB.db.tar.zst.sig"
upload_asset "$WORK/$DB.db"
upload_asset "$WORK/$DB.db.sig"
# Read the database back over the SAME URL pacman uses and confirm the archive
# really holds what we just merged. Publishing is two dozen HTTP calls against
# a box on the far side of a LAN; "the upload returned 201" is not the same
# claim as "the phone will see this".
if download_asset "$DB.db.tar.zst" "$WORK/live.db.tar.zst"; then
db_entries "$WORK/live.db.tar.zst" | sort > "$WORK/live.tsv"
sort "$WORK/after.tsv" > "$WORK/expect.tsv"
if ! diff -q "$WORK/expect.tsv" "$WORK/live.tsv" >/dev/null; then
echo "FATAL: the published $DB database does not match what was uploaded:" >&2
diff "$WORK/expect.tsv" "$WORK/live.tsv" >&2 || true
exit 1
fi
log "$ARCH: verified live database matches ($(wc -l < "$WORK/live.tsv") entries)"
else
echo "FATAL: could not read back $DB.db.tar.zst after publishing it" >&2
exit 1
fi
# Now — and only now, with the new database live and verified — drop our own
# superseded builds. Doing it earlier would leave a window where the published
# database referenced a package that no longer existed.
if [ -n "$SUPERSEDED" ]; then
while IFS= read -r old; do
[ -n "$old" ] || continue
delete_asset "$old"
delete_asset "$old.sig"
done <<< "$SUPERSEDED"
fi
# Legacy whole-archive manifest from the delete-and-recreate era. Harmless but
# misleading — it claims to cover packages this producer never built.
delete_asset "$DB-repo.sha256"
done
for f in "${EXTRA_ASSETS[@]+"${EXTRA_ASSETS[@]}"}"; do
[ -f "$f" ] || { log "extra asset $f missing, skipping"; continue; }
upload_asset "$f"
done
# --- manifest line in the release body ---------------------------------------
# The body is the only place that records WHO put what in the archive. With the
# tag pinned at creation it is also the only place a reader can see that the
# archive moved at all.
BODY=$(printf '%s' "$REL_JSON" | python3 -c "
import json,sys
try: print(json.load(sys.stdin).get('body',''))
except Exception: print('')")
NEW_BODY=$(python3 - "$BODY" "$PRODUCER" "$PRODUCER_VERSION" <<'PY'
import sys
body, producer, version = sys.argv[1], sys.argv[2], sys.argv[3]
line = f"- {producer}: {version}"
kept = [l for l in body.splitlines() if not l.startswith(f"- {producer}: ")]
while kept and not kept[-1].strip():
kept.pop()
print("\n".join(kept + [line]))
PY
)
curl -sf -X PATCH -H "$AUTH" -H "Content-Type: application/json" \
"$API/repos/$ARCHIVE_REPO/releases/$REL_ID" \
-d "$(python3 -c "import json,sys; print(json.dumps({'body': sys.argv[1]}))" "$NEW_BODY")" \
-o /dev/null
log "$PRODUCER $PRODUCER_VERSION published into edge additively"