fix(rebuild): re-index and purge the CDN after a rebuild
A rebuild replaces an object in place: a package whose build failed was published as its CRAN source, and the rebuilt binary takes exactly the same URL. Two things then hide the result from clients. The slot's index still advertises the old MD5 and, for anything served from source, no Built stamp, because weekly-rebuild-missing never re-indexed. And the pull zone caches tarballs for ~370 days, while purge_cdn_cache.sh only purges the five index files, so the edge keeps serving the source tarball for up to a year with nothing about it looking wrong. Observed after rebuilding AATtools 0.0.3: the pipeline reported a successful upload while the edge still served the CRAN source, etag ea8127... and no Meta/. - re-index the slot at the end of a rebuild, flat and per-minor, detecting the codename from the image rather than adding OS_ID to 18 matrix rows - add scripts/purge_cdn_zone.sh and call it afterwards. One zone purge covers every replaced object and all three hostnames, which share pull zone 3857050; purging per URL would be ~13.5k rate-limited calls per arch where one missed call leaves a silently stale package - purge on failure too, since a rebuild that died part-way still replaced objects and those are exactly the ones a stale edge keeps hiding
This commit is contained in:
parent
6b0a987a6a
commit
7dc84d590b
2 changed files with 85 additions and 0 deletions
|
|
@ -163,6 +163,17 @@ steps:
|
|||
- UVR_R_BIN=/opt/R/$R_VERSION/bin/R local/uvr-install.sh httr2
|
||||
- /opt/R/$R_VERSION/bin/R -q -e 'source("local/fetch-rebuild-packages-from-issue.R")'
|
||||
- $XVFB $XVFB_ARGS -- /opt/R/$R_VERSION/bin/R -q -e "sink(stdout(), type = 'message'); options(crayon.enabled = TRUE, Ncpus = $NCPUS, future.globals.onReference = NULL); pkgs <- readLines('/tmp/rebuild_pkgs.txt'); if (length(pkgs) == 0) { cat('Nothing to rebuild\n'); q('no') }; excluded <- jsonlite::fromJSON('local/excluded-packages.json')[['package']]; pkgs <- setdiff(pkgs, excluded); cat(sprintf('Rebuilding %d packages\n', length(pkgs))); n <- length(pkgs); for (i in seq_along(pkgs)) { x <- pkgs[i]; cat(sprintf('[%d/%d] %s\n', i, n, x)); tryCatch(bincraft::build_binary_package(x, tag_limit = 1L, patches = 'local/patches', s3_endpoint = 'https://s3.eu-central-003.backblazeb2.com', s3_region = 'eu-central-003', s3_bucket = 'devxy-rpkgs-binaries', s3_access_key_id = Sys.getenv('B2_S3_ACCESS_KEY'), s3_secret_access_key = Sys.getenv('B2_S3_SECRET_KEY'), metadata_db_host = 'r-binaries.devxy.io', metadata_db_name = 'build_metadata', metadata_db_table = 'single_builds', metadata_db_user = 'rpkgs', metadata_db_password = Sys.getenv('PGPASS'), metadata_db_sslmode = 'require', metadata_db_port = 15432, archive = TRUE, upload = TRUE, store_build_metadata = TRUE), error = function(e) cat(sprintf('ERROR building %s - %s\n', x, conditionMessage(e)))) }" 2>&1
|
||||
# A rebuild replaces objects in place, so the slot's index still advertises
|
||||
# the old MD5 and, for anything that had been served from source, no Built
|
||||
# stamp. Re-index here rather than waiting for the next process-updates
|
||||
# run, or the rebuilt binaries stay invisible to clients until then.
|
||||
# The codename is detected from the image's /etc/os-release.
|
||||
- /opt/R/$R_VERSION/bin/R -q -e 'library(bincraft); upload_package_index(s3_endpoint = "https://s3.eu-central-003.backblazeb2.com", s3_region = "eu-central-003", s3_bucket = "devxy-rpkgs-binaries", s3_access_key_id = Sys.getenv("B2_S3_ACCESS_KEY"), s3_secret_access_key = Sys.getenv("B2_S3_SECRET_KEY"))'
|
||||
- |
|
||||
for RBIN in /opt/R/[0-9]*/bin/R; do
|
||||
RMINOR=$(basename "$(dirname "$(dirname "$RBIN")")" | cut -d. -f1-2)
|
||||
/opt/R/$R_VERSION/bin/R -q -e "library(bincraft); upload_package_index(r_minor = '$RMINOR', s3_endpoint = 'https://s3.eu-central-003.backblazeb2.com', s3_region = 'eu-central-003', s3_bucket = 'devxy-rpkgs-binaries', s3_access_key_id = Sys.getenv('B2_S3_ACCESS_KEY'), s3_secret_access_key = Sys.getenv('B2_S3_SECRET_KEY'))" || true
|
||||
done
|
||||
backend_options:
|
||||
docker:
|
||||
resources:
|
||||
|
|
@ -172,3 +183,25 @@ steps:
|
|||
limits:
|
||||
memory: 18Gi
|
||||
cpu: 3000m
|
||||
|
||||
- name: Purge CDN cache
|
||||
image: reg.devxy.io/docker.io/library/alpine:3.24
|
||||
environment:
|
||||
OTEL_R_TRACES_EXPORTER: none
|
||||
OTEL_R_LOGS_EXPORTER: none
|
||||
OTEL_R_METRICS_EXPORTER: none
|
||||
BUNNYNET_API_KEY:
|
||||
from_secret: BUNNYNET_API_KEY
|
||||
REPO_RO_TOKEN:
|
||||
from_secret: REPO_RO_TOKEN
|
||||
# All hostnames on the zone share this id, so one purge covers
|
||||
# cran.devxy.io, cran.allianceswisspass.devxy.io and cran.rpkgs.com.
|
||||
BUNNY_PULLZONE: '3857050'
|
||||
commands:
|
||||
- apk add --no-cache -q bash curl git
|
||||
- git clone -q https://pat-s:$$REPO_RO_TOKEN@git.devxy.io/devxy/build-cran-binaries.git .
|
||||
- bash scripts/purge_cdn_zone.sh "$BUNNYNET_API_KEY" "$BUNNY_PULLZONE"
|
||||
# A rebuild that died part-way still replaced objects, and those are exactly
|
||||
# the ones a stale edge would keep hiding, so purge either way.
|
||||
when:
|
||||
- status: [success, failure]
|
||||
|
|
|
|||
52
scripts/purge_cdn_zone.sh
Executable file
52
scripts/purge_cdn_zone.sh
Executable file
|
|
@ -0,0 +1,52 @@
|
|||
#!/usr/bin/env bash
|
||||
#
|
||||
# Purge the entire BunnyCDN pull zone.
|
||||
#
|
||||
# `purge_cdn_cache.sh` purges the five index files by URL, which is right after
|
||||
# a normal update: new packages arrive at new URLs, so only the index is stale.
|
||||
#
|
||||
# A rebuild is different. It replaces an object *in place*: a package whose
|
||||
# build failed was published as its CRAN source, and the rebuilt binary takes
|
||||
# exactly the same URL. The zone caches tarballs for ~370 days
|
||||
# (`cache_expiration_time` in cdn.tf), so without a purge every client keeps
|
||||
# receiving the source tarball for up to a year, and nothing about it looks
|
||||
# wrong from the outside.
|
||||
#
|
||||
# Purging per URL would mean one API call per replaced package -- ~13.5k per
|
||||
# arch against a rate-limited endpoint, where a single missed call leaves a
|
||||
# silently stale package. One zone purge is a single call regardless of how many
|
||||
# objects were replaced. The cost is a cold cache for everything else, which is
|
||||
# why this is not used by the daily update path.
|
||||
#
|
||||
# All hostnames on the zone (cran.devxy.io, cran.allianceswisspass.devxy.io,
|
||||
# cran.rpkgs.com) share pull zone 3857050, so one purge covers all of them.
|
||||
#
|
||||
# Usage:
|
||||
# purge_cdn_zone.sh <BUNNYNET_API_KEY> <pull_zone_id>
|
||||
#
|
||||
set -euo pipefail
|
||||
|
||||
if (($# < 2)); then
|
||||
echo "usage: $0 <api_key> <pull_zone_id>" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
api_key="$1"
|
||||
zone_id="$2"
|
||||
|
||||
echo "Purging BunnyCDN pull zone ${zone_id}"
|
||||
|
||||
status=$(
|
||||
curl -sS -o /tmp/purge_zone_response.txt -w '%{http_code}' -X POST \
|
||||
-H "AccessKey: ${api_key}" \
|
||||
-H "Content-Length: 0" \
|
||||
"https://api.bunny.net/pullzone/${zone_id}/purgeCache"
|
||||
)
|
||||
|
||||
if [[ "${status}" != "200" && "${status}" != "204" ]]; then
|
||||
echo "Purge of pull zone ${zone_id} failed with HTTP ${status}:" >&2
|
||||
cat /tmp/purge_zone_response.txt >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Purged pull zone ${zone_id} (HTTP ${status})"
|
||||
Loading…
Reference in a new issue