From 6b0a987a6a04c6f88ddc57fb944642a8754e50ed Mon Sep 17 00:00:00 2001 From: pat-s Date: Sun, 9 Aug 2026 16:25:13 +0000 Subject: [PATCH 1/2] fix(build): keep source fallbacks out of the S3 package cache The cache handed to build_binary_package() as s3_package_cache was the raw bucket listing, and the build list subtracted the same listing. Neither could tell a binary from a package whose build failed and was published as its CRAN source, so every source fallback read as "already built" and was skipped for good - which is how alpine324 accumulated ~13.5k of them. - drop objects the slot's index reports as served from source, which bincraft marks by leaving the Built stamp off - build s3_dt from the filtered listing too, since it is subtracted from the build list and would otherwise exclude the packages that need building - log how many objects were dropped Archived objects have no index record and are kept: unknown means binary, never "rebuild it". A slot last indexed by a bincraft that predates the Built change stamps everything, so its cache is unchanged from before. --- local/packages-to-build.R | 59 +++++++++++++++++++++++++++++++++++---- 1 file changed, 53 insertions(+), 6 deletions(-) diff --git a/local/packages-to-build.R b/local/packages-to-build.R index 887952d..3b508fb 100644 --- a/local/packages-to-build.R +++ b/local/packages-to-build.R @@ -89,14 +89,61 @@ s3_pkgs <- s3fs::s3_dir_ls( recurse = TRUE ) -# Save the raw S3 file listing for the build step to use as s3_package_cache +file_names <- basename(s3_pkgs) + +# An object occupying a key is not proof a binary was built: a package whose +# build failed has its CRAN source published under exactly that name. Left in +# the cache, `build_binary_package()` reads it as "already built" and skips the +# package forever, which is how alpine324 accumulated ~13.5k source tarballs. +# +# bincraft stamps `Built` only on records it actually built, so the slot's own +# index distinguishes them. A slot last indexed by a bincraft that predates that +# fix stamps `Built` on everything, so the cache is then unchanged from before. +# Archived objects have no index record and are kept: unknown means binary, +# never "rebuild it". +index_url <- sprintf( + "https://cran.rpkgs.com/%s/%s/latest/src/contrib/PACKAGES.gz", + arch, + codename +) +source_served <- tryCatch( + { + con_idx <- gzcon(url(index_url, open = "rb")) + on.exit(close(con_idx), add = TRUE) + idx <- read.dcf(con_idx, fields = c("Package", "Version", "Built")) + sprintf( + "%s_%s.tar.gz", + idx[is.na(idx[, "Built"]), "Package"], + idx[is.na(idx[, "Built"]), "Version"] + ) + }, + error = function(e) { + cat(sprintf( + "WARNING: could not read %s (%s); keeping the full S3 cache\n", + index_url, + conditionMessage(e) + )) + character(0) + } +) + +binary_cache <- setdiff(file_names, source_served) +cat(sprintf( + "S3 cache: %d objects, %d served as CRAN source, %d usable binaries\n", + length(file_names), + length(file_names) - length(binary_cache), + length(binary_cache) +)) + +# Save the S3 file listing for the build step to use as s3_package_cache. # This avoids loading s3fs/reticulate in the build container, saving memory for # the dependency-installer subprocesses -saveRDS(basename(s3_pkgs), "/mnt/cache/packages/s3_cache.rds") - -file_names <- basename(s3_pkgs) -matches <- regexec("^([A-Za-z0-9.]+)_([0-9][^/]*)\\.tar\\.gz$", file_names) -parts <- regmatches(file_names, matches) +saveRDS(binary_cache, "/mnt/cache/packages/s3_cache.rds") +# Built from the filtered listing, not the raw one: `s3_dt` is subtracted from +# the build list below, so a source fallback left in here would exclude the very +# package that needs building. +matches <- regexec("^([A-Za-z0-9.]+)_([0-9][^/]*)\\.tar\\.gz$", binary_cache) +parts <- regmatches(binary_cache, matches) parts <- parts[sapply(parts, length) == 3] s3_dt <- data.table( Package = sapply(parts, `[`, 2), -- 2.54.0 From 7dc84d590b8119104c9f1082bf9f4f382bea28f8 Mon Sep 17 00:00:00 2001 From: pat-s Date: Sun, 9 Aug 2026 18:31:23 +0000 Subject: [PATCH 2/2] fix(rebuild): re-index and purge the CDN after a rebuild A rebuild replaces an object in place: a package whose build failed was published as its CRAN source, and the rebuilt binary takes exactly the same URL. Two things then hide the result from clients. The slot's index still advertises the old MD5 and, for anything served from source, no Built stamp, because weekly-rebuild-missing never re-indexed. And the pull zone caches tarballs for ~370 days, while purge_cdn_cache.sh only purges the five index files, so the edge keeps serving the source tarball for up to a year with nothing about it looking wrong. Observed after rebuilding AATtools 0.0.3: the pipeline reported a successful upload while the edge still served the CRAN source, etag ea8127... and no Meta/. - re-index the slot at the end of a rebuild, flat and per-minor, detecting the codename from the image rather than adding OS_ID to 18 matrix rows - add scripts/purge_cdn_zone.sh and call it afterwards. One zone purge covers every replaced object and all three hostnames, which share pull zone 3857050; purging per URL would be ~13.5k rate-limited calls per arch where one missed call leaves a silently stale package - purge on failure too, since a rebuild that died part-way still replaced objects and those are exactly the ones a stale edge keeps hiding --- .crow/weekly-rebuild-missing.yaml | 33 ++++++++++++++++++++ scripts/purge_cdn_zone.sh | 52 +++++++++++++++++++++++++++++++ 2 files changed, 85 insertions(+) create mode 100755 scripts/purge_cdn_zone.sh diff --git a/.crow/weekly-rebuild-missing.yaml b/.crow/weekly-rebuild-missing.yaml index 0353901..c86fe5a 100644 --- a/.crow/weekly-rebuild-missing.yaml +++ b/.crow/weekly-rebuild-missing.yaml @@ -163,6 +163,17 @@ steps: - UVR_R_BIN=/opt/R/$R_VERSION/bin/R local/uvr-install.sh httr2 - /opt/R/$R_VERSION/bin/R -q -e 'source("local/fetch-rebuild-packages-from-issue.R")' - $XVFB $XVFB_ARGS -- /opt/R/$R_VERSION/bin/R -q -e "sink(stdout(), type = 'message'); options(crayon.enabled = TRUE, Ncpus = $NCPUS, future.globals.onReference = NULL); pkgs <- readLines('/tmp/rebuild_pkgs.txt'); if (length(pkgs) == 0) { cat('Nothing to rebuild\n'); q('no') }; excluded <- jsonlite::fromJSON('local/excluded-packages.json')[['package']]; pkgs <- setdiff(pkgs, excluded); cat(sprintf('Rebuilding %d packages\n', length(pkgs))); n <- length(pkgs); for (i in seq_along(pkgs)) { x <- pkgs[i]; cat(sprintf('[%d/%d] %s\n', i, n, x)); tryCatch(bincraft::build_binary_package(x, tag_limit = 1L, patches = 'local/patches', s3_endpoint = 'https://s3.eu-central-003.backblazeb2.com', s3_region = 'eu-central-003', s3_bucket = 'devxy-rpkgs-binaries', s3_access_key_id = Sys.getenv('B2_S3_ACCESS_KEY'), s3_secret_access_key = Sys.getenv('B2_S3_SECRET_KEY'), metadata_db_host = 'r-binaries.devxy.io', metadata_db_name = 'build_metadata', metadata_db_table = 'single_builds', metadata_db_user = 'rpkgs', metadata_db_password = Sys.getenv('PGPASS'), metadata_db_sslmode = 'require', metadata_db_port = 15432, archive = TRUE, upload = TRUE, store_build_metadata = TRUE), error = function(e) cat(sprintf('ERROR building %s - %s\n', x, conditionMessage(e)))) }" 2>&1 + # A rebuild replaces objects in place, so the slot's index still advertises + # the old MD5 and, for anything that had been served from source, no Built + # stamp. Re-index here rather than waiting for the next process-updates + # run, or the rebuilt binaries stay invisible to clients until then. + # The codename is detected from the image's /etc/os-release. + - /opt/R/$R_VERSION/bin/R -q -e 'library(bincraft); upload_package_index(s3_endpoint = "https://s3.eu-central-003.backblazeb2.com", s3_region = "eu-central-003", s3_bucket = "devxy-rpkgs-binaries", s3_access_key_id = Sys.getenv("B2_S3_ACCESS_KEY"), s3_secret_access_key = Sys.getenv("B2_S3_SECRET_KEY"))' + - | + for RBIN in /opt/R/[0-9]*/bin/R; do + RMINOR=$(basename "$(dirname "$(dirname "$RBIN")")" | cut -d. -f1-2) + /opt/R/$R_VERSION/bin/R -q -e "library(bincraft); upload_package_index(r_minor = '$RMINOR', s3_endpoint = 'https://s3.eu-central-003.backblazeb2.com', s3_region = 'eu-central-003', s3_bucket = 'devxy-rpkgs-binaries', s3_access_key_id = Sys.getenv('B2_S3_ACCESS_KEY'), s3_secret_access_key = Sys.getenv('B2_S3_SECRET_KEY'))" || true + done backend_options: docker: resources: @@ -172,3 +183,25 @@ steps: limits: memory: 18Gi cpu: 3000m + + - name: Purge CDN cache + image: reg.devxy.io/docker.io/library/alpine:3.24 + environment: + OTEL_R_TRACES_EXPORTER: none + OTEL_R_LOGS_EXPORTER: none + OTEL_R_METRICS_EXPORTER: none + BUNNYNET_API_KEY: + from_secret: BUNNYNET_API_KEY + REPO_RO_TOKEN: + from_secret: REPO_RO_TOKEN + # All hostnames on the zone share this id, so one purge covers + # cran.devxy.io, cran.allianceswisspass.devxy.io and cran.rpkgs.com. + BUNNY_PULLZONE: '3857050' + commands: + - apk add --no-cache -q bash curl git + - git clone -q https://pat-s:$$REPO_RO_TOKEN@git.devxy.io/devxy/build-cran-binaries.git . + - bash scripts/purge_cdn_zone.sh "$BUNNYNET_API_KEY" "$BUNNY_PULLZONE" + # A rebuild that died part-way still replaced objects, and those are exactly + # the ones a stale edge would keep hiding, so purge either way. + when: + - status: [success, failure] diff --git a/scripts/purge_cdn_zone.sh b/scripts/purge_cdn_zone.sh new file mode 100755 index 0000000..391a814 --- /dev/null +++ b/scripts/purge_cdn_zone.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# +# Purge the entire BunnyCDN pull zone. +# +# `purge_cdn_cache.sh` purges the five index files by URL, which is right after +# a normal update: new packages arrive at new URLs, so only the index is stale. +# +# A rebuild is different. It replaces an object *in place*: a package whose +# build failed was published as its CRAN source, and the rebuilt binary takes +# exactly the same URL. The zone caches tarballs for ~370 days +# (`cache_expiration_time` in cdn.tf), so without a purge every client keeps +# receiving the source tarball for up to a year, and nothing about it looks +# wrong from the outside. +# +# Purging per URL would mean one API call per replaced package -- ~13.5k per +# arch against a rate-limited endpoint, where a single missed call leaves a +# silently stale package. One zone purge is a single call regardless of how many +# objects were replaced. The cost is a cold cache for everything else, which is +# why this is not used by the daily update path. +# +# All hostnames on the zone (cran.devxy.io, cran.allianceswisspass.devxy.io, +# cran.rpkgs.com) share pull zone 3857050, so one purge covers all of them. +# +# Usage: +# purge_cdn_zone.sh +# +set -euo pipefail + +if (($# < 2)); then + echo "usage: $0 " >&2 + exit 2 +fi + +api_key="$1" +zone_id="$2" + +echo "Purging BunnyCDN pull zone ${zone_id}" + +status=$( + curl -sS -o /tmp/purge_zone_response.txt -w '%{http_code}' -X POST \ + -H "AccessKey: ${api_key}" \ + -H "Content-Length: 0" \ + "https://api.bunny.net/pullzone/${zone_id}/purgeCache" +) + +if [[ "${status}" != "200" && "${status}" != "204" ]]; then + echo "Purge of pull zone ${zone_id} failed with HTTP ${status}:" >&2 + cat /tmp/purge_zone_response.txt >&2 + exit 1 +fi + +echo "Purged pull zone ${zone_id} (HTTP ${status})" -- 2.54.0