build-cran-binaries/local/packages-to-build.R
pat-s 6b0a987a6a
fix(build): keep source fallbacks out of the S3 package cache
The cache handed to build_binary_package() as s3_package_cache was the raw
bucket listing, and the build list subtracted the same listing. Neither could
tell a binary from a package whose build failed and was published as its CRAN
source, so every source fallback read as "already built" and was skipped for
good - which is how alpine324 accumulated ~13.5k of them.

- drop objects the slot's index reports as served from source, which bincraft
  marks by leaving the Built stamp off
- build s3_dt from the filtered listing too, since it is subtracted from the
  build list and would otherwise exclude the packages that need building
- log how many objects were dropped

Archived objects have no index record and are kept: unknown means binary, never
"rebuild it". A slot last indexed by a bincraft that predates the Built change
stamps everything, so its cache is unchanged from before.
2026-08-09 16:25:13 +00:00

259 lines
7.4 KiB
R

options(error = function() {
cat("ERROR:", geterrmessage(), "\n", file = stdout())
traceback(2)
q(status = 1)
})
library(bincraft, quietly = TRUE)
suppressPackageStartupMessages(library(dplyr))
library(DBI, quietly = TRUE)
library(purrr, quietly = TRUE)
library(progressr, quietly = TRUE)
suppressPackageStartupMessages(library(data.table))
# Sys.setenv("OS" = "alpine")
# Sys.setenv("OS_VERSION" = "3.22")
# Sys.setenv("ARCH" = "arm64")
arch <- Sys.getenv("ARCH")
# target: alpine-322, ubuntu-2404, redhat-9, etc.
platform <- paste(
Sys.getenv("OS"),
gsub("[.]", "", Sys.getenv("OS_VERSION")),
sep = "-"
)
# Use bincraft's codename detection for S3 paths (e.g. "rhel10" not "redhat10")
codename <- bincraft::set_codename(NULL)
con <- DBI::dbConnect(
RPostgres::Postgres(),
dbname = "build_metadata",
host = "r-binaries.devxy.io",
port = 15432,
user = "rpkgs",
password = Sys.getenv("PGPASS"),
sslmode = "require"
)
cran_archive <- tools::CRAN_archive_db()
cran_release <- tools::CRAN_package_db()
# Subset cran_archive to only those packages
cran_archive_in_release <- cran_archive[
names(cran_archive) %in% cran_release$Package
]
archive_versions <- rbindlist(
lapply(names(cran_archive), function(pkg) {
df <- as.data.table(cran_archive_in_release[[pkg]])
file_names <- rownames(cran_archive_in_release[[pkg]]) # Get row names from the original data.frame!
versions <- sub(".*_(.*)\\.tar\\.gz$", "\\1", file_names)
data.table(
Package = pkg,
Version = versions,
mtime = df$mtime
)
}),
fill = TRUE
)
# Select the 4 most recent archive versions by mtime for each package
# Combined with the 1 release version = 5 versions per package
archive_versions <- archive_versions[
order(Package, -as.numeric(mtime))
][,
head(.SD, 4),
by = Package
][,
.(Package, Version)
]
# Now get release versions (assuming cran_release has Package and Version columns)
release_versions <- data.table(
Package = cran_release$Package,
Version = as.character(cran_release$Version)
)
pkgs_to_build <- unique(rbind(archive_versions, release_versions, fill = TRUE))
setorder(pkgs_to_build, Package, Version)
### Get all packages in S3
s3fs::s3_file_system(
aws_access_key_id = Sys.getenv("B2_S3_ACCESS_KEY"),
aws_secret_access_key = Sys.getenv("B2_S3_SECRET_KEY"),
endpoint = "https://s3.eu-central-003.backblazeb2.com",
region_name = "eu-central-003",
refresh = TRUE
)
s3_pkgs <- s3fs::s3_dir_ls(
sprintf("devxy-rpkgs-binaries/%s/%s/latest/src/contrib", arch, codename),
recurse = TRUE
)
file_names <- basename(s3_pkgs)
# An object occupying a key is not proof a binary was built: a package whose
# build failed has its CRAN source published under exactly that name. Left in
# the cache, `build_binary_package()` reads it as "already built" and skips the
# package forever, which is how alpine324 accumulated ~13.5k source tarballs.
#
# bincraft stamps `Built` only on records it actually built, so the slot's own
# index distinguishes them. A slot last indexed by a bincraft that predates that
# fix stamps `Built` on everything, so the cache is then unchanged from before.
# Archived objects have no index record and are kept: unknown means binary,
# never "rebuild it".
index_url <- sprintf(
"https://cran.rpkgs.com/%s/%s/latest/src/contrib/PACKAGES.gz",
arch,
codename
)
source_served <- tryCatch(
{
con_idx <- gzcon(url(index_url, open = "rb"))
on.exit(close(con_idx), add = TRUE)
idx <- read.dcf(con_idx, fields = c("Package", "Version", "Built"))
sprintf(
"%s_%s.tar.gz",
idx[is.na(idx[, "Built"]), "Package"],
idx[is.na(idx[, "Built"]), "Version"]
)
},
error = function(e) {
cat(sprintf(
"WARNING: could not read %s (%s); keeping the full S3 cache\n",
index_url,
conditionMessage(e)
))
character(0)
}
)
binary_cache <- setdiff(file_names, source_served)
cat(sprintf(
"S3 cache: %d objects, %d served as CRAN source, %d usable binaries\n",
length(file_names),
length(file_names) - length(binary_cache),
length(binary_cache)
))
# Save the S3 file listing for the build step to use as s3_package_cache.
# This avoids loading s3fs/reticulate in the build container, saving memory for
# the dependency-installer subprocesses
saveRDS(binary_cache, "/mnt/cache/packages/s3_cache.rds")
# Built from the filtered listing, not the raw one: `s3_dt` is subtracted from
# the build list below, so a source fallback left in here would exclude the very
# package that needs building.
matches <- regexec("^([A-Za-z0-9.]+)_([0-9][^/]*)\\.tar\\.gz$", binary_cache)
parts <- regmatches(binary_cache, matches)
parts <- parts[sapply(parts, length) == 3]
s3_dt <- data.table(
Package = sapply(parts, `[`, 2),
Version = sapply(parts, `[`, 3)
)
### Get all packages with build errors
sql_query <- paste0(
# nolint
"SELECT error_occurred FROM ",
"single_builds",
" WHERE name = $1 AND tag = $2 AND platform = $3 AND arch = $4"
)
# Function to query for a single package-version
query_error <- function(pkg, ver) {
purrr::insistently(
~ DBI::dbGetQuery(
con,
sql_query,
params = list(pkg, ver, platform, arch)
),
rate = purrr::rate_backoff(
pause_base = 1L,
pause_cap = 60L,
pause_min = 1L,
max_times = 10L,
jitter = FALSE
),
quiet = FALSE
)()
}
# Fetch all relevant columns from the database
errored_pkgs <- DBI::dbGetQuery(
con,
sprintf(
"SELECT name, tag FROM single_builds WHERE error_occurred = TRUE and platform='%s' and arch='%s'",
platform,
arch
)
)
errored_pkgs <- as.data.table(errored_pkgs)
setkey(pkgs_to_build, Package, Version)
setnames(errored_pkgs, c("Package", "Version"))
setkey(errored_pkgs, Package, Version)
### Final subsetting
pkgs_no_error <- pkgs_to_build[!errored_pkgs]
# Return the full (Package, Version) pairs that need building
pkgs <- pkgs_no_error[!s3_dt]
# Deduplicate
pkgs <- unique(pkgs)
setorder(pkgs, Package, Version)
### R-minor sensitivity (classify once per package, applied to all versions)
source(file.path("local", "r-minor-helpers.R"))
risky_deps <- bincraft::abi_risky_linking_deps()
release_meta <- data.table(
Package = cran_release$Package,
NeedsCompilation = cran_release$NeedsCompilation,
LinkingTo = cran_release$LinkingTo
)
meta <- release_meta[Package %in% unique(pkgs$Package)]
meta[,
triage := mapply(
classify_from_metadata,
NeedsCompilation,
LinkingTo,
MoreArgs = list(risky_deps = risky_deps)
)
]
# Only the "ambiguous" compiled packages need a source grep.
ambiguous <- meta[triage == "ambiguous", Package]
sensitive_ambiguous <- character()
if (length(ambiguous) > 0L) {
tmp_src <- file.path(tempdir(), "abi_src")
dir.create(tmp_src, showWarnings = FALSE, recursive = TRUE)
sens <- vapply(
ambiguous,
function(pkg) {
out <- tryCatch(
{
dl <- utils::download.packages(
pkg,
destdir = tmp_src,
repos = "https://cloud.r-project.org",
quiet = TRUE
)
isTRUE(as.logical(bincraft::needs_per_minor_recompile(dl[1L, 2L])))
},
error = function(e) TRUE
) # fail safe: treat as sensitive
out
},
logical(1L)
)
sensitive_ambiguous <- ambiguous[sens]
}
sensitive_pkgs <- unique(c(
meta[triage == "sensitive", Package],
sensitive_ambiguous
))
pkgs[, r_minor_sensitive := Package %in% sensitive_pkgs]
sprintf(
"R-minor-sensitive packages: %s of %s",
length(sensitive_pkgs),
uniqueN(pkgs$Package)
)