Some checks are pending
ci/crow/cron/process-updates/13 Pipeline was successful
ci/crow/cron/process-updates/14 Pipeline was successful
ci/crow/cron/process-updates/15 Pipeline was successful
ci/crow/manual/weekly-rebuild-missing/5 Pipeline is running
ci/crow/cron/process-updates/16 Pipeline was successful
ci/crow/manual/weekly-rebuild-missing/6 Pipeline is running
ci/crow/cron/process-updates/18 Pipeline was successful
ci/crow/cron/process-updates/11 Pipeline was successful
ci/crow/cron/process-updates/12 Pipeline was successful
ci/crow/cron/process-updates/17 Pipeline was successful
ci/crow/cron/process-updates/6 Pipeline was successful
ci/crow/cron/process-updates/5 Pipeline was successful
ci/crow/cron/process-updates/8 Pipeline was successful
ci/crow/cron/process-updates/9 Pipeline was successful
ci/crow/cron/process-updates/3 Pipeline was successful
ci/crow/cron/process-updates/1 Pipeline was successful
ci/crow/cron/process-updates/2 Pipeline was successful
ci/crow/cron/process-updates/7 Pipeline was successful
## Problem
This is the gap flagged in rpkgs/bincraft#106. `build_binary_package()` has a fast path that compares against `s3_package_cache` instead of querying S3 per package, and that cache is produced here:
```r
s3_pkgs <- s3fs::s3_dir_ls(".../latest/src/contrib", recurse = TRUE)
saveRDS(basename(s3_pkgs), "/mnt/cache/packages/s3_cache.rds")
```
A raw bucket listing cannot tell a binary from a package whose build failed and was published as its CRAN source — the two occupy the same key. So every source fallback reads as "already built" and is skipped for good. That is how `alpine324` accumulated ~13.5k of them.
The same listing feeds `s3_dt`, which is subtracted from the build list at line 194 (`pkgs <- pkgs_no_error[!s3_dt]`). That one matters more: it excludes the very packages that need building, before `build_binary_package()` is even called.
## What this changes
Drops from the listing every object the slot's own index reports as served from source. bincraft leaves the `Built` stamp off exactly those records (rpkgs/bincraft#105), so the index already carries the answer and no credentials, downloads or extra API calls are needed.
Both consumers are fixed: the saved cache and `s3_dt`.
The cache stays a plain filename vector, so the build container still needs no `s3fs`/reticulate — that was the point of saving it in the first place.
Two deliberately conservative edges:
- archived objects have no index record, so they are kept. Unknown means binary, never "rebuild it".
- if the index cannot be read, the full listing is kept and a warning is printed, so a CDN blip cannot mass-schedule a rebuild.
## Verification
The script parses, and the new block run against the live indices:
```
amd64/alpine324: index=24235 source-served=13542 e.g. AATtools_0.0.3.tar.gz, ABCDscores_7.0.0.tar.gz
amd64/noble: index=24681 source-served=0
```
`alpine324` is re-indexed by bincraft 5.1.1, so 13 542 objects drop out and those packages become buildable. `noble` has not been re-indexed yet, so every record still carries `Built`, nothing is dropped, and its behaviour is exactly what it is today — the safe failure mode this relies on.
The new log line makes it visible per run:
```
S3 cache: N objects, M served as CRAN source, K usable binaries
```
## Sequencing
Needs rpkgs/bincraft#106 (and a release) before a rebuild actually builds: this fixes the bulk build path's list, #106 fixes the per-package pre-build skip.
Reviewed-on: #159
259 lines
7.4 KiB
R
259 lines
7.4 KiB
R
options(error = function() {
|
|
cat("ERROR:", geterrmessage(), "\n", file = stdout())
|
|
traceback(2)
|
|
q(status = 1)
|
|
})
|
|
|
|
library(bincraft, quietly = TRUE)
|
|
suppressPackageStartupMessages(library(dplyr))
|
|
library(DBI, quietly = TRUE)
|
|
library(purrr, quietly = TRUE)
|
|
library(progressr, quietly = TRUE)
|
|
suppressPackageStartupMessages(library(data.table))
|
|
|
|
# Sys.setenv("OS" = "alpine")
|
|
# Sys.setenv("OS_VERSION" = "3.22")
|
|
# Sys.setenv("ARCH" = "arm64")
|
|
|
|
arch <- Sys.getenv("ARCH")
|
|
# target: alpine-322, ubuntu-2404, redhat-9, etc.
|
|
platform <- paste(
|
|
Sys.getenv("OS"),
|
|
gsub("[.]", "", Sys.getenv("OS_VERSION")),
|
|
sep = "-"
|
|
)
|
|
# Use bincraft's codename detection for S3 paths (e.g. "rhel10" not "redhat10")
|
|
codename <- bincraft::set_codename(NULL)
|
|
|
|
con <- DBI::dbConnect(
|
|
RPostgres::Postgres(),
|
|
dbname = "build_metadata",
|
|
host = "r-binaries.devxy.io",
|
|
port = 15432,
|
|
user = "rpkgs",
|
|
password = Sys.getenv("PGPASS"),
|
|
sslmode = "require"
|
|
)
|
|
|
|
cran_archive <- tools::CRAN_archive_db()
|
|
cran_release <- tools::CRAN_package_db()
|
|
# Subset cran_archive to only those packages
|
|
cran_archive_in_release <- cran_archive[
|
|
names(cran_archive) %in% cran_release$Package
|
|
]
|
|
|
|
archive_versions <- rbindlist(
|
|
lapply(names(cran_archive), function(pkg) {
|
|
df <- as.data.table(cran_archive_in_release[[pkg]])
|
|
file_names <- rownames(cran_archive_in_release[[pkg]]) # Get row names from the original data.frame!
|
|
versions <- sub(".*_(.*)\\.tar\\.gz$", "\\1", file_names)
|
|
data.table(
|
|
Package = pkg,
|
|
Version = versions,
|
|
mtime = df$mtime
|
|
)
|
|
}),
|
|
fill = TRUE
|
|
)
|
|
|
|
# Select the 4 most recent archive versions by mtime for each package
|
|
# Combined with the 1 release version = 5 versions per package
|
|
archive_versions <- archive_versions[
|
|
order(Package, -as.numeric(mtime))
|
|
][,
|
|
head(.SD, 4),
|
|
by = Package
|
|
][,
|
|
.(Package, Version)
|
|
]
|
|
|
|
# Now get release versions (assuming cran_release has Package and Version columns)
|
|
release_versions <- data.table(
|
|
Package = cran_release$Package,
|
|
Version = as.character(cran_release$Version)
|
|
)
|
|
|
|
pkgs_to_build <- unique(rbind(archive_versions, release_versions, fill = TRUE))
|
|
setorder(pkgs_to_build, Package, Version)
|
|
|
|
### Get all packages in S3
|
|
s3fs::s3_file_system(
|
|
aws_access_key_id = Sys.getenv("B2_S3_ACCESS_KEY"),
|
|
aws_secret_access_key = Sys.getenv("B2_S3_SECRET_KEY"),
|
|
endpoint = "https://s3.eu-central-003.backblazeb2.com",
|
|
region_name = "eu-central-003",
|
|
refresh = TRUE
|
|
)
|
|
s3_pkgs <- s3fs::s3_dir_ls(
|
|
sprintf("devxy-rpkgs-binaries/%s/%s/latest/src/contrib", arch, codename),
|
|
recurse = TRUE
|
|
)
|
|
|
|
file_names <- basename(s3_pkgs)
|
|
|
|
# An object occupying a key is not proof a binary was built: a package whose
|
|
# build failed has its CRAN source published under exactly that name. Left in
|
|
# the cache, `build_binary_package()` reads it as "already built" and skips the
|
|
# package forever, which is how alpine324 accumulated ~13.5k source tarballs.
|
|
#
|
|
# bincraft stamps `Built` only on records it actually built, so the slot's own
|
|
# index distinguishes them. A slot last indexed by a bincraft that predates that
|
|
# fix stamps `Built` on everything, so the cache is then unchanged from before.
|
|
# Archived objects have no index record and are kept: unknown means binary,
|
|
# never "rebuild it".
|
|
index_url <- sprintf(
|
|
"https://cran.rpkgs.com/%s/%s/latest/src/contrib/PACKAGES.gz",
|
|
arch,
|
|
codename
|
|
)
|
|
source_served <- tryCatch(
|
|
{
|
|
con_idx <- gzcon(url(index_url, open = "rb"))
|
|
on.exit(close(con_idx), add = TRUE)
|
|
idx <- read.dcf(con_idx, fields = c("Package", "Version", "Built"))
|
|
sprintf(
|
|
"%s_%s.tar.gz",
|
|
idx[is.na(idx[, "Built"]), "Package"],
|
|
idx[is.na(idx[, "Built"]), "Version"]
|
|
)
|
|
},
|
|
error = function(e) {
|
|
cat(sprintf(
|
|
"WARNING: could not read %s (%s); keeping the full S3 cache\n",
|
|
index_url,
|
|
conditionMessage(e)
|
|
))
|
|
character(0)
|
|
}
|
|
)
|
|
|
|
binary_cache <- setdiff(file_names, source_served)
|
|
cat(sprintf(
|
|
"S3 cache: %d objects, %d served as CRAN source, %d usable binaries\n",
|
|
length(file_names),
|
|
length(file_names) - length(binary_cache),
|
|
length(binary_cache)
|
|
))
|
|
|
|
# Save the S3 file listing for the build step to use as s3_package_cache.
|
|
# This avoids loading s3fs/reticulate in the build container, saving memory for
|
|
# the dependency-installer subprocesses
|
|
saveRDS(binary_cache, "/mnt/cache/packages/s3_cache.rds")
|
|
# Built from the filtered listing, not the raw one: `s3_dt` is subtracted from
|
|
# the build list below, so a source fallback left in here would exclude the very
|
|
# package that needs building.
|
|
matches <- regexec("^([A-Za-z0-9.]+)_([0-9][^/]*)\\.tar\\.gz$", binary_cache)
|
|
parts <- regmatches(binary_cache, matches)
|
|
parts <- parts[sapply(parts, length) == 3]
|
|
s3_dt <- data.table(
|
|
Package = sapply(parts, `[`, 2),
|
|
Version = sapply(parts, `[`, 3)
|
|
)
|
|
|
|
### Get all packages with build errors
|
|
|
|
sql_query <- paste0(
|
|
# nolint
|
|
"SELECT error_occurred FROM ",
|
|
"single_builds",
|
|
" WHERE name = $1 AND tag = $2 AND platform = $3 AND arch = $4"
|
|
)
|
|
# Function to query for a single package-version
|
|
query_error <- function(pkg, ver) {
|
|
purrr::insistently(
|
|
~ DBI::dbGetQuery(
|
|
con,
|
|
sql_query,
|
|
params = list(pkg, ver, platform, arch)
|
|
),
|
|
rate = purrr::rate_backoff(
|
|
pause_base = 1L,
|
|
pause_cap = 60L,
|
|
pause_min = 1L,
|
|
max_times = 10L,
|
|
jitter = FALSE
|
|
),
|
|
quiet = FALSE
|
|
)()
|
|
}
|
|
|
|
# Fetch all relevant columns from the database
|
|
errored_pkgs <- DBI::dbGetQuery(
|
|
con,
|
|
sprintf(
|
|
"SELECT name, tag FROM single_builds WHERE error_occurred = TRUE and platform='%s' and arch='%s'",
|
|
platform,
|
|
arch
|
|
)
|
|
)
|
|
errored_pkgs <- as.data.table(errored_pkgs)
|
|
setkey(pkgs_to_build, Package, Version)
|
|
setnames(errored_pkgs, c("Package", "Version"))
|
|
setkey(errored_pkgs, Package, Version)
|
|
|
|
### Final subsetting
|
|
pkgs_no_error <- pkgs_to_build[!errored_pkgs]
|
|
# Return the full (Package, Version) pairs that need building
|
|
pkgs <- pkgs_no_error[!s3_dt]
|
|
# Deduplicate
|
|
pkgs <- unique(pkgs)
|
|
setorder(pkgs, Package, Version)
|
|
|
|
### R-minor sensitivity (classify once per package, applied to all versions)
|
|
source(file.path("local", "r-minor-helpers.R"))
|
|
risky_deps <- bincraft::abi_risky_linking_deps()
|
|
|
|
release_meta <- data.table(
|
|
Package = cran_release$Package,
|
|
NeedsCompilation = cran_release$NeedsCompilation,
|
|
LinkingTo = cran_release$LinkingTo
|
|
)
|
|
|
|
meta <- release_meta[Package %in% unique(pkgs$Package)]
|
|
meta[,
|
|
triage := mapply(
|
|
classify_from_metadata,
|
|
NeedsCompilation,
|
|
LinkingTo,
|
|
MoreArgs = list(risky_deps = risky_deps)
|
|
)
|
|
]
|
|
|
|
# Only the "ambiguous" compiled packages need a source grep.
|
|
ambiguous <- meta[triage == "ambiguous", Package]
|
|
sensitive_ambiguous <- character()
|
|
if (length(ambiguous) > 0L) {
|
|
tmp_src <- file.path(tempdir(), "abi_src")
|
|
dir.create(tmp_src, showWarnings = FALSE, recursive = TRUE)
|
|
sens <- vapply(
|
|
ambiguous,
|
|
function(pkg) {
|
|
out <- tryCatch(
|
|
{
|
|
dl <- utils::download.packages(
|
|
pkg,
|
|
destdir = tmp_src,
|
|
repos = "https://cloud.r-project.org",
|
|
quiet = TRUE
|
|
)
|
|
isTRUE(as.logical(bincraft::needs_per_minor_recompile(dl[1L, 2L])))
|
|
},
|
|
error = function(e) TRUE
|
|
) # fail safe: treat as sensitive
|
|
out
|
|
},
|
|
logical(1L)
|
|
)
|
|
sensitive_ambiguous <- ambiguous[sens]
|
|
}
|
|
|
|
sensitive_pkgs <- unique(c(
|
|
meta[triage == "sensitive", Package],
|
|
sensitive_ambiguous
|
|
))
|
|
pkgs[, r_minor_sensitive := Package %in% sensitive_pkgs]
|
|
sprintf(
|
|
"R-minor-sensitive packages: %s of %s",
|
|
length(sensitive_pkgs),
|
|
uniqueN(pkgs$Package)
|
|
)
|