build-cran-binaries/local/build-all.R
pat-s 1cd86b65e1
All checks were successful
ci/crow/cron/process-updates-alpine-323-amd64 Pipeline was successful
ci/crow/cron/process-updates-alpine-323-arm64 Pipeline was successful
ci/crow/cron/process-updates-redhat-9-arm64 Pipeline was successful
ci/crow/cron/process-updates-ubuntu-2204-amd64 Pipeline was successful
ci/crow/cron/process-updates-ubuntu-2204-arm64 Pipeline was successful
ci/crow/cron/process-updates-ubuntu-2404-amd64 Pipeline was successful
feat(build): skip already-built package versions on workflow restart (#91)
## Summary

When a `build-all-*` workflow is restarted, the build job re-reads the static `pkgs_to_build.rds` that the install-deps step produced once, so it cycles over every package an interrupted run already built. This adds a DB-based skip filter so a restart only processes what is genuinely left.

- At job start, `build-all.R` queries the `single_builds` metadata table for `(name, tag)` already built successfully (`error_occurred = FALSE`) on this `platform`/`arch`, and drops those pairs from the chunk before the build loop. It logs how many it skipped.
- One indexed query, one round trip, run before the pak forks — no extra S3 listing and no new Python/s3fs memory pressure (`RPostgres`/`DBI` are already used in the container).
- Errored versions are intentionally **not** skipped, so transient failures still get retried on restart.

## Dependency

Correctness depends on a `error_occurred = FALSE` row meaning the binary is actually published. That guarantee is added in rpkgs/bincraft#56 (success row written only after a confirmed S3 upload). This PR should land together with / after a bincraft release including that fix.

Reviewed-on: #91
2026-06-16 07:27:20 +00:00

103 lines
3.6 KiB
R

sink(stdout(), type = "message")
options(crayon.enabled = TRUE, future.globals.onReference = NULL)
source(file.path("local", "r-minor-helpers.R"))
args <- commandArgs(trailingOnly = TRUE)
parsed <- parse_build_args(args)
split_into <- parsed$split_into
split_index <- parsed$split_index
ncpus <- parsed$ncpus
sensitive_only <- parsed$sensitive_only
options(Ncpus = ncpus)
# Load bincraft eagerly to avoid lazy-load memory spike during first build call
library(bincraft, quietly = TRUE)
library(future)
plan("sequential")
pkgs <- if (sensitive_only) {
readRDS("/mnt/cache/packages/r_minor_sensitive_pkgs.rds")
} else {
readRDS("/mnt/cache/packages/pkgs_to_build.rds")
}
# Back-compat: tolerate an older RDS without the column (treat all as non-sensitive)
if (is.null(pkgs$r_minor_sensitive)) {
pkgs$r_minor_sensitive <- FALSE
}
sprintf(
"Total# of remaining package versions: %s (sensitive_only=%s)",
nrow(pkgs),
sensitive_only
)
# Split into chunks for this worker
chunks <- split(pkgs, cut(seq_len(nrow(pkgs)), split_into, labels = FALSE))
chunk <- chunks[[split_index]]
sprintf("# of package versions for this job: %s", nrow(chunk))
# Exclude known problematic packages (single source of truth)
exclude <- jsonlite::fromJSON("local/excluded-packages.json")[["package"]]
chunk <- chunk[!chunk$Package %in% exclude, ]
# Skip package versions already built in a previous run.
# pkgs_to_build.rds is a static snapshot from the install-deps step, so on a
# restart it still lists everything an interrupted run already produced. The
# metadata DB reflects that progress, so we re-derive the remaining set here.
platform <- paste(Sys.getenv("OS"), gsub("[.]", "", Sys.getenv("OS_VERSION")), sep = "-")
arch <- Sys.getenv("ARCH")
con <- DBI::dbConnect(
RPostgres::Postgres(),
dbname = "build_metadata",
host = "r-binaries.devxy.io",
port = 15432,
user = "rpkgs",
password = Sys.getenv("PGPASS"),
sslmode = "require"
)
built <- DBI::dbGetQuery(
con,
"SELECT name, tag FROM single_builds WHERE platform = $1 AND arch = $2 AND error_occurred = FALSE",
params = list(platform, arch)
)
DBI::dbDisconnect(con)
before <- nrow(chunk)
chunk <- chunk[!paste(chunk$Package, chunk$Version) %in% paste(built$name, built$tag), ]
sprintf("Skipped %d already-built package versions; %d remaining for this job", before - nrow(chunk), nrow(chunk))
# Read pre-computed S3 listing from install-deps step
# This avoids loading s3fs/reticulate/Python in the build container,
# saving significant memory for pak subprocess forks
s3_cache <- readRDS("/mnt/cache/packages/s3_cache.rds")
sprintf("S3 cache: %s files", length(s3_cache))
n <- nrow(chunk)
mapply(
function(pkg, ver, sens, i) {
cat(sprintf("[%d/%d] %s_%s (r_minor_sensitive=%s)\n", i, n, pkg, ver, sens))
bincraft::build_binary_package(
pkg,
tag = ver,
is_r_minor_sensitive = isTRUE(sens),
s3_endpoint = "https://s3.eu-central-003.backblazeb2.com",
s3_region = "eu-central-003",
s3_bucket = "devxy-rpkgs-binaries",
s3_access_key_id = Sys.getenv("B2_S3_ACCESS_KEY"),
s3_secret_access_key = Sys.getenv("B2_S3_SECRET_KEY"),
s3_package_cache = s3_cache,
metadata_db_host = "r-binaries.devxy.io",
metadata_db_name = "build_metadata",
metadata_db_table = "single_builds",
metadata_db_user = "rpkgs",
metadata_db_password = Sys.getenv("PGPASS"),
metadata_db_sslmode = "require",
metadata_db_port = 15432,
archive = TRUE,
upload = TRUE,
store_build_metadata = TRUE
)
},
chunk$Package,
chunk$Version,
chunk$r_minor_sensitive,
seq_len(n)
)