Some checks failed
ci/crow/cron/process-updates/7 Pipeline was successful
ci/crow/cron/process-updates/3 Pipeline was successful
ci/crow/cron/process-updates/9 Pipeline was canceled
ci/crow/cron/process-updates/4 Pipeline was successful
ci/crow/manual/build-all-versions-install-deps/2 Pipeline was successful
ci/crow/cron/process-updates/13 Pipeline was canceled
ci/crow/cron/process-updates/8 Pipeline failed
ci/crow/cron/process-updates/10 Pipeline was successful
ci/crow/cron/process-updates/17 Pipeline was canceled
ci/crow/cron/process-updates/11 Pipeline was successful
ci/crow/cron/process-updates/5 Pipeline failed
ci/crow/manual/build-all-versions-install-deps/1 Pipeline was successful
ci/crow/cron/process-updates/15 Pipeline failed
ci/crow/cron/process-updates/1 Pipeline was successful
ci/crow/cron/process-updates/14 Pipeline was successful
ci/crow/manual/build-all-versions/2 Pipeline was successful
ci/crow/cron/process-updates/16 Pipeline was successful
ci/crow/manual/build-all-versions/3 Pipeline failed
ci/crow/manual/build-all-versions/1 Pipeline failed
ci/crow/cron/process-updates/18 Pipeline was successful
ci/crow/manual/build-all-versions/6 Pipeline failed
ci/crow/manual/build-all-versions/5 Pipeline was canceled
ci/crow/manual/build-all-versions/8 Pipeline was canceled
ci/crow/cron/process-updates/12 Pipeline was canceled
ci/crow/manual/build-all-versions/7 Pipeline was canceled
ci/crow/cron/process-updates/6 Pipeline failed
ci/crow/cron/process-updates/2 Pipeline failed
ci/crow/manual/build-all-versions/4 Pipeline was canceled
## Problem
The arm64 `build-all` pipeline fills the macmini (gaia) host disk despite an 8h prune.
Root cause is not images or job volumes: it is the persistent dep-cache volume, specifically `pkgcache/R/pkgcache/_metadata`, which grew to ~165 GB.
`{pkgcache}` mints a new content hash for the "patched" binaries repo on every PACKAGES change, so each per-package build writes a fresh ~70 MB `pkgs-<hash>.rds` (+ `patched-<hash>/`) that is never evicted (2407 snapshots observed).
When the disk hits 100% OrbStack stops and the on-host prune can no longer connect to the daemon, so it never self-heals.
## Change (Workstream A of the disk-fill fix)
- Add `trim_pkgcache_metadata()` to `local/r-minor-helpers.R`: keeps the newest `keep` (default 20) `patched-*`/`pkgs-*.rds` entries under `_metadata`, deleting only entries older than `min_age_secs` (default 600s) so it never races the up-to-4 concurrent split-jobs sharing the volume.
Preserves `pkg/` downloads and the stable CRAN/BioC/INLA repo dirs.
No-op when `R_PKG_CACHE_DIR` is empty (amd64) or `_metadata` is absent (first run).
- Call it every 25 packages inside the build loop in `local/build-all.R`.
- Add a defensive start-of-run cleanup of `_metadata/patched-*` + `pkgs-*.rds` to the two workflows that mount the persistent volume (`build-all-versions.yaml`, `build-all-versions-install-deps.yaml`).
Only these paths are touched; `process-updates.yaml`/`weekly-rebuild-missing.yaml` (no persistent volume) are unchanged.
Follow-ups (separate workstreams): on-host self-healing prune watcher + OrbStack disk cap (ansible), and Prometheus/Grafana alerting (k8s-talos).
Upstream: bincraft patched-repo hash churn is the true source fix.
New unit tests (6) for the helper; full suite 24/24 green.
Reviewed-on: #110
165 lines
6.2 KiB
R
165 lines
6.2 KiB
R
sink(stdout(), type = "message")
|
|
options(crayon.enabled = TRUE, future.globals.onReference = NULL)
|
|
source(file.path("local", "r-minor-helpers.R"))
|
|
|
|
args <- commandArgs(trailingOnly = TRUE)
|
|
parsed <- parse_build_args(args)
|
|
split_into <- parsed$split_into
|
|
split_index <- parsed$split_index
|
|
ncpus <- parsed$ncpus
|
|
sensitive_only <- parsed$sensitive_only
|
|
options(Ncpus = ncpus)
|
|
|
|
# Load bincraft eagerly to avoid lazy-load memory spike during first build call
|
|
library(bincraft, quietly = TRUE)
|
|
library(future)
|
|
plan("sequential")
|
|
|
|
# The install-deps step precomputes the package snapshot into /mnt/cache, but
|
|
# that volume is per-agent: a job landing on a fresh agent (or racing
|
|
# install-deps) finds it empty. Recompute the snapshot here when any part is
|
|
# missing, so the first job on an agent repopulates the cache for the jobs that
|
|
# follow; concurrent jobs that also miss simply redo the work. Write via a
|
|
# temp file + atomic rename so a concurrent reader never sees a half-written rds.
|
|
package_cache_files <- c(
|
|
"/mnt/cache/packages/pkgs_to_build.rds",
|
|
"/mnt/cache/packages/r_minor_sensitive_pkgs.rds",
|
|
"/mnt/cache/packages/s3_cache.rds"
|
|
)
|
|
if (!all(file.exists(package_cache_files))) {
|
|
message("Package snapshot missing from cache; recomputing via packages-to-build.R")
|
|
dir.create("/mnt/cache/packages", showWarnings = FALSE, recursive = TRUE)
|
|
save_rds_atomic <- function(obj, path) {
|
|
tmp <- paste0(path, ".tmp.", Sys.getpid())
|
|
saveRDS(obj, tmp)
|
|
file.rename(tmp, path)
|
|
}
|
|
source(file.path("local", "packages-to-build.R"))
|
|
save_rds_atomic(pkgs, "/mnt/cache/packages/pkgs_to_build.rds")
|
|
save_rds_atomic(pkgs[r_minor_sensitive == TRUE], "/mnt/cache/packages/r_minor_sensitive_pkgs.rds")
|
|
message("Package snapshot recomputed.")
|
|
}
|
|
|
|
pkgs <- if (sensitive_only) {
|
|
readRDS("/mnt/cache/packages/r_minor_sensitive_pkgs.rds")
|
|
} else {
|
|
readRDS("/mnt/cache/packages/pkgs_to_build.rds")
|
|
}
|
|
# Back-compat: tolerate an older RDS without the column (treat all as non-sensitive)
|
|
if (is.null(pkgs$r_minor_sensitive)) {
|
|
pkgs$r_minor_sensitive <- FALSE
|
|
}
|
|
sprintf(
|
|
"Total# of remaining package versions: %s (sensitive_only=%s)",
|
|
nrow(pkgs),
|
|
sensitive_only
|
|
)
|
|
|
|
# Split into chunks for this worker
|
|
chunks <- split(pkgs, cut(seq_len(nrow(pkgs)), split_into, labels = FALSE))
|
|
chunk <- chunks[[split_index]]
|
|
sprintf("# of package versions for this job: %s", nrow(chunk))
|
|
|
|
# Exclude known problematic packages (single source of truth)
|
|
exclude <- jsonlite::fromJSON("local/excluded-packages.json")[["package"]]
|
|
chunk <- chunk[!chunk$Package %in% exclude, ]
|
|
|
|
# Skip package versions already attempted in a previous run (built or errored).
|
|
# pkgs_to_build.rds is a static snapshot from the install-deps step, so on a
|
|
# restart it still lists everything an interrupted run already produced. The
|
|
# metadata DB reflects that progress, so we re-derive the remaining set here.
|
|
# We exclude *all* attempted versions, not just successful ones: a previously
|
|
# errored version is skipped by build_binary_package() anyway, so leaving it in
|
|
# the chunk only makes the job cycle through it one-by-one for no benefit.
|
|
# Derive platform + arch from the running container, mirroring the codename ->
|
|
# platform mapping bincraft uses internally. The OS/OS_VERSION selectors are
|
|
# workflow-level CI variables that are not injected into the container
|
|
# environment, so Sys.getenv() would return "" and this pre-filter would query
|
|
# platform "-" and skip nothing.
|
|
codename <- bincraft::set_codename(NULL)
|
|
platform <- switch(
|
|
codename,
|
|
jammy = "ubuntu-2204",
|
|
noble = "ubuntu-2404",
|
|
resolute = "ubuntu-2604",
|
|
rhel10 = "redhat-10",
|
|
rhel9 = "redhat-9",
|
|
rhel8 = "redhat-8",
|
|
alpine320 = "alpine-320",
|
|
alpine321 = "alpine-321",
|
|
alpine322 = "alpine-322",
|
|
alpine323 = "alpine-323",
|
|
alpine324 = "alpine-324",
|
|
alpine325 = "alpine-325",
|
|
alpine326 = "alpine-326",
|
|
NA_character_
|
|
)
|
|
local_machine <- Sys.info()[["machine"]]
|
|
arch <- if (grepl("arm64|aarch64", local_machine)) "arm64" else "amd64"
|
|
con <- DBI::dbConnect(
|
|
RPostgres::Postgres(),
|
|
dbname = "build_metadata",
|
|
host = "r-binaries.devxy.io",
|
|
port = 15432,
|
|
user = "rpkgs",
|
|
password = Sys.getenv("PGPASS"),
|
|
sslmode = "require"
|
|
)
|
|
built <- DBI::dbGetQuery(
|
|
con,
|
|
"SELECT name, tag FROM single_builds WHERE platform = $1 AND arch = $2",
|
|
params = list(platform, arch)
|
|
)
|
|
DBI::dbDisconnect(con)
|
|
before <- nrow(chunk)
|
|
chunk <- chunk[!paste(chunk$Package, chunk$Version) %in% paste(built$name, built$tag), ]
|
|
sprintf("Skipped %d already-attempted package versions; %d remaining for this job", before - nrow(chunk), nrow(chunk))
|
|
|
|
# Read pre-computed S3 listing from install-deps step
|
|
# This avoids loading s3fs/reticulate/Python in the build container,
|
|
# saving significant memory for pak subprocess forks
|
|
s3_cache <- readRDS("/mnt/cache/packages/s3_cache.rds")
|
|
sprintf("S3 cache: %s files", length(s3_cache))
|
|
|
|
n <- nrow(chunk)
|
|
# Every `trim_every` packages, bound the pkgcache _metadata dir so a full-platform
|
|
# run does not accumulate thousands of ~70 MB snapshots and fill the host disk.
|
|
# No-op on amd64 (R_PKG_CACHE_DIR is empty / cache not persisted).
|
|
trim_every <- 25L
|
|
mapply(
|
|
function(pkg, ver, sens, i) {
|
|
cat(sprintf("[%d/%d] %s_%s (r_minor_sensitive=%s)\n", i, n, pkg, ver, sens))
|
|
bincraft::build_binary_package(
|
|
pkg,
|
|
tag = ver,
|
|
is_r_minor_sensitive = isTRUE(sens),
|
|
s3_endpoint = "https://s3.eu-central-003.backblazeb2.com",
|
|
s3_region = "eu-central-003",
|
|
s3_bucket = "devxy-rpkgs-binaries",
|
|
s3_access_key_id = Sys.getenv("B2_S3_ACCESS_KEY"),
|
|
s3_secret_access_key = Sys.getenv("B2_S3_SECRET_KEY"),
|
|
s3_package_cache = s3_cache,
|
|
metadata_db_host = "r-binaries.devxy.io",
|
|
metadata_db_name = "build_metadata",
|
|
metadata_db_table = "single_builds",
|
|
metadata_db_user = "rpkgs",
|
|
metadata_db_password = Sys.getenv("PGPASS"),
|
|
metadata_db_sslmode = "require",
|
|
metadata_db_port = 15432,
|
|
archive = TRUE,
|
|
patches = "local/patches",
|
|
upload = TRUE,
|
|
store_build_metadata = TRUE
|
|
)
|
|
if (i %% trim_every == 0L) {
|
|
removed <- trim_pkgcache_metadata()
|
|
if (removed > 0L) {
|
|
cat(sprintf(" [pkgcache trim] removed %d stale _metadata entries\n", removed))
|
|
}
|
|
}
|
|
},
|
|
chunk$Package,
|
|
chunk$Version,
|
|
chunk$r_minor_sensitive,
|
|
seq_len(n)
|
|
)
|