mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-08-29 02:22:10 +03:00
Compare commits
9 Commits
dependabot
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
24c0643a94 | ||
|
|
226538fa27 | ||
|
|
f907b5ea8e | ||
|
|
5b38ec717d | ||
|
|
f564b64f7d | ||
|
|
9dc8eab70e | ||
|
|
e71be03398 | ||
|
|
09de69edc7 | ||
|
|
e4683cd22d |
43
.github/workflows/ci.yml
vendored
43
.github/workflows/ci.yml
vendored
@@ -606,13 +606,24 @@ jobs:
|
||||
# Dynamic runner: when the release captain flips the USE_VPS_RUNNER repo var to
|
||||
# 'true' (scripts/vps/release-runner-up.sh does it after the self-hosted VM is
|
||||
# online), the heavy jobs run on the dedicated 32-core VPS runners (label
|
||||
# omni-release) instead of queueing on the 20-concurrent-job hosted pool.
|
||||
# omni-build) instead of queueing on the 20-concurrent-job hosted pool.
|
||||
# Safety: fork PRs NEVER reach the self-hosted runner — the expression falls
|
||||
# back to ubuntu-latest unless the PR head repo is this repository (push /
|
||||
# dispatch events are own-origin by definition). Any failure path (VM down,
|
||||
# var unset/false) also falls back to ubuntu-latest.
|
||||
runs-on: ${{ (vars.USE_VPS_RUNNER == 'true' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository)) && fromJSON('["self-hosted","omni-release"]') || 'ubuntu-latest' }}
|
||||
runs-on: ${{ (vars.USE_VPS_RUNNER == 'true' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository)) && fromJSON('["self-hosted","omni-build"]') || 'ubuntu-latest' }}
|
||||
needs: changes
|
||||
# The .113 pool runs ONE next-build with room to spare and two at the edge: the
|
||||
# box has 31 GB and a single next-build peaks at 14–16 GB RSS. On 2026-08-28
|
||||
# 13:50Z the kernel OOM-killed main's build while a PR build ran beside it
|
||||
# (five Build jobs had been queued by a burst of PRs). Two lanes: main keeps
|
||||
# its own so a release is never queued behind PR traffic; PR builds serialize
|
||||
# among themselves. GitHub keeps one running + one pending per group and
|
||||
# CANCELS older pendings — a cancelled PR build is re-runnable; a dead main
|
||||
# build costs the publish its artefact and a 40-minute rebuild that OOMs.
|
||||
concurrency:
|
||||
group: heavy-build-${{ github.ref == 'refs/heads/main' && 'main' || 'pr' }}
|
||||
cancel-in-progress: false
|
||||
if: ${{ github.event_name != 'pull_request' || (needs.changes.outputs.code == 'true' && github.event.pull_request.draft == false) }}
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
@@ -646,14 +657,14 @@ jobs:
|
||||
# Keep standalone/node_modules intact: package/electron jobs consume the
|
||||
# Next-traced standalone tree and must not replace it with root node_modules.
|
||||
run: |
|
||||
tar -czf /tmp/e2e-build.tar.gz \
|
||||
tar -czf "$RUNNER_TEMP/e2e-build.tar.gz" \
|
||||
--exclude='.build/next/cache' \
|
||||
.build/next
|
||||
- name: Upload Next.js build for downstream jobs
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: next-build
|
||||
path: /tmp/e2e-build.tar.gz
|
||||
path: ${{ runner.temp }}/e2e-build.tar.gz
|
||||
retention-days: 1
|
||||
|
||||
package-artifact:
|
||||
@@ -676,10 +687,14 @@ jobs:
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: next-build
|
||||
path: /tmp/
|
||||
# Workspace-relative on purpose: the matrix below includes windows-latest, whose
|
||||
# default shell is pwsh, where $RUNNER_TEMP is empty (it is $env:RUNNER_TEMP) —
|
||||
# #11896's first cut broke the Electron smoke on exactly that. A relative path
|
||||
# works in bash and pwsh alike; hosted workspaces are ephemeral.
|
||||
path: next-build-artifact
|
||||
- name: Extract Next.js build artifact
|
||||
run: |
|
||||
tar -xzf /tmp/e2e-build.tar.gz
|
||||
tar -xzf next-build-artifact/e2e-build.tar.gz
|
||||
# build:cli consumes the downloaded .build/next standalone artifact and assembles dist/;
|
||||
# it only rebuilds if the downloaded standalone artifact is missing.
|
||||
- run: npm run build:cli
|
||||
@@ -767,10 +782,14 @@ jobs:
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: next-build
|
||||
path: /tmp/
|
||||
# Workspace-relative on purpose: the matrix below includes windows-latest, whose
|
||||
# default shell is pwsh, where $RUNNER_TEMP is empty (it is $env:RUNNER_TEMP) —
|
||||
# #11896's first cut broke the Electron smoke on exactly that. A relative path
|
||||
# works in bash and pwsh alike; hosted workspaces are ephemeral.
|
||||
path: next-build-artifact
|
||||
- name: Extract Next.js build artifact
|
||||
run: |
|
||||
tar -xzf /tmp/e2e-build.tar.gz
|
||||
tar -xzf next-build-artifact/e2e-build.tar.gz
|
||||
- name: Install Electron dependencies
|
||||
working-directory: electron
|
||||
run: npm install --no-audit --no-fund
|
||||
@@ -1230,10 +1249,14 @@ jobs:
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: next-build
|
||||
path: /tmp/
|
||||
# Workspace-relative on purpose: the matrix below includes windows-latest, whose
|
||||
# default shell is pwsh, where $RUNNER_TEMP is empty (it is $env:RUNNER_TEMP) —
|
||||
# #11896's first cut broke the Electron smoke on exactly that. A relative path
|
||||
# works in bash and pwsh alike; hosted workspaces are ephemeral.
|
||||
path: next-build-artifact
|
||||
- name: Extract Next.js build artifact
|
||||
run: |
|
||||
tar -xzf /tmp/e2e-build.tar.gz
|
||||
tar -xzf next-build-artifact/e2e-build.tar.gz
|
||||
# WS4.1: duration-balanced shards (LPT over config/quality/e2e-timings.json).
|
||||
# Measured skew of plain --shard was 14× (24m47s vs 1m47s) — E2E was the CI
|
||||
# critical path. The balancer self-verifies completeness and exits non-zero on
|
||||
|
||||
4
.github/workflows/nightly-release-green.yml
vendored
4
.github/workflows/nightly-release-green.yml
vendored
@@ -68,7 +68,7 @@ jobs:
|
||||
# this runs on the dedicated VPS runner — clean env (no operator OMNIROUTE_API_KEY,
|
||||
# no local noauth CLIs => zero machine-specific false positives) and no contention.
|
||||
# Nightly cron normally finds the var false (VM off) and falls back to hosted.
|
||||
runs-on: ${{ (vars.USE_VPS_RUNNER == 'true' && fromJSON('["self-hosted","omni-release"]')) || 'ubuntu-latest' }}
|
||||
runs-on: ${{ (vars.USE_VPS_RUNNER == 'true' && fromJSON('["self-hosted","omni-build"]')) || 'ubuntu-latest' }}
|
||||
env:
|
||||
JWT_SECRET: ci-nightly-secret-with-sufficient-length-for-validation
|
||||
API_KEY_SECRET: ci-nightly-api-key-secret-long
|
||||
@@ -217,7 +217,7 @@ jobs:
|
||||
# On a push, only run for a push to main — a push to release/* is handled by
|
||||
# release-green above. Schedule/dispatch always run (they also sweep main).
|
||||
if: ${{ github.event_name != 'push' || github.ref_name == 'main' }}
|
||||
runs-on: ${{ (vars.USE_VPS_RUNNER == 'true' && fromJSON('["self-hosted","omni-release"]')) || 'ubuntu-latest' }}
|
||||
runs-on: ${{ (vars.USE_VPS_RUNNER == 'true' && fromJSON('["self-hosted","omni-build"]')) || 'ubuntu-latest' }}
|
||||
env:
|
||||
JWT_SECRET: ci-nightly-secret-with-sufficient-length-for-validation
|
||||
API_KEY_SECRET: ci-nightly-api-key-secret-long
|
||||
|
||||
44
.github/workflows/npm-publish.yml
vendored
44
.github/workflows/npm-publish.yml
vendored
@@ -23,11 +23,12 @@ on:
|
||||
- next
|
||||
- historic
|
||||
publish_mode:
|
||||
description: "staged = npm stage publish (owner approves with 2FA after the staged boot-verify); direct = legacy immediate publish (emergency fallback only)"
|
||||
description: "auto = publish through npm Trusted Publishing (OIDC, no token, no 2FA prompt — the default); staged = npm stage publish (owner approves with 2FA); direct = legacy token publish (emergency fallback only)"
|
||||
required: false
|
||||
default: "staged"
|
||||
default: "auto"
|
||||
type: choice
|
||||
options:
|
||||
- auto
|
||||
- staged
|
||||
- direct
|
||||
workflow_call:
|
||||
@@ -62,7 +63,7 @@ jobs:
|
||||
# mid-"Creating an optimized production build" while v3.8.48 had still fit in 16min.
|
||||
# This job never runs on `pull_request`, so the fork-safety clause is always true here;
|
||||
# it is kept verbatim so the expression stays greppable against ci.yml.
|
||||
runs-on: ${{ (vars.USE_VPS_RUNNER == 'true' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository)) && fromJSON('["self-hosted","omni-release"]') || 'ubuntu-latest' }}
|
||||
runs-on: ${{ (vars.USE_VPS_RUNNER == 'true' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository)) && fromJSON('["self-hosted","omni-build"]') || 'ubuntu-latest' }}
|
||||
outputs:
|
||||
version: ${{ steps.resolve.outputs.version }}
|
||||
tag: ${{ steps.resolve.outputs.tag }}
|
||||
@@ -204,8 +205,11 @@ jobs:
|
||||
exit 0
|
||||
fi
|
||||
RUN=""
|
||||
# $RUNNER_TEMP, never /tmp: on the .113 pool /tmp is a 12 GB tmpfs (RAM). Parking
|
||||
# this 1.3 GB artefact there took 27–32 min of the 76-min publish job — the
|
||||
# same bytes upload from disk in 2 min. RUNNER_TEMP is per-runner and on disk.
|
||||
for candidate in $CANDIDATES; do
|
||||
if gh run download "$candidate" --repo "$REPO" --name next-build --dir /tmp/next-build 2>/dev/null; then
|
||||
if gh run download "$candidate" --repo "$REPO" --name next-build --dir "$RUNNER_TEMP/next-build" 2>/dev/null; then
|
||||
RUN="$candidate"
|
||||
break
|
||||
fi
|
||||
@@ -215,8 +219,8 @@ jobs:
|
||||
echo "::notice::none of the candidate runs still carries next-build (1-day retention) — falling back to a full build"
|
||||
exit 0
|
||||
fi
|
||||
tar -xzf /tmp/next-build/e2e-build.tar.gz -C .
|
||||
rm -rf /tmp/next-build
|
||||
tar -xzf "$RUNNER_TEMP/next-build/e2e-build.tar.gz" -C .
|
||||
rm -rf "$RUNNER_TEMP/next-build"
|
||||
if [ -f .build/next/standalone/server.js ]; then
|
||||
echo "✅ standalone tree restored from CI run $RUN — build:cli will skip next build"
|
||||
else
|
||||
@@ -404,8 +408,34 @@ jobs:
|
||||
fi
|
||||
npm --version
|
||||
|
||||
# Trusted Publishing (OIDC): npm mints a short-lived credential for THIS run from
|
||||
# GitHub's id-token — no NPM_TOKEN secret, no 2FA prompt, provenance included, and
|
||||
# it is the bypass npm sanctions now that tokens which skip 2FA are being retired
|
||||
# (gh.io/npm-gat-bypass2fa-deprecation). Requires the package's Trusted Publisher to
|
||||
# be configured on npmjs.com (owner: diegosouzapw/OmniRoute, workflow
|
||||
# npm-publish.yml) and a github-hosted runner — which is why this job exists.
|
||||
# Without that configuration `npm publish` fails with ENEEDAUTH: re-dispatch with
|
||||
# publish_mode=staged or direct. Automatic publishing was the flow up to v3.8.48;
|
||||
# v3.8.49 moved to staged (WS1.3) to keep a leaked token from publishing alone —
|
||||
# OIDC gives the same guarantee without the manual approve.
|
||||
- name: Publish to npm (Trusted Publishing / OIDC — automatic)
|
||||
if: github.event_name != 'workflow_dispatch' || inputs.publish_mode == 'auto'
|
||||
env:
|
||||
VERSION: ${{ needs.publish.outputs.version }}
|
||||
TAG: ${{ needs.publish.outputs.tag }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
TARBALL="omniroute-${VERSION}.tgz"
|
||||
test -f "$TARBALL" || { echo "tarball $TARBALL did not arrive from the publish job" >&2; ls -la; exit 1; }
|
||||
# Deliberately NO NODE_AUTH_TOKEN in this step: npm >= 11.5 detects the GitHub
|
||||
# OIDC token itself. Always pass --tag explicitly (defense in depth: an older
|
||||
# VERSION can never claim `@latest`).
|
||||
npm publish "$TARBALL" --provenance --access public --tag "$TAG" --ignore-scripts
|
||||
echo "✅ Published omniroute@$VERSION (dist-tag=$TAG) via Trusted Publishing"
|
||||
|
||||
- name: Publish to npm (staged — owner approves with 2FA)
|
||||
if: github.event_name != 'workflow_dispatch' || inputs.publish_mode != 'direct'
|
||||
# Only on an explicit request now: Trusted Publishing below is the default.
|
||||
if: github.event_name == 'workflow_dispatch' && inputs.publish_mode == 'staged'
|
||||
env:
|
||||
VERSION: ${{ needs.publish.outputs.version }}
|
||||
TAG: ${{ needs.publish.outputs.tag }}
|
||||
|
||||
4
changelog.d/features/npm-trusted-publishing-oidc.md
Normal file
4
changelog.d/features/npm-trusted-publishing-oidc.md
Normal file
@@ -0,0 +1,4 @@
|
||||
- The npm publish is automatic again, through npm Trusted Publishing (OIDC): the hosted
|
||||
`stage-npm` job publishes with a short-lived credential minted from GitHub's id-token —
|
||||
no `NPM_TOKEN`, no 2FA prompt, provenance attached. `publish_mode=staged` (owner
|
||||
approves with 2FA) and `direct` (token) remain available on `workflow_dispatch`.
|
||||
@@ -0,0 +1,5 @@
|
||||
- Fixed the Alibaba free-tier allowlist test that went red on its own once the
|
||||
shipped catalog's `validUntil` (2026-08-27) passed, leaving every PR and `main`
|
||||
with a failing `Unit Tests (1/8)`. The test now builds its own packs with dates
|
||||
it controls, and covers the expired-pack fallback that production has actually
|
||||
been serving.
|
||||
@@ -0,0 +1,4 @@
|
||||
- Added a unit test that fails seven days before any dated pack under `config/`
|
||||
(`validUntil` and sibling keys) lapses, naming the file and key. The Alibaba
|
||||
free-tier pack expired on 2026-08-27 and turned every PR red the next morning
|
||||
with no commit involved; renewal now happens on someone's terms, not the clock's.
|
||||
@@ -0,0 +1,3 @@
|
||||
- `check:workflows` now fails (under `--strict`/`--ratchet`) when any job routed to a
|
||||
self-hosted runner publishes with `--provenance` — npm rejects that with `422` at the
|
||||
registry, which in v3.8.50 only surfaced after the tag and Docker images were public.
|
||||
@@ -0,0 +1,5 @@
|
||||
- `scripts/ops/runner-janitor.sh` now proves a path is idle with one `lsof`
|
||||
snapshot and removes stale leftovers itself (tmpfs after 3 h — it is RAM — disk
|
||||
after 24 h), kills orphan `next-build` processes, prunes checkouts of stopped
|
||||
runners, and alerts on memory pressure; `--dry-run` shows exactly what it would
|
||||
do. `docs/ops/RUNNER_BOX.md` reconciled to the measured box (31 GB, 10 listeners).
|
||||
@@ -0,0 +1,5 @@
|
||||
- The `next-build` artefact (1.3 GB) is now written and read under `$RUNNER_TEMP`
|
||||
(per-runner, on disk) instead of `/tmp`, which on the self-hosted pool is a
|
||||
12 GB tmpfs in RAM. Landing it there took 27–32 of the publish job's 76 minutes,
|
||||
and the fixed `/tmp/e2e-build.tar.gz` name let E2E jobs on different runners
|
||||
overwrite each other's download.
|
||||
3
changelog.d/maintenance/11897-ci-heavy-build-lane.md
Normal file
3
changelog.d/maintenance/11897-ci-heavy-build-lane.md
Normal file
@@ -0,0 +1,3 @@
|
||||
- The CI `build` job now runs in two concurrency lanes — `main` and pull requests —
|
||||
so a release build is never queued behind (or OOM-killed beside) PR builds on the
|
||||
self-hosted pool, which holds one `next-build` comfortably and two at the edge.
|
||||
4
changelog.d/maintenance/ci-omni-build-runner-label.md
Normal file
4
changelog.d/maintenance/ci-omni-build-runner-label.md
Normal file
@@ -0,0 +1,4 @@
|
||||
- Every CI job that runs a `next build` (`build`, the npm `publish`, both release-green
|
||||
validations) now targets the `omni-build` runner label, which only two of the eight
|
||||
self-hosted runners carry. The box holds one build comfortably and two at the edge; a
|
||||
third now queues on GitHub instead of being OOM-killed by the kernel.
|
||||
@@ -1,12 +1,12 @@
|
||||
---
|
||||
title: "Release Checklist"
|
||||
version: 3.8.40
|
||||
lastUpdated: 2026-06-28
|
||||
version: 3.8.51
|
||||
lastUpdated: 2026-08-28
|
||||
---
|
||||
|
||||
# Release Checklist
|
||||
|
||||
> **Last updated:** 2026-06-28 — v3.8.40
|
||||
> **Last updated:** 2026-08-28 — v3.8.51
|
||||
> Streamlined release flow that leverages Claude Code skills for automation.
|
||||
>
|
||||
> **Keep the queue/branch green between releases:** see [RELEASE_GREEN.md](./RELEASE_GREEN.md)
|
||||
@@ -37,7 +37,21 @@ npm run test:e2e # optional but recommended
|
||||
/capture-release-evidences-cc
|
||||
```
|
||||
|
||||
## npm Staged Publishing (default since v3.8.49 — WS1.3/D2)
|
||||
## npm Trusted Publishing (default since v3.8.51) — staged on request, direct as fallback
|
||||
|
||||
`npm-publish.yml` publishes through **npm Trusted Publishing (OIDC)** by default: the
|
||||
`stage-npm` job (github-hosted) exchanges GitHub's id-token for a short-lived npm
|
||||
credential for that run — no long-lived npm token in the repository secrets, no 2FA prompt, provenance attached.
|
||||
That is the bypass npm sanctions now that tokens which skip 2FA are being retired;
|
||||
it restores the fully automatic flow the project had up to v3.8.48 while keeping the
|
||||
WS1.3 guarantee (a leaked token cannot publish alone — there is no token).
|
||||
|
||||
**One-time setup (owner):** npmjs.com → package `omniroute` → Settings → *Trusted
|
||||
Publisher* → GitHub: owner `diegosouzapw`, repo `OmniRoute`, workflow `npm-publish.yml`
|
||||
(environment: none). Until that exists, the automatic step fails with `ENEEDAUTH`:
|
||||
re-dispatch with `publish_mode=staged` (below) or `direct`.
|
||||
|
||||
### Staged publishing (on request — `publish_mode=staged`)
|
||||
|
||||
The npm-publish workflow no longer publishes directly: it boots the packed tarball
|
||||
(`check:pack-boot`) and then runs `npm stage publish` — the exact bytes are parked on
|
||||
|
||||
@@ -4,32 +4,66 @@ title: Self-Hosted Runner Box Operations
|
||||
|
||||
# Self-Hosted Runner Box Operations (.113 pool)
|
||||
|
||||
The self-hosted pool (`self-hosted, omni-release` labels) runs on the 16 GB box at
|
||||
`192.168.0.113`. Two failure modes recurred on release days and were, until v3.8.49,
|
||||
manual discipline; the **janitor script codifies them** (WS3.3 of the quality plan):
|
||||
The self-hosted pool (`self-hosted, omni-release` on all eight runners; `omni-build` on two) runs on the **.113** box.
|
||||
Measured 2026-08-28 (v3.8.50 postmortem, Parte III):
|
||||
|
||||
1. **Orphaned temp/work dirs** filling the disk → disk-full SQLite errors mid-job.
|
||||
2. **>4 concurrent runners** → OOM-killed jobs (8-wide killed jobs twice on the
|
||||
v3.8.47 release day; 4-wide is the proven ceiling).
|
||||
| resource | value | what it means for scheduling |
|
||||
| --------- | ---------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| RAM / CPU | **31 GB / 32 cores** (was 16 GB when this doc was first written) | one `next-build` peaks at **~14 GB** → 2 concurrent heavy builds saturate the box, 3 take it down (2026-08-28 06:42Z: load 56, two jobs lost) |
|
||||
| swap | 15 GB | it swapped its way through the v3.8.50 publish; pressure shows in `/proc/pressure/memory` |
|
||||
| `/tmp` | **12 GB tmpfs = RAM** | anything parked there is memory; leftovers are swept after 3 h |
|
||||
| disk | 188 GB | `_work` checkouts of 8 runners reach ~70 GB with no cap |
|
||||
| runners | **10 listeners**: 8 OmniRoute + OmniHeuris + OmniMind | all share the memory above |
|
||||
|
||||
## Install the janitor (one-time, on the box)
|
||||
|
||||
```bash
|
||||
sudo mkdir -p /opt/omniroute-ops
|
||||
sudo cp scripts/ops/runner-janitor.sh /opt/omniroute-ops/
|
||||
sudo chmod +x /opt/omniroute-ops/runner-janitor.sh
|
||||
( sudo crontab -l 2>/dev/null; echo '*/30 * * * * /opt/omniroute-ops/runner-janitor.sh >> /var/log/runner-janitor.log 2>&1' ) | sudo crontab -
|
||||
scp scripts/ops/runner-janitor.sh root@192.168.0.113:/opt/omniroute-ops/runner-janitor.sh
|
||||
ssh root@192.168.0.113 'chmod +x /opt/omniroute-ops/runner-janitor.sh; apt-get install -y lsof'
|
||||
# cron (root): every 30 min, log to /var/log/runner-janitor.log
|
||||
*/30 * * * * MAX_ACTIVE_RUNNERS=8 /opt/omniroute-ops/runner-janitor.sh >> /var/log/runner-janitor.log 2>&1
|
||||
```
|
||||
|
||||
What it does every 30min: sweeps runner temp leftovers older than 24h, alerts at
|
||||
≥85% root-disk usage, and alerts when more than the runner ceiling (default 4, tunable
|
||||
via the script's own environment) of `Runner.Listener` processes are up. Alerts land in `/var/log/runner-janitor.log`
|
||||
with a non-zero exit (grep for `⚠`).
|
||||
`lsof` is required: the janitor proves a path is idle with one snapshot of open
|
||||
files before removing it, and without the tool it removes nothing and says so
|
||||
(exit 1). Try any change with `--dry-run` first — it prints exactly what it would
|
||||
do and touches nothing.
|
||||
|
||||
What it does every run: sweeps our own leftovers (`runner-*`, `omniroute-*`,
|
||||
`next-build*`, `e2e-build.tar.gz`) after **3 h on tmpfs** and 24 h on disk
|
||||
`_work/_temp`; kills a `next-build` older than 75 min (no job runs that long — on
|
||||
2026-08-27 one ran 70 min after GitHub had declared its job lost); prunes 48 h-old
|
||||
checkouts of runners whose unit is **stopped**; alerts on disk ≥ 85 %, memory PSI
|
||||
`full/avg60` ≥ 10 %, and more listeners than `MAX_ACTIVE_RUNNERS` (with an
|
||||
omniroute/other breakdown). Exit 1 = attention needed; read the log.
|
||||
|
||||
## Runner units: KillMode
|
||||
|
||||
The runner's default `KillMode=process` leaves `Runner.Worker → npm → next-build`
|
||||
alive when a unit is stopped or restarted — an orphan build keeps eating RAM and
|
||||
CPU with no job attached. Every OmniRoute unit carries a drop-in
|
||||
(`/etc/systemd/system/actions.runner.diegosouzapw-OmniRoute.<name>.service.d/10-killmode.conf`)
|
||||
with `KillMode=mixed`: SIGTERM to the listener first, SIGKILL to the whole cgroup at
|
||||
`TimeoutStop`. It takes effect on the unit's next restart — restart **one runner at
|
||||
a time, only when idle**, with the idle check and the restart in the same command.
|
||||
|
||||
## Operating rules
|
||||
|
||||
- **Ceiling: 4 runners** on the 16 GB box. Runners 5–8 stay STOPPED except for
|
||||
explicit off-peak experiments — never during a release window.
|
||||
- Stopping a runner mid-job cancels the job (observed live): `systemctl stop`
|
||||
only when its runner is idle (`Runner.Listener` without a `Runner.Worker` child).
|
||||
- **Heavy-build ceiling: 2 at a time — enforced by label.** Every job that runs a
|
||||
`next build` (`ci.yml` `build`, `npm-publish.yml` `publish`, both `nightly-release-green`
|
||||
validations) targets `[self-hosted, omni-build]`, and only **two** runners carry that
|
||||
label (`omniroute-113-5`, `omniroute-113-6`, added through the runners API — no
|
||||
re-registration). The other six keep `omni-release` and take nothing heavy; GitHub
|
||||
queues a third build instead of the kernel killing one. Pair with the `heavy-build-*`
|
||||
concurrency lanes in `ci.yml`. To add capacity, label another runner — never raise
|
||||
the count past what 31 GB holds (one next-build ≈ 14–16 GB).
|
||||
- **Never clean `/tmp` or `_work` by hand while any runner is busy.** A
|
||||
check-then-delete with a gap between the two is how a live Build job lost its
|
||||
`_work` on 2026-08-27. The janitor does the check and the removal in one step;
|
||||
let it.
|
||||
- Stopping a runner mid-job cancels the job (observed live): `systemctl stop` only
|
||||
when its listener has no `Runner.Worker` child — and do it in one command.
|
||||
- Workflows must not park artefacts in `/tmp` (it is RAM). Download to
|
||||
`$RUNNER_TEMP` (on disk, per runner) — the 1.3 GB `next-build` artefact took 27–32
|
||||
minutes to land on the tmpfs and 2 minutes to upload from disk.
|
||||
- The `.15` VPS is homologation-only — never runs CI runners.
|
||||
|
||||
@@ -42,6 +42,7 @@ import { execFileSync, spawnSync } from "node:child_process";
|
||||
import fs from "node:fs";
|
||||
import path from "node:path";
|
||||
import { pathToFileURL } from "node:url";
|
||||
import { findProvenanceOnSelfHosted, formatProvenanceFinding } from "./lib/provenanceRunner.mjs";
|
||||
|
||||
const ROOT = process.cwd();
|
||||
const WORKFLOWS_DIR = path.join(ROOT, ".github", "workflows");
|
||||
@@ -275,6 +276,23 @@ export function runZizmor(workflowsDir) {
|
||||
// Main
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Hard rule (not a lint count): `--provenance` inside a job that runs on a
|
||||
* self-hosted runner. npm answers 422 at the registry, and in v3.8.50 that
|
||||
* answer only came after the tag, the GitHub Release and the Docker images were
|
||||
* already out. Blocks under --strict AND --ratchet (the CI mode); plain mode
|
||||
* reports it like everything else.
|
||||
* @param {string[]} files absolute workflow paths
|
||||
*/
|
||||
export function runProvenanceRunnerCheck(files) {
|
||||
const findings = [];
|
||||
for (const file of files) {
|
||||
const text = fs.readFileSync(file, "utf8");
|
||||
findings.push(...findProvenanceOnSelfHosted(text, path.relative(ROOT, file)));
|
||||
}
|
||||
return findings;
|
||||
}
|
||||
|
||||
function main() {
|
||||
const hasActionlint = isBinaryAvailable("actionlint");
|
||||
const hasZizmor = isBinaryAvailable("zizmor");
|
||||
@@ -350,6 +368,16 @@ function main() {
|
||||
}
|
||||
}
|
||||
|
||||
const provenanceFindings = runProvenanceRunnerCheck(workflowFiles);
|
||||
if (provenanceFindings.length > 0) {
|
||||
console.error(
|
||||
`[check-workflows] provenance×self-hosted: ${provenanceFindings.length} finding(s) — HARD RULE:`
|
||||
);
|
||||
provenanceFindings.forEach((f) => console.error(` ${formatProvenanceFinding(f)}`));
|
||||
} else if (!QUIET) {
|
||||
console.log("[check-workflows] provenance×self-hosted: OK (0 findings)");
|
||||
}
|
||||
|
||||
const total = actionlintCount + zizmorCount;
|
||||
process.stdout.write(`workflowFindings=${total}\n`);
|
||||
process.stdout.write(`actionlintFindings=${actionlintCount}\n`);
|
||||
@@ -357,6 +385,15 @@ function main() {
|
||||
// Read this line with the count above: a finding total is only reproducible against the
|
||||
// version that produced it. See zizmorVersion().
|
||||
process.stdout.write(`zizmorVersion=${hasZizmor ? zizmorVersion() : "absent"}\n`);
|
||||
process.stdout.write(`provenanceRunnerFindings=${provenanceFindings.length}\n`);
|
||||
if ((STRICT || RATCHET) && provenanceFindings.length > 0) {
|
||||
console.error(
|
||||
`\n[check-workflows] FAIL — ${provenanceFindings.length} job(s) publish with --provenance from a self-hosted runner.\n` +
|
||||
" npm rejects that with 422 at the registry. Move the upload step to a github-hosted job\n" +
|
||||
" (see .github/workflows/npm-publish.yml `stage-npm` for the pattern)."
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
if (STRICT && total > 0) {
|
||||
console.error(`\n[check-workflows] FAIL — ${total} workflow finding(s) total (--strict mode).`);
|
||||
|
||||
90
scripts/check/lib/configExpiry.mjs
Normal file
90
scripts/check/lib/configExpiry.mjs
Normal file
@@ -0,0 +1,90 @@
|
||||
/**
|
||||
* scripts/check/lib/configExpiry.mjs
|
||||
*
|
||||
* Finds dated validity fields in JSON config packs so a test can fail BEFORE
|
||||
* they lapse. Origin: config/alibaba-free-tier-allowlist.json carried
|
||||
* `"validUntil": "2026-08-27"`; on 2026-08-28 the loader started (correctly)
|
||||
* rejecting the pack and a test that asserted "the shipped pack loads" turned
|
||||
* every PR and main red with no commit involved (#11866). A time bomb, not a
|
||||
* regression — and the only kind of defect a diff review can never catch.
|
||||
*
|
||||
* Pure helpers; the repo-wide assertion lives in
|
||||
* tests/unit/config-expiry-time-bomb.test.ts.
|
||||
*/
|
||||
import fs from "node:fs";
|
||||
import path from "node:path";
|
||||
|
||||
export const EXPIRY_KEY =
|
||||
/^(validUntil|valid_until|validTo|valid_to|expiresAt|expires_at|expiry|expires)$/;
|
||||
const DAY_MS = 86_400_000;
|
||||
|
||||
/**
|
||||
* Walks a parsed JSON value and returns every string-valued expiry field.
|
||||
* @returns {{ file: string, keyPath: string, raw: string, expiresAt: number|null }[]}
|
||||
*/
|
||||
export function collectExpiryFields(value, file, keyPath = []) {
|
||||
const out = [];
|
||||
if (Array.isArray(value)) {
|
||||
value.forEach((v, i) => out.push(...collectExpiryFields(v, file, [...keyPath, String(i)])));
|
||||
return out;
|
||||
}
|
||||
if (!value || typeof value !== "object") return out;
|
||||
for (const [key, v] of Object.entries(value)) {
|
||||
const kp = [...keyPath, key];
|
||||
if (EXPIRY_KEY.test(key) && typeof v === "string") {
|
||||
const ms = Date.parse(v);
|
||||
out.push({ file, keyPath: kp.join("."), raw: v, expiresAt: Number.isFinite(ms) ? ms : null });
|
||||
} else if (v && typeof v === "object") {
|
||||
out.push(...collectExpiryFields(v, file, kp));
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** @returns {"expired"|"expiring"|"ok"|"unparseable"} */
|
||||
export function classifyExpiry(field, nowMs, warnDays = 7) {
|
||||
if (field.expiresAt === null) return "unparseable";
|
||||
if (field.expiresAt < nowMs) return "expired";
|
||||
if (field.expiresAt < nowMs + warnDays * DAY_MS) return "expiring";
|
||||
return "ok";
|
||||
}
|
||||
|
||||
/** All *.json under dir, recursively, skipping node_modules. Sorted for stable output. */
|
||||
export function walkJsonFiles(dir) {
|
||||
const out = [];
|
||||
const stack = [dir];
|
||||
while (stack.length > 0) {
|
||||
const current = stack.pop();
|
||||
let entries;
|
||||
try {
|
||||
entries = fs.readdirSync(current, { withFileTypes: true });
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
for (const e of entries) {
|
||||
const full = path.join(current, e.name);
|
||||
if (e.isDirectory()) {
|
||||
if (e.name !== "node_modules") stack.push(full);
|
||||
} else if (e.isFile() && e.name.endsWith(".json")) {
|
||||
out.push(full);
|
||||
}
|
||||
}
|
||||
}
|
||||
return out.sort();
|
||||
}
|
||||
|
||||
/**
|
||||
* Scans every JSON file under `dir`; `file` in the result is relative to `dir`
|
||||
* with forward slashes, so allowlists can key on it portably.
|
||||
*/
|
||||
export function scanConfigExpiry(dir) {
|
||||
return walkJsonFiles(dir).flatMap((f) => {
|
||||
let parsed;
|
||||
try {
|
||||
parsed = JSON.parse(fs.readFileSync(f, "utf8"));
|
||||
} catch {
|
||||
return []; // not this scanner's job to validate JSON
|
||||
}
|
||||
return collectExpiryFields(parsed, path.relative(dir, f).split(path.sep).join("/"));
|
||||
});
|
||||
}
|
||||
83
scripts/check/lib/provenanceRunner.mjs
Normal file
83
scripts/check/lib/provenanceRunner.mjs
Normal file
@@ -0,0 +1,83 @@
|
||||
/**
|
||||
* scripts/check/lib/provenanceRunner.mjs
|
||||
*
|
||||
* npm refuses `--provenance` from a self-hosted runner:
|
||||
*
|
||||
* 422 Unprocessable Entity - Error verifying sigstore provenance bundle:
|
||||
* Unsupported GitHub Actions runner environment: "self-hosted".
|
||||
* Only "github-hosted" runners are supported when publishing with provenance.
|
||||
*
|
||||
* v3.8.50 hit this at the very end of a 76-minute publish job — after the tag,
|
||||
* the GitHub Release and the Docker images were already public — because
|
||||
* `USE_VPS_RUNNER` had been turned on (2026-08-02) with no release in between to
|
||||
* surface it. The combination is greppable, so it must fail in CI the moment a
|
||||
* workflow introduces it, not four weeks later at the registry.
|
||||
*
|
||||
* Pure: takes workflow YAML text, returns the offending (job, step) pairs.
|
||||
*/
|
||||
import { load as yamlLoad } from "js-yaml";
|
||||
|
||||
const SELF_HOSTED = /\bself-hosted\b/;
|
||||
const EXPRESSION = /\$\{\{/;
|
||||
// Lookahead, not \b: `--provenance-file=…` is a different flag (a pre-built
|
||||
// bundle) and must not match — a word boundary sits between "e" and "-".
|
||||
const PROVENANCE = /(^|\s)--provenance(?=\s|=|$)/m;
|
||||
|
||||
/**
|
||||
* Classifies a job's `runs-on` value.
|
||||
* @returns {"self-hosted"|"hosted"|"unknown"}
|
||||
* "unknown" = an expression with no literal `self-hosted` in it (e.g.
|
||||
* `${{ matrix.os }}`); the check does not guess, it skips.
|
||||
*/
|
||||
export function classifyRunsOn(runsOn) {
|
||||
if (runsOn == null) return "unknown";
|
||||
if (typeof runsOn === "string") {
|
||||
if (SELF_HOSTED.test(runsOn)) return "self-hosted";
|
||||
return EXPRESSION.test(runsOn) ? "unknown" : "hosted";
|
||||
}
|
||||
if (Array.isArray(runsOn)) {
|
||||
return runsOn.some((v) => typeof v === "string" && SELF_HOSTED.test(v))
|
||||
? "self-hosted"
|
||||
: "hosted";
|
||||
}
|
||||
if (typeof runsOn === "object") {
|
||||
// { group: ..., labels: ... } form
|
||||
const labels = runsOn.labels;
|
||||
return classifyRunsOn(Array.isArray(labels) ? labels : labels == null ? "" : String(labels));
|
||||
}
|
||||
return "unknown";
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string} yamlText
|
||||
* @param {string} fileName used only for reporting
|
||||
* @returns {{ file: string, job: string, step: string }[]}
|
||||
*/
|
||||
export function findProvenanceOnSelfHosted(yamlText, fileName = "<workflow>") {
|
||||
let doc;
|
||||
try {
|
||||
doc = yamlLoad(yamlText);
|
||||
} catch {
|
||||
// actionlint owns syntax; an unparseable file is not this rule's finding.
|
||||
return [];
|
||||
}
|
||||
const jobs =
|
||||
doc && typeof doc === "object" && doc.jobs && typeof doc.jobs === "object" ? doc.jobs : {};
|
||||
const findings = [];
|
||||
for (const [jobName, job] of Object.entries(jobs)) {
|
||||
if (!job || typeof job !== "object") continue;
|
||||
if (classifyRunsOn(job["runs-on"]) !== "self-hosted") continue;
|
||||
const steps = Array.isArray(job.steps) ? job.steps : [];
|
||||
steps.forEach((step, i) => {
|
||||
if (step && typeof step.run === "string" && PROVENANCE.test(step.run)) {
|
||||
findings.push({ file: fileName, job: jobName, step: step.name || `#${i + 1}` });
|
||||
}
|
||||
});
|
||||
}
|
||||
return findings;
|
||||
}
|
||||
|
||||
/** Human-readable line per finding, used by the CLI. */
|
||||
export function formatProvenanceFinding(f) {
|
||||
return `${f.file}: job "${f.job}", step "${f.step}" runs \`--provenance\` on a self-hosted runner — npm rejects that (422). Move the upload to a github-hosted job.`;
|
||||
}
|
||||
@@ -1,53 +1,172 @@
|
||||
#!/usr/bin/env bash
|
||||
# runner-janitor — self-hosted runner box hygiene (WS3.3, v3.8.49 quality plan).
|
||||
# runner-janitor — self-hosted runner box hygiene for the .113 pool.
|
||||
#
|
||||
# The .113 runner box has recurring failure modes that until now were manual
|
||||
# discipline: orphaned tmpfs/work dirs filling the disk, and >4 concurrent
|
||||
# runners OOM-killing jobs (16 GB box; incidents on the v3.8.47 release day).
|
||||
# Install via cron on the box (see docs/ops/RUNNER_BOX.md):
|
||||
# */30 * * * * /opt/omniroute-ops/runner-janitor.sh >> /var/log/runner-janitor.log 2>&1
|
||||
# Runs from cron every 30 min (see docs/ops/RUNNER_BOX.md). It ACTS on what it
|
||||
# can prove is safe and ALERTS on what needs an operator decision. Reads of
|
||||
# "is this in use?" and the removal happen in the same command, never in two
|
||||
# passes: a check-then-delete with a gap is how a live Build job lost its _work
|
||||
# on 2026-08-27.
|
||||
#
|
||||
# Measured box (2026-08-28): 31 GB RAM, 32 cores, 15 GB swap, /tmp = 12 GB
|
||||
# tmpfs (RAM!), 188 GB disk. A single `next-build` peaks at ~14 GB, so two
|
||||
# concurrent heavy builds saturate the box and three take it down (06:42Z that
|
||||
# day: load 56, two jobs lost). The v3.8.50 postmortem (Parte III) has the numbers.
|
||||
#
|
||||
# What it does, in order:
|
||||
# 1) sweep stale artefacts our tooling leaves behind — tmpfs bases after 3 h
|
||||
# (they hold RAM), disk _work/_temp bases after 24 h; only names we create,
|
||||
# only when no process has them open
|
||||
# 2) kill zombie builds: a `next-build` older than ZOMBIE_BUILD_MAX_MIN has no
|
||||
# job attached (a real Build step measures ~26 min). On 2026-08-27 one ran
|
||||
# 70 minutes after GitHub had already declared its job lost, eating 3.6 GB
|
||||
# and a full core set. KillMode=mixed on the units covers systemctl
|
||||
# stop/restart; this covers the lost-connection path.
|
||||
# 3) prune 48 h-old checkouts under _work of runners whose unit is INACTIVE
|
||||
# (stopped runners cannot be mid-job; active ones are never touched)
|
||||
# 4) alert: root disk >= DISK_ALERT_PCT, memory PSI full/avg60 >= threshold,
|
||||
# Runner.Listener count above the ceiling (with a per-project breakdown —
|
||||
# the box also hosts OmniHeuris and OmniMind runners)
|
||||
#
|
||||
# Usage: runner-janitor.sh [--dry-run] [--help]
|
||||
# Exit codes: 0 healthy · 1 attention needed (printed to stdout for the log).
|
||||
set -euo pipefail
|
||||
|
||||
MAX_ACTIVE_RUNNERS="${MAX_ACTIVE_RUNNERS:-4}"
|
||||
DISK_ALERT_PCT="${DISK_ALERT_PCT:-85}"
|
||||
WORK_DIR_MAX_AGE_HOURS="${WORK_DIR_MAX_AGE_HOURS:-24}"
|
||||
STATUS=0
|
||||
|
||||
echo "[janitor] $(date -u +%FT%TZ) start"
|
||||
|
||||
# 1) Sweep stale runner temp/work leftovers (>24h — no legitimate job runs that long).
|
||||
# Hardened for a root cron on world-writable paths: never follow a symlinked base
|
||||
# (a compromised runner could plant one), -P + -xdev so the sweep cannot traverse
|
||||
# out of the filesystem, and patterns narrowed to names OUR tooling creates
|
||||
# (no generic tmp* — unrelated system temp files are out of scope).
|
||||
for base in /tmp /home/*/actions-runner*/_work/_temp; do
|
||||
[ -d "$base" ] || continue
|
||||
[ -L "$base" ] && { echo "[janitor] skip symlinked base: $base"; continue; }
|
||||
find -P "$base" -xdev -maxdepth 1 \( -name 'runner-*' -o -name 'omniroute-*' \) \
|
||||
! -type l -mmin +$((WORK_DIR_MAX_AGE_HOURS * 60)) -exec rm -rf {} + 2>/dev/null || true
|
||||
DRY_RUN=0
|
||||
for arg in "$@"; do
|
||||
case "$arg" in
|
||||
--dry-run) DRY_RUN=1 ;;
|
||||
-h|--help)
|
||||
sed -n '2,32p' "$0" | sed 's/^# \{0,1\}//'
|
||||
exit 0 ;;
|
||||
*) echo "unknown argument: $arg" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
echo "[janitor] stale temp sweep done"
|
||||
|
||||
# 2) Disk pressure — alert loudly before SQLITE_FULL kills jobs mid-run.
|
||||
USAGE=$(df --output=pcent / | tail -1 | tr -dc '0-9')
|
||||
if [ "$USAGE" -ge "$DISK_ALERT_PCT" ]; then
|
||||
echo "[janitor] ⚠ ROOT DISK ${USAGE}% >= ${DISK_ALERT_PCT}% — clean before the next heavy run"
|
||||
STATUS=1
|
||||
MAX_ACTIVE_RUNNERS="${MAX_ACTIVE_RUNNERS:-8}"
|
||||
DISK_ALERT_PCT="${DISK_ALERT_PCT:-85}"
|
||||
TMPFS_MAX_AGE_HOURS="${TMPFS_MAX_AGE_HOURS:-3}"
|
||||
WORK_TEMP_MAX_AGE_HOURS="${WORK_TEMP_MAX_AGE_HOURS:-24}"
|
||||
WORK_CHECKOUT_MAX_AGE_HOURS="${WORK_CHECKOUT_MAX_AGE_HOURS:-48}"
|
||||
ZOMBIE_BUILD_MAX_MIN="${ZOMBIE_BUILD_MAX_MIN:-75}"
|
||||
ZOMBIE_BUILD_COMM="${ZOMBIE_BUILD_COMM:-next-build}"
|
||||
PSI_FULL_AVG60_ALERT="${PSI_FULL_AVG60_ALERT:-10}"
|
||||
# Overridable so the unit test can point everything at a fixture tree.
|
||||
JANITOR_TMP_BASES="${JANITOR_TMP_BASES-/tmp}"
|
||||
JANITOR_WORK_TEMP_BASES="${JANITOR_WORK_TEMP_BASES-/opt/actions-runner*/_work/_temp /home/*/actions-runner*/_work/_temp}"
|
||||
JANITOR_RUNNER_DIRS="${JANITOR_RUNNER_DIRS-/opt/actions-runner*}"
|
||||
JANITOR_PSI_FILE="${JANITOR_PSI_FILE:-/proc/pressure/memory}"
|
||||
JANITOR_DF_PATH="${JANITOR_DF_PATH:-/}"
|
||||
|
||||
STATUS=0
|
||||
say() { echo "[janitor] $*"; }
|
||||
|
||||
# "Is anything using this?" — ONE snapshot of every open path on the box
|
||||
# (lsof -Fn), then a prefix match per candidate. `lsof +D <dir>` walks the whole
|
||||
# tree instead and took minutes on a 5 GB leftover — unusable from cron. An
|
||||
# absent lsof means "cannot prove idle": the sweep keeps the path and says so.
|
||||
LSOF_BIN="${JANITOR_LSOF:-lsof}"
|
||||
have_busy_tools() { command -v "$LSOF_BIN" >/dev/null 2>&1; }
|
||||
SNAP=""
|
||||
cleanup() { [ -n "$SNAP" ] && rm -f -- "$SNAP"; }
|
||||
trap cleanup EXIT
|
||||
# One lsof for the whole run (~13 s / 83k lines on the box), kept ONLY for the
|
||||
# bases we sweep — 460 candidates grepping a re-printed 83k-line string was the
|
||||
# slow part, not lsof itself.
|
||||
snapshot_open_paths() {
|
||||
have_busy_tools || return 0
|
||||
SNAP=$(mktemp) || return 0
|
||||
local prefixes="" b
|
||||
for b in $JANITOR_TMP_BASES $JANITOR_WORK_TEMP_BASES; do [ -d "$b" ] && prefixes="$prefixes"$'\n'"$b/"; done
|
||||
# -F n: one "n<path>" line per open file; -w: no warnings
|
||||
"$LSOF_BIN" -w -Fn 2>/dev/null | sed -n 's/^n//p' | grep -F -f <(printf '%s' "$prefixes" | sed '/^$/d') > "$SNAP" 2>/dev/null || true
|
||||
}
|
||||
is_busy() {
|
||||
local p="$1"
|
||||
[ -n "$SNAP" ] && [ -s "$SNAP" ] || return 1
|
||||
# exact path, or anything beneath it when it is a directory
|
||||
grep -qxF -- "$p" "$SNAP" && return 0
|
||||
[ -d "$p" ] && grep -qF -- "$p/" "$SNAP"
|
||||
}
|
||||
|
||||
# sweep <base> <max-age-minutes>: only names our tooling creates, never through
|
||||
# a symlinked base, never across a filesystem, and remove+check in one step.
|
||||
sweep() {
|
||||
local base="$1" max_min="$2" p
|
||||
[ -d "$base" ] || return 0
|
||||
[ -L "$base" ] && { say "skip symlinked base: $base"; return 0; }
|
||||
while IFS= read -r -d '' p; do
|
||||
if ! have_busy_tools; then say "cannot prove idle (lsof missing — apt install lsof), kept: $p"; STATUS=1; continue; fi
|
||||
if is_busy "$p"; then say "busy, kept: $p"; continue; fi
|
||||
if [ "$DRY_RUN" -eq 1 ]; then say "would remove ($(( max_min / 60 ))h+): $p"; else rm -rf -- "$p" && say "removed ($(( max_min / 60 ))h+): $p"; fi
|
||||
done < <(find -P "$base" -xdev -mindepth 1 -maxdepth 1 \
|
||||
\( -name 'runner-*' -o -name 'omniroute-*' -o -name 'next-build*' -o -name 'e2e-build.tar.gz' \) \
|
||||
! -type l -mmin "+$max_min" -print0 2>/dev/null || true)
|
||||
}
|
||||
|
||||
say "$(date -u +%FT%TZ) start${DRY_RUN:+ (dry-run=$DRY_RUN)} busy-tools=$(have_busy_tools && echo ok || echo MISSING)"
|
||||
|
||||
# 1) stale artefacts — tmpfs is RAM, so it gets the short fuse
|
||||
snapshot_open_paths
|
||||
for base in $JANITOR_TMP_BASES; do sweep "$base" $(( TMPFS_MAX_AGE_HOURS * 60 )); done
|
||||
for base in $JANITOR_WORK_TEMP_BASES; do sweep "$base" $(( WORK_TEMP_MAX_AGE_HOURS * 60 )); done
|
||||
say "stale temp sweep done"
|
||||
|
||||
# 2) zombie builds
|
||||
ZOMBIES=0
|
||||
while read -r pid etimes comm; do
|
||||
[ -n "${pid:-}" ] || continue
|
||||
if [ "$etimes" -gt $(( ZOMBIE_BUILD_MAX_MIN * 60 )) ]; then
|
||||
say "⚠ zombie build pid=$pid comm=$comm age=$(( etimes / 60 ))min > ${ZOMBIE_BUILD_MAX_MIN}min — no job runs this long"
|
||||
if [ "$DRY_RUN" -eq 1 ]; then say "[dry-run] would: kill -TERM $pid (then -KILL)"; else
|
||||
kill -TERM "$pid" 2>/dev/null || true; sleep 10
|
||||
kill -0 "$pid" 2>/dev/null && { kill -KILL "$pid" 2>/dev/null || true; say " needed SIGKILL"; }
|
||||
fi
|
||||
ZOMBIES=$(( ZOMBIES + 1 )); STATUS=1
|
||||
fi
|
||||
done < <(ps -eo pid=,etimes=,comm= 2>/dev/null | awk -v c="$ZOMBIE_BUILD_COMM" '$3 ~ ("^" c) {print $1, $2, $3}' || true)
|
||||
say "zombie builds: $ZOMBIES"
|
||||
|
||||
# 3) old checkouts of STOPPED runners
|
||||
for d in $JANITOR_RUNNER_DIRS; do
|
||||
[ -d "$d" ] && [ -f "$d/.runner" ] || continue
|
||||
agent=$(grep -o '"agentName": *"[^"]*"' "$d/.runner" 2>/dev/null | sed 's/.*"\([^"]*\)"$/\1/')
|
||||
[ -n "$agent" ] || continue
|
||||
unit=$(systemctl list-units --plain --no-legend "actions.runner.*.${agent}.service" 2>/dev/null | awk 'NR==1{print $1}')
|
||||
[ -n "$unit" ] || continue
|
||||
if systemctl is-active --quiet "$unit"; then continue; fi
|
||||
while IFS= read -r -d '' co; do
|
||||
if [ "$DRY_RUN" -eq 1 ]; then say "would prune checkout of stopped runner $agent: $co"; else rm -rf -- "$co" && say "pruned checkout of stopped runner $agent: $co"; fi
|
||||
done < <(find -P "$d/_work" -xdev -mindepth 2 -maxdepth 2 -type d -mmin "+$(( WORK_CHECKOUT_MAX_AGE_HOURS * 60 ))" -print0 2>/dev/null || true)
|
||||
done
|
||||
|
||||
# 4a) disk
|
||||
USAGE=$(df --output=pcent "$JANITOR_DF_PATH" 2>/dev/null | tail -1 | tr -dc '0-9')
|
||||
if [ "${USAGE:-0}" -ge "$DISK_ALERT_PCT" ]; then
|
||||
say "⚠ ROOT DISK ${USAGE}% >= ${DISK_ALERT_PCT}% — clean before the next heavy run"; STATUS=1
|
||||
else
|
||||
echo "[janitor] disk ${USAGE}% OK"
|
||||
say "disk ${USAGE:-?}% OK"
|
||||
fi
|
||||
|
||||
# 3) Concurrency ceiling — 8-wide OOMed the 16 GB box twice on release day;
|
||||
# 4 is the proven ceiling. This CODIFIES the rule that was manual discipline.
|
||||
# 4b) memory pressure (PSI) — the box swapped its way through the v3.8.50 publish
|
||||
if [ -r "$JANITOR_PSI_FILE" ]; then
|
||||
FULL60=$(awk '/^full/ {for(i=1;i<=NF;i++) if ($i ~ /^avg60=/) {sub("avg60=","",$i); print $i}}' "$JANITOR_PSI_FILE" 2>/dev/null || echo "")
|
||||
if [ -n "$FULL60" ] && awk -v v="$FULL60" -v t="$PSI_FULL_AVG60_ALERT" 'BEGIN{exit !(v+0 >= t+0)}'; then
|
||||
say "⚠ MEMORY PRESSURE psi full/avg60=${FULL60}% >= ${PSI_FULL_AVG60_ALERT}% — too many heavy jobs at once"; STATUS=1
|
||||
else
|
||||
say "memory psi full/avg60=${FULL60:-n/a}% OK"
|
||||
fi
|
||||
fi
|
||||
|
||||
# 4c) concurrency ceiling — alert with a breakdown; the fix is fewer/labelled
|
||||
# runners (an operator decision), not killing listeners from cron.
|
||||
ACTIVE=$(pgrep -fc "Runner.Listener" || true)
|
||||
OMNI=$(pgrep -fc "actions-runner-omniroute[^ ]*/bin[^ ]*/Runner.Listener" || true)
|
||||
if [ "${ACTIVE:-0}" -gt "$MAX_ACTIVE_RUNNERS" ]; then
|
||||
echo "[janitor] ⚠ ${ACTIVE} Runner.Listener processes > ceiling ${MAX_ACTIVE_RUNNERS} — stop the extra runners (systemctl stop actions.runner.<name>)"
|
||||
say "⚠ ${ACTIVE} Runner.Listener processes (omniroute=${OMNI:-0}, other=$(( ${ACTIVE:-0} - ${OMNI:-0} ))) > ceiling ${MAX_ACTIVE_RUNNERS} — stop idle extras: systemctl stop <unit> only when it has no Runner.Worker child"
|
||||
STATUS=1
|
||||
else
|
||||
echo "[janitor] runners active: ${ACTIVE:-0}/${MAX_ACTIVE_RUNNERS} OK"
|
||||
say "runners active: ${ACTIVE:-0}/${MAX_ACTIVE_RUNNERS} (omniroute=${OMNI:-0}) OK"
|
||||
fi
|
||||
|
||||
echo "[janitor] done status=$STATUS"
|
||||
say "done status=$STATUS"
|
||||
exit "$STATUS"
|
||||
|
||||
@@ -7,6 +7,9 @@
|
||||
*/
|
||||
import { test } from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import {
|
||||
ALIBABA_FREE_TIER_TEXT_CAPABLE_MODELS,
|
||||
ALIBABA_NO_FREE_TIER_TEXT_MODELS,
|
||||
@@ -27,18 +30,90 @@ test("built-in allowlist includes operator free models and excludes paid blockli
|
||||
assert.equal(isAlibabaBuiltinFreeTierTextModel("qwen3.7-max"), false);
|
||||
});
|
||||
|
||||
test("allowlist JSON pack overrides embedded lists when valid", () => {
|
||||
/**
|
||||
* The shipped `config/alibaba-free-tier-allowlist.json` carries a `validUntil`,
|
||||
* so asserting against it made this test a time bomb: it went red on its own on
|
||||
* 2026-08-28, the day after the pack expired, and stayed red on every PR and on
|
||||
* `main` (#11866). Nothing had changed — the clock moved.
|
||||
*
|
||||
* Production was never affected: an expired pack falls back to the embedded
|
||||
* list by design. So the contract worth pinning is the BEHAVIOR on both sides of
|
||||
* the expiry, with packs this test owns and dates it controls — never the
|
||||
* freshness of the catalog that ships in the repo.
|
||||
*/
|
||||
function withAllowlistPack(
|
||||
pack: Record<string, unknown>,
|
||||
assertions: () => void
|
||||
): void {
|
||||
const dir = mkdtempSync(join(tmpdir(), "alibaba-allowlist-"));
|
||||
const packPath = join(dir, "allowlist.json");
|
||||
writeFileSync(packPath, JSON.stringify(pack), "utf8");
|
||||
|
||||
const previousPath = process.env.ALIBABA_FREE_TIER_ALLOWLIST_PATH;
|
||||
const packPath = `${process.cwd()}/config/alibaba-free-tier-allowlist.json`;
|
||||
process.env.ALIBABA_FREE_TIER_ALLOWLIST_PATH = packPath;
|
||||
resetAlibabaFreeTierAllowlistCache();
|
||||
try {
|
||||
assertions();
|
||||
} finally {
|
||||
if (previousPath) process.env.ALIBABA_FREE_TIER_ALLOWLIST_PATH = previousPath;
|
||||
else delete process.env.ALIBABA_FREE_TIER_ALLOWLIST_PATH;
|
||||
resetAlibabaFreeTierAllowlistCache();
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
|
||||
const pack = loadAlibabaFreeTierAllowlistPack();
|
||||
assert.ok(pack);
|
||||
assert.ok(isAlibabaFreeTierAllowlistPackValid(pack!));
|
||||
assert.ok(pack!.capable.includes("qwen3.6-plus"));
|
||||
test("allowlist JSON pack overrides embedded lists while it is still valid", () => {
|
||||
withAllowlistPack(
|
||||
{
|
||||
asOf: "2026-07-28",
|
||||
validUntil: "2999-01-01",
|
||||
capable: ["pack-only-capable-model", "qwen3.6-plus"],
|
||||
noFreeTier: ["pack-only-paid-model"],
|
||||
},
|
||||
() => {
|
||||
const pack = loadAlibabaFreeTierAllowlistPack();
|
||||
assert.ok(pack, "a pack inside its validity window must load");
|
||||
assert.ok(isAlibabaFreeTierAllowlistPackValid(pack!));
|
||||
assert.ok(pack!.capable.includes("qwen3.6-plus"));
|
||||
// Positive anchor: the pack must actually REPLACE the embedded list, not
|
||||
// merely load. `pack-only-capable-model` exists nowhere else.
|
||||
assert.equal(isAlibabaBuiltinFreeTierTextModel("pack-only-capable-model"), true);
|
||||
assert.equal(isAlibabaBuiltinNoFreeTierTextModel("pack-only-paid-model"), true);
|
||||
}
|
||||
);
|
||||
});
|
||||
|
||||
if (previousPath) process.env.ALIBABA_FREE_TIER_ALLOWLIST_PATH = previousPath;
|
||||
else delete process.env.ALIBABA_FREE_TIER_ALLOWLIST_PATH;
|
||||
resetAlibabaFreeTierAllowlistCache();
|
||||
test("an expired allowlist pack is ignored and the embedded list serves instead", () => {
|
||||
// This is the path production has actually been on since 2026-08-27, and it
|
||||
// had no coverage at all — which is why the expiry surfaced as a red test
|
||||
// rather than as a deliberate, understood fallback.
|
||||
withAllowlistPack(
|
||||
{
|
||||
asOf: "2026-07-28",
|
||||
validUntil: "2026-08-27",
|
||||
capable: ["pack-only-capable-model"],
|
||||
noFreeTier: ["pack-only-paid-model"],
|
||||
},
|
||||
() => {
|
||||
assert.equal(loadAlibabaFreeTierAllowlistPack(), null, "expired pack must not load");
|
||||
assert.equal(isAlibabaBuiltinFreeTierTextModel("pack-only-capable-model"), false);
|
||||
// The embedded list must be what answers once the pack is rejected.
|
||||
assert.equal(isAlibabaBuiltinFreeTierTextModel("qwen3.6-plus"), true);
|
||||
assert.equal(isAlibabaBuiltinNoFreeTierTextModel("qwen3.7-max"), true);
|
||||
}
|
||||
);
|
||||
});
|
||||
|
||||
test("isAlibabaFreeTierAllowlistPackValid compares against the instant it is given", () => {
|
||||
const pack = { asOf: "2026-07-28", validUntil: "2026-08-27", capable: ["x"], noFreeTier: [] };
|
||||
assert.equal(isAlibabaFreeTierAllowlistPackValid(pack, Date.parse("2026-08-26")), true);
|
||||
assert.equal(isAlibabaFreeTierAllowlistPackValid(pack, Date.parse("2026-08-28")), false);
|
||||
// No expiry declared means the pack never goes stale on its own.
|
||||
assert.equal(
|
||||
isAlibabaFreeTierAllowlistPackValid(
|
||||
{ asOf: "2026-07-28", capable: ["x"], noFreeTier: [] },
|
||||
Date.parse("2999-01-01")
|
||||
),
|
||||
true
|
||||
);
|
||||
});
|
||||
|
||||
145
tests/unit/check-workflows-provenance-runner.test.ts
Normal file
145
tests/unit/check-workflows-provenance-runner.test.ts
Normal file
@@ -0,0 +1,145 @@
|
||||
import test from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { readFileSync, readdirSync } from "node:fs";
|
||||
import { join } from "node:path";
|
||||
|
||||
import {
|
||||
classifyRunsOn,
|
||||
findProvenanceOnSelfHosted,
|
||||
} from "../../scripts/check/lib/provenanceRunner.mjs";
|
||||
|
||||
/**
|
||||
* v3.8.50, 10th publish attempt, 76 minutes in — after the tag, the GitHub
|
||||
* Release and the Docker images were already public:
|
||||
*
|
||||
* 422 Unprocessable Entity - Error verifying sigstore provenance bundle:
|
||||
* Unsupported GitHub Actions runner environment: "self-hosted".
|
||||
*
|
||||
* `USE_VPS_RUNNER` had routed the publish job to the .113 pool on 2026-08-02;
|
||||
* no release happened between 07-30 and 08-28, so nothing surfaced it. The
|
||||
* pairing is pure text, so it must fail the workflow lint on the PR that
|
||||
* introduces it.
|
||||
*/
|
||||
const ROOT = join(import.meta.dirname, "../..");
|
||||
const WORKFLOWS = join(ROOT, ".github/workflows");
|
||||
|
||||
// The exact runs-on expression npm-publish.yml used when it broke.
|
||||
const VPS_EXPR =
|
||||
"${{ (vars.USE_VPS_RUNNER == 'true' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository)) && fromJSON('[\"self-hosted\",\"omni-release\"]') || 'ubuntu-latest' }}";
|
||||
|
||||
function workflow(runsOn: string, run: string, extra = ""): string {
|
||||
return [
|
||||
"name: t",
|
||||
"on: push",
|
||||
"jobs:",
|
||||
" publish:",
|
||||
` runs-on: ${runsOn}`,
|
||||
extra,
|
||||
" steps:",
|
||||
" - name: upload",
|
||||
` run: ${run}`,
|
||||
"",
|
||||
].join("\n");
|
||||
}
|
||||
|
||||
test("classifyRunsOn: literal, array, object-with-labels and the fromJSON expression are self-hosted", () => {
|
||||
assert.equal(classifyRunsOn("self-hosted"), "self-hosted");
|
||||
assert.equal(classifyRunsOn(["self-hosted", "omni-release"]), "self-hosted");
|
||||
assert.equal(classifyRunsOn({ group: "Default", labels: ["self-hosted"] }), "self-hosted");
|
||||
assert.equal(classifyRunsOn(VPS_EXPR), "self-hosted");
|
||||
});
|
||||
|
||||
test("classifyRunsOn: hosted labels are hosted, opaque expressions are unknown (never guessed)", () => {
|
||||
assert.equal(classifyRunsOn("ubuntu-latest"), "hosted");
|
||||
assert.equal(classifyRunsOn(["ubuntu-latest"]), "hosted");
|
||||
assert.equal(classifyRunsOn("${{ matrix.os }}"), "unknown");
|
||||
assert.equal(classifyRunsOn(undefined), "unknown");
|
||||
});
|
||||
|
||||
test("flags --provenance inside a job routed to the self-hosted pool", () => {
|
||||
const found = findProvenanceOnSelfHosted(
|
||||
workflow(
|
||||
JSON.stringify(VPS_EXPR),
|
||||
'npm stage publish --provenance --access public --tag "$TAG"'
|
||||
),
|
||||
"npm-publish.yml"
|
||||
);
|
||||
assert.deepEqual(found, [{ file: "npm-publish.yml", job: "publish", step: "upload" }]);
|
||||
});
|
||||
|
||||
test("also catches the literal label and the --provenance-file form", () => {
|
||||
assert.equal(
|
||||
findProvenanceOnSelfHosted(workflow("self-hosted", "npm publish --provenance")).length,
|
||||
1
|
||||
);
|
||||
assert.equal(
|
||||
findProvenanceOnSelfHosted(
|
||||
workflow("[self-hosted, omni-release]", "npm publish --provenance-file=./p.json")
|
||||
).length,
|
||||
0,
|
||||
"--provenance-file is a different flag (a pre-built bundle) and is not what the registry rejects"
|
||||
);
|
||||
assert.equal(
|
||||
findProvenanceOnSelfHosted(workflow("self-hosted", "npm publish --provenance=true")).length,
|
||||
1
|
||||
);
|
||||
});
|
||||
|
||||
test("does not flag hosted jobs, unknown runners, or self-hosted jobs without the flag", () => {
|
||||
assert.deepEqual(
|
||||
findProvenanceOnSelfHosted(workflow("ubuntu-latest", "npm publish --provenance")),
|
||||
[]
|
||||
);
|
||||
assert.deepEqual(
|
||||
findProvenanceOnSelfHosted(workflow("${{ matrix.os }}", "npm publish --provenance")),
|
||||
[]
|
||||
);
|
||||
assert.deepEqual(
|
||||
findProvenanceOnSelfHosted(workflow("self-hosted", "npm publish --access public")),
|
||||
[]
|
||||
);
|
||||
// The word only in a step NAME or a comment is not a finding.
|
||||
assert.deepEqual(
|
||||
findProvenanceOnSelfHosted(
|
||||
[
|
||||
"name: t",
|
||||
"on: push",
|
||||
"jobs:",
|
||||
" j:",
|
||||
" runs-on: self-hosted",
|
||||
" steps:",
|
||||
" - name: provenance note",
|
||||
" run: echo hi # --provenance later",
|
||||
"",
|
||||
].join("\n")
|
||||
),
|
||||
[],
|
||||
"a comment after the command is still part of the run string — accept that the regex is conservative"
|
||||
);
|
||||
});
|
||||
|
||||
test("reusable-workflow jobs (uses:) and unparseable YAML are not this rule's findings", () => {
|
||||
const reusable = [
|
||||
"name: t",
|
||||
"on: push",
|
||||
"jobs:",
|
||||
" j:",
|
||||
" uses: ./.github/workflows/x.yml",
|
||||
"",
|
||||
].join("\n");
|
||||
assert.deepEqual(findProvenanceOnSelfHosted(reusable), []);
|
||||
assert.deepEqual(findProvenanceOnSelfHosted("jobs: [unclosed"), []);
|
||||
});
|
||||
|
||||
test("regression guard: no workflow in this repo publishes with --provenance from a self-hosted runner", () => {
|
||||
const files = readdirSync(WORKFLOWS).filter((f) => /\.ya?ml$/.test(f));
|
||||
assert.ok(files.length > 10, "expected the real workflow set");
|
||||
const findings = files.flatMap((f) =>
|
||||
findProvenanceOnSelfHosted(readFileSync(join(WORKFLOWS, f), "utf8"), f)
|
||||
);
|
||||
assert.deepEqual(
|
||||
findings,
|
||||
[],
|
||||
`npm rejects provenance from self-hosted runners (422) — move the upload to a github-hosted job: ${JSON.stringify(findings)}`
|
||||
);
|
||||
});
|
||||
135
tests/unit/config-expiry-time-bomb.test.ts
Normal file
135
tests/unit/config-expiry-time-bomb.test.ts
Normal file
@@ -0,0 +1,135 @@
|
||||
import test from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
|
||||
import {
|
||||
classifyExpiry,
|
||||
collectExpiryFields,
|
||||
scanConfigExpiry,
|
||||
} from "../../scripts/check/lib/configExpiry.mjs";
|
||||
|
||||
/**
|
||||
* Time bombs: config packs with a `validUntil` (or sibling key) that lapse with
|
||||
* no commit involved. The Alibaba free-tier pack expired on 2026-08-27 and from
|
||||
* the 28th every PR and main carried a red Unit Tests shard (#11866). Nothing a
|
||||
* diff review could have caught.
|
||||
*
|
||||
* This suite fails SEVEN DAYS BEFORE any pack under config/ lapses, naming the
|
||||
* file and key, so renewal happens on someone's terms instead of the clock's.
|
||||
*/
|
||||
const ROOT = join(import.meta.dirname, "../..");
|
||||
const CONFIG_DIR = join(ROOT, "config");
|
||||
const DAY = 86_400_000;
|
||||
const WARN_DAYS = 7;
|
||||
|
||||
/**
|
||||
* Packs known to be expired/expiring, each pinned to the issue that owns the
|
||||
* renewal decision. An entry whose pack is no longer expiring FAILS below as a
|
||||
* stale allowlist entry — remove it when the pack is renewed.
|
||||
*/
|
||||
const ALLOWLIST: Record<string, string> = {
|
||||
"alibaba-free-tier-allowlist.json":
|
||||
"#11866 — validUntil 2026-08-27 has passed; the loader already falls back to the embedded list, and renewing the curated free-tier pack is an operator data decision, not a test fix",
|
||||
};
|
||||
|
||||
const NOW = Date.UTC(2026, 7, 28); // 2026-08-28, fixed: this suite must not itself depend on the clock
|
||||
const day = (offset: number) => new Date(NOW + offset * DAY).toISOString().slice(0, 10);
|
||||
|
||||
test("collectExpiryFields: finds nested and array-nested expiry keys, ignores non-string values", () => {
|
||||
const fields = collectExpiryFields(
|
||||
{
|
||||
validUntil: day(3),
|
||||
nested: { expiresAt: day(30), other: "x" },
|
||||
list: [{ expiry: day(-1) }, { expires: 12345 }],
|
||||
expires_at: "not a date",
|
||||
},
|
||||
"pack.json"
|
||||
);
|
||||
assert.deepEqual(
|
||||
fields.map((f) => [f.keyPath, f.expiresAt === null ? null : "date"]),
|
||||
[
|
||||
["validUntil", "date"],
|
||||
["nested.expiresAt", "date"],
|
||||
["list.0.expiry", "date"],
|
||||
["expires_at", null],
|
||||
]
|
||||
);
|
||||
});
|
||||
|
||||
test("classifyExpiry: expired / expiring inside the warning window / ok / unparseable", () => {
|
||||
const f = (raw: string) => ({
|
||||
file: "p",
|
||||
keyPath: "validUntil",
|
||||
raw,
|
||||
expiresAt: Number.isFinite(Date.parse(raw)) ? Date.parse(raw) : null,
|
||||
});
|
||||
assert.equal(classifyExpiry(f(day(-1)), NOW, WARN_DAYS), "expired");
|
||||
assert.equal(
|
||||
classifyExpiry(f(day(0)), NOW, WARN_DAYS),
|
||||
"expiring",
|
||||
"lapsing today is already too late to be 'ok'"
|
||||
);
|
||||
assert.equal(classifyExpiry(f(day(6)), NOW, WARN_DAYS), "expiring");
|
||||
assert.equal(classifyExpiry(f(day(8)), NOW, WARN_DAYS), "ok");
|
||||
assert.equal(classifyExpiry(f("never"), NOW, WARN_DAYS), "unparseable");
|
||||
});
|
||||
|
||||
test("scanConfigExpiry: walks a config tree, skips node_modules and invalid JSON, keys files portably", () => {
|
||||
const dir = mkdtempSync(join(tmpdir(), "cfg-expiry-"));
|
||||
try {
|
||||
mkdirSync(join(dir, "sub"), { recursive: true });
|
||||
mkdirSync(join(dir, "node_modules", "dep"), { recursive: true });
|
||||
writeFileSync(join(dir, "a.json"), JSON.stringify({ validUntil: day(3) }));
|
||||
writeFileSync(join(dir, "sub", "b.json"), JSON.stringify({ deep: { expiresAt: day(40) } }));
|
||||
writeFileSync(
|
||||
join(dir, "node_modules", "dep", "c.json"),
|
||||
JSON.stringify({ validUntil: day(-5) })
|
||||
);
|
||||
writeFileSync(join(dir, "broken.json"), "{ not json");
|
||||
writeFileSync(join(dir, "notes.txt"), JSON.stringify({ validUntil: day(-5) }));
|
||||
const found = scanConfigExpiry(dir).map((f) => `${f.file}:${f.keyPath}`);
|
||||
assert.deepEqual(found, ["a.json:validUntil", "sub/b.json:deep.expiresAt"]);
|
||||
} finally {
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
test(`repo: no pack under config/ lapses within ${WARN_DAYS} days unless its renewal is tracked`, (t) => {
|
||||
const fields = scanConfigExpiry(CONFIG_DIR);
|
||||
// Positive anchor: the scanner must be seeing SOMETHING, or a renamed key
|
||||
// would silently turn this whole suite into a no-op.
|
||||
assert.ok(
|
||||
fields.length >= 1,
|
||||
"expected at least one dated pack under config/ (the Alibaba allowlist) — if the key was renamed, extend EXPIRY_KEY"
|
||||
);
|
||||
|
||||
const failures: string[] = [];
|
||||
const seenAllowlisted = new Set<string>();
|
||||
for (const f of fields) {
|
||||
const status = classifyExpiry(f, Date.now(), WARN_DAYS);
|
||||
const tracked = ALLOWLIST[f.file];
|
||||
if (status === "unparseable") {
|
||||
t.diagnostic(`${f.file} ${f.keyPath}="${f.raw}" is not a date — not monitored`);
|
||||
continue;
|
||||
}
|
||||
if (status === "ok") continue;
|
||||
if (tracked) {
|
||||
seenAllowlisted.add(f.file);
|
||||
t.diagnostic(`${f.file} ${f.keyPath}=${f.raw} is ${status} — tracked: ${tracked}`);
|
||||
continue;
|
||||
}
|
||||
failures.push(
|
||||
`${f.file} → ${f.keyPath}=${f.raw} is ${status}: renew the pack (or track it in ALLOWLIST with its issue)`
|
||||
);
|
||||
}
|
||||
for (const file of Object.keys(ALLOWLIST)) {
|
||||
if (!seenAllowlisted.has(file)) {
|
||||
failures.push(
|
||||
`stale ALLOWLIST entry: ${file} is no longer expired/expiring — remove it (${ALLOWLIST[file]})`
|
||||
);
|
||||
}
|
||||
}
|
||||
assert.deepEqual(failures, [], failures.join("\n"));
|
||||
});
|
||||
196
tests/unit/runner-janitor.test.ts
Normal file
196
tests/unit/runner-janitor.test.ts
Normal file
@@ -0,0 +1,196 @@
|
||||
import { describe, it } from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { spawnSync } from "node:child_process";
|
||||
import {
|
||||
existsSync,
|
||||
mkdirSync,
|
||||
mkdtempSync,
|
||||
readFileSync,
|
||||
rmSync,
|
||||
statSync,
|
||||
utimesSync,
|
||||
writeFileSync,
|
||||
} from "node:fs";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
|
||||
/**
|
||||
* scripts/ops/runner-janitor.sh runs from cron on the .113 runner box. This
|
||||
* suite pins its safety contract against a fixture tree — never the real /tmp:
|
||||
* every base, the runner dirs, the PSI file and the df path are redirected, the
|
||||
* zombie pattern is set to a name no process has, and the ceilings are lifted
|
||||
* so the outcome does not depend on the box the test happens to run on.
|
||||
*/
|
||||
const ROOT = path.resolve(import.meta.dirname, "..", "..");
|
||||
const SCRIPT = path.join(ROOT, "scripts", "ops", "runner-janitor.sh");
|
||||
const HOUR = 3_600_000;
|
||||
// The sweep needs lsof to PROVE a path is idle (one snapshot of open paths). Hosted CI
|
||||
// images ship both; a bare devbox may not. Each branch below asserts what must
|
||||
// hold in that environment — without the tools the contract is "delete nothing,
|
||||
// say why", which is exactly the behaviour worth pinning.
|
||||
const HAVE_BUSY_TOOLS =
|
||||
spawnSync("bash", ["-c", "command -v lsof"], { stdio: "ignore" }).status === 0;
|
||||
|
||||
function fixture() {
|
||||
const base = mkdtempSync(path.join(os.tmpdir(), "janitor-fixture-"));
|
||||
const old = new Date(Date.now() - 5 * HOUR);
|
||||
const mk = (name: string, dir: boolean, when: Date | null) => {
|
||||
const p = path.join(base, name);
|
||||
if (dir) {
|
||||
mkdirSync(p);
|
||||
writeFileSync(path.join(p, "x"), "x");
|
||||
} else writeFileSync(p, "x");
|
||||
if (when) utimesSync(p, when, when);
|
||||
return p;
|
||||
};
|
||||
return {
|
||||
base,
|
||||
staleTar: mk("e2e-build.tar.gz", false, old), // fixed-name artefact ci.yml/npm-publish leave behind
|
||||
staleBuild: mk("next-build-abc", true, old),
|
||||
staleUpgrade: mk("omniroute-install-upgrade-xyz", true, old),
|
||||
fresh: mk("omniroute-batch-api-fresh", true, null), // in use right now
|
||||
unrelated: mk("somebody-elses.log", false, old), // not ours — never touched
|
||||
};
|
||||
}
|
||||
|
||||
function run(args: string[], base: string, extraEnv: Record<string, string> = {}) {
|
||||
return spawnSync("bash", [SCRIPT, ...args], {
|
||||
encoding: "utf8",
|
||||
stdio: ["ignore", "pipe", "pipe"],
|
||||
env: {
|
||||
...process.env,
|
||||
JANITOR_TMP_BASES: base,
|
||||
JANITOR_WORK_TEMP_BASES: "",
|
||||
JANITOR_RUNNER_DIRS: path.join(base, "no-runners-here-*"),
|
||||
JANITOR_PSI_FILE: path.join(base, "no-psi"),
|
||||
JANITOR_DF_PATH: base,
|
||||
ZOMBIE_BUILD_COMM: "janitor-test-no-such-process",
|
||||
MAX_ACTIVE_RUNNERS: "9999",
|
||||
DISK_ALERT_PCT: "101",
|
||||
...extraEnv,
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
describe("runner-janitor.sh", () => {
|
||||
it("is executable bash with strict mode and prints usage on --help", () => {
|
||||
assert.ok(existsSync(SCRIPT));
|
||||
assert.ok(statSync(SCRIPT).mode & 0o111, "must be chmod +x (cron runs it directly)");
|
||||
const body = readFileSync(SCRIPT, "utf8");
|
||||
assert.ok(body.startsWith("#!/usr/bin/env bash"));
|
||||
assert.ok(body.includes("set -euo pipefail"));
|
||||
const help = run(["--help"], os.tmpdir());
|
||||
assert.equal(help.status, 0, help.stderr);
|
||||
assert.match(help.stdout, /--dry-run/);
|
||||
});
|
||||
|
||||
it("without lsof it cannot prove idle, so it deletes nothing and says why (exit 1)", () => {
|
||||
const f = fixture();
|
||||
try {
|
||||
const r = run([], f.base, { JANITOR_LSOF: "/nonexistent/lsof" });
|
||||
assert.equal(r.status, 1, "a janitor that cannot do its job must show up in the cron log");
|
||||
assert.match(r.stdout, /busy-tools=MISSING/);
|
||||
assert.match(
|
||||
r.stdout,
|
||||
/cannot prove idle \(lsof missing — apt install lsof\), kept: .*e2e-build\.tar\.gz/
|
||||
);
|
||||
for (const p of [f.staleTar, f.staleBuild, f.staleUpgrade, f.fresh, f.unrelated]) {
|
||||
assert.ok(existsSync(p), `must not delete ${p} when idleness cannot be proven`);
|
||||
}
|
||||
} finally {
|
||||
rmSync(f.base, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it("--dry-run names what it WOULD remove and removes nothing", (t) => {
|
||||
if (!HAVE_BUSY_TOOLS) return t.skip("lsof absent on this box — sweep branch covered in CI");
|
||||
const f = fixture();
|
||||
try {
|
||||
const r = run(["--dry-run"], f.base);
|
||||
assert.equal(r.status, 0, r.stderr + r.stdout);
|
||||
assert.match(r.stdout, /busy-tools=ok/);
|
||||
for (const p of [f.staleTar, f.staleBuild, f.staleUpgrade]) {
|
||||
assert.match(
|
||||
r.stdout,
|
||||
new RegExp(`would remove \\(3h\\+\\): ${p.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}`)
|
||||
);
|
||||
assert.doesNotMatch(
|
||||
r.stdout,
|
||||
new RegExp(`removed \\(3h\\+\\): ${p.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}`),
|
||||
"dry-run must never claim it removed something"
|
||||
);
|
||||
assert.ok(existsSync(p), `dry-run must not delete ${p}`);
|
||||
}
|
||||
assert.doesNotMatch(
|
||||
r.stdout,
|
||||
/omniroute-batch-api-fresh/,
|
||||
"a fresh dir is never a candidate"
|
||||
);
|
||||
assert.doesNotMatch(r.stdout, /somebody-elses\.log/, "only names our tooling creates");
|
||||
assert.match(r.stdout, /zombie builds: 0/);
|
||||
assert.match(r.stdout, /done status=0/);
|
||||
} finally {
|
||||
rmSync(f.base, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it("for real: sweeps the three stale artefacts, keeps the fresh one and the stranger", (t) => {
|
||||
if (!HAVE_BUSY_TOOLS) return t.skip("lsof absent on this box — sweep branch covered in CI");
|
||||
const f = fixture();
|
||||
try {
|
||||
const r = run([], f.base);
|
||||
assert.equal(r.status, 0, r.stderr + r.stdout);
|
||||
assert.ok(!existsSync(f.staleTar), "stale e2e-build.tar.gz must go (it is RAM on tmpfs)");
|
||||
assert.ok(!existsSync(f.staleBuild), "stale next-build dir must go");
|
||||
assert.ok(!existsSync(f.staleUpgrade), "stale install-upgrade dir must go");
|
||||
assert.ok(existsSync(f.fresh), "a fresh dir must survive");
|
||||
assert.ok(existsSync(f.unrelated), "files we did not create must survive even when old");
|
||||
} finally {
|
||||
rmSync(f.base, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it("tmpfs fuse is shorter than the disk fuse (RAM vs disk), both overridable", () => {
|
||||
const f = fixture();
|
||||
try {
|
||||
// With a 6h tmpfs fuse the 5h-old artefacts are NOT stale yet.
|
||||
const r = run(["--dry-run"], f.base, { TMPFS_MAX_AGE_HOURS: "6" });
|
||||
assert.doesNotMatch(
|
||||
r.stdout,
|
||||
/would remove|removed \(|cannot prove idle/,
|
||||
"nothing is stale under a 6h fuse, so no candidate is even examined"
|
||||
);
|
||||
const body = readFileSync(SCRIPT, "utf8");
|
||||
assert.match(body, /TMPFS_MAX_AGE_HOURS:-3\}/, "tmpfs default must stay short — it is RAM");
|
||||
assert.match(body, /WORK_TEMP_MAX_AGE_HOURS:-24\}/);
|
||||
} finally {
|
||||
rmSync(f.base, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it("alerts (exit 1) on disk and memory pressure thresholds without touching files", () => {
|
||||
const f = fixture();
|
||||
try {
|
||||
writeFileSync(
|
||||
path.join(f.base, "psi"),
|
||||
"some avg10=0.00 avg60=0.00 avg300=0.00 total=1\nfull avg10=0.00 avg60=23.50 avg300=9.00 total=1\n"
|
||||
);
|
||||
const r = run(["--dry-run"], f.base, {
|
||||
JANITOR_PSI_FILE: path.join(f.base, "psi"),
|
||||
DISK_ALERT_PCT: "0",
|
||||
});
|
||||
assert.equal(r.status, 1, "attention needed must be exit 1 for the cron log");
|
||||
assert.match(r.stdout, /MEMORY PRESSURE psi full\/avg60=23\.50%/);
|
||||
assert.ok(existsSync(f.fresh) && existsSync(f.unrelated));
|
||||
assert.match(r.stdout, /ROOT DISK \d+% >= 0%/);
|
||||
assert.ok(existsSync(f.staleTar), "alerting never deletes");
|
||||
} finally {
|
||||
rmSync(f.base, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it("rejects unknown arguments instead of silently running", () => {
|
||||
const r = run(["--yolo"], os.tmpdir());
|
||||
assert.equal(r.status, 2);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user