mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-08-18 21:22:28 +03:00
* feat(db): add a job registry for scheduled background work Background jobs each ship their own timer today, so there is no list of what is scheduled, no history of what ran, and no way to pause one without an environment variable and a restart. The registry gives them one home: a jobs table holding the schedule, a job_runs table holding the outcomes, and a loopback-only API to inspect and control both. Cron jobs read their expression through an optional cronGetter rather than the stored column, so an operator changing OMNIROUTE_WARMUP_CRON does not need the row rewritten. register() is an idempotent upsert that refreshes the schedule but never overwrites `enabled` or `created_at`, which is what lets a job be re-registered on every boot without discarding the operator's toggle. Run history is pruned per job rather than globally, and safeRun records a failure for a handler that throws as well as one that returns success:false, so a crashing job leaves a trail instead of a gap. The API is under /api/jobs and gated to loopback in the route guard. It can trigger a run and flip a job off, which is runtime administration and does not belong on a remotely reachable surface. Signed-off-by: Minxi Hou <houminxi@gmail.com> * feat(jobs): move the budget reset and token health check onto the registry Both jobs owned their own timer and started themselves as an import side effect, so nothing could report whether they were running, when they last ran, or why a run failed. They now register with the job registry and are started from it, which also means their schedule and run history are visible through /api/jobs. startAll() runs each interval job's first tick synchronously, so both entry points start the registry only after initializeCloudSync() has been awaited. The old wiring reached that ordering two different ways: the budget reset was started after the init call, and the health check's first sweep sat behind a 10s timer. Replacing both with one startAll() would otherwise have moved the two handlers in front of the initialisation they run against. Both entry points also register the same pair of jobs. Registering one and not the other is how a background job goes missing without anything failing. sweep() now returns how many connections it swept, so the health check can record a real records_affected the way the budget reset does. The migration documents that column as a per-job count, and hardcoding zero would have left one of the two jobs reporting a number the schema promises but the code never produces. A skipped or empty sweep reports zero. Every existing caller ignores the return value. The token health check keeps its own disable semantics: the handler still calls isHealthCheckDisabled() before sweeping, so OMNIROUTE_DISABLE_TOKEN_HEALTHCHECK, the production-build phase and the automated-test guard behave as before. Its registry adapter lives in src/lib/jobs/ next to the budget reset rather than in tokenHealthCheck.ts, which is already above its frozen size ceiling on the base branch and should not grow further. The adapter lets a failing sweep throw rather than reporting it itself, matching the budget reset: safeRun records a thrown error as a failure run with its message. The warmup job is seeded disabled. Its handler arrives with the warmup scheduler, and startAll() filters on enabled before it looks for a handler, so seeding it enabled here would warn about the missing handler on every boot. * fix: allowlist cron-parser dep and document OMNIROUTE_RUNNOW_TIMEOUT_MS env var Co-authored-by: diegosouzapw <diegosouza.pw@gmail.com> --------- Signed-off-by: Minxi Hou <houminxi@gmail.com> Co-authored-by: Minxi Hou <houminxi@gmail.com> Co-authored-by: diegosouzapw <diegosouzapw@users.noreply.github.com>
176 lines
5.6 KiB
TypeScript
176 lines
5.6 KiB
TypeScript
/**
|
|
* JobRegistry persistence layer.
|
|
*
|
|
* CRUD for the `jobs` and `job_runs` tables (migration 136). All timestamps are
|
|
* ISO-8601 strings; the registry computes thresholds in JS (never SQL datetime
|
|
* arithmetic) so comparisons are plain string compares.
|
|
*
|
|
* Column naming follows the rest of src/lib/db: snake_case in SQLite, camelCase in
|
|
* the returned objects (mapRow). `enabled` is stored INTEGER (0/1), `config` is a
|
|
* JSON string parsed/stringified at the boundary.
|
|
*/
|
|
|
|
import { getDbInstance } from "./core";
|
|
import type { JobRecord, JobRun } from "../jobRegistry/core";
|
|
|
|
function mapJob(row: any): JobRecord {
|
|
return {
|
|
id: row.id,
|
|
type: row.type,
|
|
cron: row.cron,
|
|
intervalMs: row.interval_ms,
|
|
enabled: row.enabled === 1,
|
|
envFlag: row.env_flag,
|
|
config: row.config ? JSON.parse(row.config) : {},
|
|
createdAt: row.created_at,
|
|
updatedAt: row.updated_at,
|
|
};
|
|
}
|
|
|
|
function mapRun(row: any): JobRun {
|
|
return {
|
|
id: row.id,
|
|
jobId: row.job_id,
|
|
startedAt: row.started_at,
|
|
finishedAt: row.finished_at,
|
|
status: row.status,
|
|
errorMessage: row.error_message,
|
|
recordsAffected: row.records_affected ?? 0,
|
|
durationMs: row.duration_ms,
|
|
};
|
|
}
|
|
|
|
export function getAllJobs(): JobRecord[] {
|
|
const db = getDbInstance();
|
|
const rows = db.prepare("SELECT * FROM jobs ORDER BY id").all();
|
|
return rows.map(mapJob);
|
|
}
|
|
|
|
export function getJob(id: string): JobRecord | null {
|
|
const db = getDbInstance();
|
|
const row = db.prepare("SELECT * FROM jobs WHERE id = ?").get(id);
|
|
return row ? mapJob(row) : null;
|
|
}
|
|
|
|
/**
|
|
* Idempotent register: INSERT OR IGNORE on first sight, then a column-level UPDATE
|
|
* that refreshes scheduling fields (type/cron/interval/env_flag/config) and bumps
|
|
* updated_at - but NEVER overwrites `enabled` (the user's API-driven toggle) nor
|
|
* `created_at`. Pass enabled=true for new jobs; the UPDATE simply skips the column.
|
|
*/
|
|
export function upsertJob(job: JobRecord): void {
|
|
const db = getDbInstance();
|
|
db.prepare(
|
|
`INSERT INTO jobs (id, type, cron, interval_ms, enabled, env_flag, config, created_at, updated_at)
|
|
VALUES (?, ?, ?, ?, ?, ?, ?, ?, datetime('now'))
|
|
ON CONFLICT(id) DO UPDATE SET
|
|
type = excluded.type,
|
|
cron = excluded.cron,
|
|
interval_ms = excluded.interval_ms,
|
|
env_flag = excluded.env_flag,
|
|
config = excluded.config,
|
|
updated_at = datetime('now')`
|
|
).run(
|
|
job.id,
|
|
job.type,
|
|
job.cron,
|
|
job.intervalMs,
|
|
job.enabled ? 1 : 0,
|
|
job.envFlag,
|
|
JSON.stringify(job.config ?? {}),
|
|
job.createdAt
|
|
);
|
|
}
|
|
|
|
export function updateJobEnabled(id: string, enabled: boolean): void {
|
|
const db = getDbInstance();
|
|
db.prepare("UPDATE jobs SET enabled = ?, updated_at = datetime('now') WHERE id = ?").run(
|
|
enabled ? 1 : 0,
|
|
id
|
|
);
|
|
}
|
|
|
|
export interface RecordRunOptions {
|
|
startedAt?: string;
|
|
durationMs?: number;
|
|
errorMessage?: string;
|
|
recordsAffected?: number;
|
|
}
|
|
|
|
/**
|
|
* Insert a completed run. `startedAt` is captured by the registry before the handler
|
|
* runs; `finishedAt` is derived from startedAt + durationMs so the two never drift.
|
|
* A status='running' insert leaves finishedAt NULL (used to mark in-flight work).
|
|
*/
|
|
export function recordRun(
|
|
jobId: string,
|
|
status: JobRun["status"],
|
|
opts: RecordRunOptions = {}
|
|
): void {
|
|
const db = getDbInstance();
|
|
const startedAt = opts.startedAt ?? new Date().toISOString();
|
|
const finishedAt =
|
|
status === "running"
|
|
? null
|
|
: opts.startedAt && opts.durationMs != null
|
|
? new Date(new Date(opts.startedAt).getTime() + opts.durationMs).toISOString()
|
|
: new Date().toISOString();
|
|
db.prepare(
|
|
`INSERT INTO job_runs (job_id, started_at, finished_at, status, error_message, records_affected, duration_ms)
|
|
VALUES (?, ?, ?, ?, ?, ?, ?)`
|
|
).run(
|
|
jobId,
|
|
startedAt,
|
|
finishedAt,
|
|
status,
|
|
opts.errorMessage ?? null,
|
|
opts.recordsAffected ?? 0,
|
|
opts.durationMs ?? null
|
|
);
|
|
}
|
|
|
|
/**
|
|
* Dual-dimension pruning: keep the most recent `maxRuns` OR anything younger than
|
|
* `maxDays`. A run is deleted only when it is BOTH outside the recent-N window AND
|
|
* older than maxDays - so a job that runs rarely keeps its full history.
|
|
* All thresholds are ISO-8601 string compares (no SQL datetime math).
|
|
*/
|
|
export function pruneRuns(jobId: string, maxRuns = 100, maxDays = 30): void {
|
|
const db = getDbInstance();
|
|
const threshold = new Date(Date.now() - maxDays * 86_400_000).toISOString();
|
|
db.prepare(
|
|
`DELETE FROM job_runs
|
|
WHERE job_id = ?
|
|
AND id NOT IN (
|
|
SELECT id FROM job_runs WHERE job_id = ? ORDER BY started_at DESC LIMIT ?
|
|
)
|
|
AND started_at < ?`
|
|
).run(jobId, jobId, maxRuns, threshold);
|
|
}
|
|
|
|
export function getRuns(jobId: string, limit = 20): JobRun[] {
|
|
const db = getDbInstance();
|
|
const rows = db
|
|
.prepare("SELECT * FROM job_runs WHERE job_id = ? ORDER BY started_at DESC LIMIT ?")
|
|
.all(jobId, limit);
|
|
return rows.map(mapRun);
|
|
}
|
|
|
|
/**
|
|
* Startup repair: any run still marked `running` whose started_at is older than
|
|
* `timeoutMinutes` is a leftover from a crashed/hung process. Mark it `failure` so
|
|
* the history is accurate and the slot frees up. The 5-minute default avoids
|
|
* clobbering a genuinely in-flight run on a slow box.
|
|
*/
|
|
export function cleanupOrphanedRuns(timeoutMinutes = 5): void {
|
|
const db = getDbInstance();
|
|
const threshold = new Date(Date.now() - timeoutMinutes * 60_000).toISOString();
|
|
db.prepare(
|
|
`UPDATE job_runs
|
|
SET status = 'failure',
|
|
error_message = 'orphaned: exceeded timeout',
|
|
finished_at = datetime('now')
|
|
WHERE status = 'running' AND started_at < ?`
|
|
).run(threshold);
|
|
}
|