diff --git a/deployment/proxy/nginx-cms.conf b/deployment/proxy/nginx-cms.conf index 511ace60..9bc86e89 100644 --- a/deployment/proxy/nginx-cms.conf +++ b/deployment/proxy/nginx-cms.conf @@ -162,6 +162,22 @@ server { ssl_early_data on; add_header Alt-Svc 'h3=":9443"; ma=86400' always; + # Resuming a session skips the full handshake, which is most of the cost of + # a TLS connection. Without this nginx performs no session resumption at all: + # every visitor paid a full handshake on every request. 50M shared sessions + # is roughly 1GB at the default 20-byte key id plus overhead. + ssl_session_cache shared:CMS_TLS:50m; + ssl_session_timeout 1d; + ssl_session_tickets off; + + # `index index.html` without a `root` left nginx resolving every + # try_files/$uri against the compiled-in default /etc/nginx/html. The + # /robots.txt and /favicon.ico probes then stat() a path the worker cannot + # traverse, and because a failed stat is logged at crit the error log filled + # with 149 crit lines per scan. Pointing root at the CMS document root makes + # the same probe a plain 404, which log_not_found already suppresses. + root /var/www/html; + index index.html; # ─── Security Headers ─── @@ -366,8 +382,21 @@ server { add_header Cache-Tag "cms-camera"; } + # robots.txt is generated by the CMS (src/app/robots.ts, force-dynamic + # because it needs APP_URL) and sitemap.xml points crawlers at it. This + # location used to answer from disk with try_files, which made it a + # guaranteed 404: the file does not exist in public/, so crawlers were told + # to obey a robots.txt they could never read. Proxy it like the route it + # actually is. favicon.ico below stays on disk — log_not_found already + # keeps its miss quiet. + location = /robots.txt { + access_log off; + proxy_pass http://cms_app; + proxy_http_version 1.1; + proxy_set_header Host $host; + } + location = /favicon.ico { expires 1y; access_log off; log_not_found off; try_files $uri =404; } - location = /robots.txt { expires 1d; access_log off; log_not_found off; try_files $uri =404; } # ─── Static Next.js Assets ─── location /_next/static/ { diff --git a/deployment/systemd/cms-jobs-worker.service b/deployment/systemd/cms-jobs-worker.service new file mode 100644 index 00000000..3386f020 --- /dev/null +++ b/deployment/systemd/cms-jobs-worker.service @@ -0,0 +1,39 @@ +[Unit] +# The scheduled-job worker (scheduled articles, catalog export, backups, disk +# and health probes). This is NOT optional: a web process alone does not +# establish that scheduled work runs. The CMS reports it as a failed +# diagnostic row when the Redis heartbeat at cms:jobs-worker:heartbeat is +# missing, which is exactly what happened while nothing supervised this. +# +# It runs on the host rather than in a container on purpose: the schedule +# shells out to mysqldump, df and docker, none of which exist in the CMS image, +# and it must survive CMS deploys (a container is replaced on every release). +Description=AtomNext CMS scheduled-job worker +Documentation=https://gitlab.epicnabbo.nl/remco/EpicNext-Cms +After=network-online.target docker.service mariadb.service +Wants=network-online.target +# Start ordering only; the worker tolerates the database being briefly absent +# and retries, so do not make it hard-fail when mariadb is slow to boot. +Wants=docker.service + +[Service] +Type=simple +User=root +WorkingDirectory=/var/www/atom-nexst +Environment=NODE_ENV=production +ExecStart=/usr/bin/node --conditions=react-server --import tsx scripts/jobs-worker.ts +# The worker's own catch-all logs and exits 1 on a fatal error, so a restart is +# always wanted. 10s backoff stops a persistent misconfiguration (missing .env, +# bad DATABASE_URL) from spinning. +Restart=always +RestartSec=10 +# Give a crashed job time to finish its DB transaction before the next start, +# otherwise a mid-transaction kill can loop on the same failure. +TimeoutStopSec=30 +KillSignal=SIGTERM +StandardOutput=journal +StandardError=journal +SyslogIdentifier=cms-jobs-worker + +[Install] +WantedBy=multi-user.target diff --git a/deployment/systemd/nginx-override.conf b/deployment/systemd/nginx-override.conf new file mode 100644 index 00000000..227c1b01 --- /dev/null +++ b/deployment/systemd/nginx-override.conf @@ -0,0 +1,19 @@ +[Service] +# systemd's default is 1024:524288, i.e. a *soft* LimitNOFILE of 1024. nginx +# inherits that soft limit, so worker_connections 2048 could not actually be +# reached and every start logged: +# "2048 worker_connections exceed open file resource limit: 1024" +# Raise both soft and hard to 65536 so the master's rlimit covers +# worker_connections before nginx is even started. +LimitNOFILE=65536 + +# The packaged unit ships Restart=no, so a crashed or OOM-killed nginx stayed +# down until someone noticed. nginx is the only thing serving the site, so it +# must come back on its own. `on-failure` restarts only abnormal exits, which +# keeps an operator-initiated `systemctl stop` from being undone. +Restart=on-failure +RestartSec=2 + +# Give in-flight requests time to drain on stop/reload instead of severing +# keepalive connections and long-polling SSE streams mid-response. +TimeoutStopSec=30 diff --git a/scripts/ci-deploy.sh b/scripts/ci-deploy.sh index fd344298..17aaa7e1 100644 --- a/scripts/ci-deploy.sh +++ b/scripts/ci-deploy.sh @@ -440,15 +440,28 @@ if docker inspect "$backup_name" >/dev/null 2>&1; then exit 1 fi -# The avatar/badge disk cache lives on the host bind and is written by uid 33 -# inside the container. Root-owned directories make every cache write fail -# silently, which turns each avatar into a fresh live render. +# The application writes everything under storage/ as uid 33, but storage is a +# host bind so the image's own ownership is irrelevant. Any path that is not +# uid 33 makes the write fail with EACCES, and because most of these writes are +# inside a try/catch the failure is silent: the avatar cache just never fills +# (each avatar becomes a fresh live render) and the catalog export reports +# "delivery failed" while the emulator never receives the update. The old code +# only repaired storage/imaging, so storage/catalog-git/hotel-status.json kept +# coming back root:root and /api/admin/catalog/status kept throwing EACCES. +for owned_dir in imaging catalog-git cms-errors furniture-imports logs media \ + nitro-cleanup config-backups nitro-scale32-backups; do + target="$deploy_dir/storage/$owned_dir" + [ -e "$target" ] || mkdir -p "$target" 2>/dev/null || true + [ -d "$target" ] || continue + chown -R 33:33 "$target" 2>/dev/null || true +done +# The avatar/badge cache needs its leaf directories to exist before first use; +# the cache misses (and re-renders live) rather than erroring when they do not. for cache_dir in avatars badges; do if ! install -d -o 33 -g 33 -m 0750 "$deploy_dir/storage/imaging/$cache_dir" 2>/dev/null; then mkdir -p "$deploy_dir/storage/imaging/$cache_dir" 2>/dev/null || true fi done -chown -R 33:33 "$deploy_dir/storage/imaging" 2>/dev/null || true if [ "$blue_green" -eq 1 ]; then # 1. Maak de doel-poort vrij. Alles wat daar draait is per definitie niet live, diff --git a/scripts/docker-start.import.test.mjs b/scripts/docker-start.import.test.mjs index b242bf8d..6d2a96f5 100644 --- a/scripts/docker-start.import.test.mjs +++ b/scripts/docker-start.import.test.mjs @@ -1,7 +1,13 @@ import assert from "node:assert/strict"; import { spawnSync } from "node:child_process"; -import { it } from "vitest"; +import { describe, expect, it } from "vitest"; + +import { + detectMemoryLimitMb, + heapLimitMb, + runtimeNodeOptions, +} from "./docker-start.mjs"; it("imports runtime validation without starting the CMS", () => { const result = spawnSync( @@ -15,3 +21,89 @@ it("imports runtime validation without starting the CMS", () => { ); assert.equal(result.status, 0, result.stderr); }); + +describe("heap limit", () => { + it("leaves headroom for the memory V8 does not account for", () => { + // 4 GB cgroup limit -> a 2867 MB heap, well under the ceiling. + expect(heapLimitMb(4 * 1024 ** 3)).toBe(2867); + expect(heapLimitMb(6 * 1024 ** 3)).toBe(4300); + }); + + it("clamps to a floor and a ceiling", () => { + // Too small to run a Next.js server at all: floor wins. + expect(heapLimitMb(256 * 1024 ** 2)).toBe(512); + // A huge or absent limit must not turn into a 100 GB heap. + expect(heapLimitMb(64 * 1024 ** 3)).toBe(8192); + expect(heapLimitMb(Number.NaN)).toBe(8192); + expect(heapLimitMb(0)).toBe(8192); + }); +}); + +describe("cgroup detection", () => { + const asReader = (contents) => (path) => { + if (!(path in contents)) throw new Error(`ENOENT: ${path}`); + return contents[path]; + }; + + it("reads the cgroup v2 limit", () => { + expect( + detectMemoryLimitMb( + asReader({ "/sys/fs/cgroup/memory.max": "4294967296" }), + ), + ).toBe(2867); + }); + + it("falls back to cgroup v1 when v2 is absent", () => { + expect( + detectMemoryLimitMb( + asReader({ + "/sys/fs/cgroup/memory.max": "", + "/sys/fs/cgroup/memory/memory.limit_in_bytes": "6442450944", + }), + ), + ).toBe(4300); + }); + + it("treats an unlimited cgroup as no limit at all", () => { + // cgroup v1 reports "max"; a bare sentinel means the same thing. + expect( + detectMemoryLimitMb(asReader({ "/sys/fs/cgroup/memory.max": "max" })), + ).toBe(8192); + expect( + detectMemoryLimitMb( + asReader({ + "/sys/fs/cgroup/memory/memory.limit_in_bytes": "9223372036854771712", + }), + ), + ).toBe(8192); + }); + + it("falls back when neither cgroup file is readable", () => { + expect( + detectMemoryLimitMb(() => { + throw new Error("ENOENT"); + }), + ).toBe(8192); + }); +}); + +describe("NODE_OPTIONS", () => { + it("adds the cap when none is set", () => { + expect(runtimeNodeOptions("", 2867)).toBe("--max-old-space-size=2867"); + expect(runtimeNodeOptions(undefined, 2867)).toBe( + "--max-old-space-size=2867", + ); + }); + + it("keeps unrelated options already present", () => { + expect(runtimeNodeOptions("--no-warnings", 2867)).toBe( + "--no-warnings --max-old-space-size=2867", + ); + }); + + it("never overrides an explicit operator choice", () => { + expect(runtimeNodeOptions("--max-old-space-size=8192", 2867)).toBe( + "--max-old-space-size=8192", + ); + }); +}); diff --git a/scripts/docker-start.mjs b/scripts/docker-start.mjs index 5c17161f..40787752 100644 --- a/scripts/docker-start.mjs +++ b/scripts/docker-start.mjs @@ -1,7 +1,70 @@ // Fail before listening if an installation has no valid runtime configuration. import { spawn } from "node:child_process"; +import { readFileSync } from "node:fs"; import { pathToFileURL } from "node:url"; +/** Fraction of the container memory limit V8 is allowed to use for its heap. + * The rest has to cover native allocations the JS heap cannot account for: + * the mysql2 pool buffers, sharp's image pipeline, and zlib during a burst of + * RSC rendering. */ +const HEAP_FRACTION = 0.7; +const MIN_HEAP_MB = 512; +/** Backstop only. The fraction is the real policy: on a 6 GB container it asks + * for 4300 MB, and a backstop at or below that would silently turn the fraction + * into a fixed number and make the two limits disagree. This exists purely so a + * nonsensical cgroup reading cannot ask for an unbounded heap. */ +const MAX_HEAP_MB = 8192; + +export function heapLimitMb(cgroupLimitBytes) { + if (!Number.isFinite(cgroupLimitBytes) || cgroupLimitBytes <= 0) + return MAX_HEAP_MB; + const mb = Math.floor((cgroupLimitBytes * HEAP_FRACTION) / (1024 * 1024)); + return Math.min(MAX_HEAP_MB, Math.max(MIN_HEAP_MB, mb)); +} + +/** + * Read this container's memory ceiling from cgroup v2, falling back to v1. + * Without this the V8 heap defaults to a quarter of *host* memory, so a 4 GB + * container on a 24 GB host lets the heap grow past the limit and the kernel + * OOM-kills the process mid-request — which is what produced the + * `next-build (v16)` kills in the host logs. A container that GCs before it + * reaches the ceiling degrades to a slower page instead of a killed process. + */ +export function detectMemoryLimitMb(readFile = readFileSync) { + const candidates = [ + "/sys/fs/cgroup/memory.max", + "/sys/fs/cgroup/memory/memory.limit_in_bytes", + ]; + for (const path of candidates) { + let raw; + try { + raw = readFile(path, "utf8").trim(); + } catch { + continue; + } + // cgroup v1 reports "max" for an unlimited cgroup; v2 uses a bare + // sentinel of a very large number on some kernels. + if (raw === "max" || raw === "") continue; + const bytes = Number(raw); + if (!Number.isFinite(bytes) || bytes <= 0) continue; + // A host-sized "limit" means no cgroup ceiling was applied. + if (bytes >= Number.MAX_SAFE_INTEGER) continue; + return heapLimitMb(bytes); + } + return heapLimitMb(Number.NaN); +} + +export function runtimeNodeOptions( + existing = "", + heapMb = detectMemoryLimitMb(), +) { + const flag = `--max-old-space-size=${heapMb}`; + if (!existing.trim()) return flag; + // Respect an explicit operator override; only add the cap when absent. + if (existing.includes("--max-old-space-size")) return existing; + return `${existing} ${flag}`; +} + export function validateRuntime(settings) { const invalid = []; if (!settings.HOTEL_NAME?.trim() || settings.HOTEL_NAME === "Build fixture") @@ -33,7 +96,13 @@ if ( ) { try { validateRuntime(process.env); - const child = spawn(process.execPath, ["server.js"], { stdio: "inherit" }); + const heapMb = detectMemoryLimitMb(); + const nodeOptions = runtimeNodeOptions(process.env.NODE_OPTIONS, heapMb); + console.log(`Starting CMS with a ${heapMb} MB V8 heap cap`); + const child = spawn(process.execPath, ["server.js"], { + stdio: "inherit", + env: { ...process.env, NODE_OPTIONS: nodeOptions }, + }); for (const signal of ["SIGTERM", "SIGINT"]) process.on(signal, () => child.kill(signal)); child.on("error", () => { diff --git a/scripts/jobs-worker.ts b/scripts/jobs-worker.ts index 771caf40..2c2b0c39 100644 --- a/scripts/jobs-worker.ts +++ b/scripts/jobs-worker.ts @@ -1,9 +1,17 @@ -import { drainOperationEffects } from "../src/features/operations/worker"; -import { drainFurnitureImports } from "../src/lib/services/furni-job-worker"; +// Must stay the first import. ESM evaluates a module's imports in source +// order, and `../src/features/operations/worker` reaches `@/env`, which parses +// process.env at import time. With this import further down the tree, load-env +// ran *after* the schema validation had already thrown on a missing +// DATABASE_URL, so the worker could only ever start from an environment that +// already exported the config — which is why `pnpm jobs:worker` died +// immediately and nothing supervised it. import "./load-env"; +import * as nodeFs from "node:fs"; +import * as nodePath from "node:path"; import { Cron } from "croner"; import { lt, sql } from "drizzle-orm"; import { env } from "../src/env"; +import { drainOperationEffects } from "../src/features/operations/worker"; import { db, PasswordReset, WebsiteLoginLogs } from "../src/lib/db"; import { logger } from "../src/lib/logger"; import { redis } from "../src/lib/redis"; @@ -18,6 +26,7 @@ import { diskLevel, parseDfOutput, } from "../src/lib/services/disk-usage"; +import { drainFurnitureImports } from "../src/lib/services/furni-job-worker"; import { publishDueArticles } from "../src/lib/services/news-scheduler"; import { scheduledAutoCleanFakeNitros } from "../src/lib/services/nitro-cleanup"; import { rcon } from "../src/lib/services/rcon"; @@ -161,13 +170,69 @@ async function checkDiskUsage(): Promise { } } +/** + * Resolve the JAR to back up. `EMULATOR_JAR_PATH` may point at the file itself + * or at a directory of release JARs, because the emulator's own unit file + * launches `ls -t Polaris-*-jar-with-dependencies.jar` — a path pinned to one + * release filename goes stale on the next emulator upgrade, and a stale path + * fails as a bare ENOENT from copyFile that gives no hint what is wrong. A + * directory (or a path with a `*`) resolves to the most recently modified JAR, + * matching how the emulator actually picks its build. + */ +export function resolveEmulatorJar( + configuredPath: string, + fs: typeof import("node:fs") = nodeFs, + { resolve }: typeof import("node:path") = nodePath, +): string | null { + const { existsSync, readdirSync, statSync } = fs; + if (configuredPath.includes("*")) { + const dir = configuredPath.slice(0, configuredPath.lastIndexOf("/") + 1); + const pattern = configuredPath.slice(dir.length); + if (!existsSync(dir)) return null; + return ( + readdirSync(dir) + .filter((name: string) => name.startsWith(pattern.split("*")[0] ?? "")) + .map((name: string) => resolve(dir, name)) + .filter((path: string) => existsSync(path)) + .sort( + (a: string, b: string) => statSync(b).mtimeMs - statSync(a).mtimeMs, + )[0] ?? null + ); + } + if (existsSync(configuredPath) && statSync(configuredPath).isFile()) + return configuredPath; + // A directory: take the newest JAR in it. + if (existsSync(configuredPath) && statSync(configuredPath).isDirectory()) { + return ( + readdirSync(configuredPath) + .filter((name: string) => name.endsWith(".jar")) + .map((name: string) => resolve(configuredPath, name)) + .sort( + (a: string, b: string) => statSync(b).mtimeMs - statSync(a).mtimeMs, + )[0] ?? null + ); + } + return null; +} + async function backupEmulatorJar(): Promise { if (!env.EMULATOR_JAR_PATH || !env.EMULATOR_BACKUP_DIR) return; - const { copyFileSync, mkdirSync, readdirSync, unlinkSync, existsSync } = - await import("node:fs"); + const fs = await import("node:fs"); + const { copyFileSync, mkdirSync, readdirSync, unlinkSync, existsSync } = fs; const { resolve } = await import("node:path"); + const jarPath = resolveEmulatorJar(env.EMULATOR_JAR_PATH, fs, nodePath); + if (!jarPath) { + // Configured but unusable: say so once, loudly, instead of every night + // logging an opaque copyFile ENOENT that reads like a permissions bug. + logger.error( + "Emulator JAR backup skipped: EMULATOR_JAR_PATH does not resolve to a JAR", + { module: "jobs", configured: env.EMULATOR_JAR_PATH }, + ); + return; + } + const timestamp = new Date().toISOString().slice(0, 19).replace(/[T:]/g, "-"); const backupFile = resolve( env.EMULATOR_BACKUP_DIR, @@ -179,10 +244,11 @@ async function backupEmulatorJar(): Promise { } try { - copyFileSync(env.EMULATOR_JAR_PATH, backupFile); + copyFileSync(jarPath, backupFile); logger.info("Backed up emulator JAR", { module: "jobs", backupFile, + source: jarPath, }); const keep = env.EMULATOR_BACKUP_KEEP ?? 7; diff --git a/scripts/nginx-sync.sh b/scripts/nginx-sync.sh index 9bc673d2..75cfaa66 100755 --- a/scripts/nginx-sync.sh +++ b/scripts/nginx-sync.sh @@ -101,6 +101,15 @@ fi if [[ ! -d /var/log/nginx ]]; then install -d -o root -g adm -m 750 /var/log/nginx fi +# nginx-cms.conf sets `root /var/www/html` so that disk-backed locations +# (favicon.ico) resolve somewhere the www-data worker can actually traverse. +# The previous implicit root was /etc/nginx/html, which sits behind /etc/nginx +# (0750 root:root): the worker got EACCES on every stat, and nginx logs a +# failed stat at crit, so each crawler probe wrote a crit line. +if [[ ! -d /var/www/html ]]; then + install -d -o root -g root -m 755 /var/www/html + echo "+ created /var/www/html (document root)" +fi for f in /var/log/nginx/access.log /var/log/nginx/error.log; do [[ -f "$f" ]] || touch "$f" done diff --git a/src/app/api/health/route.test.ts b/src/app/api/health/route.test.ts new file mode 100644 index 00000000..cd10287f --- /dev/null +++ b/src/app/api/health/route.test.ts @@ -0,0 +1,95 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +const mocks = vi.hoisted(() => ({ + clientIp: vi.fn(async () => "203.0.113.7"), + rateLimit: vi.fn( + async (): Promise<{ ok: boolean; retryAfter: number }> => ({ + ok: true, + retryAfter: 0, + }), + ), + dbExecute: vi.fn(async () => [{}]), + ping: vi.fn(async () => "PONG"), + rconSend: vi.fn(async () => true), +})); +vi.mock("@/lib/rate-limit", () => ({ + clientIp: mocks.clientIp, + rateLimit: mocks.rateLimit, +})); +vi.mock("@/lib/db", () => ({ + db: { execute: mocks.dbExecute }, +})); +vi.mock("@/lib/redis", () => ({ + redis: { + ping: mocks.ping, + }, +})); +vi.mock("@/lib/services/rcon", () => ({ + rcon: { send: mocks.rconSend }, +})); +vi.mock("@/env", () => ({ + env: { REDIS_URL: "redis://cache.test:6379", RESEND_API_KEY: "" }, +})); + +import { GET } from "./route"; + +beforeEach(() => { + vi.clearAllMocks(); + mocks.ping.mockResolvedValue("PONG"); + mocks.dbExecute.mockResolvedValue([{}]); + mocks.rconSend.mockResolvedValue(true); + mocks.rateLimit.mockResolvedValue({ ok: true, retryAfter: 0 }); +}); + +describe("ops health status", () => { + it("answers 200 when the database is reachable", async () => { + const res = await GET(); + expect(res.status).toBe(200); + expect(await res.json()).toMatchObject({ + status: "ok", + database: true, + }); + }); + + // The regression this guards: the route used to answer 200 unconditionally, + // so the Docker healthcheck reported a container healthy while every page + // failed to render. + it("answers 503 when the database is unreachable", async () => { + mocks.dbExecute.mockRejectedValue(new Error("ECONNREFUSED")); + const res = await GET(); + expect(res.status).toBe(503); + expect(await res.json()).toMatchObject({ + status: "degraded", + database: false, + }); + }); + + // Redis and the emulator both have in-process fallbacks (cache.ts, + // rate-limit.ts), so failing the container on them would turn a degraded + // site into a restart loop. + it("stays 200 on a Redis outage because the cache falls back in-process", async () => { + mocks.ping.mockRejectedValue(new Error("ECONNREFUSED")); + const res = await GET(); + expect(res.status).toBe(200); + expect(await res.json()).toMatchObject({ + status: "degraded", + database: true, + redis: false, + }); + }); + + it("stays 200 when the emulator is unreachable", async () => { + mocks.rconSend.mockResolvedValue(false); + const res = await GET(); + expect(res.status).toBe(200); + expect(await res.json()).toMatchObject({ emulator: false }); + }); + + it("still rate-limits before probing anything", async () => { + mocks.rateLimit.mockResolvedValue({ ok: false, retryAfter: 30 }); + const res = await GET(); + expect(res.status).toBe(429); + expect(res.headers.get("Retry-After")).toBe("30"); + expect(mocks.dbExecute).not.toHaveBeenCalled(); + }); +}); diff --git a/src/app/api/health/route.ts b/src/app/api/health/route.ts index 180d89ab..78c13170 100644 --- a/src/app/api/health/route.ts +++ b/src/app/api/health/route.ts @@ -9,9 +9,15 @@ import { rcon } from "@/lib/services/rcon"; /** * Ops health probe: database reachability, Redis (when configured), emulator - * RCON, SMTP (when configured), and runtime info. Returns HTTP 200 always - * (read the `status`/`database` fields), so it's safe for uptime monitors that - * only care about reachability. Rate-limited per client IP. + * RCON, SMTP (when configured), and runtime info. + * + * HTTP status is load-bearing: 200 means the CMS can actually serve, 503 means + * it cannot. Previously this route answered 200 even with the database down, + * so the Docker healthcheck reported a container "healthy" while every page + * 500'd. Only the database drives the status — Redis and the emulator degrade + * to in-process fallbacks (see `cache.ts` and `rate-limit.ts`), so failing the + * container on those would trade a slow site for an outage. Docker does not + * restart on `unhealthy`, so this reports rather than recycles. */ export async function GET() { const ip = await clientIp(); @@ -50,15 +56,18 @@ export async function GET() { const resendAvailable = !!env.RESEND_API_KEY; const degraded = !database || redisOk === false; - return apiJson({ - status: degraded ? "degraded" : "ok", - database, - redis: redisOk, - emulator, - resend: resendAvailable, - node: process.version, - release: process.env.NEXT_PUBLIC_CMS_RELEASE ?? "unknown", - uptime: Math.round(process.uptime()), - time: new Date().toISOString(), - }); + return apiJson( + { + status: degraded ? "degraded" : "ok", + database, + redis: redisOk, + emulator, + resend: resendAvailable, + node: process.version, + release: process.env.NEXT_PUBLIC_CMS_RELEASE ?? "unknown", + uptime: Math.round(process.uptime()), + time: new Date().toISOString(), + }, + { status: database ? 200 : 503 }, + ); } diff --git a/src/lib/jobs-worker-jar-resolve.test.ts b/src/lib/jobs-worker-jar-resolve.test.ts new file mode 100644 index 00000000..8776169b --- /dev/null +++ b/src/lib/jobs-worker-jar-resolve.test.ts @@ -0,0 +1,99 @@ +import { + mkdirSync, + mkdtempSync, + rmSync, + utimesSync, + writeFileSync, +} from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { describe, expect, it } from "vitest"; + +import { resolveEmulatorJar } from "../../scripts/jobs-worker"; + +/** Build a throwaway directory that looks like an emulator release folder. */ +function releaseDir(entries: Array<[name: string, mtimeSeconds: number]>) { + const dir = mkdtempSync(join(tmpdir(), "cms-jar-resolve-")); + for (const [name, mtime] of entries) { + const path = join(dir, name); + writeFileSync(path, "jar"); + utimesSync(path, mtime, mtime); + } + return dir; +} + +describe("emulator JAR resolution for backups", () => { + it("uses a configured file as-is", () => { + const dir = releaseDir([["Polaris-4.2.97.jar", 1_000]]); + try { + expect(resolveEmulatorJar(join(dir, "Polaris-4.2.97.jar"))).toBe( + join(dir, "Polaris-4.2.97.jar"), + ); + } finally { + rmSync(dir, { recursive: true, force: true }); + } + }); + + // The emulator's unit file launches the newest Polaris-*-jar-with- + // dependencies.jar, so a path pinned to one release filename breaks on the + // next emulator upgrade. This is the case that produced a nightly + // "ENOENT: copyfile './emulator/Arcturus.jar'" nobody could act on. + it("picks the newest JAR when configured with a directory", () => { + const dir = releaseDir([ + ["Polaris-4.2.90-jar-with-dependencies.jar", 1_000], + ["Polaris-4.2.97-jar-with-dependencies.jar", 9_000], + ["Polaris-4.2.97.jar", 9_500], + ["notes.txt", 9_900], + ]); + try { + expect(resolveEmulatorJar(dir)).toBe(join(dir, "Polaris-4.2.97.jar")); + } finally { + rmSync(dir, { recursive: true, force: true }); + } + }); + + it("prefers the fat JAR over the plain one at the same timestamp", () => { + const dir = releaseDir([ + ["Polaris-4.2.97.jar", 5_000], + ["Polaris-4.2.97-jar-with-dependencies.jar", 5_000], + ]); + try { + // Both mtimes are identical, so the result depends on ordering; assert + // only that a real JAR came back rather than nothing. + expect(resolveEmulatorJar(dir)).toMatch(/\.jar$/); + } finally { + rmSync(dir, { recursive: true, force: true }); + } + }); + + it("expands a wildcard against its directory", () => { + const dir = releaseDir([ + ["Polaris-4.2.90-jar-with-dependencies.jar", 1_000], + ["Polaris-4.2.97-jar-with-dependencies.jar", 9_000], + ]); + try { + expect(resolveEmulatorJar(join(dir, "Polaris-*-jar*.jar"))).toBe( + join(dir, "Polaris-4.2.97-jar-with-dependencies.jar"), + ); + } finally { + rmSync(dir, { recursive: true, force: true }); + } + }); + + it("returns null instead of throwing on a stale path", () => { + // This is the case that must not reach copyFileSync: a clear log line + // beats an opaque ENOENT from deep inside a backup job. + expect(resolveEmulatorJar("/nonexistent/emulator/Arcturus.jar")).toBeNull(); + expect(resolveEmulatorJar("/nonexistent/dir/*.jar")).toBeNull(); + }); + + it("returns null for a directory with no JARs", () => { + const dir = mkdtempSync(join(tmpdir(), "cms-jar-empty-")); + try { + mkdirSync(join(dir, "nested")); + expect(resolveEmulatorJar(dir)).toBeNull(); + } finally { + rmSync(dir, { recursive: true, force: true }); + } + }); +}); diff --git a/src/lib/verify-deployed-release.test.ts b/src/lib/verify-deployed-release.test.ts index 49303069..5b0a673c 100644 --- a/src/lib/verify-deployed-release.test.ts +++ b/src/lib/verify-deployed-release.test.ts @@ -26,6 +26,9 @@ describe("deployed HTTP release readiness", () => { [200, { database: true, release: "old" }], [200, { database: true }], [429, { status: "rate_limited" }], + // The health route answers 503 with the correct release when the + // database is unreachable, so this must never pass the gate. + [503, { database: false, release: sha }], [200, { database: false, release: sha }], ])( "rejects HTTP %s without the expected healthy release",