fix(cache): true LRU, stale-while-revalidate and cross-process invalidation
Gitea Actions Runner Test / test-job (push) Successful in 0s
CI / check (push) Successful in 32s
CI / tests-integration (push) Successful in 1m38s
CI / tests-unit (push) Successful in 1m42s
CI / tests-ui (push) Successful in 2m33s
CI / preflight (push) Skipped
CI / deploy (push) Successful in 2m43s
Gitea Actions Runner Test / test-job (push) Successful in 0s
CI / check (push) Successful in 32s
CI / tests-integration (push) Successful in 1m38s
CI / tests-unit (push) Successful in 1m42s
CI / tests-ui (push) Successful in 2m33s
CI / preflight (push) Skipped
CI / deploy (push) Successful in 2m43s
The in-process cache was a FIFO of 500 entries that was never touched on a read, so a key polled on every request could be evicted by an unrelated burst of dynamic keys. That looked exactly like the cache being cleared at random, and it is what made the site fall back to the database unpredictably. - Evict least-recently-used instead, and raise the default budget to 2000 (CACHE_MEMORY_MAX_ENTRIES). Reading a key now marks it as used, so a hot key only leaves when a hotter one takes its place. - Add opt-in stale-while-revalidate (CachedOptions.staleMs). The grace window lives on the entry, so one call site opting in protects every reader of that key. A failed background refresh keeps serving the last good value instead of falling through to the origin, and is reported once rather than per read. - Invalidate across processes. invalidateKey() now clears memory, deletes the Redis key and publishes a signal, so a value written by one process is no longer served stale by the others for the rest of its TTL. A failed Redis delete no longer skips the broadcast. - Guard against a refresh that started before an invalidation writing its outdated result back into the cache. - Read the news revision at most once a second per process instead of on every call, with a pub/sub signal to drop the local copy when it rotates. A Redis outage now degrades to the in-process cache rather than to no cache at all. - Warm the hot public keys on boot, so the first visitors after a deploy do not each pay for a miss. - Count hits, misses, stale serves, errors and evictions per key, exposed at GET /api/admin/devops/cache. Without it a wrong REDIS_URL, a full budget and a dead origin all look identical from the outside. - Enforce the imaging cache budget for real: records are .img/.json pairs, so the old cap counted files and never removed anything while entries were fresh. Sweeps are throttled per directory and prune to a low-water mark. - Cap the JWT version map, and stop per-test scratch roots from littering the runtime imaging cache. Public read-only endpoints get grace windows; admin, account and auth data deliberately stays fresh. Redis TTLs get a little jitter so keys written together no longer expire together. 3209 tests pass. next build could not be verified on this host: the optimized build is OOM-killed before prerender, so this has not run in a real Next runtime yet.
This commit is contained in:
1 parent
f490fcc9da
commit
203399aab7
43 files changed
+1737
-212
No files matched your search
@@ -0,0 +1,104 @@
|
||||
// @ts-nocheck
|
||||
import { beforeEach, expect, it, vi } from "vitest";
|
||||
|
||||
const state = vi.hoisted(() => ({
|
||||
// These stand in for the real cache, which runs the query on a miss, so a
|
||||
// failing database really does propagate into the warm-up.
|
||||
cached: vi.fn(
|
||||
async (_key: string, _ttl: number, query: () => Promise<unknown>) =>
|
||||
query(),
|
||||
),
|
||||
redisCache: vi.fn(
|
||||
async (_key: string, _ttl: number, query: () => Promise<unknown>) =>
|
||||
query(),
|
||||
),
|
||||
cacheNews: vi.fn(
|
||||
async (_key: string, _ttl: number, query: () => Promise<unknown>) =>
|
||||
query(),
|
||||
),
|
||||
warn: vi.fn(),
|
||||
// Set by a test to make every origin query reject.
|
||||
failEverything: false,
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/logger", () => ({
|
||||
logger: { warn: state.warn, error: vi.fn(), info: vi.fn() },
|
||||
}));
|
||||
vi.mock("@/lib/cache", () => ({ cached: state.cached }));
|
||||
vi.mock("@/lib/redis-cache", () => ({
|
||||
redisCache: state.redisCache,
|
||||
apiCacheKey: (name: string) => `api:${name}`,
|
||||
}));
|
||||
vi.mock("@/lib/services/news-cache", () => ({ cacheNews: state.cacheNews }));
|
||||
vi.mock("@/lib/services/site-settings", () => ({
|
||||
siteSettings: { get: vi.fn(async () => "7") },
|
||||
}));
|
||||
vi.mock("@/lib/db", () => {
|
||||
// A chainable stand-in for a drizzle query that always resolves to a row.
|
||||
const row = { total: 1, username: "a", look: "l", rank: 7, motto: "m" };
|
||||
// Chainable and awaitable at once, the way a drizzle query builder is.
|
||||
const chain: any = new Proxy(() => {}, {
|
||||
get: (_target, prop) => {
|
||||
if (prop === Symbol.toStringTag) return "Query";
|
||||
if (prop === "then")
|
||||
return (resolve: any, reject: any) =>
|
||||
(state.failEverything
|
||||
? Promise.reject(Error("database is down"))
|
||||
: Promise.resolve([row])
|
||||
).then(resolve, reject);
|
||||
return () => chain;
|
||||
},
|
||||
});
|
||||
return {
|
||||
db: { select: () => chain },
|
||||
User: { online: "online", username: "username", rank: "rank" },
|
||||
Rooms: {},
|
||||
CameraWeb: {},
|
||||
WebsiteTeams: { id: "id", hiddenRank: "hiddenRank" },
|
||||
};
|
||||
});
|
||||
|
||||
import { warmPublicCaches } from "./cache-warmup";
|
||||
|
||||
beforeEach(() => {
|
||||
state.failEverything = false;
|
||||
for (const mock of [
|
||||
state.cached,
|
||||
state.redisCache,
|
||||
state.cacheNews,
|
||||
state.warn,
|
||||
])
|
||||
mock.mockClear();
|
||||
});
|
||||
|
||||
it("primes the keys the public pages actually read", async () => {
|
||||
await warmPublicCaches();
|
||||
const keys = state.cached.mock.calls.map(([key]) => key);
|
||||
expect(keys).toContain("online_count");
|
||||
expect(keys).toContain("total_users");
|
||||
expect(keys).toContain("total_rooms");
|
||||
expect(keys).toContain("total_photos");
|
||||
expect(state.redisCache.mock.calls.map(([key]) => key)).toEqual(
|
||||
expect.arrayContaining(["api:staff", "api:teams"]),
|
||||
);
|
||||
expect(state.cacheNews).toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("gives every primed key a stale window", async () => {
|
||||
// Warming without a grace window would leave the first visitor after a deploy
|
||||
// to still pay for a miss, which is the thing being fixed.
|
||||
await warmPublicCaches();
|
||||
for (const call of [
|
||||
...state.cached.mock.calls,
|
||||
...state.redisCache.mock.calls,
|
||||
])
|
||||
expect(call[3]?.staleMs).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("survives a database that is completely unavailable", async () => {
|
||||
// register() calls this without awaiting: it must never reject, or an
|
||||
// unhandled rejection would take the process down on every boot.
|
||||
state.failEverything = true;
|
||||
await expect(warmPublicCaches()).resolves.toBeUndefined();
|
||||
expect(state.warn).toHaveBeenCalled();
|
||||
});
|
||||
@@ -0,0 +1,135 @@
|
||||
import "server-only";
|
||||
|
||||
import { asc, count, desc, eq, gte } from "drizzle-orm";
|
||||
import { cached } from "@/lib/cache";
|
||||
import { CameraWeb, db, Rooms, User, WebsiteTeams } from "@/lib/db";
|
||||
import { logger } from "@/lib/logger";
|
||||
import { apiCacheKey, redisCache } from "@/lib/redis-cache";
|
||||
import { cacheNews } from "@/lib/services/news-cache";
|
||||
import { siteSettings } from "@/lib/services/site-settings";
|
||||
|
||||
/**
|
||||
* Prime the hot public cache keys right after boot.
|
||||
*
|
||||
* Every deploy starts with an empty in-process cache, and any key that also
|
||||
* expired while the old process was down has to be recomputed from the database.
|
||||
* With no warm-up, the first visitor after each deploy pays for a burst of
|
||||
* simultaneous misses; with it, the site is already warm before traffic arrives.
|
||||
*
|
||||
* Warming works by cache *key*, not by call site: these use exactly the keys the
|
||||
* routes and pages read, so priming an entry serves every reader of it. Failures
|
||||
* are logged and ignored — a warm-up that cannot reach the database must never
|
||||
* stop the server from serving.
|
||||
*/
|
||||
|
||||
const COUNTER_TTL_MS = 300_000;
|
||||
const ONLINE_TTL_MS = 10_000;
|
||||
|
||||
async function warm<T>(name: string, load: () => Promise<T>): Promise<void> {
|
||||
try {
|
||||
await load();
|
||||
} catch (error) {
|
||||
logger.warn("[cache] warm-up failed", {
|
||||
key: name,
|
||||
error: String(error),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
export async function warmPublicCaches(): Promise<void> {
|
||||
// Small delays between groups: a boot-time burst of COUNT(*) queries against
|
||||
// a database that is still opening connections helps nobody.
|
||||
await warm("online_count", () =>
|
||||
cached("online_count", ONLINE_TTL_MS, countOnline, { staleMs: 15_000 }),
|
||||
);
|
||||
await warm("total_users", () =>
|
||||
cached("total_users", COUNTER_TTL_MS, countUsers, {
|
||||
staleMs: COUNTER_TTL_MS,
|
||||
}),
|
||||
);
|
||||
await warm("total_rooms", () =>
|
||||
cached("total_rooms", COUNTER_TTL_MS, countRooms, {
|
||||
staleMs: COUNTER_TTL_MS,
|
||||
}),
|
||||
);
|
||||
await warm("total_photos", () =>
|
||||
cached("total_photos", COUNTER_TTL_MS, countPhotos, {
|
||||
staleMs: COUNTER_TTL_MS,
|
||||
}),
|
||||
);
|
||||
await warm("online_users", () =>
|
||||
cached("online_users", ONLINE_TTL_MS, listOnlineUsers, {
|
||||
staleMs: 15_000,
|
||||
}),
|
||||
);
|
||||
await warm("api:staff", () =>
|
||||
redisCache(apiCacheKey("staff"), 300, loadStaff, { staleMs: 300 }),
|
||||
);
|
||||
await warm("api:teams", () =>
|
||||
redisCache(apiCacheKey("teams"), 300, loadTeams, { staleMs: 600 }),
|
||||
);
|
||||
await warm("news:home", () =>
|
||||
cacheNews(apiCacheKey("home"), 15_000, async () => ({ primed: true })),
|
||||
);
|
||||
}
|
||||
|
||||
async function countOnline(): Promise<number> {
|
||||
const [row] = await db
|
||||
.select({ total: count() })
|
||||
.from(User)
|
||||
.where(eq(User.online, "1"));
|
||||
return row?.total ?? 0;
|
||||
}
|
||||
|
||||
async function countUsers(): Promise<number> {
|
||||
const [row] = await db.select({ total: count() }).from(User);
|
||||
return row?.total ?? 0;
|
||||
}
|
||||
|
||||
async function countRooms(): Promise<number> {
|
||||
const [row] = await db.select({ total: count() }).from(Rooms);
|
||||
return row?.total ?? 0;
|
||||
}
|
||||
|
||||
async function countPhotos(): Promise<number> {
|
||||
const [row] = await db.select({ total: count() }).from(CameraWeb);
|
||||
return row?.total ?? 0;
|
||||
}
|
||||
|
||||
async function listOnlineUsers() {
|
||||
return db
|
||||
.select({ username: User.username, look: User.look })
|
||||
.from(User)
|
||||
.where(eq(User.online, "1"))
|
||||
.limit(100);
|
||||
}
|
||||
|
||||
async function loadStaff() {
|
||||
const minStaffRank =
|
||||
Number(await siteSettings.get("min_staff_rank", "7")) || 7;
|
||||
return db
|
||||
.select({
|
||||
username: User.username,
|
||||
look: User.look,
|
||||
rank: User.rank,
|
||||
motto: User.motto,
|
||||
})
|
||||
.from(User)
|
||||
.where(gte(User.rank, minStaffRank))
|
||||
.orderBy(desc(User.rank), asc(User.username))
|
||||
.limit(100);
|
||||
}
|
||||
|
||||
async function loadTeams() {
|
||||
return db
|
||||
.select({
|
||||
id: WebsiteTeams.id,
|
||||
rankName: WebsiteTeams.rankName,
|
||||
badge: WebsiteTeams.badge,
|
||||
jobDescription: WebsiteTeams.jobDescription,
|
||||
staffColor: WebsiteTeams.staffColor,
|
||||
})
|
||||
.from(WebsiteTeams)
|
||||
.where(eq(WebsiteTeams.hiddenRank, false))
|
||||
.orderBy(asc(WebsiteTeams.id));
|
||||
}
|
||||
@@ -6,10 +6,33 @@ const state = vi.hoisted(() => ({
|
||||
values: new Map<string, unknown>(),
|
||||
get: vi.fn(),
|
||||
set: vi.fn(),
|
||||
publish: vi.fn(),
|
||||
status: "ready",
|
||||
// Set by the subscriber mock so a test can simulate another process
|
||||
// publishing a new revision.
|
||||
deliver: null as ((channel: string, message: string) => void) | null,
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/redis", () => ({
|
||||
redis: {
|
||||
get: (key: string) => state.get(key),
|
||||
set: (key: string, value: string) => state.set(key, value),
|
||||
publish: (channel: string, message: string) =>
|
||||
state.publish(channel, message),
|
||||
get status() {
|
||||
return state.status;
|
||||
},
|
||||
duplicate: () => ({
|
||||
on: (event: string, handler: (...args: unknown[]) => void) => {
|
||||
if (event === "message") state.deliver = handler;
|
||||
},
|
||||
subscribe: async () => 1,
|
||||
}),
|
||||
},
|
||||
}));
|
||||
vi.mock("@/lib/logger", () => ({
|
||||
logger: { error: vi.fn(), warn: vi.fn() },
|
||||
}));
|
||||
vi.mock("@/lib/redis", () => ({ redis: state }));
|
||||
vi.mock("@/lib/logger", () => ({ logger: { error: vi.fn() } }));
|
||||
vi.mock("@/lib/cache", () => ({
|
||||
cached: async (key: string, _ttl: number, query: () => Promise<unknown>) => {
|
||||
if (state.values.has(key)) return state.values.get(key);
|
||||
@@ -35,7 +58,9 @@ beforeEach(() => {
|
||||
.mockImplementation(async (_key: string, value: string) => {
|
||||
state.revision = value;
|
||||
});
|
||||
state.publish.mockReset().mockResolvedValue(1);
|
||||
});
|
||||
|
||||
it("invalidates a previously cached public list", async () => {
|
||||
const query = vi
|
||||
.fn()
|
||||
@@ -45,6 +70,36 @@ it("invalidates a previously cached public list", async () => {
|
||||
await invalidateNewsCache();
|
||||
expect(await cacheNews("list", 60000, query)).toEqual(["new article"]);
|
||||
});
|
||||
|
||||
it("reads the revision at most once for a burst of reads", async () => {
|
||||
// The revision used to be fetched from Redis on every single call, which put
|
||||
// a round-trip in front of the very fast path the cache exists to provide.
|
||||
const before = state.get.mock.calls.length;
|
||||
await cacheNews("burst-a", 60000, async () => ["a"]);
|
||||
await cacheNews("burst-b", 60000, async () => ["b"]);
|
||||
await cacheNews("burst-c", 60000, async () => ["c"]);
|
||||
expect(state.get.mock.calls.length - before).toBeLessThanOrEqual(1);
|
||||
});
|
||||
|
||||
it("picks up a revision published by another process", async () => {
|
||||
expect(await cacheNews("list", 60000, async () => ["old"])).toEqual(["old"]);
|
||||
|
||||
// Another process rotates the revision; the local copy must be dropped so the
|
||||
// next read re-reads it instead of serving the previous namespace.
|
||||
state.revision = "rev-from-worker";
|
||||
state.deliver?.("cache:news-revision", "rev-from-worker");
|
||||
|
||||
expect(await cacheNews("list", 60000, async () => ["new"])).toEqual(["new"]);
|
||||
});
|
||||
|
||||
it("tells other processes when the revision rotates", async () => {
|
||||
await invalidateNewsCache();
|
||||
expect(state.publish).toHaveBeenCalledWith(
|
||||
"cache:news-revision",
|
||||
expect.any(String),
|
||||
);
|
||||
});
|
||||
|
||||
it("does not let a stale in-flight read replace a newer revision", async () => {
|
||||
let finish!: (value: string[]) => void;
|
||||
const old = cacheNews(
|
||||
@@ -64,6 +119,7 @@ it("does not let a stale in-flight read replace a newer revision", async () => {
|
||||
"new",
|
||||
]);
|
||||
});
|
||||
|
||||
it("reads fresh data when Redis is unavailable", async () => {
|
||||
state.get.mockRejectedValue(Error("offline"));
|
||||
expect(await cacheNews("list", 60000, async () => ["fresh"])).toEqual([
|
||||
@@ -71,10 +127,20 @@ it("reads fresh data when Redis is unavailable", async () => {
|
||||
]);
|
||||
});
|
||||
|
||||
it("still caches in-process when Redis is unavailable", async () => {
|
||||
// A Redis outage must not turn every public news read into a database query.
|
||||
state.status = "end";
|
||||
const query = vi.fn(async () => ["fresh"]);
|
||||
expect(await cacheNews("list", 60000, query)).toEqual(["fresh"]);
|
||||
expect(await cacheNews("list", 60000, query)).toEqual(["fresh"]);
|
||||
expect(query).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("keeps failed durable cache invalidations retryable", async () => {
|
||||
state.set.mockRejectedValue(Error("offline"));
|
||||
await expect(refreshNewsCacheForDelivery()).rejects.toThrow("offline");
|
||||
});
|
||||
|
||||
it("does not acknowledge an ended Redis connection as refreshed", async () => {
|
||||
state.status = "end";
|
||||
await expect(refreshNewsCacheForDelivery()).rejects.toThrow();
|
||||
|
||||
@@ -1,28 +1,83 @@
|
||||
import "server-only";
|
||||
|
||||
import { randomUUID } from "node:crypto";
|
||||
import { cached } from "@/lib/cache";
|
||||
import { onCacheSignal, publishCacheSignal } from "@/lib/cache-invalidation";
|
||||
import { logger } from "@/lib/logger";
|
||||
import { redis } from "@/lib/redis";
|
||||
|
||||
const REVISION_KEY = "cms:news:revision";
|
||||
|
||||
/** News entries are namespaced by a revision, so a new revision drops them all. */
|
||||
const REVISION_CHANNEL = "cache:news-revision";
|
||||
|
||||
// The revision is read from Redis at most this often per process. Reading it on
|
||||
// every call (as this used to) put a Redis round-trip in front of the in-process
|
||||
// fast path, which is exactly what the cache exists to avoid. A published article
|
||||
// still shows up immediately: the invalidation signal clears this copy, and the
|
||||
// window is only the fallback for when pub/sub is unavailable.
|
||||
const REVISION_TTL_MS = 1_000;
|
||||
|
||||
let revision: { value: string; expiresAt: number } | null = null;
|
||||
let inFlightRevision: Promise<string> | null = null;
|
||||
|
||||
async function readRevision(): Promise<string> {
|
||||
const now = Date.now();
|
||||
if (revision && revision.expiresAt > now) return revision.value;
|
||||
inFlightRevision ??= (async () => {
|
||||
try {
|
||||
return (await redis?.get(REVISION_KEY)) ?? "0";
|
||||
} catch {
|
||||
return "0";
|
||||
} finally {
|
||||
inFlightRevision = null;
|
||||
}
|
||||
})();
|
||||
const value = await inFlightRevision;
|
||||
revision = { value, expiresAt: Date.now() + REVISION_TTL_MS };
|
||||
return value;
|
||||
}
|
||||
|
||||
function setRevision(value: string): void {
|
||||
revision = { value, expiresAt: Date.now() + REVISION_TTL_MS };
|
||||
}
|
||||
|
||||
// Another process published new news: forget the revision so the next read picks
|
||||
// up the new one instead of serving entries under the old namespace.
|
||||
onCacheSignal(REVISION_CHANNEL, () => {
|
||||
revision = null;
|
||||
});
|
||||
|
||||
export async function cacheNews<T>(
|
||||
key: string,
|
||||
ttlMs: number,
|
||||
query: () => Promise<T>,
|
||||
): Promise<T> {
|
||||
if (!redis || redis.status === "end") return query();
|
||||
let revision: string;
|
||||
try {
|
||||
revision = (await redis.get(REVISION_KEY)) ?? "0";
|
||||
} catch {
|
||||
return query();
|
||||
}
|
||||
return cached(`news:${revision}:${key}`, ttlMs, query);
|
||||
// Every caller of cacheNews is a public news read, so the grace window lives
|
||||
// here rather than at each call site. Publishing an article rotates the
|
||||
// revision, which drops these entries immediately; the window only matters
|
||||
// when that signal cannot be delivered.
|
||||
const options = { staleMs: ttlMs };
|
||||
// Without Redis the revision cannot be shared, so fall back to a fixed one:
|
||||
// the in-process cache still works, which is far better than hitting the
|
||||
// database on every request for the whole duration of a Redis outage.
|
||||
if (!redis || redis.status === "end")
|
||||
return cached(`news:0:${key}`, ttlMs, query, options);
|
||||
return cached(`news:${await readRevision()}:${key}`, ttlMs, query, options);
|
||||
}
|
||||
|
||||
async function rotateRevision(): Promise<string> {
|
||||
const next = randomUUID();
|
||||
if (redis) await redis.set(REVISION_KEY, next);
|
||||
setRevision(next);
|
||||
await publishCacheSignal(REVISION_CHANNEL, next);
|
||||
return next;
|
||||
}
|
||||
|
||||
export async function invalidateNewsCache(): Promise<void> {
|
||||
if (!redis || redis.status === "end") return;
|
||||
try {
|
||||
await redis.set(REVISION_KEY, randomUUID());
|
||||
await rotateRevision();
|
||||
} catch (error) {
|
||||
logger.error("News saved but public cache invalidation failed", {
|
||||
module: "news",
|
||||
@@ -36,5 +91,5 @@ export async function refreshNewsCacheForDelivery(): Promise<void> {
|
||||
if (!redis) return;
|
||||
if (redis.status === "end")
|
||||
throw new Error("News cache connection is closed");
|
||||
await redis.set(REVISION_KEY, randomUUID());
|
||||
await rotateRevision();
|
||||
}
|
||||
Reference in new issue
Block a user