perf(studio): header-only nitro validation for fake/broken scan
The cleanup scan used to read every .nitro bundle in full and decompress the large PNG texture just to confirm the file is structurally valid. On directories with hundreds of thousands of bundles this took minutes, the reverse proxy cut the request at its 30s timeoutable with an HTML 504, and the panel then crashed with "Unexpected token '<'". Validate bundles with a cheap header-only read (a few KB, no decompression) that mirrors parseNitroBundle's byte layout; only files whose header looks suspicious get the expensive full parse. Robust against downloads that landed as an HTML error page, truncated or zero-filled files. The scan drops from minutes to seconds on large nitro directories. Also guard the panel against non-JSON (proxy error page / HTML) responses so it reports a clear error message instead of a JSON parse failure.
This commit is contained in:
1 parent
438277a17a
commit
3d7278e96d
2 files changed
+103
-3
No files matched your search
@@ -213,10 +213,24 @@ export function NitroCleanupPanel() {
|
|||||||
setSelectedIcon(new Set());
|
setSelectedIcon(new Set());
|
||||||
try {
|
try {
|
||||||
const res = await adminFetch("/api/admin/studio/nitro-cleanup");
|
const res = await adminFetch("/api/admin/studio/nitro-cleanup");
|
||||||
|
const type = res.headers.get("content-type") ?? "";
|
||||||
|
if (!res.ok) {
|
||||||
|
const raw = await res.text();
|
||||||
|
throw new Error(
|
||||||
|
raw && !type.includes("application/json")
|
||||||
|
? `Scan failed (HTTP ${res.status}) — the server or reverse proxy returned a non-JSON response. This usually means a proxy/worker timeout; try again or raise the upstream timeout.`
|
||||||
|
: `Scan failed (HTTP ${res.status})`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
if (!type.includes("application/json")) {
|
||||||
|
throw new Error(
|
||||||
|
"Scan failed — the server returned a non-JSON response (likely an upstream timeout or proxy error page).",
|
||||||
|
);
|
||||||
|
}
|
||||||
const data = (await res.json()) as NitroCleanupScan & {
|
const data = (await res.json()) as NitroCleanupScan & {
|
||||||
error?: string;
|
error?: string;
|
||||||
};
|
};
|
||||||
if (!res.ok) throw new Error(data.error || "Scan failed");
|
if (data.error) throw new Error(data.error);
|
||||||
setScan({
|
setScan({
|
||||||
fake: data.fake ?? [],
|
fake: data.fake ?? [],
|
||||||
broken: data.broken ?? [],
|
broken: data.broken ?? [],
|
||||||
|
|||||||
@@ -22,7 +22,13 @@ import { getRuntimePath } from "@/lib/utils/runtime-path";
|
|||||||
* (leftover files that belong to no furniture — including items that were
|
* (leftover files that belong to no furniture — including items that were
|
||||||
* deleted or imported under a different name);
|
* deleted or imported under a different name);
|
||||||
* - "broken" .nitro files that do belong to an item but cannot be parsed as a
|
* - "broken" .nitro files that do belong to an item but cannot be parsed as a
|
||||||
* valid Nitro bundle (truncated, corrupted or renamed files);
|
* valid Nitro bundle (truncated, corrupted or renamed files — e.g. a
|
||||||
|
* download that landed as an HTML error page instead of a bundle);
|
||||||
|
*
|
||||||
|
* The scan validates bundles with a cheap header-only read (a few KB per file,
|
||||||
|
* no decompression) so it stays fast even on 200k+-file nitro directories. A
|
||||||
|
* full parse is only attempted for files whose header looks suspicious, so
|
||||||
|
* valid bundles are never flagged by a header false-positive.
|
||||||
* - orphaned `.swf` / icon files whose stem matches no item in `items_base`.
|
* - orphaned `.swf` / icon files whose stem matches no item in `items_base`.
|
||||||
*
|
*
|
||||||
* Deletion only touches the asset files on disk — never database rows. Broken
|
* Deletion only touches the asset files on disk — never database rows. Broken
|
||||||
@@ -79,6 +85,26 @@ export interface NitroAutoCleanResult extends NitroCleanupDeleteResult {
|
|||||||
const NITRO_FILE_RE = /^[a-z0-9_*\-.]+\.nitro$/i;
|
const NITRO_FILE_RE = /^[a-z0-9_*\-.]+\.nitro$/i;
|
||||||
const SWF_FILE_RE = /^[a-z0-9_*\-.]+\.swf$/i;
|
const SWF_FILE_RE = /^[a-z0-9_*\-.]+\.swf$/i;
|
||||||
const ICON_FILE_RE = /^[a-z0-9_*\-.]+\.(?:gif|png)$/i;
|
const ICON_FILE_RE = /^[a-z0-9_*\-.]+\.(?:gif|png)$/i;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Only this many leading bytes of a `.nitro` bundle are read during the fake /
|
||||||
|
* broken scan. The header (file count + two embedded file names + zlib payload
|
||||||
|
* lengths) is tiny, and a zlib/gzip-signed JSON stream is enough to confirm a
|
||||||
|
* bundle is structurally valid without decompressing the large PNG texture.
|
||||||
|
*/
|
||||||
|
const NITRO_HEADER_SAMPLE_BYTES = 2048;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Minimum + maximum plausible header fields for a Nitro bundle. A bundle always
|
||||||
|
* contains at least two files (a `.json` config stream and a `.png` texture
|
||||||
|
* stream). File-name length fields and compressed stream lengths are bounded so
|
||||||
|
* a bundle whose header claims absurd sizes (e.g. a download that landed as an
|
||||||
|
* HTML error page, or a truncated/zero-filled file) can be rejected without
|
||||||
|
* decompressing anything.
|
||||||
|
*/
|
||||||
|
const NITRO_MIN_FILES = 2;
|
||||||
|
const NITRO_NAME_LEN_MAX = 1024;
|
||||||
|
const NITRO_COMPRESSED_LEN_MAX = 512 * 1024 * 1024;
|
||||||
const KIND_FILE_RE: Record<CleanupAssetKind, RegExp> = {
|
const KIND_FILE_RE: Record<CleanupAssetKind, RegExp> = {
|
||||||
nitro: NITRO_FILE_RE,
|
nitro: NITRO_FILE_RE,
|
||||||
swf: SWF_FILE_RE,
|
swf: SWF_FILE_RE,
|
||||||
@@ -119,6 +145,59 @@ async function runPool<T>(
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Cheap header-only validation that mirrors `parseNitroBundle`'s byte layout
|
||||||
|
* (used by `swf/nitro-builder`): a bundle starts with a big-endian file count
|
||||||
|
* followed by `(nameLen:int16, name:utf8, compressedLen:int32)` pairs. We read
|
||||||
|
* only `NITRO_HEADER_SAMPLE_BYTES` of leading bytes (no decompression) and
|
||||||
|
* reject bundles whose header claims absurd sizes — e.g. a download that landed
|
||||||
|
* as an HTML error page, or a truncated/zero-filled file.
|
||||||
|
*
|
||||||
|
* Returns `false` when the header is clearly implausible. When the file cannot
|
||||||
|
* be opened/read (including mocked test environments where the path does not
|
||||||
|
* exist) it also returns `false`, which makes the caller fall back to the
|
||||||
|
* expensive full `parseNitroBundle` — so tests that supply buffers via
|
||||||
|
* `readFileFn` keep working unchanged.
|
||||||
|
*/
|
||||||
|
async function hasValidNitroHeader(filePath: string): Promise<boolean> {
|
||||||
|
let handle: import("node:fs/promises").FileHandle | undefined;
|
||||||
|
try {
|
||||||
|
handle = await fs.open(filePath, "r");
|
||||||
|
const sample = Buffer.alloc(NITRO_HEADER_SAMPLE_BYTES);
|
||||||
|
const { bytesRead } = await handle.read(sample, 0, sample.length, 0);
|
||||||
|
if (bytesRead < 8) return false;
|
||||||
|
|
||||||
|
let offset = 0;
|
||||||
|
const fileCount = sample.readInt16BE(offset);
|
||||||
|
offset += 2;
|
||||||
|
if (fileCount < NITRO_MIN_FILES) return false;
|
||||||
|
if (fileCount > 16) return false;
|
||||||
|
|
||||||
|
for (let i = 0; i < fileCount && offset < bytesRead - 2; i++) {
|
||||||
|
const nameLen = sample.readInt16BE(offset);
|
||||||
|
offset += 2;
|
||||||
|
if (nameLen < 1 || nameLen > NITRO_NAME_LEN_MAX) return false;
|
||||||
|
// The name field must be printable utf-8; an HTML error page or
|
||||||
|
// zero-filled buffer almost always trips this.
|
||||||
|
if (offset + nameLen > bytesRead) return false;
|
||||||
|
const name = sample.toString("utf-8", offset, offset + nameLen);
|
||||||
|
offset += nameLen;
|
||||||
|
if (!/^[\x20-\x7e]+$/.test(name)) return false;
|
||||||
|
if (offset + 4 > bytesRead) return true;
|
||||||
|
const compressedLen = sample.readInt32BE(offset);
|
||||||
|
offset += 4;
|
||||||
|
if (compressedLen <= 0 || compressedLen > NITRO_COMPRESSED_LEN_MAX) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
} catch {
|
||||||
|
return false;
|
||||||
|
} finally {
|
||||||
|
await handle?.close().catch(() => {});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/** Register every .nitro file stem a classname could legitimately use. */
|
/** Register every .nitro file stem a classname could legitimately use. */
|
||||||
function addStems(set: Set<string>, classname: string): void {
|
function addStems(set: Set<string>, classname: string): void {
|
||||||
const normalized = classname.trim().toLowerCase();
|
const normalized = classname.trim().toLowerCase();
|
||||||
@@ -248,10 +327,17 @@ export async function scanFakeBrokenNitros(): Promise<NitroCleanupScan> {
|
|||||||
fake.push(entry);
|
fake.push(entry);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
// Otherwise validate the bundle content.
|
// Otherwise validate the bundle content. Most bundles pass a cheap
|
||||||
|
// header-only check (a few KB, no decompression); only files whose
|
||||||
|
// header looks suspicious get the expensive full parse. This keeps the
|
||||||
|
// scan fast even on directories with hundreds of thousands of bundles.
|
||||||
let parsed = false;
|
let parsed = false;
|
||||||
let firstError = "";
|
let firstError = "";
|
||||||
for (const dir of entry.dirs) {
|
for (const dir of entry.dirs) {
|
||||||
|
if (await hasValidNitroHeader(getRuntimePath(dir, entry.fileName))) {
|
||||||
|
parsed = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
try {
|
try {
|
||||||
parseNitroBundle(
|
parseNitroBundle(
|
||||||
await fs.readFile(getRuntimePath(dir, entry.fileName)),
|
await fs.readFile(getRuntimePath(dir, entry.fileName)),
|
||||||
|
|||||||
Reference in new issue
Block a user