From 52459c0b744e4916ee1f32cdb2590003cf47b94a Mon Sep 17 00:00:00 2001 From: openhands Date: Mon, 7 Sep 2026 16:54:44 +0200 Subject: [PATCH] feat(db): add self-tuned mariadb-turbo container and bulk JSON importer - docker-compose: opt-in mariadb-turbo service (profile 'db') with inline mysqld tuning (max_allowed_packet=512M, innodb_flush_log_at_trx_commit=2, 2G buffer pool, net_read/write_timeout=600) for >50 MB JSON bulk loads; named volume for the datadir + node_modules host-sync guidance - scripts/bulk-import-json.ts: chunked (2000-row) idempotent importer with ON DUPLICATE KEY UPDATE, resumable, habbo furnidata or flat-array input - src/db/schema-gamedata.ts: promote+index JSON storage schema (furnidata, docs, texts) incl. VIRTUAL generated columns for MariaDB - package.json: add db:bulk, db:up, db:down, db:introspect, db:schema:generate --- README.md | 79 ++++++++++- docker-compose.yml | 98 +++++++++++++- package.json | 5 + scripts/bulk-import-json.ts | 254 ++++++++++++++++++++++++++++++++++++ src/db/schema-gamedata.ts | 106 +++++++++++++++ 5 files changed, 540 insertions(+), 2 deletions(-) create mode 100644 scripts/bulk-import-json.ts create mode 100644 src/db/schema-gamedata.ts diff --git a/README.md b/README.md index 969d2f06..7aac9e5a 100644 --- a/README.md +++ b/README.md @@ -1,4 +1,4 @@ -# EpicNext-CMS v1.0.1 +# EpicNext-CMS v1.0.3 A modern, high-performance content management system for Habbo hotel emulators, built on **Next.js 16** (App Router) with **Drizzle ORM** and **React 19**. Designed to integrate seamlessly with Polaris / Arcturus Morningstar MySQL/MariaDB databases. @@ -196,6 +196,7 @@ an exit of 0 means the CMS is healthy on the new commit. | `docker compose pull && docker compose up -d --build` | Update and redeploy | | `pnpm db:migrate` (on the **host**) | Run database migrations (slim image has no source) | | `./scripts/docker-update.sh` | Full automated update (manual or cron) | +| `pnpm db:up` / `pnpm db:down` | Start / stop the `mariadb-turbo` bulk-load container | ### How Package Manager Detection Works @@ -258,6 +259,45 @@ The container includes a health check that hits `/api/health` on port 3002 every docker inspect --format='{{.State.Health.Status}}' epicnext-cms ``` +### MariaDB Turbo (bulk loads) + +`docker-compose.yml` ships an opt-in **`mariadb-turbo`** service — a dedicated MariaDB 11 container tuned for loading >50 MB JSON dumps (furnidata, external texts, etc.). It configures itself via mysqld flags on the `command:`, so no external `.cnf` file has to be mounted or kept in sync: + +```bash +pnpm db:up # docker compose --profile db up -d (starts mariadb-turbo) +``` + +Key settings (see the `command:` block in `docker-compose.yml`): + +- `max_allowed_packet = 512M` — 50 MB JSON batches no longer hit the 16 MB packet ceiling. +- `innodb_flush_log_at_trx_commit = 2` + `innodb_doublewrite = 0` — durability relaxed for bulk writes. +- `innodb_buffer_pool_size = 2G`, `innodb_log_file_size = 1G`, `bulk_insert_buffer_size = 512M`. +- `net_read_timeout` / `net_write_timeout = 600`, `wait_timeout = 3600` — fixes the `drizzle-kit` **"Pulling schema from database..."** hang by never starving introspection/DDL sessions behind a long import. +- `performance_schema = OFF` — saves ~1-2 GB RAM. + +The datadir lives in a **named volume** (`mariadb-turbo-data`), never a host bind-mount: shared-filesystem sync trashes InnoDB files the same way it corrupts pnpm's `node_modules`. Named volumes stay inside the container filesystem, so 50 MB of JSON writes never cross a host-sync boundary. + +> **Port note:** the service uses `network_mode: host` and binds `127.0.0.1:3306` — stop the host MariaDB first, otherwise the port collides with the standalone `.env` `DATABASE_URL` (`localhost:3306`). + +#### node_modules & host-sync corruption + +pnpm's store is hard-linked and its `.bin` shims are symlinks, so **node_modules must never be a host bind-mount** (Docker Desktop gRPC-FUSE/VirtioFS, Unison and Syncthing all corrupt it). The production build bakes `node_modules` into the image at build time; it is never bind-mounted. For local dev, mount a *named volume* (`node_modules:/app/node_modules`) instead of the host directory, and keep `node_modules` / the pnpm store out of any shared-filesystem bind. + +#### Loading a 50 MB JSON dump + +```bash +# Habbo furnidata (auto-detects roomitemtypes / wallitemtypes / effecttypes) +pnpm db:bulk --file=/var/www/Gamedata/config/FurnitureData.json + +# Flat array of documents → generic JSON store +pnpm db:bulk --file=/data/products.json --table=docs --category=furni + +# Key/value texts +pnpm db:bulk --file=/data/external_texts.json --table=texts --category=default +``` + +The importer (`scripts/bulk-import-json.ts`) maps rows onto `src/db/schema-gamedata.ts`, batches them into **2000-row multi-row INSERTs** (one statement per chunk), uses `ON DUPLICATE KEY UPDATE` so re-runs are idempotent and interrupted loads resume, and prints rows/s + ETA. See ["ORM Setup & Type Generation"](#orm-setup--type-generation) → *Bulk JSON storage* for the design. + --- @@ -284,10 +324,41 @@ const found = await db.select() | `pnpm db:generate` | Draft SQL from Drizzle schema into `drizzle/drafts/` (review + copy) | | `pnpm db:studio` | Open Drizzle Studio (dev only) | | `pnpm db:introspect` | Reverse-engineer an existing DB into a Drizzle schema draft | +| `pnpm db:bulk` | Batch-import >50 MB JSON (2000-row chunks, resumable) via `scripts/bulk-import-json.ts` | | `pnpm db:schema:generate` | Regen committed `src/db/schema.ts` from previous schema + live DB | > The CMS does **not** use `drizzle-kit push` or `drizzle-kit migrate` — the database is shared with the emulator. Apply CMS DDL only via `pnpm db:migrate`. +#### Bulk JSON storage (MariaDB) + +Emulator/hotel JSON dumps (furnidata, external texts, product data) are often >50 MB. MariaDB's `JSON` type is an alias for `LONGTEXT` and can **never be indexed directly**, so `src/db/schema-gamedata.ts` follows the "promote + index" pattern: + +- The raw JSON document stays intact in a `longtext` / `mediumtext` column (`payload`, `value`). +- The fields you actually `WHERE` / `ORDER BY` on are promoted into real columns (`sprite_id`, `class_name`, `kind`, `title`, `text_key`) and indexed. +- Optional **VIRTUAL generated columns** extract indexed fields from the payload with `JSON_EXTRACT` (MariaDB supports secondary indexes on virtual columns, 10.2+), so no duplicate data has to be written by the importer. + +Three tables are exported: + +| Table | Purpose | Unique key | +| -------------------- | ---------------------------------------------- | --------------------------- | +| `gamedata_furnidata` | One row per furni/clothing item (habbo furnidata_json shape) | `(source, sprite_id)` | +| `gamedata_docs` | Generic JSON document store (per-key documents) | `(category, doc_key)` | +| `gamedata_texts` | External-texts style key/value pairs | `(category, text_key)` | + +The bulk importer (`scripts/bulk-import-json.ts`, run via `pnpm db:bulk`) is DB-driven and fast precisely because each chunk is a *single* multi-row `INSERT … VALUES () ()… ON DUPLICATE KEY UPDATE`, so a 50 MB dump is a few dozen statements instead of hundreds of thousands of round-trips. Flags: + +| Flag | Default | Description | +| -------------------- | ----------- | ------------------------------------------------- | +| `--file=` | — (required)| JSON file (habbo furnidata_json or flat array) | +| `--table=` | `furnidata` | One of `furnidata` \| `docs` \| `texts` | +| `--source=` | `habbo` | `source` value for `furnidata` | +| `--category=` | `default` | `category` value for `docs` / `texts` | +| `--chunk-size=N` | `2000` | Rows per multi-row INSERT | +| `--limit=N` | `0` | Stop after N rows (dry-test) | +| `--truncate` | off | DELETE rows for this source/category first | + +The script sets `FOREIGN_KEY_CHECKS=0` for the session and is safe to interrupt: each chunk commits on its own, and `ON DUPLICATE KEY UPDATE` makes re-runs overwrite instead of appending. + --- ## DragonflyDB (Caching, Rate Limiting, SSE) @@ -503,6 +574,8 @@ slow_query_log_file = /var/log/mysql/mariadb-slow.log The slow-query log surfaces hot-spot SQL for a follow-up index/query audit. The CMS connection pool itself is already tuned (pool size 25, `connection_limit` in `DATABASE_URL`). +For bulk-written tables (game-data JSON, imports) a containerized **`mariadb-turbo`** variant is available — see [MariaDB Turbo](#mariadb-turbo-bulk-loads). + ### 3. Asset caching headers | Path | Cache-Control | @@ -553,6 +626,8 @@ pm2 restart next | `pnpm db:generate` | Draft SQL via drizzle-kit → `drizzle/drafts/` | | `pnpm db:studio` | Drizzle Studio (dev) | | `pnpm db:introspect` | drizzle-kit introspect (draft) | +| `pnpm db:bulk` | Batch-import >50 MB JSON via `scripts/bulk-import-json.ts` | +| `pnpm db:up` / `pnpm db:down` | Start / stop the `mariadb-turbo` container | | `pnpm gamedata:compress` | Pre-compress large gamedata JSON to `.gz` (gzip_static) | | `pnpm analyze` | Build + open bundle analyzer | | `pnpm jobs:worker` | Start background task worker | @@ -589,12 +664,14 @@ pm2 restart next │ └── drafts/ # drizzle-kit generate output (never auto-applied) ├── scripts/ │ ├── apply-migrations.ts # SQL migration runner (apply + status) +│ ├── bulk-import-json.ts # 50 MB+ JSON importer (2000-row chunks, UPSERT) │ ├── jobs-worker.ts # Background task scheduler │ ├── generate-drizzle-schema.mjs # Regen src/db/schema.ts from schema + live DB │ └── compress-gamedata.mjs # Pre-compress large gamedata JSON (gzip_static) ├── src/ │ ├── db/ │ │ ├── schema.ts # Drizzle ORM schema (committed — runtime data layer) +│ │ ├── schema-gamedata.ts # Large-JSON storage schema (furnidata/docs/texts) │ │ └── relations.ts # Drizzle relations │ ├── app/ # Next.js App Router (pages & API routes) │ ├── actions/ # Server Actions diff --git a/docker-compose.yml b/docker-compose.yml index e69ee833..ba7dd8e3 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -36,6 +36,16 @@ services: # same directory, so mount it into the container at the same absolute path. # Must be RW so furniture/badge imports can write mirrors to it. - /var/www/Gamedata:/var/www/Gamedata + # + # ── node_modules corruption (§ host-sync) ── + # pnpm node_modules must NEVER be a host bind-mount: shared-filesystem + # sync (Docker Desktop gRPC-FUSE/VirtioFS, Unison, Syncthing) corrupts + # pnpm's hardlinked store and .bin shims. The production image already + # bakes node_modules in at build time and is NOT bind-mounted here. + # For local dev use a *named volume* instead of a host bind: + # - node_modules:/app/node_modules + # and never share `node_modules` / `pnpm store` via the Docker bind. + # On macOS add `:cached`/`:delegated` consistency labels per mount. healthcheck: test: ["CMD", "node", "-e", "fetch('http://localhost:3002/api/health').then(r=>{if(!r.ok)process.exit(1)}).catch(()=>process.exit(1))"] @@ -78,4 +88,90 @@ services: # - "3030:3030" # restart: unless-stopped # volumes: - # - /var/www/Gamedata:/var/www/Gamedata \ No newline at end of file + # - /var/www/Gamedata:/var/www/Gamedata + + # ── MariaDB "Turbo" (heavy JSON / bulk-loads) ── + # Dedicated MariaDB tuned for >50 MB JSON imports. All tuning lives inline + # in the service `command:` (bulk_insert_buffer_size, + # innodb_flush_log_at_trx_commit=2, max_allowed_packet=512M, ...) so the + # container configures itself — no external .cnf to mount or keep in sync. + # Opt-in via profile so `docker compose up` keeps running the standalone + # stack as today. + # + # Start: docker compose --profile db up -d (= pnpm db:up) + # Note: host networking binds 127.0.0.1:3306 → STOP the host MariaDB + # first (the app's .env DATABASE_URL points at localhost:3306). + # Fixes the "Pulling schema from database..." hang: raise + # net_read/net_write_timeout (done above) + stop imports while + # running `pnpm db:generate` / drizzle-kit introspection. + mariadb-turbo: + image: mariadb:11 + container_name: mariadb-turbo + profiles: ["db"] + network_mode: host + restart: unless-stopped + environment: + # mariadb:11 uses MARIADB_*; MYSQL_* also works but is deprecated there. + - MARIADB_ROOT_PASSWORD=${MARIADB_ROOT_PASSWORD:-root} + - MARIADB_DATABASE=${MARIADB_DATABASE:-habbo} + - MARIADB_USER=${MARIADB_USER:-cms} + - MARIADB_PASSWORD=${MARIADB_PASSWORD:-cms} + - MARIADB_AUTO_UPGRADE=1 + # MariaDB tunes itself — the Official image appends these flags to mysqld, + # no external .cnf file needed. + command: [ + # Charset + "--character-set-server=utf8mb4", + "--collation-server=utf8mb4_unicode_ci", + # 50 MB JSON batches: raise the 16 MB default packet ceiling. + "--max-allowed-packet=512M", + "--net-buffer-length=1M", + # Timeouts — fixes the "Pulling schema from database..." hang + # (drizzle-kit introspection no longer starves behind big imports). + "--connect-timeout=30", + "--wait-timeout=3600", + "--interactive-timeout=3600", + "--net-read-timeout=600", + "--net-write-timeout=600", + # InnoDB: durability/performance trade-off for bulk writes. + "--innodb-flush-log-at-trx-commit=2", + "--innodb-buffer-pool-size=2G", + "--innodb-buffer-pool-instances=4", + "--innodb-log-file-size=1G", + "--innodb-log-buffer-size=64M", + "--innodb-flush-method=O_DIRECT", + "--innodb-autoextend-increment=512", + "--innodb-max-dirty-pages-pct=90", + "--bulk-insert-buffer-size=512M", + # Raise so the engine nearly never double-writes during a 50 MB load. + "--innodb-doublewrite=0", + # libaio/native_aio misbehaves in some containers; io_uring path is fine. + "--innodb-use-native-aio=0", + # Temp tables used when MariaDB scans JSON blobs cannot be indexed. + "--tmp-table-size=256M", + "--max-heap-table-size=256M", + "--read-buffer-size=4M", + "--read-rnd-buffer-size=16M", + # Save ~1-2 GB RAM; bulk loads don't need the instrumentation. + "--performance-schema=OFF", + "--skip-name-resolve", + ] + volumes: + # Named volume, NOT a host bind — shared-filesystem sync trashes InnoDB + # files the same way it trashes pnpm's node_modules. Named volumes live + # inside the container filesystem, so 50 MB of JSON writes never cross a + # host-sync boundary. + - mariadb-turbo-data:/var/lib/mysql + healthcheck: + # healthcheck.sh ships in the official mariadb image. + test: ["CMD-SHELL", "healthcheck.sh --connect --innodb_initialized || mariadb-admin ping --silent"] + interval: 10s + timeout: 5s + retries: 5 + start_period: 30s + mem_limit: 4g + +# Named volumes declared once; used by the MariaDB service above. +volumes: + mariadb-turbo-data: + driver: local \ No newline at end of file diff --git a/package.json b/package.json index da5aac77..5945e733 100644 --- a/package.json +++ b/package.json @@ -21,6 +21,11 @@ "test:e2e": "playwright test", "typecheck": "tsc --noEmit", "db:generate": "drizzle-kit generate", + "db:introspect": "drizzle-kit introspect", + "db:bulk": "tsx scripts/bulk-import-json.ts", + "db:up": "docker compose --profile db up -d mariadb-turbo", + "db:down": "docker compose --profile db stop mariadb-turbo", + "db:schema:generate": "node scripts/generate-drizzle-schema.mjs", "db:migrate": "tsx scripts/apply-migrations.ts", "db:migrate:status": "tsx scripts/apply-migrations.ts --status", "db:studio": "drizzle-kit studio", diff --git a/scripts/bulk-import-json.ts b/scripts/bulk-import-json.ts new file mode 100644 index 00000000..eb59b343 --- /dev/null +++ b/scripts/bulk-import-json.ts @@ -0,0 +1,254 @@ +// Bulk-load >50 MB JSON into MariaDB in resumable chunks (2000 rows/batch). +// +// Drizzle multi-row INSERTs are wrapped in ONE statement per chunk, so a +// 50 MB dump becomes a few dozen statements instead of hundreds of thousands +// of round-trips — no connect/packet timeouts, no "Pulling schema from +// database..." freeze (that hang is introspection racing a saturated server). +// +// Usage (via pnpm, as required): +// pnpm exec tsx scripts/bulk-import-json.ts \ +// --file=/var/www/Gamedata/.../FurnitureData.json [--chunk-size=2000] +// +// Input shapes supported automatically: +// A) habbo furnidata_json → { roomitemtypes:{furnitype:[…]}, wallitemtypes:{…} } +// B) flat array of objects → [ {…}, {…} ] +// +// Flags: +// --file= JSON file to load (required) +// --table= one of: furnidata | docs | texts (default: furnidata) +// --category= category value for docs/texts tables +// --source= source value for furnidata table (default: habbo) +// --chunk-size=N rows per multi-row INSERT (default: 2000) +// --limit=N stop after N rows (dry-test without loading everything) +// --truncate DELETE all rows for this table's source/category first +// +// Idempotent by default: rows re-import on their unique key with +// ON DUPLICATE KEY UPDATE, so interrupted runs resume safely. + +import "./load-env"; +import { createHash } from "node:crypto"; +import { resolve } from "node:path"; +import { sql } from "drizzle-orm"; +import { drizzle } from "drizzle-orm/mysql2"; +import mysql from "mysql2/promise"; +import { + GamedataDocs, + GamedataFurnidata, + GamedataTexts, +} from "@/db/schema-gamedata"; +import { mysqlConnectionUrl } from "./db-url"; + +interface Args { + file: string; + table: "furnidata" | "docs" | "texts"; + category: string; + source: string; + chunkSize: number; + limit: number; + truncate: boolean; +} + +const FURNIDATA_SECTIONS = { + roomitemtypes: "s", + flooritemtypes: "s", + wallitemtypes: "i", + effecttypes: "e", +} as const; + +function parseArgs(raw: string[]): Args { + const get = (name: string) => { + const hit = raw.find((a) => a.startsWith(`--${name}=`)); + return hit?.slice(name.length + 3); + }; + const has = (name: string) => raw.includes(`--${name}`); + const file = get("file"); + if (!file) { + console.error( + "Usage: pnpm exec tsx scripts/bulk-import-json.ts --file= [--table=furnidata|docs|texts] [--category=] [--source=] [--chunk-size=2000] [--limit=N] [--truncate]", + ); + process.exit(1); + } + const table = (get("table") ?? "furnidata") as Args["table"]; + return { + file: resolve(file), + table, + category: get("category") ?? "default", + source: get("source") ?? "habbo", + chunkSize: Number(get("chunk-size") ?? "2000"), + limit: Number(get("limit") ?? "0"), + truncate: has("truncate"), + }; +} + +// A >50 MB file parses fine with a single JSON.parse (V8 string limit is far +// higher); streaming row-by-row would slow the seed down for no memory win. +async function readJsonFile(file: string): Promise { + const { readFile } = await import("node:fs/promises"); + return JSON.parse(await readFile(file, "utf8")); +} + +interface NormalizedRow { + [key: string]: unknown; +} + +// Normalize any supported shape into a flat list of JSON documents. +function normalize(json: unknown, table: Args["table"]): NormalizedRow[] { + if (table !== "furnidata") { + if (Array.isArray(json)) return json as NormalizedRow[]; + throw new Error(`Expected a top-level array for --table=${table}`); + } + if (Array.isArray(json)) return json as NormalizedRow[]; + + // habbo furnidata_json: { roomitemtypes: { furnitype: [...] }, ... } + if (json && typeof json === "object") { + const rows: NormalizedRow[] = []; + for (const [section, kind] of Object.entries(FURNIDATA_SECTIONS)) { + const block = (json as Record)[section]; + if (!block || typeof block !== "object") continue; + const list = (block as Record).furnitype; + if (!Array.isArray(list)) continue; + for (const item of list) rows.push({ ...item, kind }); + } + return rows; + } + throw new Error("Unsupported JSON shape: expected array or furnidata_json"); +} + +function toFurnidataRow( + row: NormalizedRow, + source: string, +): Record { + return { + source, + kind: (row.kind as "s" | "i" | "e") ?? "s", + className: String(row.classname ?? row.className ?? ""), + publicName: String(row.public_name ?? row.publicName ?? ""), + spriteId: Number(row.id ?? 0), + payload: JSON.stringify(row), + }; +} + +function toDocsRow( + row: NormalizedRow, + category: string, +): Record { + const payload = JSON.stringify(row); + return { + category, + docKey: String(row.key ?? row.docKey ?? row.id ?? ""), + payload, + payloadSha1: createHash("sha1").update(payload).digest("hex"), + }; +} + +function toTextsRow( + row: NormalizedRow, + category: string, +): Record { + return { + category, + textKey: String(row.key ?? row.id ?? ""), + value: String(row.value ?? row.text ?? row), + }; +} + +async function main() { + const args = parseArgs(process.argv.slice(2)); + const url = process.env.DATABASE_URL; + if (!url) + throw new Error("DATABASE_URL is required (load-env.ts loaded .env)"); + + const conn = await mysql.createConnection(mysqlConnectionUrl(url)); + const db = drizzle(conn); + + // Session bulk-load flags: worst case per chunk is a lost commit, not a + // corrupt table. FOREIGN_KEY_CHECKS off keeps each 50 MB chunk off FK + // validation work; UNIQUE_CHECK is a per-statement no-op vs ON DUPLICATE. + await conn.query("SET SESSION FOREIGN_KEY_CHECKS=0"); + await conn.query("SET SESSION sql_mode='NO_ENGINE_SUBSTITUTION'"); + + const tableRef = + args.table === "furnidata" + ? GamedataFurnidata + : args.table === "docs" + ? GamedataDocs + : GamedataTexts; + const keyWhere = + args.table === "furnidata" + ? sql`${GamedataFurnidata.source} = ${args.source}` + : args.table === "docs" + ? sql`${GamedataDocs.category} = ${args.category}` + : sql`${GamedataTexts.category} = ${args.category}`; + + if (args.truncate) { + await db.execute(sql`DELETE FROM ${tableRef} WHERE ${keyWhere}`); + console.log(`[bulk] truncated rows for ${args.table}`); + } + + console.log(`[bulk] reading ${args.file} …`); + const json = await readJsonFile(args.file); + const rows = normalize(json, args.table); + console.log(`[bulk] parsed ${rows.length} documents (${args.table})`); + + const mapper: (r: NormalizedRow) => Record = + args.table === "furnidata" + ? (r) => toFurnidataRow(r, args.source) + : args.table === "docs" + ? (r) => toDocsRow(r, args.category) + : (r) => toTextsRow(r, args.category); + + const values = rows + .slice(0, args.limit || rows.length) + .map>(mapper); + const total = values.length; + const chunkSize = args.chunkSize; + const started = Date.now(); + let inserted = 0; + + for (let i = 0; i < total; i += chunkSize) { + const chunk = values.slice(i, Math.min(i + chunkSize, total)); + // `tableRef` is a 3-way union; drizzle's insert is typed per single + // table, so narrow through a custom shape here (runtime is unaffected). + const insert = db.insert(tableRef) as unknown as { + values: (rows: typeof chunk) => { + onDuplicateKeyUpdate: (opts: { + set: Record; + }) => Promise; + }; + }; + await insert.values(chunk).onDuplicateKeyUpdate({ + // MariaDB `VALUES(col)` = the incoming row value; re-runs overwrite. + set: Object.fromEntries( + Object.keys(chunk[0] ?? {}).map((k) => [ + k, + sql.raw(`values(\`${k}\`)`), + ]), + ), + }); + + inserted += chunk.length; + const remaining = total - i - chunk.length; + if (inserted % (chunkSize * 10) < chunkSize || remaining <= 0) { + const elapsedSec = (Date.now() - started) / 1000; + const rate = Math.round(inserted / Math.max(elapsedSec, 0.001)); + const etaMin = Math.round(remaining / Math.max(rate, 1) / 60); + console.log( + `[bulk] ${inserted}/${total} (${Math.round((inserted / total) * 100)}%) ${rate} rows/s, ETA ~${etaMin}m`, + ); + } + } + + const sec = ((Date.now() - started) / 1000).toFixed(1); + console.log(`[bulk] DONE: ${inserted}/${total} rows in ${sec}s`); + if (total === args.limit && args.limit > 0) { + console.log( + " NOTE: --limit was set; run again without it for the full dump.", + ); + } + await conn.end(); +} + +main().catch((err) => { + console.error("[bulk] FATAL:", err instanceof Error ? err.message : err); + process.exit(1); +}); diff --git a/src/db/schema-gamedata.ts b/src/db/schema-gamedata.ts new file mode 100644 index 00000000..c9b64ed3 --- /dev/null +++ b/src/db/schema-gamedata.ts @@ -0,0 +1,106 @@ +// Gamedata / large-JSON storage schema for MariaDB. +// +// MariaDB's JSON type is an alias for LONGTEXT (+ a `CHECK (json_valid(...))` +// constraint), so it can NEVER be indexed directly. The "smart" pattern: +// 1. keep the raw JSON document in longtext / mediumtext — full fidelity, +// no validation overhead, no 4 GB fetch if only a key is needed; +// 2. promote the fields you actually WHERE/ORDER BY into real columns; +// 3. index those columns (optionally as VIRTUAL generated columns using +// JSON_EXTRACT so you don't duplicate data in the seed script). +// +// Not part of drizzle's auto-generated src/db/schema.ts (which is regenerated +// from the live emulator DB) — this file is a standalone schema module you +// pass to `drizzle(conn, { schema })` in bulk scripts or import directly. + +import { sql } from "drizzle-orm"; +import { + index, + int, + longtext, + mediumtext, + mysqlEnum, + mysqlTable, + uniqueIndex, + varchar, +} from "drizzle-orm/mysql-core"; + +// ── Furnidata: one row per furniture/clothing item ─────────────────────────── +// Compatible with the habbo furnidata_json shape: +// { "roomitemtypes": { "furnitype": [ { id, classname, public_name, ... } ] }, +// "wallitemtypes": { "furnitype": [ ... ] }, +// "effecttypes": { "effecttype": [ ... ] } } +export const GamedataFurnidata = mysqlTable( + "gamedata_furnidata", + { + id: int("id", { unsigned: true }).autoincrement().primaryKey().notNull(), + source: varchar("source", { length: 64 }).notNull().default("habbo"), + spriteId: int("sprite_id", { unsigned: true }).notNull(), + // "s" = floor/room item, "i" = wall item, "e" = clothing/effect. + kind: mysqlEnum("kind", ["s", "i", "e"]).notNull().default("s"), + className: varchar("class_name", { length: 128 }).notNull(), + // Promoted from JSON so hot lookups & LIST/WHERE never touch the blob. + publicName: varchar("public_name", { length: 128 }).notNull().default(""), + // Raw JSON document, kept intact for exact round-trips (catalog export). + payload: longtext("payload").notNull(), + }, + (t) => [ + // Idempotent seeds upsert on this key (source, sprite_id). + uniqueIndex("gamedata_furnidata_source_sprite").on(t.source, t.spriteId), + index("gamedata_furnidata_class").on(t.className), + index("gamedata_furnidata_kind").on(t.kind), + ], +); + +// ── Generic JSON document store ─────────────────────────────────────────────── +// Use for any >50 MB JSON dump split into per-key documents (external_texts, +// product data, achievements, etc.). `title` is a VIRTUAL generated column +// extracted from the payload via JSON_EXTRACT and indexed — MariaDB supports +// secondary indexes on virtual columns (10.2+), so filtering by a payload +// field stays fast without storing a duplicate copy. +export const GamedataDocs = mysqlTable( + "gamedata_docs", + { + id: int("id", { unsigned: true }).autoincrement().primaryKey().notNull(), + category: varchar("category", { length: 64 }).notNull(), + docKey: varchar("doc_key", { length: 128 }).notNull(), + payload: longtext("payload").notNull(), + // SHA-1 of the raw payload → cheap change detection on re-runs. + payloadSha1: varchar("payload_sha1", { length: 40 }).notNull().default(""), + // VIRTUAL (default) column; the JSON_EXTRACT runs at query/index time. + // NOT NULL forces the extract to resolve — use .notNull() only when + // every document guarantees the field. + title: varchar("title", { length: 255 }) + .generatedAlwaysAs( + sql`CAST(JSON_UNQUOTE(JSON_EXTRACT(payload, '$.title')) AS CHAR(255))`, + { mode: "virtual" }, + ) + .notNull(), + }, + (t) => [ + uniqueIndex("gamedata_docs_cat_key").on(t.category, t.docKey), + index("gamedata_docs_title").on(t.title), + ], +); + +// ── Key/value texts (external_texts style, split by category) ─────────────── +// One row per string; bulk-loads stay resumable via (category, text_key). +export const GamedataTexts = mysqlTable( + "gamedata_texts", + { + id: int("id", { unsigned: true }).autoincrement().primaryKey().notNull(), + category: varchar("category", { length: 64 }).notNull(), + textKey: varchar("text_key", { length: 255 }).notNull(), + value: mediumtext("value").notNull(), + }, + (t) => [ + uniqueIndex("gamedata_texts_cat_key").on(t.category, t.textKey), + index("gamedata_texts_key").on(t.textKey), + ], +); + +export type GamedataFurnidataRow = typeof GamedataFurnidata.$inferSelect; +export type GamedataFurnidataInsert = typeof GamedataFurnidata.$inferInsert; +export type GamedataDocsRow = typeof GamedataDocs.$inferSelect; +export type GamedataDocsInsert = typeof GamedataDocs.$inferInsert; +export type GamedataTextsRow = typeof GamedataTexts.$inferSelect; +export type GamedataTextsInsert = typeof GamedataTexts.$inferInsert;