feat(server): port the UO models, utils and schema fragment

The data half of the extraction: 8 model directories, 13 utils, the shard
stream catalog and the 27-table schema fragment with its purge.

server/core.js is what makes the port a one-line import change per file rather
than a signature change per function. Ported code requires its dependencies at
file scope -- `const { query } = require('../../core')` -- which runs before
register() has been called and before any ctx exists. So every member is a
stable function that resolves ctx when CALLED, and nothing may be destructured
off ctx at init either, because core is free to hand over a getter.

Two helpers are vendored rather than taken from ctx, and the line between them
is the point. utils/excerpt.js is core's deriveExcerpt -- nine lines of pure
text handling. Core's sanitiser next to it was NOT copied: a second copy of a
security control diverges silently the moment either is fixed. announceLinks.js
vendors legError and articleUrl the same way, but baseUrl could not be: core's
reads APP_BASE_URL, and §2.7 forbids a module reading core's environment, so it
comes off ctx.site.baseUrl.

The schema fragment is core's 27 shard_*/uo_link_* statements, verbs CREATE,
ALTER and UPDATE only, every CREATE TABLE guarded. Two of its tables carry a
foreign key INTO users, which is allowed and is why the replay order matters --
core's schema is in place before this runs. The reverse never occurs and must
not: it would make core unable to boot without a module installed.

One real port bug caught by the integration run, not by tests: the atlas art
map resolved `../../../db/data`, which pointed at core's tree when this file
lived there and points outside server/ now. A path that happens to resolve is
exactly what survives a green suite, because the absent-file branch returns {}
and looks like the normal case.

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
2026-08-11 12:06:26 -05:00
committed by Claude
parent 47809854ef
commit fe3251a543
40 changed files with 7967 additions and 3 deletions

287
server/utils/clilocParse.js Normal file
View File

@@ -0,0 +1,287 @@
// Cliloc parsing — the pure half.
//
// A "cliloc" is UO's localization table: an integer id mapped to a display
// string. Items carry a `LabelNumber` rather than a name, so without this table
// the site can only render `id 1023721` where the game shows "quarter staff".
// The shard already sends the id on every equipment entry (`char.profile`'s
// `cliloc` field) and will send one per marketplace listing — the *number* was
// never the missing piece, the *table* was.
//
// This module is fs-free on purpose, exactly like `spawnAtlasParse.js`: the
// suite runs in CI where there is no UO client, so every parser here is driven
// from inline fixtures. `clilocSource.js` is the only thing that touches disk.
//
// ── Two input formats, and why ─────────────────────────────────────────────
//
// The client's own `Cliloc.enu` is COMPRESSED (Mythic format) on any modern
// client, and decompressing it is a bit-level port of an inverse-BWT coder that
// nothing in this stack needs at runtime. ServUO's own bundled `Ultima.StringList`
// cannot read it either — which is why `VendorSearch.GetItemName` is already inert
// on such a shard and the plugin could not supply names even if we asked it to.
//
// So the operator converts once, from their own client, and points the site at
// the result (see docs/website/CLILOCS.md). Two shapes are accepted because
// different tools produce different things:
//
// • PLAIN BINARY — the pre-compression cliloc layout: a 6-byte header, then
// records of {int32 number, byte flag, uint16 length, UTF-8 bytes}.
// • DELIMITED TEXT — `number<TAB|,|;>text` per line, which is what the common
// GUI exports emit. Quoted CSV fields and a header row are tolerated.
//
// Nothing derived from the client is ever committed: the converted file lives at
// an operator-supplied path and is gitignored, the same rule the spawn atlas art
// map already follows.
/** Raised for a file we can identify but deliberately refuse to guess at. */
class ClilocFormatError extends Error {
constructor(message, code) {
super(message)
this.name = 'ClilocFormatError'
this.code = code
}
}
/**
* Bumped when this parser produces DIFFERENT data from an IDENTICAL source file.
*
* Stored beside the source hash so the boot path can tell "same file, but the
* parser moved on" from "same file, nothing to do". Without it a corrected parse
* would ship and never reach an install whose cliloc file never changes — the
* trap `spawnAtlasSource.PARSER_VERSION` documents.
*/
const PARSER_VERSION = 1
// The plain layout's header is `02 00 00 00 01 00` — a 4-byte version and a
// 2-byte language marker. Only the size matters for parsing; the values are
// checked to sniff the format, not to validate it.
const HEADER_BYTES = 6
const RECORD_HEADER_BYTES = 7 // int32 number + byte flag + uint16 length
// Every compressed cliloc file the client ships begins with a DWORD whose high
// byte is 0x8E (the XOR key UOFiddler calls `HeaderXorKey`, 0x8E2C9A3D). That is
// the single cheapest way to tell an operator they exported the wrong file —
// without it, the plain parser happily reads compressed bytes as ~19k records of
// negative ids and 60 KB "strings" before dying somewhere in the middle, and the
// resulting error names the wrong problem.
const MYTHIC_HIGH_BYTE = 0x8e
/** True when `buffer` is a Mythic-compressed cliloc rather than the plain layout. */
function isCompressedCliloc(buffer) {
return buffer.length >= 4 && buffer[3] === MYTHIC_HIGH_BYTE
}
/**
* Parse the plain binary cliloc layout.
*
* Strict about truncation, and that strictness is load-bearing: a half-copied or
* partly-written file is the realistic failure here, and it must fail loudly
* rather than import a silently short table that then renders half the world as
* `id 1023721`. A record that runs past the end of the buffer throws.
*/
function parseClilocBinary(buffer) {
if (!Buffer.isBuffer(buffer)) throw new ClilocFormatError('Not a buffer', 'NOT_BUFFER')
if (isCompressedCliloc(buffer)) {
throw new ClilocFormatError(
'This is a compressed (Mythic-format) cliloc file, which the site cannot read. ' +
'Convert it to the plain format first — see docs/website/CLILOCS.md.',
'COMPRESSED',
)
}
if (buffer.length < HEADER_BYTES) {
throw new ClilocFormatError('File is shorter than a cliloc header', 'TRUNCATED')
}
const entries = []
let offset = HEADER_BYTES
while (offset < buffer.length) {
if (offset + RECORD_HEADER_BYTES > buffer.length) {
throw new ClilocFormatError(
`Truncated record header at byte ${offset} (${entries.length} entries read)`,
'TRUNCATED',
)
}
const number = buffer.readInt32LE(offset)
const flag = buffer.readUInt8(offset + 4)
// The length is written by the client as an unsigned 16-bit value. Reading it
// signed (as ServUO's own SDK does) turns any string over 32 KB into a
// negative length; real tables top out around 12 KB, so this has no effect on
// current data and costs nothing to get right.
const length = buffer.readUInt16LE(offset + 5)
offset += RECORD_HEADER_BYTES
if (offset + length > buffer.length) {
throw new ClilocFormatError(
`Truncated record body at byte ${offset} (${entries.length} entries read)`,
'TRUNCATED',
)
}
entries.push({ number, flag, text: buffer.toString('utf8', offset, offset + length) })
offset += length
}
return entries
}
// A delimited line splits on the FIRST separator only: cliloc text is full of
// commas ("a scroll of magery, unfinished") and splitting on all of them would
// truncate every such entry at its first comma.
const TEXT_SEPARATORS = ['\t', ',', ';']
/** Unwrap one CSV field: strip surrounding quotes and unescape doubled quotes. */
function unquote(value) {
const trimmed = value.trim()
if (trimmed.length >= 2 && trimmed.startsWith('"') && trimmed.endsWith('"')) {
return trimmed.slice(1, -1).replace(/""/g, '"')
}
return trimmed
}
/**
* Parse a delimited text export: `number<sep>text` per line.
*
* Tolerant by design — this is whatever an operator's GUI tool produced, not a
* format we control. A header row, blank lines, `#` comments and a trailing
* flags column are all ignored. A line whose first field is not an integer is
* skipped rather than fatal, because that is exactly what a header row is.
*
* The one thing it will NOT do is return an empty table quietly: a file that
* yields no entries at all is a wrong file, not an empty one.
*/
function parseClilocText(text) {
const entries = []
for (const line of String(text).split(/\r?\n/)) {
// The line is deliberately NOT trimmed before the separator search. Roughly
// half of a real cliloc table is empty strings (unused ids), which export as
// `1005008<TAB>` — and trimming eats that trailing separator, leaving a bare
// number that then looks like a header row and is skipped. That silently
// dropped 55,994 of 123,490 entries. Individual FIELDS are trimmed instead,
// by `unquote`.
if (line.trim() === '' || line.trimStart().startsWith('#')) continue
// Pick the separator that actually appears first, so a tab-delimited line
// whose text contains a comma still splits on the tab.
let cut = -1
for (const sep of TEXT_SEPARATORS) {
const at = line.indexOf(sep)
if (at !== -1 && (cut === -1 || at < cut)) cut = at
}
if (cut === -1) continue
// An EMPTY first field must not become id 0: `Number('')` is 0, not NaN, so
// a line that merely starts with a separator would otherwise import as a
// bogus cliloc 0 instead of being skipped.
const head = unquote(line.slice(0, cut))
if (head === '') continue
const number = Number(head)
if (!Number.isInteger(number)) continue // header row, or a wrapped line
let rest = line.slice(cut + 1)
// Some exports carry `number,flag,text`. A bare integer in the second field
// is a flag; anything else is the text itself (and a text field that IS just
// a number is indistinguishable, so it stays as the text — the safer miss).
let flag = 0
for (const sep of TEXT_SEPARATORS) {
const at = rest.indexOf(sep)
if (at === -1) continue
const head = unquote(rest.slice(0, at))
if (/^\d{1,3}$/.test(head) && rest.slice(at + 1).trim() !== '') {
flag = Number(head)
rest = rest.slice(at + 1)
}
break
}
entries.push({ number, flag, text: unquote(rest) })
}
if (entries.length === 0) {
throw new ClilocFormatError('No cliloc entries found in the text export', 'EMPTY')
}
return entries
}
/**
* Parse either supported shape, sniffing which one this is.
*
* The sniff is on the binary header rather than the file extension: operators
* name these things whatever they like, and an `.enu` that is really a TSV (or a
* `.txt` that is really binary) should still import.
*/
function parseCliloc(buffer) {
const buf = Buffer.isBuffer(buffer) ? buffer : Buffer.from(buffer)
if (isCompressedCliloc(buf)) {
throw new ClilocFormatError(
'This is a compressed (Mythic-format) cliloc file, which the site cannot read. ' +
'Convert it to the plain format first — see docs/website/CLILOCS.md.',
'COMPRESSED',
)
}
// The plain layout always opens with version 2 / language 1. Anything else is
// treated as text, which is the recoverable guess: a mis-sniffed text file
// yields "no entries found", while a mis-sniffed binary yields nonsense.
if (buf.length >= HEADER_BYTES && buf.readInt32LE(0) === 2 && buf.readUInt16LE(4) === 1) {
return parseClilocBinary(buf)
}
return parseClilocText(buf.toString('utf8'))
}
// ── Display ────────────────────────────────────────────────────────────────
// Cliloc strings interpolate arguments the client supplies out of an item's
// property list: `~1_val~`, `~2_NAME~`, `~1_ITEM~`. We never have those — the
// bridge sends the id, not the packet — so a name carrying them must be reduced
// to what is actually knowable rather than shown with the raw tokens in it.
const PLACEHOLDER_RE = /~\d+_[^~]*~/g
/**
* Reduce a raw cliloc string to something displayable.
*
* Placeholders are dropped and the leftover punctuation tidied, so
* `"[~1_stuff~]"` becomes `""` (correctly nothing — the whole string was the
* argument) and `"cold damage ~1_val~%"` becomes `"cold damage"`.
*
* **Punctuation is only tidied when a placeholder was actually removed.** The
* trailing `%` above is the unit belonging to the number we never had, and the
* brackets in `[~1_stuff~]` only ever wrapped the argument — but a string with
* no placeholder has no such debris, and trimming it anyway corrupts real names.
* A shard's `"Runic Gateway Sigil (v2)"` came back as `"(v2"` while this was
* unconditional.
*
* Returns `''` when nothing survives, which callers treat as "no name" and fall
* back to the item id — better than showing a bracket.
*/
const DEBRIS = /^[\s\-–—,.;:%[\]()]+|[\s\-–—,.;:%[\]()]+$/g
function displayText(raw) {
if (raw == null) return ''
const source = String(raw)
const hadPlaceholder = PLACEHOLDER_RE.test(source)
PLACEHOLDER_RE.lastIndex = 0 // the regex is global; `test` advances it
if (!hadPlaceholder) return source.replace(/\s+/g, ' ').trim()
return source
.replace(PLACEHOLDER_RE, ' ')
.replace(/\s+/g, ' ')
.replace(/\s+([,.;:!?])/g, '$1')
.replace(DEBRIS, '')
.trim()
}
/** True when a raw cliloc string is nothing but interpolated arguments. */
const isPlaceholderOnly = (raw) => raw != null && String(raw).trim() !== '' && displayText(raw) === ''
module.exports = {
ClilocFormatError,
PARSER_VERSION,
HEADER_BYTES,
isCompressedCliloc,
parseCliloc,
parseClilocBinary,
parseClilocText,
displayText,
isPlaceholderOnly,
}