The data half of the extraction: 8 model directories, 13 utils, the shard
stream catalog and the 27-table schema fragment with its purge.
server/core.js is what makes the port a one-line import change per file rather
than a signature change per function. Ported code requires its dependencies at
file scope -- `const { query } = require('../../core')` -- which runs before
register() has been called and before any ctx exists. So every member is a
stable function that resolves ctx when CALLED, and nothing may be destructured
off ctx at init either, because core is free to hand over a getter.
Two helpers are vendored rather than taken from ctx, and the line between them
is the point. utils/excerpt.js is core's deriveExcerpt -- nine lines of pure
text handling. Core's sanitiser next to it was NOT copied: a second copy of a
security control diverges silently the moment either is fixed. announceLinks.js
vendors legError and articleUrl the same way, but baseUrl could not be: core's
reads APP_BASE_URL, and §2.7 forbids a module reading core's environment, so it
comes off ctx.site.baseUrl.
The schema fragment is core's 27 shard_*/uo_link_* statements, verbs CREATE,
ALTER and UPDATE only, every CREATE TABLE guarded. Two of its tables carry a
foreign key INTO users, which is allowed and is why the replay order matters --
core's schema is in place before this runs. The reverse never occurs and must
not: it would make core unable to boot without a module installed.
One real port bug caught by the integration run, not by tests: the atlas art
map resolved `../../../db/data`, which pointed at core's tree when this file
lived there and points outside server/ now. A path that happens to resolve is
exactly what survives a green suite, because the absent-file branch returns {}
and looks like the normal case.
Co-Authored-By: Claude <noreply@anthropic.com>
288 lines
12 KiB
JavaScript
288 lines
12 KiB
JavaScript
// Cliloc parsing — the pure half.
|
|
//
|
|
// A "cliloc" is UO's localization table: an integer id mapped to a display
|
|
// string. Items carry a `LabelNumber` rather than a name, so without this table
|
|
// the site can only render `id 1023721` where the game shows "quarter staff".
|
|
// The shard already sends the id on every equipment entry (`char.profile`'s
|
|
// `cliloc` field) and will send one per marketplace listing — the *number* was
|
|
// never the missing piece, the *table* was.
|
|
//
|
|
// This module is fs-free on purpose, exactly like `spawnAtlasParse.js`: the
|
|
// suite runs in CI where there is no UO client, so every parser here is driven
|
|
// from inline fixtures. `clilocSource.js` is the only thing that touches disk.
|
|
//
|
|
// ── Two input formats, and why ─────────────────────────────────────────────
|
|
//
|
|
// The client's own `Cliloc.enu` is COMPRESSED (Mythic format) on any modern
|
|
// client, and decompressing it is a bit-level port of an inverse-BWT coder that
|
|
// nothing in this stack needs at runtime. ServUO's own bundled `Ultima.StringList`
|
|
// cannot read it either — which is why `VendorSearch.GetItemName` is already inert
|
|
// on such a shard and the plugin could not supply names even if we asked it to.
|
|
//
|
|
// So the operator converts once, from their own client, and points the site at
|
|
// the result (see docs/website/CLILOCS.md). Two shapes are accepted because
|
|
// different tools produce different things:
|
|
//
|
|
// • PLAIN BINARY — the pre-compression cliloc layout: a 6-byte header, then
|
|
// records of {int32 number, byte flag, uint16 length, UTF-8 bytes}.
|
|
// • DELIMITED TEXT — `number<TAB|,|;>text` per line, which is what the common
|
|
// GUI exports emit. Quoted CSV fields and a header row are tolerated.
|
|
//
|
|
// Nothing derived from the client is ever committed: the converted file lives at
|
|
// an operator-supplied path and is gitignored, the same rule the spawn atlas art
|
|
// map already follows.
|
|
|
|
/** Raised for a file we can identify but deliberately refuse to guess at. */
|
|
class ClilocFormatError extends Error {
|
|
constructor(message, code) {
|
|
super(message)
|
|
this.name = 'ClilocFormatError'
|
|
this.code = code
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Bumped when this parser produces DIFFERENT data from an IDENTICAL source file.
|
|
*
|
|
* Stored beside the source hash so the boot path can tell "same file, but the
|
|
* parser moved on" from "same file, nothing to do". Without it a corrected parse
|
|
* would ship and never reach an install whose cliloc file never changes — the
|
|
* trap `spawnAtlasSource.PARSER_VERSION` documents.
|
|
*/
|
|
const PARSER_VERSION = 1
|
|
|
|
// The plain layout's header is `02 00 00 00 01 00` — a 4-byte version and a
|
|
// 2-byte language marker. Only the size matters for parsing; the values are
|
|
// checked to sniff the format, not to validate it.
|
|
const HEADER_BYTES = 6
|
|
const RECORD_HEADER_BYTES = 7 // int32 number + byte flag + uint16 length
|
|
|
|
// Every compressed cliloc file the client ships begins with a DWORD whose high
|
|
// byte is 0x8E (the XOR key UOFiddler calls `HeaderXorKey`, 0x8E2C9A3D). That is
|
|
// the single cheapest way to tell an operator they exported the wrong file —
|
|
// without it, the plain parser happily reads compressed bytes as ~19k records of
|
|
// negative ids and 60 KB "strings" before dying somewhere in the middle, and the
|
|
// resulting error names the wrong problem.
|
|
const MYTHIC_HIGH_BYTE = 0x8e
|
|
|
|
/** True when `buffer` is a Mythic-compressed cliloc rather than the plain layout. */
|
|
function isCompressedCliloc(buffer) {
|
|
return buffer.length >= 4 && buffer[3] === MYTHIC_HIGH_BYTE
|
|
}
|
|
|
|
/**
|
|
* Parse the plain binary cliloc layout.
|
|
*
|
|
* Strict about truncation, and that strictness is load-bearing: a half-copied or
|
|
* partly-written file is the realistic failure here, and it must fail loudly
|
|
* rather than import a silently short table that then renders half the world as
|
|
* `id 1023721`. A record that runs past the end of the buffer throws.
|
|
*/
|
|
function parseClilocBinary(buffer) {
|
|
if (!Buffer.isBuffer(buffer)) throw new ClilocFormatError('Not a buffer', 'NOT_BUFFER')
|
|
if (isCompressedCliloc(buffer)) {
|
|
throw new ClilocFormatError(
|
|
'This is a compressed (Mythic-format) cliloc file, which the site cannot read. ' +
|
|
'Convert it to the plain format first — see docs/website/CLILOCS.md.',
|
|
'COMPRESSED',
|
|
)
|
|
}
|
|
if (buffer.length < HEADER_BYTES) {
|
|
throw new ClilocFormatError('File is shorter than a cliloc header', 'TRUNCATED')
|
|
}
|
|
|
|
const entries = []
|
|
let offset = HEADER_BYTES
|
|
|
|
while (offset < buffer.length) {
|
|
if (offset + RECORD_HEADER_BYTES > buffer.length) {
|
|
throw new ClilocFormatError(
|
|
`Truncated record header at byte ${offset} (${entries.length} entries read)`,
|
|
'TRUNCATED',
|
|
)
|
|
}
|
|
const number = buffer.readInt32LE(offset)
|
|
const flag = buffer.readUInt8(offset + 4)
|
|
// The length is written by the client as an unsigned 16-bit value. Reading it
|
|
// signed (as ServUO's own SDK does) turns any string over 32 KB into a
|
|
// negative length; real tables top out around 12 KB, so this has no effect on
|
|
// current data and costs nothing to get right.
|
|
const length = buffer.readUInt16LE(offset + 5)
|
|
offset += RECORD_HEADER_BYTES
|
|
|
|
if (offset + length > buffer.length) {
|
|
throw new ClilocFormatError(
|
|
`Truncated record body at byte ${offset} (${entries.length} entries read)`,
|
|
'TRUNCATED',
|
|
)
|
|
}
|
|
entries.push({ number, flag, text: buffer.toString('utf8', offset, offset + length) })
|
|
offset += length
|
|
}
|
|
|
|
return entries
|
|
}
|
|
|
|
// A delimited line splits on the FIRST separator only: cliloc text is full of
|
|
// commas ("a scroll of magery, unfinished") and splitting on all of them would
|
|
// truncate every such entry at its first comma.
|
|
const TEXT_SEPARATORS = ['\t', ',', ';']
|
|
|
|
/** Unwrap one CSV field: strip surrounding quotes and unescape doubled quotes. */
|
|
function unquote(value) {
|
|
const trimmed = value.trim()
|
|
if (trimmed.length >= 2 && trimmed.startsWith('"') && trimmed.endsWith('"')) {
|
|
return trimmed.slice(1, -1).replace(/""/g, '"')
|
|
}
|
|
return trimmed
|
|
}
|
|
|
|
/**
|
|
* Parse a delimited text export: `number<sep>text` per line.
|
|
*
|
|
* Tolerant by design — this is whatever an operator's GUI tool produced, not a
|
|
* format we control. A header row, blank lines, `#` comments and a trailing
|
|
* flags column are all ignored. A line whose first field is not an integer is
|
|
* skipped rather than fatal, because that is exactly what a header row is.
|
|
*
|
|
* The one thing it will NOT do is return an empty table quietly: a file that
|
|
* yields no entries at all is a wrong file, not an empty one.
|
|
*/
|
|
function parseClilocText(text) {
|
|
const entries = []
|
|
for (const line of String(text).split(/\r?\n/)) {
|
|
// The line is deliberately NOT trimmed before the separator search. Roughly
|
|
// half of a real cliloc table is empty strings (unused ids), which export as
|
|
// `1005008<TAB>` — and trimming eats that trailing separator, leaving a bare
|
|
// number that then looks like a header row and is skipped. That silently
|
|
// dropped 55,994 of 123,490 entries. Individual FIELDS are trimmed instead,
|
|
// by `unquote`.
|
|
if (line.trim() === '' || line.trimStart().startsWith('#')) continue
|
|
|
|
// Pick the separator that actually appears first, so a tab-delimited line
|
|
// whose text contains a comma still splits on the tab.
|
|
let cut = -1
|
|
for (const sep of TEXT_SEPARATORS) {
|
|
const at = line.indexOf(sep)
|
|
if (at !== -1 && (cut === -1 || at < cut)) cut = at
|
|
}
|
|
if (cut === -1) continue
|
|
|
|
// An EMPTY first field must not become id 0: `Number('')` is 0, not NaN, so
|
|
// a line that merely starts with a separator would otherwise import as a
|
|
// bogus cliloc 0 instead of being skipped.
|
|
const head = unquote(line.slice(0, cut))
|
|
if (head === '') continue
|
|
const number = Number(head)
|
|
if (!Number.isInteger(number)) continue // header row, or a wrapped line
|
|
|
|
let rest = line.slice(cut + 1)
|
|
// Some exports carry `number,flag,text`. A bare integer in the second field
|
|
// is a flag; anything else is the text itself (and a text field that IS just
|
|
// a number is indistinguishable, so it stays as the text — the safer miss).
|
|
let flag = 0
|
|
for (const sep of TEXT_SEPARATORS) {
|
|
const at = rest.indexOf(sep)
|
|
if (at === -1) continue
|
|
const head = unquote(rest.slice(0, at))
|
|
if (/^\d{1,3}$/.test(head) && rest.slice(at + 1).trim() !== '') {
|
|
flag = Number(head)
|
|
rest = rest.slice(at + 1)
|
|
}
|
|
break
|
|
}
|
|
|
|
entries.push({ number, flag, text: unquote(rest) })
|
|
}
|
|
|
|
if (entries.length === 0) {
|
|
throw new ClilocFormatError('No cliloc entries found in the text export', 'EMPTY')
|
|
}
|
|
return entries
|
|
}
|
|
|
|
/**
|
|
* Parse either supported shape, sniffing which one this is.
|
|
*
|
|
* The sniff is on the binary header rather than the file extension: operators
|
|
* name these things whatever they like, and an `.enu` that is really a TSV (or a
|
|
* `.txt` that is really binary) should still import.
|
|
*/
|
|
function parseCliloc(buffer) {
|
|
const buf = Buffer.isBuffer(buffer) ? buffer : Buffer.from(buffer)
|
|
|
|
if (isCompressedCliloc(buf)) {
|
|
throw new ClilocFormatError(
|
|
'This is a compressed (Mythic-format) cliloc file, which the site cannot read. ' +
|
|
'Convert it to the plain format first — see docs/website/CLILOCS.md.',
|
|
'COMPRESSED',
|
|
)
|
|
}
|
|
|
|
// The plain layout always opens with version 2 / language 1. Anything else is
|
|
// treated as text, which is the recoverable guess: a mis-sniffed text file
|
|
// yields "no entries found", while a mis-sniffed binary yields nonsense.
|
|
if (buf.length >= HEADER_BYTES && buf.readInt32LE(0) === 2 && buf.readUInt16LE(4) === 1) {
|
|
return parseClilocBinary(buf)
|
|
}
|
|
return parseClilocText(buf.toString('utf8'))
|
|
}
|
|
|
|
// ── Display ────────────────────────────────────────────────────────────────
|
|
|
|
// Cliloc strings interpolate arguments the client supplies out of an item's
|
|
// property list: `~1_val~`, `~2_NAME~`, `~1_ITEM~`. We never have those — the
|
|
// bridge sends the id, not the packet — so a name carrying them must be reduced
|
|
// to what is actually knowable rather than shown with the raw tokens in it.
|
|
const PLACEHOLDER_RE = /~\d+_[^~]*~/g
|
|
|
|
/**
|
|
* Reduce a raw cliloc string to something displayable.
|
|
*
|
|
* Placeholders are dropped and the leftover punctuation tidied, so
|
|
* `"[~1_stuff~]"` becomes `""` (correctly nothing — the whole string was the
|
|
* argument) and `"cold damage ~1_val~%"` becomes `"cold damage"`.
|
|
*
|
|
* **Punctuation is only tidied when a placeholder was actually removed.** The
|
|
* trailing `%` above is the unit belonging to the number we never had, and the
|
|
* brackets in `[~1_stuff~]` only ever wrapped the argument — but a string with
|
|
* no placeholder has no such debris, and trimming it anyway corrupts real names.
|
|
* A shard's `"Runic Gateway Sigil (v2)"` came back as `"(v2"` while this was
|
|
* unconditional.
|
|
*
|
|
* Returns `''` when nothing survives, which callers treat as "no name" and fall
|
|
* back to the item id — better than showing a bracket.
|
|
*/
|
|
const DEBRIS = /^[\s\-–—,.;:%[\]()]+|[\s\-–—,.;:%[\]()]+$/g
|
|
|
|
function displayText(raw) {
|
|
if (raw == null) return ''
|
|
const source = String(raw)
|
|
const hadPlaceholder = PLACEHOLDER_RE.test(source)
|
|
PLACEHOLDER_RE.lastIndex = 0 // the regex is global; `test` advances it
|
|
|
|
if (!hadPlaceholder) return source.replace(/\s+/g, ' ').trim()
|
|
|
|
return source
|
|
.replace(PLACEHOLDER_RE, ' ')
|
|
.replace(/\s+/g, ' ')
|
|
.replace(/\s+([,.;:!?])/g, '$1')
|
|
.replace(DEBRIS, '')
|
|
.trim()
|
|
}
|
|
|
|
/** True when a raw cliloc string is nothing but interpolated arguments. */
|
|
const isPlaceholderOnly = (raw) => raw != null && String(raw).trim() !== '' && displayText(raw) === ''
|
|
|
|
module.exports = {
|
|
ClilocFormatError,
|
|
PARSER_VERSION,
|
|
HEADER_BYTES,
|
|
isCompressedCliloc,
|
|
parseCliloc,
|
|
parseClilocBinary,
|
|
parseClilocText,
|
|
displayText,
|
|
isPlaceholderOnly,
|
|
}
|