// Cliloc parsing — the pure half. // // A "cliloc" is UO's localization table: an integer id mapped to a display // string. Items carry a `LabelNumber` rather than a name, so without this table // the site can only render `id 1023721` where the game shows "quarter staff". // The shard already sends the id on every equipment entry (`char.profile`'s // `cliloc` field) and will send one per marketplace listing — the *number* was // never the missing piece, the *table* was. // // This module is fs-free on purpose, exactly like `spawnAtlasParse.js`: the // suite runs in CI where there is no UO client, so every parser here is driven // from inline fixtures. `clilocSource.js` is the only thing that touches disk. // // ── Two input formats, and why ───────────────────────────────────────────── // // The client's own `Cliloc.enu` is COMPRESSED (Mythic format) on any modern // client, and decompressing it is a bit-level port of an inverse-BWT coder that // nothing in this stack needs at runtime. ServUO's own bundled `Ultima.StringList` // cannot read it either — which is why `VendorSearch.GetItemName` is already inert // on such a shard and the plugin could not supply names even if we asked it to. // // So the operator converts once, from their own client, and points the site at // the result (see docs/website/CLILOCS.md). Two shapes are accepted because // different tools produce different things: // // • PLAIN BINARY — the pre-compression cliloc layout: a 6-byte header, then // records of {int32 number, byte flag, uint16 length, UTF-8 bytes}. // • DELIMITED TEXT — `numbertext` per line, which is what the common // GUI exports emit. Quoted CSV fields and a header row are tolerated. // // Nothing derived from the client is ever committed: the converted file lives at // an operator-supplied path and is gitignored, the same rule the spawn atlas art // map already follows. /** Raised for a file we can identify but deliberately refuse to guess at. */ class ClilocFormatError extends Error { constructor(message, code) { super(message) this.name = 'ClilocFormatError' this.code = code } } /** * Bumped when this parser produces DIFFERENT data from an IDENTICAL source file. * * Stored beside the source hash so the boot path can tell "same file, but the * parser moved on" from "same file, nothing to do". Without it a corrected parse * would ship and never reach an install whose cliloc file never changes — the * trap `spawnAtlasSource.PARSER_VERSION` documents. */ const PARSER_VERSION = 1 // The plain layout's header is `02 00 00 00 01 00` — a 4-byte version and a // 2-byte language marker. Only the size matters for parsing; the values are // checked to sniff the format, not to validate it. const HEADER_BYTES = 6 const RECORD_HEADER_BYTES = 7 // int32 number + byte flag + uint16 length // Every compressed cliloc file the client ships begins with a DWORD whose high // byte is 0x8E (the XOR key UOFiddler calls `HeaderXorKey`, 0x8E2C9A3D). That is // the single cheapest way to tell an operator they exported the wrong file — // without it, the plain parser happily reads compressed bytes as ~19k records of // negative ids and 60 KB "strings" before dying somewhere in the middle, and the // resulting error names the wrong problem. const MYTHIC_HIGH_BYTE = 0x8e /** True when `buffer` is a Mythic-compressed cliloc rather than the plain layout. */ function isCompressedCliloc(buffer) { return buffer.length >= 4 && buffer[3] === MYTHIC_HIGH_BYTE } /** * Parse the plain binary cliloc layout. * * Strict about truncation, and that strictness is load-bearing: a half-copied or * partly-written file is the realistic failure here, and it must fail loudly * rather than import a silently short table that then renders half the world as * `id 1023721`. A record that runs past the end of the buffer throws. */ function parseClilocBinary(buffer) { if (!Buffer.isBuffer(buffer)) throw new ClilocFormatError('Not a buffer', 'NOT_BUFFER') if (isCompressedCliloc(buffer)) { throw new ClilocFormatError( 'This is a compressed (Mythic-format) cliloc file, which the site cannot read. ' + 'Convert it to the plain format first — see docs/website/CLILOCS.md.', 'COMPRESSED', ) } if (buffer.length < HEADER_BYTES) { throw new ClilocFormatError('File is shorter than a cliloc header', 'TRUNCATED') } const entries = [] let offset = HEADER_BYTES while (offset < buffer.length) { if (offset + RECORD_HEADER_BYTES > buffer.length) { throw new ClilocFormatError( `Truncated record header at byte ${offset} (${entries.length} entries read)`, 'TRUNCATED', ) } const number = buffer.readInt32LE(offset) const flag = buffer.readUInt8(offset + 4) // The length is written by the client as an unsigned 16-bit value. Reading it // signed (as ServUO's own SDK does) turns any string over 32 KB into a // negative length; real tables top out around 12 KB, so this has no effect on // current data and costs nothing to get right. const length = buffer.readUInt16LE(offset + 5) offset += RECORD_HEADER_BYTES if (offset + length > buffer.length) { throw new ClilocFormatError( `Truncated record body at byte ${offset} (${entries.length} entries read)`, 'TRUNCATED', ) } entries.push({ number, flag, text: buffer.toString('utf8', offset, offset + length) }) offset += length } return entries } // A delimited line splits on the FIRST separator only: cliloc text is full of // commas ("a scroll of magery, unfinished") and splitting on all of them would // truncate every such entry at its first comma. const TEXT_SEPARATORS = ['\t', ',', ';'] /** Unwrap one CSV field: strip surrounding quotes and unescape doubled quotes. */ function unquote(value) { const trimmed = value.trim() if (trimmed.length >= 2 && trimmed.startsWith('"') && trimmed.endsWith('"')) { return trimmed.slice(1, -1).replace(/""/g, '"') } return trimmed } /** * Parse a delimited text export: `numbertext` per line. * * Tolerant by design — this is whatever an operator's GUI tool produced, not a * format we control. A header row, blank lines, `#` comments and a trailing * flags column are all ignored. A line whose first field is not an integer is * skipped rather than fatal, because that is exactly what a header row is. * * The one thing it will NOT do is return an empty table quietly: a file that * yields no entries at all is a wrong file, not an empty one. */ function parseClilocText(text) { const entries = [] for (const line of String(text).split(/\r?\n/)) { // The line is deliberately NOT trimmed before the separator search. Roughly // half of a real cliloc table is empty strings (unused ids), which export as // `1005008` — and trimming eats that trailing separator, leaving a bare // number that then looks like a header row and is skipped. That silently // dropped 55,994 of 123,490 entries. Individual FIELDS are trimmed instead, // by `unquote`. if (line.trim() === '' || line.trimStart().startsWith('#')) continue // Pick the separator that actually appears first, so a tab-delimited line // whose text contains a comma still splits on the tab. let cut = -1 for (const sep of TEXT_SEPARATORS) { const at = line.indexOf(sep) if (at !== -1 && (cut === -1 || at < cut)) cut = at } if (cut === -1) continue // An EMPTY first field must not become id 0: `Number('')` is 0, not NaN, so // a line that merely starts with a separator would otherwise import as a // bogus cliloc 0 instead of being skipped. const head = unquote(line.slice(0, cut)) if (head === '') continue const number = Number(head) if (!Number.isInteger(number)) continue // header row, or a wrapped line let rest = line.slice(cut + 1) // Some exports carry `number,flag,text`. A bare integer in the second field // is a flag; anything else is the text itself (and a text field that IS just // a number is indistinguishable, so it stays as the text — the safer miss). let flag = 0 for (const sep of TEXT_SEPARATORS) { const at = rest.indexOf(sep) if (at === -1) continue const head = unquote(rest.slice(0, at)) if (/^\d{1,3}$/.test(head) && rest.slice(at + 1).trim() !== '') { flag = Number(head) rest = rest.slice(at + 1) } break } entries.push({ number, flag, text: unquote(rest) }) } if (entries.length === 0) { throw new ClilocFormatError('No cliloc entries found in the text export', 'EMPTY') } return entries } /** * Parse either supported shape, sniffing which one this is. * * The sniff is on the binary header rather than the file extension: operators * name these things whatever they like, and an `.enu` that is really a TSV (or a * `.txt` that is really binary) should still import. */ function parseCliloc(buffer) { const buf = Buffer.isBuffer(buffer) ? buffer : Buffer.from(buffer) if (isCompressedCliloc(buf)) { throw new ClilocFormatError( 'This is a compressed (Mythic-format) cliloc file, which the site cannot read. ' + 'Convert it to the plain format first — see docs/website/CLILOCS.md.', 'COMPRESSED', ) } // The plain layout always opens with version 2 / language 1. Anything else is // treated as text, which is the recoverable guess: a mis-sniffed text file // yields "no entries found", while a mis-sniffed binary yields nonsense. if (buf.length >= HEADER_BYTES && buf.readInt32LE(0) === 2 && buf.readUInt16LE(4) === 1) { return parseClilocBinary(buf) } return parseClilocText(buf.toString('utf8')) } // ── Display ──────────────────────────────────────────────────────────────── // Cliloc strings interpolate arguments the client supplies out of an item's // property list: `~1_val~`, `~2_NAME~`, `~1_ITEM~`. We never have those — the // bridge sends the id, not the packet — so a name carrying them must be reduced // to what is actually knowable rather than shown with the raw tokens in it. const PLACEHOLDER_RE = /~\d+_[^~]*~/g /** * Reduce a raw cliloc string to something displayable. * * Placeholders are dropped and the leftover punctuation tidied, so * `"[~1_stuff~]"` becomes `""` (correctly nothing — the whole string was the * argument) and `"cold damage ~1_val~%"` becomes `"cold damage"`. * * **Punctuation is only tidied when a placeholder was actually removed.** The * trailing `%` above is the unit belonging to the number we never had, and the * brackets in `[~1_stuff~]` only ever wrapped the argument — but a string with * no placeholder has no such debris, and trimming it anyway corrupts real names. * A shard's `"Runic Gateway Sigil (v2)"` came back as `"(v2"` while this was * unconditional. * * Returns `''` when nothing survives, which callers treat as "no name" and fall * back to the item id — better than showing a bracket. */ const DEBRIS = /^[\s\-–—,.;:%[\]()]+|[\s\-–—,.;:%[\]()]+$/g function displayText(raw) { if (raw == null) return '' const source = String(raw) const hadPlaceholder = PLACEHOLDER_RE.test(source) PLACEHOLDER_RE.lastIndex = 0 // the regex is global; `test` advances it if (!hadPlaceholder) return source.replace(/\s+/g, ' ').trim() return source .replace(PLACEHOLDER_RE, ' ') .replace(/\s+/g, ' ') .replace(/\s+([,.;:!?])/g, '$1') .replace(DEBRIS, '') .trim() } /** True when a raw cliloc string is nothing but interpolated arguments. */ const isPlaceholderOnly = (raw) => raw != null && String(raw).trim() !== '' && displayText(raw) === '' module.exports = { ClilocFormatError, PARSER_VERSION, HEADER_BYTES, isCompressedCliloc, parseCliloc, parseClilocBinary, parseClilocText, displayText, isPlaceholderOnly, }