// Spawn atlas parsers — pure functions over strings, no `fs`, no dependencies. // // These back the CLI build script (`scripts/buildSpawnAtlas.js`), which is the // only thing that reads a ServUO tree. Keeping every parser pure and fs-free is // what lets the test suite cover them in CI, where no ServUO tree exists: the // tests hand these functions literal XML strings. // // Four source shapes, two very different parsing strategies: // // Spawns/*.xml ~10.5 MB across 13 files, FLAT records // → streaming regex, never a DOM. See parsePoints(). // Data/Regions.xml 129 KB, genuinely nested inside // Data/Locations/*.xml nested / // Config/ChampionSpawns.xml 4.8 KB, / // → the small recursive tokenizer below. // // The server has zero XML dependencies and this adds none. The tokenizer is // deliberately a *subset* parser: it handles the constructs these four files // actually use (elements, attributes, self-closing tags, comments, the XML // declaration, CDATA, the five predefined entities plus numeric refs) and // nothing else. It is not a general-purpose XML parser and must not be reused // as one — no namespaces, no DTDs, no entity declarations. // ── Entities ─────────────────────────────────────────────────────────────── const NAMED_ENTITIES = { amp: '&', lt: '<', gt: '>', quot: '"', apos: "'", } // Region and location names carry apostrophes ("Mondain's Legacy", "Wrong's // Level 3"), so entity decoding is load-bearing here, not decorative. function decodeEntities(text) { if (!text.includes('&')) return text return text.replace(/&(#x?[0-9a-fA-F]+|[a-zA-Z]+);/g, (match, body) => { if (body[0] === '#') { const code = body[1] === 'x' || body[1] === 'X' ? Number.parseInt(body.slice(2), 16) : Number.parseInt(body.slice(1), 10) return Number.isFinite(code) ? String.fromCodePoint(code) : match } const named = NAMED_ENTITIES[body.toLowerCase()] return named === undefined ? match : named }) } // ── The tokenizer ────────────────────────────────────────────────────────── const ATTR_RE = /([\w:.-]+)\s*=\s*("([^"]*)"|'([^']*)')/g function parseAttrs(source) { const attrs = {} ATTR_RE.lastIndex = 0 let match while ((match = ATTR_RE.exec(source)) !== null) { const raw = match[3] !== undefined ? match[3] : match[4] attrs[match[1]] = decodeEntities(raw) } return attrs } /** * Parse a small nested XML document into `{ name, attrs, children, text }`. * * Intended for Regions.xml / Locations / ChampionSpawns.xml only — never for * the multi-megabyte Spawns files. Returns the root element, or `null` for a * document with no elements. * * Mismatched or stray closing tags are ignored rather than thrown on: these are * hand-maintained shard config files, and one malformed region should degrade * to a missing region, not abort a build that is otherwise fine. */ function parseXml(source) { const text = String(source) const root = { name: '#document', attrs: {}, children: [], text: '' } const stack = [root] let i = 0 while (i < text.length) { const lt = text.indexOf('<', i) if (lt === -1) { appendText(stack[stack.length - 1], text.slice(i)) break } if (lt > i) appendText(stack[stack.length - 1], text.slice(i, lt)) // Comment, declaration/DOCTYPE, or CDATA — skipped wholesale. if (text.startsWith('', lt + 4) i = end === -1 ? text.length : end + 3 continue } if (text.startsWith('', lt + 9) const stop = end === -1 ? text.length : end appendRawText(stack[stack.length - 1], text.slice(lt + 9, stop)) i = end === -1 ? text.length : end + 3 continue } if (text.startsWith('', lt + 2) i = end === -1 ? text.length : end + 2 continue } if (text.startsWith('', lt + 2) i = end === -1 ? text.length : end + 1 continue } const gt = findTagEnd(text, lt) if (gt === -1) { // Unterminated tag: nothing sane is left to read. break } const inner = text.slice(lt + 1, gt) if (inner[0] === '/') { const name = inner.slice(1).trim() // Pop to the nearest matching open element. If there is no match the tag // is stray and we drop it rather than unwinding the whole stack. for (let depth = stack.length - 1; depth > 0; depth -= 1) { if (stack[depth].name === name) { stack.length = depth break } } i = gt + 1 continue } const selfClosing = inner.endsWith('/') const body = selfClosing ? inner.slice(0, -1) : inner const space = body.search(/\s/) const name = (space === -1 ? body : body.slice(0, space)).trim() const node = { name, attrs: space === -1 ? {} : parseAttrs(body.slice(space)), children: [], text: '', } stack[stack.length - 1].children.push(node) if (!selfClosing) stack.push(node) i = gt + 1 } return root.children.length > 0 ? root.children[0] : null } // `>` inside a quoted attribute value must not end the tag. function findTagEnd(text, from) { let quote = null for (let i = from + 1; i < text.length; i += 1) { const ch = text[i] if (quote) { if (ch === quote) quote = null } else if (ch === '"' || ch === "'") { quote = ch } else if (ch === '>') { return i } } return -1 } function appendText(node, chunk) { if (chunk.trim() === '') return appendRawText(node, decodeEntities(chunk)) } function appendRawText(node, chunk) { node.text = node.text ? `${node.text}${chunk}` : chunk } function childrenNamed(node, name) { if (!node || !node.children) return [] return node.children.filter((child) => child.name === name) } // ── Facet names ──────────────────────────────────────────────────────────── // The three sources disagree about facet spelling and nothing in the files // reconciles them: `Spawns/*.xml` `` and `Regions.xml` `` both // say `TerMur`/`Tokuno`, while `Data/Locations/*.xml` spells the same facets // `Ter Mur` and `Tokuno Islands`. Left alone this is silent — the landmark // fallback simply never matches on those two facets and every unregioned spawn // in Ter Mur and Tokuno reads "Wilderness" — so every facet name entering the // atlas is canonicalised through here first. const FACET_CANONICAL = new Map([ ['felucca', 'Felucca'], ['trammel', 'Trammel'], ['ilshenar', 'Ilshenar'], ['malas', 'Malas'], ['tokuno', 'Tokuno'], ['tokunoislands', 'Tokuno'], ['termur', 'TerMur'], ]) /** * Canonicalise a facet name to the `` spelling the atlas keys on. * Unknown facets pass through trimmed rather than being dropped, so a custom * shard facet still gets an atlas rather than vanishing. */ function normalizeFacet(value) { const raw = String(value ?? '').trim() if (raw === '') return '' return FACET_CANONICAL.get(raw.toLowerCase().replace(/[\s_-]+/g, '')) ?? raw } // ── Small coercions ──────────────────────────────────────────────────────── function toInt(value, fallback = 0) { const n = Number.parseInt(value, 10) return Number.isFinite(n) ? n : fallback } function toBool(value) { return String(value).trim().toLowerCase() === 'true' } /** * URL-safe slug used as the creature primary key and in `/atlas/:slug`. * Spawn type tokens are C# class names, so they are already ASCII-ish; this * mainly lowercases and collapses punctuation. */ function slugify(value) { return String(value) .trim() .toLowerCase() .replace(/[^a-z0-9]+/g, '-') .replace(/^-+|-+$/g, '') } // ── Objects2 ─────────────────────────────────────────────────────────────── /** * Parse a `` value into `[{ type, max }]`. * * The format is one or more segments joined by `:OBJ=`, each segment being * `Type:MX=n:SB=0:RT=0:...` — the type is the token before the first `:`, and * every following token is a `KEY=value` pair. Verified against trammel.xml, * where a single point carries six types: * * Giantserpent:MX=1:...:OBJ=Giantspider:MX=1:...:OBJ=Boar:MX=1:... * * Splitting on `:` alone would shred this, which is why the `:OBJ=` split comes * first. `MX` is that type's own max count and is what the atlas displays; * every other flag (spawn/trigger/refractory bookkeeping) is dropped. * * The type token itself may carry XmlSpawner directives appended to the class * name — property assignments after `/` and an amount/argument list after `,`: * * Agralem/Name/Agralem alchemist/z/-50 Fairy,{RND,4,8} * GargishRefugee/hue/34532 greatape,true GargishRouser,1 * * Taken literally these produce creatures that do not exist ("alchemist/z/-50") * AND split real ones in two, because `Fairy` and `Fairy,{RND,4,8}` slug apart — * 71 of 845 entries were affected before this was stripped. Only the leading * class name identifies the creature, so everything from the first `/` or `,` * is dropped. */ /** Reduce an XmlSpawner type token to the bare class name. */ function stripSpawnerDirectives(token) { const cut = String(token).search(/[/,]/) return (cut === -1 ? String(token) : String(token).slice(0, cut)).trim() } function parseObjects2(value) { const source = String(value ?? '').trim() if (source === '') return [] return source .split(':OBJ=') .map((segment) => { const tokens = segment.split(':') const type = stripSpawnerDirectives(tokens.shift() ?? '') if (type === '') return null let max = 1 for (const token of tokens) { const eq = token.indexOf('=') if (eq === -1) continue if (token.slice(0, eq).trim().toUpperCase() === 'MX') { max = toInt(token.slice(eq + 1), 1) } } return { type, max } }) .filter((entry) => entry !== null) } // ── Spawns/*.xml ─────────────────────────────────────────────────────────── const POINT_RE = /([\s\S]*?)<\/Points>/g function tagValue(block, name) { const match = block.match(new RegExp(`<${name}>([\\s\\S]*?)`)) return match ? decodeEntities(match[1]).trim() : '' } /** * Parse a `Spawns/.xml` file into spawn point records. * * Deliberately regex/streaming and NOT `parseXml` — these files total ~10.5 MB * and putting them through a DOM builder would allocate a node per element for * ~40 fields on every one of ~6,500 records to keep 14 of them. The records are * flat, so a per-record regex sweep is both correct and cheap. * * Only the fields the site can actually show are kept. Everything to do with * triggering, refractory windows, proximity, sequential spawning, sounds and * `UniqueId` is dropped here rather than downstream — that is what holds the * committed artifact under 1 MB. * * NOTE: the facet comes from each record's own ``, never from the file * name. `Eodon.xml`, `GravewaterLake.xml` and the other named-area files all * carry TerMur/Trammel points, so there are 13 files but only 6 facets. */ function parsePoints(source) { const text = String(source) const points = [] POINT_RE.lastIndex = 0 let match while ((match = POINT_RE.exec(text)) !== null) { const block = match[1] const facet = normalizeFacet(tagValue(block, 'Map')) if (facet === '') continue points.push({ name: tagValue(block, 'Name'), facet, x: toInt(tagValue(block, 'X')), y: toInt(tagValue(block, 'Y')), width: toInt(tagValue(block, 'Width')), height: toInt(tagValue(block, 'Height')), range: toInt(tagValue(block, 'Range')), maxCount: toInt(tagValue(block, 'MaxCount')), minDelay: toInt(tagValue(block, 'MinDelay')), maxDelay: toInt(tagValue(block, 'MaxDelay')), // Time-of-day gating: TODMode 0 means "always", in which case the start // and end values are meaningless and the site must not render them. todStart: toInt(tagValue(block, 'TODStart')), todEnd: toInt(tagValue(block, 'TODEnd')), todMode: toInt(tagValue(block, 'TODMode')), // A spawner switched off in-world spawns nothing; the build filters these // out so the atlas describes what actually appears, not what is merely // configured. Parsed here so the decision stays in the build script. running: toBool(tagValue(block, 'IsRunning')), types: parseObjects2(tagValue(block, 'Objects2')), }) } return points } // ── Data/Regions.xml ─────────────────────────────────────────────────────── /** * Flatten `Data/Regions.xml` into `[{ facet, name, type, priority, parent, rects }]`. * * Regions nest: a `` may contain further `` elements, and the * inner ones frequently omit `name` and `priority` (`` * inside "Prism of Light"). Unnamed regions are skipped — they cannot label a * spawn point — but their children are still walked, and a child that omits * `priority` inherits its parent's rather than defaulting to 0, which would * quietly sort it below every top-level region. */ function parseRegions(source) { const root = parseXml(source) const regions = [] if (!root) return regions for (const facetNode of childrenNamed(root, 'Facet')) { const facet = normalizeFacet(facetNode.attrs.name) if (facet === '') continue walkRegions(facetNode, facet, null, 0, regions) } return regions } function walkRegions(node, facet, parentName, parentPriority, out) { for (const regionNode of childrenNamed(node, 'region')) { const name = regionNode.attrs.name || '' const priority = Object.hasOwn(regionNode.attrs, 'priority') ? toInt(regionNode.attrs.priority, parentPriority) : parentPriority if (name !== '') { const rects = childrenNamed(regionNode, 'rect').map((rect) => ({ x: toInt(rect.attrs.x), y: toInt(rect.attrs.y), width: toInt(rect.attrs.width), height: toInt(rect.attrs.height), })) // A named region with no rects (some exist purely to carry music or a // `go` point) can never contain anything, so it is not worth indexing. if (rects.length > 0) { out.push({ facet, name, type: regionNode.attrs.type || '', priority, parent: parentName, rects, }) } } walkRegions(regionNode, facet, name === '' ? parentName : name, priority, out) } } // ── Data/Locations/*.xml ─────────────────────────────────────────────────── /** * Flatten a `Data/Locations/.xml` into landmark points. * * The file nests `` arbitrarily deep and puts coordinates only on * ``: Trammel → Dungeons → Covetous → "Level 1". The outermost parent is * the facet itself and is dropped from `path`; `group` is the innermost * enclosing parent ("Covetous"), which is the label worth showing — "Covetous" * reads better than "Level 1" when naming where a spawn is. */ function parseLocations(source, facetHint = '') { const root = parseXml(source) const landmarks = [] if (!root) return landmarks for (const top of childrenNamed(root, 'parent')) { const facet = normalizeFacet(top.attrs.name || facetHint) walkLocations(top, facet, [], landmarks) } return landmarks } function walkLocations(node, facet, path, out) { for (const child of childrenNamed(node, 'child')) { const name = child.attrs.name || '' if (name === '') continue out.push({ facet, name, group: path.length > 0 ? path[path.length - 1] : name, path: [...path], x: toInt(child.attrs.x), y: toInt(child.attrs.y), z: toInt(child.attrs.z), }) } for (const parent of childrenNamed(node, 'parent')) { const name = parent.attrs.name || '' walkLocations(parent, facet, name === '' ? path : [...path, name], out) } } // ── Config/ChampionSpawns.xml ────────────────────────────────────────────── /** * Parse `Config/ChampionSpawns.xml` into champion altar records. * * This is the shard's *configured* champion roster — which altars exist, where, * and which type each is pinned to. It is static content and distinct from the * live `champ.update` feed the bridge already carries: this says "there is an * Unholy Terror altar in Deceit", the feed says "it is on level 3 right now". * * A spawn with no `type` is randomised on every activation, which the site must * render as "random" rather than as an empty type. */ function parseChampions(source) { const root = parseXml(source) const champions = [] if (!root) return champions for (const spawnNode of childrenNamed(root, 'spawn')) { const location = childrenNamed(spawnNode, 'location')[0] const attrs = location ? location.attrs : {} champions.push({ name: spawnNode.attrs.name || '', group: spawnNode.attrs.group || '', type: spawnNode.attrs.type || '', randomType: !spawnNode.attrs.type, facet: normalizeFacet(attrs.map), x: toInt(attrs.x), y: toInt(attrs.y), z: toInt(attrs.z), radius: toInt(attrs.radius), }) } return champions } // ── Placement ────────────────────────────────────────────────────────────── const DEFAULT_LANDMARK_RADIUS = 200 function inRect(x, y, rect) { return ( x >= rect.x && x < rect.x + rect.width && y >= rect.y && y < rect.y + rect.height ) } function rectArea(rect) { return Math.max(1, rect.width) * Math.max(1, rect.height) } /** * Group parsed regions and landmarks by facet once, so the per-point resolve * below is a scan of one facet instead of the whole world. With ~6,500 points * and a few thousand rects this stays comfortably sub-second; there is no need * for a spatial index and none is worth the complexity. */ function buildPlacementIndex(regions, landmarks) { const byFacet = new Map() const facet = (name) => { if (!byFacet.has(name)) byFacet.set(name, { regions: [], landmarks: [] }) return byFacet.get(name) } for (const region of regions) facet(region.facet).regions.push(region) for (const landmark of landmarks) facet(landmark.facet).landmarks.push(landmark) return byFacet } /** * Turn a raw coordinate into a human place name. * * This is the transform the whole atlas exists for: it is what makes a row read * "Lizardman — Despise, Felucca" instead of "Lizardman — 5411, 1234". * * Resolution order: * 1. The highest-`priority` named region whose rect contains the point. Ties * break toward the SMALLEST rect, so a specific room inside a dungeon wins * over the dungeon-wide rect it sits in. * 2. Otherwise the nearest landmark within `landmarkRadius` tiles, labelled by * its group ("Covetous"), not the individual marker ("Level 1"). * 3. Otherwise "Wilderness". The radius cap is what keeps step 3 reachable — * without it the nearest landmark is always *some* landmark, however far, * and open countryside would get labelled with a dungeon on the far side * of the map. */ function resolveRegion(x, y, facetName, index, options = {}) { const radius = options.landmarkRadius ?? DEFAULT_LANDMARK_RADIUS const bucket = index.get(facetName) const result = { region: null, landmark: null, label: 'Wilderness' } if (!bucket) return result let best = null let bestPriority = -Infinity let bestArea = Infinity for (const region of bucket.regions) { for (const rect of region.rects) { if (!inRect(x, y, rect)) continue const area = rectArea(rect) if (region.priority > bestPriority || (region.priority === bestPriority && area < bestArea)) { best = region bestPriority = region.priority bestArea = area } } } if (best) { result.region = best.name result.label = best.name return result } let nearest = null let nearestDistance = Infinity const limit = radius * radius for (const landmark of bucket.landmarks) { const dx = landmark.x - x const dy = landmark.y - y const distance = dx * dx + dy * dy if (distance < nearestDistance) { nearest = landmark nearestDistance = distance } } if (nearest && nearestDistance <= limit) { result.landmark = nearest.group || nearest.name result.label = result.landmark } return result } module.exports = { parseXml, parseObjects2, parsePoints, parseRegions, parseLocations, parseChampions, buildPlacementIndex, resolveRegion, normalizeFacet, slugify, decodeEntities, DEFAULT_LANDMARK_RADIUS, }