Uninstall-with-purge ran purge.sql while the module was still started: the tables went, and the module kept serving and ingesting against a schema that no longer existed until lifecycle.stop() finished — up to the five-second hook budget. For module-uo that is the uo-link WebSocket writing shard events into dropped tables, and requests in flight answering 500 where a stopped module answers 404. Nothing required the old order. The comment justified it as "purge while the SQL is still readable", but removeDir is the only step that touches the filesystem, so purge.sql stays readable until after the stop. The 400 for a module that ships no purge.sql is now resolved before anything is stopped, so a refused request leaves the module exactly as it found it. Found while proving Phase 4's acceptance criterion 2 against the real module-uo v0.3.0 release on an empty database (MODULE_SYSTEM.md §2.7.2). 742 server tests (+1); routes.manifest.json and swagger-output.json byte-identical. AI disclosure: this contribution was AI-assisted (Claude Code). Co-Authored-By: Claude <noreply@anthropic.com>
434 lines
20 KiB
JavaScript
434 lines
20 KiB
JavaScript
// ── Admin: installed modules ───────────────────────────────────────────────
|
|
//
|
|
// Phase 4, slice 1 of docs/website/MODULE_SYSTEM.md §2.7.2. Admin-only, and more
|
|
// so than anything else in this directory: installing a module puts JavaScript on
|
|
// the volume that core will `require` into its own process on the next boot. That
|
|
// is the feature — it is what "an operator never builds anything" means (§1.14) —
|
|
// but it is worth being plain that this controller is remote code execution with
|
|
// an audit trail, not a settings screen.
|
|
//
|
|
// What guards it, in the order an attacker would meet them:
|
|
//
|
|
// 1. `requireRole('admin')` on every route, on top of the group's staff gate.
|
|
// 2. An `https`-only host allowlist, re-checked on every redirect hop, so a
|
|
// pasted URL cannot be pointed at the compose network or a metadata service
|
|
// (install.js).
|
|
// 3. The sha256 the release published, compared against the bytes that arrived.
|
|
// 4. A full inspection of the archive before a byte of it is unpacked, and an
|
|
// unpack into a scratch directory that is only moved into place once the
|
|
// bundle has agreed with the manifest about what it is (archive.js).
|
|
// 5. Every action here writes to core's one audit log.
|
|
//
|
|
// The one thing this file cannot do is mount anything. §1.12 makes the volume the
|
|
// mounting source of truth, read once at require time, so install and uninstall
|
|
// take effect on the next boot — which is why `restart` is a route here rather
|
|
// than a sentence in a tooltip (decision 1).
|
|
|
|
const modules = require('../../../model/modules/modules.model')
|
|
const activity = require('../../../model/activity/activity.model')
|
|
const settings = require('../../../model/settings/settings.model')
|
|
const loader = require('../../../modules/loader')
|
|
const lifecycle = require('../../../modules/lifecycle')
|
|
const install = require('../../../modules/install')
|
|
const declared = require('../../../modules/declared')
|
|
// A namespace import, like every other require in this file, and not
|
|
// `const { runPurge } = …`: destructuring at require time captures the function
|
|
// rather than the module, which makes it the one dependency here that cannot be
|
|
// substituted. That matters because the two tests worth having about purge are
|
|
// about the ORDER it runs in relative to the directory being removed.
|
|
const schema = require('../../../modules/schema')
|
|
|
|
const log = require('../../../utils/logger')('admin-modules')
|
|
|
|
// The allowlist setting. Seeded from MODULE_SOURCE_HOSTS on first boot and
|
|
// admin-managed from then on (decision 6) — db/seed.js writes it once and never
|
|
// overwrites it, so changing the variable later does not silently reach in and
|
|
// undo an operator's choice. The key itself lives in install.js, which is now
|
|
// read by boot-time declared-set resolution as well as by this controller.
|
|
const HOSTS_KEY = install.HOSTS_SETTING
|
|
|
|
// A hostname, not a URL: no scheme, no path, no port, no wildcard. Deliberately
|
|
// strict — every character allowed here is a character that can appear in the
|
|
// host of a URL this server will fetch and execute the contents of.
|
|
const HOSTNAME = /^[a-z0-9]([a-z0-9-]*[a-z0-9])?(\.[a-z0-9]([a-z0-9-]*[a-z0-9])?)*$/
|
|
|
|
async function allowedHosts() {
|
|
return install.parseHosts(await settings.get(HOSTS_KEY))
|
|
}
|
|
|
|
/**
|
|
* One module, as the admin screen needs it.
|
|
*
|
|
* Four sources have to be reconciled, and which one answers which question is
|
|
* the whole of §2.4:
|
|
*
|
|
* - the ROW says what the operator decided and what the last boot recorded;
|
|
* - the LOADER says what is mounted and answering right now;
|
|
* - the VOLUME says whether there is still a directory there at all;
|
|
* - the DECLARATION (slice 3) says what this container's environment asks for,
|
|
* which is the only one of the four an admin cannot change from this screen.
|
|
*
|
|
* They can legitimately disagree, and the screen has to show that rather than
|
|
* pick a winner. A row `enabled` with a loader state of `disabled` is a module
|
|
* the operator has just switched back on and which is waiting for a restart —
|
|
* exactly the case decision 3 creates, and it would be a lie to render it as
|
|
* either "running" or "off". A declared module with no row and no directory is
|
|
* the newest of those disagreements: MODULES asked for it and resolution could
|
|
* not get it, so the screen carries the reason rather than showing nothing.
|
|
*/
|
|
function present(row, live, onVolume, declaration = null) {
|
|
const id = row ? row.id : live ? live.id : declaration.id
|
|
return {
|
|
id,
|
|
name: row ? row.name : live ? live.name : id,
|
|
version: row ? row.version : live ? live.version : null,
|
|
// What the database records.
|
|
state: row ? row.state : null,
|
|
failureStage: row ? row.failureStage : (live && live.stage) || null,
|
|
failureReason: row ? row.failureReason : (live && live.reason) || null,
|
|
source: row ? row.source : null,
|
|
sha256: row ? row.sha256 : null,
|
|
installedAt: row ? row.installedAt : null,
|
|
startedAt: row ? row.startedAt : null,
|
|
// What is actually mounted in this process, and what it is answering.
|
|
liveState: live ? live.state : null,
|
|
// The version RUNNING, which is not always the version installed: an upgrade
|
|
// writes new files and a new row while the old code stays loaded until the
|
|
// restart. Without this the screen would report the new version as
|
|
// "Running", which is the same lie in a different place.
|
|
liveVersion: live ? live.version : null,
|
|
capabilities: live ? live.capabilities : [],
|
|
// What is on the volume.
|
|
onVolume,
|
|
canPurge: onVolume && Boolean(install.purgeFile(id)),
|
|
// What the environment declares. `declaredVersion` is what MODULES pins, not
|
|
// what is installed — they differ exactly while a resolution is failing, and
|
|
// `declaredError` says why. Uninstalling a declared module from this screen
|
|
// removes its directory and disables its row; the next boot puts the
|
|
// directory back and leaves the row disabled, so the screen says so rather
|
|
// than letting the files reappear unexplained.
|
|
declared: Boolean(declaration),
|
|
declaredVersion: declaration ? declaration.version : null,
|
|
declaredError: declaration && declaration.action === 'failed' ? declaration.message : null,
|
|
}
|
|
}
|
|
|
|
// GET /admin/modules — every module core knows about, from all three sources,
|
|
// plus the source allowlist the install form needs.
|
|
async function list(req, res) {
|
|
try {
|
|
const rows = await modules.list()
|
|
// The loader throws rather than returning [] before load() has run (§7.6),
|
|
// and this controller is reachable from a process where that is true —
|
|
// `npm run seed` never gets here, but a test harness might.
|
|
const live = loader.isLoaded() ? loader.list() : []
|
|
const byId = new Map(live.map((m) => [m.id, m]))
|
|
const declaredById = new Map(declared.state().map((d) => [d.id, d]))
|
|
|
|
const seen = new Set()
|
|
const out = []
|
|
for (const row of rows) {
|
|
seen.add(row.id)
|
|
out.push(
|
|
present(row, byId.get(row.id) || null, install.isInstalled(row.id), declaredById.get(row.id)),
|
|
)
|
|
}
|
|
// A directory on the volume that has no row yet — a hand-placed install
|
|
// before its first boot. It has to be listed, or the screen would show
|
|
// nothing for a module whose routes are already being served.
|
|
for (const m of live) {
|
|
if (!seen.has(m.id)) {
|
|
seen.add(m.id)
|
|
out.push(present(null, m, true, declaredById.get(m.id)))
|
|
}
|
|
}
|
|
// A module MODULES declares that has neither. Resolution failed and left
|
|
// nothing behind — the case an operator most needs told, because from the
|
|
// screen's other three sources it is indistinguishable from never having
|
|
// asked for it.
|
|
for (const d of declaredById.values()) {
|
|
if (!seen.has(d.id)) out.push(present(null, null, install.isInstalled(d.id), d))
|
|
}
|
|
|
|
return res.json({ modules: out, sourceHosts: await allowedHosts() })
|
|
} catch (err) {
|
|
log.error('list modules', err)
|
|
return res.status(500).json({ message: 'Internal Server Error' })
|
|
}
|
|
}
|
|
|
|
// POST /admin/modules — install (or upgrade) from an install-manifest URL.
|
|
async function create(req, res) {
|
|
const url = String(req.body.url || '').trim()
|
|
try {
|
|
const hosts = await allowedHosts()
|
|
const result = await install.install({ url, hosts })
|
|
|
|
// Provenance is written here and nowhere else: the boot reconcile records a
|
|
// module with NULL source/sha256 and leaves what it is not given, precisely
|
|
// so that a refresh cannot overwrite what an install knew (lifecycle.js).
|
|
const row = await modules.recordInstalled({
|
|
id: result.id,
|
|
name: result.name,
|
|
version: result.version,
|
|
source: result.source,
|
|
sha256: result.sha256,
|
|
})
|
|
|
|
await activity.log({
|
|
req,
|
|
userId: req.user.id,
|
|
action: 'module.install',
|
|
detail: { id: result.id, version: result.version, source: url, sha256: result.sha256, replaced: result.replaced },
|
|
})
|
|
log.warn('module installed — it will mount on the next restart', {
|
|
id: result.id,
|
|
version: result.version,
|
|
by: req.user.username,
|
|
})
|
|
|
|
return res.status(201).json({ module: row, restartRequired: true, replaced: result.replaced })
|
|
} catch (err) {
|
|
if (err.name === 'InstallError' || err.name === 'ArchiveError') {
|
|
// The operator pasted a URL and something about what came back was wrong.
|
|
// The message is the useful part and is written to be read by them.
|
|
log.warn('module install refused', { url, reason: err.message })
|
|
return res.status(err.status || 400).json({ message: err.message })
|
|
}
|
|
log.error('install module', err)
|
|
return res.status(500).json({ message: 'Internal Server Error' })
|
|
}
|
|
}
|
|
|
|
// POST /admin/modules/:id/enable — switch a module back on, for the next boot.
|
|
//
|
|
// Deliberately does NOT touch the loader's record. Disable ran the module's
|
|
// onShutdown (decision 3), and there is no onBoot re-dispatch to undo that: a
|
|
// module whose sockets were closed and timers cleared cannot be made to serve
|
|
// again by flipping a flag, and pretending otherwise would put it back on the
|
|
// nav with a torn-down world behind it. The row moves; the restart starts it.
|
|
async function enable(req, res) {
|
|
const { id } = req.params
|
|
try {
|
|
const row = await modules.enable(id)
|
|
if (!row) return res.status(404).json({ message: 'No such module.' })
|
|
|
|
await activity.log({ req, userId: req.user.id, action: 'module.enable', detail: { id } })
|
|
return res.json({ module: row, restartRequired: true })
|
|
} catch (err) {
|
|
if (err.name === 'ModuleStateError') return res.status(409).json({ message: err.message })
|
|
log.error('enable module', err)
|
|
return res.status(500).json({ message: 'Internal Server Error' })
|
|
}
|
|
}
|
|
|
|
// POST /admin/modules/:id/disable — stop it now.
|
|
//
|
|
// The one action on this screen that takes effect without a restart, and the
|
|
// reason it does is that it is the one an operator reaches for when something is
|
|
// going wrong. Its routes answer 404 from the moment this returns, and its
|
|
// onShutdown has already run.
|
|
async function disable(req, res) {
|
|
const { id } = req.params
|
|
try {
|
|
const current = await modules.get(id)
|
|
if (!current) return res.status(404).json({ message: 'No such module.' })
|
|
|
|
const { stopped, error } = await lifecycle.stop(id)
|
|
const row = await modules.get(id)
|
|
|
|
await activity.log({
|
|
req,
|
|
userId: req.user.id,
|
|
action: 'module.disable',
|
|
detail: { id, hookRan: stopped, hookError: error },
|
|
})
|
|
log.warn('module disabled by an operator', { id, hookRan: stopped, by: req.user.username })
|
|
|
|
// `shutdownError` is reported rather than swallowed: the module IS disabled
|
|
// either way, and an operator whose module could not close cleanly should be
|
|
// told so while they still have the logs to look at.
|
|
return res.json({ module: row, stopped, shutdownError: error })
|
|
} catch (err) {
|
|
log.error('disable module', err)
|
|
return res.status(500).json({ message: 'Internal Server Error' })
|
|
}
|
|
}
|
|
|
|
// DELETE /admin/modules/:id[?purge=true] — uninstall.
|
|
//
|
|
// Non-destructive by default (§2.5): the directory goes, the row stays
|
|
// `disabled`, and the module's tables and data are left alone.
|
|
//
|
|
// The purge option is here rather than as a follow-up action because it cannot
|
|
// be a follow-up action (decision 5): `purge.sql` is a file inside the directory
|
|
// this is about to delete, so after an uninstall there is nothing left to purge
|
|
// with. Ticking the box is the last moment the file exists.
|
|
//
|
|
// The order below is the whole of it: locate the purge file, STOP, purge, then
|
|
// delete. Only the last step is destructive to the filesystem, so the file stays
|
|
// readable the whole way through — which is why stopping comes first. Dropping a
|
|
// module's tables while it is still started leaves it serving and ingesting
|
|
// against a schema that no longer exists: module-uo's uo-link WebSocket keeps
|
|
// writing shard events for as long as its onShutdown takes to run, and requests
|
|
// in flight answer 500 where a stopped module answers 404.
|
|
async function remove(req, res) {
|
|
const { id } = req.params
|
|
const purge = req.query.purge === 'true' || req.query.purge === '1'
|
|
try {
|
|
const current = await modules.get(id)
|
|
const onVolume = install.isInstalled(id)
|
|
if (!current && !onVolume) return res.status(404).json({ message: 'No such module.' })
|
|
|
|
// Resolved before anything happens, because this branch is a 400: a request
|
|
// that is going to be refused must not stop the module on its way out.
|
|
let purgeSql = null
|
|
if (purge) {
|
|
purgeSql = install.purgeFile(id)
|
|
if (!purgeSql) {
|
|
return res.status(400).json({
|
|
message: 'This module ships no purge.sql, so its data cannot be deleted. Uninstall without purging instead.',
|
|
})
|
|
}
|
|
}
|
|
|
|
// Stop it before its tables and then its files vanish. A module whose
|
|
// directory is deleted out from under a running onShutdown is being asked to
|
|
// tear down a world whose code may already be half-unreadable — and its
|
|
// sockets would otherwise stay open until the restart, holding a connection
|
|
// on behalf of a module that no longer exists on disk.
|
|
await lifecycle.stop(id)
|
|
|
|
const purged = purgeSql ? await schema.runPurge(purgeSql) : null
|
|
const removed = await install.removeDir(id)
|
|
|
|
// A purge leaves nothing: no directory, no tables, no data. Keeping a
|
|
// `disabled` row for that is a tombstone with nothing to offer and a Purge
|
|
// button that would fail. A plain uninstall keeps its row, which is what
|
|
// makes the retained data visible and reinstallable.
|
|
if (purge) await modules.remove(id)
|
|
|
|
await activity.log({
|
|
req,
|
|
userId: req.user.id,
|
|
action: purge ? 'module.purge' : 'module.uninstall',
|
|
detail: { id, purged, removed },
|
|
})
|
|
log.warn(`module ${purge ? 'uninstalled and purged' : 'uninstalled'}`, {
|
|
id,
|
|
statements: purged,
|
|
by: req.user.username,
|
|
})
|
|
|
|
return res.json({ id, removed, purged, restartRequired: true })
|
|
} catch (err) {
|
|
log.error('uninstall module', err)
|
|
return res.status(500).json({ message: err.message || 'Internal Server Error' })
|
|
}
|
|
}
|
|
|
|
// POST /admin/modules/:id/purge — drop a still-installed module's data.
|
|
//
|
|
// Refuses unless the module is already disabled, and that guard is the point:
|
|
// dropping the tables under a module that is still serving requests leaves it
|
|
// answering out of a world that no longer exists. Disabling first is one click
|
|
// and makes the destructive step happen against something that has stopped.
|
|
async function purge(req, res) {
|
|
const { id } = req.params
|
|
try {
|
|
const current = await modules.get(id)
|
|
if (!current) return res.status(404).json({ message: 'No such module.' })
|
|
if (current.state !== 'disabled') {
|
|
return res.status(409).json({
|
|
message: 'Disable this module before purging its data, so nothing is serving out of the tables being dropped.',
|
|
})
|
|
}
|
|
|
|
const file = install.purgeFile(id)
|
|
if (!file) {
|
|
return res.status(400).json({ message: 'This module ships no purge.sql, so its data cannot be deleted.' })
|
|
}
|
|
|
|
const statements = await schema.runPurge(file)
|
|
await activity.log({ req, userId: req.user.id, action: 'module.purge', detail: { id, statements } })
|
|
log.warn('module data purged', { id, statements, by: req.user.username })
|
|
|
|
return res.json({ id, purged: statements })
|
|
} catch (err) {
|
|
log.error('purge module', err)
|
|
return res.status(500).json({ message: err.message || 'Internal Server Error' })
|
|
}
|
|
}
|
|
|
|
// PUT /admin/modules/sources — the host allowlist.
|
|
async function setSources(req, res) {
|
|
const hosts = install.parseHosts(req.body.hosts)
|
|
const bad = hosts.find((h) => !HOSTNAME.test(h))
|
|
if (bad) return res.status(400).json({ message: `"${bad}" is not a valid hostname.` })
|
|
|
|
try {
|
|
const before = await allowedHosts()
|
|
await settings.set(HOSTS_KEY, hosts.join(','), req.user.id)
|
|
await activity.log({
|
|
req,
|
|
userId: req.user.id,
|
|
action: 'module.sources',
|
|
detail: { before, after: hosts },
|
|
})
|
|
log.warn('module source allowlist changed', { before, after: hosts, by: req.user.username })
|
|
return res.json({ sourceHosts: hosts })
|
|
} catch (err) {
|
|
log.error('set module sources', err)
|
|
return res.status(500).json({ message: 'Internal Server Error' })
|
|
}
|
|
}
|
|
|
|
// POST /admin/modules/restart — restart the server process.
|
|
//
|
|
// Decision 1. Install, uninstall and re-enable all only take effect at boot
|
|
// because §1.12 reads the volume at require time, and §2.4 promises recovery
|
|
// "with no shell access to the box" — which a banner saying "please restart your
|
|
// container" does not deliver.
|
|
//
|
|
// It reaches server.js's existing SIGTERM handler rather than doing the work
|
|
// itself: that handler stops the modules, the workers and the listeners in the
|
|
// right order and closes the pool and the log file before exiting 0, and going
|
|
// through it means there is exactly one graceful-shutdown path that this route
|
|
// cannot drift from.
|
|
//
|
|
// It gets there by EMITTING the event, not by signalling the process, and that
|
|
// is not a detail. `process.kill(process.pid, 'SIGTERM')` is what this did
|
|
// first, and it works on Linux — but **Windows has no POSIX signals, and Node
|
|
// documents SIGTERM there as unconditional termination of the target process**.
|
|
// So on a Windows host the restart killed the server outright: no module
|
|
// `onShutdown`, no pool close, no log flush. Verified by running it — the
|
|
// process was gone and the shutdown handler had logged nothing.
|
|
//
|
|
// `process.on('SIGTERM', …)` is an ordinary EventEmitter listener, so
|
|
// `process.emit('SIGTERM')` invokes exactly the same handler on every platform
|
|
// without involving the OS at all. Deployment is Linux containers and would
|
|
// never have shown this; development is not.
|
|
//
|
|
// What brings the process BACK is the supervisor, not this. The shipped
|
|
// docker-compose.yml declares `restart: unless-stopped` on `app`, which restarts
|
|
// on a clean exit as well as a crash. A bare `npm start` does not come back, and
|
|
// the screen says so before it asks.
|
|
function restart(req, res) {
|
|
log.warn('restart requested from the admin panel', { by: req.user.username })
|
|
|
|
// Logged and answered first. Once the signal is raised the response has no
|
|
// listener left to flush through, so the operator would be told nothing.
|
|
res.status(202).json({ restarting: true })
|
|
|
|
activity
|
|
.log({ req, userId: req.user.id, action: 'module.restart', detail: {} })
|
|
.catch((err) => log.error('failed to record the restart in the audit log', err))
|
|
.finally(() => {
|
|
// A beat, so the 202 is on the wire. `unref` so this timer is not itself
|
|
// something keeping the process alive.
|
|
setTimeout(() => process.emit('SIGTERM'), 250).unref()
|
|
})
|
|
}
|
|
|
|
module.exports = { list, create, enable, disable, remove, purge, setSources, restart, HOSTS_KEY }
|