Files
website/server/src/model/events/eventRunControls.model.js
wtclaude 7b570c8ea1
All checks were successful
PR Checks / bot-tests (pull_request) Successful in 30s
PR Checks / server-tests (pull_request) Successful in 5m26s
PR Checks / client-build (pull_request) Successful in 8m30s
feat(events): the minimal admin surface (Phase 3)
Three screens, an Events nav group and the six live run controls Phase 1 left
absent on purpose because nothing was in flight. An admin can now author,
publish, start and watch an event that announces things and cues a human; a
moderator can stop one that is going wrong.

Six controls, not eight. `advance` is absent because a phase today advances when
its steps go terminal — the per-step skip already does that — and Phase 5 is what
gives a phase an advance condition. Cancel takes `{ reason }`, not `{ cleanup }`,
until Phase 8's ledger exists. Every control is a compare-and-set on the status it
may act from, so a console rendered thirty seconds ago cannot act on a run that
has moved.

Fixes a defect in the Phase 2 runner: `advanceRun` drained up to
EVENT_STEPS_PER_TICK steps while only checking the run's status at the top of the
tick, so a pause pressed mid-batch did nothing for up to 24 more steps.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01T6t8mrAWhZU5vnyYgZTMtL
2026-09-02 08:39:35 -05:00

297 lines
13 KiB
JavaScript

// ── The live run controls ──────────────────────────────────────────────────
//
// EVENTS.md §I ("live controls that are honest"), §K and §L. Six of them: pause,
// resume and cancel act on a run; confirm, skip and retry act on one step. They
// arrive in Phase 3 because Phase 2 is what gave them something to act on — a
// run that announces, waits and completes on its own is exactly the run that
// needs no control, and a run that paused on a failed world write is the one
// that does.
//
// **Two of §I's six run-level controls are deliberately not here.**
// `advance` — force a phase forward — has no honest meaning yet: a phase today
// advances when its steps go terminal, and the per-step skip already does that
// one step at a time. Phase 5 is what gives a phase an `advance` CONDITION, and
// that is the first moment "force it anyway" means something an operator could
// predict. `cleanup` needs Phase 8's resource ledger; there is nothing to
// revert, so cancel takes `{ reason }` and gains `cleanup` when there is
// something for it to do. Both are absent rather than inert, which is the
// posture Phase 1 set and Phase 2 kept.
//
// **Every control is guarded on the status it may act from, and the guard is a
// WHERE clause rather than a read-then-write.** A run console rendered thirty
// seconds ago describes a run that has since moved — the runner ticks every
// fifteen — so a control that checked in JavaScript and then wrote would race
// the tick it exists to interrupt. `transition()` and the four step statements
// are all compare-and-set, and a `false` from one of them is reported as a 409
// naming the status the run is actually in.
//
// **Who may press them is `admin` + `moderator` (§K, §N2), and it is the widest
// gate in this feature on purpose.** Starting a run commits the deployment to
// everything the definition contains, unattended — that wants the narrowest gate
// there is. Stopping one is incident response at 2am, and it wants the widest.
const runsDb = require('./eventRuns.db')
const stepsDb = require('./eventRunSteps.db')
const logDb = require('./eventRunLog.db')
const MAX_REASON = 500
const clean = (raw) => {
const text = typeof raw === 'string' ? raw.trim() : ''
return text ? text.slice(0, MAX_REASON) : null
}
const conflict = (message) => ({ ok: false, status: 409, errors: [message] })
/** The run, or a 404 shaped the way every other model here shapes one. */
async function loadRun(runId) {
const run = await runsDb.getById(runId)
return run || null
}
/**
* A step of THIS run, or null.
*
* Scoped to the run rather than fetched by id alone: the step id arrives from a
* URL under a run id, and a control that acted on a step belonging to a
* different run would be a real one — the console's step ids are not secret and
* the two paths would otherwise never be compared.
*/
async function loadStep(runId, stepId) {
const step = await stepsDb.getById(stepId)
if (!step || Number(step.run_id) !== Number(runId)) return null
return step
}
// ── Run-level ─────────────────────────────────────────────────────────────
/**
* Pause a run in flight.
*
* `starting` and `running` only — §K's "live control of a run **in flight**". A
* `scheduled` run has not begun, and the thing to do with an occurrence that
* should not happen is cancel it: pausing one would leave a run that is neither
* going to start nor visibly abandoned, and resuming it after its grace window
* had passed would produce a `missed` from a button labelled resume.
*
* The claim is cleared with the transition. A tick may be working the run at
* this exact moment; it will find its guarded writes returning zero rows and
* hand back a lease it no longer holds, both of which are no-ops. What it will
* NOT do is dispatch the rest of its batch — `advanceRun` re-reads the status
* between steps precisely so this control means what it says.
*/
async function pause(runId, { reason } = {}, userId = null) {
const run = await loadRun(runId)
if (!run) return { ok: false, status: 404, errors: ['no such run'] }
if (run.status === 'paused') return conflict('this run is already paused')
const note = clean(reason)
if (!(await runsDb.transition(run.id, ['starting', 'running'], 'paused', { clearClaim: true }))) {
return conflict(`a ${run.status} run cannot be paused`)
}
await logDb.write({
runId: run.id,
kind: 'run.status',
phase: run.current_phase,
detail: { from: run.status, to: 'paused', control: 'pause', by: userId, reason: note },
})
return { ok: true, run: await runsDb.getById(run.id) }
}
/**
* Resume a paused run.
*
* Where it goes back to is derived rather than remembered: `current_phase` is
* set by the transition into `running` and by nothing else, so a paused run that
* has one was running and a paused run that has none never got past `starting`.
* Both statuses are in `findDue`, so the next tick picks the run up either way,
* and there is no fourth column recording what a run was paused *from* — a
* column that could disagree with the run's own history.
*
* **`last_error` is cleared and `health` is not.** The error is what the pause
* was about and an operator has just dealt with it; leaving it on the banner
* would have a healthy run permanently accused of a failure that is in the log
* where it belongs. Health is a different claim — that this run has already had
* trouble — and it stays true no matter who pressed resume.
*/
async function resume(runId, options = {}, userId = null) {
const run = await loadRun(runId)
if (!run) return { ok: false, status: 404, errors: ['no such run'] }
if (run.status !== 'paused') return conflict(`a ${run.status} run is not paused`)
const to = run.current_phase ? 'running' : 'starting'
if (!(await runsDb.transition(run.id, 'paused', to, { error: null }))) {
return conflict('this run stopped being paused')
}
await logDb.write({
runId: run.id,
kind: 'run.status',
phase: run.current_phase,
detail: { from: 'paused', to, control: 'resume', by: userId },
})
return { ok: true, run: await runsDb.getById(run.id) }
}
/**
* Cancel a run.
*
* Legal from every non-terminal status including `scheduled`, because "this
* event is not happening" is a decision an operator makes before it starts as
* often as during it.
*
* `cancelOpen` then closes out the steps that will never run — the pending ones
* and any parked cue. A step with a LIVE lease is left exactly where it is:
* something is dispatching it, nothing can recall a command already sent (§L),
* and a second writer on that row would race the process that owns it. It
* finishes into a cancelled run, which is honest.
*/
async function cancel(runId, { reason } = {}, userId = null) {
const run = await loadRun(runId)
if (!run) return { ok: false, status: 404, errors: ['no such run'] }
if (runsDb.TERMINAL.includes(run.status)) return conflict(`this run is already ${run.status}`)
const note = clean(reason)
const from = ['scheduled', 'starting', 'running', 'paused', 'ending']
if (!(await runsDb.transition(run.id, from, 'cancelled', { error: note || 'cancelled by staff' }))) {
return conflict('this run is no longer cancellable')
}
const closed = await stepsDb.cancelOpen(run.id)
await logDb.write({
runId: run.id,
kind: 'run.status',
phase: run.current_phase,
detail: { from: run.status, to: 'cancelled', control: 'cancel', by: userId, reason: note, cancelledSteps: closed },
})
return { ok: true, run: await runsDb.getById(run.id), cancelledSteps: closed }
}
// ── Step-level ────────────────────────────────────────────────────────────
/**
* Confirm a parked step — the GM cue's other half.
*
* `core.cue` posts an instruction and parks: the step stays `running` with a
* NULL lease, genuinely in flight with nothing holding it, so no sweep takes it
* back and a cue posted on Friday is still waiting on Monday. This is what ends
* it, and it is the control that makes the whole system useful before any module
* automates anything — a GM does the target-driven part in-client and says so
* here.
*
* The outcome is `done`, not `skipped`: a person saying they did the thing is
* the step having succeeded. The note is what they did, and it is kept.
*/
async function confirmStep(runId, stepId, { note } = {}, userId = null) {
const run = await loadRun(runId)
if (!run) return { ok: false, status: 404, errors: ['no such run'] }
const step = await loadStep(runId, stepId)
if (!step) return { ok: false, status: 404, errors: ['no such step on this run'] }
const text = clean(note)
if (!(await stepsDb.confirmParked(step.id, text))) {
return conflict(`this step is ${step.status} and is not waiting on anyone`)
}
await logDb.write({
runId: run.id,
stepId: step.id,
kind: 'step.status',
phase: step.phase,
detail: { to: 'done', action: step.action_id, control: 'confirm', by: userId, note: text },
})
return { ok: true, step: await stepsDb.getById(step.id) }
}
/**
* Skip a step: one that has not started, or a parked cue nobody is going to do.
*
* This is what the `skipped` status was reserved for (§L) — which is also why
* the three `on_failure` dispositions all write `failed` instead. A status
* meaning both "a human decided against this" and "this was attempted three
* times and never worked" would make the console's summary line unreadable.
*
* A `failed` step is not skippable and does not need to be: `nextOpenStep`
* already passes over one, so resuming a run carries the phase past it.
*/
async function skipStep(runId, stepId, { reason } = {}, userId = null) {
const run = await loadRun(runId)
if (!run) return { ok: false, status: 404, errors: ['no such run'] }
if (runsDb.TERMINAL.includes(run.status)) return conflict(`this run is ${run.status}`)
const step = await loadStep(runId, stepId)
if (!step) return { ok: false, status: 404, errors: ['no such step on this run'] }
const note = clean(reason)
if (!(await stepsDb.skipByHuman(step.id, note))) {
return conflict(`a ${step.status} step cannot be skipped`)
}
await logDb.write({
runId: run.id,
stepId: step.id,
kind: 'step.status',
phase: step.phase,
detail: { to: 'skipped', action: step.action_id, control: 'skip', by: userId, reason: note },
})
return { ok: true, step: await stepsDb.getById(step.id) }
}
/**
* Re-queue the failed step a run is stopped at, and resume the run — one action.
*
* **The two halves are one control because there is no state in which you would
* want half of it.** Retry is legal only from `paused`, and a paused run is
* paused *at* this step; re-queueing without resuming would leave the run in
* precisely the state it was already in, with a button the operator now has to
* find. Splitting them would read as honesty and behave as a trap.
*
* Two guards, and the second is the one worth explaining. The step must be the
* furthest one its phase has reached — `lastStartedSeq` — because a `failed`
* step under an `on_failure` of `skip` is one the run has already moved PAST.
* `nextOpenStep` selects `pending` and `running` only, so the runner steps over
* a failed row; re-queueing an earlier one puts a `pending` step behind the
* cursor, where it sits for ever.
*/
async function retryStep(runId, stepId, options = {}, userId = null) {
const run = await loadRun(runId)
if (!run) return { ok: false, status: 404, errors: ['no such run'] }
if (run.status !== 'paused') {
return conflict(`a step can only be retried while its run is paused; this run is ${run.status}`)
}
const step = await loadStep(runId, stepId)
if (!step) return { ok: false, status: 404, errors: ['no such step on this run'] }
if (step.status !== 'failed') return conflict(`a ${step.status} step cannot be retried`)
if (step.phase !== run.current_phase) {
return conflict('this step belongs to a phase the run has already left')
}
const furthest = await stepsDb.lastStartedSeq(run.id, step.phase)
if (furthest === null || Number(furthest) !== Number(step.seq)) {
return conflict('the run is not stopped at this step; only the step a phase is stopped at can be retried')
}
if (!(await stepsDb.requeue(step.id))) return conflict('this step is no longer failed')
await logDb.write({
runId: run.id,
stepId: step.id,
kind: 'step.status',
phase: step.phase,
detail: { to: 'pending', action: step.action_id, control: 'retry', by: userId, attemptsReset: step.attempts },
})
const resumed = await resume(runId, {}, userId)
return {
ok: true,
step: await stepsDb.getById(step.id),
// A resume that did not take is reported rather than swallowed: the step IS
// re-queued either way, and an operator told "retried" about a run that is
// still paused would be told something false.
resumed: Boolean(resumed.ok),
run: resumed.run || (await runsDb.getById(run.id)),
}
}
module.exports = { pause, resume, cancel, confirmStep, skipStep, retryStep }