// ── The live run controls ────────────────────────────────────────────────── // // EVENTS.md §I ("live controls that are honest"), §K and §L. Six of them: pause, // resume and cancel act on a run; confirm, skip and retry act on one step. They // arrive in Phase 3 because Phase 2 is what gave them something to act on — a // run that announces, waits and completes on its own is exactly the run that // needs no control, and a run that paused on a failed world write is the one // that does. // // **Two of §I's six run-level controls are deliberately not here.** // `advance` — force a phase forward — has no honest meaning yet: a phase today // advances when its steps go terminal, and the per-step skip already does that // one step at a time. Phase 5 is what gives a phase an `advance` CONDITION, and // that is the first moment "force it anyway" means something an operator could // predict. `cleanup` needs Phase 8's resource ledger; there is nothing to // revert, so cancel takes `{ reason }` and gains `cleanup` when there is // something for it to do. Both are absent rather than inert, which is the // posture Phase 1 set and Phase 2 kept. // // **Every control is guarded on the status it may act from, and the guard is a // WHERE clause rather than a read-then-write.** A run console rendered thirty // seconds ago describes a run that has since moved — the runner ticks every // fifteen — so a control that checked in JavaScript and then wrote would race // the tick it exists to interrupt. `transition()` and the four step statements // are all compare-and-set, and a `false` from one of them is reported as a 409 // naming the status the run is actually in. // // **Who may press them is `admin` + `moderator` (§K, §N2), and it is the widest // gate in this feature on purpose.** Starting a run commits the deployment to // everything the definition contains, unattended — that wants the narrowest gate // there is. Stopping one is incident response at 2am, and it wants the widest. const runsDb = require('./eventRuns.db') const stepsDb = require('./eventRunSteps.db') const logDb = require('./eventRunLog.db') const MAX_REASON = 500 const clean = (raw) => { const text = typeof raw === 'string' ? raw.trim() : '' return text ? text.slice(0, MAX_REASON) : null } const conflict = (message) => ({ ok: false, status: 409, errors: [message] }) /** The run, or a 404 shaped the way every other model here shapes one. */ async function loadRun(runId) { const run = await runsDb.getById(runId) return run || null } /** * A step of THIS run, or null. * * Scoped to the run rather than fetched by id alone: the step id arrives from a * URL under a run id, and a control that acted on a step belonging to a * different run would be a real one — the console's step ids are not secret and * the two paths would otherwise never be compared. */ async function loadStep(runId, stepId) { const step = await stepsDb.getById(stepId) if (!step || Number(step.run_id) !== Number(runId)) return null return step } // ── Run-level ───────────────────────────────────────────────────────────── /** * Pause a run in flight. * * `starting` and `running` only — §K's "live control of a run **in flight**". A * `scheduled` run has not begun, and the thing to do with an occurrence that * should not happen is cancel it: pausing one would leave a run that is neither * going to start nor visibly abandoned, and resuming it after its grace window * had passed would produce a `missed` from a button labelled resume. * * The claim is cleared with the transition. A tick may be working the run at * this exact moment; it will find its guarded writes returning zero rows and * hand back a lease it no longer holds, both of which are no-ops. What it will * NOT do is dispatch the rest of its batch — `advanceRun` re-reads the status * between steps precisely so this control means what it says. */ async function pause(runId, { reason } = {}, userId = null) { const run = await loadRun(runId) if (!run) return { ok: false, status: 404, errors: ['no such run'] } if (run.status === 'paused') return conflict('this run is already paused') const note = clean(reason) if (!(await runsDb.transition(run.id, ['starting', 'running'], 'paused', { clearClaim: true }))) { return conflict(`a ${run.status} run cannot be paused`) } await logDb.write({ runId: run.id, kind: 'run.status', phase: run.current_phase, detail: { from: run.status, to: 'paused', control: 'pause', by: userId, reason: note }, }) return { ok: true, run: await runsDb.getById(run.id) } } /** * Resume a paused run. * * Where it goes back to is derived rather than remembered: `current_phase` is * set by the transition into `running` and by nothing else, so a paused run that * has one was running and a paused run that has none never got past `starting`. * Both statuses are in `findDue`, so the next tick picks the run up either way, * and there is no fourth column recording what a run was paused *from* — a * column that could disagree with the run's own history. * * **`last_error` is cleared and `health` is not.** The error is what the pause * was about and an operator has just dealt with it; leaving it on the banner * would have a healthy run permanently accused of a failure that is in the log * where it belongs. Health is a different claim — that this run has already had * trouble — and it stays true no matter who pressed resume. */ async function resume(runId, options = {}, userId = null) { const run = await loadRun(runId) if (!run) return { ok: false, status: 404, errors: ['no such run'] } if (run.status !== 'paused') return conflict(`a ${run.status} run is not paused`) const to = run.current_phase ? 'running' : 'starting' if (!(await runsDb.transition(run.id, 'paused', to, { error: null }))) { return conflict('this run stopped being paused') } await logDb.write({ runId: run.id, kind: 'run.status', phase: run.current_phase, detail: { from: 'paused', to, control: 'resume', by: userId }, }) return { ok: true, run: await runsDb.getById(run.id) } } /** * Cancel a run. * * Legal from every non-terminal status including `scheduled`, because "this * event is not happening" is a decision an operator makes before it starts as * often as during it. * * `cancelOpen` then closes out the steps that will never run — the pending ones * and any parked cue. A step with a LIVE lease is left exactly where it is: * something is dispatching it, nothing can recall a command already sent (§L), * and a second writer on that row would race the process that owns it. It * finishes into a cancelled run, which is honest. */ async function cancel(runId, { reason } = {}, userId = null) { const run = await loadRun(runId) if (!run) return { ok: false, status: 404, errors: ['no such run'] } if (runsDb.TERMINAL.includes(run.status)) return conflict(`this run is already ${run.status}`) const note = clean(reason) const from = ['scheduled', 'starting', 'running', 'paused', 'ending'] if (!(await runsDb.transition(run.id, from, 'cancelled', { error: note || 'cancelled by staff' }))) { return conflict('this run is no longer cancellable') } const closed = await stepsDb.cancelOpen(run.id) await logDb.write({ runId: run.id, kind: 'run.status', phase: run.current_phase, detail: { from: run.status, to: 'cancelled', control: 'cancel', by: userId, reason: note, cancelledSteps: closed }, }) return { ok: true, run: await runsDb.getById(run.id), cancelledSteps: closed } } // ── Step-level ──────────────────────────────────────────────────────────── /** * Confirm a parked step — the GM cue's other half. * * `core.cue` posts an instruction and parks: the step stays `running` with a * NULL lease, genuinely in flight with nothing holding it, so no sweep takes it * back and a cue posted on Friday is still waiting on Monday. This is what ends * it, and it is the control that makes the whole system useful before any module * automates anything — a GM does the target-driven part in-client and says so * here. * * The outcome is `done`, not `skipped`: a person saying they did the thing is * the step having succeeded. The note is what they did, and it is kept. */ async function confirmStep(runId, stepId, { note } = {}, userId = null) { const run = await loadRun(runId) if (!run) return { ok: false, status: 404, errors: ['no such run'] } const step = await loadStep(runId, stepId) if (!step) return { ok: false, status: 404, errors: ['no such step on this run'] } const text = clean(note) if (!(await stepsDb.confirmParked(step.id, text))) { return conflict(`this step is ${step.status} and is not waiting on anyone`) } await logDb.write({ runId: run.id, stepId: step.id, kind: 'step.status', phase: step.phase, detail: { to: 'done', action: step.action_id, control: 'confirm', by: userId, note: text }, }) return { ok: true, step: await stepsDb.getById(step.id) } } /** * Skip a step: one that has not started, or a parked cue nobody is going to do. * * This is what the `skipped` status was reserved for (§L) — which is also why * the three `on_failure` dispositions all write `failed` instead. A status * meaning both "a human decided against this" and "this was attempted three * times and never worked" would make the console's summary line unreadable. * * A `failed` step is not skippable and does not need to be: `nextOpenStep` * already passes over one, so resuming a run carries the phase past it. */ async function skipStep(runId, stepId, { reason } = {}, userId = null) { const run = await loadRun(runId) if (!run) return { ok: false, status: 404, errors: ['no such run'] } if (runsDb.TERMINAL.includes(run.status)) return conflict(`this run is ${run.status}`) const step = await loadStep(runId, stepId) if (!step) return { ok: false, status: 404, errors: ['no such step on this run'] } const note = clean(reason) if (!(await stepsDb.skipByHuman(step.id, note))) { return conflict(`a ${step.status} step cannot be skipped`) } await logDb.write({ runId: run.id, stepId: step.id, kind: 'step.status', phase: step.phase, detail: { to: 'skipped', action: step.action_id, control: 'skip', by: userId, reason: note }, }) return { ok: true, step: await stepsDb.getById(step.id) } } /** * Re-queue the failed step a run is stopped at, and resume the run — one action. * * **The two halves are one control because there is no state in which you would * want half of it.** Retry is legal only from `paused`, and a paused run is * paused *at* this step; re-queueing without resuming would leave the run in * precisely the state it was already in, with a button the operator now has to * find. Splitting them would read as honesty and behave as a trap. * * Two guards, and the second is the one worth explaining. The step must be the * furthest one its phase has reached — `lastStartedSeq` — because a `failed` * step under an `on_failure` of `skip` is one the run has already moved PAST. * `nextOpenStep` selects `pending` and `running` only, so the runner steps over * a failed row; re-queueing an earlier one puts a `pending` step behind the * cursor, where it sits for ever. */ async function retryStep(runId, stepId, options = {}, userId = null) { const run = await loadRun(runId) if (!run) return { ok: false, status: 404, errors: ['no such run'] } if (run.status !== 'paused') { return conflict(`a step can only be retried while its run is paused; this run is ${run.status}`) } const step = await loadStep(runId, stepId) if (!step) return { ok: false, status: 404, errors: ['no such step on this run'] } if (step.status !== 'failed') return conflict(`a ${step.status} step cannot be retried`) if (step.phase !== run.current_phase) { return conflict('this step belongs to a phase the run has already left') } const furthest = await stepsDb.lastStartedSeq(run.id, step.phase) if (furthest === null || Number(furthest) !== Number(step.seq)) { return conflict('the run is not stopped at this step; only the step a phase is stopped at can be retried') } if (!(await stepsDb.requeue(step.id))) return conflict('this step is no longer failed') await logDb.write({ runId: run.id, stepId: step.id, kind: 'step.status', phase: step.phase, detail: { to: 'pending', action: step.action_id, control: 'retry', by: userId, attemptsReset: step.attempts }, }) const resumed = await resume(runId, {}, userId) return { ok: true, step: await stepsDb.getById(step.id), // A resume that did not take is reported rather than swallowed: the step IS // re-queued either way, and an operator told "retried" about a run that is // still paused would be told something false. resumed: Boolean(resumed.ok), run: resumed.run || (await runsDb.getById(run.id)), } } module.exports = { pause, resume, cancel, confirmStep, skipStep, retryStep }