feat(events): the runner (Phase 2)
`utils/eventRunner.js`, the eighth poller, wired into server.js beside engagementWorker. Its tick reclaims stale leases, sweeps occurrences past their grace window into `missed`, advances each due run through its phases, and drains that phase's steps in `seq` order. The three core actions from Phase 1 get real bodies, so a published event started from the existing run route now announces, waits and completes on its own. No routes are added: a runner has no surface, and the live controls stay Phase 3's. Four things the org lead settled (2026-09-02): a parked step is `running` with a NULL lease; `await: 'human'` and `holdFor` are ordinary success-envelope members rather than special cases keyed on an action id; a run whose concurrency key is held stays `scheduled` and lets its grace window decide; and `n` in §L's `retry(n)` is a runner constant. Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -92,4 +92,201 @@ const statusCounts = async (runId) => {
|
||||
return Object.fromEntries(rows.map((r) => [r.status, Number(r.n)]))
|
||||
}
|
||||
|
||||
module.exports = { listForRun, getById, materialisePhase, statusCounts, idempotencyKey }
|
||||
// ── Phase 2: draining a step ───────────────────────────────────────────────
|
||||
//
|
||||
// The CAS claim, the lease, the attempt counter and the terminal writes. Phase 1
|
||||
// left all of it out rather than stubbing it, and this is where it lands.
|
||||
//
|
||||
// **Two rules govern everything below, and both were paid for once already.**
|
||||
//
|
||||
// 1. `attempts` is incremented by the CLAIM and by nothing else, and no recovery
|
||||
// path ever resets it. Engagement Phase 14's defect was a stale-row sweep that
|
||||
// returned rows to their start state: the attempt ceiling became unreachable,
|
||||
// so the row cycled forever, never reached a terminal status, and was
|
||||
// therefore never eligible for any retention sweep.
|
||||
// 2. A PARKED step is `running` with a NULL lease, and the reclaim only ever
|
||||
// touches a lease that is non-NULL and expired (the org lead's answer,
|
||||
// 2026-09-02). That is what lets a GM cue wait for a human overnight without a
|
||||
// sweep re-dispatching the instruction every fifteen minutes.
|
||||
|
||||
// A run's steps in authored order, for the phase the run is currently in.
|
||||
const listForPhase = async (runId, phase) =>
|
||||
(
|
||||
await query(
|
||||
'SELECT * FROM event_run_steps WHERE run_id = ? AND phase = ? ORDER BY seq, id',
|
||||
[runId, phase],
|
||||
)
|
||||
).map(hydrate)
|
||||
|
||||
/**
|
||||
* The next step of a phase that the runner may work on, or null.
|
||||
*
|
||||
* **Steps within a phase are strictly serial.** This returns the lowest-`seq`
|
||||
* step that is not terminal, and the runner does nothing with step N+1 until N
|
||||
* has finished — which is the only reading under which `core.wait` means anything
|
||||
* at all, and the only one under which a cue can gate what follows it.
|
||||
*
|
||||
* A parked or running step is returned too, so the caller can see that the phase
|
||||
* is occupied rather than concluding it is finished.
|
||||
*/
|
||||
const nextOpenStep = async (runId, phase) => {
|
||||
const [row] = await query(
|
||||
`SELECT * FROM event_run_steps
|
||||
WHERE run_id = ? AND phase = ?
|
||||
AND status IN ('pending','running')
|
||||
ORDER BY seq, id LIMIT 1`,
|
||||
[runId, phase],
|
||||
)
|
||||
return hydrate(row) || null
|
||||
}
|
||||
|
||||
/**
|
||||
* Take ownership of one pending step: the CAS `pending -> running`, plus a lease.
|
||||
*
|
||||
* `due_at` is honoured here rather than in the caller's filter so that the whole
|
||||
* decision — is it mine, is it due — is one statement the database arbitrates. A
|
||||
* NULL `due_at` is due now, which is what materialisation writes for every step
|
||||
* that is not sitting behind a `core.wait`.
|
||||
*/
|
||||
async function claim(id, owner, leaseUntil, now) {
|
||||
const result = await query(
|
||||
`UPDATE event_run_steps
|
||||
SET status = 'running', attempts = attempts + 1, claimed_by = ?, claim_expires_at = ?,
|
||||
started_at = COALESCE(started_at, NOW())
|
||||
WHERE id = ? AND status = 'pending' AND (due_at IS NULL OR due_at <= ?)`,
|
||||
[owner, leaseUntil, id, now],
|
||||
)
|
||||
return Number(result?.affectedRows || 0) === 1
|
||||
}
|
||||
|
||||
/**
|
||||
* Park a claimed step: it stays `running`, and its lease goes NULL.
|
||||
*
|
||||
* This is the whole mechanism behind `core.cue`. The step is genuinely in flight
|
||||
* — an instruction has been posted and nothing else in the phase may proceed —
|
||||
* but no process is holding it, so the reclaim must not take it back. A NULL
|
||||
* lease says exactly that, and `reclaimStale` below is written to agree.
|
||||
*/
|
||||
const park = (id, note) =>
|
||||
query(
|
||||
`UPDATE event_run_steps SET claim_expires_at = NULL, last_error = ?
|
||||
WHERE id = ? AND status = 'running'`,
|
||||
[note ? String(note).slice(0, 500) : null, id],
|
||||
)
|
||||
|
||||
/**
|
||||
* Release a claimed step back to `pending` for a later attempt.
|
||||
*
|
||||
* `attempts` is untouched — it was already incremented by the claim, which is the
|
||||
* only place that may. Backoff is flat rather than exponential for the reason the
|
||||
* outbox's is: `due_at` is also the event's own clock, and a doubling backoff
|
||||
* pushes a step arbitrarily far past the moment the event was about.
|
||||
*/
|
||||
const reschedule = (id, dueAt, error) =>
|
||||
query(
|
||||
`UPDATE event_run_steps
|
||||
SET status = 'pending', due_at = ?, claimed_by = NULL, claim_expires_at = NULL, last_error = ?
|
||||
WHERE id = ? AND status = 'running'`,
|
||||
[dueAt, error ? String(error).slice(0, 500) : null, id],
|
||||
)
|
||||
|
||||
/** A terminal outcome for one step: done, failed, skipped, refused or cancelled. */
|
||||
const finish = (id, status, error) =>
|
||||
query(
|
||||
`UPDATE event_run_steps
|
||||
SET status = ?, last_error = ?, finished_at = NOW(),
|
||||
claimed_by = NULL, claim_expires_at = NULL
|
||||
WHERE id = ? AND status = 'running'`,
|
||||
[status, error ? String(error).slice(0, 500) : null, id],
|
||||
)
|
||||
|
||||
/**
|
||||
* Delay the next not-yet-started step of a phase — what `core.wait` actually does.
|
||||
*
|
||||
* The wait step itself completes normally; the pause is the NEXT step's `due_at`,
|
||||
* owned by the runner. A `perform()` that slept would hold its claim for the
|
||||
* duration and turn a five-minute pause into a five-minute lease, which is the
|
||||
* one shape this must not have.
|
||||
*
|
||||
* Guarded on `status = 'pending'` and on the current `due_at` being sooner, so a
|
||||
* re-dispatch of a wait whose ack was lost cannot push the following step further
|
||||
* out a second time.
|
||||
*/
|
||||
const holdNext = async (runId, phase, afterSeq, dueAt) => {
|
||||
const result = await query(
|
||||
`UPDATE event_run_steps
|
||||
SET due_at = ?
|
||||
WHERE run_id = ? AND phase = ? AND seq > ? AND status = 'pending'
|
||||
AND (due_at IS NULL OR due_at < ?)
|
||||
ORDER BY seq LIMIT 1`,
|
||||
[dueAt, runId, phase, afterSeq, dueAt],
|
||||
)
|
||||
return Number(result?.affectedRows || 0) === 1
|
||||
}
|
||||
|
||||
/**
|
||||
* Recover steps whose claim outlived the process that took it.
|
||||
*
|
||||
* **`attempts` is not reset and the lease being NULL is not staleness.** The
|
||||
* first is Engagement Phase 14's rule; the second is what makes a parked cue
|
||||
* survive. A step that has already burned its attempts leaves `running` as
|
||||
* `failed` rather than being handed back, and in that order — a reclaim that ran
|
||||
* first would return it to `pending` and it would be retried forever.
|
||||
*/
|
||||
const reclaimStale = async (now, maxAttempts = 0) => {
|
||||
let failed = 0
|
||||
if (Number(maxAttempts) > 0) {
|
||||
const gaveUp = await query(
|
||||
`UPDATE event_run_steps
|
||||
SET status = 'failed', last_error = 'gave up after repeated interruptions',
|
||||
finished_at = NOW(), claimed_by = NULL, claim_expires_at = NULL
|
||||
WHERE status = 'running'
|
||||
AND claim_expires_at IS NOT NULL AND claim_expires_at < ?
|
||||
AND attempts >= ?`,
|
||||
[now, Math.floor(maxAttempts)],
|
||||
)
|
||||
failed = Number(gaveUp?.affectedRows || 0)
|
||||
}
|
||||
const reclaimed = await query(
|
||||
`UPDATE event_run_steps
|
||||
SET status = 'pending', claimed_by = NULL, claim_expires_at = NULL
|
||||
WHERE status = 'running' AND claim_expires_at IS NOT NULL AND claim_expires_at < ?`,
|
||||
[now],
|
||||
)
|
||||
return { failed, reclaimed: Number(reclaimed?.affectedRows || 0) }
|
||||
}
|
||||
|
||||
/**
|
||||
* Cancel every step of a run that has not started (Phase 3's cancel, and the
|
||||
* abort_run disposition).
|
||||
*
|
||||
* A `running` step is deliberately left alone, parked or not: nothing can recall
|
||||
* a command already sent, and a second writer on that row would race the process
|
||||
* that owns it (§L).
|
||||
*/
|
||||
const cancelPending = async (runId) => {
|
||||
const result = await query(
|
||||
`UPDATE event_run_steps
|
||||
SET status = 'cancelled', finished_at = NOW()
|
||||
WHERE run_id = ? AND status = 'pending'`,
|
||||
[runId],
|
||||
)
|
||||
return Number(result?.affectedRows || 0)
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
listForRun,
|
||||
listForPhase,
|
||||
getById,
|
||||
materialisePhase,
|
||||
statusCounts,
|
||||
idempotencyKey,
|
||||
nextOpenStep,
|
||||
claim,
|
||||
park,
|
||||
reschedule,
|
||||
finish,
|
||||
holdNext,
|
||||
reclaimStale,
|
||||
cancelPending,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user