Two new tables — event_action_settings (the deployment switchboard) and event_run_budget (what a run has spent and the most it may) — plus verified_at and verified_by on event_versions. The whole authorisation decision moves behind one function, events/authorize.js: role, enablement, cap, and the shard's own switch named as the layer core deliberately does not duplicate. Three routes, none moved: GET/PUT /admin/events/actions (admin in both directions) and POST /admin/events/:id/verify (admin, editor — a dry run dispatches nothing). Four decisions, settled by the org lead 2026-09-03: - The default-off line falls between inspect and change, not between notify and inspect. Read literally, §K shipped core.wait disabled. The same line is the role floor. - The tightest cap wins where two actions spend one dimension, pinned into the run at creation with the action it came from. - A refusal follows the step's on_failure and takes health to degraded — its own status and its own log kind, because a refusal is not an outage. - The verify gate is enforced for scheduled starts only: a human pressing Start now is the review the gate exists to require. Derived and flagged for review: a dry run fails rather than warns on a disabled action or an over-cap plan, and the unattended path does not re-check the starter's role. +111 tests (1921/1847/73/1 — the one failure pre-existing and environmental), including a 403 walk over the real router and two concurrent spends against one cap on a real MariaDB. The live walk found two defects, both fixed here: the run console route dropped the budget it was handed, and the role refusal used a plural verb over a one-item list. Co-Authored-By: Claude <noreply@anthropic.com> Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01T6t8mrAWhZU5vnyYgZTMtL
1288 lines
54 KiB
JavaScript
1288 lines
54 KiB
JavaScript
// ── The event runner (EVENTS_PLAN.md Phase 2) ──────────────────────────────
|
||
//
|
||
// The phase's shipped claim, first: **a manually started event that broadcasts,
|
||
// waits, and completes.** Then the properties around it that are not behaviour
|
||
// so much as promises — the ones §E and §L make, and the two this codebase has
|
||
// already paid for once:
|
||
//
|
||
// • a reclaim never resets `attempts` (Engagement Phase 14's defect)
|
||
// • a parked GM cue is not stale, however long it waits
|
||
// • no shape a failure can take reads as success (§F)
|
||
// • a step naming an unregistered action fails terminal with the module named
|
||
// and degrades the run — never a silent skip (§L)
|
||
// • the three `on_failure` dispositions do three different things to the RUN
|
||
// • the idempotency key does not vary by attempt (§E)
|
||
//
|
||
// **The three tables are stubbed at the `.db` layer** and the runner's own logic
|
||
// runs for real against them — the shape `engagementEngine.test.js` uses. What a
|
||
// stub cannot prove is the raw SQL whose correctness IS a server contract: the
|
||
// two CAS claims, the lease reclaim's two-statement order, and `holdNext`'s
|
||
// guard. Those run against a real MariaDB in `eventRunnerSql.test.js`, which
|
||
// skips when there is none. A stub reproduces the reading, not the server.
|
||
//
|
||
// Point the DB at a closed port before requiring anything: the registries reach
|
||
// utils/discordAnnounce, which builds the pool at require time.
|
||
process.env.DB_HOST = '127.0.0.1'
|
||
process.env.DB_PORT = '59999'
|
||
|
||
const { test, beforeEach, afterEach, after } = require('node:test')
|
||
const assert = require('node:assert/strict')
|
||
|
||
const registries = require('../src/modules/registries')
|
||
const runner = require('../src/utils/eventRunner')
|
||
const { classify } = require('../src/events/dispatch')
|
||
const runsDb = require('../src/model/events/eventRuns.db')
|
||
const stepsDb = require('../src/model/events/eventRunSteps.db')
|
||
const logDb = require('../src/model/events/eventRunLog.db')
|
||
const versionsDb = require('../src/model/events/eventVersions.db')
|
||
const definitionsDb = require('../src/model/events/eventDefinitions.db')
|
||
const gatesDb = require('../src/model/events/eventPhaseGates.db')
|
||
// Phase 6 put a permission check in front of every dispatch, and it reads two
|
||
// tables. **For the third time in this feature, a leg the stubbing file did not
|
||
// know about is a ten-second ECONNREFUSED that says nothing about the route it
|
||
// was testing** — Phase 4's expansion leg and Phase 5's gate read did the same.
|
||
// The rule the three of them add up to: when the runner or a model gains a leg,
|
||
// every file that stubs the layer under it needs the stub.
|
||
const settingsDb = require('../src/model/events/eventActionSettings.db')
|
||
const budgetDb = require('../src/model/events/eventRunBudget.db')
|
||
const gates = require('../src/events/gates')
|
||
const db = require('../src/utils/db')
|
||
|
||
after(() => db.close())
|
||
|
||
const T0 = new Date('2026-09-02T12:00:00Z')
|
||
const later = (ms) => new Date(T0.getTime() + ms)
|
||
|
||
// ── In-memory stand-ins for the three tables ───────────────────────────────
|
||
|
||
let store
|
||
const originals = {}
|
||
|
||
for (const [name, mod] of [['runsDb', runsDb], ['stepsDb', stepsDb], ['logDb', logDb], ['versionsDb', versionsDb], ['definitionsDb', definitionsDb], ['gatesDb', gatesDb], ['settingsDb', settingsDb], ['budgetDb', budgetDb]]) {
|
||
originals[name] = { mod, fns: { ...mod } }
|
||
}
|
||
|
||
const restoreOriginals = () => {
|
||
for (const { mod, fns } of Object.values(originals)) Object.assign(mod, fns)
|
||
}
|
||
|
||
const TERMINAL_RUN = ['completed', 'cancelled', 'failed', 'missed']
|
||
const clone = (o) => JSON.parse(JSON.stringify(o, (k, v) => v))
|
||
|
||
function installStubs() {
|
||
store = {
|
||
runs: new Map(),
|
||
steps: new Map(),
|
||
log: [],
|
||
versions: new Map(),
|
||
definitions: new Map(),
|
||
gates: new Map(),
|
||
settings: new Map(),
|
||
budget: new Map(),
|
||
nextStepId: 1,
|
||
nextGateId: 1,
|
||
}
|
||
|
||
// Phase 4 put a schedule-expansion leg in front of the tick. This file is
|
||
// about what the runner does with runs that ALREADY exist, so it has nothing
|
||
// to expand — but the leg is a real query, and left unstubbed every `tick()`
|
||
// here would reach for the dead-port pool and wait on it. Answering with an
|
||
// empty list is what keeps this file measuring the runner rather than a
|
||
// connection timeout.
|
||
Object.assign(definitionsDb, { findSchedulable: async () => [] })
|
||
|
||
// Snapshots, not live references. A SQL SELECT hands back a copy, and the
|
||
// runner reads `step.attempts` as the value BEFORE its own claim incremented
|
||
// it — returning references here would make the retry budget off by one in the
|
||
// stub only, which is exactly the class of thing a stub must not invent.
|
||
const snapRun = (r) => ({ ...r })
|
||
const snapStep = (s) => ({ ...s, params: { ...(s.params || {}) } })
|
||
|
||
runsDb.findDue = async (now) =>
|
||
[...store.runs.values()]
|
||
.filter((r) => ['scheduled', 'starting', 'running', 'ending'].includes(r.status) && r.scheduled_for <= now)
|
||
.sort((a, b) => a.scheduled_for - b.scheduled_for || a.id - b.id)
|
||
.map(snapRun)
|
||
|
||
runsDb.findMissed = async (now) =>
|
||
[...store.runs.values()]
|
||
.filter((r) => {
|
||
const grace = store.definitions.get(r.definition_id)?.grace_seconds ?? 900
|
||
return r.status === 'scheduled' && r.scheduled_for.getTime() + grace * 1000 < now.getTime()
|
||
})
|
||
.map(snapRun)
|
||
|
||
runsDb.claimStart = async (id, owner, lease) => {
|
||
const r = store.runs.get(id)
|
||
if (!r || r.status !== 'scheduled') return false
|
||
Object.assign(r, { status: 'starting', claimed_by: owner, claim_expires_at: lease, started_at: r.started_at || T0 })
|
||
return true
|
||
}
|
||
|
||
runsDb.claimTick = async (id, owner, lease, now) => {
|
||
const r = store.runs.get(id)
|
||
if (!r || !['starting', 'running', 'ending'].includes(r.status)) return false
|
||
// No owner-matches escape: a live lease is not re-enterable, not even by the
|
||
// process that took it. The stub agrees with the statement on purpose.
|
||
if (r.claim_expires_at && r.claim_expires_at >= now) return false
|
||
Object.assign(r, { claimed_by: owner, claim_expires_at: lease })
|
||
return true
|
||
}
|
||
|
||
runsDb.releaseClaim = async (id, owner) => {
|
||
const r = store.runs.get(id)
|
||
if (!r || r.claimed_by !== owner) return false
|
||
Object.assign(r, { claimed_by: null, claim_expires_at: null })
|
||
return true
|
||
}
|
||
|
||
runsDb.transition = async (id, from, to, opts = {}) => {
|
||
const r = store.runs.get(id)
|
||
const froms = Array.isArray(from) ? from : [from]
|
||
if (!r || !froms.includes(r.status)) return false
|
||
r.status = to
|
||
if (opts.phase !== undefined) r.current_phase = opts.phase
|
||
if (opts.error !== undefined) r.last_error = opts.error
|
||
if (TERMINAL_RUN.includes(to)) {
|
||
r.ended_at = r.ended_at || T0
|
||
r.claimed_by = null
|
||
r.claim_expires_at = null
|
||
} else if (opts.clearClaim) {
|
||
r.claimed_by = null
|
||
r.claim_expires_at = null
|
||
}
|
||
return true
|
||
}
|
||
|
||
runsDb.statusOf = async (id) => store.runs.get(id)?.status || null
|
||
|
||
// **Escalation only**, which is the statement's own guard rather than a
|
||
// convenience of this stub: `FIELD(health, 'ok','degraded','stalled') < rank`.
|
||
// A stub that let health move backwards would make a `stalled` run quietly
|
||
// become `degraded` again on the next retry, in the tests only.
|
||
const HEALTH_RANK = { ok: 1, degraded: 2, stalled: 3 }
|
||
runsDb.setHealth = async (id, health) => {
|
||
const r = store.runs.get(id)
|
||
if (!r || !HEALTH_RANK[health]) return false
|
||
if (HEALTH_RANK[r.health] >= HEALTH_RANK[health]) return false
|
||
r.health = health
|
||
return true
|
||
}
|
||
|
||
runsDb.concurrencyHolder = async (key, exceptId) => {
|
||
if (!key) return null
|
||
const held = [...store.runs.values()].find(
|
||
(r) => r.concurrency_key === key && r.id !== exceptId && ['starting', 'running', 'paused', 'ending'].includes(r.status),
|
||
)
|
||
return held ? { id: held.id, status: held.status, definition_id: held.definition_id } : null
|
||
}
|
||
|
||
runsDb.reclaimStale = async (now) => {
|
||
let n = 0
|
||
for (const r of store.runs.values()) {
|
||
if (['starting', 'running', 'ending'].includes(r.status) && r.claim_expires_at && r.claim_expires_at < now) {
|
||
r.claimed_by = null
|
||
r.claim_expires_at = null
|
||
n += 1
|
||
}
|
||
}
|
||
return n
|
||
}
|
||
|
||
stepsDb.materialisePhase = async (runId, phase, steps) => {
|
||
steps.forEach((s, i) => {
|
||
// INSERT IGNORE against uq_evstep_slot (run_id, phase, seq).
|
||
const exists = [...store.steps.values()].find((x) => x.run_id === runId && x.phase === phase && x.seq === i)
|
||
if (exists) return
|
||
const id = store.nextStepId++
|
||
store.steps.set(id, {
|
||
id,
|
||
run_id: runId,
|
||
phase,
|
||
seq: i,
|
||
action_id: s.actionId,
|
||
params: s.params || {},
|
||
action_version: s.actionVersion || 1,
|
||
status: 'pending',
|
||
due_at: null,
|
||
attempts: 0,
|
||
on_failure: s.onFailure || 'pause',
|
||
idempotency_key: stepsDb.idempotencyKey(runId, id),
|
||
claimed_by: null,
|
||
claim_expires_at: null,
|
||
last_error: null,
|
||
})
|
||
})
|
||
return [...store.steps.values()].filter((s) => s.run_id === runId).map(snapStep)
|
||
}
|
||
|
||
stepsDb.nextOpenStep = async (runId, phase) => {
|
||
const s = [...store.steps.values()]
|
||
.filter((x) => x.run_id === runId && x.phase === phase && ['pending', 'running'].includes(x.status))
|
||
.sort((a, b) => a.seq - b.seq || a.id - b.id)[0]
|
||
return s ? snapStep(s) : null
|
||
}
|
||
|
||
stepsDb.claim = async (id, owner, lease, now) => {
|
||
const s = store.steps.get(id)
|
||
if (!s || s.status !== 'pending') return false
|
||
if (s.due_at && s.due_at > now) return false
|
||
Object.assign(s, { status: 'running', attempts: s.attempts + 1, claimed_by: owner, claim_expires_at: lease })
|
||
return true
|
||
}
|
||
|
||
stepsDb.park = async (id) => {
|
||
const s = store.steps.get(id)
|
||
if (s && s.status === 'running') s.claim_expires_at = null
|
||
}
|
||
|
||
stepsDb.reschedule = async (id, dueAt, error) => {
|
||
const s = store.steps.get(id)
|
||
if (!s || s.status !== 'running') return
|
||
// `attempts` is untouched: the claim already incremented it, and nothing else
|
||
// may. This is the stub agreeing with the statement, not with the runner.
|
||
Object.assign(s, { status: 'pending', due_at: dueAt, claimed_by: null, claim_expires_at: null, last_error: error })
|
||
}
|
||
|
||
stepsDb.finish = async (id, status, error) => {
|
||
const s = store.steps.get(id)
|
||
if (!s || s.status !== 'running') return
|
||
Object.assign(s, { status, last_error: error, claimed_by: null, claim_expires_at: null, finished_at: T0 })
|
||
}
|
||
|
||
stepsDb.holdNext = async (runId, phase, afterSeq, dueAt) => {
|
||
const s = [...store.steps.values()]
|
||
.filter((x) => x.run_id === runId && x.phase === phase && x.seq > afterSeq && x.status === 'pending')
|
||
.filter((x) => !x.due_at || x.due_at < dueAt)
|
||
.sort((a, b) => a.seq - b.seq)[0]
|
||
if (!s) return false
|
||
s.due_at = dueAt
|
||
return true
|
||
}
|
||
|
||
stepsDb.reclaimStale = async (now, maxAttempts = 0) => {
|
||
let failed = 0
|
||
let reclaimed = 0
|
||
// Give up first, reclaim second — the order the statement uses, and the
|
||
// reason `MAX_ATTEMPTS` is reachable at all.
|
||
for (const s of store.steps.values()) {
|
||
if (s.status === 'running' && s.claim_expires_at && s.claim_expires_at < now && maxAttempts > 0 && s.attempts >= maxAttempts) {
|
||
Object.assign(s, { status: 'failed', last_error: 'gave up after repeated interruptions', claimed_by: null, claim_expires_at: null })
|
||
failed += 1
|
||
}
|
||
}
|
||
for (const s of store.steps.values()) {
|
||
if (s.status === 'running' && s.claim_expires_at && s.claim_expires_at < now) {
|
||
// NOT reset: attempts survives the reclaim.
|
||
Object.assign(s, { status: 'pending', claimed_by: null, claim_expires_at: null })
|
||
reclaimed += 1
|
||
}
|
||
}
|
||
return { failed, reclaimed }
|
||
}
|
||
|
||
stepsDb.cancelPending = async (runId) => {
|
||
let n = 0
|
||
for (const s of store.steps.values()) {
|
||
if (s.run_id === runId && s.status === 'pending') {
|
||
s.status = 'cancelled'
|
||
n += 1
|
||
}
|
||
}
|
||
return n
|
||
}
|
||
|
||
stepsDb.listForRun = async (runId) =>
|
||
[...store.steps.values()].filter((s) => s.run_id === runId).sort((a, b) => a.seq - b.seq).map(snapStep)
|
||
|
||
logDb.write = async (line) => {
|
||
store.log.push(line)
|
||
return true
|
||
}
|
||
logDb.pruneTerminal = async () => 0
|
||
|
||
versionsDb.getById = async (id) => store.versions.get(id) || null
|
||
|
||
// ── The phase gates (Phase 5) ────────────────────────────────────────────
|
||
//
|
||
// `open` is INSERT IGNORE against (run_id, phase), and `count`/`satisfy` both
|
||
// carry `WHERE satisfied_at IS NULL` — the stub reproduces the guards rather
|
||
// than the convenience, because a gate that could be satisfied twice would
|
||
// advance a phase twice and no test here would see it.
|
||
const gateKey = (runId, phase) => `${runId}|${phase}`
|
||
|
||
gatesDb.open = async ({ runId, phase, kind, afterSeconds = null, triggerId = null, conditions = null, needed = 1, now = T0 }) => {
|
||
const key = gateKey(runId, phase)
|
||
if (store.gates.has(key)) return false
|
||
store.gates.set(key, {
|
||
id: store.nextGateId++,
|
||
run_id: runId,
|
||
phase,
|
||
kind,
|
||
after_seconds: afterSeconds,
|
||
trigger_id: triggerId,
|
||
conditions,
|
||
needed,
|
||
tally: 0,
|
||
entered_at: now,
|
||
due_at: kind === 'after' ? new Date(now.getTime() + afterSeconds * 1000) : null,
|
||
last_event: null,
|
||
last_event_at: null,
|
||
satisfied_at: null,
|
||
satisfied_by: null,
|
||
forced_by: null,
|
||
})
|
||
return true
|
||
}
|
||
|
||
gatesDb.forPhase = async (runId, phase) => {
|
||
const g = store.gates.get(gateKey(runId, phase))
|
||
return g ? { ...g } : null
|
||
}
|
||
|
||
gatesDb.byId = async (id) => {
|
||
const g = [...store.gates.values()].find((x) => x.id === id)
|
||
return g ? { ...g } : null
|
||
}
|
||
|
||
gatesDb.listForRun = async (runId) =>
|
||
[...store.gates.values()].filter((g) => g.run_id === runId).map((g) => ({ ...g }))
|
||
|
||
gatesDb.openForTrigger = async (triggerId) =>
|
||
[...store.gates.values()]
|
||
.filter((g) => g.trigger_id === triggerId && !g.satisfied_at)
|
||
.filter((g) => {
|
||
const r = store.runs.get(g.run_id)
|
||
return r && ['running', 'paused'].includes(r.status) && r.current_phase === g.phase
|
||
})
|
||
.map((g) => ({ ...g }))
|
||
|
||
gatesDb.count = async (id, { lastEvent = null, now = T0 } = {}) => {
|
||
const g = [...store.gates.values()].find((x) => x.id === id)
|
||
if (!g || g.satisfied_at) return { counted: false, satisfied: false }
|
||
g.tally += 1
|
||
g.last_event = lastEvent
|
||
g.last_event_at = now
|
||
if (g.tally >= g.needed) {
|
||
g.satisfied_at = now
|
||
g.satisfied_by = 'condition'
|
||
}
|
||
return { counted: true, satisfied: Boolean(g.satisfied_at), tally: g.tally }
|
||
}
|
||
|
||
gatesDb.noteNearMiss = async (id, { lastEvent = null, now = T0 } = {}) => {
|
||
const g = [...store.gates.values()].find((x) => x.id === id)
|
||
if (!g || g.satisfied_at) return false
|
||
g.last_event = lastEvent
|
||
g.last_event_at = now
|
||
return true
|
||
}
|
||
|
||
gatesDb.satisfy = async (id, by, { userId = null, now = T0 } = {}) => {
|
||
const g = [...store.gates.values()].find((x) => x.id === id)
|
||
if (!g || g.satisfied_at) return false
|
||
Object.assign(g, { satisfied_at: now, satisfied_by: by, forced_by: userId })
|
||
return true
|
||
}
|
||
|
||
// ── Phase 6's two tables ──
|
||
//
|
||
// `store.settings` is empty by default, which is not a gap: an empty
|
||
// switchboard is what a fresh deployment HAS, and `authorize.isEnabled` then
|
||
// answers from the risk class. Every test in this file that does not set a
|
||
// switch is therefore exercising the default posture, which is the posture
|
||
// almost every deployment will run under.
|
||
settingsDb.get = async (actionId) => store.settings.get(actionId) || null
|
||
settingsDb.byIds = async (ids) =>
|
||
new Map(
|
||
[...new Set(ids || [])]
|
||
.filter((id) => store.settings.has(id))
|
||
.map((id) => [id, store.settings.get(id)]),
|
||
)
|
||
|
||
budgetDb.seed = async (runId, dimensions) => {
|
||
for (const [dimension, d] of Object.entries(dimensions || {})) {
|
||
const key = `${runId}:${dimension}`
|
||
if (store.budget.has(key)) continue
|
||
store.budget.set(key, { run_id: runId, dimension, consumed: 0, cap: d.cap, effective_from: d.from || null })
|
||
}
|
||
return Object.keys(dimensions || {}).length
|
||
}
|
||
// The conditional increment, read the way the server reads it — the guard is
|
||
// evaluated against the PRE-update value, and a NULL cap is uncapped.
|
||
budgetDb.spend = async (runId, dimension, amount) => {
|
||
if (!(amount > 0)) return true
|
||
const row = store.budget.get(`${runId}:${dimension}`)
|
||
if (!row) return false
|
||
if (row.cap !== null && row.consumed + amount > row.cap) return false
|
||
row.consumed += amount
|
||
return true
|
||
}
|
||
budgetDb.refund = async (runId, dimension, amount) => {
|
||
const row = store.budget.get(`${runId}:${dimension}`)
|
||
if (row && amount > 0) row.consumed = Math.max(row.consumed - amount, 0)
|
||
}
|
||
budgetDb.forRun = async (runId) =>
|
||
[...store.budget.values()].filter((b) => b.run_id === runId).sort((a, b) => a.dimension.localeCompare(b.dimension))
|
||
}
|
||
|
||
// ── Fixtures ───────────────────────────────────────────────────────────────
|
||
|
||
let nextRunId = 1
|
||
|
||
function seedRun(phases, { scheduledFor = T0, graceSeconds = 900, concurrencyKey = null, status = 'scheduled' } = {}) {
|
||
const id = nextRunId++
|
||
store.definitions.set(id, { id, grace_seconds: graceSeconds })
|
||
store.versions.set(id, { id, spec: { schedule: { kind: 'manual' }, phases } })
|
||
store.runs.set(id, {
|
||
id,
|
||
definition_id: id,
|
||
version_id: id,
|
||
scope: '',
|
||
status,
|
||
health: 'ok',
|
||
cleanup_status: 'not_required',
|
||
current_phase: null,
|
||
scheduled_for: scheduledFor,
|
||
concurrency_key: concurrencyKey,
|
||
params: null,
|
||
rehearsal: 0,
|
||
claimed_by: null,
|
||
claim_expires_at: null,
|
||
last_error: null,
|
||
started_at: null,
|
||
ended_at: null,
|
||
})
|
||
// Phase 1's `create()` materialises the FIRST phase at creation rather than at
|
||
// start, so a seeded run has to as well — otherwise every test here would be
|
||
// exercising a shape the admin route cannot produce.
|
||
const first = phases[0]
|
||
if (first) void stepsDb.materialisePhase(id, first.key, first.steps || [])
|
||
return id
|
||
}
|
||
|
||
const step = (actionId, params = {}, onFailure = 'skip') => ({ actionId, params, onFailure, actionVersion: 1 })
|
||
const run = (id) => store.runs.get(id)
|
||
const stepsOf = (id) => [...store.steps.values()].filter((s) => s.run_id === id).sort((a, b) => a.seq - b.seq)
|
||
const kinds = (id) => store.log.filter((l) => l.runId === id).map((l) => l.kind)
|
||
const gateOf = (id, phase) => store.gates.get(`${id}|${phase}`)
|
||
|
||
// A registered test action whose behaviour the test dictates.
|
||
let scripted
|
||
|
||
beforeEach(() => {
|
||
registries._reset()
|
||
installStubs()
|
||
nextRunId = 1
|
||
scripted = {}
|
||
})
|
||
|
||
afterEach(() => {
|
||
restoreOriginals()
|
||
registries._reset()
|
||
})
|
||
|
||
/** Register actions the way a module does, through the real staging area. */
|
||
const register = (entries, owner = 'test') => {
|
||
const api = registries.stage(owner)
|
||
api.registerEventActions(entries)
|
||
registries.apply(api.staged)
|
||
}
|
||
|
||
// ── Registering actions the tests drive ────────────────────────────────────
|
||
//
|
||
// Registered through the real registry rather than by stubbing `eventAction`,
|
||
// because the shape check at registration is part of what the runner relies on:
|
||
// an action that would not register is not one the runner has to survive.
|
||
|
||
const scriptedAction = (id, extra = {}) => ({
|
||
id,
|
||
label: id,
|
||
risk: 'notify',
|
||
reversible: 'none',
|
||
version: 1,
|
||
budgetMs: 1000,
|
||
params: [],
|
||
perform: async (envelope) => {
|
||
;(scripted[id] ||= { calls: [] }).calls.push(envelope)
|
||
const answer = scripted[id].answers?.shift() ?? scripted[id].answer
|
||
if (typeof answer === 'function') return answer(envelope)
|
||
return answer ?? { ok: true }
|
||
},
|
||
...extra,
|
||
})
|
||
|
||
test('a manually started event announces, waits and completes', async () => {
|
||
register([scriptedAction('test.announce'), scriptedAction('test.wait')])
|
||
scripted['test.wait'] = { calls: [], answer: { ok: true, holdFor: 300 } }
|
||
|
||
const id = seedRun([
|
||
{ key: 'main', label: 'Main', steps: [step('test.announce'), step('test.wait'), step('test.announce')] },
|
||
])
|
||
|
||
// Tick one: announce, then wait, then stop against the held third step.
|
||
await runner.tick(T0)
|
||
let s = stepsOf(id)
|
||
assert.equal(s[0].status, 'done')
|
||
assert.equal(s[1].status, 'done')
|
||
assert.equal(s[2].status, 'pending', 'the step after a wait must not run in the same tick')
|
||
assert.equal(s[2].due_at.getTime(), later(300_000).getTime(), 'the wait is the NEXT step due_at')
|
||
assert.equal(run(id).status, 'running')
|
||
assert.equal(run(id).claimed_by, null, 'a run left in flight gives its lease back')
|
||
|
||
// Tick two, still inside the wait: nothing moves.
|
||
await runner.tick(later(120_000))
|
||
assert.equal(stepsOf(id)[2].status, 'pending')
|
||
assert.equal(run(id).status, 'running')
|
||
|
||
// Tick three, past it: the last step runs and the run completes.
|
||
await runner.tick(later(301_000))
|
||
assert.equal(stepsOf(id)[2].status, 'done')
|
||
assert.equal(run(id).status, 'completed')
|
||
assert.equal(run(id).health, 'ok')
|
||
assert.ok(kinds(id).includes('phase.completed'))
|
||
assert.ok(kinds(id).includes('run.status'))
|
||
})
|
||
|
||
test('a run passes through `ending` on its way to completed', async () => {
|
||
register([scriptedAction('test.noop')])
|
||
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('test.noop')] }])
|
||
await runner.tick(T0)
|
||
|
||
const transitions = store.log.filter((l) => l.runId === id && l.kind === 'run.status').map((l) => l.detail.to)
|
||
assert.deepEqual(transitions, ['starting', 'running', 'ending', 'completed'])
|
||
})
|
||
|
||
test('phases run in order and the next one is materialised on entry', async () => {
|
||
register([scriptedAction('test.noop')])
|
||
|
||
const id = seedRun([
|
||
{ key: 'opening', label: 'Opening', steps: [step('test.noop')] },
|
||
{ key: 'closing', label: 'Closing', steps: [step('test.noop'), step('test.noop')] },
|
||
])
|
||
|
||
await runner.tick(T0)
|
||
assert.equal(run(id).status, 'completed')
|
||
assert.deepEqual(stepsOf(id).map((s) => s.phase), ['opening', 'closing', 'closing'])
|
||
assert.ok(stepsOf(id).every((s) => s.status === 'done'))
|
||
})
|
||
|
||
test('a wait as the last step of a phase holds the NEXT phase, rather than meaning nothing', async () => {
|
||
register([scriptedAction('test.noop'), scriptedAction('test.wait')])
|
||
scripted['test.wait'] = { calls: [], answer: { ok: true, holdFor: 300 } }
|
||
|
||
const id = seedRun([
|
||
{ key: 'opening', label: 'Opening', steps: [step('test.noop'), step('test.wait')] },
|
||
{ key: 'closing', label: 'Closing', steps: [step('test.noop')] },
|
||
])
|
||
|
||
await runner.tick(T0)
|
||
const closing = stepsOf(id).filter((x) => x.phase === 'closing')
|
||
assert.equal(closing.length, 1, 'the next phase is materialised')
|
||
assert.equal(closing[0].status, 'pending')
|
||
assert.equal(
|
||
closing[0].due_at.getTime(),
|
||
later(300_000).getTime(),
|
||
'the hold crosses the phase boundary; dropping it would start the next phase at once',
|
||
)
|
||
assert.equal(run(id).status, 'running')
|
||
|
||
await runner.tick(later(301_000))
|
||
assert.equal(run(id).status, 'completed')
|
||
})
|
||
|
||
test('a GM cue parks: the step stays running with no lease, and the reclaim leaves it alone', async () => {
|
||
register([scriptedAction('test.cue'), scriptedAction('test.after')])
|
||
scripted['test.cue'] = { calls: [], answer: { ok: true, await: 'human' } }
|
||
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('test.cue'), step('test.after')] }])
|
||
|
||
await runner.tick(T0)
|
||
const cue = stepsOf(id)[0]
|
||
assert.equal(cue.status, 'running')
|
||
assert.equal(cue.claim_expires_at, null, 'a parked step carries no lease')
|
||
assert.equal(stepsOf(id)[1].status, 'pending', 'nothing after a cue proceeds')
|
||
assert.ok(kinds(id).includes('step.parked'))
|
||
|
||
// A week later the reclaim has still not touched it, and the cue has been
|
||
// dispatched exactly once. This is the whole point of a NULL lease.
|
||
await runner.tick(later(7 * 24 * 60 * 60 * 1000))
|
||
assert.equal(stepsOf(id)[0].status, 'running')
|
||
assert.equal(stepsOf(id)[0].attempts, 1)
|
||
assert.equal(scripted['test.cue'].calls.length, 1)
|
||
assert.equal(run(id).status, 'running')
|
||
})
|
||
|
||
test('a reclaim returns a stale step without resetting attempts', async () => {
|
||
register([scriptedAction('test.slow')])
|
||
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('test.slow')] }])
|
||
|
||
// Simulate a process that claimed the step and died: running, lease in the past.
|
||
await runner.tick(T0)
|
||
const s = stepsOf(id)[0]
|
||
Object.assign(store.steps.get(s.id), { status: 'running', attempts: 2, claim_expires_at: later(-1000) })
|
||
|
||
await stepsDb.reclaimStale(T0, runner.MAX_ATTEMPTS)
|
||
assert.equal(store.steps.get(s.id).status, 'pending')
|
||
assert.equal(store.steps.get(s.id).attempts, 2, 'Engagement Phase 14: a reclaim must never reset attempts')
|
||
})
|
||
|
||
test('a step whose attempts are spent leaves `running` as failed rather than being handed back', async () => {
|
||
register([scriptedAction('test.slow')])
|
||
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('test.slow')] }])
|
||
await runner.tick(T0)
|
||
const s = stepsOf(id)[0]
|
||
Object.assign(store.steps.get(s.id), { status: 'running', attempts: runner.MAX_ATTEMPTS, claim_expires_at: later(-1000) })
|
||
|
||
const { failed, reclaimed } = await stepsDb.reclaimStale(T0, runner.MAX_ATTEMPTS)
|
||
assert.equal(failed, 1)
|
||
assert.equal(reclaimed, 0, 'a row that gave up must not also be reclaimed, or it retries forever')
|
||
assert.equal(store.steps.get(s.id).status, 'failed')
|
||
})
|
||
|
||
test('a transient failure retries on a flat backoff with the same idempotency key, then applies on_failure', async () => {
|
||
register([scriptedAction('test.flaky')])
|
||
scripted['test.flaky'] = { calls: [], answer: { ok: false, retry: true, error: 'relay is down' } }
|
||
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('test.flaky', {}, 'skip')] }])
|
||
const key = () => stepsOf(id)[0].idempotency_key
|
||
|
||
await runner.tick(T0)
|
||
const firstKey = key()
|
||
assert.equal(stepsOf(id)[0].status, 'pending')
|
||
assert.equal(stepsOf(id)[0].attempts, 1)
|
||
assert.equal(stepsOf(id)[0].due_at.getTime(), later(runner.RETRY_MS).getTime())
|
||
assert.equal(run(id).health, 'degraded', 'degraded from the first retry, not from the eventual failure')
|
||
|
||
await runner.tick(later(runner.RETRY_MS))
|
||
assert.equal(stepsOf(id)[0].attempts, 2)
|
||
|
||
await runner.tick(later(2 * runner.RETRY_MS))
|
||
assert.equal(stepsOf(id)[0].attempts, runner.MAX_ATTEMPTS)
|
||
assert.equal(stepsOf(id)[0].status, 'failed', 'all three dispositions write the step failed')
|
||
assert.equal(run(id).status, 'completed', 'on_failure: skip lets the run finish')
|
||
assert.equal(run(id).health, 'degraded')
|
||
|
||
assert.equal(key(), firstKey, 'the idempotency key does not vary by attempt')
|
||
assert.equal(new Set(scripted['test.flaky'].calls.map((c) => c.idempotencyKey)).size, 1)
|
||
})
|
||
|
||
test('on_failure: pause stops the run and the tick never picks it up again', async () => {
|
||
register([scriptedAction('test.bad'), scriptedAction('test.after')])
|
||
scripted['test.bad'] = { calls: [], answer: { ok: false, retry: false, error: 'the world is half changed' } }
|
||
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('test.bad', {}, 'pause'), step('test.after')] }])
|
||
|
||
await runner.tick(T0)
|
||
assert.equal(run(id).status, 'paused')
|
||
assert.equal(stepsOf(id)[0].status, 'failed')
|
||
assert.equal(stepsOf(id)[1].status, 'pending', 'a paused run leaves its remaining steps alone')
|
||
|
||
await runner.tick(later(60_000))
|
||
assert.equal(run(id).status, 'paused', 'only Phase 3 resume moves a paused run')
|
||
assert.equal(scripted['test.after']?.calls?.length ?? 0, 0)
|
||
})
|
||
|
||
test('on_failure: abort_run fails the run and cancels what has not started', async () => {
|
||
register([scriptedAction('test.bad'), scriptedAction('test.after')])
|
||
scripted['test.bad'] = { calls: [], answer: { ok: false, retry: false, error: 'no' } }
|
||
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('test.bad', {}, 'abort_run'), step('test.after')] }])
|
||
|
||
await runner.tick(T0)
|
||
assert.equal(run(id).status, 'failed')
|
||
assert.equal(stepsOf(id)[0].status, 'failed')
|
||
assert.equal(stepsOf(id)[1].status, 'cancelled')
|
||
})
|
||
|
||
test('a step naming an unregistered action fails terminal with the module named, and degrades the run', async () => {
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('gone.verb', {}, 'skip')] }])
|
||
|
||
await runner.tick(T0)
|
||
const s = stepsOf(id)[0]
|
||
assert.equal(s.status, 'failed', 'never a silent skip (§L)')
|
||
assert.equal(s.attempts, 1, 'a dormant action is terminal, so it is not retried')
|
||
assert.match(s.last_error, /gone\.verb/)
|
||
assert.equal(run(id).health, 'degraded')
|
||
})
|
||
|
||
test('a held concurrency key holds the run at scheduled, and logs the reason once', async () => {
|
||
register([scriptedAction('test.noop'), scriptedAction('test.cue')])
|
||
scripted['test.cue'] = { calls: [], answer: { ok: true, await: 'human' } }
|
||
|
||
// The holder is parked on a cue, which is what keeps it genuinely in flight. A
|
||
// holder with no steps would complete itself on this same tick — correct
|
||
// behaviour, and a fixture that proved nothing.
|
||
const holder = seedRun([{ key: 'main', label: 'Main', steps: [step('test.cue')] }], {
|
||
concurrencyKey: 'invasion:Yew',
|
||
})
|
||
const waiting = seedRun([{ key: 'main', label: 'Main', steps: [step('test.noop')] }], { concurrencyKey: 'invasion:Yew' })
|
||
|
||
await runner.tick(T0)
|
||
assert.equal(run(waiting).status, 'scheduled')
|
||
assert.match(run(waiting).last_error, new RegExp(`run ${holder}`))
|
||
assert.equal(kinds(waiting).filter((k) => k === 'run.blocked').length, 1)
|
||
|
||
// Still held, and still one line: a line per tick would bury the one that matters.
|
||
await runner.tick(later(15_000))
|
||
assert.equal(kinds(waiting).filter((k) => k === 'run.blocked').length, 1)
|
||
|
||
// The holder finishes, and the next tick starts the run that was waiting.
|
||
store.runs.get(holder).status = 'completed'
|
||
store.runs.get(holder).claim_expires_at = null
|
||
await runner.tick(later(30_000))
|
||
assert.equal(run(waiting).status, 'completed')
|
||
})
|
||
|
||
test('an occurrence past its own grace window is missed, never a late silent start', async () => {
|
||
register([scriptedAction('test.noop')])
|
||
|
||
const late = seedRun([{ key: 'main', label: 'Main', steps: [step('test.noop')] }], { graceSeconds: 600 })
|
||
const inside = seedRun([{ key: 'main', label: 'Main', steps: [step('test.noop')] }], { graceSeconds: 3600 })
|
||
|
||
// Both are due; the process has been down for half an hour.
|
||
await runner.tick(later(30 * 60 * 1000))
|
||
|
||
assert.equal(run(late).status, 'missed')
|
||
assert.equal(stepsOf(late)[0].status, 'cancelled')
|
||
assert.equal(run(inside).status, 'completed', 'inside its window it starts late and says so')
|
||
})
|
||
|
||
test('a run in `ending` when the process died is completed by the next tick', async () => {
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [] }], { status: 'ending' })
|
||
store.runs.get(id).current_phase = 'main'
|
||
|
||
await runner.tick(T0)
|
||
assert.equal(run(id).status, 'completed')
|
||
})
|
||
|
||
test('a live lease is not re-enterable, not even by the process that took it', async () => {
|
||
register([scriptedAction('test.cue')])
|
||
scripted['test.cue'] = { calls: [], answer: { ok: true, await: 'human' } }
|
||
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('test.cue')] }])
|
||
await runner.tick(T0)
|
||
|
||
// Put a live lease back on the run, as an overrunning tick would have.
|
||
Object.assign(store.runs.get(id), { claimed_by: runner.OWNER, claim_expires_at: later(60_000) })
|
||
const taken = await runsDb.claimTick(id, runner.OWNER, later(120_000), T0)
|
||
assert.equal(taken, false, 'the CAS is what protects a tick that overran into the next one')
|
||
})
|
||
|
||
// ── §F: no shape a failure can take reads as success ───────────────────────
|
||
|
||
test('classify: every failure shape is a failure', () => {
|
||
assert.equal(classify(undefined, 'a').outcome, 'retry')
|
||
assert.equal(classify(null, 'a').outcome, 'retry')
|
||
assert.equal(classify('ok', 'a').outcome, 'retry')
|
||
assert.equal(classify(['ok'], 'a').outcome, 'retry')
|
||
assert.equal(classify({}, 'a').outcome, 'retry', 'a missing ok is not a success')
|
||
assert.equal(classify({ ok: 'yes' }, 'a').outcome, 'retry', 'ok must be true, not truthy')
|
||
assert.equal(classify({ ok: false }, 'a').outcome, 'retry')
|
||
assert.equal(classify({ ok: false, retry: false }, 'a').outcome, 'terminal')
|
||
assert.equal(classify({ __timedOut: true, error: 'slow' }, 'a').outcome, 'retry')
|
||
})
|
||
|
||
test('classify: the two success shapes that are not "finished"', () => {
|
||
assert.equal(classify({ ok: true }, 'a').outcome, 'done')
|
||
assert.equal(classify({ ok: true }, 'a').holdSeconds, 0)
|
||
assert.equal(classify({ ok: true, await: 'human' }, 'a').outcome, 'parked')
|
||
assert.equal(classify({ ok: true, holdFor: 90 }, 'a').holdSeconds, 90)
|
||
assert.equal(classify({ ok: true, holdFor: '90' }, 'a').holdSeconds, 90)
|
||
assert.equal(classify({ ok: true, holdFor: -1 }, 'a').outcome, 'terminal', 'a bad holdFor is not a silent zero')
|
||
assert.equal(classify({ ok: true, holdFor: 'soon' }, 'a').outcome, 'terminal')
|
||
assert.ok(classify({ ok: true, holdFor: 1e12 }, 'a').holdSeconds <= 7 * 24 * 60 * 60, 'holdFor is bounded')
|
||
})
|
||
|
||
test('an action that throws is a transient failure, not a crashed tick', async () => {
|
||
register([
|
||
scriptedAction('test.thrower', {
|
||
perform: async () => {
|
||
throw new Error('boom')
|
||
},
|
||
}),
|
||
])
|
||
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('test.thrower', {}, 'skip')] }])
|
||
await runner.tick(T0)
|
||
|
||
assert.equal(stepsOf(id)[0].status, 'pending')
|
||
assert.match(stepsOf(id)[0].last_error, /boom/)
|
||
assert.equal(run(id).status, 'running', 'one bad action does not stop the deployment')
|
||
})
|
||
|
||
test('an action that never answers is cut off at its declared budget', async () => {
|
||
register([
|
||
scriptedAction('test.hang', { budgetMs: 30, perform: () => new Promise(() => {}) }),
|
||
])
|
||
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('test.hang', {}, 'skip')] }])
|
||
await runner.tick(T0)
|
||
|
||
assert.equal(stepsOf(id)[0].status, 'pending', 'a timeout is transient')
|
||
assert.match(stepsOf(id)[0].last_error, /budget/)
|
||
})
|
||
|
||
test('the dispatch envelope carries what §F says it carries', async () => {
|
||
register([scriptedAction('test.echo')])
|
||
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('test.echo', {})] }])
|
||
store.runs.get(id).scope = 'atlantic'
|
||
await runner.tick(T0)
|
||
|
||
const [envelope] = scripted['test.echo'].calls
|
||
assert.equal(envelope.runId, id)
|
||
assert.equal(envelope.scope, 'atlantic')
|
||
assert.equal(envelope.verify, false)
|
||
assert.equal(typeof envelope.idempotencyKey, 'string')
|
||
assert.equal(envelope.idempotencyKey.length, 40)
|
||
assert.deepEqual(Object.keys(envelope).sort(), ['actor', 'idempotencyKey', 'params', 'runId', 'scope', 'stepId', 'verify'])
|
||
})
|
||
|
||
test('a run paused mid-batch stops there rather than draining the rest of the phase', async () => {
|
||
// The whole value of a pause is that it takes effect NOW. `advanceRun` drains
|
||
// up to STEPS_PER_TICK steps from one run inside a single tick, so a status
|
||
// re-read only at the top of the tick would answer a pause by dispatching
|
||
// another two dozen steps. Written by pausing from inside an action's own
|
||
// `perform`, which is the only moment that race is reproducible.
|
||
register([
|
||
scriptedAction('test.pauser', {
|
||
perform: async ({ runId }) => {
|
||
await runsDb.transition(runId, ['starting', 'running'], 'paused', { clearClaim: true })
|
||
return { ok: true }
|
||
},
|
||
}),
|
||
scriptedAction('test.after'),
|
||
])
|
||
|
||
const id = seedRun([
|
||
{
|
||
key: 'main',
|
||
label: 'Main',
|
||
steps: [step('test.pauser'), step('test.after'), step('test.after')],
|
||
},
|
||
])
|
||
|
||
await runner.tick(T0)
|
||
|
||
assert.equal(run(id).status, 'paused')
|
||
const [first, second, third] = stepsOf(id)
|
||
assert.equal(first.status, 'done', 'the step that was already dispatched finishes')
|
||
assert.equal(second.status, 'pending', 'nothing after it ran')
|
||
assert.equal(third.status, 'pending')
|
||
assert.equal(scripted['test.after'], undefined, 'the later action was never called')
|
||
})
|
||
|
||
test('a resumed run picks up from the step it stopped at', async () => {
|
||
register([scriptedAction('test.a'), scriptedAction('test.b')])
|
||
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [step('test.a'), step('test.b')] }])
|
||
await runner.tick(T0)
|
||
assert.equal(run(id).status, 'completed')
|
||
|
||
// And the mirror of it: a run parked at `paused` is not picked up at all, which
|
||
// is what `findDue`'s omission of the status buys.
|
||
const other = seedRun([{ key: 'main', label: 'Main', steps: [step('test.a')] }], { status: 'paused' })
|
||
await runner.tick(T0)
|
||
assert.equal(stepsOf(other)[0].status, 'pending', 'a paused run is not swept')
|
||
})
|
||
|
||
|
||
// ── Phase 5: advance conditions ────────────────────────────────────────────
|
||
//
|
||
// The claim: a phase with a gate waits for it, a phase without one does not
|
||
// change at all, and NOTHING advances a phase but its condition or a human.
|
||
|
||
test('an `after` gate holds a phase whose steps are all done, and releases it on its deadline', async () => {
|
||
register([scriptedAction('test.a'), scriptedAction('test.b')])
|
||
|
||
const id = seedRun([
|
||
{ key: 'one', label: 'One', steps: [step('test.a')], advance: { after: '30m' } },
|
||
{ key: 'two', label: 'Two', steps: [step('test.b')] },
|
||
])
|
||
|
||
await runner.tick(T0)
|
||
assert.equal(run(id).status, 'running')
|
||
assert.equal(run(id).current_phase, 'one', 'the phase did not advance on its steps alone')
|
||
assert.equal(stepsOf(id)[0].status, 'done', 'but its step ran')
|
||
assert.equal(scripted['test.b'], undefined, 'and the next phase has not started')
|
||
|
||
// A tick a minute later changes nothing: the deadline is computed once, at
|
||
// entry, and is not re-derived from a `now` that has moved.
|
||
await runner.tick(later(60_000))
|
||
assert.equal(run(id).current_phase, 'one')
|
||
|
||
await runner.tick(later(30 * 60_000 + 1))
|
||
assert.equal(run(id).status, 'completed')
|
||
assert.equal(gateOf(id, 'one').satisfied_by, 'elapsed')
|
||
const line = store.log.find((l) => l.runId === id && l.kind === 'phase.advanced')
|
||
assert.equal(line.detail.because, 'elapsed')
|
||
assert.ok(line.detail.waitedSeconds >= 1800, 'the log says how long it actually waited')
|
||
})
|
||
|
||
test('a gate is in ADDITION to the steps, never instead of them', async () => {
|
||
// The gate opens immediately; the phase must still not advance, because a
|
||
// phase whose steps are still running is not finished. The near miss is a
|
||
// reading under which a boss that spawned early would carry a run past an
|
||
// announcement that had not been made.
|
||
register([scriptedAction('test.slow', { perform: async () => ({ ok: true, await: 'human' }) }), scriptedAction('test.b')])
|
||
|
||
const id = seedRun([
|
||
{ key: 'one', label: 'One', steps: [step('test.slow')], advance: { after: '1s' } },
|
||
{ key: 'two', label: 'Two', steps: [step('test.b')] },
|
||
])
|
||
|
||
await runner.tick(T0)
|
||
await runner.tick(later(60_000))
|
||
|
||
assert.equal(run(id).current_phase, 'one')
|
||
assert.equal(gateOf(id, 'one').satisfied_at, null, 'the gate was never even consulted')
|
||
assert.equal(scripted['test.b'], undefined)
|
||
})
|
||
|
||
test('a phase with no gate advances exactly as it did before Phase 5', async () => {
|
||
register([scriptedAction('test.a'), scriptedAction('test.b')])
|
||
const id = seedRun([
|
||
{ key: 'one', label: 'One', steps: [step('test.a')] },
|
||
{ key: 'two', label: 'Two', steps: [step('test.b')] },
|
||
])
|
||
await runner.tick(T0)
|
||
assert.equal(run(id).status, 'completed')
|
||
assert.equal(store.gates.size, 0, 'and no gate row was written for it')
|
||
})
|
||
|
||
test('an `on` gate counts firings from the emit path, and needs all of them', async () => {
|
||
register([scriptedAction('test.a'), scriptedAction('test.b')])
|
||
|
||
const id = seedRun([
|
||
{
|
||
key: 'one',
|
||
label: 'One',
|
||
steps: [step('test.a')],
|
||
advance: { on: 'test.trigger', where: { variable: 'region', cmp: 'eq', value: 'Yew' }, count: 2 },
|
||
},
|
||
{ key: 'two', label: 'Two', steps: [step('test.b')] },
|
||
])
|
||
|
||
await runner.tick(T0)
|
||
assert.equal(run(id).current_phase, 'one')
|
||
|
||
const fire = (region) =>
|
||
gates.observe({ triggerId: 'test.trigger', occurredAt: T0.toISOString(), subject: null, data: { region } })
|
||
|
||
// The near miss: it is recorded, it is logged, and it does not count.
|
||
await fire('Britain')
|
||
assert.equal(gateOf(id, 'one').tally, 0, 'a firing the condition rejects is not progress')
|
||
assert.equal(gateOf(id, 'one').last_event.matched, false, 'but it IS recorded — "wrong region" and "nothing happened" are different answers')
|
||
|
||
await fire('Yew')
|
||
assert.equal(gateOf(id, 'one').tally, 1)
|
||
assert.equal(gateOf(id, 'one').satisfied_at, null, 'one of two is not two')
|
||
|
||
await runner.tick(later(1000))
|
||
assert.equal(run(id).current_phase, 'one', 'and the runner agrees')
|
||
|
||
await fire('Yew')
|
||
assert.equal(gateOf(id, 'one').satisfied_by, 'condition')
|
||
|
||
await runner.tick(later(2000))
|
||
assert.equal(run(id).status, 'completed')
|
||
|
||
const evaluated = store.log.filter((l) => l.runId === id && l.kind === 'condition.evaluated')
|
||
assert.equal(evaluated.length, 3, 'every firing is logged, matched or not')
|
||
assert.deepEqual(evaluated.map((l) => l.detail.matched), [false, true, true])
|
||
assert.deepEqual(evaluated.map((l) => l.detail.seen), [0, 1, 2], 'and the tally logged is the one the row holds')
|
||
})
|
||
|
||
test('a phase forced by a human is not logged as advanced twice', async () => {
|
||
// `phase.advanced` is written by whoever made the DECISION. The advance
|
||
// control writes it with the actor and the reason; a second line from the
|
||
// tick that then acts on the satisfied gate made the console show the phase
|
||
// advancing twice, the less informative one last. Found in the live walk.
|
||
register([scriptedAction('test.a')])
|
||
const id = seedRun([
|
||
{ key: 'one', label: 'One', steps: [step('test.a')], advance: { on: 'test.trigger', count: 1 } },
|
||
{ key: 'two', label: 'Two', steps: [] },
|
||
])
|
||
await runner.tick(T0)
|
||
|
||
// What the control does: satisfy the gate as `forced` and stop.
|
||
await gatesDb.satisfy(gateOf(id, 'one').id, 'forced', { userId: 7, now: later(1000) })
|
||
await runner.tick(later(2000))
|
||
|
||
const advanced = store.log.filter((l) => l.runId === id && l.kind === 'phase.advanced')
|
||
assert.equal(advanced.length, 0, 'the tick writes no line of its own for a decision it did not make')
|
||
assert.equal(run(id).current_phase, 'two', 'and it still advances the phase')
|
||
})
|
||
|
||
test('a firing that arrives after the gate closed does not keep counting', async () => {
|
||
register([scriptedAction('test.a')])
|
||
const id = seedRun([
|
||
{ key: 'one', label: 'One', steps: [step('test.a')], advance: { on: 'test.trigger', count: 1 } },
|
||
{ key: 'two', label: 'Two', steps: [] },
|
||
])
|
||
await runner.tick(T0)
|
||
await gates.observe({ triggerId: 'test.trigger', occurredAt: T0.toISOString(), data: {} })
|
||
assert.equal(gateOf(id, 'one').tally, 1)
|
||
|
||
await gates.observe({ triggerId: 'test.trigger', occurredAt: T0.toISOString(), data: {} })
|
||
assert.equal(gateOf(id, 'one').tally, 1, 'the guard is `WHERE satisfied_at IS NULL`, not a read-then-write')
|
||
})
|
||
|
||
test('an `on` gate that waits past the threshold marks the run stalled, once', async () => {
|
||
register([scriptedAction('test.a')])
|
||
const id = seedRun([
|
||
{ key: 'one', label: 'One', steps: [step('test.a')], advance: { on: 'test.trigger', count: 1 } },
|
||
{ key: 'two', label: 'Two', steps: [] },
|
||
])
|
||
|
||
await runner.tick(T0)
|
||
assert.equal(run(id).health, 'ok', 'a phase that has just begun waiting is not stalled')
|
||
|
||
await runner.tick(later(gates.STALL_MS + 1000))
|
||
assert.equal(run(id).health, 'stalled')
|
||
|
||
await runner.tick(later(gates.STALL_MS + 60_000))
|
||
const health = store.log.filter((l) => l.runId === id && l.kind === 'run.health')
|
||
assert.equal(health.length, 1, 'and it is logged once, not once per tick')
|
||
})
|
||
|
||
test('an `after` gate is never stalled, however long it was authored to wait', async () => {
|
||
register([scriptedAction('test.a')])
|
||
const id = seedRun([
|
||
{ key: 'one', label: 'One', steps: [step('test.a')], advance: { after: '2d' } },
|
||
{ key: 'two', label: 'Two', steps: [] },
|
||
])
|
||
await runner.tick(T0)
|
||
await runner.tick(later(gates.STALL_MS * 3))
|
||
assert.equal(run(id).health, 'ok', 'a phase waiting out the delay it was given is working, not stalled')
|
||
})
|
||
|
||
test('health only ever escalates, so a retry after a stall does not demote it', async () => {
|
||
assert.equal(await runsDb.setHealth(seedRun([{ key: 'main', label: 'Main', steps: [] }]), 'degraded'), true)
|
||
const id = seedRun([{ key: 'main', label: 'Main', steps: [] }])
|
||
assert.equal(await runsDb.setHealth(id, 'stalled'), true)
|
||
assert.equal(await runsDb.setHealth(id, 'degraded'), false, 'a later degradation cannot undo a stall')
|
||
assert.equal(run(id).health, 'stalled')
|
||
assert.equal(await runsDb.setHealth(id, 'ok'), false, 'and nothing returns a run to healthy')
|
||
})
|
||
|
||
test('re-entering a phase does not open a second gate', async () => {
|
||
// The INSERT IGNORE that makes recovery from a died-mid-entry process
|
||
// uneventful — the same property `materialisePhase` has.
|
||
register([scriptedAction('test.a')])
|
||
const id = seedRun([
|
||
{ key: 'one', label: 'One', steps: [step('test.a')], advance: { after: '1h' } },
|
||
{ key: 'two', label: 'Two', steps: [] },
|
||
])
|
||
await runner.tick(T0)
|
||
const entered = gateOf(id, 'one').entered_at
|
||
await runner.tick(later(60_000))
|
||
await runner.tick(later(120_000))
|
||
assert.equal(store.gates.size, 1)
|
||
assert.equal(gateOf(id, 'one').entered_at, entered, 'and the deadline it was given does not move')
|
||
})
|
||
|
||
// ── Phase 6: enablement and caps, in front of the dispatch ─────────────────
|
||
//
|
||
// The runner gained one thing this phase: it asks `mayInvoke` before it asks a
|
||
// module to do anything. What follows is the behaviour that produces, and the
|
||
// three properties that are decisions rather than mechanisms.
|
||
|
||
const seedBudget = (runId, dimension, { consumed = 0, cap = null } = {}) =>
|
||
store.budget.set(`${runId}:${dimension}`, { run_id: runId, dimension, consumed, cap, effective_from: null })
|
||
|
||
const budgetOf = (runId, dimension) => store.budget.get(`${runId}:${dimension}`)
|
||
|
||
const setSwitch = (id, enabled, caps = {}) =>
|
||
store.settings.set(id, { action_id: id, enabled: enabled ? 1 : 0, caps })
|
||
|
||
test('a disabled action is REFUSED, not failed, and the two look different on the record', async () => {
|
||
// Decision 3 (org lead, 2026-09-03): a refusal takes the same disposition a
|
||
// failure takes, and says a different thing. `refused` and `step.refused` are
|
||
// what let an operator reading a stopped run at 2am see at a glance that
|
||
// nothing is broken — the deployment simply does not permit what was asked.
|
||
register([scriptedAction('test.change', { risk: 'change', label: 'Change things' })])
|
||
const id = seedRun([{ key: 'main', steps: [step('test.change')] }])
|
||
|
||
await runner.tick(T0)
|
||
|
||
assert.equal(stepsOf(id)[0].status, 'refused')
|
||
assert.equal(stepsOf(id)[0].last_error, '"Change things" is not enabled on this deployment')
|
||
assert.ok(kinds(id).includes('step.refused'))
|
||
assert.ok(!kinds(id).includes('step.status'), 'a refusal is not a step.status line')
|
||
assert.equal(scripted['test.change'], undefined, 'a refused action is never dispatched')
|
||
})
|
||
|
||
test('a refusal follows the step’s on_failure, exactly as a failure does', async () => {
|
||
// The whole of decision 3. `change` defaults to `pause`, so the run stops where
|
||
// it stands and waits for a human to raise the cap or edit the plan.
|
||
register([scriptedAction('test.change', { risk: 'change' }), scriptedAction('test.after')])
|
||
const id = seedRun([
|
||
{ key: 'main', steps: [step('test.change', {}, 'pause'), step('test.after')] },
|
||
])
|
||
|
||
await runner.tick(T0)
|
||
|
||
assert.equal(run(id).status, 'paused')
|
||
assert.equal(run(id).health, 'degraded')
|
||
assert.equal(stepsOf(id)[1].status, 'pending', 'nothing after a pausing refusal runs')
|
||
})
|
||
|
||
test('a refusal with on_failure skip lets the run carry on, degraded', async () => {
|
||
register([scriptedAction('test.change', { risk: 'change' }), scriptedAction('test.after')])
|
||
const id = seedRun([{ key: 'main', steps: [step('test.change', {}, 'skip'), step('test.after')] }])
|
||
|
||
await runner.tick(T0)
|
||
|
||
assert.equal(stepsOf(id)[0].status, 'refused')
|
||
assert.equal(stepsOf(id)[1].status, 'done')
|
||
assert.equal(run(id).status, 'completed')
|
||
assert.equal(run(id).health, 'degraded')
|
||
})
|
||
|
||
test('an enabled world-changing action runs, because the switch is the whole gate', async () => {
|
||
// The other half of the default-off posture, and the one that proves the switch
|
||
// is read rather than the risk class being a refusal on its own.
|
||
register([scriptedAction('test.change', { risk: 'change' })])
|
||
setSwitch('test.change', true)
|
||
const id = seedRun([{ key: 'main', steps: [step('test.change')] }])
|
||
|
||
await runner.tick(T0)
|
||
|
||
assert.equal(stepsOf(id)[0].status, 'done')
|
||
assert.equal(run(id).status, 'completed')
|
||
})
|
||
|
||
test('a step over its cap is refused with the dimension and the numbers on the log line', async () => {
|
||
// "You asked for 40 and this deployment allows 30" is an authoring error, and
|
||
// it has to arrive as those words rather than as a stack trace.
|
||
register([
|
||
scriptedAction('test.spawn', { risk: 'change', cost: (p) => ({ 'x.creatures': p.count }) }),
|
||
])
|
||
setSwitch('test.spawn', true, { 'x.creatures': 30 })
|
||
const id = seedRun([{ key: 'main', steps: [step('test.spawn', { count: 12 }, 'skip')] }])
|
||
seedBudget(id, 'x.creatures', { consumed: 28, cap: 30 })
|
||
|
||
await runner.tick(T0)
|
||
|
||
assert.equal(stepsOf(id)[0].status, 'refused')
|
||
const line = store.log.find((l) => l.runId === id && l.kind === 'step.refused')
|
||
assert.equal(line.detail.code, 'cap')
|
||
assert.equal(line.detail.dimension, 'x.creatures')
|
||
assert.equal(line.detail.requested, 12)
|
||
assert.equal(line.detail.cap, 30)
|
||
assert.equal(line.detail.consumed, 28)
|
||
assert.equal(budgetOf(id, 'x.creatures').consumed, 28, 'a refused step spends nothing')
|
||
})
|
||
|
||
test('two steps drawing on one cap spend it once each, and the second is refused when it will not fit', async () => {
|
||
// The stub's half of the plan's acceptance criterion. The SERVER's half — two
|
||
// spends arriving genuinely at once — is in `eventRunnerSql.test.js`, because
|
||
// the guard lives in a WHERE and a stub reproduces the reading rather than the
|
||
// server.
|
||
register([scriptedAction('test.spawn', { risk: 'change', cost: (p) => ({ 'x.creatures': p.count }) })])
|
||
setSwitch('test.spawn', true, { 'x.creatures': 30 })
|
||
const id = seedRun([
|
||
{
|
||
key: 'main',
|
||
steps: [
|
||
step('test.spawn', { count: 20 }, 'skip'),
|
||
step('test.spawn', { count: 20 }, 'skip'),
|
||
step('test.spawn', { count: 10 }, 'skip'),
|
||
],
|
||
},
|
||
])
|
||
seedBudget(id, 'x.creatures', { consumed: 0, cap: 30 })
|
||
|
||
await runner.tick(T0)
|
||
|
||
const s = stepsOf(id)
|
||
assert.equal(s[0].status, 'done')
|
||
assert.equal(s[1].status, 'refused', 'the second 20 does not fit under 30')
|
||
assert.equal(s[2].status, 'done', 'and a later step that DOES fit still runs')
|
||
assert.equal(budgetOf(id, 'x.creatures').consumed, 30)
|
||
})
|
||
|
||
test('a retry does not pay the cap twice', async () => {
|
||
// The spend happens on the first attempt only. A retry re-dispatches the same
|
||
// idempotent operation against the same key, and charging a cap for a flaky
|
||
// socket would exhaust a deployment's allowance through unreliability rather
|
||
// than through effect.
|
||
register([scriptedAction('test.spawn', { risk: 'change', cost: () => ({ 'x.creatures': 5 }) })])
|
||
setSwitch('test.spawn', true, { 'x.creatures': 30 })
|
||
scripted['test.spawn'] = { calls: [], answers: [{ ok: false, retry: true, error: 'shard busy' }] }
|
||
|
||
const id = seedRun([{ key: 'main', steps: [step('test.spawn', {}, 'skip')] }])
|
||
seedBudget(id, 'x.creatures', { consumed: 0, cap: 30 })
|
||
|
||
await runner.tick(T0)
|
||
assert.equal(stepsOf(id)[0].status, 'pending')
|
||
assert.equal(budgetOf(id, 'x.creatures').consumed, 5, 'the first attempt spends')
|
||
|
||
// Past the retry backoff: the second attempt succeeds and must not spend again.
|
||
await runner.tick(later(61_000))
|
||
assert.equal(stepsOf(id)[0].status, 'done')
|
||
assert.equal(budgetOf(id, 'x.creatures').consumed, 5, 'the retry must not pay twice')
|
||
})
|
||
|
||
test('a step that spent and then failed for good keeps its spend', async () => {
|
||
// The corollary, and it is deliberate: the attempt may have half-run, and a
|
||
// refund would be core asserting that it did not.
|
||
register([scriptedAction('test.spawn', { risk: 'change', cost: () => ({ 'x.creatures': 5 }) })])
|
||
setSwitch('test.spawn', true, { 'x.creatures': 30 })
|
||
scripted['test.spawn'] = { calls: [], answer: { ok: false, retry: false, error: 'no' } }
|
||
|
||
const id = seedRun([{ key: 'main', steps: [step('test.spawn', {}, 'skip')] }])
|
||
seedBudget(id, 'x.creatures', { consumed: 0, cap: 30 })
|
||
|
||
await runner.tick(T0)
|
||
assert.equal(stepsOf(id)[0].status, 'failed')
|
||
assert.equal(budgetOf(id, 'x.creatures').consumed, 5)
|
||
})
|
||
|
||
test('a step spending a dimension its run has no budget row for is refused', async () => {
|
||
// Fail-closed. A run whose version names a costing action always has that
|
||
// dimension seeded — uncapped ones included, as a row with a NULL cap — so a
|
||
// missing row means the step is spending something its own version never
|
||
// declared.
|
||
register([scriptedAction('test.spawn', { risk: 'change', cost: () => ({ 'x.ghosts': 1 }) })])
|
||
setSwitch('test.spawn', true)
|
||
const id = seedRun([{ key: 'main', steps: [step('test.spawn', {}, 'skip')] }])
|
||
|
||
await runner.tick(T0)
|
||
|
||
assert.equal(stepsOf(id)[0].status, 'refused')
|
||
const line = store.log.find((l) => l.runId === id && l.kind === 'step.refused')
|
||
assert.equal(line.detail.code, 'unbudgeted')
|
||
})
|
||
|
||
test('an uncapped dimension counts without ever refusing', async () => {
|
||
register([scriptedAction('test.spawn', { risk: 'change', cost: () => ({ 'x.creatures': 99 }) })])
|
||
setSwitch('test.spawn', true)
|
||
const id = seedRun([{ key: 'main', steps: [step('test.spawn'), step('test.spawn')] }])
|
||
seedBudget(id, 'x.creatures', { consumed: 0, cap: null })
|
||
|
||
await runner.tick(T0)
|
||
|
||
assert.deepEqual(stepsOf(id).map((s) => s.status), ['done', 'done'])
|
||
assert.equal(budgetOf(id, 'x.creatures').consumed, 198)
|
||
})
|
||
|
||
test('the runner never re-checks the role of whoever started the run', async () => {
|
||
// §K's "a demoted user loses access at once" is about reaching a ROUTE. A run
|
||
// already in flight is deliberately not re-gated against its starter's current
|
||
// role: demoting an admin at midnight must not silently strand every event they
|
||
// started. Cancel is the control for a run that should stop.
|
||
register([scriptedAction('test.burn', { risk: 'irreversible' })])
|
||
setSwitch('test.burn', true)
|
||
const id = seedRun([{ key: 'main', steps: [step('test.burn')] }])
|
||
|
||
await runner.tick(T0)
|
||
|
||
assert.equal(stepsOf(id)[0].status, 'done')
|
||
})
|