From 15f80c704227d29a635d15307ec68cbe2546f326 Mon Sep 17 00:00:00 2001 From: Wolfy-J Date: Tue, 22 Sep 2026 22:58:56 -0400 Subject: [PATCH 01/10] fix(runner): single durable activation ownership record owned by the orchestrator --- src/_index.yaml | 2 + src/consts.lua | 1 + .../10_add_activation_ownership.lua | 68 +++ src/migrations/_index.yaml | 14 + src/persist/activation_repo.lua | 210 +++++-- src/persist/activation_repo_test.lua | 222 ++++++- src/persist/commit.lua | 72 ++- src/persist/commit_test.lua | 59 ++ src/persist/ops.lua | 52 +- src/persist/ops_test.lua | 134 ++++ src/runner/_index.yaml | 12 +- src/runner/orchestrator.lua | 281 ++++++--- .../orchestrator_completion_flush_test.lua | 3 +- .../orchestrator_process_event_test.lua | 4 + src/runner/orchestrator_test.lua | 234 ++++++- src/runner/overseer.lua | 539 ++++++---------- src/runner/overseer_state.lua | 578 +++--------------- src/runner/overseer_state_test.lua | 402 +++--------- src/runner/overseer_test.lua | 278 +++++++-- src/runner/runtime_epoch.lua | 15 + src/runner/workflow_state.lua | 27 +- src/runner/workflow_state_test.lua | 58 +- src/security/_index.yaml | 18 + test/_index.yaml | 41 ++ test/restricted_caller_test.lua | 98 +++ 25 files changed, 1979 insertions(+), 1443 deletions(-) create mode 100644 src/migrations/10_add_activation_ownership.lua create mode 100644 src/runner/runtime_epoch.lua create mode 100644 test/restricted_caller_test.lua diff --git a/src/_index.yaml b/src/_index.yaml index 8a248ab..77de79b 100644 --- a/src/_index.yaml +++ b/src/_index.yaml @@ -268,6 +268,8 @@ entries: path: ".meta.target_db" - entry: userspace.dataflow.migrations:09_add_iteration_uniqueness_guard path: ".meta.target_db" + - entry: userspace.dataflow.migrations:10_add_activation_ownership + path: ".meta.target_db" - entry: userspace.dataflow.runner:overseer.service path: ".lifecycle.depends_on +=" - entry: userspace.dataflow.env:retention_db diff --git a/src/consts.lua b/src/consts.lua index acebdbd..b518ca3 100644 --- a/src/consts.lua +++ b/src/consts.lua @@ -4,6 +4,7 @@ local consts = {} consts.HOST_ID = "app:processes" consts.APP_DB = "app:db" consts.ORCHESTRATOR = "userspace.dataflow.runner:orchestrator" +consts.RUNTIME_EPOCH_READER = "userspace.dataflow.runner:runtime_epoch" -- Topic constants for actor state transitions consts.TOPIC = { diff --git a/src/migrations/10_add_activation_ownership.lua b/src/migrations/10_add_activation_ownership.lua new file mode 100644 index 0000000..e4cc834 --- /dev/null +++ b/src/migrations/10_add_activation_ownership.lua @@ -0,0 +1,68 @@ +local function execute_or_error(db, query) + local success, err = db:execute(query) + if err then error(err) end + return success +end + +-- The orchestrator-owned ownership record. owner_token identifies one +-- orchestrator incarnation, owner_pid its process, owner_epoch (added by +-- migration 08) the runtime it runs in, and owner_phase is running or +-- released; a row without a phase has never been owned. +local SQLITE_COLUMNS = { + { name = "owner_token", type = "TEXT" }, + { name = "owner_pid", type = "TEXT" }, + { name = "owner_phase", type = "TEXT" }, +} + +local POSTGRES_COLUMNS = { + { name = "owner_token", type = "TEXT" }, + { name = "owner_pid", type = "TEXT" }, + { name = "owner_phase", type = "TEXT" }, +} + +local function sqlite_columns(db) + local columns, columns_err = db:query("PRAGMA table_info(dataflow_activations)") + if columns_err then error(columns_err) end + local present = {} + for _, column in ipairs(columns or {}) do present[column.name] = true end + return present +end + +return require("migration").define(function() + migration("Record the orchestrator that owns an activation", function() + database("postgres", function() + up(function(db) + for _, column in ipairs(POSTGRES_COLUMNS) do + execute_or_error(db, "ALTER TABLE dataflow_activations ADD COLUMN IF NOT EXISTS " .. + column.name .. " " .. column.type) + end + end) + down(function(db) + for _, column in ipairs(POSTGRES_COLUMNS) do + execute_or_error(db, "ALTER TABLE dataflow_activations DROP COLUMN IF EXISTS " .. + column.name) + end + end) + end) + + database("sqlite", function() + up(function(db) + local present = sqlite_columns(db) + for _, column in ipairs(SQLITE_COLUMNS) do + if not present[column.name] then + execute_or_error(db, "ALTER TABLE dataflow_activations ADD COLUMN " .. + column.name .. " " .. column.type) + end + end + end) + down(function(db) + local present = sqlite_columns(db) + for _, column in ipairs(SQLITE_COLUMNS) do + if present[column.name] then + execute_or_error(db, "ALTER TABLE dataflow_activations DROP COLUMN " .. column.name) + end + end + end) + end) + end) +end) diff --git a/src/migrations/_index.yaml b/src/migrations/_index.yaml index 1fc6aed..2832352 100644 --- a/src/migrations/_index.yaml +++ b/src/migrations/_index.yaml @@ -198,3 +198,17 @@ entries: imports: migration: wippy.migration:migration method: migrate + + - name: 10_add_activation_ownership + kind: function.lua + meta: + type: migration + tags: [dataflows, activation, durability] + description: Record the orchestrator incarnation that owns an activation and whether it is running or released + depends_on: [ns:wippy.migration] + target_db: app:db + timestamp: "2026-09-22T12:00:00Z" + source: file://10_add_activation_ownership.lua + imports: + migration: wippy.migration:migration + method: migrate diff --git a/src/persist/activation_repo.lua b/src/persist/activation_repo.lua index 626d822..b338fc4 100644 --- a/src/persist/activation_repo.lua +++ b/src/persist/activation_repo.lua @@ -11,6 +11,12 @@ local TERMINAL_STATUS = { [consts.STATUS.TERMINATED] = true, } +local PHASE = { + RUNNING = "running", + RELEASED = "released", +} +activation_repo.OWNER_PHASE = PHASE + local TERMINAL_VALUES = { consts.STATUS.COMPLETED_SUCCESS, consts.STATUS.COMPLETED_FAILURE, @@ -150,6 +156,9 @@ local function normalize_row(row: any) generation = tonumber(row.generation), desired_active = row.desired_active == true or tonumber(row.desired_active) == 1, owner_epoch = row.owner_epoch and tostring(row.owner_epoch) or nil, + owner_token = row.owner_token and tostring(row.owner_token) or nil, + owner_pid = row.owner_pid and tostring(row.owner_pid) or nil, + owner_phase = row.owner_phase and tostring(row.owner_phase) or nil, launch_args = launch_args, requested_at = tostring(row.requested_at), updated_at = tostring(row.updated_at), @@ -189,6 +198,7 @@ end local function get_tx(tx, dataflow_id) local rows, query_err = tx_query(tx, [[ SELECT dataflow_id, generation, desired_active, owner_epoch, + owner_token, owner_pid, owner_phase, launch_args, requested_at, updated_at FROM dataflow_activations WHERE dataflow_id = ? LIMIT 1 ]], { dataflow_id }) @@ -244,7 +254,6 @@ local function advance_activation_tx(tx, dataflow_id, launch_args: any, now_valu ON CONFLICT(dataflow_id) DO UPDATE SET generation = dataflow_activations.generation + 1, desired_active = excluded.desired_active, - owner_epoch = NULL, launch_args = %s, requested_at = excluded.requested_at, updated_at = excluded.updated_at @@ -417,14 +426,56 @@ function activation_repo.activate_due_tx(tx, dataflow_id, wake_key, now_value) return { changed = false, terminal = false, promoted = false, due = false }, nil end -function activation_repo.release_if_generation_tx(tx, dataflow_id, generation, now_value) +-- An ownership fence compares the owner columns with the values a writer +-- observed: an absent token or phase matches a row without one. +local function owner_fence_sql(fence: any, params: { any }): string + local clauses = {} + if fence.owner_token ~= nil then + table.insert(clauses, "owner_token = ?") + table.insert(params, fence.owner_token) + else + table.insert(clauses, "owner_token IS NULL") + end + if fence.owner_phase ~= nil then + table.insert(clauses, "owner_phase = ?") + table.insert(params, fence.owner_phase) + else + table.insert(clauses, "owner_phase IS NULL") + end + if fence.generation ~= nil then + table.insert(clauses, "generation = ?") + table.insert(params, fence.generation) + end + return table.concat(clauses, " AND ") +end + +local function validate_fence(fence: any): (boolean?, string?) + if type(fence) ~= "table" then return nil, "ownership fence is required" end + if fence.owner_token ~= nil and (type(fence.owner_token) ~= "string" or fence.owner_token == "") then + return nil, "owner_token must be a non-empty string" + end + if fence.owner_phase ~= nil and fence.owner_phase ~= PHASE.RUNNING and + fence.owner_phase ~= PHASE.RELEASED then + return nil, "owner_phase must be running or released" + end + if fence.generation ~= nil then + local generation = tonumber(fence.generation) + if not generation or generation < 1 or generation % 1 ~= 0 then + return nil, "generation must be a positive integer" + end + end + return true, nil +end + +-- Release the active request when the ownership fence still holds: marks the +-- owner released and the activation inactive. A fence that no longer matches +-- reports whether ownership changed or only the generation advanced. +function activation_repo.release_owner_tx(tx, dataflow_id, fence, now_value) if not tx then return nil, "transaction is required" end local valid, validation_err = validate_id(dataflow_id) if not valid then return nil, validation_err end - generation = tonumber(generation) - if not generation or generation < 1 or generation % 1 ~= 0 then - return nil, "generation must be a positive integer" - end + valid, validation_err = validate_fence(fence) + if not valid then return nil, validation_err end valid, validation_err = validate_timestamp(now_value, "updated_at") if not valid then return nil, validation_err end @@ -436,82 +487,122 @@ function activation_repo.release_if_generation_tx(tx, dataflow_id, generation, n return terminal, nil end + local params: { any } = { false, PHASE.RELEASED, now_value, dataflow_id, true } + local predicate = owner_fence_sql(fence, params) local result, update_err = tx_execute(tx, [[ UPDATE dataflow_activations - SET desired_active = ?, launch_args = NULL, updated_at = ? - WHERE dataflow_id = ? AND generation = ? AND desired_active = ? - AND EXISTS ( - SELECT 1 FROM dataflows - WHERE dataflow_id = ? AND status NOT IN (?, ?, ?, ?) - ) - ]], { - false, now_value, dataflow_id, generation, true, dataflow_id, - TERMINAL_VALUES[1], TERMINAL_VALUES[2], TERMINAL_VALUES[3], TERMINAL_VALUES[4], - }) + SET desired_active = ?, owner_phase = ?, launch_args = NULL, updated_at = ? + WHERE dataflow_id = ? AND desired_active = ? AND ]] .. predicate, params) if update_err then return nil, "failed to release activation: " .. tostring(update_err) end - if result and (result.rows_affected or 0) > 0 then - return { changed = true, released = true, generation = generation, terminal = false }, nil - end local current, current_err = get_tx(tx, dataflow_id) if current_err then return nil, current_err end + if result and (result.rows_affected or 0) > 0 then + return { + changed = true, + released = true, + terminal = false, + generation = current and current.generation or fence.generation, + }, nil + end return { changed = false, released = false, terminal = false, generation = current and current.generation or nil, + owner_changed = current == nil or current.owner_token ~= fence.owner_token or + current.owner_phase ~= fence.owner_phase, }, nil end --- Fence process ownership before spawn. A generation can be claimed only from --- the exact epoch observed by the overseer. The write happens before process --- creation, so an overseer crash between claim and spawn is classified as a --- same-runtime loss rather than retried into a process flood. -function activation_repo.claim_epoch_tx( - tx, dataflow_id, generation, observed_epoch, runtime_epoch, now_value) +-- Admit an orchestrator that already holds the canonical name as the owner of +-- the current request, before it mutates anything. Admission is refused for a +-- terminal or inactive activation, a request older than the one the process +-- was started for, and while another owner of the same runtime is running. The +-- same token is admitted again, so an unacknowledged admission can be retried. +function activation_repo.admit_owner_tx(tx, dataflow_id, min_generation, owner, now_value) if not tx then return nil, "transaction is required" end local valid, validation_err = validate_id(dataflow_id) if not valid then return nil, validation_err end - generation = tonumber(generation) - if not generation or generation < 1 or generation % 1 ~= 0 then - return nil, "generation must be a positive integer" + min_generation = tonumber(min_generation) + if not min_generation or min_generation < 1 or min_generation % 1 ~= 0 then + return nil, "min_generation must be a positive integer" end - if type(runtime_epoch) ~= "string" or runtime_epoch == "" then - return nil, "runtime_epoch is required" + if type(owner) ~= "table" then return nil, "owner is required" end + for _, field in ipairs({ "token", "pid", "runtime_epoch" }) do + if type(owner[field]) ~= "string" or owner[field] == "" then + return nil, "owner " .. field .. " is required" + end end valid, validation_err = validate_timestamp(now_value, "updated_at") if not valid then return nil, validation_err end local status, status_err = activation_repo.lock_workflow_tx(tx, dataflow_id) if status_err then return nil, status_err end - local terminal = terminal_result_from_status(status) - if terminal then - terminal.claimed = false - return terminal, nil + local found, current_err = get_tx(tx, dataflow_id) + if current_err then return nil, current_err end + if not found then + return { + dataflow_id = dataflow_id, + admitted = false, + refused = TERMINAL_STATUS[status] and "terminal" or "inactive", + }, nil end + local current = found :: any - local epoch_predicate = "owner_epoch IS NULL" - local params = { runtime_epoch, now_value, dataflow_id, generation, true } - if observed_epoch ~= nil then - if type(observed_epoch) ~= "string" or observed_epoch == "" then - return nil, "observed_epoch must be nil or a non-empty string" - end - epoch_predicate = "owner_epoch = ?" - table.insert(params, observed_epoch) + local refused: string? = nil + if TERMINAL_STATUS[status] then + refused = "terminal" + elseif current.owner_token == owner.token and current.owner_phase == PHASE.RUNNING then + refused = nil + elseif current.desired_active ~= true then + refused = "inactive" + elseif current.generation < min_generation then + refused = "stale" + elseif current.owner_phase == PHASE.RUNNING and current.owner_epoch == owner.runtime_epoch then + refused = "owned" + end + if refused then + current.admitted = false + current.refused = refused + return current, nil end + if current.owner_token == owner.token then + current.admitted = true + return current, nil + end + local result, update_err = tx_execute(tx, [[ UPDATE dataflow_activations - SET owner_epoch = ?, updated_at = ? - WHERE dataflow_id = ? AND generation = ? AND desired_active = ? - AND ]] .. epoch_predicate, params) - if update_err then return nil, "failed to claim activation epoch: " .. tostring(update_err) end + SET owner_token = ?, owner_pid = ?, owner_epoch = ?, owner_phase = ?, updated_at = ? + WHERE dataflow_id = ? AND generation = ? + ]], { owner.token, owner.pid, owner.runtime_epoch, PHASE.RUNNING, now_value, + dataflow_id, current.generation }) + if update_err then return nil, "failed to admit owner: " .. tostring(update_err) end + if not result or (result.rows_affected or 0) ~= 1 then return nil, "owner admission was not recorded" end + current.owner_token = owner.token + current.owner_pid = owner.pid + current.owner_epoch = owner.runtime_epoch + current.owner_phase = PHASE.RUNNING + current.admitted = true + return current, nil +end +-- Fence an orchestrator-owned transaction: the token must still own the +-- activation as a running owner. Takes the workflow lock first. +function activation_repo.verify_owner_tx(tx, dataflow_id, owner_token) + if not tx then return nil, "transaction is required" end + local valid, validation_err = validate_id(dataflow_id) + if not valid then return nil, validation_err end + if type(owner_token) ~= "string" or owner_token == "" then return nil, "owner_token is required" end + local _, status_err = activation_repo.lock_workflow_tx(tx, dataflow_id) + if status_err then return nil, status_err end local current, current_err = get_tx(tx, dataflow_id) if current_err then return nil, current_err end - if not current then return nil, "activation row missing after epoch claim" end - current.claimed = result ~= nil and (result.rows_affected or 0) == 1 - current.terminal = false - return current, nil + if not current or current.owner_token ~= owner_token or current.owner_phase ~= PHASE.RUNNING then + return nil, "orchestrator ownership lost" + end + return true, nil end function activation_repo.consume_wake_tx(tx, dataflow_id, wake_key, generation) @@ -579,6 +670,19 @@ function activation_repo.disable_terminal_tx(tx, dataflow_id, now_value) return cleanup_terminal_tx(tx, dataflow_id, status, now_value) end +-- Read the activation after taking the workflow lock, so a transaction holding +-- it (an admission or a release) finishes first. +function activation_repo.read_locked_tx(tx, dataflow_id) + if not tx then return nil, "transaction is required" end + local valid, validation_err = validate_id(dataflow_id) + if not valid then return nil, validation_err end + local status, status_err = activation_repo.lock_workflow_tx(tx, dataflow_id) + if status_err then return nil, status_err end + local activation, activation_err = get_tx(tx, dataflow_id) + if activation_err then return nil, activation_err end + return { activation = activation, status = status }, nil +end + function activation_repo.get(dataflow_id) local valid, id_err = validate_id(dataflow_id) if not valid then return nil, id_err end @@ -586,6 +690,7 @@ function activation_repo.get(dataflow_id) if db_err then return nil, db_err end local rows, query_err = db_query(db, [[ SELECT dataflow_id, generation, desired_active, owner_epoch, + owner_token, owner_pid, owner_phase, launch_args, requested_at, updated_at FROM dataflow_activations WHERE dataflow_id = ? LIMIT 1 ]], { dataflow_id }) @@ -598,8 +703,9 @@ function activation_repo.list_active() local db, db_err = sql.get(consts.APP_DB) if db_err then return nil, db_err end local rows, query_err = db_query(db, [[ - SELECT a.dataflow_id, a.generation, a.desired_active, a.owner_epoch, a.launch_args, - a.requested_at, a.updated_at + SELECT a.dataflow_id, a.generation, a.desired_active, a.owner_epoch, + a.owner_token, a.owner_pid, a.owner_phase, + a.launch_args, a.requested_at, a.updated_at FROM dataflow_activations a JOIN dataflows d ON d.dataflow_id = a.dataflow_id WHERE a.desired_active = ? AND d.status NOT IN (?, ?, ?, ?) diff --git a/src/persist/activation_repo_test.lua b/src/persist/activation_repo_test.lua index 2115a77..72a5069 100644 --- a/src/persist/activation_repo_test.lua +++ b/src/persist/activation_repo_test.lua @@ -138,39 +138,195 @@ local function define_tests() test.is_nil(select(1, activation_repo.get(id))) end) - test.it("claims an activation from one runtime epoch exactly once", function() + local function owner(token: string, runtime_epoch: string): any + return { token = token, pid = "pid-" .. token, runtime_epoch = runtime_epoch } + end + + test.it("admits one owner per runtime and lets the same token retry its admission", function() local id = create_dataflow(consts.STATUS.RUNNING) local requested = test.not_nil(select(1, transaction(function(tx) return activation_repo.request_activation_tx(tx, id, {}, now()) end))) :: any - test.is_nil(requested.owner_epoch) + test.is_nil(requested.owner_token) + test.is_nil(requested.owner_phase) local first = test.not_nil(select(1, transaction(function(tx) - return activation_repo.claim_epoch_tx( - tx, id, 1, nil, "runtime-a", now(1)) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t1", "runtime-a"), now(1)) end))) :: any - test.is_true(first.claimed) + test.is_true(first.admitted) + test.eq(first.generation, 1) + test.eq(first.owner_token, "t1") + test.eq(first.owner_pid, "pid-t1") test.eq(first.owner_epoch, "runtime-a") + test.eq(first.owner_phase, "running") - local duplicate = test.not_nil(select(1, transaction(function(tx) - return activation_repo.claim_epoch_tx( - tx, id, 1, nil, "runtime-a", now(2)) + local rival = test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t2", "runtime-a"), now(2)) + end))) :: any + test.is_false(rival.admitted) + test.eq(rival.refused, "owned") + test.eq(rival.owner_token, "t1") + + local retried = test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t1", "runtime-a"), now(3)) + end))) :: any + test.is_true(retried.admitted) + test.eq(retried.owner_token, "t1") + + local rebooted = test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t3", "runtime-b"), now(4)) + end))) :: any + test.is_true(rebooted.admitted, "a running owner of an earlier runtime is gone") + test.eq(rebooted.owner_token, "t3") + test.eq(rebooted.owner_epoch, "runtime-b") + end) + + test.it("refuses admission for terminal, inactive and older-than-requested activations", function() + local id = create_dataflow(consts.STATUS.RUNNING) + test.not_nil(select(1, transaction(function(tx) + return activation_repo.request_activation_tx(tx, id, {}, now()) + end))) + local stale = test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 2, owner("t1", "runtime-a"), now(1)) + end))) :: any + test.is_false(stale.admitted) + test.eq(stale.refused, "stale") + test.is_nil(stale.owner_token) + + test.is_true((test.not_nil(select(1, transaction(function(tx) + return activation_repo.release_owner_tx(tx, id, { generation = 1 }, now(2)) + end))) :: any).released) + local inactive = test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t1", "runtime-a"), now(3)) + end))) :: any + test.is_false(inactive.admitted) + test.eq(inactive.refused, "inactive") + + local done = create_dataflow(consts.STATUS.COMPLETED_SUCCESS) + local terminal = test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, done, 1, owner("t1", "runtime-a"), now(4)) + end))) :: any + test.is_false(terminal.admitted) + test.eq(terminal.refused, "terminal") + end) + + test.it("keeps the live owner across generation advances", function() + local id = create_dataflow(consts.STATUS.RUNNING) + test.not_nil(select(1, transaction(function(tx) + return activation_repo.request_activation_tx(tx, id, {}, now()) + end))) + test.is_true((test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t1", "runtime-a"), now(1)) + end))) :: any).admitted) + local signaled = test.not_nil(select(1, transaction(function(tx) + return activation_repo.activate_for_signal_tx( + tx, id, "signal:" .. uuid.v7(), now(2), now(2)) + end))) :: any + test.eq(signaled.generation, 2) + local requested = test.not_nil(select(1, transaction(function(tx) + return activation_repo.request_activation_tx(tx, id, {}, now(3)) + end))) :: any + test.eq(requested.generation, 3) + for _, row in ipairs({ signaled, requested, test.not_nil(find_active(id)) }) do + test.eq((row :: any).owner_token, "t1") + test.eq((row :: any).owner_phase, "running") + test.eq((row :: any).owner_epoch, "runtime-a") + end + end) + + test.it("releases only for the owning token at the current generation", function() + local id = create_dataflow(consts.STATUS.RUNNING) + test.not_nil(select(1, transaction(function(tx) + return activation_repo.request_activation_tx(tx, id, {}, now()) + end))) + test.is_true((test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t1", "runtime-a"), now(1)) + end))) :: any).admitted) + test.not_nil(select(1, transaction(function(tx) + return activation_repo.activate_for_signal_tx( + tx, id, "signal:" .. uuid.v7(), now(2), now(2)) + end))) + + local stale = test.not_nil(select(1, transaction(function(tx) + return activation_repo.release_owner_tx(tx, id, + { owner_token = "t1", owner_phase = "running", generation = 1 }, now(3)) end))) :: any - test.is_false(duplicate.claimed) - test.eq(duplicate.owner_epoch, "runtime-a") + test.is_false(stale.released) + test.eq(stale.generation, 2) + test.is_false(stale.owner_changed) - local reboot = test.not_nil(select(1, transaction(function(tx) - return activation_repo.claim_epoch_tx( - tx, id, 1, "runtime-a", "runtime-b", now(3)) + local foreign = test.not_nil(select(1, transaction(function(tx) + return activation_repo.release_owner_tx(tx, id, + { owner_token = "t2", owner_phase = "running", generation = 2 }, now(4)) end))) :: any - test.is_true(reboot.claimed) - test.eq(reboot.owner_epoch, "runtime-b") + test.is_false(foreign.released) + test.is_true(foreign.owner_changed) - local next_generation = test.not_nil(select(1, transaction(function(tx) - return activation_repo.request_activation_tx(tx, id, {}, now(4)) + local released = test.not_nil(select(1, transaction(function(tx) + return activation_repo.release_owner_tx(tx, id, + { owner_token = "t1", owner_phase = "running", generation = 2 }, now(5)) end))) :: any - test.eq(next_generation.generation, 2) - test.is_nil(next_generation.owner_epoch) + test.is_true(released.released) + test.eq(released.generation, 2) + local row = test.not_nil(select(1, activation_repo.get(id))) :: any + test.is_false(row.desired_active) + test.eq(row.owner_phase, "released") + test.eq(row.owner_token, "t1") + + test.not_nil(select(1, transaction(function(tx) + return activation_repo.request_activation_tx(tx, id, {}, now(6)) + end))) + local successor = test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 3, owner("t4", "runtime-a"), now(7)) + end))) :: any + test.is_true(successor.admitted, "a released owner is replaced in the same runtime") + end) + + test.it("fails a lost owner by token alone at the latest generation", function() + local id = create_dataflow(consts.STATUS.RUNNING) + test.not_nil(select(1, transaction(function(tx) + return activation_repo.request_activation_tx(tx, id, {}, now()) + end))) + test.is_true((test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t1", "runtime-a"), now(1)) + end))) :: any).admitted) + for offset = 2, 3 do + test.not_nil(select(1, transaction(function(tx) + return activation_repo.activate_for_signal_tx( + tx, id, "signal:" .. uuid.v7(), now(offset), now(offset)) + end))) + end + local failed = test.not_nil(select(1, transaction(function(tx) + return activation_repo.release_owner_tx(tx, id, + { owner_token = "t1", owner_phase = "running" }, now(4)) + end))) :: any + test.is_true(failed.released) + test.eq(failed.generation, 3) + end) + + test.it("verifies the owning token inside a transaction", function() + local id = create_dataflow(consts.STATUS.RUNNING) + test.not_nil(select(1, transaction(function(tx) + return activation_repo.request_activation_tx(tx, id, {}, now()) + end))) + test.is_true((test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t1", "runtime-a"), now(1)) + end))) :: any).admitted) + test.is_true(select(1, transaction(function(tx) + return activation_repo.verify_owner_tx(tx, id, "t1") + end))) + local _, foreign_err = transaction(function(tx) + return activation_repo.verify_owner_tx(tx, id, "t2") + end) + test.contains(tostring(foreign_err), "ownership lost") + test.is_true((test.not_nil(select(1, transaction(function(tx) + return activation_repo.release_owner_tx(tx, id, + { owner_token = "t1", owner_phase = "running", generation = 1 }, now(2)) + end))) :: any).released) + local _, released_err = transaction(function(tx) + return activation_repo.verify_owner_tx(tx, id, "t1") + end) + test.contains(tostring(released_err), "ownership lost") end) test.it("advances only when a newly inserted signal wake wins", function() @@ -199,19 +355,43 @@ local function define_tests() test.eq(wake_generation(id, second_key), 2) local stale_release = test.not_nil(select(1, transaction(function(tx) - return activation_repo.release_if_generation_tx(tx, id, 1, now(3)) + return activation_repo.release_owner_tx(tx, id, { generation = 1 }, now(3)) end))) :: any test.is_false(stale_release.released) test.eq(stale_release.generation, 2) test.is_true((test.not_nil(select(1, activation_repo.get(id))) :: any).desired_active) local current_release = test.not_nil(select(1, transaction(function(tx) - return activation_repo.release_if_generation_tx(tx, id, 2, now(4)) + return activation_repo.release_owner_tx(tx, id, { generation = 2 }, now(4)) end))) :: any test.is_true(current_release.released) test.is_false((test.not_nil(select(1, activation_repo.get(id))) :: any).desired_active) end) + test.it("reads the activation and workflow status under the workflow lock", function() + local id = create_dataflow(consts.STATUS.RUNNING) + test.not_nil(select(1, transaction(function(tx) + return activation_repo.request_activation_tx(tx, id, {}, now()) + end))) + test.is_true((test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t1", "runtime-a"), now(1)) + end))) :: any).admitted) + + local observed = test.not_nil(select(1, transaction(function(tx) + return activation_repo.read_locked_tx(tx, id) + end))) :: any + test.eq(observed.status, consts.STATUS.RUNNING) + local activation = test.not_nil(observed.activation) :: any + test.eq(activation.generation, 1) + test.eq(activation.owner_token, "t1") + test.eq(activation.owner_phase, "running") + + local _, missing_err = transaction(function(tx) + return activation_repo.read_locked_tx(tx, uuid.v7()) + end) + test.contains(tostring(missing_err), "dataflow not found") + end) + test.it("promotes a due timer exactly once and consumes only its fenced row", function() local id = create_dataflow(consts.STATUS.WAITING) local wake_key = "yield:" .. uuid.v7() diff --git a/src/persist/commit.lua b/src/persist/commit.lua index ca0e515..a1f0706 100644 --- a/src/persist/commit.lua +++ b/src/persist/commit.lua @@ -90,30 +90,53 @@ function commit.request_activation(dataflow_id, launch_args, options) return activation, nil end --- Terminalize an activation only if the generation that lost its runtime owner --- still owns the workflow. A newer start/signal generation wins the fence and --- is never failed by a stale EXIT event. -function commit.fail_activation(dataflow_id, generation, failure) +-- Record an orchestrator start as the owner of the current request (see +-- activation_repo.admit_owner_tx). owner = { token, pid, runtime_epoch }. +function commit.admit_owner(dataflow_id, min_generation, owner) + local db, db_err = get_db() + if db_err then return nil, db_err end + local tx, begin_err = db:begin() + if begin_err then + db:release() + return nil, "Failed to begin transaction: " .. tostring(begin_err) + end + local admission, admit_err = activation_repo.admit_owner_tx( + tx, dataflow_id, min_generation, owner, commit._get_current_timestamp()) + if admit_err then + tx:rollback() + db:release() + return nil, admit_err + end + local _, commit_err = tx:commit() + if commit_err then + tx:rollback() + db:release() + return nil, "Failed to commit owner admission: " .. tostring(commit_err) + end + db:release() + return admission, nil +end + +-- Terminalize an activation only while the ownership the overseer observed +-- still holds. fence = { token, phase, generation }: a running owner's token +-- alone covers every newer request; any other fence names one generation. +function commit.fail_activation(dataflow_id, fence, failure) if type(dataflow_id) ~= "string" or dataflow_id == "" then return nil, "Dataflow ID is required" end - generation = tonumber(generation) - if not generation or generation < 1 or generation % 1 ~= 0 then - return nil, "Activation generation must be a positive integer" - end + if type(fence) ~= "table" then return nil, "Ownership fence is required" end if type(failure) ~= "table" then return nil, "Failure details are required" end - local commands: { any } = { - { - type = ops.COMMAND_TYPES.COMPLETE_WORKFLOW, - payload = { - activation_generation = generation, - status = consts.STATUS.COMPLETED_FAILURE, - metadata = { runtime_failure = failure }, - merge_metadata = true, - }, - }, + local payload: { [string]: any } = { + status = consts.STATUS.COMPLETED_FAILURE, + metadata = { runtime_failure = failure }, + merge_metadata = true, + activation_generation = fence.generation, } + if fence.token ~= nil or fence.phase ~= nil then + payload.owner = { token = fence.token, phase = fence.phase } + end + local commands: { any } = { { type = ops.COMMAND_TYPES.COMPLETE_WORKFLOW, payload = payload } } local execute: any = commit.execute local result, execute_err = execute(dataflow_id, uuid.v7(), commands) if execute_err then return nil, execute_err end @@ -476,7 +499,9 @@ end -- @param dataflow_id (string): ID of the dataflow to operate on -- @param op_id (string): Operation ID (generated if nil) -- @param commands (table): Array of command tables --- @param options (table): Optional parameters +-- @param options (table): Optional parameters: publish (boolean) and +-- owner_token (string), which fences the batch to the orchestrator that owns +-- the activation -- @return (table, string): Result of operations and error message if failed function commit.execute(dataflow_id, op_id, commands, options) if not dataflow_id then @@ -514,6 +539,15 @@ function commit.execute(dataflow_id, op_id, commands, options) return nil, "Failed to begin transaction: " .. err_tx end + if options.owner_token ~= nil then + local _, owner_err = activation_repo.verify_owner_tx(tx, dataflow_id, options.owner_token) + if owner_err then + tx:rollback() + db:release() + return nil, tostring(owner_err) + end + end + -- Execute the operations local result, err_op = commit.tx_execute(tx, dataflow_id, op_id, commands, { publish = false }) diff --git a/src/persist/commit_test.lua b/src/persist/commit_test.lua index 57b3bb9..772ce55 100644 --- a/src/persist/commit_test.lua +++ b/src/persist/commit_test.lua @@ -1286,6 +1286,65 @@ local function define_tests() test.not_nil(result) test.eq(count_process_messages("dataflow.overseer", "dataflow.activation.changed"), 1) end) + + it("applies an owner-fenced batch only while its token owns the activation", function() + if test_ctx.tx then test_ctx.tx:rollback(); test_ctx.tx = nil end + if test_ctx.db then test_ctx.db:release(); test_ctx.db = nil end + local dataflow_id = create_isolated_dataflow() + test.not_nil(select(1, commit.request_activation(dataflow_id, {}, { notify = false }))) + local admitted, admit_err = commit.admit_owner(dataflow_id, 1, { + token = "owner-a", pid = "pid-a", runtime_epoch = "runtime-a", + }) + test.is_nil(admit_err) + test.is_true((test.not_nil(admitted) :: any).admitted) + + local function set_status(status: string, owner_token: string): (any, string?) + return commit.execute(dataflow_id, nil, { { + type = ops.COMMAND_TYPES.UPDATE_WORKFLOW, + payload = { dataflow_id = dataflow_id, status = status }, + } }, { publish = false, owner_token = owner_token }) + end + local foreign, foreign_err = set_status(consts.STATUS.RUNNING, "owner-b") + test.is_nil(foreign) + test.contains(tostring(foreign_err), "ownership lost") + local function status(): string + local db = test.not_nil(select(1, sql.get("app:db"))) :: any + local rows, query_err = db:query(rebind( + "SELECT status FROM dataflows WHERE dataflow_id = ?", db:type()), { dataflow_id }) + db:release() + test.is_nil(query_err) + test.eq(#rows, 1) + return tostring(rows[1].status) + end + test.eq(status(), "active") + + local owned, owned_err = set_status(consts.STATUS.RUNNING, "owner-a") + test.is_nil(owned_err) + test.not_nil(owned) + test.eq(status(), consts.STATUS.RUNNING) + end) + + it("fails an activation only while the observed owner still holds it", function() + if test_ctx.tx then test_ctx.tx:rollback(); test_ctx.tx = nil end + if test_ctx.db then test_ctx.db:release(); test_ctx.db = nil end + local dataflow_id = create_isolated_dataflow() + test.not_nil(select(1, commit.request_activation(dataflow_id, {}, { notify = false }))) + test.is_true((test.not_nil(select(1, commit.admit_owner(dataflow_id, 1, { + token = "owner-a", pid = "pid-a", runtime_epoch = "runtime-a", + }))) :: any).admitted) + test.not_nil(select(1, commit.request_activation(dataflow_id, {}, { notify = false }))) + + local stale, stale_err = commit.fail_activation(dataflow_id, + { token = "owner-b", phase = "running" }, { reason = "runtime_owner_lost" }) + test.is_nil(stale_err) + test.is_false((test.not_nil(stale) :: any).completed) + + local failed, failed_err = commit.fail_activation(dataflow_id, + { token = "owner-a", phase = "running" }, { reason = "runtime_owner_lost" }) + test.is_nil(failed_err) + test.is_true((test.not_nil(failed) :: any).completed) + test.eq((test.not_nil(failed) :: any).current_generation, 2) + end) end) describe("Edge Cases and Error Handling", function() diff --git a/src/persist/ops.lua b/src/persist/ops.lua index f91f0c6..46fe49f 100644 --- a/src/persist/ops.lua +++ b/src/persist/ops.lua @@ -1130,6 +1130,18 @@ end handlers[constants.COMMAND_TYPES.UPDATE_WORKFLOW] = update_workflow +-- payload.owner = { token, phase } names the ownership the writer holds; an +-- absent owner matches an activation that was never owned. +local function ownership_fence(payload: any, generation: number?): (any?, string?) + local owner = payload.owner + if owner ~= nil and type(owner) ~= "table" then return nil, "owner must be a table" end + return { + owner_token = owner and owner.token or nil, + owner_phase = owner and owner.phase or nil, + generation = generation, + }, nil +end + handlers[constants.COMMAND_TYPES.PASSIVATE_WORKFLOW] = function(tx, dataflow_id, op_id, command) if not dataflow_id or dataflow_id == "" then return nil, "Workflow ID is required" @@ -1141,9 +1153,21 @@ handlers[constants.COMMAND_TYPES.PASSIVATE_WORKFLOW] = function(tx, dataflow_id, if not generation or generation < 1 or generation % 1 ~= 0 then return nil, "Activation generation must be a positive integer" end + -- Signal wakes whose NODE_SIGNAL the releasing owner applied without a + -- waiting yield. They are removed only together with a successful release. + local signal_wake_keys = payload.signal_wake_keys or {} + if type(signal_wake_keys) ~= "table" then return nil, "signal_wake_keys must be an array" end + for _, wake_key in ipairs(signal_wake_keys) do + if type(wake_key) ~= "string" or not wake_key:match("^signal:.+") then + return nil, "Passivation may only remove signal wake keys: " .. tostring(wake_key) + end + end + + local fence, fence_err = ownership_fence(payload, generation) + if fence_err then return nil, fence_err end local now_ts = time.now():format(time.RFC3339NANO) - local release, release_err = activation_repo.release_if_generation_tx(tx, wf_id, generation, now_ts) + local release, release_err = activation_repo.release_owner_tx(tx, wf_id, fence, now_ts) if release_err then return nil, "Failed to release activation: " .. tostring(release_err) end if not release.released then return { @@ -1152,6 +1176,7 @@ handlers[constants.COMMAND_TYPES.PASSIVATE_WORKFLOW] = function(tx, dataflow_id, op_id = op_id, released = false, terminal = release.terminal == true, + owner_changed = release.owner_changed == true, current_generation = release.generation, } end @@ -1166,6 +1191,14 @@ handlers[constants.COMMAND_TYPES.PASSIVATE_WORKFLOW] = function(tx, dataflow_id, return nil, "Workflow not found while passivating" end + for _, wake_key in ipairs(signal_wake_keys) do + local _, wake_err = sql.builder.delete("dataflow_wakes") + :where("dataflow_id = ?", wf_id) + :where("wake_key = ?", wake_key) + :run_with(tx):exec() + if wake_err then return nil, "Failed to remove applied signal wake: " .. tostring(wake_err) end + end + return { dataflow_id = wf_id, changes_made = true, @@ -1184,17 +1217,24 @@ handlers[constants.COMMAND_TYPES.COMPLETE_WORKFLOW] = function(tx, dataflow_id, end local payload = command.payload or {} local wf_id = payload.dataflow_id or dataflow_id - local generation = tonumber(payload.activation_generation) - if not generation or generation < 1 or generation % 1 ~= 0 then - return nil, "Activation generation must be a positive integer" + -- A running owner's token alone fences its failure at whatever generation + -- is current; every other completion is fenced to one generation. + local generation = nil + if payload.activation_generation ~= nil or not (type(payload.owner) == "table" and payload.owner.token) then + generation = tonumber(payload.activation_generation) + if not generation or generation < 1 or generation % 1 ~= 0 then + return nil, "Activation generation must be a positive integer" + end end local status = payload.status local terminal = status == constants.STATUS.COMPLETED_SUCCESS or status == constants.STATUS.COMPLETED_FAILURE if not terminal then return nil, "Completion status must be completed or failed" end + local fence, fence_err = ownership_fence(payload, generation) + if fence_err then return nil, fence_err end local now_ts = time.now():format(time.RFC3339NANO) - local release, release_err = activation_repo.release_if_generation_tx(tx, wf_id, generation, now_ts) + local release, release_err = activation_repo.release_owner_tx(tx, wf_id, fence, now_ts) if release_err then return nil, "Failed to fence workflow completion: " .. tostring(release_err) end if not release.released then return { @@ -1203,9 +1243,11 @@ handlers[constants.COMMAND_TYPES.COMPLETE_WORKFLOW] = function(tx, dataflow_id, op_id = op_id, completed = false, terminal = release.terminal == true, + owner_changed = release.owner_changed == true, current_generation = release.generation, } end + generation = release.generation local update_result, update_err = update_workflow( tx, dataflow_id, op_id, { diff --git a/src/persist/ops_test.lua b/src/persist/ops_test.lua index f0fcb50..3cd93ea 100644 --- a/src/persist/ops_test.lua +++ b/src/persist/ops_test.lua @@ -1164,6 +1164,140 @@ local function define_tests() test.eq(rows[1].status, ops.STATUS.RUNNING) end) + it("passivates only for the owning token and marks its ownership released", function() + local resources = setup_test_resources() + local tx = get_test_transaction() + local timestamp = time.now():format(time.RFC3339NANO) + test.not_nil(select(1, activation_repo.request_activation_tx( + tx, resources.dataflow_id, {}, timestamp))) + local admitted = test.not_nil(select(1, activation_repo.admit_owner_tx( + tx, resources.dataflow_id, 1, + { token = "owner-a", pid = "pid-a", runtime_epoch = "runtime-a" }, timestamp))) :: any + test.is_true(admitted.admitted) + local _, running_err = tx:execute(rebind( + "UPDATE dataflows SET status = ? WHERE dataflow_id = ?", tx:db_type()), + { ops.STATUS.RUNNING, resources.dataflow_id }) + test.is_nil(running_err) + + local foreign, foreign_err = ops.execute(tx, resources.dataflow_id, nil, { + type = ops.COMMAND_TYPES.PASSIVATE_WORKFLOW, + payload = { + activation_generation = 1, + owner = { token = "owner-b", phase = "running" }, + }, + }) + test.is_nil(foreign_err) + test.is_false(foreign.results[1].released) + test.is_true(foreign.results[1].owner_changed) + + local result, err = ops.execute(tx, resources.dataflow_id, nil, { + type = ops.COMMAND_TYPES.PASSIVATE_WORKFLOW, + payload = { + activation_generation = 1, + owner = { token = "owner-a", phase = "running" }, + }, + }) + test.is_nil(err) + test.is_true(result.results[1].released) + local rows, query_err = txq(tx, [[ + SELECT owner_token, owner_phase, desired_active FROM dataflow_activations + WHERE dataflow_id = ? + ]], { resources.dataflow_id }) + test.is_nil(query_err) + test.eq(rows[1].owner_token, "owner-a") + test.eq(rows[1].owner_phase, "released") + test.is_false(db_bool(rows[1].desired_active)) + end) + + it("removes only the captured signal wakes in the releasing passivation", function() + local resources = setup_test_resources() + local tx = get_test_transaction() + local timestamp = time.now():format(time.RFC3339NANO) + local activation, activation_err = activation_repo.request_activation_tx( + tx, resources.dataflow_id, {}, timestamp) + test.is_nil(activation_err) + test.eq(activation.generation, 1) + local _, running_err = tx:execute(rebind( + "UPDATE dataflows SET status = ? WHERE dataflow_id = ?", tx:db_type()), + { ops.STATUS.RUNNING, resources.dataflow_id }) + test.is_nil(running_err) + for _, wake_key in ipairs({ "signal:captured", "signal:uncaptured", "yield:deadline" }) do + local _, wake_err = tx:execute(rebind([[ + INSERT INTO dataflow_wakes(dataflow_id, wake_key, wake_at, activation_generation) + VALUES (?, ?, ?, ?) + ]], tx:db_type()), { resources.dataflow_id, wake_key, "2099-01-01T00:00:00Z", 1 }) + test.is_nil(wake_err) + end + + local result, err = ops.execute(tx, resources.dataflow_id, nil, { + type = ops.COMMAND_TYPES.PASSIVATE_WORKFLOW, + payload = { activation_generation = 1, signal_wake_keys = { "signal:captured" } }, + }) + test.is_nil(err) + test.is_true(result.results[1].released) + + local wakes, wake_query_err = txq(tx, [[ + SELECT wake_key FROM dataflow_wakes WHERE dataflow_id = ? ORDER BY wake_key + ]], { resources.dataflow_id }) + test.is_nil(wake_query_err) + test.eq(#wakes, 2) + test.eq(wakes[1].wake_key, "signal:uncaptured") + test.eq(wakes[2].wake_key, "yield:deadline") + end) + + it("keeps captured signal wakes when a concurrent signal wins the passivation fence", function() + local resources = setup_test_resources() + local tx = get_test_transaction() + local timestamp = time.now():format(time.RFC3339NANO) + local _, activation_err = activation_repo.request_activation_tx( + tx, resources.dataflow_id, {}, timestamp) + test.is_nil(activation_err) + local _, running_err = tx:execute(rebind( + "UPDATE dataflows SET status = ? WHERE dataflow_id = ?", tx:db_type()), + { ops.STATUS.RUNNING, resources.dataflow_id }) + test.is_nil(running_err) + local _, captured_err = tx:execute(rebind([[ + INSERT INTO dataflow_wakes(dataflow_id, wake_key, wake_at, activation_generation) + VALUES (?, ?, ?, ?) + ]], tx:db_type()), { resources.dataflow_id, "signal:captured", timestamp, 1 }) + test.is_nil(captured_err) + local signal, signal_err = activation_repo.activate_for_signal_tx( + tx, resources.dataflow_id, "signal:concurrent", timestamp, timestamp) + test.is_nil(signal_err) + test.eq(signal.generation, 2) + + local result, err = ops.execute(tx, resources.dataflow_id, nil, { + type = ops.COMMAND_TYPES.PASSIVATE_WORKFLOW, + payload = { activation_generation = 1, signal_wake_keys = { "signal:captured" } }, + }) + test.is_nil(err) + test.is_false(result.results[1].released) + test.eq(result.results[1].current_generation, 2) + + local wakes, wake_query_err = txq(tx, [[ + SELECT wake_key FROM dataflow_wakes WHERE dataflow_id = ? ORDER BY wake_key + ]], { resources.dataflow_id }) + test.is_nil(wake_query_err) + test.eq(#wakes, 2) + test.eq(wakes[1].wake_key, "signal:captured") + test.eq(wakes[2].wake_key, "signal:concurrent") + end) + + it("rejects passivation cleanup of a wake that is not a signal wake", function() + local resources = setup_test_resources() + local tx = get_test_transaction() + local _, activation_err = activation_repo.request_activation_tx( + tx, resources.dataflow_id, {}, time.now():format(time.RFC3339NANO)) + test.is_nil(activation_err) + + local result, err = ops.execute(tx, resources.dataflow_id, nil, { + type = ops.COMMAND_TYPES.PASSIVATE_WORKFLOW, + payload = { activation_generation = 1, signal_wake_keys = { "yield:deadline" } }, + }) + test.is_nil(result) + test.contains(tostring(err), "signal wake") + end) + it("serializes terminal cleanup before later signal activation", function() local resources = setup_test_resources() local tx = get_test_transaction() diff --git a/src/runner/_index.yaml b/src/runner/_index.yaml index b1e1dd3..281bdbc 100644 --- a/src/runner/_index.yaml +++ b/src/runner/_index.yaml @@ -33,7 +33,6 @@ entries: execution_frame: userspace.dataflow:execution_frame scheduler: userspace.dataflow.runner:scheduler workflow_state: userspace.dataflow.runner:workflow_state - wake_repo: userspace.dataflow.persist:wake_repo overseer: userspace.dataflow.runner:overseer method: run @@ -168,6 +167,17 @@ entries: overseer: userspace.dataflow.runner:overseer method: run_tests + - name: runtime_epoch + kind: function.lua + meta: + comment: Returns the full-runtime boot epoch under the module-owned reader group, whatever the caller scope. + source: file://runtime_epoch.lua + modules: [env] + security: + groups: + - userspace.dataflow.security:epoch_reader + method: run + - name: overseer_state kind: library.lua meta: diff --git a/src/runner/orchestrator.lua b/src/runner/orchestrator.lua index e2230ea..71a388a 100644 --- a/src/runner/orchestrator.lua +++ b/src/runner/orchestrator.lua @@ -13,7 +13,6 @@ local orchestrator = { commit = require("commit"), activation_repo = require("activation_repo"), execution_frame = require("execution_frame"), - wake_repo = require("wake_repo"), overseer = require("overseer"), } @@ -29,7 +28,6 @@ type Runtime = { commit: any, activation_repo: any, execution_frame: any, - wake_repo: any, overseer: any, } @@ -44,6 +42,7 @@ type OrchestratorState = { scope: any, on_complete_id: string?, activation_generation: number, + owner_token: string, running: boolean, exit_result: any, final_status: string?, @@ -58,6 +57,8 @@ type ParkArmState = { scope: any, } +local OWNER_RUNNING = "running" + local TERMINAL_STATUS = { [consts.STATUS.COMPLETED_SUCCESS] = true, [consts.STATUS.COMPLETED_FAILURE] = true, @@ -168,6 +169,19 @@ local function stop_for_existing_terminal(state: OrchestratorState, projection: return false end +-- Another orchestrator owns the activation, or this one already released it: +-- this life must not write or schedule anything further. +local function stop_for_lost_ownership(state: OrchestratorState) + state.running = false + state.exit_result = { + success = false, + dataflow_id = state.dataflow_id, + error = "Orchestrator ownership lost", + } + state.final_status = nil + return false +end + local function adopt_projection_generation(state: OrchestratorState, projection: any) local current_generation = projection and tonumber( projection.current_generation or projection.generation) @@ -201,6 +215,7 @@ local function persist_fenced_failure( type = consts.COMMAND_TYPES.COMPLETE_WORKFLOW, payload = { activation_generation = state.activation_generation, + owner = { token = state.owner_token, phase = OWNER_RUNNING }, status = consts.STATUS.COMPLETED_FAILURE, metadata = { error = failure_message }, }, @@ -234,6 +249,9 @@ local function persist_fenced_failure( if projection and projection.terminal == true then return stop_for_existing_terminal(state, projection), false end + if projection and projection.owner_changed == true then + return stop_for_lost_ownership(state), false + end local _, generation_err = adopt_projection_generation(state, projection) if generation_err then @@ -365,6 +383,78 @@ local function load_startup_pending_commits(state: OrchestratorState) return true end +local function passivation_projection(result: any): any + local results = result and result.results or nil + if type(results) ~= "table" then return nil end + for _, projection in ipairs(results) do + if type(projection) == "table" and projection.released ~= nil then return projection end + end + return nil +end + +---Release the owned activation. A newer request keeps this owner running; a +---committed release ends this life: it gives up the canonical name and exits. +---@return boolean continue Whether this life keeps running +---@return boolean reschedule Whether a newer generation must be scheduled now +local function passivate(state: OrchestratorState): (boolean, boolean) + local wake_keys = state.workflow_state:unclaimed_signal_wake_keys() + state.workflow_state:queue_commands({ + type = consts.COMMAND_TYPES.PASSIVATE_WORKFLOW, + payload = { + activation_generation = state.activation_generation, + owner = { token = state.owner_token, phase = OWNER_RUNNING }, + signal_wake_keys = wake_keys, + }, + }) + local result, persist_err = state.workflow_state:persist() + if persist_err then + -- The release may have committed; the overseer resolves it from the + -- durable ownership record once this life has exited. + state.running = false + state.exit_result = { + success = false, + dataflow_id = state.dataflow_id, + error = "Failed to persist waiting status: " .. tostring(persist_err), + } + return false, false + end + + local projection = passivation_projection(result) + if not projection or projection.released ~= true then + if projection and projection.terminal == true then + return stop_for_existing_terminal(state, projection), false + end + if projection and projection.owner_changed == true then + return stop_for_lost_ownership(state), false + end + local _, generation_err = adopt_projection_generation(state, projection) + if generation_err then + state.running = false + state.exit_result = { + success = false, + dataflow_id = state.dataflow_id, + error = generation_err, + } + return false, false + end + return true, true + end + + state.workflow_state:release_signal_wake_keys(wake_keys) + -- NODE_YIELD projected its deadline atomically; the passivation leaves it + -- untouched so a concurrent NODE_SIGNAL replacement survives. + state.runtime.process.registry.unregister("dataflow." .. state.dataflow_id) + state.runtime.overseer.notify() + state.running = false + state.exit_result = { + success = true, + pending = true, + passivated = true, + dataflow_id = state.dataflow_id, + } + return false, false +end + ---Call scheduler and handle the result immediately ---@param state table Orchestrator state ---@return boolean continue Whether to continue processing @@ -407,36 +497,8 @@ local function call_scheduler_and_handle(state: OrchestratorState) end -- re-enter the loop: yield satisfied, state changed, re-schedule elseif decision.type == state.runtime.scheduler.DECISION_TYPE.PASSIVATE then - state.workflow_state:queue_commands({ - type = consts.COMMAND_TYPES.PASSIVATE_WORKFLOW, - payload = { activation_generation = state.activation_generation }, - }) - local passivate_result, status_err = state.workflow_state:persist() - if status_err then - state.running = false - state.exit_result = { - success = false, - dataflow_id = state.dataflow_id, - error = "Failed to persist waiting status: " .. tostring(status_err), - } - return false - end - - local projection = passivate_result and passivate_result.results and passivate_result.results[1] or nil - if not projection or projection.released ~= true then - if projection and projection.terminal == true then - return stop_for_existing_terminal(state, projection) - end - local _, generation_err = adopt_projection_generation(state, projection) - if generation_err then - state.running = false - state.exit_result = { - success = false, - dataflow_id = state.dataflow_id, - error = generation_err, - } - return false - end + local continue, reschedule = passivate(state) + if reschedule then -- A start, signal, or due deadline advanced the durable -- generation while this life was deciding to park. The failed -- CAS is the handoff: reload durable work and schedule again @@ -444,32 +506,7 @@ local function call_scheduler_and_handle(state: OrchestratorState) state.reschedule_requested = true goto continue_scheduler end - local unclaimed_wakes = {} - if type(state.workflow_state.take_unclaimed_signal_wake_keys) == "function" then - unclaimed_wakes = state.workflow_state:take_unclaimed_signal_wake_keys() - end - for _, wake_key in ipairs(unclaimed_wakes) do - local _, cleanup_err = state.runtime.wake_repo.remove(state.dataflow_id, wake_key) - if cleanup_err then - logger:warn("unclaimed signal wake cleanup failed", { - dataflow_id = state.dataflow_id, - wake_key = wake_key, - error = tostring(cleanup_err), - }) - end - end - -- NODE_YIELD projected its deadline atomically. Do not rewrite the - -- wake here: a concurrent NODE_SIGNAL may already have replaced it - -- with an immediate wake between this decision and persistence. - state.runtime.overseer.notify() - state.running = false - state.exit_result = { - success = true, - pending = true, - passivated = true, - dataflow_id = state.dataflow_id, - } - return false + return continue else return true end @@ -662,6 +699,7 @@ function handle_complete_workflow(state: OrchestratorState, payload: any) type = consts.COMMAND_TYPES.COMPLETE_WORKFLOW, payload = { activation_generation = state.activation_generation, + owner = { token = state.owner_token, phase = OWNER_RUNNING }, status = final_status, metadata = { error = not success and detailed_error or nil } } @@ -683,6 +721,9 @@ function handle_complete_workflow(state: OrchestratorState, payload: any) if projection and projection.terminal == true then return stop_for_existing_terminal(state, projection), false end + if projection and projection.owner_changed == true then + return stop_for_lost_ownership(state), false + end local _, generation_err = adopt_projection_generation(state, projection) if generation_err then state.exit_result = { @@ -698,7 +739,8 @@ function handle_complete_workflow(state: OrchestratorState, payload: any) -- re-evaluation cannot act on in-memory outcomes that never persisted; -- completion is only decided without live node processes, so a reload -- orphans nothing. - local fresh_state, fresh_err = state.runtime.workflow_state.new(state.dataflow_id) + local fresh_state, fresh_err = state.runtime.workflow_state.new( + state.dataflow_id, { owner_token = state.owner_token }) local loaded = nil if fresh_state then loaded, fresh_err = fresh_state:load_state() @@ -1159,6 +1201,53 @@ local function duplicate_owner_result(dataflow_id) } end +-- The overseer passes the runtime epoch with a spawn; a synchronous caller's +-- scope may not read it, so that path asks the module-owned reader. +local function runtime_epoch_for(runtime: Runtime, args: any): (string?, string?) + local provided = args and args.runtime_epoch + if type(provided) == "string" and provided ~= "" then return provided, nil end + local read, read_err = runtime.funcs.new():call(consts.RUNTIME_EPOCH_READER) + if read_err then return nil, tostring(read_err) end + local epoch = type(read) == "table" and read.epoch or nil + if type(epoch) ~= "string" or epoch == "" then return nil, "runtime epoch is not ready" end + return epoch, nil +end + +-- Admit this process as the owner of the current request. An admission whose +-- outcome is unknown is resolved by rereading the token it would have written. +local function admit_owner( + runtime: Runtime, + args: any, + dataflow_id: string, + min_generation: number, + pid: string +): (any?, string?) + local runtime_epoch, epoch_err = runtime_epoch_for(runtime, args) + if epoch_err or not runtime_epoch then + return nil, "Dataflow runtime epoch is unavailable: " .. tostring(epoch_err) + end + local token = uuid.v7() + local admission, admission_err = runtime.commit.admit_owner(dataflow_id, min_generation, { + token = token, + pid = pid, + runtime_epoch = runtime_epoch, + }) + if not admission_err and admission then + admission.owner_token = token + return admission, nil + end + local current, current_err = runtime.activation_repo.get(dataflow_id) + if current_err then + return nil, "Orchestrator admission outcome unknown: " .. tostring(admission_err) .. + "; " .. tostring(current_err) + end + if current and current.owner_token == token and current.owner_phase == OWNER_RUNNING then + current.admitted = true + return current, nil + end + return nil, "Orchestrator admission failed: " .. tostring(admission_err) +end + ---Main orchestrator function ---@param args table Arguments containing dataflow_id and optional init_func_id ---@return table result Orchestration result with success/error @@ -1173,7 +1262,6 @@ local function run(args, runtime_override: any?) commit = bound.commit, activation_repo = bound.activation_repo, execution_frame = bound.execution_frame, - wake_repo = bound.wake_repo, overseer = bound.overseer, } local dataflow_id_raw = args and args.dataflow_id @@ -1233,7 +1321,48 @@ local function run(args, runtime_override: any?) end runtime.process.set_options({ trap_links = true, upgradable = false }) - local ws, ws_err = runtime.workflow_state.new(dataflow_id) + -- Holding the name, become the durable owner of the current request before + -- anything is loaded or written; every later write is fenced by the token. + local admission, admission_err = admit_owner( + runtime, args, dataflow_id, activation_generation, tostring(self_pid)) + if admission_err or not admission then + return { + success = false, + dataflow_id = dataflow_id, + error = tostring(admission_err or "admission returned no result"), + } + end + if admission.admitted ~= true then + runtime.process.registry.unregister(process_name) + if admission.refused == "terminal" then + local _, cleanup_err = runtime.commit.disable_terminal_activation(dataflow_id) + if cleanup_err then + return { + success = false, + dataflow_id = dataflow_id, + error = "Failed to disable stale terminal activation: " .. tostring(cleanup_err), + } + end + return { + success = true, + dataflow_id = dataflow_id, + message = "Dataflow already in terminal state", + } + end + if admission.refused == "owned" then return duplicate_owner_result(dataflow_id) end + return { + success = true, + pending = true, + dataflow_id = dataflow_id, + message = "Stale or inactive workflow activation", + } + end + -- A newer request may have won before this process started; the owner + -- adopts it rather than exiting into a false owner loss. + activation_generation = tonumber(admission.generation) or activation_generation + local owner_token = tostring(admission.owner_token) + + local ws, ws_err = runtime.workflow_state.new(dataflow_id, { owner_token = owner_token }) if ws_err then return { success = false, error = "Failed to create workflow state: " .. ws_err } end @@ -1254,6 +1383,7 @@ local function run(args, runtime_override: any?) scope = nil, on_complete_id = nil, activation_generation = activation_generation, + owner_token = owner_token, running = true, exit_result = nil, runtime = runtime, @@ -1283,7 +1413,7 @@ local function run(args, runtime_override: any?) error = "Failed to disable stale terminal activation: " .. tostring(cleanup_err), } end - runtime.process.registry.unregister("dataflow." .. dataflow_id) + runtime.process.registry.unregister(process_name) return { success = true, dataflow_id = dataflow_id, @@ -1291,31 +1421,7 @@ local function run(args, runtime_override: any?) } end - -- A process may start after a newer activation generation has already won. - -- The canonical named process owns that handoff: adopt a newer durable - -- fence before executing rather than exiting and creating a false runtime - -- owner loss. A generation older than the spawn request is invalid. - local activation, activation_err = runtime.activation_repo.get(dataflow_id) - if activation_err then - return { - success = false, - dataflow_id = dataflow_id, - error = "Failed to load workflow activation: " .. tostring(activation_err), - } - end - local durable_generation = activation and tonumber(activation.generation) or nil - if not activation or activation.desired_active ~= true or not durable_generation or - durable_generation < activation_generation then - runtime.process.registry.unregister("dataflow." .. dataflow_id) - return { - success = true, - pending = true, - dataflow_id = dataflow_id, - message = "Stale or inactive workflow activation", - } - end - activation_generation = durable_generation - local durable_launch_args = type(activation.launch_args) == "table" and activation.launch_args or nil + local durable_launch_args = type(admission.launch_args) == "table" and admission.launch_args or nil if durable_launch_args then if type(durable_launch_args.init_func_id) == "string" then init_func_id = durable_launch_args.init_func_id @@ -1324,7 +1430,6 @@ local function run(args, runtime_override: any?) args.on_complete = durable_launch_args.on_complete end end - state.activation_generation = activation_generation -- The spawn path is already monitored atomically. This notification lets -- a synchronous client-owned invocation be adopted by the same overseer; diff --git a/src/runner/orchestrator_completion_flush_test.lua b/src/runner/orchestrator_completion_flush_test.lua index 28e0cb0..50ece5e 100644 --- a/src/runner/orchestrator_completion_flush_test.lua +++ b/src/runner/orchestrator_completion_flush_test.lua @@ -83,7 +83,6 @@ local function define_tests() end, }, execution_frame = execution_frame, - wake_repo = { remove = function(): (boolean, nil) return true, nil end }, overseer = { notify = function(): (boolean, nil) return true, nil end }, funcs = { new = function(): any @@ -198,6 +197,7 @@ local function define_tests() local result = orchestrator.run({ dataflow_id = dataflow_id, activation_generation = probes.generation, + runtime_epoch = "runtime-test", }, runtime) :: any test.is_true(probes.aborted, "exit-batch transaction abort was exercised") @@ -225,6 +225,7 @@ local function define_tests() local result = orchestrator.run({ dataflow_id = dataflow_id, activation_generation = stale_generation, + runtime_epoch = "runtime-test", }, runtime) :: any test.is_true(probes.aborted, "exit-batch transaction abort was exercised") diff --git a/src/runner/orchestrator_process_event_test.lua b/src/runner/orchestrator_process_event_test.lua index 3841b50..75acdba 100644 --- a/src/runner/orchestrator_process_event_test.lua +++ b/src/runner/orchestrator_process_event_test.lua @@ -104,6 +104,9 @@ local function define_tests() } runtime.commit = { get_pending_commits = function(_dataflow_id: string): ({ string }?, string?) return {}, nil end, + admit_owner = function(): (any, string?) + return { generation = 1, desired_active = true, admitted = true }, nil + end, } runtime.overseer = { notify = function(): (boolean, string?) return true, nil end, @@ -170,6 +173,7 @@ local function define_tests() local result = orchestrator.run({ dataflow_id = "failure-reason-workflow", activation_generation = 1, + runtime_epoch = "runtime-test", }, runtime) :: any test.is_false(result.success) diff --git a/src/runner/orchestrator_test.lua b/src/runner/orchestrator_test.lua index 2470697..dba8d0d 100644 --- a/src/runner/orchestrator_test.lua +++ b/src/runner/orchestrator_test.lua @@ -4,6 +4,9 @@ local consts = require("consts") type HarnessOptions = { activation: any?, + admission_error: string?, + admission_refused: string?, + durable_owner_token: string?, actor_id: string?, channel_events: { any }?, failed_node_errors: string?, @@ -15,6 +18,8 @@ type HarnessOptions = { persist_results: { any }?, registry_owner: string?, scheduler_decisions: { any }?, + signal_wake_keys: { string }?, + trace: { string }?, state_error: string?, status: string?, } @@ -25,8 +30,13 @@ local function harness(options: HarnessOptions?): any local scheduler_index = 0 local event_index = 0 local workflow_state: any = {} + local trace: { string } = cfg.trace or {} + local registered_owner: string? = cfg.registry_owner + local admitted_token: string? = nil + local function record(entry: string) table.insert(trace, entry) end workflow_state.load_state = function(self: any): (any?, string?) + record("load_state") if cfg.load_error then return nil, cfg.load_error end return self, nil end @@ -49,7 +59,23 @@ local function harness(options: HarnessOptions?): any end workflow_state.get_failed_node_errors = function(): string? return cfg.failed_node_errors end workflow_state.track_process = function(self: any): any return self end - workflow_state.queue_commands = function(self: any): any return self end + workflow_state.queue_commands = function(self: any, commands: any): any + if type(commands) == "table" and commands.type == consts.COMMAND_TYPES.PASSIVATE_WORKFLOW then + local owner = commands.payload.owner or {} + record("queue_passivation:" .. table.concat(commands.payload.signal_wake_keys or {}, ",") .. + ":" .. tostring(owner.token == admitted_token and owner.phase == "running")) + end + return self + end + workflow_state.unclaimed_signal_wake_keys = function(): { string } + local keys: { string } = {} + for _, key in ipairs(cfg.signal_wake_keys or {}) do table.insert(keys, key) end + return keys + end + workflow_state.release_signal_wake_keys = function(self: any, keys: { string }): any + record("release_wakes:" .. table.concat(keys, ",")) + return self + end workflow_state.queue_completion = function(self: any): any return self end workflow_state.discard_queued_commands = function(self: any): any return self end workflow_state.get_node = function(): any return { type = "test_node", status = consts.STATUS.PENDING } end @@ -63,18 +89,23 @@ local function harness(options: HarnessOptions?): any workflow_state.persist = function(): (any?, string?) persist_index = persist_index + 1 local configured = cfg.persist_results and cfg.persist_results[persist_index] - if configured then - if configured.error then return nil, tostring(configured.error) end - return configured.value, nil + if configured and configured.error then + record("persist_failed") + return nil, tostring(configured.error) end - return { changes_made = true, results = { { completed = true, released = true } } }, nil + record("commit") + local value = configured and configured.value or + { changes_made = true, results = { { completed = true, released = true } } } + return value, nil end local inbox = { case_receive = function(): any return { channel = "inbox" } end } local events = { case_receive = function(): any return { channel = "events" } end } local runtime: any = { workflow_state = { - new = function(): (any?, string?) + new = function(_id: string, options: any?): (any?, string?) + local token = options and options.owner_token or nil + record("state:" .. tostring(token ~= nil and token == admitted_token)) if cfg.state_error then return nil, cfg.state_error end return workflow_state, nil end, @@ -97,11 +128,20 @@ local function harness(options: HarnessOptions?): any process = { registry = { lookup = function(): (string?, any?) - if cfg.registry_owner then return cfg.registry_owner, nil end + record("lookup") + if registered_owner then return registered_owner, nil end return nil, "not_found: name not registered" end, - register = function(): (boolean, nil) return true, nil end, - unregister = function() end, + register = function(): (boolean, nil) + record("register") + registered_owner = "orchestrator-pid" + return true, nil + end, + unregister = function(): any + record("unregister") + registered_owner = nil + return true + end, }, pid = function(): string return "orchestrator-pid" end, set_options = function() end, @@ -134,10 +174,30 @@ local function harness(options: HarnessOptions?): any return cfg.pending_commits or {}, nil end, disable_terminal_activation = function(): (any, nil) return { terminal = true }, nil end, + admit_owner = function(_id: string, min_generation: number, owner: any): (any?, string?) + local current = cfg.activation or { generation = 1, desired_active = true } + local refused = cfg.admission_refused + if not refused and current.desired_active ~= true then refused = "inactive" end + if not refused and current.generation < min_generation then refused = "stale" end + record("admit:" .. tostring(current.generation) .. ":" .. tostring(owner.runtime_epoch) .. + ":" .. tostring(refused or "admitted")) + admitted_token = owner.token + if cfg.admission_error then return nil, cfg.admission_error end + local row: any = { admitted = refused == nil, refused = refused } + for key, value in pairs(current) do row[key] = value end + return row, nil + end, }, activation_repo = { get = function(): (any, nil) - return cfg.activation or { generation = 1, desired_active = true }, nil + record("reread") + local row: any = {} + for key, value in pairs(cfg.activation or { generation = 1, desired_active = true }) do + row[key] = value + end + row.owner_token = cfg.durable_owner_token or admitted_token + row.owner_phase = "running" + return row, nil end, }, execution_frame = { @@ -151,12 +211,22 @@ local function harness(options: HarnessOptions?): any local executor: any = {} executor.with_actor = function(self: any): any return self end executor.with_scope = function(self: any): any return self end - executor.call = function(): (any, nil) return {}, nil end + executor.call = function(_self: any, id: string): (any, nil) + if id == consts.RUNTIME_EPOCH_READER then + record("epoch_reader") + return { epoch = "runtime-test" }, nil + end + return {}, nil + end return executor end, }, - wake_repo = { remove = function(): (boolean, nil) return true, nil end }, - overseer = { notify = function(): (boolean, nil) return true, nil end }, + overseer = { + notify = function(): (boolean, nil) + record("notify") + return true, nil + end, + }, } return runtime end @@ -166,9 +236,31 @@ local function run(runtime: any, args: any?): any for key, value in pairs(args or {}) do call_args[key] = value end call_args.dataflow_id = call_args.dataflow_id or "workflow-1" call_args.activation_generation = call_args.activation_generation or 1 + if call_args.runtime_epoch == false then + call_args.runtime_epoch = nil + else + call_args.runtime_epoch = call_args.runtime_epoch or "runtime-spawn" + end return orchestrator.run(call_args, runtime) end +local function since(trace: { string }, marker: string): { string } + local tail: { string } = {} + local found = false + for _, entry in ipairs(trace) do + if entry == marker then found = true end + if found then table.insert(tail, entry) end + end + return tail +end + +local function contains(trace: { string }, entry: string): boolean + for _, value in ipairs(trace) do + if value == entry then return true end + end + return false +end + local function define_tests() describe("Orchestrator protocol", function() it("rejects a missing dataflow id", function() @@ -290,6 +382,122 @@ local function define_tests() test.is_true(result.passivated) end) + it("admits itself under the canonical name before loading or mutating state", function() + local trace: { string } = {} + local result = run(harness({ + trace = trace, + activation = { generation = 3, desired_active = true }, + }), { activation_generation = 2 }) + test.is_true(result.success) + test.eq(table.concat(since(trace, "register"), " ", 1, 4), + "register admit:3:runtime-spawn:admitted state:true load_state") + end) + + it("reads the runtime epoch through the module reader when started synchronously", function() + local trace: { string } = {} + local result = run(harness({ trace = trace }), { runtime_epoch = false }) + test.is_true(result.success) + local admitted = since(trace, "epoch_reader") + test.eq(admitted[2], "admit:1:runtime-test:admitted") + end) + + it("leaves before loading state when admission is refused", function() + for _, case in ipairs({ + { refused = "owned", message = "already running" }, + { refused = "stale", message = "Stale" }, + { refused = "inactive", message = "Stale" }, + }) do + local trace: { string } = {} + local result = run(harness({ trace = trace, admission_refused = case.refused })) + test.is_true(result.pending) + test.contains(result.message, case.message) + test.is_false(contains(trace, "load_state")) + test.is_true(contains(trace, "unregister")) + end + local terminal_trace: { string } = {} + local terminal = run(harness({ trace = terminal_trace, admission_refused = "terminal" })) + test.is_true(terminal.success) + test.contains(terminal.message, "terminal") + test.is_false(contains(terminal_trace, "load_state")) + end) + + it("resolves an unacknowledged admission from its durable token", function() + local trace: { string } = {} + local result = run(harness({ trace = trace, admission_error = "connection reset" })) + test.is_true(result.success) + test.eq(table.concat(since(trace, "admit:1:runtime-spawn:admitted"), " ", 1, 4), + "admit:1:runtime-spawn:admitted reread state:true load_state") + + local lost: { string } = {} + local refused = run(harness({ + trace = lost, admission_error = "connection reset", durable_owner_token = "other-token", + })) + test.is_false(refused.success) + test.contains(refused.error, "admission") + test.is_false(contains(lost, "load_state")) + end) + + it("releases ownership, then gives up the name and exits", function() + local trace: { string } = {} + local result = run(harness({ + trace = trace, + signal_wake_keys = { "signal:a", "signal:b" }, + scheduler_decisions = { { type = "passivate", payload = {} } }, + persist_results = { + { value = { results = { { released = true, generation = 1 } } } }, + }, + })) + test.is_true(result.success) + test.is_true(result.passivated) + test.eq(table.concat(since(trace, "queue_passivation:signal:a,signal:b:true"), " "), + "queue_passivation:signal:a,signal:b:true commit " .. + "release_wakes:signal:a,signal:b unregister notify") + end) + + it("keeps the name and wakes and reschedules when passivation loses the generation", function() + local trace: { string } = {} + local result = run(harness({ + trace = trace, + signal_wake_keys = { "signal:a" }, + scheduler_decisions = { + { type = "passivate", payload = {} }, + { type = "complete_workflow", payload = { success = true, message = "rescheduled" } }, + }, + persist_results = { + { value = { results = { { released = false, current_generation = 2 } } } }, + { value = { results = { { completed = true, generation = 2 } } } }, + }, + })) + test.is_true(result.success) + test.eq(result.output.message, "rescheduled") + test.is_false(contains(trace, "unregister")) + test.is_false(contains(trace, "release_wakes:signal:a")) + end) + + it("stops without further work when its ownership was lost or the release is uncertain", function() + for _, persisted in ipairs({ + { value = { results = { { released = false, owner_changed = true, current_generation = 1 } } } }, + { error = "Failed to persist commands: connection reset" }, + }) do + local trace: { string } = {} + local result = run(harness({ + trace = trace, + signal_wake_keys = { "signal:a" }, + scheduler_decisions = { + { type = "passivate", payload = {} }, + { type = "complete_workflow", payload = { success = true, message = "unreachable" } }, + }, + persist_results = { persisted }, + })) + test.is_false(result.success) + test.is_nil(result.passivated) + test.is_false(contains(trace, "release_wakes:signal:a")) + test.is_false(contains(trace, "notify")) + local after = since(trace, "queue_passivation:signal:a:true") + test.eq(#after, 2) + end + end) + it("runtime cancellation stops the life without inventing business cancellation", function() local result = run(harness({ scheduler_decisions = { { type = "no_work", payload = {} } }, diff --git a/src/runner/overseer.lua b/src/runner/overseer.lua index 4f96c49..f823cfd 100644 --- a/src/runner/overseer.lua +++ b/src/runner/overseer.lua @@ -27,40 +27,29 @@ local SAFETY_INTERVAL = "30s" local SCAN_LIMIT = 100 local RUNTIME_EPOCH_ENV = "userspace.dataflow.env:runtime_epoch" -type OwnerReference = { - dataflow_id: string, - generation: number, -} - -type OwnershipRecord = { - dataflow_id: string, - generation: number, - phase: string, - pid: string?, - claim_required: boolean, - claim_from_epoch: string?, - candidate_pid: string?, -} - type OwnershipState = { - by_dataflow: { [string]: OwnershipRecord }, by_pid: { [string]: string }, + by_dataflow: { [string]: string }, +} + +type Fence = { + token: string?, + phase: string?, + generation: number?, } type Decision = { kind: string, reason: string, - dataflow_id: string?, + dataflow_id: string, generation: number?, pid: string?, - message: string?, - observed_epoch: string?, + fence: Fence?, } type Runtime = { ownership: OwnershipState, nudges: { [string]: Nudge }, - known: { [string]: boolean }, bootstrapped: boolean, epoch: string?, } @@ -72,15 +61,10 @@ type Nudge = { wake_at: string?, } -type Activation = { - dataflow_id: string?, - generation: number?, - desired_active: boolean?, - owner_epoch: string?, - launch_args: table?, - promoted: boolean?, - requested_at: string?, - updated_at: string?, +type Observation = { + activation: any, + status: string?, + registered_pid: string?, } type Workflow = { @@ -96,11 +80,8 @@ type WakeRow = { wake_at: string, } -local TERMINAL_STATUS = { - [consts.STATUS.COMPLETED_SUCCESS] = true, - [consts.STATUS.COMPLETED_FAILURE] = true, - [consts.STATUS.CANCELLED] = true, - [consts.STATUS.TERMINATED] = true, +type ReconcileOptions = { + message: string?, } local function schema_not_ready(err: any): boolean @@ -160,10 +141,6 @@ local function call_with_tx(fn: (any) -> (any?, string?)): (any?, string?) return M.with_tx(fn) end -local function is_terminal(status: string?): boolean - return TERMINAL_STATUS[tostring(status or "")] == true -end - local function log_flow(message: string, dataflow_id: string, err: any) logger:warn(message, { dataflow_id = dataflow_id, @@ -193,18 +170,6 @@ local function lookup_owner(dataflow_id: string): (string?, string?) return pid and tostring(pid) or nil, nil end -local function apply_transition( - runtime: Runtime, - next_state: OwnershipState?, - decision: Decision?, - err: any -): (Decision?, string?) - if err then return nil, tostring(err) end - if not next_state or not decision then return nil, "overseer transition returned no result" end - runtime.ownership = next_state :: OwnershipState - return decision :: Decision, nil -end - local function clone_launch_args(value: { [string]: any }?): { [string]: any } local result: { [string]: any } = {} for key, item in pairs(value or {}) do result[key] = item end @@ -215,7 +180,6 @@ function M.new_runtime(epoch: string?): Runtime return { ownership = M.overseer_state.new() :: OwnershipState, nudges = {}, - known = {}, bootstrapped = false, epoch = epoch, } @@ -255,264 +219,157 @@ local function failure_message(event: any): string return "active orchestrator exited before reaching a durable terminal or waiting state" end -function M.drive_decision(runtime: Runtime, initial: Decision?): (boolean?, string?) - local decision = initial - for _ = 1, 10 do - if not decision or decision.kind == M.overseer_state.ACTION.NONE then - if decision and decision.reason == "owner_monitored" and decision.pid then - local owner = M.overseer_state.owner_for_pid( - runtime.ownership, decision.pid) :: OwnerReference? - if owner then - local delivered, delivery_err = deliver_nudge( - runtime, owner.dataflow_id, decision.pid) - if not delivered then - log_flow("owner nudge delivery failed", owner.dataflow_id, delivery_err) - end - end - end - return true, nil - end - - local dataflow_id = decision.dataflow_id - local generation = decision.generation - if not dataflow_id or not generation then return nil, "decision identity is missing" end - - if decision.kind == M.overseer_state.ACTION.INSPECT_OWNER then - local pid, lookup_err = lookup_owner(dataflow_id) - if lookup_err then return nil, "canonical owner lookup failed: " .. lookup_err end - decision = select(1, apply_transition(runtime, - M.overseer_state.on_owner_observation(runtime.ownership, { - dataflow_id = dataflow_id, - generation = generation, - registered_pid = pid, - message = decision.message, - }))) - - elseif decision.kind == M.overseer_state.ACTION.CLAIM then - if not runtime.epoch then return nil, "runtime epoch is unavailable" end - local claimed, claim_err = call_with_tx(function(tx) - local result, err = M.activation_repo.claim_epoch_tx( - tx, dataflow_id, generation, decision.observed_epoch, - runtime.epoch, now_value()) - return result, err and tostring(err) or nil - end) - if claim_err then return nil, tostring(claim_err) end - decision = select(1, apply_transition(runtime, - M.overseer_state.on_claim_observation(runtime.ownership, { - dataflow_id = dataflow_id, - generation = generation, - claimed = claimed ~= nil and claimed.claimed == true, - }))) - - elseif decision.kind == M.overseer_state.ACTION.REFRESH then - local current, current_err = M.activation_repo.get(dataflow_id) - local workflow, workflow_err = M.dataflow_repo.get(dataflow_id) - if current_err or workflow_err then - return nil, tostring(current_err or workflow_err) - end - if not current then return true, nil end - return M.reconcile_activation(runtime, current, workflow) - - elseif decision.kind == M.overseer_state.ACTION.MONITOR then - if not decision.pid then return nil, "monitor decision has no PID" end - local ok, monitored, monitor_err = pcall(M.process.monitor, decision.pid) - local monitor_ok = ok and (monitored == true or - is_already_monitoring(monitored) or is_already_monitoring(monitor_err)) - local registered_pid = nil - if not monitor_ok then registered_pid = select(1, lookup_owner(dataflow_id)) end - decision = select(1, apply_transition(runtime, - M.overseer_state.on_monitor_observation(runtime.ownership, { - dataflow_id = dataflow_id, - generation = generation, - pid = decision.pid, - monitor_ok = monitor_ok, - registered_pid = registered_pid, - error = not ok and tostring(monitored) or tostring(monitor_err or "monitor failed"), - }))) +local MAX_RECONCILE_PASSES = 4 + +-- Read the activation and the canonical name together under the workflow lock, +-- so an admission or release in flight finishes before the name is checked. +local function observe(dataflow_id: string): (Observation?, string?) + local observed, observe_err = call_with_tx(function(tx) + local locked, locked_err = M.activation_repo.read_locked_tx(tx, dataflow_id) + if locked_err then return nil, tostring(locked_err) end + local pid, lookup_err = lookup_owner(dataflow_id) + if lookup_err then return nil, "canonical owner lookup failed: " .. lookup_err end + return { + activation = locked and locked.activation or nil, + status = locked and locked.status or nil, + registered_pid = pid, + }, nil + end) + if observe_err then return nil, tostring(observe_err) end + return observed :: Observation, nil +end - elseif decision.kind == M.overseer_state.ACTION.STOP then - if not decision.pid then return nil, "stop decision has no PID" end - local registered_pid, lookup_err = lookup_owner(dataflow_id) - if lookup_err then return nil, "terminal owner lookup failed: " .. lookup_err end - if not registered_pid then - decision = nil - goto continue_decision - end - local target_pid = registered_pid - local cancel_ok, cancelled, cancel_err = pcall(M.process.cancel, target_pid, "5s") - if not cancel_ok or cancelled ~= true then - local cancel_failure = cancel_ok and cancel_err or cancelled - if is_not_found(cancel_failure) then - decision = nil - else - local terminate_ok, terminated, terminate_err = pcall( - M.process.terminate, target_pid) - local terminate_failure = terminate_ok and terminate_err or terminated - if (not terminate_ok or terminated ~= true) and - not is_not_found(terminate_failure) then - return nil, "failed to stop terminal orchestrator: " .. tostring( - terminate_err or terminated or cancel_err or cancelled) - end - decision = nil - end - else - decision = nil - end +local function stop_owner(pid: string): (boolean?, string?) + local cancel_ok, cancelled, cancel_err = pcall(M.process.cancel, pid, "5s") + if cancel_ok and cancelled == true then return true, nil end + local cancel_failure = cancel_ok and cancel_err or cancelled + if is_not_found(cancel_failure) then return true, nil end + local terminate_ok, terminated, terminate_err = pcall(M.process.terminate, pid) + local terminate_failure = terminate_ok and terminate_err or terminated + if (not terminate_ok or terminated ~= true) and not is_not_found(terminate_failure) then + return nil, "failed to stop orchestrator: " .. tostring( + terminate_err or terminated or cancel_err or cancelled) + end + return true, nil +end - elseif decision.kind == M.overseer_state.ACTION.SPAWN then - local activation, activation_err = M.activation_repo.get(dataflow_id) - local workflow, workflow_err = M.dataflow_repo.get(dataflow_id) - if activation_err or workflow_err or not activation or not workflow then - decision = select(1, apply_transition(runtime, - M.overseer_state.on_spawn_observation(runtime.ownership, { - dataflow_id = dataflow_id, - generation = generation, - error = "durable spawn state unavailable: " .. tostring( - activation_err or workflow_err or "missing row"), - }))) - elseif activation.desired_active ~= true or - tonumber(activation.generation) ~= generation or is_terminal(workflow.status) then - decision = select(1, apply_transition(runtime, - M.overseer_state.on_activation(runtime.ownership, { - dataflow_id = dataflow_id, - generation = tonumber(activation.generation) or generation, - desired_active = activation.desired_active == true, - status = tostring(workflow.status), - owner_epoch = activation.owner_epoch and - tostring(activation.owner_epoch) or nil, - runtime_epoch = runtime.epoch, - }))) - else - local actor, scope, frame_err = M.execution_frame.reconstruct( - workflow.actor_id, workflow.actor_context) - if frame_err or not actor or not scope then - decision = select(1, apply_transition(runtime, - M.overseer_state.on_spawn_observation(runtime.ownership, { - dataflow_id = dataflow_id, - generation = generation, - error = "execution frame reconstruction failed: " .. tostring( - frame_err or "missing actor or scope"), - }))) - else - local args = clone_launch_args(activation.launch_args :: { [string]: any }?) - args.dataflow_id = dataflow_id - args.activation_generation = generation - local spawn_ok, spawn_pid, spawn_err = pcall(function() - return M.process.with_context({}) - :with_name("dataflow." .. dataflow_id) - :with_actor(actor) - :with_scope(scope) - :spawn_monitored(tostring(M.consts.ORCHESTRATOR), tostring(M.consts.HOST_ID), args) - end) - if not spawn_ok then - spawn_err = tostring(spawn_pid) - spawn_pid = nil - end - local registered_pid = select(1, lookup_owner(dataflow_id)) - decision = select(1, apply_transition(runtime, - M.overseer_state.on_spawn_observation(runtime.ownership, { - dataflow_id = dataflow_id, - generation = generation, - spawn_pid = spawn_pid and tostring(spawn_pid) or nil, - registered_pid = registered_pid, - error = spawn_pid and nil or tostring(spawn_err or "spawn returned no PID"), - }))) - end - end +local function monitor_owner(runtime: Runtime, dataflow_id: string, pid: string): boolean + local ok, monitored, monitor_err = pcall(M.process.monitor, pid) + local monitor_ok = ok and (monitored == true or + is_already_monitoring(monitored) or is_already_monitoring(monitor_err)) + if not monitor_ok then return false end + M.overseer_state.track(runtime.ownership, dataflow_id, pid) + local delivered, delivery_err = deliver_nudge(runtime, dataflow_id, pid) + if not delivered then log_flow("owner nudge delivery failed", dataflow_id, delivery_err) end + return true +end - elseif decision.kind == M.overseer_state.ACTION.FAIL then - local failure, failure_err = M.commit.fail_activation(dataflow_id, generation, { - source = "dataflow.overseer", - reason = decision.reason, - message = decision.message, - failed_at = now_value(), - }) - if failure_err then return nil, failure_err end - decision = select(1, apply_transition(runtime, - M.overseer_state.on_failed(runtime.ownership, { - dataflow_id = dataflow_id, - generation = generation, - }))) - if failure and failure.completed ~= true then - local current, current_err = M.activation_repo.get(dataflow_id) - local workflow, workflow_err = M.dataflow_repo.get(dataflow_id) - if current_err or workflow_err then - return nil, tostring(current_err or workflow_err) - end - if current then - local reconciled, reconcile_err = M.reconcile_activation(runtime, current, workflow) - if not reconciled then return nil, reconcile_err end - end - end - else - return nil, "unknown overseer decision " .. tostring(decision.kind) - end - ::continue_decision:: - end - return nil, "overseer decision chain exceeded safety bound" +local function fail(dataflow_id: string, fence: Fence, reason: string, message: string): (any?, string?) + return M.commit.fail_activation(dataflow_id, fence, { + source = "dataflow.overseer", + reason = reason, + message = message, + failed_at = now_value(), + }) end -function M.reconcile_activation( +-- Spawn the canonical orchestrator for the observed request. Returns the +-- spawned pid, or a name conflict marker, or the reason spawning is impossible. +local function spawn_owner( runtime: Runtime, - raw_activation: any, - raw_workflow: any?, - nudge: Nudge? -): (boolean?, string?) - if type(raw_activation) ~= "table" then return nil, "activation must be a table" end - local normalized_launch_args: table? = nil - if type(raw_activation.launch_args) == "table" then - normalized_launch_args = raw_activation.launch_args :: table + dataflow_id: string, + activation: any, + generation: number +): (string?, boolean, string?) + local raw_workflow, workflow_err = M.dataflow_repo.get(dataflow_id) + if workflow_err or not raw_workflow then + return nil, false, "durable spawn state unavailable: " .. tostring(workflow_err or "missing row") end - local activation: Activation = { - dataflow_id = raw_activation.dataflow_id and - tostring(raw_activation.dataflow_id) or nil, - generation = tonumber(raw_activation.generation), - desired_active = raw_activation.desired_active == true, - owner_epoch = raw_activation.owner_epoch and tostring(raw_activation.owner_epoch) or nil, - launch_args = normalized_launch_args, - promoted = raw_activation.promoted == true, - requested_at = raw_activation.requested_at and - tostring(raw_activation.requested_at) or nil, - updated_at = raw_activation.updated_at and tostring(raw_activation.updated_at) or nil, - } - local workflow: Workflow? = nil - if type(raw_workflow) == "table" then - workflow = { - dataflow_id = raw_workflow.dataflow_id and tostring(raw_workflow.dataflow_id) or nil, - actor_id = raw_workflow.actor_id and tostring(raw_workflow.actor_id) or nil, - actor_context = raw_workflow.actor_context, - status = raw_workflow.status and tostring(raw_workflow.status) or nil, - } + local workflow = raw_workflow :: Workflow + local actor, scope, frame_err = M.execution_frame.reconstruct(workflow.actor_id, workflow.actor_context) + if frame_err or not actor or not scope then + return nil, false, "execution frame reconstruction failed: " .. tostring( + frame_err or "missing actor or scope") end - local dataflow_id = tostring(activation.dataflow_id or "") - local generation = tonumber(activation.generation) - if dataflow_id == "" or not generation then return nil, "activation identity is invalid" end - if not runtime.epoch then return nil, "runtime epoch is unavailable" end - runtime.known[dataflow_id] = true + local launch_args: { [string]: any }? = nil + if type(activation.launch_args) == "table" then + launch_args = activation.launch_args :: { [string]: any } + end + local args = clone_launch_args(launch_args) + args.dataflow_id = dataflow_id + args.activation_generation = generation + args.runtime_epoch = runtime.epoch + local spawn_ok, spawn_pid, spawn_err = pcall(function() + return M.process.with_context({}) + :with_name("dataflow." .. dataflow_id) + :with_actor(actor) + :with_scope(scope) + :spawn_monitored(tostring(M.consts.ORCHESTRATOR), tostring(M.consts.HOST_ID), args) + end) + if spawn_ok and spawn_pid then return tostring(spawn_pid), false, nil end + local failure = spawn_ok and spawn_err or spawn_pid + local registered_pid = select(1, lookup_owner(dataflow_id)) + if registered_pid then return nil, true, nil end + return nil, false, "orchestrator spawn failed: " .. tostring(failure or "spawn returned no PID") +end - local current = M.overseer_state.owner_for_dataflow(runtime.ownership, dataflow_id) - if activation.desired_active == true and (not current or generation > current.generation) then - runtime.nudges[dataflow_id] = nudge or { +-- Converge one dataflow on its durable ownership record. Every pass starts +-- from a fresh locked observation; nothing decided earlier is needed. +function M.reconcile(runtime: Runtime, dataflow_id: string, options: ReconcileOptions?): (boolean?, string?) + if not runtime.epoch then return nil, "runtime epoch is unavailable" end + local message = options and options.message or "active orchestrator disappeared during runtime" + for _ = 1, MAX_RECONCILE_PASSES do + local observed, observe_err = observe(dataflow_id) + if observe_err or not observed then return nil, tostring(observe_err) end + local activation = observed.activation + if not activation then + M.overseer_state.forget_dataflow(runtime.ownership, dataflow_id) + return true, nil + end + local decision = M.overseer_state.decide({ dataflow_id = dataflow_id, - generation = generation, - } - elseif activation.desired_active ~= true or (workflow and is_terminal(workflow.status)) then - runtime.nudges[dataflow_id] = nil + status = observed.status, + desired_active = activation.desired_active == true, + generation = tonumber(activation.generation), + owner_token = activation.owner_token, + owner_phase = activation.owner_phase, + owner_epoch = activation.owner_epoch, + registered_pid = observed.registered_pid, + runtime_epoch = runtime.epoch, + }) :: Decision + + if decision.kind == M.overseer_state.ACTION.NONE then + runtime.nudges[dataflow_id] = nil + return true, nil + elseif decision.kind == M.overseer_state.ACTION.STOP then + runtime.nudges[dataflow_id] = nil + return stop_owner(tostring(decision.pid)) + elseif decision.kind == M.overseer_state.ACTION.MONITOR then + if monitor_owner(runtime, dataflow_id, tostring(decision.pid)) then return true, nil end + elseif decision.kind == M.overseer_state.ACTION.FAIL then + local failed, fail_err = fail(dataflow_id, decision.fence :: Fence, decision.reason, message) + if fail_err then return nil, tostring(fail_err) end + if failed and (failed.completed == true or failed.terminal == true) then return true, nil end + elseif decision.kind == M.overseer_state.ACTION.SPAWN then + local pid, conflict, spawn_err = spawn_owner( + runtime, dataflow_id, activation, tonumber(decision.generation) or 1) + if pid then + M.overseer_state.track(runtime.ownership, dataflow_id, pid) + local delivered, delivery_err = deliver_nudge(runtime, dataflow_id, pid) + if not delivered then log_flow("owner nudge delivery failed", dataflow_id, delivery_err) end + return true, nil + end + if not conflict then + local failed, fail_err = fail(dataflow_id, decision.fence :: Fence, + "orchestrator_spawn_failed", tostring(spawn_err)) + if fail_err then return nil, tostring(fail_err) end + if failed and (failed.completed == true or failed.terminal == true) then return true, nil end + end + else + return nil, "unknown overseer decision " .. tostring(decision.kind) + end end - - local next_state, next_decision, transition_err = M.overseer_state.on_activation( - runtime.ownership, { - dataflow_id = dataflow_id, - generation = generation, - desired_active = activation.desired_active == true, - status = workflow and tostring(workflow.status) or nil, - owner_epoch = activation.owner_epoch and tostring(activation.owner_epoch) or nil, - runtime_epoch = runtime.epoch, - }) - local decision, state_err = apply_transition( - runtime, next_state, next_decision, transition_err) - if state_err then return nil, state_err end - return M.drive_decision(runtime, decision) + return nil, "ownership of " .. dataflow_id .. " did not settle" end local function pending_due(now: string, limit: number): ({ WakeRow }?, string?) @@ -556,13 +413,13 @@ function M.promote_due(runtime: Runtime): (number?, string?) tostring(row.dataflow_id), "generation is missing") goto continue_due end - local ok, reconcile_err = M.reconcile_activation( - runtime, activation :: Activation, nil, { + runtime.nudges[tostring(row.dataflow_id)] = { dataflow_id = tostring(row.dataflow_id), generation = promoted_generation, wake_key = tostring(row.wake_key), wake_at = row.wake_at and tostring(row.wake_at) or nil, - }) + } + local ok, reconcile_err = M.reconcile(runtime, tostring(row.dataflow_id)) if not ok then log_flow("promoted activation reconciliation failed", tostring(row.dataflow_id), reconcile_err) @@ -576,37 +433,20 @@ end function M.reconcile_all(runtime: Runtime): (number?, string?) local active, list_err = M.activation_repo.list_active() if list_err then return nil, tostring(list_err) end - local active_ids: { [string]: boolean } = {} + local seen: { [string]: boolean } = {} local active_count = 0 for _, activation in ipairs(active or {}) do active_count = active_count + 1 local id = tostring(activation.dataflow_id) - active_ids[id] = true - local ok, reconcile_err = M.reconcile_activation(runtime, activation) + seen[id] = true + local ok, reconcile_err = M.reconcile(runtime, id) if not ok then log_flow("active activation reconciliation failed", id, reconcile_err) end end - - local inactive_ids = {} - for dataflow_id in pairs(runtime.known) do - if not active_ids[dataflow_id] then table.insert(inactive_ids, dataflow_id) end - end - for _, dataflow_id in ipairs(inactive_ids) do - local owner = M.overseer_state.owner_for_dataflow(runtime.ownership, dataflow_id) - if owner then - local activation, activation_err = M.activation_repo.get(dataflow_id) - local workflow, workflow_err = M.dataflow_repo.get(dataflow_id) - if activation_err or workflow_err then - log_flow("inactive activation reconciliation failed", dataflow_id, - activation_err or workflow_err) - else - local snapshot = activation or { - dataflow_id = dataflow_id, - generation = owner.generation, - desired_active = false, - } - local ok, reconcile_err = M.reconcile_activation(runtime, snapshot, workflow) - if not ok then log_flow("inactive state application failed", dataflow_id, reconcile_err) end - end + -- A monitored process whose activation is no longer active is stopped. + for _, id in ipairs(M.overseer_state.tracked(runtime.ownership)) do + if not seen[id] then + local ok, reconcile_err = M.reconcile(runtime, id) + if not ok then log_flow("inactive activation reconciliation failed", id, reconcile_err) end end end return active_count, nil @@ -632,12 +472,7 @@ function M.handle_activation_hint(runtime: Runtime, payload: any): (boolean?, st payload.dataflow_id == "" then return nil, "activation hint identity is invalid" end - local activation, activation_err = M.activation_repo.get(payload.dataflow_id) - if activation_err then return nil, tostring(activation_err) end - if not activation then return nil, "activation row is missing" end - local workflow, workflow_err = M.dataflow_repo.get(payload.dataflow_id) - if workflow_err then return nil, tostring(workflow_err) end - return M.reconcile_activation(runtime, activation, workflow) + return M.reconcile(runtime, payload.dataflow_id) end function M.safety_reconcile(runtime: Runtime): (number?, string?) @@ -648,42 +483,10 @@ end function M.handle_exit(runtime: Runtime, event: any): (boolean?, string?) local pid = event and event.from and tostring(event.from) or nil - local owner = pid and M.overseer_state.owner_for_pid(runtime.ownership, pid) or nil - if not owner or not pid then return true, nil end - - local activation, activation_err = M.activation_repo.get(owner.dataflow_id) - local workflow, workflow_err = M.dataflow_repo.get(owner.dataflow_id) - if activation_err or workflow_err then return nil, tostring(activation_err or workflow_err) end - - local generation = activation and tonumber(activation.generation) or nil - if generation and generation ~= owner.generation then - local next_state, next_decision, transition_err = M.overseer_state.on_exit( - runtime.ownership, { - pid = pid, - generation = owner.generation, - desired_active = false, - status = workflow and tostring(workflow.status) or nil, - }) - local _, remove_err = apply_transition( - runtime, next_state, next_decision, transition_err) - if remove_err then return nil, remove_err end - return M.reconcile_activation(runtime, activation, workflow) - end - - local desired_active = activation ~= nil and activation.desired_active == true and - workflow ~= nil and not is_terminal(workflow.status) - local next_state, next_decision, transition_err = M.overseer_state.on_exit( - runtime.ownership, { - pid = pid, - generation = owner.generation, - desired_active = desired_active, - status = workflow and tostring(workflow.status) or nil, - message = failure_message(event), - }) - local decision, state_err = apply_transition( - runtime, next_state, next_decision, transition_err) - if state_err then return nil, state_err end - return M.drive_decision(runtime, decision) + if not pid then return true, nil end + local dataflow_id = M.overseer_state.forget_pid(runtime.ownership, pid) + if not dataflow_id then return true, nil end + return M.reconcile(runtime, dataflow_id, { message = failure_message(event) }) end function M.next_pending_wake(): (any?, string?) diff --git a/src/runner/overseer_state.lua b/src/runner/overseer_state.lua index cbf08e8..ef31eb2 100644 --- a/src/runner/overseer_state.lua +++ b/src/runner/overseer_state.lua @@ -1,534 +1,148 @@ --- Pure ownership state for the Dataflow overseer. +-- Pure decisions for the Dataflow overseer. -- --- Durable activation is the source of truth. This state only tracks which PID --- owns an activation generation inside the current runtime. A missing owner is --- spawned while acquiring a newly observed generation (including boot). Once a --- generation has been observed, losing its owner is a terminal failure; it is --- never restarted in the same runtime. +-- The activation row carries the ownership record, written only by the +-- orchestrator: owner_token (one orchestrator incarnation), owner_epoch (its +-- runtime) and owner_phase (running or released). The canonical name +-- dataflow. is held by at most one process. From one locked observation of +-- both, the overseer decides without remembering earlier decisions: +-- - terminal or inactive: stop whoever holds the name; +-- - a name holder: monitor it; +-- - a running owner of this runtime without the name: it died, so fail the +-- activation fenced by its token, which covers every newer request; +-- - otherwise (never owned, released, or owned in an earlier runtime): spawn. local M = {} -type OwnershipRecord = { - dataflow_id: string, - generation: number, - phase: string, - pid: string?, - claim_required: boolean, - claim_from_epoch: string?, - candidate_pid: string?, +M.ACTION = { + NONE = "none", + STOP = "stop", + MONITOR = "monitor", + FAIL = "fail", + SPAWN = "spawn", } -type State = { - by_dataflow: { [string]: OwnershipRecord }, - by_pid: { [string]: string }, -} +local OWNER_RUNNING = "running" -type IdentityInput = { - dataflow_id: string, - generation: number, +local TERMINAL_STATUS = { + completed = true, + failed = true, + cancelled = true, + terminated = true, } -type ActivationInput = { +type Observation = { dataflow_id: string, - generation: number, - desired_active: boolean, status: string?, - terminal: boolean?, + desired_active: boolean, + generation: number?, + owner_token: string?, + owner_phase: string?, owner_epoch: string?, - runtime_epoch: string, -} - -type OwnerObservationInput = { - dataflow_id: string, - generation: number, registered_pid: string?, - message: string?, -} - -type ClaimObservationInput = { - dataflow_id: string, - generation: number, - claimed: boolean, -} - -type SpawnObservationInput = { - dataflow_id: string, - generation: number, - registered_pid: string?, - spawn_pid: string?, - error: string?, -} - -type MonitorObservationInput = { - dataflow_id: string, - generation: number, - pid: string, - monitor_ok: boolean, - registered_pid: string?, - error: string?, + runtime_epoch: string, } -type ExitInput = { - pid: string, +type Fence = { + token: string?, + phase: string?, generation: number?, - desired_active: boolean, - status: string?, - terminal: boolean?, - message: string?, } type Decision = { kind: string, reason: string, - dataflow_id: string?, - generation: number?, - current_generation: number?, - observed_generation: number?, - pid: string?, - phase: string?, - message: string?, - observed_epoch: string?, -} - -type DecisionDetails = { - dataflow_id: string?, + dataflow_id: string, generation: number?, - current_generation: number?, - observed_generation: number?, pid: string?, - phase: string?, - message: string?, - observed_epoch: string?, + fence: Fence?, } -M.ACTION = { - NONE = "none", - INSPECT_OWNER = "inspect_owner", - CLAIM = "claim", - REFRESH = "refresh", - SPAWN = "spawn", - MONITOR = "monitor", - STOP = "stop", - FAIL = "fail", -} - -local TERMINAL_STATUS = { - completed = true, - failed = true, - cancelled = true, - terminated = true, +type State = { + by_pid: { [string]: string }, + by_dataflow: { [string]: string }, } -local function copy_record(record: OwnershipRecord): OwnershipRecord - return { - dataflow_id = record.dataflow_id, - generation = record.generation, - phase = record.phase, - pid = record.pid, - claim_required = record.claim_required, - claim_from_epoch = record.claim_from_epoch, - candidate_pid = record.candidate_pid, - } -end - -local function get_record(state: State, dataflow_id: string): OwnershipRecord? - return state.by_dataflow[dataflow_id] -end - -local function none(reason: string, details: DecisionDetails?): Decision - details = details or {} - return { - kind = "none", - reason = reason, - dataflow_id = details.dataflow_id, - generation = details.generation, - current_generation = details.current_generation, - observed_generation = details.observed_generation, - pid = details.pid, - phase = details.phase, - message = details.message, - observed_epoch = details.observed_epoch, - } -end - -local function identity(input: IdentityInput): IdentityInput - local dataflow_id = input.dataflow_id - local generation = input.generation - assert(dataflow_id ~= "", "dataflow_id is required") - assert(generation >= 1 and generation % 1 == 0, "generation must be a positive integer") - return { dataflow_id = dataflow_id, generation = generation } +local function is_terminal(status: string?): boolean + return TERMINAL_STATUS[string.lower(tostring(status or ""))] == true end -local function is_terminal(input: { status: string?, terminal: boolean? }): boolean - return input.terminal == true or - TERMINAL_STATUS[string.lower(tostring(input.status or ""))] == true -end - -local function unbind(state: State, record: OwnershipRecord?) - if record and record.pid ~= nil then - state.by_pid[tostring(record.pid)] = nil - record.pid = nil - end -end - -local function remove(state: State, dataflow_id: string) - local record = get_record(state, dataflow_id) - if record then unbind(state, record) end - state.by_dataflow[dataflow_id] = nil -end - -local function current(state: State, dataflow_id: string, generation: number): (OwnershipRecord?, Decision) - local record = get_record(state, dataflow_id) - if not record then return nil, none("unknown_activation") end - if record.generation ~= generation then - return nil, none("stale_generation", { - current_generation = record.generation, - observed_generation = generation, - }) - end - return record, none("current_generation") -end - -local function fail(record: OwnershipRecord, reason: string, message: string?): Decision - record.phase = "failure_requested" - return { - kind = M.ACTION.FAIL, - reason = reason, - message = tostring(message or reason), - dataflow_id = record.dataflow_id, - generation = record.generation, - } -end +function M.decide(observation: Observation): Decision + local id = observation.dataflow_id + local pid = observation.registered_pid + if pid == "" then pid = nil end -local function bind(state: State, record: OwnershipRecord, pid: string) - unbind(state, record) - local key = tostring(pid) - local displaced_id = state.by_pid[key] - if displaced_id then - local other = state.by_dataflow[displaced_id] - if other then - other.pid = nil - other.phase = "verification_requested" + if is_terminal(observation.status) or observation.desired_active ~= true then + if pid then + return { + kind = M.ACTION.STOP, + reason = is_terminal(observation.status) and "terminal_owner_stop" or "inactive_owner_stop", + dataflow_id = id, + pid = pid, + } end + return { kind = M.ACTION.NONE, reason = "not_active", dataflow_id = id } end - record.pid = key - record.phase = "monitored" - state.by_pid[key] = record.dataflow_id -end -function M.new(): State - local records: { [string]: OwnershipRecord } = {} - local pids: { [string]: string } = {} - return { by_dataflow = records, by_pid = pids } -end - -function M.on_activation(state: State, input: ActivationInput): (State, Decision, string?) - local id = identity(input) - local next_state: State = state - local record = get_record(next_state, id.dataflow_id) - - if record and id.generation < record.generation then - return next_state, none("stale_activation", { - current_generation = record.generation, - observed_generation = id.generation, - }), nil + if pid then + return { kind = M.ACTION.MONITOR, reason = "registered_owner", dataflow_id = id, pid = pid } end - if is_terminal(input) or input.desired_active == false then - local stopped_pid = nil - if record and id.generation >= record.generation then - stopped_pid = record.pid - remove(next_state, id.dataflow_id) - end - if stopped_pid then - return next_state, { - kind = "stop", - reason = is_terminal(input) and "terminal_owner_stop" or "inactive_owner_stop", - dataflow_id = id.dataflow_id, - generation = id.generation, - pid = stopped_pid, - }, nil - end - return next_state, none(is_terminal(input) and "terminal" or "inactive"), nil + if observation.owner_phase == OWNER_RUNNING and observation.owner_token ~= nil and + observation.owner_epoch == observation.runtime_epoch then + return { + kind = M.ACTION.FAIL, + reason = "runtime_owner_lost", + dataflow_id = id, + generation = observation.generation, + fence = { token = observation.owner_token, phase = OWNER_RUNNING }, + } end - assert(input.desired_active == true, "desired_active must be a boolean") - if record and id.generation == record.generation then - if record.phase == "monitored" then - record.phase = "verification_requested" - return next_state, { - kind = M.ACTION.INSPECT_OWNER, - reason = "verify_active_owner", - dataflow_id = id.dataflow_id, - generation = id.generation, - }, nil - end - if record.phase == "failure_requested" then - return next_state, fail(record, "retry_failure_persistence", - "active generation previously lost its runtime owner"), nil - end - return next_state, none("activation_in_flight", { phase = record.phase }), nil - end - - if record and id.generation > record.generation then - -- A live orchestrator may own a sequence of activation generations - -- (for example, one signal after another). Process monitoring belongs - -- to that stable owner, not to an individual generation; retain the - -- monitor while the durable generation fence advances. - if record.pid ~= nil then - record.generation = id.generation - record.phase = "acquisition_requested" - record.claim_required = input.owner_epoch ~= input.runtime_epoch - record.claim_from_epoch = input.owner_epoch - record.candidate_pid = nil - return next_state, { - kind = M.ACTION.INSPECT_OWNER, - reason = "advance_monitored_owner", - dataflow_id = id.dataflow_id, - generation = id.generation, - }, nil - end - - -- Once an observed owner is lost, a later activation cannot turn that - -- same-runtime failure into a restart. Fence the newest generation so - -- failure persistence wins even if a signal raced the EXIT event. - if record.phase == "failure_requested" or record.phase == "verification_requested" then - record.generation = id.generation - return next_state, fail(record :: OwnershipRecord, "runtime_owner_lost", - "active orchestrator disappeared during runtime"), nil - end - end - - remove(next_state, id.dataflow_id) - next_state.by_dataflow[id.dataflow_id] = { - dataflow_id = id.dataflow_id, - generation = id.generation, - phase = "acquisition_requested", - claim_required = input.owner_epoch ~= input.runtime_epoch, - claim_from_epoch = input.owner_epoch, + return { + kind = M.ACTION.SPAWN, + reason = "activation_unowned", + dataflow_id = id, + generation = observation.generation, + fence = { + token = observation.owner_token, + phase = observation.owner_phase, + generation = observation.generation, + }, } - return next_state, { - kind = M.ACTION.INSPECT_OWNER, - reason = "acquire_activation", - dataflow_id = id.dataflow_id, - generation = id.generation, - }, nil -end - -function M.on_owner_observation(state: State, input: OwnerObservationInput): (State, Decision, string?) - local id = identity(input) - local next_state: State = state - local record, stale = current(next_state, id.dataflow_id, id.generation) - if not record then return next_state, stale, nil end - - local pid = input.registered_pid - if pid ~= nil and tostring(pid) ~= "" then - if record.pid ~= nil and tostring(record.pid) == tostring(pid) and - record.claim_required == false then - record.phase = "monitored" - return next_state, none("existing_owner_verified", { - dataflow_id = id.dataflow_id, - generation = id.generation, - pid = tostring(pid), - }), nil - end - if record.phase == "acquisition_requested" and record.claim_required then - record.phase = "claim_requested" - record.candidate_pid = tostring(pid) - local claim_reason = "new_owner_adoption_claim" - local observed_epoch: string? = record.claim_from_epoch - if observed_epoch ~= nil then claim_reason = "reboot_owner_adoption_claim" end - return next_state, { - kind = M.ACTION.CLAIM, - reason = claim_reason, - dataflow_id = id.dataflow_id, - generation = id.generation, - observed_epoch = observed_epoch, - }, nil - end - record.phase = "monitor_requested" - return next_state, { - kind = M.ACTION.MONITOR, - reason = "registered_owner", - dataflow_id = id.dataflow_id, - generation = id.generation, - pid = tostring(pid), - }, nil - end - - if record.pid ~= nil then - unbind(next_state, record) - return next_state, fail(record, "runtime_owner_lost", - input.message or "active orchestrator disappeared during runtime"), nil - end - - if record.phase == "acquisition_requested" and record.claim_required then - record.phase = "claim_requested" - local claim_reason = "new_activation_claim" - local observed_epoch: string? = record.claim_from_epoch - if observed_epoch ~= nil then claim_reason = "reboot_recovery_claim" end - return next_state, { - kind = M.ACTION.CLAIM, - reason = claim_reason, - dataflow_id = id.dataflow_id, - generation = id.generation, - observed_epoch = observed_epoch, - }, nil - end - if record.phase == "acquisition_requested" then - return next_state, fail(record, "same_runtime_owner_missing", - "active orchestrator disappeared in the current runtime epoch"), nil - end - return next_state, fail(record, "runtime_owner_lost", - input.message or "active orchestrator disappeared during runtime"), nil -end - -function M.on_claim_observation(state: State, input: ClaimObservationInput): (State, Decision, string?) - local id = identity(input) - local next_state: State = state - local record, stale = current(next_state, id.dataflow_id, id.generation) - if not record then return next_state, stale, nil end - if input.claimed == true then - record.claim_required = false - if record.candidate_pid then - if record.pid ~= nil and tostring(record.pid) == tostring(record.candidate_pid) then - record.phase = "monitored" - return next_state, none("existing_owner_claimed", { - dataflow_id = id.dataflow_id, - generation = id.generation, - pid = record.pid, - }), nil - end - record.phase = "monitor_requested" - return next_state, { - kind = M.ACTION.MONITOR, - reason = "activation_owner_epoch_claimed", - dataflow_id = id.dataflow_id, - generation = id.generation, - pid = record.candidate_pid, - }, nil - end - record.phase = "spawn_requested" - return next_state, { - kind = M.ACTION.SPAWN, - reason = "activation_epoch_claimed", - dataflow_id = id.dataflow_id, - generation = id.generation, - }, nil - end - remove(next_state, id.dataflow_id) - return next_state, { - kind = M.ACTION.REFRESH, - reason = "activation_epoch_claim_lost", - dataflow_id = id.dataflow_id, - generation = id.generation, - }, nil end -function M.on_spawn_observation(state: State, input: SpawnObservationInput): (State, Decision, string?) - local id = identity(input) - local next_state: State = state - local record, stale = current(next_state, id.dataflow_id, id.generation) - if not record then return next_state, stale, nil end - - local registered_pid = input.registered_pid - if registered_pid ~= nil and tostring(registered_pid) == "" then registered_pid = nil end - local spawn_pid = input.spawn_pid - if spawn_pid ~= nil and tostring(spawn_pid) == "" then spawn_pid = nil end - local candidate = registered_pid or spawn_pid - if candidate then - record.phase = "monitor_requested" - return next_state, { - kind = M.ACTION.MONITOR, - reason = registered_pid and "canonical_owner_after_spawn" or "spawned_owner", - dataflow_id = id.dataflow_id, - generation = id.generation, - pid = tostring(candidate), - }, nil - end - return next_state, fail(record, "orchestrator_spawn_failed", - input.error or "orchestrator spawn returned no owner"), nil +function M.new(): State + return { by_pid = {}, by_dataflow = {} } end -function M.on_monitor_observation(state: State, input: MonitorObservationInput): (State, Decision, string?) - local id = identity(input) - local next_state: State = state - local record, stale = current(next_state, id.dataflow_id, id.generation) - if not record then return next_state, stale, nil end - - local pid = input.pid - if input.monitor_ok == true and pid ~= nil and tostring(pid) ~= "" then - bind(next_state, record, pid) - return next_state, none("owner_monitored", { pid = tostring(pid) }), nil - end - local registered_pid = input.registered_pid - if registered_pid and tostring(registered_pid) ~= "" and - tostring(registered_pid) ~= tostring(pid or "") then - record.phase = "monitor_requested" - return next_state, { - kind = M.ACTION.MONITOR, - reason = "owner_changed_during_monitor", - dataflow_id = id.dataflow_id, - generation = id.generation, - pid = tostring(registered_pid), - }, nil - end - return next_state, fail(record, "orchestrator_monitor_failed", - input.error or "canonical orchestrator could not be monitored"), nil +-- Record the monitored process of a dataflow; an earlier process stays routable +-- until its EXIT arrives. +function M.track(state: State, dataflow_id: string, pid: string) + state.by_pid[pid] = dataflow_id + state.by_dataflow[dataflow_id] = pid end -function M.on_exit(state: State, input: ExitInput): (State, Decision, string?) - local pid = input.pid - assert(pid ~= "", "pid is required") - local next_state: State = state - local dataflow_id = next_state.by_pid[pid] - if not dataflow_id then return next_state, none("stale_exit"), nil end - local record = next_state.by_dataflow[dataflow_id] - if not record then - next_state.by_pid[pid] = nil - return next_state, none("stale_exit_owner"), nil - end - if input.generation and tonumber(input.generation) ~= record.generation then - return next_state, none("stale_exit_generation"), nil - end - if tostring(record.pid or "") ~= pid then - next_state.by_pid[pid] = nil - return next_state, none("stale_exit_owner"), nil - end - unbind(next_state, record) - if is_terminal(input) or input.desired_active == false then - remove(next_state, record.dataflow_id) - return next_state, none(is_terminal(input) and "terminal_exit" or "inactive_exit"), nil - end - assert(input.desired_active == true, "desired_active must be a boolean") - record.phase = "verification_requested" - return next_state, { - kind = M.ACTION.INSPECT_OWNER, - reason = "verify_after_exit", - message = input.message, - dataflow_id = record.dataflow_id, - generation = record.generation, - }, nil +function M.forget_pid(state: State, pid: string): string? + local dataflow_id = state.by_pid[pid] + if not dataflow_id then return nil end + state.by_pid[pid] = nil + if state.by_dataflow[dataflow_id] == pid then state.by_dataflow[dataflow_id] = nil end + return dataflow_id end -function M.on_failed(state: State, input: IdentityInput): (State, Decision, string?) - local id = identity(input) - local next_state: State = state - local record = next_state.by_dataflow[id.dataflow_id] - if record and record.generation == id.generation then remove(next_state, id.dataflow_id) end - return next_state, none("failure_persisted"), nil +function M.forget_dataflow(state: State, dataflow_id: string) + local pid = state.by_dataflow[dataflow_id] + if pid then state.by_pid[pid] = nil end + state.by_dataflow[dataflow_id] = nil end -function M.owner_for_dataflow(state: State, dataflow_id: string): OwnershipRecord? - local record = get_record(state, dataflow_id) - return record and copy_record(record) or nil +function M.pid_for(state: State, dataflow_id: string): string? + return state.by_dataflow[dataflow_id] end -function M.owner_for_pid(state: State, pid: string): { dataflow_id: string, generation: number }? - local dataflow_id = state.by_pid[pid] - if not dataflow_id then return nil end - local record = state.by_dataflow[dataflow_id] - if not record then return nil end - return { dataflow_id = dataflow_id, generation = record.generation } +function M.tracked(state: State): { string } + local ids = {} + for dataflow_id in pairs(state.by_dataflow) do table.insert(ids, dataflow_id) end + return ids end return M diff --git a/src/runner/overseer_state_test.lua b/src/runner/overseer_state_test.lua index a3fc55f..9188e2a 100644 --- a/src/runner/overseer_state_test.lua +++ b/src/runner/overseer_state_test.lua @@ -1,344 +1,102 @@ local test = require("test") local overseer: any = require("overseer_state") -local on_activation: any = overseer.on_activation -local on_owner_observation: any = overseer.on_owner_observation -local on_claim_observation: any = overseer.on_claim_observation -local on_spawn_observation: any = overseer.on_spawn_observation -local on_monitor_observation: any = overseer.on_monitor_observation -local on_exit: any = overseer.on_exit -local on_failed: any = overseer.on_failed local CURRENT_EPOCH = "runtime-current" -local function required(value: any): any - return test.not_nil(value) :: any -end - -local function activate(state, id: string, generation: number) - local next_state, decision, err = on_activation(state, { - dataflow_id = id, - generation = generation, - desired_active = true, +type Observation = { + dataflow_id: string, + status: string?, + desired_active: boolean, + generation: number?, + owner_token: string?, + owner_phase: string?, + owner_epoch: string?, + registered_pid: string?, + runtime_epoch: string, +} + +local function observe(fields: any): Observation + local observation: Observation = { + dataflow_id = "df", status = "running", + desired_active = true, + generation = 3, runtime_epoch = CURRENT_EPOCH, - }) - test.is_nil(err) - decision = required(decision) - test.eq(decision.kind, overseer.ACTION.INSPECT_OWNER) - return next_state + } + local values = fields or {} + if values.status ~= nil then observation.status = values.status end + if values.desired_active ~= nil then observation.desired_active = values.desired_active == true end + observation.owner_token = values.owner_token + observation.owner_phase = values.owner_phase + observation.owner_epoch = values.owner_epoch + observation.registered_pid = values.registered_pid + return observation end -local function acquire(state, id: string, generation: number, pid: string) - local inspected, claim = on_owner_observation(state, { - dataflow_id = id, - generation = generation, - }) - claim = required(claim) - test.eq(claim.kind, overseer.ACTION.CLAIM) - local claimed, spawn = on_claim_observation(inspected, { - dataflow_id = id, - generation = generation, - claimed = true, - }) - spawn = required(spawn) - test.eq(spawn.kind, overseer.ACTION.SPAWN) - local spawned, monitor = on_spawn_observation(claimed, { - dataflow_id = id, - generation = generation, - spawn_pid = pid, - }) - monitor = required(monitor) - test.eq(monitor.kind, overseer.ACTION.MONITOR) - local tracked, settled = on_monitor_observation(spawned, { - dataflow_id = id, - generation = generation, - pid = pid, - monitor_ok = true, - }) - settled = required(settled) - test.eq(settled.kind, overseer.ACTION.NONE) - test.eq(settled.reason, "owner_monitored") - return tracked +local function decide(observation: Observation): any + return overseer.decide(observation) end local function run_tests() - test.describe("Pure Dataflow overseer ownership state", function() - test.it("acquires a boot activation exactly once and indexes its canonical owner", function() - local state = acquire(activate(overseer.new(), "df-boot", 1), "df-boot", 1, "pid-boot") - local owner = test.not_nil(overseer.owner_for_dataflow(state, "df-boot")) - test.eq(owner.pid, "pid-boot") - test.eq(owner.generation, 1) - local reverse = test.not_nil(overseer.owner_for_pid(state, "pid-boot")) - test.eq(reverse.dataflow_id, "df-boot") - test.eq(reverse.generation, 1) - end) - - test.it("verifies a monitored duplicate notification without requesting another spawn", function() - local state = acquire(activate(overseer.new(), "df-live", 2), "df-live", 2, "pid-live") - local verifying, inspect = on_activation(state, { - dataflow_id = "df-live", generation = 2, desired_active = true, - owner_epoch = CURRENT_EPOCH, runtime_epoch = CURRENT_EPOCH, - }) - inspect = required(inspect) - test.eq(inspect.kind, overseer.ACTION.INSPECT_OWNER) - test.eq(inspect.reason, "verify_active_owner") - local tracked, verified = on_owner_observation(verifying, { - dataflow_id = "df-live", generation = 2, registered_pid = "pid-live", - }) - verified = required(verified) - test.eq(verified.kind, overseer.ACTION.NONE) - test.eq(verified.reason, "existing_owner_verified") - test.eq((test.not_nil(overseer.owner_for_dataflow(tracked, "df-live"))).pid, "pid-live") - end) - - test.it("turns a missing runtime owner into failure and never into restart", function() - local state = acquire(activate(overseer.new(), "df-fail", 3), "df-fail", 3, "pid-fail") - local exited, inspect = on_exit(state, { - pid = "pid-fail", - generation = 3, - desired_active = true, - message = "host process crashed", - }) - inspect = required(inspect) - test.eq(inspect.kind, overseer.ACTION.INSPECT_OWNER) - local failing, failure = on_owner_observation(exited, { - dataflow_id = "df-fail", - generation = 3, - message = tostring(inspect.message), - }) - failure = required(failure) - test.eq(failure.kind, overseer.ACTION.FAIL) - test.eq(failure.reason, "runtime_owner_lost") - test.eq(failure.message, "host process crashed") - test.eq((test.not_nil(overseer.owner_for_dataflow(failing, "df-fail"))).phase, - "failure_requested") - end) - - test.it("adopts a canonical race winner instead of failing or duplicating it", function() - local state = acquire(activate(overseer.new(), "df-race", 4), "df-race", 4, "pid-loser") - local exited, inspect = on_exit(state, { - pid = "pid-loser", generation = 4, desired_active = true, - }) - inspect = required(inspect) - local next_state, monitor = on_owner_observation(exited, { - dataflow_id = "df-race", generation = 4, registered_pid = "pid-winner", - }) - monitor = required(monitor) - test.eq(inspect.kind, overseer.ACTION.INSPECT_OWNER) - test.eq(monitor.kind, overseer.ACTION.MONITOR) - test.eq(monitor.pid, "pid-winner") - test.eq((test.not_nil(overseer.owner_for_dataflow(next_state, "df-race"))).phase, - "monitor_requested") - end) - - test.it("claims a newly registered synchronous owner before adopting it", function() - local state = activate(overseer.new(), "df-direct", 1) - local claiming, claim = on_owner_observation(state, { - dataflow_id = "df-direct", - generation = 1, - registered_pid = "pid-direct", - }) - claim = required(claim) - test.eq(claim.kind, overseer.ACTION.CLAIM) - test.eq(claim.reason, "new_owner_adoption_claim") - test.is_nil(claim.observed_epoch) - - local monitoring, monitor = on_claim_observation(claiming, { - dataflow_id = "df-direct", - generation = 1, - claimed = true, - }) - monitor = required(monitor) - test.eq(monitor.kind, overseer.ACTION.MONITOR) - test.eq(monitor.reason, "activation_owner_epoch_claimed") - test.eq(monitor.pid, "pid-direct") - test.eq((test.not_nil(overseer.owner_for_dataflow( - monitoring, "df-direct"))).phase, "monitor_requested") - end) - - test.it("distinguishes full-runtime recovery from an overseer-only restart", function() - local reboot_state, reboot_inspect = on_activation(overseer.new(), { - dataflow_id = "df-reboot", - generation = 4, - desired_active = true, - owner_epoch = "runtime-before", - runtime_epoch = CURRENT_EPOCH, - }) - test.eq(required(reboot_inspect).kind, overseer.ACTION.INSPECT_OWNER) - local _, reboot_claim = on_owner_observation(reboot_state, { - dataflow_id = "df-reboot", generation = 4, - }) - reboot_claim = required(reboot_claim) - test.eq(reboot_claim.kind, overseer.ACTION.CLAIM) - test.eq(reboot_claim.reason, "reboot_recovery_claim") - test.eq(reboot_claim.observed_epoch, "runtime-before") - - local same_state, same_inspect = on_activation(overseer.new(), { - dataflow_id = "df-service-restart", - generation = 5, - desired_active = true, - owner_epoch = CURRENT_EPOCH, - runtime_epoch = CURRENT_EPOCH, - }) - test.eq(required(same_inspect).kind, overseer.ACTION.INSPECT_OWNER) - local _, failure = on_owner_observation(same_state, { - dataflow_id = "df-service-restart", generation = 5, - }) - failure = required(failure) - test.eq(failure.kind, overseer.ACTION.FAIL) - test.eq(failure.reason, "same_runtime_owner_missing") - end) - - test.it("stops a live owner after a durable terminal transition", function() - local state = acquire(activate(overseer.new(), "df-cancel", 5), "df-cancel", 5, "pid-cancel") - local stopped, decision = on_activation(state, { - dataflow_id = "df-cancel", - generation = 5, - desired_active = false, - status = "cancelled", - owner_epoch = CURRENT_EPOCH, - runtime_epoch = CURRENT_EPOCH, - }) - decision = required(decision) - test.eq(decision.kind, overseer.ACTION.STOP) - test.eq(decision.pid, "pid-cancel") - test.is_nil(overseer.owner_for_dataflow(stopped, "df-cancel")) - test.is_nil(overseer.owner_for_pid(stopped, "pid-cancel")) - end) - - test.it("never acquires inactive or terminal activations", function() - local inactive, inactive_decision = on_activation(overseer.new(), { - dataflow_id = "df-waiting", generation = 1, desired_active = false, status = "waiting", - runtime_epoch = CURRENT_EPOCH, - }) - inactive_decision = required(inactive_decision) - test.eq(inactive_decision.kind, overseer.ACTION.NONE) - test.eq(inactive_decision.reason, "inactive") - test.is_nil(overseer.owner_for_dataflow(inactive, "df-waiting")) - - local terminal, terminal_decision = on_activation(overseer.new(), { - dataflow_id = "df-done", generation = 1, desired_active = true, status = "failed", - runtime_epoch = CURRENT_EPOCH, - }) - terminal_decision = required(terminal_decision) - test.eq(terminal_decision.kind, overseer.ACTION.NONE) - test.eq(terminal_decision.reason, "terminal") - test.is_nil(overseer.owner_for_dataflow(terminal, "df-done")) - end) - - test.it("generation fences stale notifications and stale EXIT events", function() - local state = acquire(activate(overseer.new(), "df-current", 7), "df-current", 7, "pid-current") - local unchanged, stale = on_activation(state, { - dataflow_id = "df-current", generation = 6, desired_active = true, - owner_epoch = CURRENT_EPOCH, runtime_epoch = CURRENT_EPOCH, - }) - stale = required(stale) - test.eq(stale.reason, "stale_activation") - local after_exit, stale_exit = on_exit(unchanged, { - pid = "pid-current", generation = 6, desired_active = true, - }) - stale_exit = required(stale_exit) - test.eq(stale_exit.reason, "stale_exit_generation") - test.eq((test.not_nil(overseer.owner_for_dataflow(after_exit, "df-current"))).pid, - "pid-current") - end) - - test.it("advances a live owner's generation without re-monitoring its process", function() - local state = acquire(activate(overseer.new(), "df-sequential", 1), - "df-sequential", 1, "pid-sequential") - local advancing, inspect = on_activation(state, { - dataflow_id = "df-sequential", generation = 2, desired_active = true, - runtime_epoch = CURRENT_EPOCH, - }) - inspect = required(inspect) - test.eq(inspect.kind, overseer.ACTION.INSPECT_OWNER) - test.eq(inspect.reason, "advance_monitored_owner") - test.eq((test.not_nil(overseer.owner_for_pid( - advancing, "pid-sequential"))).generation, 2) - - local claiming, claim = on_owner_observation(advancing, { - dataflow_id = "df-sequential", generation = 2, - registered_pid = "pid-sequential", - }) - claim = required(claim) - test.eq(claim.kind, overseer.ACTION.CLAIM) - local tracked, settled = on_claim_observation(claiming, { - dataflow_id = "df-sequential", generation = 2, claimed = true, - }) - settled = required(settled) - test.eq(settled.kind, overseer.ACTION.NONE) - test.eq(settled.reason, "existing_owner_claimed") - local owner = test.not_nil(overseer.owner_for_dataflow(tracked, "df-sequential")) - test.eq(owner.pid, "pid-sequential") - test.eq(owner.generation, 2) - test.eq(owner.phase, "monitored") + test.describe("Pure Dataflow overseer decisions", function() + test.it("monitors whichever process holds the canonical name", function() + local decision = decide(observe({ + registered_pid = "pid-live", + owner_token = "t1", owner_phase = "running", owner_epoch = CURRENT_EPOCH, + })) + test.eq(decision.kind, overseer.ACTION.MONITOR) + test.eq(decision.pid, "pid-live") end) - test.it("does not let a racing activation resurrect a lost runtime owner", function() - local state = acquire(activate(overseer.new(), "df-racing-loss", 4), - "df-racing-loss", 4, "pid-racing-loss") - local exited = select(1, on_exit(state, { - pid = "pid-racing-loss", generation = 4, desired_active = true, - message = "killed", + test.it("fails a running owner of this runtime that lost its name, fenced by its token", function() + local decision = decide(observe({ + owner_token = "t1", owner_phase = "running", owner_epoch = CURRENT_EPOCH, })) - local failing, failure = on_activation(exited, { - dataflow_id = "df-racing-loss", generation = 5, desired_active = true, - runtime_epoch = CURRENT_EPOCH, - }) - failure = required(failure) - test.eq(failure.kind, overseer.ACTION.FAIL) - test.eq(failure.generation, 5) - test.eq(failure.reason, "runtime_owner_lost") - local owner = test.not_nil(overseer.owner_for_dataflow(failing, "df-racing-loss")) - test.eq(owner.generation, 5) - test.eq(owner.phase, "failure_requested") + test.eq(decision.kind, overseer.ACTION.FAIL) + test.eq(decision.reason, "runtime_owner_lost") + test.eq(decision.fence.token, "t1") + test.eq(decision.fence.phase, "running") + test.is_nil(decision.fence.generation, "the owner's token covers every newer request") end) - test.it("fails spawn and monitor acquisition errors without a retry action", function() - local state = activate(overseer.new(), "df-spawn-error", 1) - local inspected = select(1, on_owner_observation(state, { - dataflow_id = "df-spawn-error", generation = 1, - })) - local _, spawn_failure = on_spawn_observation(inspected, { - dataflow_id = "df-spawn-error", generation = 1, error = "host unavailable", - }) - spawn_failure = required(spawn_failure) - test.eq(spawn_failure.kind, overseer.ACTION.FAIL) - test.eq(spawn_failure.reason, "orchestrator_spawn_failed") + test.it("spawns for an unowned, released or earlier-runtime activation", function() + for _, owner in ipairs({ + {}, + { owner_token = "t1", owner_phase = "released", owner_epoch = CURRENT_EPOCH }, + { owner_token = "t1", owner_phase = "running", owner_epoch = "runtime-before" }, + }) do + local decision = decide(observe(owner)) + test.eq(decision.kind, overseer.ACTION.SPAWN) + test.eq(decision.generation, 3) + test.eq(decision.fence.token, owner.owner_token) + test.eq(decision.fence.phase, owner.owner_phase) + test.eq(decision.fence.generation, 3) + end + end) - state = activate(overseer.new(), "df-monitor-error", 1) - local _, monitor_failure = on_monitor_observation(state, { - dataflow_id = "df-monitor-error", - generation = 1, - pid = "pid-missing", - monitor_ok = false, - error = "not found", - }) - monitor_failure = required(monitor_failure) - test.eq(monitor_failure.kind, overseer.ACTION.FAIL) - test.eq(monitor_failure.reason, "orchestrator_monitor_failed") + test.it("stops a name holder of a terminal or inactive activation and otherwise does nothing", function() + for _, state in ipairs({ + { status = "failed" }, + { desired_active = false, status = "waiting" }, + }) do + local named = observe(state) + named.registered_pid = "pid-old" + local stop = decide(named) + test.eq(stop.kind, overseer.ACTION.STOP) + test.eq(stop.pid, "pid-old") + test.eq(decide(observe(state)).kind, overseer.ACTION.NONE) + end end) - test.it("retries only failure persistence and removes state after it commits", function() - local state = activate(overseer.new(), "df-persist", 8) - local inspected = select(1, on_owner_observation(state, { - dataflow_id = "df-persist", generation = 8, - })) - local failing = select(1, on_spawn_observation(inspected, { - dataflow_id = "df-persist", generation = 8, error = "spawn failed", - })) - local repeated, retry = on_activation(failing, { - dataflow_id = "df-persist", generation = 8, desired_active = true, - owner_epoch = CURRENT_EPOCH, runtime_epoch = CURRENT_EPOCH, - }) - retry = required(retry) - test.eq(retry.kind, overseer.ACTION.FAIL) - test.eq(retry.reason, "retry_failure_persistence") - local cleared, done = on_failed(repeated, { - dataflow_id = "df-persist", generation = 8, - }) - done = required(done) - test.eq(done.reason, "failure_persisted") - test.is_nil(overseer.owner_for_dataflow(cleared, "df-persist")) + test.it("tracks one monitored process per dataflow and routes its EXIT", function() + local state = overseer.new() + overseer.track(state, "df", "pid-1") + overseer.track(state, "df", "pid-2") + test.eq(overseer.pid_for(state, "df"), "pid-2") + test.eq(overseer.forget_pid(state, "pid-1"), "df") + test.eq(overseer.pid_for(state, "df"), "pid-2", "a late EXIT leaves the current owner tracked") + test.eq(overseer.forget_pid(state, "pid-2"), "df") + test.is_nil(overseer.pid_for(state, "df")) + test.is_nil(overseer.forget_pid(state, "pid-unknown")) end) end) end diff --git a/src/runner/overseer_test.lua b/src/runner/overseer_test.lua index f0ad02c..b767c9b 100644 --- a/src/runner/overseer_test.lua +++ b/src/runner/overseer_test.lua @@ -36,8 +36,8 @@ local function captures() cancels = {}, terminates = {}, failures = {}, - claims = {}, reconstructions = {}, + locked_reads = {}, } end @@ -84,7 +84,8 @@ local function process_mock(captured) return self end function spawner:spawn_monitored(source, host, args) - local pid = "pid-" .. tostring(args.dataflow_id) .. "-" .. tostring(args.activation_generation) + if captured.owners[self.name] then return nil, "name already registered" end + local pid = "pid-" .. tostring(args.dataflow_id) .. "-" .. tostring(#captured.spawns + 1) captured.owners[self.name] = pid table.insert(captured.spawns, { source = source, @@ -103,12 +104,36 @@ local function process_mock(captured) return mock end +-- The overseer never writes ownership: these helpers play the orchestrator. +local function admit(row: any, token: string, pid: string, runtime_epoch: string?) + row.owner_token = token + row.owner_pid = pid + row.owner_epoch = runtime_epoch or CURRENT_EPOCH + row.owner_phase = "running" +end + +local function release(row: any) + row.owner_phase = "released" + row.desired_active = false +end + +local function advance(row: any) + row.generation = row.generation + 1 + row.desired_active = true +end + +local function fence_matches(row: any, fence: any): boolean + if row.owner_token ~= fence.token or row.owner_phase ~= fence.phase then return false end + return fence.generation == nil or row.generation == fence.generation +end + local function run_tests() test.describe("Dataflow overseer IO", function() local originals local observed local activations: { [string]: any } = {} local workflows: { [string]: any } = {} + local locked_read_errors: { string } = {} test.before_each(function() originals = { @@ -124,6 +149,7 @@ local function run_tests() observed = captures() activations = {} :: { [string]: any } workflows = {} :: { [string]: any } + locked_read_errors = {} :: { string } overseer.process = process_mock(observed) overseer.execution_frame = { reconstruct = function(actor_id, actor_context) @@ -135,24 +161,12 @@ local function run_tests() end, } overseer.activation_repo = { - get = function(id) return activations[id], nil end, - claim_epoch_tx = function(_tx, id, generation, observed_epoch, runtime_epoch) - local row = activations[id] - local matches = row and row.generation == generation and row.desired_active and - row.owner_epoch == observed_epoch - table.insert(observed.claims, { - dataflow_id = id, - generation = generation, - observed_epoch = observed_epoch, - runtime_epoch = runtime_epoch, - claimed = matches == true, - }) - if matches then row.owner_epoch = runtime_epoch end - if not row then return nil, "activation missing" end - local result = {} - for key, value in pairs(row) do result[key] = value end - result.claimed = matches == true - return result, nil + read_locked_tx = function(_tx, id) + table.insert(observed.locked_reads, id) + local read_err = table.remove(locked_read_errors, 1) + if read_err then return nil, read_err end + local status = workflows[id] and workflows[id].status or nil + return { activation = activations[id], status = status }, nil end, list_active = function() local rows = {} @@ -169,15 +183,23 @@ local function run_tests() end, } overseer.commit = { - fail_activation = function(id, generation, failure) + fail_activation = function(id, fence, failure) + local row = activations[id] + local completed = row ~= nil and row.desired_active == true and fence_matches(row, fence) table.insert(observed.failures, { dataflow_id = id, - generation = generation, + generation = row and row.generation or nil, + fence = fence, failure = failure, + completed = completed, }) - activations[id].desired_active = false + if not completed then + return { completed = false, current_generation = row and row.generation }, nil + end + row.desired_active = false + row.owner_phase = row.owner_token and "released" or row.owner_phase workflows[id].status = overseer.consts.STATUS.COMPLETED_FAILURE - return { completed = true, current_generation = generation }, nil + return { completed = true, current_generation = row.generation }, nil end, } overseer.with_tx = function(fn) return fn({}) end @@ -188,9 +210,16 @@ local function run_tests() for key, value in pairs(originals) do overseer[key] = value end end) + local function completed_failures(): { any } + local done = {} + for _, failure in ipairs(observed.failures) do + if failure.completed then table.insert(done, failure) end + end + return done + end + test.it("recovers each durable boot activation once under its frozen actor and scope", function() activations.boot = activation("boot", 3, { init_func_id = "app:init" }) - activations.boot.owner_epoch = "runtime-before-restart" workflows.boot = workflow("boot") local runtime = overseer.new_runtime(CURRENT_EPOCH) local count, err = overseer.bootstrap(runtime) @@ -203,11 +232,10 @@ local function run_tests() test.eq(spawn.source, overseer.consts.ORCHESTRATOR) test.eq(spawn.host, overseer.consts.HOST_ID) test.eq(spawn.args.activation_generation, 3) + test.eq(spawn.args.runtime_epoch, CURRENT_EPOCH) test.eq(spawn.args.init_func_id, "app:init") test.eq(spawn.actor, "restored:actor:boot") test.eq(spawn.scope, "scope:actor:boot") - test.eq(#observed.claims, 1) - test.eq(observed.claims[1].observed_epoch, "runtime-before-restart") local second, second_err = overseer.safety_reconcile(runtime) test.is_nil(second_err) @@ -217,58 +245,145 @@ local function run_tests() test.it("adopts an existing canonical owner without reconstructing or spawning", function() activations.live = activation("live", 1) + admit(activations.live, "t-live", "pid-existing") workflows.live = workflow("live") observed.owners["dataflow.live"] = "pid-existing" - local ok, err = overseer.reconcile_activation( - overseer.new_runtime(CURRENT_EPOCH), test.not_nil(activations.live) :: any) + local ok, err = overseer.reconcile(overseer.new_runtime(CURRENT_EPOCH), "live") test.is_nil(err) test.is_true(ok) test.eq(#observed.spawns, 0) test.eq(#observed.reconstructions, 0) - test.eq(#observed.claims, 1) - test.eq(observed.claims[1].runtime_epoch, CURRENT_EPOCH) test.eq(observed.monitors[1], "pid-existing") end) - test.it("accepts the idempotent monitor result from spawn_monitored", function() + test.it("accepts the idempotent monitor result for an already monitored owner", function() activations.monitored = activation("monitored", 1) workflows.monitored = workflow("monitored") + observed.owners["dataflow.monitored"] = "pid-monitored" overseer.process.monitor = function(pid) table.insert(observed.monitors, tostring(pid)) return nil, "already monitoring pid" end - - local ok, err = overseer.reconcile_activation( - overseer.new_runtime(CURRENT_EPOCH), activations.monitored) + local runtime = overseer.new_runtime(CURRENT_EPOCH) + local ok, err = overseer.reconcile(runtime, "monitored") test.is_nil(err) test.is_true(ok) + test.eq(#observed.spawns, 0) + test.eq(#observed.failures, 0) + test.eq(overseer.overseer_state.pid_for(runtime.ownership, "monitored"), "pid-monitored") + end) + + test.it("spawns the successor at once when a released owner's EXIT is still pending", function() + activations.handoff = activation("handoff", 1) + workflows.handoff = workflow("handoff") + local runtime = overseer.new_runtime(CURRENT_EPOCH) + test.is_true(select(1, overseer.reconcile(runtime, "handoff"))) + local first = observed.spawns[1] + admit(activations.handoff, "t1", tostring(first.pid)) + + release(activations.handoff) + observed.owners["dataflow.handoff"] = nil + advance(activations.handoff) + test.is_true(select(1, overseer.handle_activation_hint(runtime, { dataflow_id = "handoff" }))) + test.eq(#observed.spawns, 2) + test.eq(observed.spawns[2].args.activation_generation, 2) + test.eq(#observed.failures, 0) + + local handled, exit_err = overseer.handle_exit(runtime, { + kind = overseer.process.event.EXIT, from = first.pid, + result = { value = { success = true, passivated = true } }, + }) + test.is_nil(exit_err) + test.is_true(handled) + test.eq(#observed.spawns, 2) + test.eq(#observed.failures, 0) + test.eq(overseer.overseer_state.pid_for(runtime.ownership, "handoff"), observed.spawns[2].pid) + end) + + test.it("waits for a released owner that still holds the name, then spawns on its EXIT", function() + activations.draining = activation("draining", 1) + workflows.draining = workflow("draining") + local runtime = overseer.new_runtime(CURRENT_EPOCH) + test.is_true(select(1, overseer.reconcile(runtime, "draining"))) + local first = observed.spawns[1] + admit(activations.draining, "t1", tostring(first.pid)) + release(activations.draining) + advance(activations.draining) + + test.is_true(select(1, overseer.handle_activation_hint(runtime, { dataflow_id = "draining" }))) + test.eq(#observed.spawns, 1, "the name holder is monitored, not replaced") + + observed.owners["dataflow.draining"] = nil + test.is_true(select(1, overseer.handle_exit(runtime, { + kind = overseer.process.event.EXIT, from = first.pid, result = { value = { passivated = true } }, + }))) + test.eq(#observed.spawns, 2) + test.eq(#observed.failures, 0) + end) + + test.it("spawns exactly once while requests advance between observation and spawn", function() + activations.burst = activation("burst", 1) + workflows.burst = workflow("burst") + local runtime = overseer.new_runtime(CURRENT_EPOCH) + local reconstruct = overseer.execution_frame.reconstruct + overseer.execution_frame.reconstruct = function(actor_id, actor_context) + for _ = 1, 5 do advance(activations.burst) end + return reconstruct(actor_id, actor_context) + end + test.is_true(select(1, overseer.reconcile(runtime, "burst"))) + for _ = 1, 5 do + advance(activations.burst) + test.is_true(select(1, overseer.handle_activation_hint(runtime, { dataflow_id = "burst" }))) + end test.eq(#observed.spawns, 1) test.eq(#observed.failures, 0) - test.eq(observed.monitors[1], observed.spawns[1].pid) end) - test.it("fails a missing owner after an overseer-only restart in the same runtime epoch", function() - activations.service_restart = activation("service_restart", 7) - activations.service_restart.owner_epoch = CURRENT_EPOCH - workflows.service_restart = workflow("service_restart") + test.it("fails a dead running owner found by a restarted overseer of the same runtime", function() + activations.dead = activation("dead", 4) + admit(activations.dead, "t-dead", "pid-gone") + advance(activations.dead) + workflows.dead = workflow("dead") - local ok, err = overseer.reconcile_activation( - overseer.new_runtime(CURRENT_EPOCH), - test.not_nil(activations.service_restart) :: any) + local ok, err = overseer.reconcile(overseer.new_runtime(CURRENT_EPOCH), "dead") test.is_nil(err) test.is_true(ok) - test.eq(#observed.claims, 0) test.eq(#observed.spawns, 0) - test.eq(#observed.failures, 1) - test.eq(observed.failures[1].failure.reason, "same_runtime_owner_missing") + local failures = completed_failures() + test.eq(#failures, 1) + test.eq(failures[1].generation, 5) + test.eq(failures[1].fence.token, "t-dead") + test.eq(failures[1].failure.reason, "runtime_owner_lost") + end) + + test.it("spawns after a restart for a released owner or an owner of an earlier runtime", function() + activations.released = activation("released", 2) + admit(activations.released, "t-released", "pid-gone") + release(activations.released) + advance(activations.released) + workflows.released = workflow("released") + activations.rebooted = activation("rebooted", 2) + admit(activations.rebooted, "t-old", "pid-old", "runtime-before") + workflows.rebooted = workflow("rebooted") + + local runtime = overseer.new_runtime(CURRENT_EPOCH) + test.is_true(select(1, overseer.reconcile(runtime, "released"))) + test.is_true(select(1, overseer.reconcile(runtime, "rebooted"))) + test.eq(#observed.failures, 0) + test.eq(#observed.spawns, 2) + test.eq(observed.spawns[1].args.activation_generation, 3) + test.eq(observed.spawns[2].args.activation_generation, 2) end) - test.it("terminalizes a runtime owner loss once and never respawns it", function() + test.it("fails the latest request once when an unreleased owner exits after advances", function() activations.crash = activation("crash", 4) workflows.crash = workflow("crash") local runtime = overseer.new_runtime(CURRENT_EPOCH) - test.is_true(select(1, overseer.reconcile_activation(runtime, activations.crash))) + test.is_true(select(1, overseer.reconcile(runtime, "crash"))) local pid = observed.spawns[1].pid + admit(activations.crash, "t-crash", pid) + advance(activations.crash) + advance(activations.crash) observed.owners["dataflow.crash"] = nil local handled, exit_err = overseer.handle_exit(runtime, { @@ -278,26 +393,30 @@ local function run_tests() }) test.is_nil(exit_err) test.is_true(handled) - test.eq(#observed.failures, 1) - test.eq(observed.failures[1].generation, 4) - test.eq(observed.failures[1].failure.message, "executor panicked") + local failures = completed_failures() + test.eq(#failures, 1) + test.eq(failures[1].generation, 6) + test.eq(failures[1].failure.message, "executor panicked") test.eq(#observed.spawns, 1) local _, safety_err = overseer.safety_reconcile(runtime) test.is_nil(safety_err) - test.eq(#observed.failures, 1) + test.eq(#completed_failures(), 1) test.eq(#observed.spawns, 1) end) - test.it("does not let a stale EXIT fail a newer activation generation", function() + test.it("adopts a successor that took the name before the old owner's EXIT", function() activations.race = activation("race", 1) workflows.race = workflow("race") local runtime = overseer.new_runtime(CURRENT_EPOCH) - test.is_true(select(1, overseer.reconcile_activation(runtime, activations.race))) + test.is_true(select(1, overseer.reconcile(runtime, "race"))) local old_pid = observed.spawns[1].pid - - activations.race = activation("race", 2) + admit(activations.race, "t-old", old_pid) + release(activations.race) + advance(activations.race) observed.owners["dataflow.race"] = "pid-new" + admit(activations.race, "t-new", "pid-new") + local handled, err = overseer.handle_exit(runtime, { kind = overseer.process.event.EXIT, from = old_pid, @@ -310,33 +429,64 @@ local function run_tests() test.eq(observed.monitors[#observed.monitors], "pid-new") end) + test.it("defers an observation error and converges on the next reconcile", function() + activations.retry = activation("retry", 1) + workflows.retry = workflow("retry") + local runtime = overseer.new_runtime(CURRENT_EPOCH) + table.insert(locked_read_errors, "database is locked") + local first, first_err = overseer.reconcile(runtime, "retry") + test.is_nil(first) + test.contains(tostring(first_err), "database is locked") + test.eq(#observed.spawns, 0) + + test.is_true(select(1, overseer.reconcile(runtime, "retry"))) + test.eq(#observed.spawns, 1) + test.eq(#observed.failures, 0) + end) + + test.it("re-observes after losing the canonical name to a concurrent owner", function() + activations.conflict = activation("conflict", 1) + workflows.conflict = workflow("conflict") + local reconstruct = overseer.execution_frame.reconstruct + overseer.execution_frame.reconstruct = function(actor_id, actor_context) + observed.owners["dataflow.conflict"] = "pid-winner" + return reconstruct(actor_id, actor_context) + end + local ok, err = overseer.reconcile(overseer.new_runtime(CURRENT_EPOCH), "conflict") + test.is_nil(err) + test.is_true(ok) + test.eq(#observed.spawns, 0) + test.eq(#observed.failures, 0) + test.eq(observed.monitors[#observed.monitors], "pid-winner") + end) + test.it("fails an unreconstructable execution frame instead of root-spawning or retrying", function() activations.frame = activation("frame", 2) workflows.frame = workflow("frame") overseer.execution_frame = { reconstruct = function() return nil, nil, "policy no longer exists" end, } - local ok, err = overseer.reconcile_activation( - overseer.new_runtime(CURRENT_EPOCH), test.not_nil(activations.frame) :: any) + local ok, err = overseer.reconcile(overseer.new_runtime(CURRENT_EPOCH), "frame") test.is_nil(err) test.is_true(ok) test.eq(#observed.spawns, 0) - test.eq(#observed.failures, 1) - test.contains(observed.failures[1].failure.message, "policy no longer exists") + local failures = completed_failures() + test.eq(#failures, 1) + test.eq(failures[1].failure.reason, "orchestrator_spawn_failed") + test.contains(failures[1].failure.message, "policy no longer exists") + test.eq(failures[1].fence.generation, 2) end) test.it("stops a monitored process after durable cancellation", function() activations.cancelled = activation("cancelled", 5) workflows.cancelled = workflow("cancelled") local runtime = overseer.new_runtime(CURRENT_EPOCH) - test.is_true(select(1, overseer.reconcile_activation(runtime, activations.cancelled))) + test.is_true(select(1, overseer.reconcile(runtime, "cancelled"))) local pid = observed.spawns[1].pid activations.cancelled.desired_active = false workflows.cancelled.status = overseer.consts.STATUS.CANCELLED - local ok, err = overseer.reconcile_activation( - runtime, test.not_nil(activations.cancelled) :: any, - test.not_nil(workflows.cancelled) :: any) + local ok, err = overseer.reconcile(runtime, "cancelled") test.is_nil(err) test.is_true(ok) test.eq(#observed.cancels, 1) diff --git a/src/runner/runtime_epoch.lua b/src/runner/runtime_epoch.lua new file mode 100644 index 0000000..2bddef8 --- /dev/null +++ b/src/runner/runtime_epoch.lua @@ -0,0 +1,15 @@ +-- Module-owned reader of the full-runtime boot epoch. The entry runs with the +-- epoch reader group, so an orchestrator started synchronously under a caller +-- scope that cannot read module environment still obtains the epoch. +local M = {} + +local RUNTIME_EPOCH_ENV = "userspace.dataflow.env:runtime_epoch" + +function M.run(): (any, string?) + local value, err = env.get(RUNTIME_EPOCH_ENV) + if err then return nil, tostring(err) end + if value == nil or tostring(value) == "" then return { epoch = nil }, nil end + return { epoch = tostring(value) }, nil +end + +return M diff --git a/src/runner/workflow_state.lua b/src/runner/workflow_state.lua index 0ccedb7..2ec8860 100644 --- a/src/runner/workflow_state.lua +++ b/src/runner/workflow_state.lua @@ -118,6 +118,8 @@ local function make_restart_metadata(previous_status, extra) return metadata end +-- options.owner_token fences every write of this state to the orchestrator +-- that owns the activation. function workflow_state.new(dataflow_id, options) if not dataflow_id or dataflow_id == "" then return nil, "Dataflow ID is required" @@ -418,7 +420,10 @@ function methods:_prune_duplicate_routed_data() return nil end - local result, err = commit.execute(self.dataflow_id, uuid.v7(), cleanup_commands, { publish = false }) + local result, err = commit.execute(self.dataflow_id, uuid.v7(), cleanup_commands, { + publish = false, + owner_token = self.options.owner_token, + }) if err then return "Failed to prune duplicate routed data: " .. err end @@ -539,7 +544,10 @@ function methods:_reset_running_nodes() self:_propagate_reset_to_dependents(reset_commands) if #reset_commands > 0 then - local result, err = commit.execute(self.dataflow_id, uuid.v7(), reset_commands, { publish = false }) + local result, err = commit.execute(self.dataflow_id, uuid.v7(), reset_commands, { + publish = false, + owner_token = self.options.owner_token, + }) if err then return "Failed to reset RUNNING nodes: " .. err end @@ -1432,13 +1440,19 @@ function methods:observe_signal_wake(_wake_key) return self end -function methods:take_unclaimed_signal_wake_keys() +function methods:unclaimed_signal_wake_keys() local keys = {} for wake_key in pairs(self.pending_signal_wake_keys) do table.insert(keys, wake_key) end - self.pending_signal_wake_keys = {} + table.sort(keys) return keys end +-- Called after a passivation that removed these durable wake rows commits. +function methods:release_signal_wake_keys(keys) + for _, wake_key in ipairs(keys or {}) do self.pending_signal_wake_keys[wake_key] = nil end + return self +end + function methods:set_input_requirements(node_id, requirements) self.input_tracker.requirements[node_id] = requirements @@ -1484,7 +1498,10 @@ function methods:persist() end local op_id = uuid.v7() - local result, err = commit.execute(self.dataflow_id, op_id, self.queued_commands, { publish = true }) + local result, err = commit.execute(self.dataflow_id, op_id, self.queued_commands, { + publish = true, + owner_token = self.options.owner_token, + }) if err then return nil, "Failed to persist commands: " .. err diff --git a/src/runner/workflow_state_test.lua b/src/runner/workflow_state_test.lua index 52177cf..b1bf1e6 100644 --- a/src/runner/workflow_state_test.lua +++ b/src/runner/workflow_state_test.lua @@ -1105,6 +1105,45 @@ local function define_tests() end) end) + describe("Ownership fence", function() + it("persists only while its owner token owns the activation", function() + test_ctx.tx:rollback() + test_ctx.tx = nil + local dataflow_id = uuid.v7() + local now_ts: string = time.now():format(time.RFC3339NANO) + local _, create_err = test_ctx.db:execute(rebind([[ + INSERT INTO dataflows ( + dataflow_id, actor_id, type, status, metadata, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?) + ]], test_ctx.db:type()), { + dataflow_id, test_ctx.actor_id, "test_workflow", "active", "{}", now_ts, now_ts, + }) + test.is_nil(create_err) + test.not_nil(select(1, commit.request_activation(dataflow_id, {}, { notify = false }))) + test.is_true((test.not_nil(select(1, commit.admit_owner(dataflow_id, 1, { + token = "owner-a", pid = "pid-a", runtime_epoch = "runtime-a", + }))) :: any).admitted) + + local function update(token: string): string? + local ws = test.not_nil(select(1, workflow_state.new(dataflow_id, { + owner_token = token, + }))) :: any + ws:queue_commands({ + type = consts.COMMAND_TYPES.UPDATE_WORKFLOW, + payload = { metadata = { writer = token } }, + }) + local _, persist_err = ws:persist() + return persist_err + end + test.contains(tostring(update("owner-b")), "ownership lost") + test.is_nil(update("owner-a")) + + local _, cleanup_err = test_ctx.db:execute(rebind( + "DELETE FROM dataflows WHERE dataflow_id = ?", test_ctx.db:type()), { dataflow_id }) + test.is_nil(cleanup_err) + end) + end) + describe("State Updates from Results", function() it("does not clean a signal wake before its durable commit is applied", function() local ws = workflow_state.new(test_ctx.dataflow_id) :: any @@ -1112,7 +1151,7 @@ local function define_tests() local wake_key = "signal:" .. signal_data_id ws:observe_signal_wake(wake_key) - test.eq(#ws:take_unclaimed_signal_wake_keys(), 0, + test.eq(#ws:unclaimed_signal_wake_keys(), 0, "mailbox observation alone cannot acknowledge the durable wake") ws:_update_state_from_results({ @@ -1128,12 +1167,27 @@ local function define_tests() }, } }, }) - local applied = ws:take_unclaimed_signal_wake_keys() + local applied = ws:unclaimed_signal_wake_keys() test.eq(#applied, 1) test.eq(applied[1], wake_key, "only an applied, unmatched signal can become a cleanup candidate") end) + it("forgets only the signal wakes a committed passivation removed", function() + local ws = workflow_state.new(test_ctx.dataflow_id) :: any + ws.pending_signal_wake_keys["signal:captured"] = true + local captured = ws:unclaimed_signal_wake_keys() + test.eq(#captured, 1) + test.eq(#ws:unclaimed_signal_wake_keys(), 1, + "capturing keys for a passivation attempt keeps them until it commits") + + ws.pending_signal_wake_keys["signal:later"] = true + ws:release_signal_wake_keys(captured) + local remaining = ws:unclaimed_signal_wake_keys() + test.eq(#remaining, 1) + test.eq(remaining[1], "signal:later") + end) + it("replays a consumed signal when its replacement re-yields in the same owner", function() local ws = workflow_state.new(test_ctx.dataflow_id) :: any ws:track_yield("signal-node", { diff --git a/src/security/_index.yaml b/src/security/_index.yaml index 5527be4..61221f8 100644 --- a/src/security/_index.yaml +++ b/src/security/_index.yaml @@ -163,6 +163,24 @@ entries: - env.get effect: allow + - name: epoch_reader + kind: registry.entry + meta: + comment: Module-owned authority of the runtime epoch reader function; it reads nothing else. + + - name: epoch_reader.runtime_epoch + kind: security.policy + meta: + comment: Reads only the module-owned in-memory full-runtime boot epoch. + groups: + - userspace.dataflow.security:epoch_reader + policy: + resources: + - userspace.dataflow.env:runtime_epoch + actions: + - env.get + effect: allow + - name: retention kind: registry.entry meta: diff --git a/test/_index.yaml b/test/_index.yaml index 3fb2e5f..d1c9ed9 100644 --- a/test/_index.yaml +++ b/test/_index.yaml @@ -123,6 +123,47 @@ entries: test: wippy.test:test method: run_tests + - name: restricted_caller_test + kind: function.lua + meta: + type: test + suite: "01_runtime_failure" + comment: Proves a caller without the module-owned runtime epoch permission runs workflows synchronously and through the overseer. + source: file://restricted_caller_test.lua + modules: [uuid, time, env, funcs, security] + imports: + client: userspace.dataflow:client + consts: userspace.dataflow:consts + test: wippy.test:test + method: run_tests + + - name: restricted_caller_probe + kind: function.lua + meta: + test_only: true + comment: Creates and runs a workflow as the restricted caller of restricted_caller_test. + source: file://restricted_caller_test.lua + modules: [uuid, time, env, funcs, security] + imports: + client: userspace.dataflow:client + consts: userspace.dataflow:consts + test: wippy.test:test + method: probe + + - name: caller_without_runtime_epoch + kind: security.policy + meta: + test_only: true + comment: Grants a restricted test caller everything except the module-owned runtime epoch. + policy: + resources: '*' + actions: '*' + effect: allow + conditions: + - field: resource + operator: ne + value: userspace.dataflow.env:runtime_epoch + - name: db kind: db.sql.sqlite meta: diff --git a/test/restricted_caller_test.lua b/test/restricted_caller_test.lua new file mode 100644 index 0000000..435a93b --- /dev/null +++ b/test/restricted_caller_test.lua @@ -0,0 +1,98 @@ +local test = require("test") +local uuid = require("uuid") +local time = require("time") +local client = require("client") +local consts = require("consts") + +local RUNTIME_EPOCH_ENV = "userspace.dataflow.env:runtime_epoch" + +local function wait_until(predicate: () -> boolean, timeout_ms: number): boolean + local attempts = math.ceil(timeout_ms / 50) + for _ = 1, attempts do + if predicate() then return true end + time.sleep("50ms") + end + return false +end + +-- Runs as the restricted caller: it may use Dataflow but cannot read the +-- module-owned runtime epoch. +local function probe(args: any): any + local _, epoch_err = env.get(RUNTIME_EPOCH_ENV) + local c, client_err = client.new() + if not c then return { error = tostring(client_err) } end + local node_id = uuid.v7() + local dataflow_id, create_err = (c :: any):create_workflow({ + { + type = consts.COMMAND_TYPES.CREATE_NODE, + payload = { + node_id = node_id, + node_type = "userspace.dataflow.node.func:node", + status = consts.STATUS.PENDING, + config = { + func_id = "userspace.dataflow.node.func:test_func", + data_targets = { { data_type = consts.DATA_TYPE.WORKFLOW_OUTPUT } }, + }, + metadata = { title = "Restricted caller probe" }, + }, + }, + { + type = consts.COMMAND_TYPES.CREATE_DATA, + payload = { + data_id = uuid.v7(), + data_type = consts.DATA_TYPE.NODE_INPUT, + node_id = node_id, + content = { message = "restricted" }, + content_type = consts.CONTENT_TYPE.JSON, + key = "default", + }, + }, + }) + if create_err then return { error = tostring(create_err) } end + local id = tostring(dataflow_id) + if args and args.mode == "start" then + local _, start_err = (c :: any):start(id) + if start_err then return { error = tostring(start_err) } end + wait_until(function() + local status = (c :: any):get_status(id) + return status == consts.STATUS.COMPLETED_SUCCESS or status == consts.STATUS.COMPLETED_FAILURE + end, 5000) + else + local _, execute_err = (c :: any):execute(id) + if execute_err then return { error = tostring(execute_err), status = (c :: any):get_status(id) } end + end + return { + epoch_denied = epoch_err ~= nil, + status = (c :: any):get_status(id), + } +end + +local function run_tests() + test.describe("Dataflow under a restricted caller scope", function() + local function restricted_scope(): any + local policy = test.not_nil(select(1, security.policy("app:caller_without_runtime_epoch"))) + return test.not_nil(select(1, security.new_scope({ policy }))) + end + + test.it("reads the runtime epoch through the module-owned reader without caller env access", function() + local read, read_err = funcs.new():with_scope(restricted_scope()) + :call("userspace.dataflow.runner:runtime_epoch") + test.is_nil(read_err) + test.not_nil((test.not_nil(read) :: any).epoch) + end) + + for _, mode in ipairs({ "execute", "start" }) do + test.it("runs a workflow to completion through " .. mode, function() + local result, err = funcs.new():with_scope(restricted_scope()) + :call("app:restricted_caller_probe", { mode = mode }) + test.is_nil(err) + local observed = test.not_nil(result) :: any + test.is_nil(observed.error) + test.is_true(observed.epoch_denied, "the caller scope cannot read the runtime epoch") + test.eq(observed.status, consts.STATUS.COMPLETED_SUCCESS) + end) + end + end) +end + +return { run_tests = test.run_cases(run_tests), probe = probe } From c258275cf69ae8336785852da80ff87ed6d5c791 Mon Sep 17 00:00:00 2001 From: Wolfy-J Date: Wed, 23 Sep 2026 11:59:46 -0400 Subject: [PATCH 02/10] fix(persist): fence owner writes on an active request and re-admit only a running token verify_owner_tx rejects writes once the workflow is terminal or the request inactive, so an orchestrator cannot overwrite a durable cancellation. Admission re-admits the same token only while it is the running owner of an active request; a released token is admitted as a new owner. Release results no longer report ownership changes nobody consumes. --- src/persist/activation_repo.lua | 27 ++++---- src/persist/activation_repo_test.lua | 92 +++++++++++++++++++++++++++- src/persist/commit_test.lua | 29 +++++++++ src/persist/ops.lua | 2 - src/persist/ops_test.lua | 7 ++- 5 files changed, 140 insertions(+), 17 deletions(-) diff --git a/src/persist/activation_repo.lua b/src/persist/activation_repo.lua index b338fc4..580fe8f 100644 --- a/src/persist/activation_repo.lua +++ b/src/persist/activation_repo.lua @@ -469,7 +469,7 @@ end -- Release the active request when the ownership fence still holds: marks the -- owner released and the activation inactive. A fence that no longer matches --- reports whether ownership changed or only the generation advanced. +-- reports the current generation. function activation_repo.release_owner_tx(tx, dataflow_id, fence, now_value) if not tx then return nil, "transaction is required" end local valid, validation_err = validate_id(dataflow_id) @@ -510,16 +510,15 @@ function activation_repo.release_owner_tx(tx, dataflow_id, fence, now_value) released = false, terminal = false, generation = current and current.generation or nil, - owner_changed = current == nil or current.owner_token ~= fence.owner_token or - current.owner_phase ~= fence.owner_phase, }, nil end -- Admit an orchestrator that already holds the canonical name as the owner of -- the current request, before it mutates anything. Admission is refused for a -- terminal or inactive activation, a request older than the one the process --- was started for, and while another owner of the same runtime is running. The --- same token is admitted again, so an unacknowledged admission can be retried. +-- was started for, and while another owner of the same runtime is running. A +-- running owner's own token is admitted again without a write, so an +-- unacknowledged admission can be retried. function activation_repo.admit_owner_tx(tx, dataflow_id, min_generation, owner, now_value) if not tx then return nil, "transaction is required" end local valid, validation_err = validate_id(dataflow_id) @@ -550,16 +549,16 @@ function activation_repo.admit_owner_tx(tx, dataflow_id, min_generation, owner, end local current = found :: any + local running_self = current.owner_token == owner.token and current.owner_phase == PHASE.RUNNING local refused: string? = nil if TERMINAL_STATUS[status] then refused = "terminal" - elseif current.owner_token == owner.token and current.owner_phase == PHASE.RUNNING then - refused = nil elseif current.desired_active ~= true then refused = "inactive" elseif current.generation < min_generation then refused = "stale" - elseif current.owner_phase == PHASE.RUNNING and current.owner_epoch == owner.runtime_epoch then + elseif current.owner_phase == PHASE.RUNNING and current.owner_epoch == owner.runtime_epoch and + not running_self then refused = "owned" end if refused then @@ -567,7 +566,7 @@ function activation_repo.admit_owner_tx(tx, dataflow_id, min_generation, owner, current.refused = refused return current, nil end - if current.owner_token == owner.token then + if running_self then current.admitted = true return current, nil end @@ -588,20 +587,24 @@ function activation_repo.admit_owner_tx(tx, dataflow_id, min_generation, owner, return current, nil end --- Fence an orchestrator-owned transaction: the token must still own the --- activation as a running owner. Takes the workflow lock first. +-- Fence an orchestrator-owned transaction: the token must still own an active +-- request of a non-terminal workflow as a running owner. Takes the workflow +-- lock first, so a cancellation or release committed before is observed. function activation_repo.verify_owner_tx(tx, dataflow_id, owner_token) if not tx then return nil, "transaction is required" end local valid, validation_err = validate_id(dataflow_id) if not valid then return nil, validation_err end if type(owner_token) ~= "string" or owner_token == "" then return nil, "owner_token is required" end - local _, status_err = activation_repo.lock_workflow_tx(tx, dataflow_id) + local status, status_err = activation_repo.lock_workflow_tx(tx, dataflow_id) if status_err then return nil, status_err end local current, current_err = get_tx(tx, dataflow_id) if current_err then return nil, current_err end if not current or current.owner_token ~= owner_token or current.owner_phase ~= PHASE.RUNNING then return nil, "orchestrator ownership lost" end + if TERMINAL_STATUS[status] or current.desired_active ~= true then + return nil, "orchestrator ownership lost: activation is not active" + end return true, nil end diff --git a/src/persist/activation_repo_test.lua b/src/persist/activation_repo_test.lua index 72a5069..44f59e5 100644 --- a/src/persist/activation_repo_test.lua +++ b/src/persist/activation_repo_test.lua @@ -5,6 +5,15 @@ local time = require("time") local activation_repo = require("activation_repo") local consts = require("dataflow_consts") +local function rebind(query: string, db_type: any): string + if db_type ~= sql.type.POSTGRES and db_type ~= "postgres" then return query end + local index = 0 + return (query:gsub("%?", function() + index = index + 1 + return "$" .. index + end)) +end + local function define_tests() test.describe("Dataflow activation repository", function() local created = {} @@ -253,14 +262,16 @@ local function define_tests() end))) :: any test.is_false(stale.released) test.eq(stale.generation, 2) - test.is_false(stale.owner_changed) local foreign = test.not_nil(select(1, transaction(function(tx) return activation_repo.release_owner_tx(tx, id, { owner_token = "t2", owner_phase = "running", generation = 2 }, now(4)) end))) :: any test.is_false(foreign.released) - test.is_true(foreign.owner_changed) + local owned = test.not_nil(select(1, activation_repo.get(id))) :: any + test.eq(owned.owner_token, "t1") + test.eq(owned.owner_phase, "running") + test.is_true(owned.desired_active) local released = test.not_nil(select(1, transaction(function(tx) return activation_repo.release_owner_tx(tx, id, @@ -329,6 +340,83 @@ local function define_tests() test.contains(tostring(released_err), "ownership lost") end) + test.it("rejects owner writes once the workflow is terminal or the request inactive", function() + local id = create_dataflow(consts.STATUS.RUNNING) + test.not_nil(select(1, transaction(function(tx) + return activation_repo.request_activation_tx(tx, id, {}, now()) + end))) + test.is_true((test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t1", "runtime-a"), now(1)) + end))) :: any).admitted) + + local db = test.not_nil(select(1, sql.get("app:db"))) :: any + local _, inactive_err = db:execute(rebind( + "UPDATE dataflow_activations SET desired_active = ? WHERE dataflow_id = ?", db:type()), + { false, id }) + test.is_nil(inactive_err) + local _, inactive_write = transaction(function(tx) + return activation_repo.verify_owner_tx(tx, id, "t1") + end) + test.contains(tostring(inactive_write), "not active") + + local _, active_err = db:execute(rebind( + "UPDATE dataflow_activations SET desired_active = ? WHERE dataflow_id = ?", db:type()), + { true, id }) + test.is_nil(active_err) + local _, cancel_err = db:execute(rebind( + "UPDATE dataflows SET status = ? WHERE dataflow_id = ?", db:type()), + { consts.STATUS.CANCELLED, id }) + db:release() + test.is_nil(cancel_err) + test.not_nil(select(1, transaction(function(tx) + return activation_repo.disable_terminal_tx(tx, id, now(2)) + end))) + local _, terminal_write = transaction(function(tx) + return activation_repo.verify_owner_tx(tx, id, "t1") + end) + test.contains(tostring(terminal_write), "not active") + end) + + test.it("re-admits the same token only as a running owner of an active request", function() + local id = create_dataflow(consts.STATUS.RUNNING) + test.not_nil(select(1, transaction(function(tx) + return activation_repo.request_activation_tx(tx, id, {}, now()) + end))) + test.is_true((test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t1", "runtime-a"), now(1)) + end))) :: any).admitted) + local db = test.not_nil(select(1, sql.get("app:db"))) :: any + local _, inactive_err = db:execute(rebind( + "UPDATE dataflow_activations SET desired_active = ? WHERE dataflow_id = ?", db:type()), + { false, id }) + db:release() + test.is_nil(inactive_err) + local inactive = test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 1, owner("t1", "runtime-a"), now(2)) + end))) :: any + test.is_false(inactive.admitted) + test.eq(inactive.refused, "inactive") + + test.not_nil(select(1, transaction(function(tx) + return activation_repo.request_activation_tx(tx, id, {}, now(3)) + end))) + test.is_true((test.not_nil(select(1, transaction(function(tx) + return activation_repo.release_owner_tx(tx, id, + { owner_token = "t1", owner_phase = "running", generation = 2 }, now(4)) + end))) :: any).released) + test.not_nil(select(1, transaction(function(tx) + return activation_repo.request_activation_tx(tx, id, {}, now(5)) + end))) + local readmitted = test.not_nil(select(1, transaction(function(tx) + return activation_repo.admit_owner_tx(tx, id, 3, owner("t1", "runtime-a"), now(6)) + end))) :: any + test.is_true(readmitted.admitted) + test.eq(readmitted.owner_phase, "running") + test.is_true(select(1, transaction(function(tx) + return activation_repo.verify_owner_tx(tx, id, "t1") + end))) + end) + test.it("advances only when a newly inserted signal wake wins", function() local id = create_dataflow(consts.STATUS.WAITING) local wake_key = "signal:" .. uuid.v7() diff --git a/src/persist/commit_test.lua b/src/persist/commit_test.lua index 772ce55..737603a 100644 --- a/src/persist/commit_test.lua +++ b/src/persist/commit_test.lua @@ -1324,6 +1324,35 @@ local function define_tests() test.eq(status(), consts.STATUS.RUNNING) end) + it("rejects an owner-fenced write after the workflow was cancelled", function() + if test_ctx.tx then test_ctx.tx:rollback(); test_ctx.tx = nil end + if test_ctx.db then test_ctx.db:release(); test_ctx.db = nil end + local dataflow_id = create_isolated_dataflow() + test.not_nil(select(1, commit.request_activation(dataflow_id, {}, { notify = false }))) + test.is_true((test.not_nil(select(1, commit.admit_owner(dataflow_id, 1, { + token = "owner-a", pid = "pid-a", runtime_epoch = "runtime-a", + }))) :: any).admitted) + local _, cancel_err = commit.execute(dataflow_id, nil, { { + type = ops.COMMAND_TYPES.UPDATE_WORKFLOW, + payload = { dataflow_id = dataflow_id, status = consts.STATUS.CANCELLED }, + } }, { publish = false }) + test.is_nil(cancel_err) + + local revived, revive_err = commit.execute(dataflow_id, nil, { { + type = ops.COMMAND_TYPES.UPDATE_WORKFLOW, + payload = { dataflow_id = dataflow_id, status = consts.STATUS.RUNNING }, + } }, { publish = false, owner_token = "owner-a" }) + test.is_nil(revived) + test.contains(tostring(revive_err), "not active") + local db = test.not_nil(select(1, sql.get("app:db"))) :: any + local rows, query_err = db:query(rebind( + "SELECT status FROM dataflows WHERE dataflow_id = ?", db:type()), { dataflow_id }) + db:release() + test.is_nil(query_err) + test.eq(#rows, 1) + test.eq(rows[1].status, consts.STATUS.CANCELLED) + end) + it("fails an activation only while the observed owner still holds it", function() if test_ctx.tx then test_ctx.tx:rollback(); test_ctx.tx = nil end if test_ctx.db then test_ctx.db:release(); test_ctx.db = nil end diff --git a/src/persist/ops.lua b/src/persist/ops.lua index 46fe49f..f6d122a 100644 --- a/src/persist/ops.lua +++ b/src/persist/ops.lua @@ -1176,7 +1176,6 @@ handlers[constants.COMMAND_TYPES.PASSIVATE_WORKFLOW] = function(tx, dataflow_id, op_id = op_id, released = false, terminal = release.terminal == true, - owner_changed = release.owner_changed == true, current_generation = release.generation, } end @@ -1243,7 +1242,6 @@ handlers[constants.COMMAND_TYPES.COMPLETE_WORKFLOW] = function(tx, dataflow_id, op_id = op_id, completed = false, terminal = release.terminal == true, - owner_changed = release.owner_changed == true, current_generation = release.generation, } end diff --git a/src/persist/ops_test.lua b/src/persist/ops_test.lua index 3cd93ea..a0ed8d5 100644 --- a/src/persist/ops_test.lua +++ b/src/persist/ops_test.lua @@ -1188,7 +1188,12 @@ local function define_tests() }) test.is_nil(foreign_err) test.is_false(foreign.results[1].released) - test.is_true(foreign.results[1].owner_changed) + local owned, owned_err = txq(tx, [[ + SELECT owner_phase, desired_active FROM dataflow_activations WHERE dataflow_id = ? + ]], { resources.dataflow_id }) + test.is_nil(owned_err) + test.eq(owned[1].owner_phase, "running") + test.is_true(db_bool(owned[1].desired_active)) local result, err = ops.execute(tx, resources.dataflow_id, nil, { type = ops.COMMAND_TYPES.PASSIVATE_WORKFLOW, From 58479fe13ad0f9efbb467742f6993f553d7df5f9 Mon Sep 17 00:00:00 2001 From: Wolfy-J Date: Wed, 23 Sep 2026 12:03:24 -0400 Subject: [PATCH 03/10] fix(runner): read the runtime epoch with the orchestrator's own authority The synchronous path called a separate epoch function under the caller's scope, which required a new funcs.call grant from existing consumers. The orchestrator entry now carries the epoch reader group, merged into whatever scope it runs under, and reads the epoch itself on every path. Admission and state-creation failures give up the canonical name, a same-runtime owner that lost its name is reported as such, and the unreachable ownership-change projection branches are removed. --- src/consts.lua | 8 +- src/persist/activation_repo.lua | 6 +- src/runner/_index.yaml | 21 +++--- src/runner/orchestrator.lua | 75 +++++++------------ .../orchestrator_completion_flush_test.lua | 7 +- .../orchestrator_process_event_test.lua | 2 +- src/runner/orchestrator_test.lua | 49 ++++++------ src/runner/overseer.lua | 4 +- src/runner/overseer_migrations_ready.lua | 5 +- src/runner/overseer_state.lua | 4 +- src/runner/runtime_epoch.lua | 15 ---- src/security/_index.yaml | 2 +- test/_index.yaml | 25 ++++++- test/restricted_caller_test.lua | 17 ++--- 14 files changed, 108 insertions(+), 132 deletions(-) delete mode 100644 src/runner/runtime_epoch.lua diff --git a/src/consts.lua b/src/consts.lua index b518ca3..7178806 100644 --- a/src/consts.lua +++ b/src/consts.lua @@ -4,7 +4,13 @@ local consts = {} consts.HOST_ID = "app:processes" consts.APP_DB = "app:db" consts.ORCHESTRATOR = "userspace.dataflow.runner:orchestrator" -consts.RUNTIME_EPOCH_READER = "userspace.dataflow.runner:runtime_epoch" +consts.RUNTIME_EPOCH_ENV = "userspace.dataflow.env:runtime_epoch" + +-- Phases of the orchestrator ownership record on an activation. +consts.OWNER_PHASE = { + RUNNING = "running", + RELEASED = "released", +} -- Topic constants for actor state transitions consts.TOPIC = { diff --git a/src/persist/activation_repo.lua b/src/persist/activation_repo.lua index 580fe8f..49b6d30 100644 --- a/src/persist/activation_repo.lua +++ b/src/persist/activation_repo.lua @@ -11,11 +11,7 @@ local TERMINAL_STATUS = { [consts.STATUS.TERMINATED] = true, } -local PHASE = { - RUNNING = "running", - RELEASED = "released", -} -activation_repo.OWNER_PHASE = PHASE +local PHASE = consts.OWNER_PHASE local TERMINAL_VALUES = { consts.STATUS.COMPLETED_SUCCESS, diff --git a/src/runner/_index.yaml b/src/runner/_index.yaml index 281bdbc..934bc7c 100644 --- a/src/runner/_index.yaml +++ b/src/runner/_index.yaml @@ -26,6 +26,12 @@ entries: - funcs - time - logger + - env + # Merged into the scope the orchestrator runs under (the caller's scope on + # the synchronous path), so it reads the module-owned runtime epoch. + security: + groups: + - userspace.dataflow.security:epoch_reader imports: activation_repo: userspace.dataflow.persist:activation_repo commit: userspace.dataflow.persist:commit @@ -152,6 +158,8 @@ entries: - wippy.migration:migration_bootloader source: file://overseer_migrations_ready.lua modules: [uuid, env] + imports: + consts: userspace.dataflow:consts method: run - name: overseer_test @@ -167,17 +175,6 @@ entries: overseer: userspace.dataflow.runner:overseer method: run_tests - - name: runtime_epoch - kind: function.lua - meta: - comment: Returns the full-runtime boot epoch under the module-owned reader group, whatever the caller scope. - source: file://runtime_epoch.lua - modules: [env] - security: - groups: - - userspace.dataflow.security:epoch_reader - method: run - - name: overseer_state kind: library.lua meta: @@ -185,6 +182,8 @@ entries: comment: Pure boot acquisition, runtime ownership, and terminal failure model. tags: [dataflow, overseer, lifecycle, pure] source: file://overseer_state.lua + imports: + consts: userspace.dataflow:consts - name: overseer_state_test kind: function.lua diff --git a/src/runner/orchestrator.lua b/src/runner/orchestrator.lua index 71a388a..2d4c985 100644 --- a/src/runner/orchestrator.lua +++ b/src/runner/orchestrator.lua @@ -57,8 +57,6 @@ type ParkArmState = { scope: any, } -local OWNER_RUNNING = "running" - local TERMINAL_STATUS = { [consts.STATUS.COMPLETED_SUCCESS] = true, [consts.STATUS.COMPLETED_FAILURE] = true, @@ -169,19 +167,6 @@ local function stop_for_existing_terminal(state: OrchestratorState, projection: return false end --- Another orchestrator owns the activation, or this one already released it: --- this life must not write or schedule anything further. -local function stop_for_lost_ownership(state: OrchestratorState) - state.running = false - state.exit_result = { - success = false, - dataflow_id = state.dataflow_id, - error = "Orchestrator ownership lost", - } - state.final_status = nil - return false -end - local function adopt_projection_generation(state: OrchestratorState, projection: any) local current_generation = projection and tonumber( projection.current_generation or projection.generation) @@ -215,7 +200,7 @@ local function persist_fenced_failure( type = consts.COMMAND_TYPES.COMPLETE_WORKFLOW, payload = { activation_generation = state.activation_generation, - owner = { token = state.owner_token, phase = OWNER_RUNNING }, + owner = { token = state.owner_token, phase = consts.OWNER_PHASE.RUNNING }, status = consts.STATUS.COMPLETED_FAILURE, metadata = { error = failure_message }, }, @@ -249,9 +234,6 @@ local function persist_fenced_failure( if projection and projection.terminal == true then return stop_for_existing_terminal(state, projection), false end - if projection and projection.owner_changed == true then - return stop_for_lost_ownership(state), false - end local _, generation_err = adopt_projection_generation(state, projection) if generation_err then @@ -402,7 +384,7 @@ local function passivate(state: OrchestratorState): (boolean, boolean) type = consts.COMMAND_TYPES.PASSIVATE_WORKFLOW, payload = { activation_generation = state.activation_generation, - owner = { token = state.owner_token, phase = OWNER_RUNNING }, + owner = { token = state.owner_token, phase = consts.OWNER_PHASE.RUNNING }, signal_wake_keys = wake_keys, }, }) @@ -424,9 +406,6 @@ local function passivate(state: OrchestratorState): (boolean, boolean) if projection and projection.terminal == true then return stop_for_existing_terminal(state, projection), false end - if projection and projection.owner_changed == true then - return stop_for_lost_ownership(state), false - end local _, generation_err = adopt_projection_generation(state, projection) if generation_err then state.running = false @@ -699,7 +678,7 @@ function handle_complete_workflow(state: OrchestratorState, payload: any) type = consts.COMMAND_TYPES.COMPLETE_WORKFLOW, payload = { activation_generation = state.activation_generation, - owner = { token = state.owner_token, phase = OWNER_RUNNING }, + owner = { token = state.owner_token, phase = consts.OWNER_PHASE.RUNNING }, status = final_status, metadata = { error = not success and detailed_error or nil } } @@ -721,9 +700,6 @@ function handle_complete_workflow(state: OrchestratorState, payload: any) if projection and projection.terminal == true then return stop_for_existing_terminal(state, projection), false end - if projection and projection.owner_changed == true then - return stop_for_lost_ownership(state), false - end local _, generation_err = adopt_projection_generation(state, projection) if generation_err then state.exit_result = { @@ -1201,30 +1177,19 @@ local function duplicate_owner_result(dataflow_id) } end --- The overseer passes the runtime epoch with a spawn; a synchronous caller's --- scope may not read it, so that path asks the module-owned reader. -local function runtime_epoch_for(runtime: Runtime, args: any): (string?, string?) - local provided = args and args.runtime_epoch - if type(provided) == "string" and provided ~= "" then return provided, nil end - local read, read_err = runtime.funcs.new():call(consts.RUNTIME_EPOCH_READER) - if read_err then return nil, tostring(read_err) end - local epoch = type(read) == "table" and read.epoch or nil - if type(epoch) ~= "string" or epoch == "" then return nil, "runtime epoch is not ready" end - return epoch, nil -end - -- Admit this process as the owner of the current request. An admission whose -- outcome is unknown is resolved by rereading the token it would have written. local function admit_owner( runtime: Runtime, - args: any, dataflow_id: string, min_generation: number, pid: string ): (any?, string?) - local runtime_epoch, epoch_err = runtime_epoch_for(runtime, args) + -- The orchestrator entry carries the epoch reader group, so the epoch is + -- read with module authority on both the spawned and the synchronous path. + local runtime_epoch, epoch_err = runtime.overseer.load_runtime_epoch() if epoch_err or not runtime_epoch then - return nil, "Dataflow runtime epoch is unavailable: " .. tostring(epoch_err) + return nil, "Dataflow runtime epoch is unavailable: " .. tostring(epoch_err or "not ready") end local token = uuid.v7() local admission, admission_err = runtime.commit.admit_owner(dataflow_id, min_generation, { @@ -1241,7 +1206,7 @@ local function admit_owner( return nil, "Orchestrator admission outcome unknown: " .. tostring(admission_err) .. "; " .. tostring(current_err) end - if current and current.owner_token == token and current.owner_phase == OWNER_RUNNING then + if current and current.owner_token == token and current.owner_phase == consts.OWNER_PHASE.RUNNING then current.admitted = true return current, nil end @@ -1324,8 +1289,9 @@ local function run(args, runtime_override: any?) -- Holding the name, become the durable owner of the current request before -- anything is loaded or written; every later write is fenced by the token. local admission, admission_err = admit_owner( - runtime, args, dataflow_id, activation_generation, tostring(self_pid)) + runtime, dataflow_id, activation_generation, tostring(self_pid)) if admission_err or not admission then + runtime.process.registry.unregister(process_name) return { success = false, dataflow_id = dataflow_id, @@ -1349,7 +1315,14 @@ local function run(args, runtime_override: any?) message = "Dataflow already in terminal state", } end - if admission.refused == "owned" then return duplicate_owner_result(dataflow_id) end + if admission.refused == "owned" then + return { + success = false, + dataflow_id = dataflow_id, + error = "The running owner of this workflow in the current runtime lost its canonical " .. + "name; the overseer resolves the activation", + } + end return { success = true, pending = true, @@ -1363,11 +1336,13 @@ local function run(args, runtime_override: any?) local owner_token = tostring(admission.owner_token) local ws, ws_err = runtime.workflow_state.new(dataflow_id, { owner_token = owner_token }) - if ws_err then - return { success = false, error = "Failed to create workflow state: " .. ws_err } - end - if not ws then - return { success = false, dataflow_id = dataflow_id, error = "Failed to create workflow state" } + if ws_err or not ws then + runtime.process.registry.unregister(process_name) + return { + success = false, + dataflow_id = dataflow_id, + error = "Failed to create workflow state: " .. tostring(ws_err or "no state returned"), + } end local workflow_state = ws :: any diff --git a/src/runner/orchestrator_completion_flush_test.lua b/src/runner/orchestrator_completion_flush_test.lua index 50ece5e..7571be9 100644 --- a/src/runner/orchestrator_completion_flush_test.lua +++ b/src/runner/orchestrator_completion_flush_test.lua @@ -83,7 +83,10 @@ local function define_tests() end, }, execution_frame = execution_frame, - overseer = { notify = function(): (boolean, nil) return true, nil end }, + overseer = { + notify = function(): (boolean, nil) return true, nil end, + load_runtime_epoch = function(): (string?, string?) return "runtime-test", nil end, + }, funcs = { new = function(): any local executor: any = {} @@ -197,7 +200,6 @@ local function define_tests() local result = orchestrator.run({ dataflow_id = dataflow_id, activation_generation = probes.generation, - runtime_epoch = "runtime-test", }, runtime) :: any test.is_true(probes.aborted, "exit-batch transaction abort was exercised") @@ -225,7 +227,6 @@ local function define_tests() local result = orchestrator.run({ dataflow_id = dataflow_id, activation_generation = stale_generation, - runtime_epoch = "runtime-test", }, runtime) :: any test.is_true(probes.aborted, "exit-batch transaction abort was exercised") diff --git a/src/runner/orchestrator_process_event_test.lua b/src/runner/orchestrator_process_event_test.lua index 75acdba..4a7affe 100644 --- a/src/runner/orchestrator_process_event_test.lua +++ b/src/runner/orchestrator_process_event_test.lua @@ -110,6 +110,7 @@ local function define_tests() } runtime.overseer = { notify = function(): (boolean, string?) return true, nil end, + load_runtime_epoch = function(): (string?, string?) return "runtime-test", nil end, } runtime.activation_repo = { get = function() @@ -173,7 +174,6 @@ local function define_tests() local result = orchestrator.run({ dataflow_id = "failure-reason-workflow", activation_generation = 1, - runtime_epoch = "runtime-test", }, runtime) :: any test.is_false(result.success) diff --git a/src/runner/orchestrator_test.lua b/src/runner/orchestrator_test.lua index dba8d0d..8da5bc3 100644 --- a/src/runner/orchestrator_test.lua +++ b/src/runner/orchestrator_test.lua @@ -211,13 +211,7 @@ local function harness(options: HarnessOptions?): any local executor: any = {} executor.with_actor = function(self: any): any return self end executor.with_scope = function(self: any): any return self end - executor.call = function(_self: any, id: string): (any, nil) - if id == consts.RUNTIME_EPOCH_READER then - record("epoch_reader") - return { epoch = "runtime-test" }, nil - end - return {}, nil - end + executor.call = function(): (any, nil) return {}, nil end return executor end, }, @@ -226,6 +220,10 @@ local function harness(options: HarnessOptions?): any record("notify") return true, nil end, + load_runtime_epoch = function(): (string?, string?) + record("epoch") + return "runtime-test", nil + end, }, } return runtime @@ -236,11 +234,6 @@ local function run(runtime: any, args: any?): any for key, value in pairs(args or {}) do call_args[key] = value end call_args.dataflow_id = call_args.dataflow_id or "workflow-1" call_args.activation_generation = call_args.activation_generation or 1 - if call_args.runtime_epoch == false then - call_args.runtime_epoch = nil - else - call_args.runtime_epoch = call_args.runtime_epoch or "runtime-spawn" - end return orchestrator.run(call_args, runtime) end @@ -389,21 +382,19 @@ local function define_tests() activation = { generation = 3, desired_active = true }, }), { activation_generation = 2 }) test.is_true(result.success) - test.eq(table.concat(since(trace, "register"), " ", 1, 4), - "register admit:3:runtime-spawn:admitted state:true load_state") + test.eq(table.concat(since(trace, "register"), " ", 1, 5), + "register epoch admit:3:runtime-test:admitted state:true load_state") end) - it("reads the runtime epoch through the module reader when started synchronously", function() + it("reads the runtime epoch under its own authority, however it was started", function() local trace: { string } = {} - local result = run(harness({ trace = trace }), { runtime_epoch = false }) + local result = run(harness({ trace = trace }), { runtime_epoch = "caller-supplied" }) test.is_true(result.success) - local admitted = since(trace, "epoch_reader") - test.eq(admitted[2], "admit:1:runtime-test:admitted") + test.eq(since(trace, "epoch")[2], "admit:1:runtime-test:admitted") end) it("leaves before loading state when admission is refused", function() for _, case in ipairs({ - { refused = "owned", message = "already running" }, { refused = "stale", message = "Stale" }, { refused = "inactive", message = "Stale" }, }) do @@ -414,6 +405,12 @@ local function define_tests() test.is_false(contains(trace, "load_state")) test.is_true(contains(trace, "unregister")) end + local owned_trace: { string } = {} + local owned = run(harness({ trace = owned_trace, admission_refused = "owned" })) + test.is_false(owned.success) + test.contains(owned.error, "lost its canonical name") + test.is_false(contains(owned_trace, "load_state")) + test.is_true(contains(owned_trace, "unregister")) local terminal_trace: { string } = {} local terminal = run(harness({ trace = terminal_trace, admission_refused = "terminal" })) test.is_true(terminal.success) @@ -425,8 +422,8 @@ local function define_tests() local trace: { string } = {} local result = run(harness({ trace = trace, admission_error = "connection reset" })) test.is_true(result.success) - test.eq(table.concat(since(trace, "admit:1:runtime-spawn:admitted"), " ", 1, 4), - "admit:1:runtime-spawn:admitted reread state:true load_state") + test.eq(table.concat(since(trace, "admit:1:runtime-test:admitted"), " ", 1, 4), + "admit:1:runtime-test:admitted reread state:true load_state") local lost: { string } = {} local refused = run(harness({ @@ -435,6 +432,14 @@ local function define_tests() test.is_false(refused.success) test.contains(refused.error, "admission") test.is_false(contains(lost, "load_state")) + test.is_true(contains(lost, "unregister")) + end) + + it("gives up the canonical name when its workflow state cannot be created", function() + local trace: { string } = {} + local result = run(harness({ trace = trace, state_error = "repository unavailable" })) + test.is_false(result.success) + test.eq(since(trace, "state:true")[2], "unregister") end) it("releases ownership, then gives up the name and exits", function() @@ -476,7 +481,7 @@ local function define_tests() it("stops without further work when its ownership was lost or the release is uncertain", function() for _, persisted in ipairs({ - { value = { results = { { released = false, owner_changed = true, current_generation = 1 } } } }, + { error = "Failed to persist commands: orchestrator ownership lost" }, { error = "Failed to persist commands: connection reset" }, }) do local trace: { string } = {} diff --git a/src/runner/overseer.lua b/src/runner/overseer.lua index f823cfd..2f635ac 100644 --- a/src/runner/overseer.lua +++ b/src/runner/overseer.lua @@ -25,7 +25,6 @@ local NAME = "dataflow.overseer" local TOPIC = "dataflow.activation.changed" local SAFETY_INTERVAL = "30s" local SCAN_LIMIT = 100 -local RUNTIME_EPOCH_ENV = "userspace.dataflow.env:runtime_epoch" type OwnershipState = { by_pid: { [string]: string }, @@ -186,7 +185,7 @@ function M.new_runtime(epoch: string?): Runtime end local function load_runtime_epoch(): (string?, string?) - local value, err = env.get(RUNTIME_EPOCH_ENV) + local value, err = env.get(M.consts.RUNTIME_EPOCH_ENV) -- The service can start before the migrations-ready bootloader. A missing -- value is expected readiness state, not an operational failure. if err then @@ -298,7 +297,6 @@ local function spawn_owner( local args = clone_launch_args(launch_args) args.dataflow_id = dataflow_id args.activation_generation = generation - args.runtime_epoch = runtime.epoch local spawn_ok, spawn_pid, spawn_err = pcall(function() return M.process.with_context({}) :with_name("dataflow." .. dataflow_id) diff --git a/src/runner/overseer_migrations_ready.lua b/src/runner/overseer_migrations_ready.lua index 888965b..e14e4bd 100644 --- a/src/runner/overseer_migrations_ready.lua +++ b/src/runner/overseer_migrations_ready.lua @@ -1,14 +1,13 @@ local M = {} local uuid = require("uuid") - -local RUNTIME_EPOCH_ENV = "userspace.dataflow.env:runtime_epoch" +local consts = require("consts") local function run() -- Memory-backed epoch state lives for exactly one application runtime. -- A restarted overseer reads the same epoch and therefore cannot mistake -- its own crash for a full application reboot. local epoch = uuid.v7() - local stored, store_err = env.set(RUNTIME_EPOCH_ENV, epoch) + local stored, store_err = env.set(consts.RUNTIME_EPOCH_ENV, epoch) if store_err or stored == false then error("failed to establish Dataflow runtime epoch: " .. tostring(store_err)) end diff --git a/src/runner/overseer_state.lua b/src/runner/overseer_state.lua index ef31eb2..6244a8e 100644 --- a/src/runner/overseer_state.lua +++ b/src/runner/overseer_state.lua @@ -20,7 +20,9 @@ M.ACTION = { SPAWN = "spawn", } -local OWNER_RUNNING = "running" +local consts = require("consts") + +local OWNER_RUNNING = consts.OWNER_PHASE.RUNNING local TERMINAL_STATUS = { completed = true, diff --git a/src/runner/runtime_epoch.lua b/src/runner/runtime_epoch.lua deleted file mode 100644 index 2bddef8..0000000 --- a/src/runner/runtime_epoch.lua +++ /dev/null @@ -1,15 +0,0 @@ --- Module-owned reader of the full-runtime boot epoch. The entry runs with the --- epoch reader group, so an orchestrator started synchronously under a caller --- scope that cannot read module environment still obtains the epoch. -local M = {} - -local RUNTIME_EPOCH_ENV = "userspace.dataflow.env:runtime_epoch" - -function M.run(): (any, string?) - local value, err = env.get(RUNTIME_EPOCH_ENV) - if err then return nil, tostring(err) end - if value == nil or tostring(value) == "" then return { epoch = nil }, nil end - return { epoch = tostring(value) }, nil -end - -return M diff --git a/src/security/_index.yaml b/src/security/_index.yaml index 61221f8..9a6afed 100644 --- a/src/security/_index.yaml +++ b/src/security/_index.yaml @@ -166,7 +166,7 @@ entries: - name: epoch_reader kind: registry.entry meta: - comment: Module-owned authority of the runtime epoch reader function; it reads nothing else. + comment: Module-owned authority merged into every orchestrator so it reads the runtime epoch; it grants nothing else. - name: epoch_reader.runtime_epoch kind: security.policy diff --git a/test/_index.yaml b/test/_index.yaml index d1c9ed9..f2ba891 100644 --- a/test/_index.yaml +++ b/test/_index.yaml @@ -150,19 +150,36 @@ entries: test: wippy.test:test method: probe - - name: caller_without_runtime_epoch + - name: existing_caller_services kind: security.policy meta: test_only: true - comment: Grants a restricted test caller everything except the module-owned runtime epoch. + comment: An existing Dataflow consumer scope. It reads no environment variables and calls only the functions granted by existing_caller_functions. policy: resources: '*' actions: '*' effect: allow conditions: - - field: resource + - field: action operator: ne - value: userspace.dataflow.env:runtime_epoch + value: env.get + - field: action + operator: ne + value: funcs.call + + - name: existing_caller_functions + kind: security.policy + meta: + test_only: true + comment: The functions an existing Dataflow consumer calls to run the restricted caller workflow. + policy: + resources: + - app:restricted_caller_probe + - userspace.dataflow.runner:orchestrator + - userspace.dataflow.node.func:test_func + actions: + - funcs.call + effect: allow - name: db kind: db.sql.sqlite diff --git a/test/restricted_caller_test.lua b/test/restricted_caller_test.lua index 435a93b..cfb8310 100644 --- a/test/restricted_caller_test.lua +++ b/test/restricted_caller_test.lua @@ -4,7 +4,6 @@ local time = require("time") local client = require("client") local consts = require("consts") -local RUNTIME_EPOCH_ENV = "userspace.dataflow.env:runtime_epoch" local function wait_until(predicate: () -> boolean, timeout_ms: number): boolean local attempts = math.ceil(timeout_ms / 50) @@ -18,7 +17,7 @@ end -- Runs as the restricted caller: it may use Dataflow but cannot read the -- module-owned runtime epoch. local function probe(args: any): any - local _, epoch_err = env.get(RUNTIME_EPOCH_ENV) + local _, epoch_err = env.get(consts.RUNTIME_EPOCH_ENV) local c, client_err = client.new() if not c then return { error = tostring(client_err) } end local node_id = uuid.v7() @@ -70,17 +69,11 @@ end local function run_tests() test.describe("Dataflow under a restricted caller scope", function() local function restricted_scope(): any - local policy = test.not_nil(select(1, security.policy("app:caller_without_runtime_epoch"))) - return test.not_nil(select(1, security.new_scope({ policy }))) + local services = test.not_nil(select(1, security.policy("app:existing_caller_services"))) + local functions = test.not_nil(select(1, security.policy("app:existing_caller_functions"))) + return test.not_nil(select(1, security.new_scope({ services, functions }))) end - test.it("reads the runtime epoch through the module-owned reader without caller env access", function() - local read, read_err = funcs.new():with_scope(restricted_scope()) - :call("userspace.dataflow.runner:runtime_epoch") - test.is_nil(read_err) - test.not_nil((test.not_nil(read) :: any).epoch) - end) - for _, mode in ipairs({ "execute", "start" }) do test.it("runs a workflow to completion through " .. mode, function() local result, err = funcs.new():with_scope(restricted_scope()) @@ -88,7 +81,7 @@ local function run_tests() test.is_nil(err) local observed = test.not_nil(result) :: any test.is_nil(observed.error) - test.is_true(observed.epoch_denied, "the caller scope cannot read the runtime epoch") + test.is_true(observed.epoch_denied, "the caller scope reads no module environment") test.eq(observed.status, consts.STATUS.COMPLETED_SUCCESS) end) end From 650ced8993e8396497b152813a152a0a710379c3 Mon Sep 17 00:00:00 2001 From: Wolfy-J Date: Wed, 23 Sep 2026 12:26:38 -0400 Subject: [PATCH 04/10] fix(runner): wake live owners from durable state, never cancel inactive holders, bound pre-admission restarts - A monitored owner is woken once for every durable generation it has not been woken for, so a request whose COMMIT message was lost, or a deadline promoted before an overseer restart, still reaches the live owner. - An inactive activation stops nobody: a released owner exits by itself and any other name holder is refused admission or admitted by a newer request, so a stale observation can no longer cancel an admitted successor. - Spawns for one unchanged ownership state are bounded; an orchestrator that keeps exiting before admission fails the activation instead of respawning. - runtime_failure_test kills the orchestrator only after its admission; the overseer is also exercised against real ownership rows. - Remove the unused wake_repo, the pid_for helper and duplicated literals. --- .../10_add_activation_ownership.lua | 39 +-- src/persist/_index.yaml | 24 -- src/persist/wake_repo.lua | 66 ---- src/persist/wake_repo_test.lua | 98 ------ src/runner/_index.yaml | 1 + src/runner/overseer.lua | 146 ++++---- src/runner/overseer_state.lua | 44 +-- src/runner/overseer_state_test.lua | 34 +- src/runner/overseer_test.lua | 318 +++++++++++++++++- test/.wippy.yaml | 2 - test/runtime_failure_test.lua | 9 +- 11 files changed, 458 insertions(+), 323 deletions(-) delete mode 100644 src/persist/wake_repo.lua delete mode 100644 src/persist/wake_repo_test.lua diff --git a/src/migrations/10_add_activation_ownership.lua b/src/migrations/10_add_activation_ownership.lua index e4cc834..8c19569 100644 --- a/src/migrations/10_add_activation_ownership.lua +++ b/src/migrations/10_add_activation_ownership.lua @@ -4,21 +4,12 @@ local function execute_or_error(db, query) return success end --- The orchestrator-owned ownership record. owner_token identifies one +-- The ownership record of an activation. owner_token identifies one -- orchestrator incarnation, owner_pid its process, owner_epoch (added by -- migration 08) the runtime it runs in, and owner_phase is running or --- released; a row without a phase has never been owned. -local SQLITE_COLUMNS = { - { name = "owner_token", type = "TEXT" }, - { name = "owner_pid", type = "TEXT" }, - { name = "owner_phase", type = "TEXT" }, -} - -local POSTGRES_COLUMNS = { - { name = "owner_token", type = "TEXT" }, - { name = "owner_pid", type = "TEXT" }, - { name = "owner_phase", type = "TEXT" }, -} +-- released; a row without a phase has never been owned. An orchestrator +-- admits itself as the running owner; a completion releases it. +local COLUMNS = { "owner_token", "owner_pid", "owner_phase" } local function sqlite_columns(db) local columns, columns_err = db:query("PRAGMA table_info(dataflow_activations)") @@ -32,15 +23,14 @@ return require("migration").define(function() migration("Record the orchestrator that owns an activation", function() database("postgres", function() up(function(db) - for _, column in ipairs(POSTGRES_COLUMNS) do + for _, column in ipairs(COLUMNS) do execute_or_error(db, "ALTER TABLE dataflow_activations ADD COLUMN IF NOT EXISTS " .. - column.name .. " " .. column.type) + column .. " TEXT") end end) down(function(db) - for _, column in ipairs(POSTGRES_COLUMNS) do - execute_or_error(db, "ALTER TABLE dataflow_activations DROP COLUMN IF EXISTS " .. - column.name) + for _, column in ipairs(COLUMNS) do + execute_or_error(db, "ALTER TABLE dataflow_activations DROP COLUMN IF EXISTS " .. column) end end) end) @@ -48,18 +38,17 @@ return require("migration").define(function() database("sqlite", function() up(function(db) local present = sqlite_columns(db) - for _, column in ipairs(SQLITE_COLUMNS) do - if not present[column.name] then - execute_or_error(db, "ALTER TABLE dataflow_activations ADD COLUMN " .. - column.name .. " " .. column.type) + for _, column in ipairs(COLUMNS) do + if not present[column] then + execute_or_error(db, "ALTER TABLE dataflow_activations ADD COLUMN " .. column .. " TEXT") end end end) down(function(db) local present = sqlite_columns(db) - for _, column in ipairs(SQLITE_COLUMNS) do - if present[column.name] then - execute_or_error(db, "ALTER TABLE dataflow_activations DROP COLUMN " .. column.name) + for _, column in ipairs(COLUMNS) do + if present[column] then + execute_or_error(db, "ALTER TABLE dataflow_activations DROP COLUMN " .. column) end end end) diff --git a/src/persist/_index.yaml b/src/persist/_index.yaml index 5eb5dad..e792b45 100644 --- a/src/persist/_index.yaml +++ b/src/persist/_index.yaml @@ -206,30 +206,6 @@ entries: test: wippy.test:test method: run_tests - - name: wake_repo - kind: library.lua - meta: - private: true - comment: Indexed persistence for independently addressable signal and yield wake transitions. - source: file://wake_repo.lua - modules: [sql] - imports: - dataflow_consts: userspace.dataflow:consts - - - name: wake_repo_test - kind: function.lua - meta: - type: test - suite: Workflow - group: Workflow - comment: Proves per-trigger registration, exact consumption, and dataflow-wide terminal cleanup. - source: file://wake_repo_test.lua - modules: [sql, uuid, time] - imports: - test: wippy.test:test - wake_repo: userspace.dataflow.persist:wake_repo - method: run_tests - # userspace.dataflow.persist:node_reader - name: node_reader kind: library.lua diff --git a/src/persist/wake_repo.lua b/src/persist/wake_repo.lua deleted file mode 100644 index 60bfc71..0000000 --- a/src/persist/wake_repo.lua +++ /dev/null @@ -1,66 +0,0 @@ -local sql = require("sql") -local consts = require("dataflow_consts") - -local wake_repo = {} - -local function get_db() - local db, err = sql.get(consts.APP_DB) - if err then return nil, err end - return db, nil -end - -function wake_repo.clear(dataflow_id) - if type(dataflow_id) ~= "string" or dataflow_id == "" then return nil, "dataflow_id is required" end - local db, db_err = get_db() - if db_err then return nil, db_err end - local executor = sql.builder.delete("dataflow_wakes") - :where("dataflow_id = ?", dataflow_id) - :run_with(db) - local _, write_err = executor:exec() - db:release() - if write_err then return nil, write_err end - return true, nil -end - -function wake_repo.next() - local db, db_err = get_db() - if db_err then return nil, db_err end - local rows, query_err = db:query( - "SELECT dataflow_id, wake_key, wake_at FROM dataflow_wakes ORDER BY wake_at ASC LIMIT 1" - ) - db:release() - if query_err then return nil, query_err end - return rows and rows[1] or nil, nil -end - -function wake_repo.due(now_value, limit) - local db, db_err = get_db() - if db_err then return nil, db_err end - local executor = sql.builder.select("dataflow_id", "wake_key", "wake_at") - :from("dataflow_wakes") - :where("wake_at <= ?", now_value) - :order_by("wake_at ASC") - :limit(tonumber(limit) or 100) - :run_with(db) - local rows, query_err = executor:query() - db:release() - if query_err then return nil, query_err end - return rows or {}, nil -end - -function wake_repo.remove(dataflow_id, wake_key) - if type(dataflow_id) ~= "string" or dataflow_id == "" then return nil, "dataflow_id is required" end - if type(wake_key) ~= "string" or wake_key == "" then return nil, "wake_key is required" end - local db, db_err = get_db() - if db_err then return nil, db_err end - local executor = sql.builder.delete("dataflow_wakes") - :where("dataflow_id = ?", dataflow_id) - :where("wake_key = ?", wake_key) - :run_with(db) - local _, write_err = executor:exec() - db:release() - if write_err then return nil, write_err end - return true, nil -end - -return wake_repo diff --git a/src/persist/wake_repo_test.lua b/src/persist/wake_repo_test.lua deleted file mode 100644 index 4bd8b80..0000000 --- a/src/persist/wake_repo_test.lua +++ /dev/null @@ -1,98 +0,0 @@ -local test = require("test") -local sql = require("sql") -local uuid = require("uuid") -local time = require("time") -local wake_repo = require("wake_repo") - -local function run_tests() - test.describe("Dataflow wake repository", function() - test.it("keeps independently addressable wake triggers per dataflow", function() - local db = test.not_nil(select(1, sql.get("app:db"))) :: any - test.is_nil(select(2, sql.builder.delete("dataflow_wakes"):run_with(db):exec())) - local id = uuid.v7() - local now = time.now():format(time.RFC3339NANO) - local db_type, type_err = db:type() - test.is_nil(type_err) - test.is_true(db_type == "sqlite" or db_type == "postgres") - local _, insert_err = sql.builder.insert("dataflows"):set_map({ - dataflow_id = id, - actor_id = "wake-test", - type = "test", - status = "running", - metadata = "{}", - created_at = now, - updated_at = now, - }):run_with(db):exec() - test.is_nil(insert_err) - db:release() - - local insert_db = test.not_nil(select(1, sql.get("app:db"))) :: any - for _, wake in ipairs({ - { key = "yield:a", at = "2998-07-12T20:02:00Z" }, - { key = "yield:b", at = "2998-07-12T20:01:00Z" }, - { key = "commit:c", at = "2998-07-12T20:03:00Z" }, - }) do - test.is_nil(select(2, sql.builder.insert("dataflow_wakes"):set_map({ - dataflow_id = id, - wake_key = wake.key, - wake_at = wake.at, - }):run_with(insert_db):exec())) - end - insert_db:release() - local row = test.not_nil(select(1, wake_repo.next())) :: any - test.eq(row.dataflow_id, id) - test.eq(row.wake_key, "yield:b") - test.eq(tostring(row.wake_at), "2998-07-12T20:01:00Z") - - test.is_true(select(1, wake_repo.remove(id, "yield:b"))) - row = test.not_nil(select(1, wake_repo.next())) :: any - test.eq(row.wake_key, "yield:a") - test.is_true(select(1, wake_repo.clear(id))) - test.is_nil(select(1, wake_repo.next())) - end) - - -- due() is reached only by the overseer, whose tests stub the repository out, so - -- until now no test ran this query against a database at all. That gap is why the query - -- shipped with sqlite-only `?` placeholders and crashlooped the wake service on postgres - -- with `syntax error at or near "ORDER"`. The fixture writes go through the builder so the - -- only thing this exercises on either dialect is due() itself. - test.it("returns wakes that are due, ordered and limited, on the live dialect", function() - local db = test.not_nil(select(1, sql.get("app:db"))) :: any - local id = uuid.v7() - local now = time.now():format(time.RFC3339NANO) - - -- due() is deliberately global: it returns every wake that is due, for any dataflow. - -- Start from an empty table so the counts below mean what they say. - test.is_nil(select(2, sql.builder.delete("dataflow_wakes"):run_with(db):exec())) - - test.is_nil(select(2, sql.builder.insert("dataflows"):set_map({ - dataflow_id = id, actor_id = "due-test", type = "test", - status = "running", metadata = "{}", created_at = now, updated_at = now, - }):run_with(db):exec())) - - for _, w in ipairs({ - { key = "due:late", at = "2020-01-01T00:02:00Z" }, - { key = "due:early", at = "2020-01-01T00:01:00Z" }, - { key = "not:due", at = "2999-01-01T00:00:00Z" }, - }) do - test.is_nil(select(2, sql.builder.insert("dataflow_wakes"):set_map({ - dataflow_id = id, wake_key = w.key, wake_at = w.at, - }):run_with(db):exec())) - end - db:release() - - local rows = test.not_nil(select(1, wake_repo.due("2020-06-01T00:00:00Z", 10))) :: any - test.eq(#rows, 2) - test.eq(rows[1].wake_key, "due:early") - test.eq(rows[2].wake_key, "due:late") - - local limited = test.not_nil(select(1, wake_repo.due("2020-06-01T00:00:00Z", 1))) :: any - test.eq(#limited, 1) - test.eq(limited[1].wake_key, "due:early") - - test.is_true(select(1, wake_repo.clear(id))) - end) - end) -end - -return { run_tests = test.run_cases(run_tests) } diff --git a/src/runner/_index.yaml b/src/runner/_index.yaml index 934bc7c..7e4758b 100644 --- a/src/runner/_index.yaml +++ b/src/runner/_index.yaml @@ -170,6 +170,7 @@ entries: group: Workflow comment: Proves boot recovery, terminal runtime failure, exact wakes, and canonical adoption. source: file://overseer_test.lua + modules: [uuid] imports: test: wippy.test:test overseer: userspace.dataflow.runner:overseer diff --git a/src/runner/overseer.lua b/src/runner/overseer.lua index 2f635ac..1d7acd3 100644 --- a/src/runner/overseer.lua +++ b/src/runner/overseer.lua @@ -31,35 +31,20 @@ type OwnershipState = { by_dataflow: { [string]: string }, } -type Fence = { - token: string?, - phase: string?, - generation: number?, -} - -type Decision = { - kind: string, - reason: string, - dataflow_id: string, - generation: number?, - pid: string?, - fence: Fence?, +-- Spawns made for one observed ownership state that never led to an admission. +type StartAttempts = { + fence: string, + count: number, } type Runtime = { ownership: OwnershipState, - nudges: { [string]: Nudge }, + starts: { [string]: StartAttempts }, + woken: { [string]: number }, bootstrapped: boolean, epoch: string?, } -type Nudge = { - dataflow_id: string, - generation: number, - wake_key: string?, - wake_at: string?, -} - type Observation = { activation: any, status: string?, @@ -178,7 +163,8 @@ end function M.new_runtime(epoch: string?): Runtime return { ownership = M.overseer_state.new() :: OwnershipState, - nudges = {}, + starts = {}, + woken = {}, bootstrapped = false, epoch = epoch, } @@ -198,13 +184,22 @@ end M.load_runtime_epoch = load_runtime_epoch -local function deliver_nudge(runtime: Runtime, dataflow_id: string, pid: string): (boolean?, string?) - local nudge = runtime.nudges[dataflow_id] - if not nudge then return true, nil end - local ok, sent, send_err = pcall(M.process.send, pid, M.consts.MESSAGE_TOPIC.WAKE, nudge) - if not ok or not sent then return nil, not ok and tostring(sent) or tostring(send_err) end - runtime.nudges[dataflow_id] = nil - return true, nil +-- A live owner absorbs newer requests. Waking it once for each durable +-- generation it has not been woken for makes it reload pending work, so no +-- request depends on a message whose sender may have died after committing. +local function wake_owner(runtime: Runtime, dataflow_id: string, pid: string, generation: number?) + if not generation then return end + local woken = runtime.woken[pid] + if woken and woken >= generation then return end + local ok, sent, send_err = pcall(M.process.send, pid, M.consts.MESSAGE_TOPIC.WAKE, { + dataflow_id = dataflow_id, + generation = generation, + }) + if not ok or not sent then + log_flow("owner wake delivery failed", dataflow_id, not ok and sent or send_err) + return + end + runtime.woken[pid] = generation end local function failure_message(event: any): string @@ -219,6 +214,9 @@ local function failure_message(event: any): string end local MAX_RECONCILE_PASSES = 4 +-- Spawns allowed for one observed ownership state; an orchestrator that exits +-- before admitting itself leaves that state unchanged. +local MAX_STARTS = 3 -- Read the activation and the canonical name together under the workflow lock, -- so an admission or release in flight finishes before the name is checked. @@ -252,18 +250,35 @@ local function stop_owner(pid: string): (boolean?, string?) return true, nil end -local function monitor_owner(runtime: Runtime, dataflow_id: string, pid: string): boolean +local function monitor_owner(runtime: Runtime, dataflow_id: string, pid: string, generation: number?): boolean local ok, monitored, monitor_err = pcall(M.process.monitor, pid) local monitor_ok = ok and (monitored == true or is_already_monitoring(monitored) or is_already_monitoring(monitor_err)) if not monitor_ok then return false end M.overseer_state.track(runtime.ownership, dataflow_id, pid) - local delivered, delivery_err = deliver_nudge(runtime, dataflow_id, pid) - if not delivered then log_flow("owner nudge delivery failed", dataflow_id, delivery_err) end + wake_owner(runtime, dataflow_id, pid, generation) return true end -local function fail(dataflow_id: string, fence: Fence, reason: string, message: string): (any?, string?) +local function fence_key(fence: any): string + return table.concat({ + tostring(fence.token or ""), tostring(fence.phase or ""), tostring(fence.generation or ""), + }, "|") +end + +-- Count a spawn for the observed ownership state; a changed state starts over. +local function start_attempt(runtime: Runtime, dataflow_id: string, fence: any): number + local key = fence_key(fence) + local attempts = runtime.starts[dataflow_id] + if attempts and attempts.fence == key then + attempts.count = attempts.count + 1 + return attempts.count + end + runtime.starts[dataflow_id] = { fence = key, count = 1 } + return 1 +end + +local function fail(dataflow_id: string, fence: any, reason: string, message: string): (any?, string?) return M.commit.fail_activation(dataflow_id, fence, { source = "dataflow.overseer", reason = reason, @@ -324,44 +339,55 @@ function M.reconcile(runtime: Runtime, dataflow_id: string, options: ReconcileOp M.overseer_state.forget_dataflow(runtime.ownership, dataflow_id) return true, nil end - local decision = M.overseer_state.decide({ + local decision: any = M.overseer_state.decide({ dataflow_id = dataflow_id, status = observed.status, desired_active = activation.desired_active == true, generation = tonumber(activation.generation), - owner_token = activation.owner_token, - owner_phase = activation.owner_phase, - owner_epoch = activation.owner_epoch, + owner_token = activation.owner_token and tostring(activation.owner_token) or nil, + owner_phase = activation.owner_phase and tostring(activation.owner_phase) or nil, + owner_epoch = activation.owner_epoch and tostring(activation.owner_epoch) or nil, registered_pid = observed.registered_pid, runtime_epoch = runtime.epoch, - }) :: Decision + }) + if decision.kind ~= M.overseer_state.ACTION.SPAWN and + decision.kind ~= M.overseer_state.ACTION.MONITOR then + runtime.starts[dataflow_id] = nil + end if decision.kind == M.overseer_state.ACTION.NONE then - runtime.nudges[dataflow_id] = nil return true, nil elseif decision.kind == M.overseer_state.ACTION.STOP then - runtime.nudges[dataflow_id] = nil return stop_owner(tostring(decision.pid)) elseif decision.kind == M.overseer_state.ACTION.MONITOR then - if monitor_owner(runtime, dataflow_id, tostring(decision.pid)) then return true, nil end + if monitor_owner(runtime, dataflow_id, tostring(decision.pid), tonumber(decision.generation)) then + return true, nil + end elseif decision.kind == M.overseer_state.ACTION.FAIL then - local failed, fail_err = fail(dataflow_id, decision.fence :: Fence, decision.reason, message) + local failed, fail_err = fail(dataflow_id, decision.fence, tostring(decision.reason), message) if fail_err then return nil, tostring(fail_err) end if failed and (failed.completed == true or failed.terminal == true) then return true, nil end elseif decision.kind == M.overseer_state.ACTION.SPAWN then - local pid, conflict, spawn_err = spawn_owner( - runtime, dataflow_id, activation, tonumber(decision.generation) or 1) - if pid then - M.overseer_state.track(runtime.ownership, dataflow_id, pid) - local delivered, delivery_err = deliver_nudge(runtime, dataflow_id, pid) - if not delivered then log_flow("owner nudge delivery failed", dataflow_id, delivery_err) end - return true, nil - end - if not conflict then - local failed, fail_err = fail(dataflow_id, decision.fence :: Fence, - "orchestrator_spawn_failed", tostring(spawn_err)) + if start_attempt(runtime, dataflow_id, decision.fence) > MAX_STARTS then + runtime.starts[dataflow_id] = nil + local failed, fail_err = fail(dataflow_id, decision.fence, "orchestrator_start_failed", message) if fail_err then return nil, tostring(fail_err) end if failed and (failed.completed == true or failed.terminal == true) then return true, nil end + else + local pid, conflict, spawn_err = spawn_owner( + runtime, dataflow_id, activation, tonumber(decision.generation) or 1) + if pid then + M.overseer_state.track(runtime.ownership, dataflow_id, pid) + -- A new orchestrator loads everything up to its generation. + runtime.woken[pid] = tonumber(decision.generation) + return true, nil + end + if not conflict then + local failed, fail_err = fail(dataflow_id, decision.fence, + "orchestrator_spawn_failed", tostring(spawn_err)) + if fail_err then return nil, tostring(fail_err) end + if failed and (failed.completed == true or failed.terminal == true) then return true, nil end + end end else return nil, "unknown overseer decision " .. tostring(decision.kind) @@ -405,25 +431,12 @@ function M.promote_due(runtime: Runtime): (number?, string?) log_flow("due wake promotion failed", tostring(row.dataflow_id), activation_err) elseif activation and activation.promoted then promoted = promoted + 1 - local promoted_generation = tonumber(activation.generation) - if not promoted_generation then - log_flow("promoted activation has invalid generation", - tostring(row.dataflow_id), "generation is missing") - goto continue_due - end - runtime.nudges[tostring(row.dataflow_id)] = { - dataflow_id = tostring(row.dataflow_id), - generation = promoted_generation, - wake_key = tostring(row.wake_key), - wake_at = row.wake_at and tostring(row.wake_at) or nil, - } local ok, reconcile_err = M.reconcile(runtime, tostring(row.dataflow_id)) if not ok then log_flow("promoted activation reconciliation failed", tostring(row.dataflow_id), reconcile_err) end end - ::continue_due:: end return promoted, nil end @@ -482,6 +495,7 @@ end function M.handle_exit(runtime: Runtime, event: any): (boolean?, string?) local pid = event and event.from and tostring(event.from) or nil if not pid then return true, nil end + runtime.woken[pid] = nil local dataflow_id = M.overseer_state.forget_pid(runtime.ownership, pid) if not dataflow_id then return true, nil end return M.reconcile(runtime, dataflow_id, { message = failure_message(event) }) diff --git a/src/runner/overseer_state.lua b/src/runner/overseer_state.lua index 6244a8e..16282f1 100644 --- a/src/runner/overseer_state.lua +++ b/src/runner/overseer_state.lua @@ -1,12 +1,16 @@ -- Pure decisions for the Dataflow overseer. -- --- The activation row carries the ownership record, written only by the --- orchestrator: owner_token (one orchestrator incarnation), owner_epoch (its --- runtime) and owner_phase (running or released). The canonical name --- dataflow. is held by at most one process. From one locked observation of --- both, the overseer decides without remembering earlier decisions: --- - terminal or inactive: stop whoever holds the name; --- - a name holder: monitor it; +-- The activation row carries the ownership record: owner_token (one +-- orchestrator incarnation), owner_epoch (its runtime) and owner_phase. Only an +-- orchestrator admits itself as the running owner; a completion releases it, +-- whether the orchestrator's own or the overseer's failure fenced by the +-- observed owner. The canonical name dataflow. is held by at most one +-- process. From one locked observation of both, the overseer decides without +-- remembering earlier decisions: +-- - terminal: stop whoever holds the name; +-- - inactive: nothing; a released owner exits by itself and any other holder +-- is refused admission, or admitted if a newer request arrives first; +-- - a name holder: monitor and wake it; -- - a running owner of this runtime without the name: it died, so fail the -- activation fenced by its token, which covers every newer request; -- - otherwise (never owned, released, or owned in an earlier runtime): spawn. @@ -72,20 +76,24 @@ function M.decide(observation: Observation): Decision local pid = observation.registered_pid if pid == "" then pid = nil end - if is_terminal(observation.status) or observation.desired_active ~= true then + if is_terminal(observation.status) then if pid then - return { - kind = M.ACTION.STOP, - reason = is_terminal(observation.status) and "terminal_owner_stop" or "inactive_owner_stop", - dataflow_id = id, - pid = pid, - } + return { kind = M.ACTION.STOP, reason = "terminal_owner_stop", dataflow_id = id, pid = pid } end - return { kind = M.ACTION.NONE, reason = "not_active", dataflow_id = id } + return { kind = M.ACTION.NONE, reason = "terminal", dataflow_id = id } + end + if observation.desired_active ~= true then + return { kind = M.ACTION.NONE, reason = "inactive", dataflow_id = id } end if pid then - return { kind = M.ACTION.MONITOR, reason = "registered_owner", dataflow_id = id, pid = pid } + return { + kind = M.ACTION.MONITOR, + reason = "registered_owner", + dataflow_id = id, + generation = observation.generation, + pid = pid, + } end if observation.owner_phase == OWNER_RUNNING and observation.owner_token ~= nil and @@ -137,10 +145,6 @@ function M.forget_dataflow(state: State, dataflow_id: string) state.by_dataflow[dataflow_id] = nil end -function M.pid_for(state: State, dataflow_id: string): string? - return state.by_dataflow[dataflow_id] -end - function M.tracked(state: State): { string } local ids = {} for dataflow_id in pairs(state.by_dataflow) do table.insert(ids, dataflow_id) end diff --git a/src/runner/overseer_state_test.lua b/src/runner/overseer_state_test.lua index 9188e2a..d67463e 100644 --- a/src/runner/overseer_state_test.lua +++ b/src/runner/overseer_state_test.lua @@ -73,17 +73,25 @@ local function run_tests() end end) - test.it("stops a name holder of a terminal or inactive activation and otherwise does nothing", function() - for _, state in ipairs({ - { status = "failed" }, - { desired_active = false, status = "waiting" }, + test.it("stops the name holder of a terminal activation", function() + local named = observe({ status = "failed" }) + named.registered_pid = "pid-old" + local stop = decide(named) + test.eq(stop.kind, overseer.ACTION.STOP) + test.eq(stop.pid, "pid-old") + test.eq(decide(observe({ status = "failed" })).kind, overseer.ACTION.NONE) + end) + + test.it("leaves the name holder of an inactive activation to exit on its own", function() + for _, holder in ipairs({ + { owner_token = "t1", owner_phase = "released", owner_epoch = CURRENT_EPOCH }, + {}, }) do - local named = observe(state) - named.registered_pid = "pid-old" - local stop = decide(named) - test.eq(stop.kind, overseer.ACTION.STOP) - test.eq(stop.pid, "pid-old") - test.eq(decide(observe(state)).kind, overseer.ACTION.NONE) + local named = observe(holder) + named.desired_active = false + named.status = "waiting" + named.registered_pid = "pid-holder" + test.eq(decide(named).kind, overseer.ACTION.NONE) end end) @@ -91,11 +99,11 @@ local function run_tests() local state = overseer.new() overseer.track(state, "df", "pid-1") overseer.track(state, "df", "pid-2") - test.eq(overseer.pid_for(state, "df"), "pid-2") + test.eq(state.by_dataflow["df"], "pid-2") test.eq(overseer.forget_pid(state, "pid-1"), "df") - test.eq(overseer.pid_for(state, "df"), "pid-2", "a late EXIT leaves the current owner tracked") + test.eq(state.by_dataflow["df"], "pid-2", "a late EXIT leaves the current owner tracked") test.eq(overseer.forget_pid(state, "pid-2"), "df") - test.is_nil(overseer.pid_for(state, "df")) + test.is_nil(state.by_dataflow["df"]) test.is_nil(overseer.forget_pid(state, "pid-unknown")) end) end) diff --git a/src/runner/overseer_test.lua b/src/runner/overseer_test.lua index b767c9b..9ae5b39 100644 --- a/src/runner/overseer_test.lua +++ b/src/runner/overseer_test.lua @@ -1,4 +1,5 @@ local test = require("test") +local uuid = require("uuid") local overseer: any = require("overseer") local CURRENT_EPOCH = "runtime-current" @@ -232,7 +233,6 @@ local function run_tests() test.eq(spawn.source, overseer.consts.ORCHESTRATOR) test.eq(spawn.host, overseer.consts.HOST_ID) test.eq(spawn.args.activation_generation, 3) - test.eq(spawn.args.runtime_epoch, CURRENT_EPOCH) test.eq(spawn.args.init_func_id, "app:init") test.eq(spawn.actor, "restored:actor:boot") test.eq(spawn.scope, "scope:actor:boot") @@ -270,7 +270,7 @@ local function run_tests() test.is_true(ok) test.eq(#observed.spawns, 0) test.eq(#observed.failures, 0) - test.eq(overseer.overseer_state.pid_for(runtime.ownership, "monitored"), "pid-monitored") + test.eq(runtime.ownership.by_dataflow["monitored"], "pid-monitored") end) test.it("spawns the successor at once when a released owner's EXIT is still pending", function() @@ -297,7 +297,7 @@ local function run_tests() test.is_true(handled) test.eq(#observed.spawns, 2) test.eq(#observed.failures, 0) - test.eq(overseer.overseer_state.pid_for(runtime.ownership, "handoff"), observed.spawns[2].pid) + test.eq(runtime.ownership.by_dataflow["handoff"], observed.spawns[2].pid) end) test.it("waits for a released owner that still holds the name, then spawns on its EXIT", function() @@ -494,11 +494,13 @@ local function run_tests() test.eq(#observed.failures, 0) end) - test.it("promotes an exact due wake once and nudges the acquired owner", function() + test.it("promotes an exact due wake once and wakes the live owner", function() local due = activation("due", 6) due.promoted = true + admit(due, "t-due", "pid-due") activations.due = due workflows.due = workflow("due") + observed.owners["dataflow.due"] = "pid-due" local calls = 0 overseer.pending_due = function() return { { dataflow_id = "due", wake_key = "yield:one", wake_at = "2026-07-24T00:00:00Z" } }, nil @@ -518,13 +520,151 @@ local function run_tests() test.is_nil(second_err) test.eq(first, 1) test.eq(second, 0) - test.eq(#observed.spawns, 1) + test.eq(#observed.spawns, 0) test.eq(#observed.sends, 1) + test.eq(observed.sends[1].pid, "pid-due") test.eq(observed.sends[1].topic, overseer.consts.MESSAGE_TOPIC.WAKE) - test.eq(observed.sends[1].payload.wake_key, "yield:one") test.eq(observed.sends[1].payload.generation, 6) end) + test.it("wakes a live owner once per durable generation, across restarts", function() + activations.quiet = activation("quiet", 1) + admit(activations.quiet, "t-quiet", "pid-quiet") + workflows.quiet = workflow("quiet") + observed.owners["dataflow.quiet"] = "pid-quiet" + -- A signal committed, but its producer died before sending COMMIT. + advance(activations.quiet) + + local runtime = overseer.new_runtime(CURRENT_EPOCH) + test.is_true(select(1, overseer.handle_activation_hint(runtime, { dataflow_id = "quiet" }))) + test.not_nil(select(1, overseer.safety_reconcile(runtime))) + test.eq(#observed.sends, 1) + advance(activations.quiet) + test.not_nil(select(1, overseer.safety_reconcile(runtime))) + test.not_nil(select(1, overseer.bootstrap(overseer.new_runtime(CURRENT_EPOCH)))) + test.eq(#observed.sends, 3) + local generations = { 2, 3, 3 } + for index, sent in ipairs(observed.sends) do + test.eq(sent.pid, "pid-quiet") + test.eq(sent.topic, overseer.consts.MESSAGE_TOPIC.WAKE) + test.eq(sent.payload.generation, generations[index]) + end + test.eq(#observed.spawns, 0) + end) + + test.it("leaves a released owner in its unregister window to exit on its own", function() + activations.window = activation("window", 1) + admit(activations.window, "t-window", "pid-window") + release(activations.window) + workflows.window = workflow("window", overseer.consts.STATUS.WAITING) + observed.owners["dataflow.window"] = "pid-window" + local ok, err = overseer.reconcile(overseer.new_runtime(CURRENT_EPOCH), "window") + test.is_nil(err) + test.is_true(ok) + test.eq(#observed.cancels, 0) + test.eq(#observed.terminates, 0) + end) + + test.it("never cancels an unadmitted name holder that a later request admits", function() + activations.successor = activation("successor", 1) + admit(activations.successor, "t-old", "pid-old") + release(activations.successor) + workflows.successor = workflow("successor", overseer.consts.STATUS.WAITING) + observed.owners["dataflow.successor"] = "pid-successor" + local runtime = overseer.new_runtime(CURRENT_EPOCH) + test.is_true(select(1, overseer.reconcile(runtime, "successor"))) + + advance(activations.successor) + admit(activations.successor, "t-successor", "pid-successor") + test.is_true(select(1, overseer.reconcile(runtime, "successor"))) + test.eq(#observed.cancels, 0) + test.eq(#observed.failures, 0) + test.eq(runtime.ownership.by_dataflow["successor"], "pid-successor") + end) + + test.it("bounds restarts of an orchestrator that exits before admission", function() + activations.unstartable = activation("unstartable", 2) + workflows.unstartable = workflow("unstartable") + local runtime = overseer.new_runtime(CURRENT_EPOCH) + test.is_true(select(1, overseer.reconcile(runtime, "unstartable"))) + for _ = 1, 5 do + local latest = observed.spawns[#observed.spawns] + observed.owners["dataflow.unstartable"] = nil + local handled, exit_err = overseer.handle_exit(runtime, { + kind = overseer.process.event.EXIT, + from = latest.pid, + result = { error = "not allowed: app:db" }, + }) + test.is_nil(exit_err) + test.is_true(handled) + end + test.eq(#observed.spawns, 3) + local failures = completed_failures() + test.eq(#failures, 1) + test.eq(failures[1].failure.reason, "orchestrator_start_failed") + test.eq(failures[1].failure.message, "not allowed: app:db") + test.eq(failures[1].fence.generation, 2) + end) + + test.it("restarts an orchestrator that exited before admission once it can start", function() + activations.flaky = activation("flaky", 1) + workflows.flaky = workflow("flaky") + local runtime = overseer.new_runtime(CURRENT_EPOCH) + test.is_true(select(1, overseer.reconcile(runtime, "flaky"))) + observed.owners["dataflow.flaky"] = nil + test.is_true(select(1, overseer.handle_exit(runtime, { + kind = overseer.process.event.EXIT, from = observed.spawns[1].pid, + result = { error = "transient start failure" }, + }))) + test.eq(#observed.spawns, 2) + admit(activations.flaky, "t-flaky", tostring(observed.spawns[2].pid)) + test.not_nil(select(1, overseer.safety_reconcile(runtime))) + test.eq(#observed.spawns, 2) + test.eq(#observed.failures, 0) + end) + + test.it("fails a running owner whose process dies between observation and monitoring", function() + activations.vanishing = activation("vanishing", 3) + admit(activations.vanishing, "t-vanishing", "pid-vanishing") + workflows.vanishing = workflow("vanishing") + observed.owners["dataflow.vanishing"] = "pid-vanishing" + overseer.process.monitor = function(pid) + table.insert(observed.monitors, tostring(pid)) + observed.owners["dataflow.vanishing"] = nil + return nil, "process not found" + end + local ok, err = overseer.reconcile(overseer.new_runtime(CURRENT_EPOCH), "vanishing") + test.is_nil(err) + test.is_true(ok) + local failures = completed_failures() + test.eq(#failures, 1) + test.eq(failures[1].failure.reason, "runtime_owner_lost") + test.eq(#observed.spawns, 0) + end) + + test.it("retries failure persistence on the next reconcile after a write error", function() + activations.persist = activation("persist", 4) + admit(activations.persist, "t-persist", "pid-gone") + workflows.persist = workflow("persist") + local fail_activation = overseer.commit.fail_activation + local attempts = 0 + overseer.commit.fail_activation = function(id, fence, failure) + attempts = attempts + 1 + if attempts == 1 then return nil, "database is locked" end + return fail_activation(id, fence, failure) + end + local runtime = overseer.new_runtime(CURRENT_EPOCH) + local first, first_err = overseer.reconcile(runtime, "persist") + test.is_nil(first) + test.contains(tostring(first_err), "database is locked") + advance(activations.persist) + test.is_true(select(1, overseer.reconcile(runtime, "persist"))) + local failures = completed_failures() + test.eq(#failures, 1) + test.eq(failures[1].generation, 5) + test.eq(#observed.spawns, 0) + end) + test.it("recovers an activation whose notification was lost on the safety scan", function() local runtime = overseer.new_runtime(CURRENT_EPOCH) activations.lost = activation("lost", 9) @@ -553,4 +693,168 @@ local function run_tests() end) end -return { run_tests = test.run_cases(run_tests) } +local function run_durable_tests() + test.describe("Dataflow overseer on durable ownership records", function() + local originals + local observed + local created: { string } = {} + -- The runtime's own overseer reconciles these rows too. Holding their + -- canonical names keeps it to monitoring this process, while the + -- overseer under test observes the scripted registry. + local live_process = overseer.process + + local function now(): string + return overseer.time.now():format(overseer.time.RFC3339NANO) + end + + local function create_dataflow(): string + local id = uuid.v7() + local db = test.not_nil(select(1, overseer.sql.get("app:db"))) :: any + local _, insert_err = overseer.sql.builder.insert("dataflows"):set_map({ + dataflow_id = id, + actor_id = "overseer-durable-test", + type = "overseer-durable-test", + status = overseer.consts.STATUS.RUNNING, + metadata = "{}", + created_at = now(), + updated_at = now(), + }):run_with(db):exec() + db:release() + test.is_nil(insert_err) + test.is_true(select(1, live_process.registry.register("dataflow." .. id))) + table.insert(created, id) + return id + end + + local function request(id: string): number + local activation = test.not_nil(select(1, overseer.commit.request_activation( + id, {}, { notify = false }))) :: any + return tonumber(activation.generation) or 0 + end + + local function admit(id: string, token: string, pid: string, runtime_epoch: string?) + local admitted = test.not_nil(select(1, overseer.commit.admit_owner(id, 1, { + token = token, pid = pid, runtime_epoch = runtime_epoch or CURRENT_EPOCH, + }))) :: any + test.is_true(admitted.admitted) + end + + local function passivate(id: string, token: string, generation: number) + local result = test.not_nil(select(1, overseer.commit.execute(id, nil, { { + type = overseer.consts.COMMAND_TYPES.PASSIVATE_WORKFLOW, + payload = { + activation_generation = generation, + owner = { token = token, phase = overseer.consts.OWNER_PHASE.RUNNING }, + }, + } }, { publish = false, owner_token = token }))) :: any + test.is_true(result.results[1].released) + end + + local function row(id: string): any + return test.not_nil(select(1, overseer.activation_repo.get(id))) + end + + test.before_each(function() + originals = { + process = overseer.process, + execution_frame = overseer.execution_frame, + commit = overseer.commit, + } + observed = captures() + overseer.process = process_mock(observed) + local durable_commit = overseer.commit + overseer.commit = setmetatable({ + fail_activation = function(id, fence, failure) + local result, err = durable_commit.fail_activation(id, fence, failure) + table.insert(observed.failures, { + dataflow_id = id, fence = fence, failure = failure, + completed = result ~= nil and result.completed == true, + }) + return result, err + end, + }, { __index = durable_commit }) + overseer.execution_frame = { + reconstruct = function(actor_id) + table.insert(observed.reconstructions, { actor_id = actor_id }) + return "restored:" .. tostring(actor_id), "scope", nil + end, + } + end) + + test.after_each(function() + for key, value in pairs(originals) do overseer[key] = value end + end) + + test.after_all(function() + local db = test.not_nil(select(1, overseer.sql.get("app:db"))) :: any + for _, id in ipairs(created) do + overseer.sql.builder.delete("dataflows"):where("dataflow_id = ?", id):run_with(db):exec() + live_process.registry.unregister("dataflow." .. id) + end + db:release() + end) + + test.it("never cancels a name holder that a later request admits", function() + local id = create_dataflow() + request(id) + admit(id, "t-old", "pid-old") + passivate(id, "t-old", 1) + observed.owners["dataflow." .. id] = "pid-successor" + local runtime = overseer.new_runtime(CURRENT_EPOCH) + test.is_true(select(1, overseer.reconcile(runtime, id))) + + test.eq(request(id), 2) + admit(id, "t-successor", "pid-successor") + test.is_true(select(1, overseer.reconcile(runtime, id))) + test.eq(#observed.cancels, 0) + test.eq(#observed.terminates, 0) + test.eq(#observed.spawns, 0) + test.eq(observed.sends[#observed.sends].pid, "pid-successor") + test.eq(observed.sends[#observed.sends].payload.generation, 2) + test.eq(row(id).owner_token, "t-successor") + test.eq(row(id).owner_phase, "running") + end) + + test.it("fails a dead running owner of this runtime at the latest request", function() + local id = create_dataflow() + request(id) + admit(id, "t-dead", "pid-dead") + test.eq(request(id), 2) + local runtime = overseer.new_runtime(CURRENT_EPOCH) + test.is_true(select(1, overseer.reconcile(runtime, id))) + test.eq(#observed.spawns, 0) + local failed = row(id) + test.is_false(failed.desired_active) + test.eq(failed.generation, 2) + test.eq(failed.owner_phase, "released") + local workflow = test.not_nil(select(1, overseer.dataflow_repo.get(id))) :: any + test.eq(workflow.status, overseer.consts.STATUS.COMPLETED_FAILURE) + end) + + test.it("replaces a released owner or an owner of an earlier runtime exactly once", function() + local released = create_dataflow() + request(released) + admit(released, "t-released", "pid-released") + passivate(released, "t-released", 1) + test.eq(request(released), 2) + local rebooted = create_dataflow() + request(rebooted) + admit(rebooted, "t-rebooted", "pid-rebooted", "runtime-before") + + local runtime = overseer.new_runtime(CURRENT_EPOCH) + for _ = 1, 2 do + test.is_true(select(1, overseer.reconcile(runtime, released))) + test.is_true(select(1, overseer.reconcile(runtime, rebooted))) + end + test.eq(#observed.spawns, 2) + test.eq(observed.spawns[1].args.activation_generation, 2) + test.eq(observed.spawns[2].args.activation_generation, 1) + test.eq(#observed.failures, 0) + end) + end) +end + +return { run_tests = test.run_cases(function() + run_tests() + run_durable_tests() +end) } diff --git a/test/.wippy.yaml b/test/.wippy.yaml index ac8922e..614fb93 100644 --- a/test/.wippy.yaml +++ b/test/.wippy.yaml @@ -100,8 +100,6 @@ override: "userspace.dataflow.persist:node_reader_test:security.policies": [app:test_policy] "userspace.dataflow.persist:ops_test:security.actor.id": dataflow.test "userspace.dataflow.persist:ops_test:security.policies": [app:test_policy] - "userspace.dataflow.persist:wake_repo_test:security.actor.id": dataflow.test - "userspace.dataflow.persist:wake_repo_test:security.policies": [app:test_policy] "userspace.dataflow.runner:orchestrator_completion_flush_test:security.actor.id": dataflow.test "userspace.dataflow.runner:orchestrator_completion_flush_test:security.policies": [app:test_policy] "userspace.dataflow.runner:orchestrator_process_event_test:security.actor.id": dataflow.test diff --git a/test/runtime_failure_test.lua b/test/runtime_failure_test.lua index 3aca790..1a22896 100644 --- a/test/runtime_failure_test.lua +++ b/test/runtime_failure_test.lua @@ -54,10 +54,15 @@ local function run_tests() local process_name = "dataflow." .. dataflow_id local owner_pid = nil + -- The orchestrator registers the name before it admits itself; only an + -- admitted owner's death is a runtime ownership loss. test.is_true(wait_until(function() owner_pid = process.registry.lookup(process_name) - return owner_pid ~= nil - end, 3000), "canonical orchestrator became observable") + if owner_pid == nil then return false end + local owned = activation_repo.get(dataflow_id) + return owned ~= nil and owned.owner_pid == tostring(owner_pid) and + owned.owner_phase == consts.OWNER_PHASE.RUNNING + end, 3000), "canonical orchestrator was admitted as the owner") local terminated, terminate_err = process.terminate( test.not_nil(owner_pid) :: string) From 50ae7910e6e93943ccd946e888a6bd426af759d7 Mon Sep 17 00:00:00 2001 From: Wolfy-J Date: Wed, 23 Sep 2026 12:57:39 -0400 Subject: [PATCH 05/10] fix(runner): keep the overseer alive when the nearest wake is already due A due wake that could not be promoted made the loop arm time.after(0), which returns no channel, and crashed the overseer on every restart. The loop now promotes a due wake immediately and arms a timer only for a future one. Due promotion discards the wakes of a dataflow that no longer exists, which SQLite leaves behind because it does not enforce the cascading foreign key; tests remove their dependent rows explicitly for the same reason. --- src/persist/activation_repo.lua | 12 +++++++++++ src/persist/activation_repo_test.lua | 30 +++++++++++++++++++++++++--- src/runner/overseer.lua | 28 ++++++++++++++++++-------- src/runner/overseer_test.lua | 15 +++++++++++++- 4 files changed, 73 insertions(+), 12 deletions(-) diff --git a/src/persist/activation_repo.lua b/src/persist/activation_repo.lua index 49b6d30..6932a16 100644 --- a/src/persist/activation_repo.lua +++ b/src/persist/activation_repo.lua @@ -358,6 +358,18 @@ function activation_repo.activate_due_tx(tx, dataflow_id, wake_key, now_value) valid, validation_err = validate_timestamp(now_value, "now") if not valid then return nil, validation_err end + -- A wake outlives its dataflow only where the database does not enforce + -- the cascading foreign key (SQLite); such a wake is discarded, never + -- promoted. + local owners, owners_err = tx_query(tx, + "SELECT dataflow_id FROM dataflows WHERE dataflow_id = ? LIMIT 1", { dataflow_id }) + if owners_err then return nil, owners_err end + if not owners or not owners[1] then + local _, discard_err = tx_execute(tx, "DELETE FROM dataflow_wakes WHERE dataflow_id = ?", { dataflow_id }) + if discard_err then return nil, "failed to discard orphaned wakes: " .. tostring(discard_err) end + return { changed = true, terminal = false, promoted = false, missing = true }, nil + end + local status, status_err = activation_repo.lock_workflow_tx(tx, dataflow_id) if status_err then return nil, status_err end local terminal = terminal_result_from_status(status) diff --git a/src/persist/activation_repo_test.lua b/src/persist/activation_repo_test.lua index 44f59e5..f85ca42 100644 --- a/src/persist/activation_repo_test.lua +++ b/src/persist/activation_repo_test.lua @@ -94,12 +94,16 @@ local function define_tests() return tonumber(rows and rows[1] and rows[1].total) or 0 end + -- SQLite connections do not enforce the cascading foreign keys, so the + -- dependent rows are removed with their dataflow explicitly. test.after_all(function() local db = test.not_nil(select(1, sql.get("app:db"))) :: any for _, id in ipairs(created) do - sql.builder.delete("dataflows") - :where("dataflow_id = ?", id) - :run_with(db):exec() + for _, table_name in ipairs({ "dataflow_wakes", "dataflow_activations", "dataflows" }) do + sql.builder.delete(table_name) + :where("dataflow_id = ?", id) + :run_with(db):exec() + end end db:release() end) @@ -540,6 +544,26 @@ local function define_tests() test.is_nil(wake_generation(id, wake_key)) end) + test.it("discards the wakes of a dataflow that no longer exists during due promotion", function() + local id = create_dataflow(consts.STATUS.WAITING) + local wake_key = "yield:" .. uuid.v7() + local db = test.not_nil(select(1, sql.get("app:db"))) :: any + test.is_nil(select(2, sql.builder.insert("dataflow_wakes"):set_map({ + dataflow_id = id, + wake_key = wake_key, + wake_at = now(-1), + }):run_with(db):exec())) + test.is_nil(select(2, sql.builder.delete("dataflows"):where("dataflow_id = ?", id):run_with(db):exec())) + db:release() + + local result = test.not_nil(select(1, transaction(function(tx) + return activation_repo.activate_due_tx(tx, id, wake_key, now()) + end))) :: any + test.is_false(result.promoted) + test.is_true(result.missing) + test.eq(wake_count(id), 0) + end) + test.it("converges terminal activation and every stale wake during due promotion", function() local id = create_dataflow(consts.STATUS.RUNNING) test.not_nil(select(1, transaction(function(tx) diff --git a/src/runner/overseer.lua b/src/runner/overseer.lua index 1d7acd3..88d6327 100644 --- a/src/runner/overseer.lua +++ b/src/runner/overseer.lua @@ -518,6 +518,15 @@ function M.notify(payload: any?): (boolean?, string?) return M.process.send(NAME, TOPIC, payload or {}) end +-- A timer for the nearest pending wake, or due = true when it is already due. +function M.arm_wake(wake: any): (any?, boolean) + local wait_ns = select(1, duration_until(tostring(wake and wake.wake_at or ""))) + if wait_ns == nil then return nil, false end + if wait_ns <= 0 then return nil, true end + local timer = M.time.after(wait_ns) + return timer, false +end + local function reconcile_or_log(runtime: Runtime, operation: (Runtime) -> (any?, string?)) local ok, err = operation(runtime) if not ok and err then @@ -542,21 +551,24 @@ function M.run(_args: any) local events = M.process.events() while true do - local safety_timer = M.time.after(SAFETY_INTERVAL) - local wake_timer = nil - local cases = { inbox:case_receive(), events:case_receive(), safety_timer:case_receive() } - + local wake_timer: any = nil local wake, wake_err = M.next_pending_wake() if not wake_err and wake then - local wait_ns = select(1, duration_until(tostring(wake.wake_at))) - if wait_ns ~= nil then - wake_timer = M.time.after(wait_ns) - table.insert(cases, wake_timer:case_receive()) + local timer, due = M.arm_wake(wake) + if due then + -- Promote a wake that is already due now; one that still cannot + -- be promoted waits for the next event or safety pass. + reconcile_or_log(runtime, runtime.bootstrapped and M.promote_due or M.bootstrap) end + wake_timer = timer elseif wake_err and not schema_not_ready(wake_err) then logger:warn("could not inspect nearest dataflow wake", { error = tostring(wake_err) }) end + local safety_timer = M.time.after(SAFETY_INTERVAL) + local cases = { inbox:case_receive(), events:case_receive(), safety_timer:case_receive() } + if wake_timer then table.insert(cases, wake_timer:case_receive()) end + local result = M.channel.select(cases) if not result.ok then break end if result.channel == events then diff --git a/src/runner/overseer_test.lua b/src/runner/overseer_test.lua index 9ae5b39..a4e36d2 100644 --- a/src/runner/overseer_test.lua +++ b/src/runner/overseer_test.lua @@ -676,6 +676,17 @@ local function run_tests() test.eq(observed.spawns[1].args.activation_generation, 9) end) + test.it("arms a timer only for a wake that is not yet due", function() + local past = overseer.time.now():add(-1 * overseer.time.SECOND):format(overseer.time.RFC3339NANO) + local future = overseer.time.now():add(60 * overseer.time.SECOND):format(overseer.time.RFC3339NANO) + local due_timer, due = overseer.arm_wake({ wake_at = past }) + test.is_nil(due_timer) + test.is_true(due) + local timer, pending = overseer.arm_wake({ wake_at = future }) + test.not_nil(timer) + test.is_false(pending) + end) + test.it("recognizes missing SQLite and PostgreSQL migration state", function() test.is_true(overseer.schema_not_ready("no such table: dataflow_activations")) test.is_true(overseer.schema_not_ready('relation "dataflow_wakes" does not exist')) @@ -788,7 +799,9 @@ local function run_durable_tests() test.after_all(function() local db = test.not_nil(select(1, overseer.sql.get("app:db"))) :: any for _, id in ipairs(created) do - overseer.sql.builder.delete("dataflows"):where("dataflow_id = ?", id):run_with(db):exec() + for _, table_name in ipairs({ "dataflow_wakes", "dataflow_activations", "dataflows" }) do + overseer.sql.builder.delete(table_name):where("dataflow_id = ?", id):run_with(db):exec() + end live_process.registry.unregister("dataflow." .. id) end db:release() From 01e860620bad55d7918b7a84d51869ebb855bb5f Mon Sep 17 00:00:00 2001 From: Wolfy-J Date: Wed, 23 Sep 2026 13:15:53 -0400 Subject: [PATCH 06/10] fix(runner): read the nearest wake deadline at full precision on PostgreSQL The PostgreSQL driver renders TIMESTAMPTZ without fractional seconds, so the overseer saw a wake as due up to a second before the database did and armed no timer for it; exact deadlines then waited for the safety scan. The deadline is now formatted in SQL at the stored precision. --- src/runner/overseer.lua | 19 ++++++++++++++++--- src/runner/overseer_test.lua | 26 ++++++++++++++++++++++++++ 2 files changed, 42 insertions(+), 3 deletions(-) diff --git a/src/runner/overseer.lua b/src/runner/overseer.lua index 88d6327..ff04de4 100644 --- a/src/runner/overseer.lua +++ b/src/runner/overseer.lua @@ -501,13 +501,26 @@ function M.handle_exit(runtime: Runtime, event: any): (boolean?, string?) return M.reconcile(runtime, dataflow_id, { message = failure_message(event) }) end +-- The wake deadline as RFC 3339 text at the precision it is stored with. The +-- PostgreSQL driver renders TIMESTAMPTZ values without fractional seconds, so +-- the deadline is formatted in SQL; SQLite stores the text as written. +function M.wake_at_column(db_type: any): string + if db_type == M.sql.type.POSTGRES or db_type == "postgres" then + return [[to_char(dataflow_wakes.wake_at AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.US"Z"')]] + end + return "dataflow_wakes.wake_at" +end + function M.next_pending_wake(): (any?, string?) local db, db_err = M.sql.get(tostring(M.consts.APP_DB)) if db_err then return nil, tostring(db_err) end - local rows, query_err = db:query([[ - SELECT dataflow_id, wake_key, wake_at FROM dataflow_wakes + local db_type, type_err = db:type() + if type_err then db:release(); return nil, tostring(type_err) end + local rows, query_err = db:query( + "SELECT dataflow_id, wake_key, " .. M.wake_at_column(db_type) .. [[ AS wake_at + FROM dataflow_wakes WHERE activation_generation IS NULL - ORDER BY wake_at ASC, dataflow_id ASC, wake_key ASC LIMIT 1 + ORDER BY dataflow_wakes.wake_at ASC, dataflow_id ASC, wake_key ASC LIMIT 1 ]]) db:release() if query_err then return nil, tostring(query_err) end diff --git a/src/runner/overseer_test.lua b/src/runner/overseer_test.lua index a4e36d2..a4ac6d4 100644 --- a/src/runner/overseer_test.lua +++ b/src/runner/overseer_test.lua @@ -807,6 +807,32 @@ local function run_durable_tests() db:release() end) + test.it("reads a pending wake deadline at full precision on the live dialect", function() + local db = test.not_nil(select(1, overseer.sql.get("app:db"))) :: any + local tx = test.not_nil(select(1, db:begin())) :: any + local id = uuid.v7() + local _, flow_err = overseer.sql.builder.insert("dataflows"):set_map({ + dataflow_id = id, actor_id = "overseer-durable-test", type = "overseer-durable-test", + status = overseer.consts.STATUS.WAITING, metadata = "{}", created_at = now(), updated_at = now(), + }):run_with(tx):exec() + test.is_nil(flow_err) + local _, wake_err = overseer.sql.builder.insert("dataflow_wakes"):set_map({ + dataflow_id = id, wake_key = "yield:precision", wake_at = "2099-01-01T00:00:00.123456Z", + }):run_with(tx):exec() + test.is_nil(wake_err) + local db_type = select(1, tx:db_type()) + local placeholder = "?" + if db_type == "postgres" then placeholder = "$1" end + local rows, query_err = tx:query("SELECT " .. overseer.wake_at_column(db_type) .. + " AS wake_at FROM dataflow_wakes WHERE dataflow_id = " .. placeholder, { id }) + tx:rollback() + db:release() + test.is_nil(query_err) + local deadline = test.not_nil(select(1, overseer.time.parse( + overseer.time.RFC3339NANO, tostring(rows[1].wake_at)))) :: any + test.eq(deadline:utc():format(overseer.time.RFC3339NANO), "2099-01-01T00:00:00.123456Z") + end) + test.it("never cancels a name holder that a later request admits", function() local id = create_dataflow() request(id) From dcab41e6b31b1a9c1c2225ee72aa77675ac4bb5c Mon Sep 17 00:00:00 2001 From: Wolfy-J Date: Wed, 23 Sep 2026 14:33:13 -0400 Subject: [PATCH 07/10] fix(runner): space orchestrator start retries and arm the next future wake - Spawns for one unchanged ownership state wait 1s, 2s, 4s and 8s between attempts, so a transient outage recovers; after five the activation fails with orchestrator_start_failed. The exhausted budget is kept until that failure is durable and the observed state changes, so a failed write never reopens it. - The service loop promotes due wakes and then arms a timer for the nearest wake that is not yet due, so a due wake that cannot be promoted never hides a later one and the loop does not spin on it. - DELETE_WORKFLOW removes the activation row explicitly, as it already did for wakes, since SQLite does not enforce the cascading foreign keys; due promotion no longer discards wakes of a missing dataflow. - Document the epoch_reader group for callers with explicit env.get denies. --- README.md | 12 +++ src/persist/activation_repo.lua | 12 --- src/persist/activation_repo_test.lua | 20 ---- src/persist/ops.lua | 8 ++ src/persist/ops_test.lua | 22 ++++ src/runner/overseer.lua | 145 +++++++++++++++++++-------- src/runner/overseer_test.lua | 141 +++++++++++++++++++++----- 7 files changed, 262 insertions(+), 98 deletions(-) diff --git a/README.md b/README.md index c2ebb0b..d146eeb 100644 --- a/README.md +++ b/README.md @@ -15,6 +15,18 @@ +## Caller security + +A workflow runs under the actor and scope of the caller that created it; the +orchestrator needs no grant beyond what that scope already allows for running the +workflow. To classify ownership it reads the module's runtime epoch +(`userspace.dataflow.env:runtime_epoch`, action `env.get`). The orchestrator entry +carries the `userspace.dataflow.security:epoch_reader` group for this, and wippy +merges that group into the caller's scope, so callers need no grant for it. A +caller policy that explicitly **denies** `env.get` on that resource (or on `*`) +overrides the merged allow and stops every orchestrator of that caller; exclude +`userspace.dataflow.env:runtime_epoch` from such a deny. + ## Durable external waits Nodes that start external work and then wait for a signal use the declarative park contract: diff --git a/src/persist/activation_repo.lua b/src/persist/activation_repo.lua index 6932a16..49b6d30 100644 --- a/src/persist/activation_repo.lua +++ b/src/persist/activation_repo.lua @@ -358,18 +358,6 @@ function activation_repo.activate_due_tx(tx, dataflow_id, wake_key, now_value) valid, validation_err = validate_timestamp(now_value, "now") if not valid then return nil, validation_err end - -- A wake outlives its dataflow only where the database does not enforce - -- the cascading foreign key (SQLite); such a wake is discarded, never - -- promoted. - local owners, owners_err = tx_query(tx, - "SELECT dataflow_id FROM dataflows WHERE dataflow_id = ? LIMIT 1", { dataflow_id }) - if owners_err then return nil, owners_err end - if not owners or not owners[1] then - local _, discard_err = tx_execute(tx, "DELETE FROM dataflow_wakes WHERE dataflow_id = ?", { dataflow_id }) - if discard_err then return nil, "failed to discard orphaned wakes: " .. tostring(discard_err) end - return { changed = true, terminal = false, promoted = false, missing = true }, nil - end - local status, status_err = activation_repo.lock_workflow_tx(tx, dataflow_id) if status_err then return nil, status_err end local terminal = terminal_result_from_status(status) diff --git a/src/persist/activation_repo_test.lua b/src/persist/activation_repo_test.lua index f85ca42..3e5722c 100644 --- a/src/persist/activation_repo_test.lua +++ b/src/persist/activation_repo_test.lua @@ -544,26 +544,6 @@ local function define_tests() test.is_nil(wake_generation(id, wake_key)) end) - test.it("discards the wakes of a dataflow that no longer exists during due promotion", function() - local id = create_dataflow(consts.STATUS.WAITING) - local wake_key = "yield:" .. uuid.v7() - local db = test.not_nil(select(1, sql.get("app:db"))) :: any - test.is_nil(select(2, sql.builder.insert("dataflow_wakes"):set_map({ - dataflow_id = id, - wake_key = wake_key, - wake_at = now(-1), - }):run_with(db):exec())) - test.is_nil(select(2, sql.builder.delete("dataflows"):where("dataflow_id = ?", id):run_with(db):exec())) - db:release() - - local result = test.not_nil(select(1, transaction(function(tx) - return activation_repo.activate_due_tx(tx, id, wake_key, now()) - end))) :: any - test.is_false(result.promoted) - test.is_true(result.missing) - test.eq(wake_count(id), 0) - end) - test.it("converges terminal activation and every stale wake during due promotion", function() local id = create_dataflow(consts.STATUS.RUNNING) test.not_nil(select(1, transaction(function(tx) diff --git a/src/persist/ops.lua b/src/persist/ops.lua index f6d122a..b289fe2 100644 --- a/src/persist/ops.lua +++ b/src/persist/ops.lua @@ -1279,6 +1279,14 @@ handlers[constants.COMMAND_TYPES.DELETE_WORKFLOW] = function(tx, dataflow_id, op if wake_err then return nil, "Failed to clear deleted dataflow wake: " .. tostring(wake_err) end local wake_index_changed = (wake_result.rows_affected or 0) > 0 + -- Dependents are removed explicitly: SQLite connections do not enforce the + -- cascading foreign keys. + local _, activation_err = sql.builder.delete("dataflow_activations") + :where("dataflow_id = ?", wf_id_to_delete) + :run_with(tx) + :exec() + if activation_err then return nil, "Failed to delete dataflow activation: " .. tostring(activation_err) end + local delete_query = sql.builder.delete("dataflows") :where("dataflow_id = ?", wf_id_to_delete) diff --git a/src/persist/ops_test.lua b/src/persist/ops_test.lua index a0ed8d5..ad93348 100644 --- a/src/persist/ops_test.lua +++ b/src/persist/ops_test.lua @@ -1164,6 +1164,28 @@ local function define_tests() test.eq(rows[1].status, ops.STATUS.RUNNING) end) + it("deletes a workflow together with its activation and wakes", function() + local resources = setup_test_resources() + local tx = get_test_transaction() + local timestamp = time.now():format(time.RFC3339NANO) + test.not_nil(select(1, activation_repo.request_activation_tx( + tx, resources.dataflow_id, {}, timestamp))) + test.not_nil(select(1, activation_repo.activate_for_signal_tx( + tx, resources.dataflow_id, "signal:" .. uuid.v7(), timestamp, timestamp))) + + local _, delete_err = ops.execute(tx, resources.dataflow_id, nil, { + type = ops.COMMAND_TYPES.DELETE_WORKFLOW, + payload = {}, + }) + test.is_nil(delete_err) + for _, table_name in ipairs({ "dataflow_activations", "dataflow_wakes", "dataflows" }) do + local rows, query_err = txq(tx, "SELECT COUNT(*) AS total FROM " .. table_name .. + " WHERE dataflow_id = ?", { resources.dataflow_id }) + test.is_nil(query_err) + test.eq(tonumber(rows[1].total), 0, table_name) + end + end) + it("passivates only for the owning token and marks its ownership released", function() local resources = setup_test_resources() local tx = get_test_transaction() diff --git a/src/runner/overseer.lua b/src/runner/overseer.lua index ff04de4..ef6da79 100644 --- a/src/runner/overseer.lua +++ b/src/runner/overseer.lua @@ -31,10 +31,12 @@ type OwnershipState = { by_dataflow: { [string]: string }, } --- Spawns made for one observed ownership state that never led to an admission. +-- Spawns made for one observed ownership state that never led to an admission, +-- and when the next one may be made. type StartAttempts = { fence: string, count: number, + retry_at: number, } type Runtime = { @@ -214,9 +216,21 @@ local function failure_message(event: any): string end local MAX_RECONCILE_PASSES = 4 --- Spawns allowed for one observed ownership state; an orchestrator that exits --- before admitting itself leaves that state unchanged. -local MAX_STARTS = 3 +-- Spawns allowed for one observed ownership state. An orchestrator that exits +-- before admitting itself leaves that state unchanged; the next spawn waits +-- for start_retry_delay, so a transient outage can pass before the last one. +local MAX_STARTS = 5 +local START_RETRY_BASE_NS = 1000000000 + +M.MAX_STARTS = MAX_STARTS + +function M.start_retry_delay(attempt: number): number + return START_RETRY_BASE_NS * math.floor(2 ^ (attempt - 1)) +end + +function M.clock(): number + return M.time.now():unix_nano() +end -- Read the activation and the canonical name together under the workflow lock, -- so an admission or release in flight finishes before the name is checked. @@ -266,16 +280,14 @@ local function fence_key(fence: any): string }, "|") end --- Count a spawn for the observed ownership state; a changed state starts over. -local function start_attempt(runtime: Runtime, dataflow_id: string, fence: any): number +-- The spawns made for the observed ownership state; a changed state starts over. +local function start_budget(runtime: Runtime, dataflow_id: string, fence: any): StartAttempts local key = fence_key(fence) local attempts = runtime.starts[dataflow_id] - if attempts and attempts.fence == key then - attempts.count = attempts.count + 1 - return attempts.count - end - runtime.starts[dataflow_id] = { fence = key, count = 1 } - return 1 + if attempts and attempts.fence == key then return attempts end + local fresh: StartAttempts = { fence = key, count = 0, retry_at = 0 } + runtime.starts[dataflow_id] = fresh + return fresh end local function fail(dataflow_id: string, fence: any, reason: string, message: string): (any?, string?) @@ -368,15 +380,21 @@ function M.reconcile(runtime: Runtime, dataflow_id: string, options: ReconcileOp if fail_err then return nil, tostring(fail_err) end if failed and (failed.completed == true or failed.terminal == true) then return true, nil end elseif decision.kind == M.overseer_state.ACTION.SPAWN then - if start_attempt(runtime, dataflow_id, decision.fence) > MAX_STARTS then - runtime.starts[dataflow_id] = nil + local attempts = start_budget(runtime, dataflow_id, decision.fence) + if attempts.count >= MAX_STARTS then + -- The budget stays exhausted until the failure is durable and the + -- observed state changes. local failed, fail_err = fail(dataflow_id, decision.fence, "orchestrator_start_failed", message) if fail_err then return nil, tostring(fail_err) end if failed and (failed.completed == true or failed.terminal == true) then return true, nil end + elseif M.clock() < attempts.retry_at then + return true, nil else local pid, conflict, spawn_err = spawn_owner( runtime, dataflow_id, activation, tonumber(decision.generation) or 1) if pid then + attempts.count = attempts.count + 1 + attempts.retry_at = M.clock() + M.start_retry_delay(attempts.count) M.overseer_state.track(runtime.ownership, dataflow_id, pid) -- A new orchestrator loads everything up to its generation. runtime.woken[pid] = tonumber(decision.generation) @@ -453,7 +471,8 @@ function M.reconcile_all(runtime: Runtime): (number?, string?) local ok, reconcile_err = M.reconcile(runtime, id) if not ok then log_flow("active activation reconciliation failed", id, reconcile_err) end end - -- A monitored process whose activation is no longer active is stopped. + -- A tracked dataflow whose activation is no longer active is reconciled too: + -- the holder of a terminal one is stopped. for _, id in ipairs(M.overseer_state.tracked(runtime.ownership)) do if not seen[id] then local ok, reconcile_err = M.reconcile(runtime, id) @@ -511,17 +530,27 @@ function M.wake_at_column(db_type: any): string return "dataflow_wakes.wake_at" end -function M.next_pending_wake(): (any?, string?) +-- The nearest pending wake, or the nearest one after a given time. +function M.next_pending_wake(after: string?): (any?, string?) local db, db_err = M.sql.get(tostring(M.consts.APP_DB)) if db_err then return nil, tostring(db_err) end local db_type, type_err = db:type() if type_err then db:release(); return nil, tostring(type_err) end + local bound = "" + local params: { any } = {} + if after ~= nil then + bound = " AND dataflow_wakes.wake_at > ?" + if db_type == M.sql.type.POSTGRES or db_type == "postgres" then + bound = " AND dataflow_wakes.wake_at > $1" + end + params = { after } + end local rows, query_err = db:query( "SELECT dataflow_id, wake_key, " .. M.wake_at_column(db_type) .. [[ AS wake_at FROM dataflow_wakes - WHERE activation_generation IS NULL + WHERE activation_generation IS NULL]] .. bound .. [[ ORDER BY dataflow_wakes.wake_at ASC, dataflow_id ASC, wake_key ASC LIMIT 1 - ]]) + ]], params) db:release() if query_err then return nil, tostring(query_err) end return rows and rows[1] or nil, nil @@ -531,15 +560,6 @@ function M.notify(payload: any?): (boolean?, string?) return M.process.send(NAME, TOPIC, payload or {}) end --- A timer for the nearest pending wake, or due = true when it is already due. -function M.arm_wake(wake: any): (any?, boolean) - local wait_ns = select(1, duration_until(tostring(wake and wake.wake_at or ""))) - if wait_ns == nil then return nil, false end - if wait_ns <= 0 then return nil, true end - local timer = M.time.after(wait_ns) - return timer, false -end - local function reconcile_or_log(runtime: Runtime, operation: (Runtime) -> (any?, string?)) local ok, err = operation(runtime) if not ok and err then @@ -554,6 +574,58 @@ local function reconcile_or_log(runtime: Runtime, operation: (Runtime) -> (any?, end end +-- Promote the wakes that are already due, then arm a timer for the nearest wake +-- that is not. A due wake that cannot be promoted never hides a later one; it is +-- retried on the next event or safety pass. +function M.wake_timer(runtime: Runtime): any? + local nearest, nearest_err = M.next_pending_wake() + if nearest_err then + if not schema_not_ready(nearest_err) then + logger:warn("could not inspect nearest dataflow wake", { error = tostring(nearest_err) }) + end + return nil + end + if not nearest then return nil end + local wait_ns = select(1, duration_until(tostring(nearest.wake_at))) + if wait_ns ~= nil and wait_ns > 0 then return M.time.after(wait_ns) end + reconcile_or_log(runtime, runtime.bootstrapped and M.promote_due or M.bootstrap) + local upcoming, upcoming_err = M.next_pending_wake(now_value()) + if upcoming_err or not upcoming then return nil end + local upcoming_ns = select(1, duration_until(tostring(upcoming.wake_at))) + if upcoming_ns == nil or upcoming_ns <= 0 then return nil end + return M.time.after(upcoming_ns) +end + +-- Spawn again for every dataflow whose start retry is due, then arm a timer for +-- the nearest one that is not. +function M.retry_starts(runtime: Runtime): (boolean?, string?) + local now = M.clock() + local due: { string } = {} + for dataflow_id, attempts in pairs(runtime.starts) do + if attempts.count > 0 and attempts.count < MAX_STARTS and attempts.retry_at <= now then + table.insert(due, dataflow_id) + end + end + for _, dataflow_id in ipairs(due) do + local ok, reconcile_err = M.reconcile(runtime, dataflow_id) + if not ok then log_flow("orchestrator start retry failed", dataflow_id, reconcile_err) end + end + return true, nil +end + +local function start_retry_timer(runtime: Runtime): any? + local now = M.clock() + local nearest: number? = nil + for _, attempts in pairs(runtime.starts) do + if attempts.count > 0 and attempts.count < MAX_STARTS and attempts.retry_at > now and + (nearest == nil or attempts.retry_at < nearest) then + nearest = attempts.retry_at + end + end + if nearest == nil then return nil end + return M.time.after(nearest - now) +end + function M.run(_args: any) local registered, register_err = M.process.registry.register(NAME) if not registered then error("overseer registration failed: " .. tostring(register_err)) end @@ -564,23 +636,14 @@ function M.run(_args: any) local events = M.process.events() while true do - local wake_timer: any = nil - local wake, wake_err = M.next_pending_wake() - if not wake_err and wake then - local timer, due = M.arm_wake(wake) - if due then - -- Promote a wake that is already due now; one that still cannot - -- be promoted waits for the next event or safety pass. - reconcile_or_log(runtime, runtime.bootstrapped and M.promote_due or M.bootstrap) - end - wake_timer = timer - elseif wake_err and not schema_not_ready(wake_err) then - logger:warn("could not inspect nearest dataflow wake", { error = tostring(wake_err) }) - end + M.retry_starts(runtime) + local wake_timer = M.wake_timer(runtime) + local retry_timer = start_retry_timer(runtime) local safety_timer = M.time.after(SAFETY_INTERVAL) local cases = { inbox:case_receive(), events:case_receive(), safety_timer:case_receive() } if wake_timer then table.insert(cases, wake_timer:case_receive()) end + if retry_timer then table.insert(cases, retry_timer:case_receive()) end local result = M.channel.select(cases) if not result.ok then break end @@ -594,6 +657,8 @@ function M.run(_args: any) end elseif result.channel == wake_timer then reconcile_or_log(runtime, runtime.bootstrapped and M.promote_due or M.bootstrap) + elseif result.channel == retry_timer then + M.retry_starts(runtime) elseif result.channel == inbox then local message = result.value local topic = message and tostring(message:topic()) or "" diff --git a/src/runner/overseer_test.lua b/src/runner/overseer_test.lua index a4ac6d4..3a226b9 100644 --- a/src/runner/overseer_test.lua +++ b/src/runner/overseer_test.lua @@ -135,6 +135,7 @@ local function run_tests() local activations: { [string]: any } = {} local workflows: { [string]: any } = {} local locked_read_errors: { string } = {} + local clock: { now: number } = { now = 0 } test.before_each(function() originals = { @@ -146,11 +147,14 @@ local function run_tests() sql = overseer.sql, with_tx = overseer.with_tx, pending_due = overseer.pending_due, + clock = overseer.clock, } observed = captures() activations = {} :: { [string]: any } workflows = {} :: { [string]: any } locked_read_errors = {} :: { string } + clock.now = 0 + overseer.clock = function(): number return clock.now end overseer.process = process_mock(observed) overseer.execution_frame = { reconstruct = function(actor_id, actor_context) @@ -582,23 +586,34 @@ local function run_tests() test.eq(runtime.ownership.by_dataflow["successor"], "pid-successor") end) - test.it("bounds restarts of an orchestrator that exits before admission", function() + local function exit_before_admission(runtime, id: string) + local latest = observed.spawns[#observed.spawns] + observed.owners["dataflow." .. id] = nil + local handled, exit_err = overseer.handle_exit(runtime, { + kind = overseer.process.event.EXIT, + from = latest.pid, + result = { error = "not allowed: app:db" }, + }) + test.is_nil(exit_err) + test.is_true(handled) + end + + test.it("spaces restarts of an orchestrator that exits before admission, then fails it", function() activations.unstartable = activation("unstartable", 2) workflows.unstartable = workflow("unstartable") local runtime = overseer.new_runtime(CURRENT_EPOCH) test.is_true(select(1, overseer.reconcile(runtime, "unstartable"))) - for _ = 1, 5 do - local latest = observed.spawns[#observed.spawns] - observed.owners["dataflow.unstartable"] = nil - local handled, exit_err = overseer.handle_exit(runtime, { - kind = overseer.process.event.EXIT, - from = latest.pid, - result = { error = "not allowed: app:db" }, - }) - test.is_nil(exit_err) - test.is_true(handled) + local limit = overseer.MAX_STARTS + for attempt = 1, limit do + exit_before_admission(runtime, "unstartable") + if attempt < limit then + test.eq(#observed.spawns, attempt, "a restart waits for its retry time") + clock.now = clock.now + overseer.start_retry_delay(attempt) + test.is_true(select(1, overseer.retry_starts(runtime))) + test.eq(#observed.spawns, attempt + 1) + end end - test.eq(#observed.spawns, 3) + test.eq(#observed.spawns, limit) local failures = completed_failures() test.eq(#failures, 1) test.eq(failures[1].failure.reason, "orchestrator_start_failed") @@ -606,16 +621,46 @@ local function run_tests() test.eq(failures[1].fence.generation, 2) end) + test.it("keeps an exhausted start budget until its failure is persisted", function() + activations.stuck = activation("stuck", 1) + workflows.stuck = workflow("stuck") + local runtime = overseer.new_runtime(CURRENT_EPOCH) + test.is_true(select(1, overseer.reconcile(runtime, "stuck"))) + for attempt = 1, overseer.MAX_STARTS - 1 do + exit_before_admission(runtime, "stuck") + clock.now = clock.now + overseer.start_retry_delay(attempt) + test.is_true(select(1, overseer.retry_starts(runtime))) + end + local fail_activation = overseer.commit.fail_activation + local attempts = 0 + overseer.commit.fail_activation = function(id, fence, failure) + attempts = attempts + 1 + if attempts <= 2 then return nil, "database is locked" end + return fail_activation(id, fence, failure) + end + observed.owners["dataflow.stuck"] = nil + local _, exit_err = overseer.handle_exit(runtime, { + kind = overseer.process.event.EXIT, from = observed.spawns[#observed.spawns].pid, + result = { error = "not allowed: app:db" }, + }) + test.contains(tostring(exit_err), "database is locked") + local _, retry_err = overseer.reconcile(runtime, "stuck") + test.contains(tostring(retry_err), "database is locked") + test.is_true(select(1, overseer.reconcile(runtime, "stuck"))) + test.eq(#observed.spawns, overseer.MAX_STARTS) + test.eq(#completed_failures(), 1) + test.eq(attempts, 3) + end) + test.it("restarts an orchestrator that exited before admission once it can start", function() activations.flaky = activation("flaky", 1) workflows.flaky = workflow("flaky") local runtime = overseer.new_runtime(CURRENT_EPOCH) test.is_true(select(1, overseer.reconcile(runtime, "flaky"))) - observed.owners["dataflow.flaky"] = nil - test.is_true(select(1, overseer.handle_exit(runtime, { - kind = overseer.process.event.EXIT, from = observed.spawns[1].pid, - result = { error = "transient start failure" }, - }))) + exit_before_admission(runtime, "flaky") + test.eq(#observed.spawns, 1) + clock.now = clock.now + overseer.start_retry_delay(1) + test.is_true(select(1, overseer.retry_starts(runtime))) test.eq(#observed.spawns, 2) admit(activations.flaky, "t-flaky", tostring(observed.spawns[2].pid)) test.not_nil(select(1, overseer.safety_reconcile(runtime))) @@ -676,15 +721,59 @@ local function run_tests() test.eq(observed.spawns[1].args.activation_generation, 9) end) - test.it("arms a timer only for a wake that is not yet due", function() - local past = overseer.time.now():add(-1 * overseer.time.SECOND):format(overseer.time.RFC3339NANO) - local future = overseer.time.now():add(60 * overseer.time.SECOND):format(overseer.time.RFC3339NANO) - local due_timer, due = overseer.arm_wake({ wake_at = past }) - test.is_nil(due_timer) - test.is_true(due) - local timer, pending = overseer.arm_wake({ wake_at = future }) - test.not_nil(timer) - test.is_false(pending) + test.it("arms the service loop for the next future wake behind a due one", function() + local now = overseer.time.now() + local due = { dataflow_id = "orphan", wake_key = "yield:due", + wake_at = now:add(-1 * overseer.time.SECOND):format(overseer.time.RFC3339NANO) } + local next_wake = { dataflow_id = "valid", wake_key = "yield:next", + wake_at = now:add(overseer.time.SECOND):format(overseer.time.RFC3339NANO) } + local promotions = 0 + local timers: { any } = {} + local selected: { number } = {} + local real_time = overseer.time + local originals_loop = { + next_pending_wake = overseer.next_pending_wake, + promote_due = overseer.promote_due, + load_runtime_epoch = overseer.load_runtime_epoch, + channel = overseer.channel, + time = overseer.time, + } + overseer.next_pending_wake = function(after: string?) + if after then return next_wake, nil end + return due, nil + end + overseer.promote_due = function() + promotions = promotions + 1 + return 0, nil + end + overseer.load_runtime_epoch = function() return CURRENT_EPOCH, nil end + overseer.time = setmetatable({ + after = function(duration) + table.insert(timers, duration) + return { case_receive = function() return "timer" end }, nil + end, + }, { __index = real_time }) + overseer.channel = { + select = function(cases) + table.insert(selected, #cases) + return { ok = false } + end, + } + overseer.process.registry.register = function() return true, nil end + overseer.process.inbox = function() return { case_receive = function() return "inbox" end } end + overseer.process.events = function() return { case_receive = function() return "events" end } end + + local ok, run_err = pcall(overseer.run, {}) + for key, value in pairs(originals_loop) do overseer[key] = value end + test.is_true(ok, tostring(run_err)) + test.eq(selected[1], 4, "inbox, events, safety and the next wake") + local wake_timer = nil + for _, duration in ipairs(timers) do + if type(duration) == "number" then wake_timer = duration end + end + local armed = test.not_nil(wake_timer) :: number + test.is_true(armed > 0 and armed <= 1000000000) + test.is_true(promotions >= 1) end) test.it("recognizes missing SQLite and PostgreSQL migration state", function() From 48095f51ac09397fe664cdd87b16cfd7784a4e0f Mon Sep 17 00:00:00 2001 From: Wolfy-J Date: Wed, 23 Sep 2026 15:04:02 -0400 Subject: [PATCH 08/10] fix(runner): drive the overseer loop from one deadline computation Each iteration settles everything that is overdue, due wakes and deferred start retries alike, repeating while it makes progress, and then arms one timer for the earliest deadline still ahead. Deadlines are read from pending rows rather than bounded by a freshly sampled now, so a wake or retry that falls due during the pass is processed before the loop blocks. A due item that makes no progress is skipped until the next event or safety pass instead of hiding later deadlines. Start budgets are retired when the observed ownership changes, including admission, and after a durable start failure; a retry deadline exists only while a spawn is deferred. --- src/runner/overseer.lua | 258 ++++++++++++++++++++--------------- src/runner/overseer_test.lua | 212 +++++++++++++++++++++------- 2 files changed, 312 insertions(+), 158 deletions(-) diff --git a/src/runner/overseer.lua b/src/runner/overseer.lua index ef6da79..4dfa364 100644 --- a/src/runner/overseer.lua +++ b/src/runner/overseer.lua @@ -32,11 +32,12 @@ type OwnershipState = { } -- Spawns made for one observed ownership state that never led to an admission, --- and when the next one may be made. +-- when the next one may be made, and whether a spawn is waiting for that time. type StartAttempts = { fence: string, count: number, - retry_at: number, + retry_at: any, + deferred: boolean, } type Runtime = { @@ -81,18 +82,13 @@ local function schema_not_ready(err: any): boolean message:find("dataflows", 1, true) ~= nil) end -local function duration_until(value: string): (number?, string?) - if value == "" then return nil, "deadline is missing" end - local deadline, err = M.time.parse(M.time.RFC3339NANO, value) - if err then deadline, err = M.time.parse(M.time.RFC3339, value) end - if err then return nil, "invalid deadline: " .. value end - local now = M.time.now() - if now:after(deadline) or now:equal(deadline) then return 0, nil end - return deadline:sub(now):nanoseconds(), nil +-- The overseer's clock; every deadline decision reads time through it. +function M.now(): any + return M.time.now() end local function now_value(): string - return M.time.now():format(M.time.RFC3339NANO) + return tostring(M.now():format(M.time.RFC3339NANO)) end local function with_tx(fn: (any) -> (any?, string?)): (any?, string?) @@ -228,10 +224,6 @@ function M.start_retry_delay(attempt: number): number return START_RETRY_BASE_NS * math.floor(2 ^ (attempt - 1)) end -function M.clock(): number - return M.time.now():unix_nano() -end - -- Read the activation and the canonical name together under the workflow lock, -- so an admission or release in flight finishes before the name is checked. local function observe(dataflow_id: string): (Observation?, string?) @@ -285,7 +277,7 @@ local function start_budget(runtime: Runtime, dataflow_id: string, fence: any): local key = fence_key(fence) local attempts = runtime.starts[dataflow_id] if attempts and attempts.fence == key then return attempts end - local fresh: StartAttempts = { fence = key, count = 0, retry_at = 0 } + local fresh: StartAttempts = { fence = key, count = 0, retry_at = nil, deferred = false } runtime.starts[dataflow_id] = fresh return fresh end @@ -362,8 +354,14 @@ function M.reconcile(runtime: Runtime, dataflow_id: string, options: ReconcileOp registered_pid = observed.registered_pid, runtime_epoch = runtime.epoch, }) - if decision.kind ~= M.overseer_state.ACTION.SPAWN and - decision.kind ~= M.overseer_state.ACTION.MONITOR then + -- A start budget belongs to one ownership state; an admission or any + -- other durable change retires it, and so does a decision to do nothing. + local budget = runtime.starts[dataflow_id] + if budget and (budget.fence ~= fence_key({ + token = activation.owner_token, phase = activation.owner_phase, + generation = tonumber(activation.generation), + }) or (decision.kind ~= M.overseer_state.ACTION.SPAWN and + decision.kind ~= M.overseer_state.ACTION.MONITOR)) then runtime.starts[dataflow_id] = nil end @@ -382,19 +380,23 @@ function M.reconcile(runtime: Runtime, dataflow_id: string, options: ReconcileOp elseif decision.kind == M.overseer_state.ACTION.SPAWN then local attempts = start_budget(runtime, dataflow_id, decision.fence) if attempts.count >= MAX_STARTS then - -- The budget stays exhausted until the failure is durable and the - -- observed state changes. + -- The budget stays exhausted until the failure is durable. local failed, fail_err = fail(dataflow_id, decision.fence, "orchestrator_start_failed", message) if fail_err then return nil, tostring(fail_err) end - if failed and (failed.completed == true or failed.terminal == true) then return true, nil end - elseif M.clock() < attempts.retry_at then + if failed and (failed.completed == true or failed.terminal == true) then + runtime.starts[dataflow_id] = nil + return true, nil + end + elseif attempts.retry_at ~= nil and M.now():before(attempts.retry_at) then + attempts.deferred = true return true, nil else local pid, conflict, spawn_err = spawn_owner( runtime, dataflow_id, activation, tonumber(decision.generation) or 1) if pid then attempts.count = attempts.count + 1 - attempts.retry_at = M.clock() + M.start_retry_delay(attempts.count) + attempts.retry_at = M.now():add(M.start_retry_delay(attempts.count)) + attempts.deferred = false M.overseer_state.track(runtime.ownership, dataflow_id, pid) -- A new orchestrator loads everything up to its generation. runtime.woken[pid] = tonumber(decision.generation) @@ -433,27 +435,40 @@ end M.pending_due = pending_due -function M.promote_due(runtime: Runtime): (number?, string?) +-- Promote one pending wake. Returns whether the wake is resolved (promoted and +-- its dataflow reconciled, already promoted, or gone) and whether this call +-- promoted it. +function M.promote_wake(runtime: Runtime, row: any): (boolean?, string?, boolean) local now = now_value() - local rows, due_err = M.pending_due(now, SCAN_LIMIT) + local activation, activation_err = call_with_tx(function(tx) + local result, err = M.activation_repo.activate_due_tx( + tx, tostring(row.dataflow_id), tostring(row.wake_key), now) + return result, err and tostring(err) or nil + end) + if activation_err then return nil, tostring(activation_err), false end + if not activation then return nil, "due wake promotion returned no result", false end + if activation.promoted then + local ok, reconcile_err = M.reconcile(runtime, tostring(row.dataflow_id)) + if not ok then + log_flow("promoted activation reconciliation failed", tostring(row.dataflow_id), reconcile_err) + end + return true, nil, true + end + if activation.due == false then return nil, "wake is not due yet", false end + return true, nil, false +end + +function M.promote_due(runtime: Runtime): (number?, string?) + local rows, due_err = M.pending_due(now_value(), SCAN_LIMIT) if due_err then return nil, due_err end local promoted = 0 for _, row in ipairs(rows or {}) do - local activation, activation_err = call_with_tx(function(tx) - local result, err = M.activation_repo.activate_due_tx( - tx, tostring(row.dataflow_id), tostring(row.wake_key), now) - return result, err and tostring(err) or nil - end) - if activation_err then - if schema_not_ready(activation_err) then return nil, activation_err end - log_flow("due wake promotion failed", tostring(row.dataflow_id), activation_err) - elseif activation and activation.promoted then + local _, promote_err, was_promoted = M.promote_wake(runtime, row) + if promote_err then + if schema_not_ready(promote_err) then return nil, promote_err end + log_flow("due wake promotion failed", tostring(row.dataflow_id), promote_err) + elseif was_promoted then promoted = promoted + 1 - local ok, reconcile_err = M.reconcile(runtime, tostring(row.dataflow_id)) - if not ok then - log_flow("promoted activation reconciliation failed", - tostring(row.dataflow_id), reconcile_err) - end end end return promoted, nil @@ -530,30 +545,21 @@ function M.wake_at_column(db_type: any): string return "dataflow_wakes.wake_at" end --- The nearest pending wake, or the nearest one after a given time. -function M.next_pending_wake(after: string?): (any?, string?) +-- Pending wakes in deadline order, with deadlines at their stored precision. +function M.pending_wakes(limit: number): ({ any }?, string?) local db, db_err = M.sql.get(tostring(M.consts.APP_DB)) if db_err then return nil, tostring(db_err) end local db_type, type_err = db:type() if type_err then db:release(); return nil, tostring(type_err) end - local bound = "" - local params: { any } = {} - if after ~= nil then - bound = " AND dataflow_wakes.wake_at > ?" - if db_type == M.sql.type.POSTGRES or db_type == "postgres" then - bound = " AND dataflow_wakes.wake_at > $1" - end - params = { after } - end local rows, query_err = db:query( "SELECT dataflow_id, wake_key, " .. M.wake_at_column(db_type) .. [[ AS wake_at FROM dataflow_wakes - WHERE activation_generation IS NULL]] .. bound .. [[ - ORDER BY dataflow_wakes.wake_at ASC, dataflow_id ASC, wake_key ASC LIMIT 1 - ]], params) + WHERE activation_generation IS NULL + ORDER BY dataflow_wakes.wake_at ASC, dataflow_id ASC, wake_key ASC LIMIT ]] .. + tostring(math.floor(limit))) db:release() if query_err then return nil, tostring(query_err) end - return rows and rows[1] or nil, nil + return ((rows or {}) :: any) :: { any }, nil end function M.notify(payload: any?): (boolean?, string?) @@ -574,56 +580,85 @@ local function reconcile_or_log(runtime: Runtime, operation: (Runtime) -> (any?, end end --- Promote the wakes that are already due, then arm a timer for the nearest wake --- that is not. A due wake that cannot be promoted never hides a later one; it is --- retried on the next event or safety pass. -function M.wake_timer(runtime: Runtime): any? - local nearest, nearest_err = M.next_pending_wake() - if nearest_err then - if not schema_not_ready(nearest_err) then - logger:warn("could not inspect nearest dataflow wake", { error = tostring(nearest_err) }) +local MAX_SETTLE_PASSES = 8 + +local function parse_deadline(value: any): any? + local text = tostring(value or "") + local deadline, err = M.time.parse(M.time.RFC3339NANO, text) + if err then deadline, err = M.time.parse(M.time.RFC3339, text) end + if err then return nil end + return deadline +end + +local function earliest(current: any?, candidate: any): any + if current == nil or candidate:before(current) then return candidate end + return current +end + +-- Act on every wake and start retry that is due, then return the earliest +-- deadline still ahead. A due item whose processing makes no progress is +-- recorded in `attempted` and skipped until the next event or safety pass, so +-- it neither hides later deadlines nor makes the loop spin. +function M.settle(runtime: Runtime, attempted: { [string]: boolean }): any? + local ahead: any? = nil + for _ = 1, MAX_SETTLE_PASSES do + local progressed = false + ahead = nil + + local skipped = 0 + for _ in pairs(attempted) do skipped = skipped + 1 end + local rows, rows_err = M.pending_wakes(SCAN_LIMIT + skipped) + if rows_err then + if not schema_not_ready(rows_err) then + logger:warn("could not inspect pending dataflow wakes", { error = tostring(rows_err) }) + end + rows = {} end - return nil - end - if not nearest then return nil end - local wait_ns = select(1, duration_until(tostring(nearest.wake_at))) - if wait_ns ~= nil and wait_ns > 0 then return M.time.after(wait_ns) end - reconcile_or_log(runtime, runtime.bootstrapped and M.promote_due or M.bootstrap) - local upcoming, upcoming_err = M.next_pending_wake(now_value()) - if upcoming_err or not upcoming then return nil end - local upcoming_ns = select(1, duration_until(tostring(upcoming.wake_at))) - if upcoming_ns == nil or upcoming_ns <= 0 then return nil end - return M.time.after(upcoming_ns) -end - --- Spawn again for every dataflow whose start retry is due, then arm a timer for --- the nearest one that is not. -function M.retry_starts(runtime: Runtime): (boolean?, string?) - local now = M.clock() - local due: { string } = {} - for dataflow_id, attempts in pairs(runtime.starts) do - if attempts.count > 0 and attempts.count < MAX_STARTS and attempts.retry_at <= now then - table.insert(due, dataflow_id) + for _, row in ipairs(rows or {}) do + local key = table.concat({ "wake", tostring(row.dataflow_id), tostring(row.wake_key), + tostring(row.wake_at) }, "|") + if not attempted[key] then + local deadline = parse_deadline(row.wake_at) + if deadline == nil then + attempted[key] = true + log_flow("pending wake has an invalid deadline", tostring(row.dataflow_id), row.wake_at) + elseif M.now():before(deadline) then + ahead = earliest(ahead, deadline) + break + else + local resolved, promote_err = M.promote_wake(runtime, row) + if not resolved then + attempted[key] = true + log_flow("due wake promotion failed", tostring(row.dataflow_id), promote_err) + end + progressed = true + end + end end - end - for _, dataflow_id in ipairs(due) do - local ok, reconcile_err = M.reconcile(runtime, dataflow_id) - if not ok then log_flow("orchestrator start retry failed", dataflow_id, reconcile_err) end - end - return true, nil -end -local function start_retry_timer(runtime: Runtime): any? - local now = M.clock() - local nearest: number? = nil - for _, attempts in pairs(runtime.starts) do - if attempts.count > 0 and attempts.count < MAX_STARTS and attempts.retry_at > now and - (nearest == nil or attempts.retry_at < nearest) then - nearest = attempts.retry_at + local due: { string } = {} + for dataflow_id, attempts in pairs(runtime.starts) do + if attempts.deferred and attempts.count < MAX_STARTS and attempts.retry_at ~= nil then + local key = "start|" .. dataflow_id .. "|" .. attempts.retry_at:format(M.time.RFC3339NANO) + if not attempted[key] then + if M.now():before(attempts.retry_at) then + ahead = earliest(ahead, attempts.retry_at) + else + attempted[key] = true + table.insert(due, dataflow_id) + end + end + end end + for _, dataflow_id in ipairs(due) do + local ok, reconcile_err = M.reconcile(runtime, dataflow_id) + if not ok then log_flow("orchestrator start retry failed", dataflow_id, reconcile_err) end + progressed = true + end + + if not progressed then return ahead end end - if nearest == nil then return nil end - return M.time.after(nearest - now) + return ahead end function M.run(_args: any) @@ -635,18 +670,26 @@ function M.run(_args: any) local inbox = M.process.inbox() local events = M.process.events() + local attempted: { [string]: boolean } = {} while true do - M.retry_starts(runtime) - local wake_timer = M.wake_timer(runtime) - local retry_timer = start_retry_timer(runtime) + -- One deadline computation: everything overdue is processed first, then + -- a single timer waits for the earliest deadline still ahead. + local deadline: any? = nil + if runtime.bootstrapped then deadline = M.settle(runtime, attempted) end + local deadline_timer: any = nil + if deadline ~= nil then + local wait_ns = deadline:sub(M.now()):nanoseconds() + if wait_ns < 1 then wait_ns = 1 end + deadline_timer = M.time.after(wait_ns) + end local safety_timer = M.time.after(SAFETY_INTERVAL) local cases = { inbox:case_receive(), events:case_receive(), safety_timer:case_receive() } - if wake_timer then table.insert(cases, wake_timer:case_receive()) end - if retry_timer then table.insert(cases, retry_timer:case_receive()) end + if deadline_timer then table.insert(cases, deadline_timer:case_receive()) end local result = M.channel.select(cases) if not result.ok then break end + if result.channel ~= deadline_timer then attempted = {} end if result.channel == events then local event = result.value if event.kind == M.process.event.CANCEL then break end @@ -655,10 +698,6 @@ function M.run(_args: any) if not ok then log_flow("orchestrator EXIT reconciliation failed", tostring(event.from or ""), exit_err) end end - elseif result.channel == wake_timer then - reconcile_or_log(runtime, runtime.bootstrapped and M.promote_due or M.bootstrap) - elseif result.channel == retry_timer then - M.retry_starts(runtime) elseif result.channel == inbox then local message = result.value local topic = message and tostring(message:topic()) or "" @@ -677,7 +716,7 @@ function M.run(_args: any) elseif topic == TOPIC then reconcile_or_log(runtime, M.promote_due) end - else + elseif result.channel ~= deadline_timer then reconcile_or_log(runtime, runtime.bootstrapped and M.safety_reconcile or M.bootstrap) end end @@ -686,6 +725,5 @@ end M.NAME = NAME M.TOPIC = TOPIC -M.duration_until = duration_until M.schema_not_ready = schema_not_ready return M diff --git a/src/runner/overseer_test.lua b/src/runner/overseer_test.lua index 3a226b9..1af24b9 100644 --- a/src/runner/overseer_test.lua +++ b/src/runner/overseer_test.lua @@ -135,7 +135,7 @@ local function run_tests() local activations: { [string]: any } = {} local workflows: { [string]: any } = {} local locked_read_errors: { string } = {} - local clock: { now: number } = { now = 0 } + local clock: { now: any } = { now = overseer.time.now() } test.before_each(function() originals = { @@ -147,14 +147,16 @@ local function run_tests() sql = overseer.sql, with_tx = overseer.with_tx, pending_due = overseer.pending_due, - clock = overseer.clock, + now = overseer.now, + pending_wakes = overseer.pending_wakes, } observed = captures() activations = {} :: { [string]: any } workflows = {} :: { [string]: any } locked_read_errors = {} :: { string } - clock.now = 0 - overseer.clock = function(): number return clock.now end + clock.now = overseer.time.now() + overseer.now = function(): any return clock.now end + overseer.pending_wakes = function() return {}, nil end overseer.process = process_mock(observed) overseer.execution_frame = { reconstruct = function(actor_id, actor_context) @@ -608,8 +610,8 @@ local function run_tests() exit_before_admission(runtime, "unstartable") if attempt < limit then test.eq(#observed.spawns, attempt, "a restart waits for its retry time") - clock.now = clock.now + overseer.start_retry_delay(attempt) - test.is_true(select(1, overseer.retry_starts(runtime))) + clock.now = clock.now:add(overseer.start_retry_delay(attempt)) + overseer.settle(runtime, {}) test.eq(#observed.spawns, attempt + 1) end end @@ -619,6 +621,7 @@ local function run_tests() test.eq(failures[1].failure.reason, "orchestrator_start_failed") test.eq(failures[1].failure.message, "not allowed: app:db") test.eq(failures[1].fence.generation, 2) + test.is_nil(runtime.starts["unstartable"], "a durable start failure retires the budget") end) test.it("keeps an exhausted start budget until its failure is persisted", function() @@ -628,8 +631,8 @@ local function run_tests() test.is_true(select(1, overseer.reconcile(runtime, "stuck"))) for attempt = 1, overseer.MAX_STARTS - 1 do exit_before_admission(runtime, "stuck") - clock.now = clock.now + overseer.start_retry_delay(attempt) - test.is_true(select(1, overseer.retry_starts(runtime))) + clock.now = clock.now:add(overseer.start_retry_delay(attempt)) + overseer.settle(runtime, {}) end local fail_activation = overseer.commit.fail_activation local attempts = 0 @@ -659,8 +662,8 @@ local function run_tests() test.is_true(select(1, overseer.reconcile(runtime, "flaky"))) exit_before_admission(runtime, "flaky") test.eq(#observed.spawns, 1) - clock.now = clock.now + overseer.start_retry_delay(1) - test.is_true(select(1, overseer.retry_starts(runtime))) + clock.now = clock.now:add(overseer.start_retry_delay(1)) + overseer.settle(runtime, {}) test.eq(#observed.spawns, 2) admit(activations.flaky, "t-flaky", tostring(observed.spawns[2].pid)) test.not_nil(select(1, overseer.safety_reconcile(runtime))) @@ -721,59 +724,172 @@ local function run_tests() test.eq(observed.spawns[1].args.activation_generation, 9) end) - test.it("arms the service loop for the next future wake behind a due one", function() - local now = overseer.time.now() - local due = { dataflow_id = "orphan", wake_key = "yield:due", - wake_at = now:add(-1 * overseer.time.SECOND):format(overseer.time.RFC3339NANO) } - local next_wake = { dataflow_id = "valid", wake_key = "yield:next", - wake_at = now:add(overseer.time.SECOND):format(overseer.time.RFC3339NANO) } - local promotions = 0 - local timers: { any } = {} - local selected: { number } = {} + -- Drives the real service loop. Each scripted step answers one select; + -- a step receives the timers armed for that iteration. + local function run_service(steps: { (any) -> any }): { any } local real_time = overseer.time - local originals_loop = { - next_pending_wake = overseer.next_pending_wake, - promote_due = overseer.promote_due, - load_runtime_epoch = overseer.load_runtime_epoch, - channel = overseer.channel, + local saved = { time = overseer.time, + channel = overseer.channel, + load_runtime_epoch = overseer.load_runtime_epoch, } - overseer.next_pending_wake = function(after: string?) - if after then return next_wake, nil end - return due, nil - end - overseer.promote_due = function() - promotions = promotions + 1 - return 0, nil - end + local inbox_channel = { case_receive = function() return "inbox" end } + local events_channel = { case_receive = function() return "events" end } + local armed: { any } = {} + local selects: { any } = {} + overseer.process.registry.register = function() return true, nil end + overseer.process.inbox = function() return inbox_channel end + overseer.process.events = function() return events_channel end overseer.load_runtime_epoch = function() return CURRENT_EPOCH, nil end overseer.time = setmetatable({ after = function(duration) - table.insert(timers, duration) - return { case_receive = function() return "timer" end }, nil + local timer: any = {} + timer.case_receive = function() return timer end + table.insert(armed, { duration = duration, channel = timer }) + return timer, nil end, }, { __index = real_time }) overseer.channel = { - select = function(cases) - table.insert(selected, #cases) - return { ok = false } + select = function() + local deadline_timer = nil + for _, timer in ipairs(armed) do + if timer.duration ~= "30s" then deadline_timer = timer end + end + local step = steps[#selects + 1] + table.insert(selects, { + spawns = #observed.spawns, + deadline = deadline_timer and deadline_timer.duration or nil, + }) + armed = {} + if not step then return { ok = false } end + return step({ + inbox = inbox_channel, + events = events_channel, + deadline = deadline_timer and deadline_timer.channel or nil, + }) end, } - overseer.process.registry.register = function() return true, nil end - overseer.process.inbox = function() return { case_receive = function() return "inbox" end } end - overseer.process.events = function() return { case_receive = function() return "events" end } end - local ok, run_err = pcall(overseer.run, {}) - for key, value in pairs(originals_loop) do overseer[key] = value end + for key, value in pairs(saved) do overseer[key] = value end test.is_true(ok, tostring(run_err)) - test.eq(selected[1], 4, "inbox, events, safety and the next wake") - local wake_timer = nil - for _, duration in ipairs(timers) do - if type(duration) == "number" then wake_timer = duration end + return selects + end + + local function hint(id: string): any + local payload = { dataflow_id = id } + return { + topic = function() return overseer.TOPIC end, + payload = function() return { data = function() return payload end } end, + } + end + + test.it("promotes a wake that fell due while the loop was promoting an earlier one", function() + local promoted: { [string]: boolean } = {} + local due_row = { dataflow_id = "a", wake_key = "yield:a", + wake_at = clock.now:add(-1 * overseer.time.SECOND):format(overseer.time.RFC3339NANO) } + local next_row = { dataflow_id = "b", wake_key = "yield:b", + wake_at = clock.now:add(1500 * overseer.time.MILLISECOND):format(overseer.time.RFC3339NANO) } + overseer.pending_wakes = function() + -- The query is slow: B's deadline passes while it runs. + clock.now = clock.now:add(2 * overseer.time.SECOND) + local rows = {} + for _, row in ipairs({ due_row, next_row }) do + if not promoted[row.wake_key] then table.insert(rows, row) end + end + return rows, nil + end + overseer.activation_repo.activate_due_tx = function(_tx, _id, key) + promoted[key] = true + return { promoted = false, already_promoted = true }, nil + end + local selects = run_service({}) + test.is_true(promoted["yield:a"]) + test.is_true(promoted["yield:b"], "B is promoted before the loop blocks") + test.is_nil(selects[1].deadline) + end) + + test.it("skips an unpromotable due wake until the next event without hiding a later one", function() + local stuck = { dataflow_id = "stuck", wake_key = "yield:stuck", + wake_at = clock.now:add(-1 * overseer.time.SECOND):format(overseer.time.RFC3339NANO) } + local later = { dataflow_id = "later", wake_key = "yield:later", + wake_at = clock.now:add(overseer.time.SECOND):format(overseer.time.RFC3339NANO) } + overseer.pending_wakes = function() return { stuck, later }, nil end + local attempts = 0 + overseer.activation_repo.activate_due_tx = function() + attempts = attempts + 1 + return nil, "database is locked" end - local armed = test.not_nil(wake_timer) :: number + local counts: { number } = {} + local selects = run_service({ + function(channels: any): any + table.insert(counts, attempts) + return { ok = true, channel = channels.inbox, value = hint("unrelated") } + end, + function(_channels: any): any + table.insert(counts, attempts) + return { ok = false } + end, + }) + test.eq(counts[1], 1) + test.eq(counts[2], 2) + local armed = test.not_nil(selects[1].deadline) :: number test.is_true(armed > 0 and armed <= 1000000000) - test.is_true(promotions >= 1) + end) + + test.it("spawns a start retry that fell due while the loop inspected wakes", function() + activations.retry = activation("retry", 1) + workflows.retry = workflow("retry") + local slow = { enabled = false } + overseer.pending_wakes = function() + if slow.enabled then clock.now = clock.now:add(2 * overseer.time.SECOND) end + return {}, nil + end + local selects = run_service({ + function(channels) + local first = observed.spawns[1] + observed.owners["dataflow.retry"] = nil + slow.enabled = true + return { ok = true, channel = channels.events, value = { + kind = overseer.process.event.EXIT, from = first.pid, + result = { error = "not allowed: app:db" }, + } } + end, + }) + test.eq(selects[1].spawns, 1) + test.eq(selects[2].spawns, 2, "the retry is spawned before the loop blocks again") + end) + + test.it("retires the start budget once the orchestrator is admitted", function() + activations.admitted = activation("admitted", 1) + workflows.admitted = workflow("admitted") + local function reads(): number + local count = 0 + for _, id in ipairs(observed.locked_reads) do + if id == "admitted" then count = count + 1 end + end + return count + end + local marks: { number } = {} + run_service({ + function(channels) + admit(activations.admitted, "t-admitted", tostring(observed.spawns[1].pid)) + return { ok = true, channel = channels.inbox, value = hint("admitted") } + end, + function(channels) + table.insert(marks, reads()) + clock.now = clock.now:add(60 * overseer.time.SECOND) + return { ok = true, channel = channels.inbox, value = hint("unrelated") } + end, + function(channels) + return { ok = true, channel = channels.inbox, value = hint("unrelated") } + end, + function(channels) + table.insert(marks, reads()) + return { ok = true, channel = channels.inbox, value = hint("unrelated") } + end, + }) + test.eq(#observed.spawns, 1) + test.eq(marks[2], marks[1], "an admitted owner is not reconciled for unrelated events") end) test.it("recognizes missing SQLite and PostgreSQL migration state", function() From 83e1b4ec868d9d2435c51ce645ef8afc103204ed Mon Sep 17 00:00:00 2001 From: Wolfy-J Date: Wed, 23 Sep 2026 15:09:51 -0400 Subject: [PATCH 09/10] fix(migrations): remove activation and wake rows of deleted workflows; prove upgrades from an older release - Migration 11 deletes activation and wake rows without a workflow, which releases before 10_add_activation_ownership left on SQLite. - restart-proof.sh takes DATAFLOW_RESTART_FROM: the first runtime runs that older release, the second upgrades its database in place. On SQLite the proof seeds rows of a deleted workflow and checks the upgrade removes them; it also checks the recovered owner record. Duplicates are counted against the workflows present before the restart, since the first runtime's probe can race its migrations. - make test-upgrade-sqlite / test-upgrade-postgres (UPGRADE_FROM, default v0.7.19) run it outside the default test target. --- Makefile | 18 ++++++- scripts/restart-proof.sh | 51 +++++++++++++++++-- src/_index.yaml | 2 + .../11_remove_orphaned_activation_rows.lua | 32 ++++++++++++ src/migrations/_index.yaml | 14 +++++ 5 files changed, 112 insertions(+), 5 deletions(-) create mode 100644 src/migrations/11_remove_orphaned_activation_rows.lua diff --git a/Makefile b/Makefile index 4cadb0e..316bdf4 100644 --- a/Makefile +++ b/Makefile @@ -14,10 +14,11 @@ DATAFLOW_PG_DATABASE ?= dataflow_test DATAFLOW_PG_USERNAME ?= dataflow DATAFLOW_PG_PASSWORD ?= dataflow WIPPY ?= wippy +UPGRADE_FROM ?= v0.7.19 TESTS ?= PUBLISH_DRY_RUN_TOKEN ?= wpy_ci_dry_run_0123456789abcdef0123456789abcdef -.PHONY: test test-sqlite test-postgres test-restart-sqlite test-restart-postgres test-static lint install verify-lock verify-package clean +.PHONY: test test-sqlite test-postgres test-restart-sqlite test-restart-postgres test-upgrade-sqlite test-upgrade-postgres test-static lint install verify-lock verify-package clean test: test-sqlite @@ -44,6 +45,21 @@ test-restart-postgres: test-static DATAFLOW_PG_PASSWORD=$(DATAFLOW_PG_PASSWORD) \ ./scripts/restart-proof.sh +# Upgrade proofs start the first runtime on an older release ($(UPGRADE_FROM)) +# and need its modules installed, so they are not part of the default test. +test-upgrade-sqlite: test-static + DATAFLOW_RESTART_FROM=$(UPGRADE_FROM) ./scripts/restart-proof.sh + +test-upgrade-postgres: test-static + DATAFLOW_RESTART_FROM=$(UPGRADE_FROM) \ + DATAFLOW_RESTART_DIALECT=postgres \ + DATAFLOW_PG_HOST=$(DATAFLOW_PG_HOST) \ + DATAFLOW_PG_PORT=$(DATAFLOW_PG_PORT) \ + DATAFLOW_PG_DATABASE=$(DATAFLOW_PG_DATABASE)_upgrade \ + DATAFLOW_PG_USERNAME=$(DATAFLOW_PG_USERNAME) \ + DATAFLOW_PG_PASSWORD=$(DATAFLOW_PG_PASSWORD) \ + ./scripts/restart-proof.sh + test-static: @command -v rg >/dev/null 2>&1 || { echo "test-static requires ripgrep (rg)"; exit 1; } @if rg -n "keeper\\.views\\.dataflow|dataflow-link|Open full view" src/session/views/state.jet; then \ diff --git a/scripts/restart-proof.sh b/scripts/restart-proof.sh index da53b40..4345486 100755 --- a/scripts/restart-proof.sh +++ b/scripts/restart-proof.sh @@ -13,6 +13,10 @@ pg_port="${DATAFLOW_PG_PORT:-5432}" pg_database="${DATAFLOW_PG_DATABASE:-dataflow_restart_test}" pg_username="${DATAFLOW_PG_USERNAME:-dataflow}" pg_password="${DATAFLOW_PG_PASSWORD:-dataflow}" +# With DATAFLOW_RESTART_FROM set to a git ref, the first runtime runs that +# older release and the second runtime upgrades its database in place. +from_ref="${DATAFLOW_RESTART_FROM:-}" +phase1_dir="$test_dir" case "$dialect" in sqlite|postgres) ;; @@ -68,7 +72,7 @@ start_runtime() { --set "vars.postgres_password=$pg_password" >"$runtime_log" 2>&1 & else wippy run -s --profile sqlite --profile "$restart_profile" \ - --set vars.sqlite_file=./.wippy/restart-proof.db >"$runtime_log" 2>&1 & + --set "vars.sqlite_file=$db_file" >"$runtime_log" 2>&1 & fi runtime_pid=$! } @@ -82,7 +86,16 @@ if [ "$dialect" = "postgres" ]; then -h "$pg_host" -p "$pg_port" -U "$pg_username" "$pg_database" fi -cd "$test_dir" +if [ -n "$from_ref" ]; then + old_root="$test_dir/.wippy/restart-from" + rm -rf "$old_root" + mkdir -p "$old_root" + git -C "$repo_dir" archive "$from_ref" | tar -x -C "$old_root" + (cd "$old_root/test" && wippy install >/dev/null) + phase1_dir="$old_root/test" +fi + +cd "$phase1_dir" start_runtime restart_create "$phase1_log" phase1="" @@ -168,6 +181,10 @@ while [ "$attempts" -lt 100 ]; do done [ "$rolling_refilled" = "1" ] || { echo "rolling window did not refill before the slow iteration completed" >&2; exit 1; } +# The first runtime's probe can race its migrations and leave an unstarted +# workflow behind; only workflows created after the restart are duplicates. +workflows_before_restart=$(query "SELECT COUNT(*) FROM dataflows;") + stop_runtime "$runtime_pid" runtime_pid="" @@ -181,6 +198,18 @@ preserved=$(query " ") [ "$preserved" = "1" ] || { echo "graceful runtime shutdown destroyed active intent" >&2; exit 1; } +orphan_id="" +if [ -n "$from_ref" ] && [ "$dialect" = "sqlite" ]; then + # Releases before the activation ownership record deleted workflows without + # their activation row on SQLite, which does not enforce the cascade. + orphan_id="00000000-0000-7000-8000-000000000001" + query "INSERT INTO dataflow_activations(dataflow_id, generation, desired_active, requested_at, updated_at) + VALUES ('$orphan_id', 1, 1, '2026-01-01T00:00:00Z', '2026-01-01T00:00:00Z'); + INSERT INTO dataflow_wakes(dataflow_id, wake_key, wake_at) + VALUES ('$orphan_id', 'yield:orphan', '2026-01-01T00:00:00Z');" +fi + +cd "$test_dir" start_runtime restart_observe "$phase2_log" second_epoch="" @@ -268,6 +297,20 @@ yield_result_rows=$(query " } workflow_count=$(query "SELECT COUNT(*) FROM dataflows;") -[ "$workflow_count" = "1" ] || { echo "restart created duplicate workflows: $workflow_count" >&2; exit 1; } +[ "$workflow_count" = "$workflows_before_restart" ] || { + echo "restart created duplicate workflows: $workflows_before_restart -> $workflow_count" >&2 + exit 1 +} + +ownership=$(query "SELECT owner_phase || '|' || CASE WHEN owner_token IS NULL THEN 'none' ELSE 'token' END + || '|' || CASE WHEN owner_pid IS NULL THEN 'none' ELSE 'pid' END + FROM dataflow_activations WHERE dataflow_id = '$dataflow_id';") +[ "$ownership" = "released|token|pid" ] || { echo "recovered workflow has unexpected ownership: $ownership" >&2; exit 1; } + +if [ -n "$orphan_id" ]; then + orphans=$(query "SELECT (SELECT COUNT(*) FROM dataflow_activations WHERE dataflow_id = '$orphan_id') + + (SELECT COUNT(*) FROM dataflow_wakes WHERE dataflow_id = '$orphan_id');") + [ "$orphans" = "0" ] || { echo "upgrade left activation or wake rows of a deleted workflow" >&2; exit 1; } +fi -echo "$dialect restart proof passed: $dataflow_id $first_epoch -> $second_epoch" +echo "$dialect restart proof passed${from_ref:+ from $from_ref}: $dataflow_id $first_epoch -> $second_epoch" diff --git a/src/_index.yaml b/src/_index.yaml index 77de79b..a5fbf3a 100644 --- a/src/_index.yaml +++ b/src/_index.yaml @@ -270,6 +270,8 @@ entries: path: ".meta.target_db" - entry: userspace.dataflow.migrations:10_add_activation_ownership path: ".meta.target_db" + - entry: userspace.dataflow.migrations:11_remove_orphaned_activation_rows + path: ".meta.target_db" - entry: userspace.dataflow.runner:overseer.service path: ".lifecycle.depends_on +=" - entry: userspace.dataflow.env:retention_db diff --git a/src/migrations/11_remove_orphaned_activation_rows.lua b/src/migrations/11_remove_orphaned_activation_rows.lua new file mode 100644 index 0000000..9cb80fe --- /dev/null +++ b/src/migrations/11_remove_orphaned_activation_rows.lua @@ -0,0 +1,32 @@ +local function execute_or_error(db, query) + local success, err = db:execute(query) + if err then error(err) end + return success +end + +-- SQLite does not enforce the cascading foreign keys, and releases before the +-- activation ownership record deleted workflows without their activation row. +-- Such rows, and wakes left the same way, belong to no workflow. +local ORPHANS = { + [[DELETE FROM dataflow_wakes WHERE NOT EXISTS ( + SELECT 1 FROM dataflows WHERE dataflows.dataflow_id = dataflow_wakes.dataflow_id)]], + [[DELETE FROM dataflow_activations WHERE NOT EXISTS ( + SELECT 1 FROM dataflows WHERE dataflows.dataflow_id = dataflow_activations.dataflow_id)]], +} + +local function remove_orphans(db) + for _, statement in ipairs(ORPHANS) do execute_or_error(db, statement) end +end + +return require("migration").define(function() + migration("Remove activation and wake rows of deleted workflows", function() + database("postgres", function() + up(remove_orphans) + down(function() end) + end) + database("sqlite", function() + up(remove_orphans) + down(function() end) + end) + end) +end) diff --git a/src/migrations/_index.yaml b/src/migrations/_index.yaml index 2832352..ffdfda2 100644 --- a/src/migrations/_index.yaml +++ b/src/migrations/_index.yaml @@ -212,3 +212,17 @@ entries: imports: migration: wippy.migration:migration method: migrate + + - name: 11_remove_orphaned_activation_rows + kind: function.lua + meta: + type: migration + tags: [dataflows, activation, durability] + description: Remove activation and wake rows whose workflow was deleted where foreign keys did not cascade + depends_on: [ns:wippy.migration] + target_db: app:db + timestamp: "2026-09-23T12:00:00Z" + source: file://11_remove_orphaned_activation_rows.lua + imports: + migration: wippy.migration:migration + method: migrate From f4d2603971560748176b424a15278486b52008e9 Mon Sep 17 00:00:00 2001 From: Wolfy-J Date: Wed, 23 Sep 2026 15:33:40 -0400 Subject: [PATCH 10/10] fix(runner): keep draining past the settle budget and run the safety pass on an absolute schedule - When settle exhausts its pass budget while still making progress, the next deadline is now, so draining continues on the next iteration after pending events instead of waiting for the safety pass. - The safety deadline is absolute and advances only when the safety pass runs; it is part of the single deadline timer, so deadline events cannot postpone it. The safety pass clears the excluded items. - A pending wake whose deadline cannot be read counts as progress, so a page of such rows cannot hide later valid wakes. register_yield_wake_tx rejects a deadline that is not an RFC 3339 time, since SQLite stores it as text. --- src/persist/_index.yaml | 2 +- src/persist/activation_repo.lua | 6 ++ src/persist/activation_repo_test.lua | 13 +++ src/runner/overseer.lua | 43 +++++----- src/runner/overseer_test.lua | 114 ++++++++++++++++++++++++++- 5 files changed, 157 insertions(+), 21 deletions(-) diff --git a/src/persist/_index.yaml b/src/persist/_index.yaml index e792b45..317bcb1 100644 --- a/src/persist/_index.yaml +++ b/src/persist/_index.yaml @@ -8,7 +8,7 @@ entries: private: true comment: Generation-fenced desired activation state for durable orchestrator lives. source: file://activation_repo.lua - modules: [sql, json] + modules: [sql, json, time] imports: dataflow_consts: userspace.dataflow:consts diff --git a/src/persist/activation_repo.lua b/src/persist/activation_repo.lua index 49b6d30..2756bf6 100644 --- a/src/persist/activation_repo.lua +++ b/src/persist/activation_repo.lua @@ -1,5 +1,6 @@ local sql = require("sql") local json = require("json") +local time = require("time") local consts = require("dataflow_consts") local activation_repo = {} @@ -643,6 +644,11 @@ function activation_repo.register_yield_wake_tx(tx, dataflow_id, yield_id, wake_ if type(yield_id) ~= "string" or yield_id == "" then return nil, "yield_id is required" end valid, validation_err = validate_timestamp(wake_at, "wake_at") if not valid then return nil, validation_err end + -- The overseer schedules wakes by this deadline; SQLite stores it as text + -- and would otherwise accept a value no clock can reach. + local _, parse_err = time.parse(time.RFC3339NANO, wake_at) + if parse_err then _, parse_err = time.parse(time.RFC3339, wake_at) end + if parse_err then return nil, "yield deadline must be an RFC 3339 time: " .. tostring(wake_at) end local result, write_err = tx_execute(tx, [[ INSERT INTO dataflow_wakes(dataflow_id, wake_key, wake_at, activation_generation) diff --git a/src/persist/activation_repo_test.lua b/src/persist/activation_repo_test.lua index 3e5722c..15843a0 100644 --- a/src/persist/activation_repo_test.lua +++ b/src/persist/activation_repo_test.lua @@ -544,6 +544,19 @@ local function define_tests() test.is_nil(wake_generation(id, wake_key)) end) + test.it("registers a yield deadline only when it is a valid RFC 3339 time", function() + local id = create_dataflow(consts.STATUS.RUNNING) + local _, malformed_err = transaction(function(tx) + return activation_repo.register_yield_wake_tx(tx, id, "malformed", "next tuesday") + end) + test.contains(tostring(malformed_err), "RFC 3339") + test.eq(wake_count(id), 0) + test.not_nil(select(1, transaction(function(tx) + return activation_repo.register_yield_wake_tx(tx, id, "valid", now(60)) + end))) + test.eq(wake_count(id), 1) + end) + test.it("converges terminal activation and every stale wake during due promotion", function() local id = create_dataflow(consts.STATUS.RUNNING) test.not_nil(select(1, transaction(function(tx) diff --git a/src/runner/overseer.lua b/src/runner/overseer.lua index 4dfa364..9b74a55 100644 --- a/src/runner/overseer.lua +++ b/src/runner/overseer.lua @@ -23,7 +23,7 @@ local M = { local NAME = "dataflow.overseer" local TOPIC = "dataflow.activation.changed" -local SAFETY_INTERVAL = "30s" +local SAFETY_INTERVAL_NS = 30000000000 local SCAN_LIMIT = 100 type OwnershipState = { @@ -596,9 +596,11 @@ local function earliest(current: any?, candidate: any): any end -- Act on every wake and start retry that is due, then return the earliest --- deadline still ahead. A due item whose processing makes no progress is --- recorded in `attempted` and skipped until the next event or safety pass, so --- it neither hides later deadlines nor makes the loop spin. +-- deadline still ahead. A due item whose processing makes no progress, or whose +-- deadline cannot be read, is recorded in `attempted` and skipped until the next +-- event or safety pass, so it neither hides later deadlines nor makes the loop +-- spin. When the pass budget runs out while work remains, the next deadline is +-- now: draining continues on the next iteration, after pending events. function M.settle(runtime: Runtime, attempted: { [string]: boolean }): any? local ahead: any? = nil for _ = 1, MAX_SETTLE_PASSES do @@ -622,6 +624,7 @@ function M.settle(runtime: Runtime, attempted: { [string]: boolean }): any? if deadline == nil then attempted[key] = true log_flow("pending wake has an invalid deadline", tostring(row.dataflow_id), row.wake_at) + progressed = true elseif M.now():before(deadline) then ahead = earliest(ahead, deadline) break @@ -658,7 +661,7 @@ function M.settle(runtime: Runtime, attempted: { [string]: boolean }): any? if not progressed then return ahead end end - return ahead + return M.now() end function M.run(_args: any) @@ -671,21 +674,27 @@ function M.run(_args: any) local events = M.process.events() local attempted: { [string]: boolean } = {} + -- The safety pass runs on an absolute schedule, so a stream of deadline + -- events cannot postpone it; it also retries every excluded item. + local next_safety_at = M.now():add(SAFETY_INTERVAL_NS) while true do + if not M.now():before(next_safety_at) then + attempted = {} + reconcile_or_log(runtime, runtime.bootstrapped and M.safety_reconcile or M.bootstrap) + next_safety_at = M.now():add(SAFETY_INTERVAL_NS) + end + -- One deadline computation: everything overdue is processed first, then -- a single timer waits for the earliest deadline still ahead. - local deadline: any? = nil - if runtime.bootstrapped then deadline = M.settle(runtime, attempted) end - local deadline_timer: any = nil - if deadline ~= nil then - local wait_ns = deadline:sub(M.now()):nanoseconds() - if wait_ns < 1 then wait_ns = 1 end - deadline_timer = M.time.after(wait_ns) + local deadline = next_safety_at + if runtime.bootstrapped then + local ahead = M.settle(runtime, attempted) + if ahead ~= nil then deadline = earliest(deadline, ahead) end end - - local safety_timer = M.time.after(SAFETY_INTERVAL) - local cases = { inbox:case_receive(), events:case_receive(), safety_timer:case_receive() } - if deadline_timer then table.insert(cases, deadline_timer:case_receive()) end + local wait_ns = deadline:sub(M.now()):nanoseconds() + if wait_ns < 1 then wait_ns = 1 end + local deadline_timer = M.time.after(wait_ns) + local cases = { inbox:case_receive(), events:case_receive(), deadline_timer:case_receive() } local result = M.channel.select(cases) if not result.ok then break end @@ -716,8 +725,6 @@ function M.run(_args: any) elseif topic == TOPIC then reconcile_or_log(runtime, M.promote_due) end - elseif result.channel ~= deadline_timer then - reconcile_or_log(runtime, runtime.bootstrapped and M.safety_reconcile or M.bootstrap) end end return { status = "shutdown" } diff --git a/src/runner/overseer_test.lua b/src/runner/overseer_test.lua index 1af24b9..47216b6 100644 --- a/src/runner/overseer_test.lua +++ b/src/runner/overseer_test.lua @@ -783,6 +783,115 @@ local function run_tests() } end + local function wake_row(id: string, offset_ms: number): any + return { + dataflow_id = id, + wake_key = "yield:" .. id, + wake_at = clock.now:add(offset_ms * overseer.time.MILLISECOND):format(overseer.time.RFC3339NANO), + } + end + + -- Pending rows in deadline order; promotion removes a row. + local function wake_table(rows: { any }): { [string]: boolean } + local promoted: { [string]: boolean } = {} + overseer.pending_wakes = function(limit: number) + local page = {} + for _, row in ipairs(rows) do + if not promoted[row.wake_key] and #page < limit then table.insert(page, row) end + end + return page, nil + end + overseer.activation_repo.activate_due_tx = function(_tx, _id, key) + promoted[key] = true + return { promoted = false, already_promoted = true }, nil + end + return promoted + end + + test.it("continues draining due wakes without a pause after a full settle", function() + local rows: { any } = {} + for index = 1, 850 do table.insert(rows, wake_row("backlog-" .. index, -1000)) end + local promoted = wake_table(rows) + local selects = run_service({ + function(channels: any): any + return { ok = true, channel = channels.deadline } + end, + }) + local continuation = test.not_nil(selects[1].deadline) :: number + test.is_true(continuation <= 1000000, "the next pass is armed at once") + local count = 0 + for _ in pairs(promoted) do count = count + 1 end + test.eq(count, 850) + end) + + test.it("retries an excluded wake on the safety deadline despite continuous deadlines", function() + local rows: { any } = { wake_row("excluded", -1000) } + for index = 1, 60 do table.insert(rows, wake_row("tick-" .. index, index * 1000)) end + local promoted = wake_table(rows) + local accept = overseer.activation_repo.activate_due_tx + local failed = { once = false } + overseer.activation_repo.activate_due_tx = function(tx, id, key) + if key == "yield:excluded" and not failed.once then + failed.once = true + return nil, "database is locked" + end + return accept(tx, id, key) + end + local steps: { (any) -> any } = {} + for _ = 1, 60 do + table.insert(steps, function(channels: any): any + clock.now = clock.now:add(overseer.time.SECOND) + return { ok = true, channel = channels.deadline } + end) + end + local at_retry = { ticks = nil :: number? } + overseer.activation_repo.activate_due_tx = (function(inner) + return function(tx, id, key) + if key == "yield:excluded" and failed.once and at_retry.ticks == nil then + local ticks = 0 + for name in pairs(promoted) do + if name ~= "yield:excluded" then ticks = ticks + 1 end + end + at_retry.ticks = ticks + end + return inner(tx, id, key) + end + end)(overseer.activation_repo.activate_due_tx) + run_service(steps) + test.is_true(promoted["yield:excluded"], "the excluded wake is retried") + local ticks = test.not_nil(at_retry.ticks) :: number + test.is_true(ticks <= 31, "retried by the safety deadline, after " .. ticks .. " deadline events") + end) + + test.it("settles a wake that appears while an earlier promotion advances the clock", function() + local rows: { any } = { wake_row("first", -1000) } + local promoted = wake_table(rows) + local accept = overseer.activation_repo.activate_due_tx + overseer.activation_repo.activate_due_tx = function(tx, id, key) + if key == "yield:first" then + clock.now = clock.now:add(2 * overseer.time.SECOND) + table.insert(rows, wake_row("appeared", -500)) + end + return accept(tx, id, key) + end + local selects = run_service({}) + test.is_true(promoted["yield:first"]) + test.is_true(promoted["yield:appeared"], "the next pass promotes it before the loop blocks") + test.eq(selects[1].spawns, 0) + end) + + test.it("never lets a page of malformed deadlines hide a valid wake", function() + local rows: { any } = {} + for index = 1, 120 do + table.insert(rows, { dataflow_id = "bad-" .. index, wake_key = "yield:bad-" .. index, + wake_at = "0000-bad-" .. string.format("%03d", index) }) + end + table.insert(rows, wake_row("valid", -1000)) + local promoted = wake_table(rows) + run_service({}) + test.is_true(promoted["yield:valid"]) + end) + test.it("promotes a wake that fell due while the loop was promoting an earlier one", function() local promoted: { [string]: boolean } = {} local due_row = { dataflow_id = "a", wake_key = "yield:a", @@ -805,7 +914,8 @@ local function run_tests() local selects = run_service({}) test.is_true(promoted["yield:a"]) test.is_true(promoted["yield:b"], "B is promoted before the loop blocks") - test.is_nil(selects[1].deadline) + local armed = test.not_nil(selects[1].deadline) :: number + test.is_true(armed > 20 * 1000000000, "only the safety deadline remains") end) test.it("skips an unpromotable due wake until the next event without hiding a later one", function() @@ -877,7 +987,7 @@ local function run_tests() end, function(channels) table.insert(marks, reads()) - clock.now = clock.now:add(60 * overseer.time.SECOND) + clock.now = clock.now:add(10 * overseer.time.SECOND) return { ok = true, channel = channels.inbox, value = hint("unrelated") } end, function(channels)