diff --git a/src/caw/executor.py b/src/caw/executor.py index bf5b92e..071dea3 100644 --- a/src/caw/executor.py +++ b/src/caw/executor.py @@ -1004,6 +1004,21 @@ async def resume_run( f"run {run_id!r} is not resumable (status: {prior_status}); " f"only an interrupted or failed run can be resumed" ) + # Cross-schema resume guard (#76): the `node.cause` column was added (#7) + # without a migration, so a run directory created before it has a `node` + # table lacking `cause`. The first terminal Node write would then crash with + # a raw `sqlite3.OperationalError` mid-resume, after the Run row has already + # flipped back to `running`, leaving an interrupted run in a worse state. + # caw is pre-1.0 with no documented state-schema-stability guarantee, so + # resume refuses such a stale directory up front with an actionable error + # rather than migrating it in place. + if not state.node_table_has_cause(): + raise ResumeError( + f"run directory {run_dir} predates a State schema change " + f"(the `node` table has no `cause` column) and cannot be resumed; " + f"this run was created by an older caw version whose State schema is " + f"not forward-compatible with this version" + ) node_statuses = state.node_statuses(run_id) max_attempts = state.max_attempt_per_node(run_id) # A `succeeded` Node is done; every other recorded Node is re-run. A re-run diff --git a/src/caw/state.py b/src/caw/state.py index 77e9d45..8b51d94 100644 --- a/src/caw/state.py +++ b/src/caw/state.py @@ -163,6 +163,20 @@ def record_attempt( (run_id, node_id, attempt, started_at, finished_at, exit_status, json.dumps(output)), ) + def node_table_has_cause(self) -> bool: + """Whether the `node` table carries the `cause` column (#76). + + The `cause` column was added (#7) via ``CREATE TABLE IF NOT EXISTS`` only, + which is a no-op against a `node` table that already exists, so a run + directory created before that column has a `node` table lacking it. Every + terminal Node write goes through ``record_node_finished``, which always + sets `cause`, so a missing column makes the FIRST such write crash with a + raw ``sqlite3.OperationalError``. Resume reads this to refuse a pre-`cause` + run directory up front with an actionable error instead (#76). + """ + columns = self._connection.execute("PRAGMA table_info(node)").fetchall() + return any(column[1] == "cause" for column in columns) + def run_status(self, run_id: str) -> str | None: """The recorded status of a Run, or ``None`` if no such Run exists. diff --git a/tests/test_executor_seam.py b/tests/test_executor_seam.py index f421074..4d90ae7 100644 --- a/tests/test_executor_seam.py +++ b/tests/test_executor_seam.py @@ -1749,3 +1749,58 @@ async def test_cancelling_a_run_terminates_the_in_flight_node_subprocess(tmp_pat (node,) = state_rows(run_dir, "SELECT * FROM node") assert node["status"] == "errored" assert read_events(run_dir)[-1]["type"] == "run_errored" + + +def drop_node_cause_column(run_dir: Path) -> None: + """Rewrite a run directory's `node` table to the pre-`cause` schema (#76). + + PR #74 added the `node.cause` column via `CREATE TABLE IF NOT EXISTS` only, + which is a no-op against a `node` table that already exists, so a run directory + created by a caw version before that column simply has no `cause` column. This + reproduces that stale schema by rebuilding the table without `cause` while + preserving its rows, so a resume reopens a genuinely pre-`cause` State. + """ + connection = sqlite3.connect(run_dir / "state.sqlite") + try: + connection.executescript( + "PRAGMA foreign_keys = OFF;" + "CREATE TABLE node_legacy (" + " run_id TEXT NOT NULL REFERENCES run (run_id)," + " node_id TEXT NOT NULL," + " status TEXT NOT NULL," + " PRIMARY KEY (run_id, node_id)" + ");" + "INSERT INTO node_legacy (run_id, node_id, status)" + " SELECT run_id, node_id, status FROM node;" + "DROP TABLE node;" + "ALTER TABLE node_legacy RENAME TO node;" + ) + connection.commit() + finally: + connection.close() + + +@pytest.mark.asyncio +async def test_resuming_a_run_whose_node_table_predates_the_cause_column_is_refused( + tmp_path: Path, +) -> None: + # Cross-schema resume guard (#76): PR #74 added `node.cause` with no migration, + # so a run directory created before that column has a `node` table without it. + # Resume then drives the first terminal node through `record_node_finished`, + # which always writes `cause`, raising a raw `sqlite3.OperationalError: no such + # column: cause` that crashes the resume. Resume must instead REFUSE up front + # with an actionable ResumeError naming the stale schema, like the other #70 + # resume guards. The first run fails so it is resume-eligible, then its `node` + # table is rewritten to the pre-`cause` schema. + runs_root = tmp_path / "runs" + workflow = shell_workflow("exit 7") + first = await execute_run(workflow, runs_root) + assert not first.succeeded, "the first run fails, so it is resume-eligible" + + run_dir = single_run_dir(runs_root) + drop_node_cause_column(run_dir) + + with pytest.raises(ResumeError, match="schema") as excinfo: + await resume_run(first.run_id, runs_root) + # The raw sqlite error must not leak; the refusal stands on its own. + assert not isinstance(excinfo.value.__cause__, sqlite3.OperationalError)