From 9d4699975d5c9ebeb19bf686f621410352276695 Mon Sep 17 00:00:00 2001 From: Praveen Perera Date: Fri, 2 Oct 2026 23:31:36 -0500 Subject: [PATCH 01/31] Remove the GPU resource loan subsystem The loan design never ran live work and is being replaced by a GPU priority queue that accepts scripts and preempts at checkpoints. This deletes loans, supervisors, return work, release watchers, trainer attempts, attestations, their routes, CLI, web views, docs, and skill references. Schema version 35 drops the loan tables and the rows those flows wrote into shared route, cancellation, and message tables. Ordinary tasks are kept, and every released schema version upgrades to the fresh schema. Two tests counted callbacks that can arrive at any time after a task ends; they now count only the callbacks they check. --- .agents/skills/homebased/SKILL.md | 2 - .agents/skills/homebased/references/errors.md | 4 +- .../skills/homebased/references/inspect.md | 2 +- .../homebased/references/resource-loans.md | 137 - .agents/skills/homebased/references/submit.md | 4 +- README.md | 8 - docs/resource-loans.md | 672 --- src/cancellation.rs | 392 +- src/cli.rs | 13 - src/cli/release_watcher.rs | 363 -- src/cli/resource.rs | 3630 ----------------- src/client.rs | 8 - src/container.rs | 8 +- src/container/docker.rs | 21 - src/container/spec.rs | 18 +- src/daemon.rs | 22 - src/daemon/actors.rs | 4 +- src/daemon/actors/resource.rs | 2057 ---------- src/daemon/actors/resource/test_support.rs | 42 - src/daemon/actors/resource/tests.rs | 667 --- src/daemon/actors/store.rs | 1065 +---- src/daemon/actors/supervisor.rs | 491 +-- src/daemon/actors/supervisor/recovery.rs | 118 +- .../actors/supervisor/resource_launch.rs | 879 ---- src/daemon/actors/supervisor/tests.rs | 3001 +------------- src/daemon/actors/task.rs | 2 +- src/daemon/api.rs | 64 +- src/daemon/cancel_delivery.rs | 143 +- src/daemon/cluster.rs | 1246 +----- src/daemon/event_sender.rs | 4 - src/daemon/inspection.rs | 126 +- src/daemon/local_submit.rs | 20 +- src/daemon/message_receiver.rs | 213 +- src/daemon/origin_submit.rs | 68 +- src/daemon/release_watcher_api.rs | 77 - src/daemon/resource_action.rs | 941 ----- src/daemon/resource_api.rs | 2292 ----------- src/daemon/resource_api/tests.rs | 1903 --------- src/daemon/resource_background.rs | 886 ---- src/daemon/resource_notice_delivery.rs | 545 --- src/daemon/resource_notice_sender.rs | 536 --- src/daemon/resource_submit.rs | 978 ----- src/daemon/web.rs | 67 +- src/digest.rs | 172 - src/domain.rs | 4 +- src/error.rs | 155 +- src/invocation.rs | 1 - src/lib.rs | 2 - src/message.rs | 94 +- src/resource.rs | 2255 ---------- src/resource/api.rs | 629 --- src/resource/background_launch.rs | 590 --- src/resource/bound_action.rs | 767 ---- src/resource/command_shape.rs | 1613 -------- src/resource/foreground.rs | 596 --- src/resource/id.rs | 149 - src/resource/initial_idle.rs | 184 - src/resource/operator_release.rs | 524 --- src/resource/ownership_lock.rs | 220 - .../ownership_lock/attempt_evidence.rs | 1071 ----- src/resource/ownership_lock/lock_probe.rs | 546 --- src/resource/release_checkpoint.rs | 204 - src/resource/release_watcher.rs | 338 -- src/resource/return_window.rs | 269 -- src/resource/store.rs | 77 - src/resource/store/acceptance.rs | 225 - src/resource/store/assigned_task.rs | 674 --- src/resource/store/cancellation.rs | 325 -- src/resource/store/checkpoint.rs | 215 - src/resource/store/codec.rs | 107 - src/resource/store/error.rs | 308 -- src/resource/store/notice.rs | 443 -- src/resource/store/provenance.rs | 302 -- src/resource/store/queue.rs | 418 -- src/resource/store/release_completion.rs | 729 ---- src/resource/store/release_loan.rs | 465 --- src/resource/store/release_watcher.rs | 340 -- src/resource/store/resources.rs | 172 - src/resource/store/return_window.rs | 375 -- src/resource/store/revision.rs | 35 - src/resource/store/rows.rs | 200 - src/resource/store/test_support.rs | 234 -- src/resource/store/tests.rs | 2574 ------------ src/resource/trainer_publication.rs | 2832 ------------- src/runner/container.rs | 5 - src/spec.rs | 2 +- src/store.rs | 1159 ++---- src/store/cancellation.rs | 297 +- src/store/container.rs | 25 - src/store/dependency.rs | 23 +- src/store/events.rs | 453 +- .../fixtures/loan_schema_v34.sql} | 7 +- src/store/identity.rs | 1340 +----- src/store/resource.rs | 325 -- src/store/resource/action_task.rs | 424 -- src/store/resource/assigned_task.rs | 400 -- src/store/resource/background.rs | 1504 ------- src/store/resource/cancellation.rs | 265 -- src/store/resource/controls.rs | 721 ---- src/store/resource/controls/test_support.rs | 56 - src/store/resource/initial_idle.rs | 224 - src/store/resource/operator_release.rs | 1010 ----- src/store/resource/release_checkpoint.rs | 508 --- src/store/resource/release_completion.rs | 421 -- src/store/resource/release_proof.rs | 218 - src/store/resource/release_watcher.rs | 543 --- src/store/resource/restore.rs | 1809 -------- src/store/resource/test_support.rs | 293 -- src/store/resource/tests.rs | 27 - src/store/resource/tests/assigned_task.rs | 1386 ------- src/store/resource/tests/background.rs | 1082 ----- src/store/resource/tests/cancellation.rs | 704 ---- .../resource/tests/completed_boundary.rs | 376 -- src/store/resource/tests/container.rs | 603 --- src/store/resource/tests/controls.rs | 317 -- src/store/resource/tests/docker.rs | 392 -- src/store/resource/tests/ended_trainer.rs | 461 --- src/store/resource/tests/fixtures.rs | 1562 ------- src/store/resource/tests/foreground.rs | 193 - src/store/resource/tests/initial_idle.rs | 211 - src/store/resource/tests/operator_release.rs | 2036 --------- src/store/resource/tests/queue_authority.rs | 119 - src/store/resource/tests/registration.rs | 84 - .../resource/tests/release_checkpoint.rs | 773 ---- .../resource/tests/release_completion.rs | 898 ---- .../resource/tests/release_transaction.rs | 392 -- src/store/resource/tests/release_watcher.rs | 1780 -------- .../tests/release_watcher_acceptance.rs | 425 -- .../resource/tests/release_watcher_binding.rs | 427 -- src/store/resource/tests/restore.rs | 1909 --------- src/store/resource/tests/return_deadline.rs | 235 -- src/store/resource/tests/supervised_launch.rs | 636 --- .../resource/tests/trainer_association.rs | 589 --- src/store/resource/trainer_association.rs | 344 -- src/store/resource/trainer_lock.rs | 251 -- src/submission.rs | 837 +--- tests/fleet.rs | 2061 +--------- tests/integration.rs | 26 +- tests/message_receiver.rs | 245 +- tests/message_sender.rs | 9 +- web/src/lib/api.ts | 17 - web/src/lib/components/ExpandToggle.svelte | 35 - .../lib/components/ResourceQueuePanel.svelte | 344 -- web/src/lib/components/TaskList.svelte | 11 - web/src/lib/daemon.svelte.ts | 191 - web/src/lib/format.ts | 11 - web/src/lib/resource-actions.ts | 28 - web/src/lib/resource-operations.svelte.ts | 60 - web/src/lib/resource-operations.test.ts | 283 -- web/src/lib/resource-operations.ts | 355 -- web/src/lib/resource-state.test.ts | 220 - web/src/lib/resource-state.ts | 400 -- web/src/lib/resources.ts | 229 -- web/src/routes/+page.svelte | 139 +- web/src/routes/files/+page.svelte | 1 - web/src/routes/resources/+page.svelte | 237 -- web/src/routes/resources/[id]/+page.svelte | 1111 ----- web/src/routes/tasks/[id]/+page.svelte | 1 - 158 files changed, 835 insertions(+), 81432 deletions(-) delete mode 100644 .agents/skills/homebased/references/resource-loans.md delete mode 100644 docs/resource-loans.md delete mode 100644 src/cli/release_watcher.rs delete mode 100644 src/cli/resource.rs delete mode 100644 src/daemon/actors/resource.rs delete mode 100644 src/daemon/actors/resource/test_support.rs delete mode 100644 src/daemon/actors/resource/tests.rs delete mode 100644 src/daemon/actors/supervisor/resource_launch.rs delete mode 100644 src/daemon/release_watcher_api.rs delete mode 100644 src/daemon/resource_action.rs delete mode 100644 src/daemon/resource_api.rs delete mode 100644 src/daemon/resource_api/tests.rs delete mode 100644 src/daemon/resource_background.rs delete mode 100644 src/daemon/resource_notice_delivery.rs delete mode 100644 src/daemon/resource_notice_sender.rs delete mode 100644 src/daemon/resource_submit.rs delete mode 100644 src/digest.rs delete mode 100644 src/resource.rs delete mode 100644 src/resource/api.rs delete mode 100644 src/resource/background_launch.rs delete mode 100644 src/resource/bound_action.rs delete mode 100644 src/resource/command_shape.rs delete mode 100644 src/resource/foreground.rs delete mode 100644 src/resource/id.rs delete mode 100644 src/resource/initial_idle.rs delete mode 100644 src/resource/operator_release.rs delete mode 100644 src/resource/ownership_lock.rs delete mode 100644 src/resource/ownership_lock/attempt_evidence.rs delete mode 100644 src/resource/ownership_lock/lock_probe.rs delete mode 100644 src/resource/release_checkpoint.rs delete mode 100644 src/resource/release_watcher.rs delete mode 100644 src/resource/return_window.rs delete mode 100644 src/resource/store.rs delete mode 100644 src/resource/store/acceptance.rs delete mode 100644 src/resource/store/assigned_task.rs delete mode 100644 src/resource/store/cancellation.rs delete mode 100644 src/resource/store/checkpoint.rs delete mode 100644 src/resource/store/codec.rs delete mode 100644 src/resource/store/error.rs delete mode 100644 src/resource/store/notice.rs delete mode 100644 src/resource/store/provenance.rs delete mode 100644 src/resource/store/queue.rs delete mode 100644 src/resource/store/release_completion.rs delete mode 100644 src/resource/store/release_loan.rs delete mode 100644 src/resource/store/release_watcher.rs delete mode 100644 src/resource/store/resources.rs delete mode 100644 src/resource/store/return_window.rs delete mode 100644 src/resource/store/revision.rs delete mode 100644 src/resource/store/rows.rs delete mode 100644 src/resource/store/test_support.rs delete mode 100644 src/resource/store/tests.rs delete mode 100644 src/resource/trainer_publication.rs rename src/{resource/store/schema.rs => store/fixtures/loan_schema_v34.sql} (99%) delete mode 100644 src/store/resource.rs delete mode 100644 src/store/resource/action_task.rs delete mode 100644 src/store/resource/assigned_task.rs delete mode 100644 src/store/resource/background.rs delete mode 100644 src/store/resource/cancellation.rs delete mode 100644 src/store/resource/controls.rs delete mode 100644 src/store/resource/controls/test_support.rs delete mode 100644 src/store/resource/initial_idle.rs delete mode 100644 src/store/resource/operator_release.rs delete mode 100644 src/store/resource/release_checkpoint.rs delete mode 100644 src/store/resource/release_completion.rs delete mode 100644 src/store/resource/release_proof.rs delete mode 100644 src/store/resource/release_watcher.rs delete mode 100644 src/store/resource/restore.rs delete mode 100644 src/store/resource/test_support.rs delete mode 100644 src/store/resource/tests.rs delete mode 100644 src/store/resource/tests/assigned_task.rs delete mode 100644 src/store/resource/tests/background.rs delete mode 100644 src/store/resource/tests/cancellation.rs delete mode 100644 src/store/resource/tests/completed_boundary.rs delete mode 100644 src/store/resource/tests/container.rs delete mode 100644 src/store/resource/tests/controls.rs delete mode 100644 src/store/resource/tests/docker.rs delete mode 100644 src/store/resource/tests/ended_trainer.rs delete mode 100644 src/store/resource/tests/fixtures.rs delete mode 100644 src/store/resource/tests/foreground.rs delete mode 100644 src/store/resource/tests/initial_idle.rs delete mode 100644 src/store/resource/tests/operator_release.rs delete mode 100644 src/store/resource/tests/queue_authority.rs delete mode 100644 src/store/resource/tests/registration.rs delete mode 100644 src/store/resource/tests/release_checkpoint.rs delete mode 100644 src/store/resource/tests/release_completion.rs delete mode 100644 src/store/resource/tests/release_transaction.rs delete mode 100644 src/store/resource/tests/release_watcher.rs delete mode 100644 src/store/resource/tests/release_watcher_acceptance.rs delete mode 100644 src/store/resource/tests/release_watcher_binding.rs delete mode 100644 src/store/resource/tests/restore.rs delete mode 100644 src/store/resource/tests/return_deadline.rs delete mode 100644 src/store/resource/tests/supervised_launch.rs delete mode 100644 src/store/resource/tests/trainer_association.rs delete mode 100644 src/store/resource/trainer_association.rs delete mode 100644 src/store/resource/trainer_lock.rs delete mode 100644 web/src/lib/components/ExpandToggle.svelte delete mode 100644 web/src/lib/components/ResourceQueuePanel.svelte delete mode 100644 web/src/lib/resource-actions.ts delete mode 100644 web/src/lib/resource-operations.svelte.ts delete mode 100644 web/src/lib/resource-operations.test.ts delete mode 100644 web/src/lib/resource-operations.ts delete mode 100644 web/src/lib/resource-state.test.ts delete mode 100644 web/src/lib/resource-state.ts delete mode 100644 web/src/lib/resources.ts delete mode 100644 web/src/routes/resources/+page.svelte delete mode 100644 web/src/routes/resources/[id]/+page.svelte diff --git a/.agents/skills/homebased/SKILL.md b/.agents/skills/homebased/SKILL.md index 45f71dc..7e8b6ab 100644 --- a/.agents/skills/homebased/SKILL.md +++ b/.agents/skills/homebased/SKILL.md @@ -16,7 +16,6 @@ description: Run long, unattended agent CLIs and general task commands through t - Always set `name` to a short goal label. Do not name the task after the agent or the CLI. - Always pass `--json` on data commands and parse the result. Every JSON object carries `api_version: 1`. - Event delivery is at-least-once. For new events, deduplicate by `(task, seq)`; for a legacy event without `seq`, use `(task, event)`. -- For GPU resource work, read [resource-loans.md](references/resource-loans.md). Check the exact pending actions at start, after compaction, and before an independent background launch. A delivered notice is not completion, and a notice whose action is absent from a complete pending result is stale. An unavailable authority is not an empty action list. - Use `homebased message send` for a direct message to a Claude Code session or Codex thread. Address a session by its id or by a task whose origin it is, never by its display name, which changes each time the session restarts. Read [messages.md](references/messages.md) for destination, source, and retry rules. - `message send --task` targets the task's origin thread, not its worker. Use `message send --worker ` to give a running Claude worker new instructions; it reads them at its next turn boundary. Use `homebased task followup` to resume a terminal Codex worker with new information. Run follow-up on the task's origin or execution machine. Only one follow-up can resume a thread at a time; wait for the active task's event after `resume_thread_busy`. - Do not poll a running task in a loop. Submit, tell the user the task id, end the turn, and wait for events. Inspect on demand only. @@ -42,7 +41,6 @@ Pick the first row that matches, then read only that file. | Configuring Fleet or discovering machines | [fleet.md](references/fleet.md) | | Sending a direct message to a session or thread | [messages.md](references/messages.md) | | A command exited non-zero, `daemon_unavailable`, or the socket is down | [errors.md](references/errors.md) | -| Supervising shared GPU work or a resource loan | [resource-loans.md](references/resource-loans.md) | | `homebased` is missing, the daemon is not installed, or the binary was rebuilt | [setup.md](references/setup.md) | ## Minimal flow diff --git a/.agents/skills/homebased/references/errors.md b/.agents/skills/homebased/references/errors.md index 628d82c..acbb1fe 100644 --- a/.agents/skills/homebased/references/errors.md +++ b/.agents/skills/homebased/references/errors.md @@ -26,10 +26,10 @@ Errors go to stderr. With `--json` they are one object: | `daemon_unavailable` | 1, retryable | Socket missing or refused. | `homebased --json daemon status`. If `socket` is `down`, read setup.md and start or restart the daemon. Tasks already running keep running and still report. | | `config_invalid` | 2 | An explicitly selected config file is missing, unreadable, or invalid TOML. | Fix the reported path or setting. Run `homebased --json config validate`. | | `invalid_spec` | 2 | Bad JSON, unknown field, wrong `api_version`, bad UUID, both or neither of `prompt`/`prompt_file`, timeout below 30m, empty command, cross-variant fields, unreadable `prompt_file`, blank prompt text, or a container mount `source` that is missing or exposes a container daemon socket on the machine that runs the task (`input.pointer` is `/workload/mounts//source`, or `/workload/mounts` when a remote machine refused it). | Fix the field at `input.pointer`. `homebased task schema` prints the schema. | -| `unknown_thread` | 2 | The spec `thread` names no Claude Code session (registry file or transcript) and no Codex thread (session file) on the submitting machine. Checked by `task submit` (also with `--dry-run`), `resource request submit`, and `resource background submit` before any work starts. | Use the exact id from submit.md step 1. `input.suggestions` lists known ids that differ by a likely typo, best first; the message names the best one. A Homebased task id is not a thread. A worker on a remote executor cannot send events to its parent's thread. | +| `unknown_thread` | 2 | The spec `thread` names no Claude Code session (registry file or transcript) and no Codex thread (session file) on the submitting machine. Checked by `task submit`, also with `--dry-run`, before any work starts. | Use the exact id from submit.md step 1. `input.suggestions` lists known ids that differ by a likely typo, best first; the message names the best one. A Homebased task id is not a thread. A worker on a remote executor cannot send events to its parent's thread. | | `thread_mismatch` | 2 | `HOMEBASED_TASK_ID` is set and the spec `thread` differs from the parent task's thread. `input.parent_task` and `input.parent_thread` name the parent. | Use `input.parent_thread`. Pass `--allow-other-thread` on submit or followup only when the events must go to another session; that thread must still exist. | | `agent_configuration` | 1 | OpenCode inherited inline configuration is malformed, has the wrong shape, or already defines the generated task agent. | Fix `OPENCODE_CONFIG_CONTENT` without putting credentials in the task spec or logs, then submit again. | -| `invalid_cwd` | 2 | `cwd` is not an existing, accessible host directory on the machine that runs the task. `input.problem` is `not_found`, `not_directory`, or `inaccessible`; `input.value` is the `cwd`. Checked at submit, including `--dry-run`, `resource request submit`, and `resource background submit`; a remote executor or resource authority checks its own file system and refuses at acceptance. | `cwd` is a host path, never a path inside a container. When it names a container mount target, `input.suggested_cwd` is the matching host path under that mount's `source`; use it, and set `workload.workdir` for the directory inside the container. | +| `invalid_cwd` | 2 | `cwd` is not an existing, accessible host directory on the machine that runs the task. `input.problem` is `not_found`, `not_directory`, or `inaccessible`; `input.value` is the `cwd`. Checked at submit, including `--dry-run`; a remote executor checks its own file system and refuses at acceptance. | `cwd` is a host path, never a path inside a container. When it names a container mount target, `input.suggested_cwd` is the matching host path under that mount's `source`; use it, and set `workload.workdir` for the directory inside the container. | | `executable_missing` | 3 | Requested program missing, not a file, or not executable. Agents also check `HOMEBASED_` overrides. | Submit from a shell where the program is on `PATH`, use an absolute path, or export `HOMEBASED_CODEX`, `HOMEBASED_CLAUDE`, or `HOMEBASED_GROK`. | | `unknown_dependency` | 3 | An `after` entry names no task submitted through this daemon. `input.task` names it. Checked by `task submit`, also with `--dry-run`. | Use the id of a task submitted from this machine. A task submitted through another machine cannot be a dependency here. | | `dependency_failed` | 5 | An `after` entry already ended without success, so the task could never start. `input.task` and `input.outcome` name it. `input.outcome` is `failed`, `blocked`, `cancelled`, `lost`, or `unknown`; `unknown` means the task ended but no record says how, as for a task that finished before an update and whose terminal event was pruned. | Handle that task's result first, then submit without it or after its replacement. | diff --git a/.agents/skills/homebased/references/inspect.md b/.agents/skills/homebased/references/inspect.md index effc977..90dbafc 100644 --- a/.agents/skills/homebased/references/inspect.md +++ b/.agents/skills/homebased/references/inspect.md @@ -68,7 +68,7 @@ The local daemon reads `output.log` on the execution machine. For a remote task, ## Dashboard -The daemon serves a read-only HTTP dashboard only when `--web-listen` / `HOMEBASED_WEB_LISTEN` is a host:port. Open that URL in a browser to see the tasks of every Fleet machine, their status, and their log tail without an agent turn. When a GPU runs or queues work, the task list shows its current task and queue at the side. +The daemon serves a read-only HTTP dashboard only when `--web-listen` / `HOMEBASED_WEB_LISTEN` is a host:port. Open that URL in a browser to see the tasks of every Fleet machine, their status, and their log tail without an agent turn. ```bash homebased --json daemon status # "web" holds the URL, or null when the dashboard is off or the socket is down diff --git a/.agents/skills/homebased/references/resource-loans.md b/.agents/skills/homebased/references/resource-loans.md deleted file mode 100644 index de130a8..0000000 --- a/.agents/skills/homebased/references/resource-loans.md +++ /dev/null @@ -1,137 +0,0 @@ -# Supervise resource loans - -Use `homebased resource --help` for the installed commands and -`homebased --json resource schema` for the accepted JSON shapes. - -## Required checks - -Use the resource CLI. Do not use ordinary `task submit` for GPU work that -belongs to a registered resource. Do not bypass an active loan or the resource -request queue. Queued requests serve by queue rank, then acceptance identity; -new requests join the back. Do not monitor a loan by polling with a model or in -a command loop. - -Queued GPU commands must run native foreground executables. Scripts, -interpreters, shell wrappers, container clients, and detach tools are refused. -For work in a pinned Docker image, submit a `container` workload with `gpus` -instead of a `docker` command. Homebased starts, watches, and removes that -container, and releases the GPU only after it confirms the removal. - -Check the exact actions for the assigned supervisor machine and thread: - -- when resource work starts -- after compaction or resume -- before launching any independent background task -- after every `HOMEBASED_RESOURCE_NOTICE` line -- after an unknown action result, when needed to resolve state - -Use `homebased --json resource pending --machine --thread -`. Read `unavailable_authorities` before interpreting `actions`. -If any authority is unavailable, the result is incomplete. Do not treat it as -no pending action and do not start independent GPU work. If all authorities -answered and `actions` is empty, there is no pending supervisor action for that -exact address; still check `resource show` for an active loan and -`resource requests` for queued work in serving order. - -Every request `cwd` is a host directory on the resource authority, even for a -`container` workload; a mount target such as `/scratch` exists only inside the -container. `resource request submit` and `resource background submit` refuse a -missing `cwd` with `invalid_cwd` and a missing mount source with -`invalid_spec`, before the request enters the queue. - -When the request's turn comes and the authority still cannot launch it, for -example because its `cwd`, a mount source, or its executable disappeared, the -task ends `failed` before launch with a `launch failed before start` reason, the -thread gets its `TASK_FAILED` event, and the queue serves the next request. A -launch whose outcome stays unknown, such as after a daemon restart mid-launch, -is failed the same way with a `launch_unconfirmed` reason once it has had no -confirmed start for 2 minutes. `resource show` reports `queue_blocked` only -until then. - -Cancel a still-queued request with `resource request cancel`. An assigned -request belongs to its task: cancel it with `homebased task cancel `, -which works whether or not the task started and serves the next request; the -`resource request cancel` error names that command. Reorder one with -`resource request move` and exactly one of `--front`, `--back`, -`--before `, or `--after `. Both need -`--expected-revision` from `resource show` and a stable `--operation-id`. A -successful move advances the resource revision even when the order does not -change. Reuse the same operation UUID and input after an unknown result. - -A notice only prompts a check, and it can arrive after its action is resolved. -The authority stops notices when their action resolves, but a notice already in -transit still arrives. Act only on an action whose `action_id` appears in a -fresh authority-backed pending result. If all authorities answered and the -notice's `action_id` is absent, the notice is stale. Ignore it, and do not -reopen or retry the action. A delivered notice does not complete an action. -Read the action phase from the same pending result. Keep the saved pending JSON unchanged for -an unknown-outcome retry. Keep the action, request, task, and operation IDs -unchanged as the runbook requires. If the authority returns a definite -rejection, do not retry with changed content under the same ID. - -## Do not bypass release or return ownership - -The resource authority owns release proof. A checkpoint by itself does not -prove that the task stopped or released its ownership lock. There is no public -`resource released` command. `ResourceActor` reconciles authority-built proof. - -A failed trainer, or a trainer cancelled without the action's saved stop, gives -the `ended_without_result` return context. The authority gives it only after -it proves that the exact saved lock is free. Do not resume that run. Choose -`after_ended_run` or `--no-resume`. A trainer with no trainer-attempt -association stays reserved with `TrainerAssociationMissing`. Do not work around -it with another launch. After the trainer ends or is lost, an operator may use -`resource operator-release --spec ` on the GPU authority machine -only after inspecting that machine and confirming that no trainer GPU work -remains. This is a human attestation, not automatic exit or lock proof. Use the -exact task, resource revision, and `state_binding` from `resource show`: - -- `no_loan` or `awaiting_release` for the registered trainer. -- `first_background_launch` with `background_launch.request_id` and - `background_launch.task_id` when `background_launch.status` is - `release_unproven`. This first launch ended before registration. The - dashboard shows `Operator release required`, even with no queued request. -- When `loan` is `restoring`, use the exact `return_execution_mode` value from - `resource show`: `direct_segment_trainer` selects `restoring_return`, and - `native_foreground` or `container` selects `restoring_foreground_return`. Use the Restoring - loan and action UUIDs and `resume_task_id` for the binding. If the field is - absent or unknown, do not guess from the decision, task status, command name, - or command text, and do not submit either binding. -- A native foreground return task keeps the Restoring loan until a successful - end with a confirmed process-group exit. A container return task keeps it - until the container exits with code 0 and Homebased confirms its removal. An - operator attestation is not exit proof, and the task keeps its saved state. - -See the [operator procedure](../../../../docs/resource-loans.md#resolve-an-unproven-ended-trainer). -Keep the document and operation ID unchanged after an unknown outcome, and -retry on the same authority. The command refuses a queued or running task. - -A new resource with no background run, loan, or release record has no idle -evidence, so its queued requests wait. Do not work around it with a background -launch or an ordinary task. After an operator inspects the authority GPU, the -operator can run `resource initial-idle --spec ` once on the -authority machine. See the [runbook step](../../../../docs/resource-loans.md#mark-a-new-resource-idle). - -A `return_required` action gives the supervisor 2 minutes, counted from when -it opened. Read `return_window.deadline_at` from pending. After the deadline, a -queued request takes the resource from the same loan, and the action is no -longer pending. The loan keeps its return context, and the next drained queue -opens a new action. Decide quickly. If the choice needs more time, run -`resource return --pending-spec --hold ` -first, for example `--hold 5m`. The hold stops at 10 minutes after the action -opened. It is not a choice, so decide before the new deadline. - -Use `resource release-watch` only for a remote supervisor action. A co-located -authority starts its watcher. Do not start another watcher. `resource return` -and `resource resolve` work for remote and co-located supervisors. Keep the -same action, request, and task IDs when an outcome is unknown. - -A native foreground return task keeps the `restoring` reservation until it -exits with code 0 and a confirmed process-group exit. A container return task -keeps it until the container exits with code 0 and Homebased confirms that the -container is removed. Neither becomes the registered training task, so do not -expect a release action for it. Use `resource resolve` for a failed end with -confirmed evidence. - -Do not use the current live training job to test, verify, or demonstrate this -workflow. Use a controlled registered task. diff --git a/.agents/skills/homebased/references/submit.md b/.agents/skills/homebased/references/submit.md index 40d0751..f7fad73 100644 --- a/.agents/skills/homebased/references/submit.md +++ b/.agents/skills/homebased/references/submit.md @@ -32,7 +32,7 @@ Copy the id; do not retype it. Submit checks the id against the Claude Code sess A `task` runs an argv array with no shell. A caller that needs shell syntax must request it explicitly, for example `["sh", "-lc", "..."]`. -A `container` runs a pinned image. Do not submit a `docker run` command as a `task`: its client can exit while the container keeps running, and resource work refuses it. For a `container`, Homebased creates the container, saves its ID, starts it, streams its logs to `output.log`, waits for it, and then removes it. The task exit code is the container exit code. +A `container` runs a pinned image. Do not submit a `docker run` command as a `task`: its client can exit while the container keeps running. For a `container`, Homebased creates the container, saves its ID, starts it, streams its logs to `output.log`, waits for it, and then removes it. The task exit code is the container exit code. ## 3. Write the prompt file (agent only) @@ -159,7 +159,7 @@ Container-only fields under `workload`: | `image` | yes | Pinned by digest: `sha256:<64 hex>` image ID or `name@sha256:<64 hex>`. A tag is refused. The image must already be on the execution machine; Homebased does not pull it. | | `entrypoint` | no | Argv array that replaces the image entrypoint. | | `args` | no | Argv array after the image. Passed unchanged; not a shell string. | -| `gpus` | no | `"all"` or an array of device indices. Required for resource work. | +| `gpus` | no | `"all"` or an array of device indices. | | `memory` | yes | Byte count, or a size such as `512m` or `24g` (binary units). Also the swap limit. At least 6 MiB. | | `user` | no | Numeric `uid:gid`. Defaults to the daemon's user. | | `workdir` | no | Absolute working directory in the container. Put a container path here, not in `cwd`. | diff --git a/README.md b/README.md index 80c24f8..acca8a3 100644 --- a/README.md +++ b/README.md @@ -428,14 +428,6 @@ one or more known peers could not be checked. Retry the lookup; the daemon only returns `task_not_found` after every machine in the current known Fleet gives a definitive negative result. -### GPU resource loans - -Use a registered resource for shared GPU work. The resource authority owns the -queue, active loan, and release proof. Read the -[GPU resource loan runbook](docs/resource-loans.md) for the real CLI commands, -JSON inputs, retry rules, and supervisor steps. Do not use the current live -training job to test this workflow. - ### Events Delivery is at-least-once. For new events, use `(task, seq)` to detect a diff --git a/docs/resource-loans.md b/docs/resource-loans.md deleted file mode 100644 index 7a51edd..0000000 --- a/docs/resource-loans.md +++ /dev/null @@ -1,672 +0,0 @@ -# GPU resource loans - -Use this runbook for an exclusive GPU registered with Homebased. The resource -authority is the machine that owns the GPU. It owns resource state, serving order, -loan state, and release proof. Version 1 runs resource work on that authority. -The assigned supervisor is one exact machine and thread. The submitting machine -owns the task callback route. - -Do not use the current live training job as an example, test, or verification -target. Use a controlled run that is registered for this resource. - -Example UUIDs, paths, and command arguments are placeholders. Replace them -before use. Do not run a code block unchanged. - -Use the CLI, not private daemon routes or direct database edits. -`homebased resource schema` prints the registration, task, return-work, -trainer-attempt, and operator-attestation JSON schemas. - -## Identities and retries - -- Choose a resource UUID before registration. Keep it for every retry. -- Choose a request UUID before the first background or queued submit. Keep the - same UUID and identical input after an unknown result. The submitting daemon - saves the Homebased task UUID in its origin route before it sends the request - to the authority. -- The authority creates loan and action UUIDs. Use the exact action UUID shown - by `resource pending`. -- The authority also saves the release watcher's request and task UUIDs before - launch. The supervisor uses the action UUID; it does not choose new watcher - identities. -- A return launch needs distinct, preallocated request and task UUIDs. Keep both - with the action UUID and identical return choice on retry. -- Cancellation and renotify each need an operation UUID. Reuse it after an - unknown result. -- Operator attestation also needs its own operation UUID. Keep the complete - saved document unchanged on retry, including the observation. -- Revision flags are compare-and-set checks. Read the current - `resource.state_revision` from `resource show`; do not guess or increment it. - -An unknown outcome means the authority may have committed the operation. It is -not a rejection. Retry with the same saved input and IDs. A definitive rejection -means the authority refused that operation. Do not change its content while -reusing the same ID. - -## Register the resource and supervisor - -Run registration on the GPU authority machine. The required input is: - -```json -{ - "id": "11111111-1111-4111-8111-111111111111", - "display_name": "shared GPU", - "supervisor": { - "machine": "22222222-2222-4222-8222-222222222222", - "thread": "33333333-3333-4333-8333-333333333333" - } -} -``` - -Save it as `resource.json`, then run: - -```sh -homebased --json resource register --spec resource.json -homebased --json resource show 11111111-1111-4111-8111-111111111111 -``` - -Registration sets the initial supervisor. To replace it, use the revision from -the latest `resource show` output: - -```sh -homebased --json resource supervisor set 11111111-1111-4111-8111-111111111111 \ - --machine 44444444-4444-4444-8444-444444444444 \ - --thread 55555555-5555-4555-8555-555555555555 \ - --expected-revision -``` - -Supervisor assignment names a machine UUID and exact thread UUID. Do not select -a supervisor from recent activity or a working directory. After an unknown -result, retry with the same supervisor and expected revision. If the authority -rejects a stale revision, read `resource show` before making a new assignment. - -## Submit background training - -`resource background submit` binds and starts the one task intended to hold the -GPU between loans. The resource owner registers it only after the task layer -confirms that it is running. Its `--spec` input is a task submit spec with no -`machine` field. -The required fields are `api_version`, `thread`, `name`, `cwd`, and -`workload`. `timeout` is optional and defaults to `1h`; it is an inactivity -timer, not a run deadline. `workload` must be a finite command: - -```json -{ - "api_version": 1, - "thread": "33333333-3333-4333-8333-333333333333", - "name": "training background", - "cwd": "/path/to/trainer", - "timeout": "1h", - "workload": { - "type": "task", - "command": [ - "python", "-m", "ops.run_segment", "run", - "--task", "/path/to/task.json", - "--input-root", "/path/to/inputs", - "--runtime-root", "/path/to/runtime", - "--image-digest", "sha256:" - ] - } -} -``` - -This shows the required direct-segment trainer shape. Replace every path and -value with those for a prepared, controlled run. The CLI rejects a machine -override. The current background implementation supports this maintained -trainer because the authority can verify its ownership lock after exit. - -Run the command on the assigned supervisor machine and exact supervisor thread: - -```sh -homebased --json resource background submit \ - 11111111-1111-4111-8111-111111111111 \ - --request-id 66666666-6666-4666-8666-666666666666 \ - --spec trainer.json -``` - -This works both when the supervisor and authority are the same machine and -when they are different machines. For a remote supervisor, its machine saves -the callback route before sending the launch to the fixed authority. Do not run -the command on the authority machine when the supervisor is remote. The -authority accepts only the exact supervisor thread, no active loan, no queued -request ahead, and the supported trainer command. An exact retry reuses the -saved launch; changed content with the same request UUID conflicts. - -## Bind the running trainer attempt - -Wait until `resource show` reports the registered background task as `running` -and the trainer holds its ownership lock. Then save the six fields from that -trainer attempt's request as `attempt.json`: - -```json -{ - "campaign_id": "campaign-01", - "campaign_revision_id": "revision-01", - "task_id": "trainer-task-01", - "attempt_id": "attempt-01", - "attempt_number": 1, - "ownership_token": "token-01" -} -``` - -The `task_id` in this JSON is the trainer's string identity. It is not the -Homebased task UUID. Bind it to that exact Homebased task: - -```sh -homebased --json resource background bind-attempt \ - 11111111-1111-4111-8111-111111111111 \ - --task-id \ - --attempt-spec attempt.json -``` - -Run this on the assigned supervisor machine and thread. The authority reads -the accepted task, reads the trainer attempt, and checks the held lock itself. -Do not invent the binding or call it before the task is confirmed running. An -exact saved binding can be retried. A different binding for that task conflicts. - -## Mark a new resource idle - -A queued request serves an unregistered resource only from saved evidence that -the GPU is free: a closed loan, a first background launch that never started a -process, or an operator attestation about an ended task. A new resource that -never had a background run has none of these. `resource show` then reports the -attention code for no idle evidence, and queued requests wait. - -For such a resource, an operator can record one initial idle attestation. Use -it only when all of these are true: - -- `resource show` shows no `registered_background_task`, no `loan`, and no - `background_launch`. -- The operator inspected the GPU on the authority machine and found no work on - it, for example with `nvidia-smi`. - -Save the document with the exact resource, authority, and `state_revision` from -`resource show`, and a new operation UUID: - -```json -{ - "operation_id": "eeeeeeee-eeee-4eee-8eee-eeeeeeeeeeee", - "resource_id": "11111111-1111-4111-8111-111111111111", - "authority_machine": "22222222-2222-4222-8222-222222222222", - "expected_state_revision": 0, - "observation": "Describe the authority GPU checks and why no work holds the GPU", - "confirmation": "operator_confirmed_gpu_free" -} -``` - -Run it on the GPU authority machine: - -```sh -homebased --json resource initial-idle --spec initial-idle.json -homebased --json resource show 11111111-1111-4111-8111-111111111111 -``` - -The authority refuses the attestation if the resource has a registered task, a -loan, a first background launch, or an operator attestation, because that -history decides the idle state. It accepts one initial attestation per -resource. The receipt is a human confirmation, not proof that a process -exited. It counts as the idle boundary only until the first loan or background -launch; after that, the normal history applies. After an unknown result, retry -with the exact same file. The same operation UUID with changed content is a -conflict. - -## Queue and cancel work - -Background and queued command inputs use the same strict `ResourceTaskSubmitSpec` -shape shown below. The command is an argv array; it is not a shell string. The -spec has no `machine` field. Use a callback `thread` and a `cwd` that is an -existing host directory on the authority machine. For a `container` workload, -`cwd` is still a host path; a mount target exists only inside the container. -The authority refuses a missing `cwd` (`invalid_cwd`) or mount source -(`invalid_spec`) at submit, before the request enters the queue. The queued command must run an inspectable native ELF or -Mach-O executable in the task's foreground process group. Scripts, shells, -interpreters, container clients, remote launchers, and detach tools are not -accepted. The authority cannot use process-group exit as release proof for -work that leaves that group. Replace the example command with a prepared native -executable before submitting. - -```json -{ - "api_version": 1, - "thread": "77777777-7777-4777-8777-777777777777", - "name": "bounded GPU job", - "cwd": "/path/to/job", - "timeout": "1h", - "workload": { - "type": "task", - "command": ["/path/to/prepared-command", "--input", "/path/to/input"] - } -} -``` - -For work in a pinned Docker image, such as checkpoint evaluation, use a -`container` workload instead of a `docker` command. Its `gpus` field is -required for resource work: - -```json -{ - "api_version": 1, - "thread": "77777777-7777-4777-8777-777777777777", - "name": "evaluate checkpoint", - "cwd": "/path/to/job", - "timeout": "1h", - "workload": { - "type": "container", - "image": "registry.example/eval@sha256:<64-hex-digest>", - "entrypoint": ["/usr/bin/python3", "-m", "eval"], - "args": ["--checkpoint", "/data/checkpoint"], - "gpus": "all", - "memory": "24g", - "mounts": [ - { "source": "/path/to/checkpoint", "target": "/data/checkpoint", "read_only": true }, - { "source": "/path/to/output", "target": "/out" } - ] - } -} -``` - -The image must already be on the authority; Homebased does not pull it. Each -mount source must exist on the authority. Docker and containerd sockets, and -directories that contain them, are refused. The container runs under -`dockerd`, outside the task's process group, so the witness is the container -itself. The authority releases the GPU only after Homebased saved the container -ID before the start, read the exited container's exit code, removed the -container, and saw that the same ID no longer exists. If the worker that -watches the container stops, the container keeps running, the task stays -`running`, and the daemon starts a worker that adopts the container. If Docker -is unavailable or the exit code cannot be read, the evidence stays -unconfirmed, the GPU stays reserved, and the supervisor gets an attention -notice. `resource background submit` does not accept a container. - -Submit with a new, stable request UUID: - -```sh -homebased --json resource request submit \ - 11111111-1111-4111-8111-111111111111 \ - --request-id 88888888-8888-4888-8888-888888888888 \ - --spec request.json -``` - -The authority assigns each request an immutable acceptance identity. Queued -requests serve by queue rank, then acceptance identity. New requests join the -back. A request can wait or activate under a loan. Do not start a second task -with ordinary `task submit` to avoid this queue. Use -`resource requests ` to read requests in serving order. - -Cancel only a request that is still queued, before activation. Get the latest -state revision with `resource show`; use a new operation UUID once, then keep it -for retries: - -```sh -homebased --json resource request cancel \ - 11111111-1111-4111-8111-111111111111 \ - 88888888-8888-4888-8888-888888888888 \ - --expected-revision \ - --operation-id 99999999-9999-4999-8999-999999999999 -``` - -Canceling the last request does not remove an active loan or its return -obligation. - -An assigned request belongs to its task, so `resource request cancel` refuses -it and names the command to use instead: `homebased task cancel ` -cancels the task whether or not it started, and the queue serves the next -request. - -A request that reaches its turn but cannot launch, for example because its -`cwd`, a mount source, or its executable disappeared after it was queued, gets -a task that ends `failed` before launch. Its `SpawnFailed` reason starts with -`launch failed before start`, the task's thread receives `TASK_FAILED`, and the -queue serves the next request. When the launch outcome is unknown, such as after -a daemon restart mid-launch, the resource actor waits at most 2 minutes -(`LAUNCH_CONFIRMATION_BOUND`) for a confirmed start. A task that is still -queued, or never got a task record, is then failed before launch with a reason -that starts with `launch_unconfirmed`, and the queue moves on. A worker must -move its task from queued to running before it spawns anything, so this cannot -hide work that already started. - -Move only a request that is still queued. Read the latest state revision with -`resource show`, then choose exactly one placement: `--front`, `--back`, -`--before `, or `--after `. Use one new operation -UUID and keep the same input and ID after an unknown result: - -```sh -homebased --json resource request move \ - 11111111-1111-4111-8111-111111111111 \ - 88888888-8888-4888-8888-888888888888 \ - --before 77777777-7777-4777-8777-777777777777 \ - --expected-revision \ - --operation-id aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa -``` - -A successful move advances the resource state revision, even when the request -stays in the same place. The public socket action uses the same revision and -operation fields as other resource actions. Its request body for the example -above is: - -```json -{ - "api_version": 1, - "operation_id": "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa", - "expected_revision": 3, - "action": { - "type": "move_queued", - "request_id": "88888888-8888-4888-8888-888888888888", - "placement": { - "type": "before", - "request_id": "77777777-7777-4777-8777-777777777777" - } - } -} -``` - -Send this body to `POST /v1/resources//actions` on the authority. - -## Check and act on supervisor actions - -Run pending from the exact assigned supervisor machine and thread. Save the -whole JSON result unchanged before an action: - -```sh -homebased --json resource pending \ - --machine \ - --thread > pending.json -``` - -Check `unavailable_authorities` first. If it is not empty, the result is -incomplete. An unavailable authority is not an empty action list. If it is -empty, use the exact `action_id`, `loan_id`, `state_revision`, phase, and return -context from this saved result. A stale action needs a fresh pending query. - -The notice is separate from the action. A `delivered` notice means only that the -notice arrived. It does not complete the release or return decision. Check -`resource show` and a fresh `resource pending` result for the authority's current -phase. Do not infer completion from a delivered notice or a finished watcher. - -The authority sends and retries a notice only while the loan waits for that -exact decision. A notice that was already in transit when the action resolved -still arrives. If all authorities answered and the notice's `action_id` is not -in the pending result, the notice is stale. Ignore it. - -### Release a background trainer - -For a remote supervisor, start the watcher for the exact `release_required` -action: - -```sh -homebased --json resource release-watch \ - --pending-spec pending.json -``` - -For a co-located supervisor, the authority's `ResourceActor` starts the bound -watcher. Do not start a second watcher. For either case, the authority builds -release proof and `ResourceActor` reconciles it. There is no public -`resource released` command and no caller-supplied release receipt. - -A newly published checkpoint is not proof that the GPU is free. The authority -must verify the exact registered task, attempt, checkpoint or final result, -confirmed process exit, and ownership-lock release. If the trainer completed -first, the return context is `already_completed`; do not resume it as the same -run. - -If the trainer failed, was cancelled without this action's saved checkpoint -stop, or exited 0 with no result publication, the return context is -`ended_without_result`. It names the task and its outcome. The authority -releases the GPU for this context only when all of these are true: - -- The exact registered task has a trainer-attempt association. -- The accepted spec and direct-segment command still match that association. -- The task is terminal, and its process-group exit is confirmed. -- The authority holds the exact saved `.segment.lock` until the transition - commits. - -If a queue exists, the next request in serving order runs next. The ended run cannot resume. -A checkpoint on disk does not make it resumable. A lost task, an unconfirmed -exit, or a held lock keeps the GPU reserved. A trainer with no trainer-attempt -association also keeps the GPU reserved. Its attention reason is -`TrainerAssociationMissing`, and only an operator can resolve it. A registered -trainer that ends before a loan opens gets a release action when a request -arrives. The same checks then apply. - -### Resolve an unproven ended trainer - -Use `operator-release` only when a trainer has ended or is lost, automatic -release proof is unavailable, and an operator has inspected the GPU authority -machine and confirmed that no work from that trainer remains. For example, a -trainer with no saved attempt association has no saved lock that the authority -can use as release proof. A trainer that ends before its confirmed start -registers it can never get an association. This command records the -operator's decision. It does not prove process exit, lock release, or a -reusable checkpoint. It cannot release a queued or running task. - -Run this command on the GPU authority machine, not on a remote supervisor. Read -the current `resource show` result first. Use its exact resource authority and -`state_revision`, and choose the `state_binding` and `task_id` from this table: - -| `resource show` state | `state_binding` | `task_id` | -| --- | --- | --- | -| No `loan`; `registered_background_task` has ended or is lost | `{"type":"no_loan"}` | `registered_background_task` | -| `loan` is `awaiting_release` | `{"type":"awaiting_release","loan_id":"","action_id":""}` | `registered_background_task` | -| No `loan`; `background_launch.status` is `release_unproven` | `{"type":"first_background_launch","request_id":""}` | `background_launch.task_id` | -| `loan` is `restoring`; `return_execution_mode` is `direct_segment_trainer` | `{"type":"restoring_return","loan_id":"","action_id":""}` | `phase.resume_task_id` | -| `loan` is `restoring`; `return_execution_mode` is `native_foreground` or `container` | `{"type":"restoring_foreground_return","loan_id":"","action_id":""}` | `phase.resume_task_id` | - -Other loan phases cannot be resolved by this command. A native foreground -return task that succeeded with a confirmed process-group exit closes its loan -automatically. A container return task that exited with code 0 and whose -removal Homebased confirmed also closes its loan automatically. If either -failed with confirmed evidence, use `resource resolve`. Use -`restoring_foreground_return` only when that proof is missing, for example -when the task is lost, its process-group exit is not confirmed, or its -container evidence is not confirmed. The two -Restoring bindings are not interchangeable. Read `resource show` and use its -exact `return_execution_mode` value to choose the binding. If that field is -absent or unknown, stop; do not infer the mode from the return decision, task -status, command name, or command text, and do not submit either Restoring -binding. The authority refuses a binding that does not match the saved mode. -The `release_unproven` launch status has the attention code -`background_launch_release_unproven`. The dashboard shows it as -`Operator release required`, not `Available`, even when no request is queued. - -Save the complete document before sending it: - -```json -{ - "operation_id": "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa", - "resource_id": "11111111-1111-4111-8111-111111111111", - "authority_machine": "22222222-2222-4222-8222-222222222222", - "task_id": "bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb", - "expected_state_revision": 7, - "state_binding": { - "type": "awaiting_release", - "loan_id": "cccccccc-cccc-4ccc-8ccc-cccccccccccc", - "action_id": "dddddddd-dddd-4ddd-8ddd-dddddddddddd" - }, - "observation": "Describe the authority GPU checks and why no trainer work remains", - "confirmation": "operator_confirmed_gpu_free" -} -``` - -Replace every example value with the exact observed value, and use the -`state_binding` from the table. The observation must describe what the operator -checked; do not copy the example text. Then run: - -```sh -homebased --json resource operator-release --spec operator-release.json -homebased --json resource show 11111111-1111-4111-8111-111111111111 -``` - -The authority saves the attestation, the task evidence it found, and the queue -or loan transition in one transaction: - -- `awaiting_release`: with queued work, the next request in serving order serves. Otherwise - the loan moves to `awaiting_return`, and the supervisor decides the return. -- `no_loan` and `first_background_launch`: the registration clears. With - queued work, the next request in serving order serves. Otherwise the receipt is the idle - boundary that a later request or first background launch uses. -- `restoring_return` and `restoring_foreground_return`: the Restoring loan - closes with the attested end, and the registration clears. With queued work, - the next request in serving order serves. Otherwise the closed loan and its receipt are the - idle boundary. The task keeps its saved state. A lost task stays lost, and an - unconfirmed process-group exit stays unconfirmed. The receipt records the - operator's attestation, not a confirmed exit. - -Until the receipt commits, the existing loan or launch keeps the resource -reserved. A task that has ended is not enough; only the saved receipt releases -it, and it stays released after a daemon restart. - -After an unknown result, retry on the same authority with the exact same file -and operation UUID, even if the resource revision has changed. An exact retry -returns the saved receipt with `replayed: true`. The same operation UUID with -changed content is a conflict. A definite refusal writes nothing. It needs a -fresh state read and, if the operator still confirms release, a new attestation -with a new operation UUID. - -### Choose what happens after the queue drains - -When the phase is `return_required`, use the `return_context` from pending. - -The supervisor has 2 minutes to decide, counted from when the return action -opened. `return_window.deadline_at` in pending and `resource show` gives the -exact time. If a request is queued after the deadline and no choice exists, the -authority serves it from the same loan and the action closes. The loan keeps -its `return_context`. When the queue drains again, a new return action opens -with a new action UUID and a new 2 minute window. A choice for the closed -action is refused as not pending. - -To think longer, hold the window before it closes: - -```sh -homebased --json resource return \ - --pending-spec pending.json \ - --hold 5m -``` - -A hold moves the deadline to now plus the given time. It never moves the -deadline more than 10 minutes after the action opened, and it never makes the -window shorter. A hold is not a choice. It does not change the resource -revision, so the saved `pending.json` stays valid for the choice. - -The return-work JSON is one of these exact tagged shapes: - -```json -{"type":"same_run_resume","stopped_task":"","recovery_ref":""} -``` - -```json -{"type":"evaluation_or_next_epoch","completed_task":"","spec":{"api_version":1,"thread":"33333333-3333-4333-8333-333333333333","name":"next training work","cwd":"/path/to/trainer","workload":{"type":"task","command":["/path/to/prepared-native-command"]}}} -``` - -```json -{"type":"new_background_work","spec":{"api_version":1,"thread":"33333333-3333-4333-8333-333333333333","name":"new training work","cwd":"/path/to/trainer","workload":{"type":"task","command":["/path/to/prepared-native-command"]}}} -``` - -```json -{"type":"after_ended_run","ended_task":"","spec":{"api_version":1,"thread":"33333333-3333-4333-8333-333333333333","name":"new training work","cwd":"/path/to/trainer","workload":{"type":"task","command":["/path/to/prepared-native-command"]}}} -``` - -`same_run_resume` is for the matching `stopped` context. The authority derives -the command from saved run records. It will not infer a resume from a checkpoint -alone; the exact release proof, checkpoint publication, trainer association, -accepted task record, and immutable inputs must still match. Use -`evaluation_or_next_epoch` only with `already_completed`, -`new_background_work` only with `idle`, and `after_ended_run` only with -`ended_without_result`. `resource schema` prints all fields. -For these three choices, the work must be a native foreground executable, the -maintained direct-segment trainer command, or a `container` workload with -`gpus`. A Python evaluation script or a wrapper command is not accepted as a -command; run it in a pinned image as a `container` instead: - -```json -{"type":"evaluation_or_next_epoch","completed_task":"","spec":{"api_version":1,"thread":"33333333-3333-4333-8333-333333333333","name":"evaluate epoch","cwd":"/path/to/trainer","workload":{"type":"container","image":"registry.example/eval@sha256:<64-hex-digest>","args":["--checkpoint","/data/checkpoint"],"gpus":"all","memory":"24g","mounts":[{"source":"/path/to/checkpoint","target":"/data/checkpoint","read_only":true}]}}} -``` - -For a remote or co-located supervisor, submit one choice and stable launch identities: - -```sh -homebased --json resource return \ - --pending-spec pending.json \ - --resume-spec return.json \ - --request-id \ - --task-id -``` - -Or explicitly close without background work and record why: - -```sh -homebased --json resource return \ - --pending-spec pending.json \ - --no-resume "" -``` - -The authority saves how the accepted task holds the GPU. A same-run resume or a -maintained direct-segment trainer command closes the `restoring` loan when its -start is confirmed, and it becomes the registered training task. A native -foreground command never becomes the registered training task. Its `restoring` -loan keeps the GPU reserved while it runs, and new requests wait. When it exits -with code 0 and a confirmed process-group exit, the loan closes and the next -queued request in serving order runs. Any other end keeps the GPU reserved for `resolve`. A -container return task has the `container` execution mode and behaves the same -way, but its witness is the removed container: the loan closes only when the -container exits with code 0 and Homebased confirms that it is removed. - -The co-located return path uses the authority's local decision owner. A remote -return keeps the callback route on the supervisor machine. Only the first -accepted launch can start a worker. After an unknown result, retry with the -same action, request, task, and work choice. A co-located `release-watch` -command is refused because the authority starts that watcher itself. - -### Resolve an ended return task - -Use `resolve` only when pending shows `restoring` with the exact loan and return -task, and the authority has confirmed that the task ended. This applies to a -direct-segment trainer that ended before its start was confirmed, and to a -native foreground or container task that ended without success. A native -foreground task needs confirmed process-group exit. A container task needs -confirmed container evidence: its exit code was read and its removal was -confirmed, or the container never started. A direct-segment trainer also needs -proof that its exact ownership lock is free: - -```sh -homebased --json resource resolve \ - --pending-spec pending.json \ - --task-id \ - --reason "" -``` - -The authority checks the task and release evidence. This is not a way to force -the GPU free while a task is running or uncertain. A lost task, a task whose -process-group exit is not confirmed, or a task whose identity changed stays -reserved. If a direct-segment return task has no lock proof, only an operator -can release it with the `restoring_return` binding. If a native foreground or -container return task is lost or has no confirmed exit evidence, only an -operator can release it with the `restoring_foreground_return` binding. See -[Resolve an unproven ended trainer](#resolve-an-unproven-ended-trainer). - -### Retry a failed notice - -Renotify only when the exact pending notice is `failed` and its action is still -pending. The authority refuses a renotify after the action resolves: - -```sh -homebased --json resource renotify \ - --pending-spec pending.json \ - --operation-id -``` - -Keep that operation UUID after an unknown result. Renotify retries delivery; it -does not repeat or complete the action. - -## Check status - -Use these read commands when needed: - -```sh -homebased --json resource show -homebased --json resource requests -homebased --json resource pending --machine --thread -``` - -Do not poll in a loop. Recheck after a notice, after compaction, before an -independent background launch, or when an operation has an unknown outcome. -Read the Homebased skill's [resource-loan procedure](../.agents/skills/homebased/references/resource-loans.md) -for the supervisor check points. diff --git a/src/cancellation.rs b/src/cancellation.rs index b98037b..c294e46 100644 --- a/src/cancellation.rs +++ b/src/cancellation.rs @@ -6,19 +6,19 @@ use uuid::Uuid; use crate::domain::{ProcessStatus, TaskId}; use crate::error::AppError; use crate::machine::MachineId; -use crate::resource::ResourceId; -use crate::submission::{ - HeldPhase, OriginRoute, RequestId, ResourceActionRoutePhase, ResourceBackgroundRoutePhase, - ResourceCancellationReceipt, ResourceRoutePhase, SubmissionState, -}; +use crate::submission::{HeldPhase, OriginRoute, RequestId, SubmissionState}; -/// Typed owner of one cancellation target -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum CancellationOwner { - /// An ordinary executor-owned task - Execution(ExecutionCancellationTarget), - /// A task that belongs to a resource authority - Resource(ResourceCancellationTarget), +/// Exact execution identity that owns the cancellation of one task +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct CancellationOwner { + /// Caller retry UUID from the exact origin route + pub request_id: RequestId, + /// Task UUID + pub task: TaskId, + /// Original submission owner + pub origin_machine: MachineId, + /// Fixed execution owner + pub execution_machine: MachineId, } /// Origin route fields that decide which owner may cancel a task @@ -33,7 +33,7 @@ pub struct CancellationRoute<'a> { pub request_id: RequestId, /// Machine that owns the route and its callbacks pub origin_machine: MachineId, - /// Fixed execution owner, or the resource authority for resource routes + /// Fixed execution owner pub execution_machine: MachineId, /// Durable submission result pub submission: &'a SubmissionState, @@ -56,13 +56,8 @@ impl<'a> From<&'a OriginRoute> for CancellationRoute<'a> { pub enum CancellationRefusal { /// The route names another task or origin RouteMismatch, - /// The executor or resource authority rejected the task before it started + /// The executor rejected the task before it started NotStarted, - /// An action-bound or background launch has no accepted task yet - /// - /// Only the launch's own retry may resolve it; a cancellation could fence - /// the fixed identity before the authority accepts it - LaunchUnresolved, /// The task is held on its origin, which alone can cancel it before launch HeldOnOrigin, } @@ -72,7 +67,7 @@ impl CancellationRefusal { #[must_use] pub fn into_error(self, task: TaskId) -> AppError { match self { - Self::RouteMismatch | Self::LaunchUnresolved => AppError::ClusterTaskConflict { task }, + Self::RouteMismatch => AppError::ClusterTaskConflict { task }, Self::NotStarted => AppError::TaskNotStarted { task }, Self::HeldOnOrigin => AppError::Usage { message: format!( @@ -84,29 +79,13 @@ impl CancellationRefusal { } } -/// Next step for one requester after ownership is known -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum CancellationPlan { - /// The retained route already proves cancellation before launch - AlreadyCancelled, - /// Save this requester-owned intent and deliver it - Deliver(Box), - /// The resource route's origin owns the intent, so ask it to save one - ForwardToOrigin(ResourceCancellationTarget), -} - impl CancellationOwner { /// Select the cancellation owner for the route reported by `route_machine` /// - /// A resource route stays resource-owned in every phase, because only its - /// origin may save the intent; [`Self::plan`] picks the delivery path from - /// the retained phase - /// /// # Errors /// /// Refuses a route for another task or origin, a rejected submission, and - /// an action-bound or background launch that the authority has not - /// accepted + /// a route still held on its origin pub fn from_route( task: TaskId, route_machine: MachineId, @@ -124,185 +103,55 @@ impl CancellationOwner { SubmissionState::Held { phase: HeldPhase::Waiting, } => Err(CancellationRefusal::HeldOnOrigin), - SubmissionState::ResourceAction { - phase: ResourceActionRoutePhase::AcceptanceUnknown, - .. - } - | SubmissionState::ResourceAction { - phase: ResourceActionRoutePhase::Rejected { .. }, - .. - } - | SubmissionState::ResourceBackground { - phase: ResourceBackgroundRoutePhase::AcceptanceUnknown, - .. - } - | SubmissionState::ResourceBackground { - phase: ResourceBackgroundRoutePhase::Rejected { .. }, - .. - } => Err(CancellationRefusal::LaunchUnresolved), - SubmissionState::Resource { resource, phase } => { - Ok(Self::Resource(ResourceCancellationTarget { - request_id: route.request_id, - task_id: task, - resource_id: *resource, - origin_machine: route.origin_machine, - authority_machine: route.execution_machine, - phase: phase.clone(), - })) - } // a launching held route may already be on its executor SubmissionState::AcceptanceUnknown | SubmissionState::Accepted | SubmissionState::Held { phase: HeldPhase::Launching, - } - | SubmissionState::ResourceAction { - phase: ResourceActionRoutePhase::Accepted, - .. - } - | SubmissionState::ResourceBackground { - phase: ResourceBackgroundRoutePhase::Accepted, - .. - } => Ok(Self::Execution(ExecutionCancellationTarget { + } => Ok(Self { request_id: route.request_id, task, origin_machine: route.origin_machine, execution_machine: route.execution_machine, - })), + }), } } - /// Machines that may retain the origin route and the execution - #[must_use] - pub const fn machines(&self) -> (MachineId, MachineId) { - match self { - Self::Execution(target) => (target.origin_machine, target.execution_machine), - Self::Resource(target) => (target.origin_machine, target.authority_machine), - } - } - - /// Decide how `requester_machine` cancels `task` under this owner + /// Build the intent that `requester_machine` saves and delivers to cancel `task` /// - /// Ordinary execution accepts an intent from any requester. A resource - /// route accepts an intent only from its origin: before activation the - /// authority cancels the queued request, and after activation the - /// authority's executor cancels the task on the ordinary path + /// Any requester may save an intent for an ordinary execution /// /// # Errors /// - /// Refuses an owner for another task and a resource route that the - /// authority rejected - pub fn plan( + /// Refuses an owner for another task + pub fn request( self, requester_machine: MachineId, task: TaskId, cancellation: Uuid, - ) -> Result { - let target = match self { - Self::Execution(target) if target.task == task => target, - Self::Resource(target) if target.task_id == task => { - return plan_resource(target, requester_machine, cancellation); - } - Self::Execution(_) | Self::Resource(_) => { - return Err(CancellationRefusal::RouteMismatch); - } - }; - Ok(CancellationPlan::Deliver(Box::new(CancellationRequest { + ) -> Result { + if self.task != task { + return Err(CancellationRefusal::RouteMismatch); + } + Ok(CancellationRequest { requester_machine, cancellation, task, - origin_machine: target.origin_machine, - execution_machine: target.execution_machine, + origin_machine: self.origin_machine, + execution_machine: self.execution_machine, target: CancellationTarget::Execution { - request_id: target.request_id, + request_id: self.request_id, }, delivery: CancellationDelivery::Pending, - }))) - } -} - -fn plan_resource( - target: ResourceCancellationTarget, - requester_machine: MachineId, - cancellation: Uuid, -) -> Result { - if target.origin_machine != requester_machine { - return Ok(CancellationPlan::ForwardToOrigin(target)); + }) } - let request_target = match &target.phase { - ResourceRoutePhase::CancelledBeforeLaunch => return Ok(CancellationPlan::AlreadyCancelled), - ResourceRoutePhase::Rejected { .. } => return Err(CancellationRefusal::NotStarted), - ResourceRoutePhase::Activated => CancellationTarget::Execution { - request_id: target.request_id, - }, - ResourceRoutePhase::AcceptanceUnknown | ResourceRoutePhase::Waiting => { - CancellationTarget::Resource(target.clone()) - } - }; - Ok(CancellationPlan::Deliver(Box::new(CancellationRequest { - requester_machine, - cancellation, - task: target.task_id, - origin_machine: target.origin_machine, - execution_machine: target.authority_machine, - target: request_target, - delivery: CancellationDelivery::Pending, - }))) -} - -/// Exact ordinary execution identity selected for cancellation -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub struct ExecutionCancellationTarget { - /// Caller retry UUID from the exact origin route - pub request_id: RequestId, - /// Task UUID - pub task: TaskId, - /// Original submission owner - pub origin_machine: MachineId, - /// Fixed execution owner - pub execution_machine: MachineId, -} - -/// Exact resource route identity selected for cancellation -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct ResourceCancellationTarget { - /// Authority request UUID - pub request_id: RequestId, - /// Preallocated global task UUID - pub task_id: TaskId, - /// Resource whose authority owns the request - pub resource_id: ResourceId, - /// Machine that owns the origin route - pub origin_machine: MachineId, - /// Machine that owns the resource and queue - pub authority_machine: MachineId, - /// Durable phase retained by the origin route - pub phase: ResourceRoutePhase, -} - -/// Immutable identity sent to a resource authority for cancellation -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceCancellationRequestIdentity { - /// Origin machine that persisted this cancellation intent - pub requester_machine: MachineId, - /// Stable cancellation UUID - pub cancellation: Uuid, - /// Authority request UUID - pub request: RequestId, - /// Preallocated global task UUID - pub task: TaskId, - /// Machine that owns the origin route - pub origin_machine: MachineId, - /// Fixed resource authority and execution owner - pub authority_machine: MachineId, - /// Resource whose queue owns the request - pub resource: ResourceId, - /// Resource route phase captured when the intent was persisted - pub target_phase: ResourceRoutePhase, } /// Durable requester-side target for one cancellation intent +/// Durable requester-side target for one cancellation intent +/// +/// The tagged form is the stored and wire shape of saved intents, so it stays +/// an enum with one variant #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] pub enum CancellationTarget { @@ -311,8 +160,6 @@ pub enum CancellationTarget { /// Caller retry UUID from the exact origin route request_id: RequestId, }, - /// Resource request that must use authority-owned cancellation - Resource(ResourceCancellationTarget), } /// One caller-owned cancellation identity @@ -343,11 +190,6 @@ pub enum CancellationDelivery { Pending, /// The executor acknowledged durable receipt Delivered { result: ExecutorCancelState }, - /// The resource authority returned a durable typed cancellation receipt - ResourceDelivered { - /// Resource request outcome, separate from executor process state - result: ResourceCancellationReceipt, - }, } /// Executor-owned state of one received cancellation @@ -404,56 +246,16 @@ impl CancellationRequest { execution_machine: self.execution_machine, } } - - /// Extract the exact authority request identity from a resource intent - #[must_use] - pub fn resource_identity(&self) -> Option { - let CancellationTarget::Resource(target) = &self.target else { - return None; - }; - Some(ResourceCancellationRequestIdentity { - requester_machine: self.requester_machine, - cancellation: self.cancellation, - request: target.request_id, - task: target.task_id, - origin_machine: target.origin_machine, - authority_machine: target.authority_machine, - resource: target.resource_id, - target_phase: target.phase.clone(), - }) - } } #[cfg(test)] mod tests { - use super::{ - CancellationOwner, CancellationPlan, CancellationRefusal, CancellationRoute, - CancellationTarget, ExecutionCancellationTarget, ResourceCancellationTarget, - }; + use super::{CancellationOwner, CancellationRefusal, CancellationRoute, CancellationTarget}; use crate::domain::TaskId; use crate::machine::MachineId; - use crate::resource::ResourceId; - use crate::submission::{RequestId, ResourceRoutePhase, SubmissionState}; + use crate::submission::{RequestId, SubmissionState}; use uuid::Uuid; - fn resource_owner( - task: TaskId, - origin_machine: MachineId, - authority_machine: MachineId, - phase: ResourceRoutePhase, - ) -> (RequestId, CancellationOwner) { - let request_id = RequestId::new(); - let owner = CancellationOwner::Resource(ResourceCancellationTarget { - request_id, - task_id: task, - resource_id: ResourceId::new(), - origin_machine, - authority_machine, - phase, - }); - (request_id, owner) - } - fn route( task: TaskId, origin_machine: MachineId, @@ -468,30 +270,6 @@ mod tests { } } - #[test] - fn activated_resource_route_stays_origin_owned() { - let task = TaskId::new(); - let origin_machine = MachineId::new(); - let submission = SubmissionState::Resource { - resource: ResourceId::new(), - phase: ResourceRoutePhase::Activated, - }; - - let owner = CancellationOwner::from_route( - task, - origin_machine, - route(task, origin_machine, &submission), - ) - .unwrap(); - let CancellationOwner::Resource(target) = owner.clone() else { - panic!("an activated resource route must keep its origin-owned target"); - }; - assert!(matches!( - owner.plan(MachineId::new(), task, Uuid::now_v7()), - Ok(CancellationPlan::ForwardToOrigin(forwarded)) if forwarded == target - )); - } - #[test] fn rejected_or_mismatched_routes_are_refused() { let task = TaskId::new(); @@ -527,87 +305,17 @@ mod tests { ); } - #[test] - fn resource_cancellation_uses_the_typed_path_before_activation() { - let task = TaskId::new(); - for phase in [ - ResourceRoutePhase::AcceptanceUnknown, - ResourceRoutePhase::Waiting, - ] { - let origin_machine = MachineId::new(); - let (_, owner) = resource_owner(task, origin_machine, MachineId::new(), phase); - - let Ok(CancellationPlan::Deliver(request)) = - owner.plan(origin_machine, task, Uuid::now_v7()) - else { - panic!("a pre-activation resource request needs resource cancellation"); - }; - assert!(matches!(request.target, CancellationTarget::Resource(_))); - } - } - - #[test] - fn activated_resource_cancellation_uses_the_executor_task_path() { - let task = TaskId::new(); - let origin_machine = MachineId::new(); - let authority_machine = MachineId::new(); - let (request_id, owner) = resource_owner( - task, - origin_machine, - authority_machine, - ResourceRoutePhase::Activated, - ); - - let Ok(CancellationPlan::Deliver(request)) = - owner.plan(origin_machine, task, Uuid::now_v7()) - else { - panic!("activated resource work must use ordinary execution cancellation"); - }; - assert_eq!(request.execution_machine, authority_machine); - assert_eq!(request.target, CancellationTarget::Execution { request_id }); - } - - #[test] - fn settled_resource_routes_do_not_build_a_cancellation_request() { - let task = TaskId::new(); - let origin_machine = MachineId::new(); - let (_, rejected) = resource_owner( - task, - origin_machine, - MachineId::new(), - ResourceRoutePhase::Rejected { - reason: "resource request rejected".into(), - }, - ); - let (_, cancelled) = resource_owner( - task, - origin_machine, - MachineId::new(), - ResourceRoutePhase::CancelledBeforeLaunch, - ); - - assert_eq!( - rejected.plan(origin_machine, task, Uuid::now_v7()), - Err(CancellationRefusal::NotStarted) - ); - assert_eq!( - cancelled.plan(origin_machine, task, Uuid::now_v7()), - Ok(CancellationPlan::AlreadyCancelled) - ); - } - #[test] fn cancellation_owner_task_mismatch_is_refused() { - let task = TaskId::new(); - let (_, owner) = resource_owner( - task, - MachineId::new(), - MachineId::new(), - ResourceRoutePhase::Waiting, - ); + let owner = CancellationOwner { + request_id: RequestId::new(), + task: TaskId::new(), + origin_machine: MachineId::new(), + execution_machine: MachineId::new(), + }; assert_eq!( - owner.plan(MachineId::new(), TaskId::new(), Uuid::now_v7()), + owner.request(MachineId::new(), TaskId::new(), Uuid::now_v7()), Err(CancellationRefusal::RouteMismatch) ); } @@ -619,18 +327,16 @@ mod tests { let origin_machine = MachineId::new(); let execution_machine = MachineId::new(); let cancellation = Uuid::now_v7(); - let owner = CancellationOwner::Execution(ExecutionCancellationTarget { + let owner = CancellationOwner { request_id, task, origin_machine, execution_machine, - }); - - let Ok(CancellationPlan::Deliver(request)) = - owner.plan(MachineId::new(), task, cancellation) - else { - panic!("ordinary execution must use its generic cancellation intent"); }; + + let request = owner + .request(MachineId::new(), task, cancellation) + .expect("ordinary execution must use its generic cancellation intent"); assert_eq!(request.task, task); assert_eq!(request.origin_machine, origin_machine); assert_eq!(request.execution_machine, execution_machine); diff --git a/src/cli.rs b/src/cli.rs index 134f239..cae6d9b 100644 --- a/src/cli.rs +++ b/src/cli.rs @@ -5,8 +5,6 @@ pub mod daemon; pub mod fleet; pub mod message; pub mod notify; -pub mod release_watcher; -pub mod resource; pub mod t3; pub mod task; pub mod update; @@ -95,12 +93,6 @@ pub enum Command { #[command(subcommand)] command: notify::NotifyCommand, }, - /// Register and manage shared exclusive resources - Resource { - /// Resource subcommand - #[command(subcommand)] - command: resource::ResourceCommand, - }, /// Submit, inspect, cancel, and report tasks Task { /// Task subcommand @@ -117,9 +109,6 @@ pub enum Command { Update(update::UpdateArgs), /// Print the version Version, - /// Hidden authority-bound resource release watcher run as one task - #[command(name = "resource-release-watcher", hide = true)] - ResourceReleaseWatcher(release_watcher::ReleaseWatcherArgs), /// Hidden worker parent of one agent #[command(name = "task-run", hide = true)] TaskRun { @@ -274,11 +263,9 @@ async fn dispatch(cli: Cli) -> Result { Command::Fleet { command } => fleet::run(&ctx, command).await, Command::Message { command } => message::run(&ctx, command).await, Command::Notify { command } => notify::run(&ctx, command).await, - Command::Resource { command } => resource::run(&ctx, command).await, Command::Task { command } => task::run(&ctx, command).await, Command::T3 { command } => t3::run(&ctx, command).await, Command::Update(args) => update::run(&ctx, args).await, - Command::ResourceReleaseWatcher(args) => release_watcher::run(&ctx, args).await, Command::Version => { version(&ctx)?; Ok(ExitCode::SUCCESS) diff --git a/src/cli/release_watcher.rs b/src/cli/release_watcher.rs deleted file mode 100644 index 483d44c..0000000 --- a/src/cli/release_watcher.rs +++ /dev/null @@ -1,363 +0,0 @@ -//! Hidden `resource-release-watcher` command run as one authority-bound Homebased task - -use std::path::Path; -use std::process::ExitCode; -use std::time::Duration; - -use crate::cli::Ctx; -use crate::client::Client; -use crate::domain::TaskId; -use crate::error::AppError; -use crate::resource::release_watcher::{ - RELEASE_WATCHER_POLL_PATH, RELEASE_WATCHER_PROTOCOL_VERSION, ReleaseWatcherCommand, - ReleaseWatcherPollOutcome, ReleaseWatcherPollRequest, ReleaseWatcherPollResponse, -}; -use crate::resource::{ActionId, ReleaseWatcherTaskId, ResourceId, ResourceRevision}; - -const TASK_ID_ENV: &str = "HOMEBASED_TASK_ID"; - -// a checkpoint arrives tens of minutes apart, so a short first poll catches quick -// transitions and the capped backoff keeps the long wait cheap -const WATCHER_POLL_TIMING: PollTiming = PollTiming { - initial: Duration::from_millis(250), - max: Duration::from_secs(10), -}; - -/// Exact release identities from the authority-built watcher command -#[derive(Debug, clap::Args)] -pub struct ReleaseWatcherArgs { - /// Resource whose release action owns this watcher - #[arg(long)] - resource_id: ResourceId, - /// Stable release action identity - #[arg(long)] - action_id: ActionId, - /// Resource revision saved with the release action - #[arg(long)] - state_revision: u64, - /// Exact registered trainer task - #[arg(long)] - trainer_task_id: TaskId, - /// Preallocated task identity of this watcher - #[arg(long)] - watcher_task_id: TaskId, -} - -impl ReleaseWatcherArgs { - fn command(&self) -> ReleaseWatcherCommand { - ReleaseWatcherCommand { - resource_id: self.resource_id, - action_id: self.action_id, - state_revision: ResourceRevision::new(self.state_revision), - trainer_task_id: self.trainer_task_id, - watcher_task_id: ReleaseWatcherTaskId::new(self.watcher_task_id), - } - } -} - -/// Poll the local authority until this watcher's part of the release action ends -pub async fn run(ctx: &Ctx, args: ReleaseWatcherArgs) -> Result { - let command = args.command(); - check_task_identity( - std::env::var(TASK_ID_ENV).ok().as_deref(), - command.watcher_task_id, - )?; - let client = Client::new(ctx.home.sock_path()); - let request = ReleaseWatcherPollRequest::new(command); - let outcome = poll_until_final(&client, &request, ctx.home.root(), WATCHER_POLL_TIMING).await?; - - Ok(report(&outcome)) -} - -/// Refuse to act unless Homebased runs this process as the bound watcher task -fn check_task_identity( - task_id: Option<&str>, - expected: ReleaseWatcherTaskId, -) -> Result<(), AppError> { - let expected = expected.as_task_id(); - match task_id.map(str::parse::) { - Some(Ok(task_id)) if task_id == expected => Ok(()), - Some(Ok(task_id)) => Err(AppError::Usage { - message: format!("release watcher {expected} is running as task {task_id}"), - }), - Some(Err(_)) | None => Err(AppError::Usage { - message: format!("release watcher {expected} must run as its Homebased task"), - }), - } -} - -#[derive(Debug, Clone, Copy)] -struct PollTiming { - initial: Duration, - max: Duration, -} - -/// Send the same poll until the authority returns a final outcome -/// -/// Socket outages and unknown replies are retried with the identical request, so -/// a lost reply resolves through the authority's saved decision. Only changed -/// observations are logged, and the backoff resets when the observation changes -async fn poll_until_final( - client: &Client, - request: &ReleaseWatcherPollRequest, - home_root: &Path, - timing: PollTiming, -) -> Result { - let mut delay = timing.initial; - let mut last_observation = None; - loop { - let observation = match client - .post_json::<_, ReleaseWatcherPollResponse>(RELEASE_WATCHER_POLL_PATH, request) - .await - { - Ok(response) if response.protocol_version == RELEASE_WATCHER_PROTOCOL_VERSION => { - if response.outcome.is_final() { - return Ok(response.outcome); - } - PollObservation::Waiting(response.outcome) - } - Ok(_) => PollObservation::UncertainReply, - Err(AppError::DaemonUnavailable { .. }) => PollObservation::DaemonUnavailable, - Err(AppError::Internal { .. }) => PollObservation::UncertainReply, - Err(error) => return Err(error), - }; - // an absent state directory cannot come back with this watcher's records - if observation.is_retry() && !home_root.exists() { - return Err(AppError::DaemonUnavailable { - message: format!("state directory {} was removed", home_root.display()), - }); - } - - if last_observation.as_ref() != Some(&observation) { - println!("{}", observation.describe()); - last_observation = Some(observation); - delay = timing.initial; - } - tokio::time::sleep(delay).await; - delay = delay.saturating_mul(2).min(timing.max); - } -} - -#[derive(Debug, Clone, PartialEq, Eq)] -enum PollObservation { - Waiting(ReleaseWatcherPollOutcome), - DaemonUnavailable, - UncertainReply, -} - -impl PollObservation { - fn is_retry(&self) -> bool { - matches!(self, Self::DaemonUnavailable | Self::UncertainReply) - } - - fn describe(&self) -> String { - match self { - Self::Waiting(ReleaseWatcherPollOutcome::WatcherNotRunning) => { - "waiting for this watcher's running task identity".into() - } - Self::Waiting(ReleaseWatcherPollOutcome::WaitingForTrainerStart) => { - "waiting for the trainer task to start".into() - } - Self::Waiting(ReleaseWatcherPollOutcome::WaitingForCheckpoint) => { - "waiting for a new complete checkpoint".into() - } - Self::Waiting(ReleaseWatcherPollOutcome::CompletedResultAwaitingTrainerExit) => { - "final result published; waiting for the trainer to exit without a stop".into() - } - Self::Waiting(outcome) => format!("waiting: {outcome:?}"), - Self::DaemonUnavailable => "daemon socket unavailable; retrying the same poll".into(), - Self::UncertainReply => "uncertain daemon reply; retrying the same poll".into(), - } - } -} - -fn report(outcome: &ReleaseWatcherPollOutcome) -> ExitCode { - match outcome { - ReleaseWatcherPollOutcome::StopCommitted { - generation_id, - cancel_requested_at, - } => { - println!( - "trainer stop committed after checkpoint {generation_id} at {cancel_requested_at}" - ); - ExitCode::SUCCESS - } - ReleaseWatcherPollOutcome::TrainerCompleted => { - println!("trainer completed with its final result; no stop was requested"); - ExitCode::SUCCESS - } - ReleaseWatcherPollOutcome::ReleaseSettled => { - println!("release action is already settled"); - ExitCode::SUCCESS - } - ReleaseWatcherPollOutcome::Attention { reason } => { - eprintln!("release watcher needs attention: {reason:?}"); - ExitCode::FAILURE - } - ReleaseWatcherPollOutcome::WatcherNotRunning - | ReleaseWatcherPollOutcome::WaitingForTrainerStart - | ReleaseWatcherPollOutcome::WaitingForCheckpoint - | ReleaseWatcherPollOutcome::CompletedResultAwaitingTrainerExit => { - eprintln!("release watcher stopped before a final outcome: {outcome:?}"); - ExitCode::FAILURE - } - } -} - -#[cfg(test)] -mod tests { - use std::sync::{Arc, Mutex}; - - use axum::Router; - use axum::body::Bytes; - use axum::http::StatusCode; - use axum::response::IntoResponse; - use axum::routing::post; - use clap::Parser; - use tempfile::tempdir; - - use super::{PollTiming, check_task_identity, poll_until_final}; - use crate::cli::{Cli, Command}; - use crate::client::Client; - use crate::domain::TaskId; - use crate::error::AppError; - use crate::resource::release_watcher::{ - RELEASE_WATCHER_POLL_PATH, RELEASE_WATCHER_PROTOCOL_VERSION, ReleaseWatcherCommand, - ReleaseWatcherPollOutcome, ReleaseWatcherPollRequest, ReleaseWatcherPollResponse, - }; - use crate::resource::{ActionId, ReleaseWatcherTaskId, ResourceId, ResourceRevision}; - use std::path::Path; - use std::time::Duration; - - const FAST: PollTiming = PollTiming { - initial: Duration::from_millis(10), - max: Duration::from_millis(40), - }; - - fn command() -> ReleaseWatcherCommand { - ReleaseWatcherCommand { - resource_id: ResourceId::new(), - action_id: ActionId::new(), - state_revision: ResourceRevision::new(2), - trainer_task_id: TaskId::new(), - watcher_task_id: ReleaseWatcherTaskId::new(TaskId::new()), - } - } - - #[test] - fn canonical_argv_parses_as_the_hidden_command() { - let command = command(); - let argv = command.argv(Path::new("/usr/local/bin/homebased")).unwrap(); - let cli = Cli::try_parse_from(argv).unwrap(); - let Command::ResourceReleaseWatcher(args) = cli.command else { - panic!("canonical argv must select the hidden watcher command"); - }; - assert_eq!(args.command(), command); - } - - #[test] - fn watcher_refuses_to_run_outside_its_bound_task() { - let watcher = ReleaseWatcherTaskId::new(TaskId::new()); - assert!(check_task_identity(Some(&watcher.as_task_id().to_string()), watcher).is_ok()); - assert!(matches!( - check_task_identity(Some(&TaskId::new().to_string()), watcher), - Err(AppError::Usage { .. }) - )); - assert!(matches!( - check_task_identity(Some("not-a-task"), watcher), - Err(AppError::Usage { .. }) - )); - assert!(matches!( - check_task_identity(None, watcher), - Err(AppError::Usage { .. }) - )); - } - - #[tokio::test] - async fn socket_outage_and_unknown_replies_retry_the_identical_poll() { - let directory = tempdir().unwrap(); - let socket = directory.path().join("homebased.sock"); - let request = ReleaseWatcherPollRequest::new(command()); - let bodies = Arc::new(Mutex::new(Vec::::new())); - - let server_socket = socket.clone(); - let server_bodies = bodies.clone(); - let server = tokio::spawn(async move { - // the daemon socket is absent for the watcher's first polls - tokio::time::sleep(Duration::from_millis(60)).await; - let listener = tokio::net::UnixListener::bind(&server_socket).unwrap(); - let router = Router::new().route( - RELEASE_WATCHER_POLL_PATH, - post(move |body: Bytes| { - let bodies = server_bodies.clone(); - async move { - let count = { - let mut bodies = bodies.lock().unwrap(); - bodies.push(body); - bodies.len() - }; - let outcome = match count { - 1 => { - return (StatusCode::INTERNAL_SERVER_ERROR, "lost").into_response(); - } - 2 => ReleaseWatcherPollOutcome::WaitingForCheckpoint, - _ => ReleaseWatcherPollOutcome::StopCommitted { - generation_id: "generation-2".into(), - cancel_requested_at: chrono::Utc::now(), - }, - }; - axum::Json(ReleaseWatcherPollResponse { - protocol_version: RELEASE_WATCHER_PROTOCOL_VERSION, - outcome, - }) - .into_response() - } - }), - ); - axum::serve(listener, router).await.unwrap(); - }); - - let outcome = tokio::time::timeout( - Duration::from_secs(10), - poll_until_final(&Client::new(socket), &request, directory.path(), FAST), - ) - .await - .unwrap() - .unwrap(); - server.abort(); - - assert!(matches!( - outcome, - ReleaseWatcherPollOutcome::StopCommitted { ref generation_id, .. } - if generation_id == "generation-2" - )); - let bodies = bodies.lock().unwrap(); - assert_eq!(bodies.len(), 3); - for body in bodies.iter() { - assert_eq!( - serde_json::from_slice::(body).unwrap(), - request - ); - } - } - - #[tokio::test] - async fn removed_state_directory_ends_an_outage_retry() { - let directory = tempdir().unwrap(); - let home = directory.path().join("home"); - let request = ReleaseWatcherPollRequest::new(command()); - - let result = tokio::time::timeout( - Duration::from_secs(5), - poll_until_final( - &Client::new(home.join("homebased.sock")), - &request, - &home, - FAST, - ), - ) - .await - .unwrap(); - assert!(matches!(result, Err(AppError::DaemonUnavailable { .. }))); - } -} diff --git a/src/cli/resource.rs b/src/cli/resource.rs deleted file mode 100644 index dd118e5..0000000 --- a/src/cli/resource.rs +++ /dev/null @@ -1,3630 +0,0 @@ -//! Resource CLI commands over the daemon Unix socket - -use crate::resource::CommandSpecError; -use crate::resource::trainer_publication::AttemptBinding; -use std::io::{self, Read}; -use std::path::PathBuf; -use std::process::ExitCode; -use std::str::FromStr; -use std::time::Duration; - -use chrono::SecondsFormat; -use clap::{Args, Subcommand}; -use serde::{Deserialize, Serialize, de::DeserializeOwned}; -use serde_json::{Value, json}; -use uuid::Uuid; - -use crate::callback::destination::SubmitOrigin; -use crate::client::Client; -use crate::daemon::fleet_api::MachinesBody; -use crate::domain::{API_VERSION, THREAD_ENV_VARS, TaskEnv, TaskId, ThreadId}; -use crate::error::AppError; -use crate::machine::MachineId; -use crate::resource::api::{ - BrowserResourceAction, InitialIdleBody, InitialIdleResponse, OperatorReleaseBody, - OperatorReleaseResponse, PendingActionList, PendingActionPhase, PendingActionView, - QueuePlacement, RESOURCE_PENDING_PATH, RESOURCE_REGISTER_PATH, RequestCancelBody, - ResourceActionBody, ResourceBackgroundSubmitOutcome, ResourceBackgroundSubmitResponse, - ResourceDetail, ResourceRegisterBody, ResourceRegistration, ResourceRequestSubmitOutcome, - ResourceRequestSubmitResponse, SupervisorReplacementBody, TrainerAttemptBody, - TrainerAttemptResponse, UnavailableAuthority, -}; -use crate::resource::bound_action::{ - LocalReturnAcceptance, RESOURCE_ACTION_SUBMIT_PATH, ResourceActionChoice, ResourceActionKind, - ResourceActionRejection, ResourceActionSubmitOutcome, ResourceActionSubmitRequest, - ResourceActionSubmitResponse, -}; -use crate::resource::initial_idle::InitialIdleAttestation; -use crate::resource::operator_release::OperatorGpuFreeAttestation; -use crate::resource::{ - ActionId, CommandSpec, ResourceId, ResourceRevision, ReturnContext, ReturnDecision, - ReturnDecisionRejection, ReturnLaunch, ReturnWork, SupervisorActionAuthority, - SupervisorAddress, SupervisorNotice, SupervisorNoticeDelivery, -}; -use crate::spec::{self, NormalizedSpec}; -use crate::submission::RequestId; - -use super::{Ctx, OutputMode}; - -/// Resource commands. All requests use the local daemon's Unix socket -#[derive(Debug, Subcommand)] -#[command(after_help = crate::cli::AFTER_HELP)] -pub enum ResourceCommand { - /// Print schemas for resource registration, work, trainer attempts, and operator release - Schema, - /// Register a resource on the local authority machine from a JSON spec - Register { - /// Registration spec file, or `-` for stdin - #[arg(long)] - spec: String, - }, - /// Record that a resource with no history starts with a free GPU - /// - /// Use it once, on the authority machine, for a newly registered resource - /// that has no background run, loan, or release record, after you inspected - /// its GPU. Queued requests then serve. This is a human confirmation, not - /// automatic proof. Reuse the exact same document after an unknown result - InitialIdle { - /// Complete InitialIdleAttestation JSON file, or `-` for stdin - #[arg(long, required = true, value_name = "FILE")] - spec: String, - }, - /// Record a human GPU-free confirmation on the local authority machine - /// - /// Inspect the authority GPU before you use this command. This is a human - /// confirmation, not automatic proof that GPU work stopped. Reuse the exact - /// same document, operation_id, and observation after an unknown result - OperatorRelease { - /// Complete OperatorGpuFreeAttestation JSON file, or `-` for stdin - #[arg(long, required = true, value_name = "FILE")] - spec: String, - }, - /// Show one resource and its authoritative state - Show { - /// Full resource UUID - #[arg(value_name = "RESOURCE_ID")] - resource_id: Uuid, - }, - /// Change a resource's assigned supervisor - Supervisor { - /// Supervisor operation - #[command(subcommand)] - command: SupervisorCommand, - }, - /// Submit command work as a resource background task - Background { - /// Background-task operation - #[command(subcommand)] - command: BackgroundCommand, - }, - /// Submit or cancel a resource request - Request { - /// Request operation - #[command(subcommand)] - command: RequestCommand, - }, - /// List resource requests in serving order - Requests { - /// Full resource UUID - #[arg(value_name = "RESOURCE_ID")] - resource_id: Uuid, - }, - /// List actions assigned to one exact supervisor address - Pending { - /// Supervisor machine UUID - #[arg(long)] - machine: MachineId, - /// Exact supervisor thread UUID - #[arg(long)] - thread: ThreadId, - }, - /// Start the watcher for a pending release action on another machine - /// - /// When the supervisor runs on the resource authority machine, the authority - /// starts the watcher itself and this command is refused - ReleaseWatch { - /// Stable release action UUID - #[arg(value_name = "ACTION_ID")] - action_id: Uuid, - /// Saved JSON from `resource pending --json`; reuse it for an unknown-result retry - #[arg(long, required = true, value_name = "FILE")] - pending_spec: String, - }, - /// Make an explicit return choice for a pending action, or hold it open longer - /// - /// Works on the same machine as the resource authority or on another machine - /// A retry with the same identities replays the saved result and never starts - /// the task again; a different choice for the same action is a conflict - /// - /// Queued requests take the resource 2 minutes after the return action opens - /// unless a choice exists. `--hold` moves that deadline to now plus the given - /// time, up to 10 minutes after the action opened, and is not a choice - Return { - /// Stable return action UUID - #[arg(value_name = "ACTION_ID")] - action_id: Uuid, - /// Saved JSON from `resource pending --json`; reuse it for an unknown-result retry - #[arg(long, required = true, value_name = "FILE")] - pending_spec: String, - /// JSON file containing one tagged ReturnWork choice, not a task-submit spec - #[arg( - long, - conflicts_with_all = ["no_resume", "hold"], - requires_all = ["request_id", "task_id"] - )] - resume_spec: Option, - /// Close the return action without starting background work and record this reason - #[arg( - long, - value_name = "REASON", - conflicts_with_all = ["resume_spec", "hold"], - required_unless_present_any = ["resume_spec", "hold"] - )] - no_resume: Option, - /// Keep queued requests waiting this much longer for a choice, such as `5m` - #[arg( - long, - value_name = "DURATION", - value_parser = humantime::parse_duration, - conflicts_with_all = ["resume_spec", "no_resume"] - )] - hold: Option, - /// Stable retry identity for the return task - #[arg(long, requires = "resume_spec")] - request_id: Option, - /// Preallocated task identity for the return task - #[arg(long, requires = "resume_spec")] - task_id: Option, - }, - /// Resolve a return task that ended before its start was confirmed - /// - /// Works on the same machine as the resource authority or on another machine - /// A retry with the same task and reason replays the saved closure - Resolve { - /// Full loan UUID - #[arg(value_name = "LOAN_ID")] - loan_id: Uuid, - /// Saved JSON from `resource pending --json`; reuse it for an unknown-result retry - #[arg(long, required = true, value_name = "FILE")] - pending_spec: String, - /// Exact bound return task UUID - #[arg(long)] - task_id: TaskId, - /// Durable supervisor resolution - #[arg(long)] - reason: String, - }, - /// Retry delivery of the notice for one pending action - Renotify { - /// Stable action UUID whose notice must be retried - #[arg(value_name = "ACTION_ID")] - action_id: Uuid, - /// Saved JSON from `resource pending --json`; reuse it for an unknown-result retry - #[arg(long, required = true, value_name = "FILE")] - pending_spec: String, - /// Stable operation UUID. Reuse this value after an unknown response - #[arg(long, required = true)] - operation_id: Uuid, - }, -} - -/// Supervisor-assignment commands -#[derive(Debug, Subcommand)] -pub enum SupervisorCommand { - /// Set the exact supervisor machine and thread - Set { - /// Full resource UUID - #[arg(value_name = "RESOURCE_ID")] - resource_id: Uuid, - /// Supervisor machine UUID - #[arg(long)] - machine: MachineId, - /// Exact supervisor thread UUID - #[arg(long)] - thread: ThreadId, - /// Compare-and-set revision from `resource show`; reuse it with the same supervisor on retry - #[arg(long, required = true)] - expected_revision: u64, - }, -} - -/// Background-task commands -#[derive(Debug, Subcommand)] -pub enum BackgroundCommand { - /// Submit a command as the registered background task - Submit { - /// Full resource UUID - #[arg(value_name = "RESOURCE_ID")] - resource_id: Uuid, - /// Stable request UUID. Reuse it after an unknown response - #[arg(long, required = true)] - request_id: Uuid, - /// Task spec file, or `-` for stdin - #[arg(long)] - spec: String, - /// Let a Homebased worker send events to a thread other than its parent task's thread - #[arg(long)] - allow_other_thread: bool, - }, - /// Bind a trainer attempt to one registered Homebased task - BindAttempt { - /// Full resource UUID - #[arg(value_name = "RESOURCE_ID")] - resource_id: Uuid, - /// Exact registered Homebased task UUID - #[arg(long, required = true)] - task_id: TaskId, - /// Strict trainer AttemptBinding JSON file, or `-` for stdin - #[arg(long, required = true, value_name = "FILE")] - attempt_spec: String, - }, -} - -/// Resource request commands -#[derive(Debug, Subcommand)] -pub enum RequestCommand { - /// Submit one command to the authority-managed request queue - Submit { - /// Full resource UUID - #[arg(value_name = "RESOURCE_ID")] - resource_id: Uuid, - /// Stable request UUID. Reuse it after an unknown response - #[arg(long, required = true)] - request_id: Uuid, - /// Task spec file, or `-` for stdin - #[arg(long)] - spec: String, - /// Let a Homebased worker send events to a thread other than its parent task's thread - #[arg(long)] - allow_other_thread: bool, - }, - /// Cancel one queued request before task activation - Cancel { - /// Full resource UUID - #[arg(value_name = "RESOURCE_ID")] - resource_id: Uuid, - /// Full request UUID, not a task UUID - #[arg(value_name = "REQUEST_ID")] - request_id: Uuid, - /// Compare-and-set revision from `resource show` - #[arg(long, required = true)] - expected_revision: u64, - /// Stable operation UUID. Reuse it after an unknown response - #[arg(long, required = true)] - operation_id: Uuid, - }, - /// Move one queued request to a new place in the serving order - Move { - /// Full resource UUID - #[arg(value_name = "RESOURCE_ID")] - resource_id: Uuid, - /// Full request UUID, not a task UUID - #[arg(value_name = "REQUEST_ID")] - request_id: Uuid, - /// Compare-and-set revision from `resource show` - #[arg(long, required = true)] - expected_revision: u64, - /// Stable operation UUID. Reuse it after an unknown response - #[arg(long, required = true)] - operation_id: Uuid, - /// New place in the queue - #[command(flatten)] - placement: PlacementArgs, - }, -} - -/// Exactly one queue placement flag of `resource request move` -#[derive(Debug, Args)] -#[group(required = true, multiple = false)] -pub struct PlacementArgs { - /// Move the request to the front of the queue - #[arg(long)] - front: bool, - /// Move the request to the back of the queue - #[arg(long)] - back: bool, - /// Move the request directly before this queued request - #[arg(long, value_name = "REQUEST_ID")] - before: Option, - /// Move the request directly after this queued request - #[arg(long, value_name = "REQUEST_ID")] - after: Option, -} - -impl PlacementArgs { - /// Convert the one flag that clap accepted into a queue placement - fn placement(&self) -> Result { - if let Some(anchor) = self.before { - validate_uuid("--before", anchor)?; - return Ok(QueuePlacement::Before { - request_id: RequestId(anchor), - }); - } - if let Some(anchor) = self.after { - validate_uuid("--after", anchor)?; - return Ok(QueuePlacement::After { - request_id: RequestId(anchor), - }); - } - // the required single-choice group leaves only front or back here - Ok(if self.front { - QueuePlacement::Front - } else { - QueuePlacement::Back - }) - } -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct ResourceRegistrationSpec { - /// Stable identity used to make registration retries idempotent - id: Uuid, - /// Human-readable resource name - display_name: String, - /// Exact assigned supervisor - supervisor: SupervisorAddress, -} - -#[derive(Debug, Serialize)] -#[serde(deny_unknown_fields)] -struct ResourceSubmitBody { - api_version: u32, - request_id: RequestId, - spec: NormalizedSpec, - env: TaskEnv, - callback_cwd: PathBuf, -} - -struct ActionContext { - resource_id: ResourceId, - action_id: ActionId, - authority: SupervisorActionAuthority, - phase: PendingActionPhase, - return_context: Option, - notice: Option, -} - -/// Run one resource command -pub async fn run(ctx: &Ctx, command: ResourceCommand) -> Result { - match command { - ResourceCommand::Schema => schema(ctx), - ResourceCommand::Register { spec } => register(ctx, &spec).await, - ResourceCommand::OperatorRelease { spec } => operator_release(ctx, &spec).await, - ResourceCommand::InitialIdle { spec } => initial_idle(ctx, &spec).await, - ResourceCommand::Show { resource_id } => show(ctx, resource_id).await, - ResourceCommand::Supervisor { command } => supervisor(ctx, command).await, - ResourceCommand::Background { command } => background(ctx, command).await, - ResourceCommand::Request { command } => request(ctx, command).await, - ResourceCommand::Requests { resource_id } => requests(ctx, resource_id).await, - ResourceCommand::Pending { machine, thread } => pending(ctx, machine, thread).await, - ResourceCommand::ReleaseWatch { - action_id, - pending_spec, - } => release_watch(ctx, action_id, &pending_spec).await, - ResourceCommand::Return { - action_id, - pending_spec, - resume_spec, - no_resume, - hold, - request_id, - task_id, - } => { - let choice = match hold { - Some(hold) => ReturnChoice::Hold(hold), - None => ReturnChoice::Decide { - resume_path: resume_spec.as_deref(), - no_resume: no_resume.as_deref(), - request_uuid: request_id, - task_id, - }, - }; - return_action(ctx, action_id, &pending_spec, choice).await - } - ResourceCommand::Resolve { - loan_id, - pending_spec, - task_id, - reason, - } => resolve(ctx, loan_id, &pending_spec, task_id, reason).await, - ResourceCommand::Renotify { - action_id, - pending_spec, - operation_id, - } => renotify(ctx, action_id, &pending_spec, operation_id).await, - } -} - -fn schema(ctx: &Ctx) -> Result { - let value = json!({ - "$schema": "https://json-schema.org/draft/2020-12/schema", - "title": "Homebased resource CLI inputs", - "description": "ResourceRegistrationSpec is used by resource register. ResourceTaskSubmitSpec is used by resource background submit and resource request submit. ResourceReturnWorkSpec is used by resource return --resume-spec. ResourceTrainerAttemptBinding is used by resource background bind-attempt. OperatorGpuFreeAttestation is a human confirmation after inspecting the authority GPU, not automatic proof. InitialIdleAttestation is used by resource initial-idle for a resource with no history; it is also a human confirmation.", - "$defs": { - "ResourceRegistrationSpec": resource_registration_schema(), - "ResourceTaskSubmitSpec": resource_task_schema()?, - "ResourceReturnWorkSpec": resource_return_work_schema()?, - "ResourceTrainerAttemptBinding": resource_trainer_attempt_binding_schema(), - "OperatorGpuFreeAttestation": operator_gpu_free_attestation_schema(), - "InitialIdleAttestation": initial_idle_attestation_schema(), - }, - }); - emit(ctx, value, None, None)?; - Ok(ExitCode::SUCCESS) -} - -fn resource_registration_schema() -> Value { - json!({ - "title": "ResourceRegistrationSpec", - "type": "object", - "additionalProperties": false, - "required": ["id", "display_name", "supervisor"], - "properties": { - "id": { "type": "string", "format": "uuid", "description": "Stable resource UUID. Keep it unchanged when retrying registration." }, - "display_name": { "type": "string", "minLength": 1 }, - "supervisor": { - "type": "object", - "additionalProperties": false, - "required": ["machine", "thread"], - "properties": { - "machine": { "type": "string", "format": "uuid" }, - "thread": { "type": "string", "format": "uuid" } - } - }, - } - }) -} - -fn resource_task_schema() -> Result { - let mut schema = spec::schema_json()?; - let command_schema = schema - .pointer("/properties/workload/oneOf") - .and_then(Value::as_array) - .and_then(|branches| { - branches.iter().find(|branch| { - branch - .pointer("/properties/type/const") - .and_then(Value::as_str) - == Some("task") - }) - }) - .and_then(|branch| branch.pointer("/properties/command")) - .cloned() - .ok_or_else(|| invalid_daemon_response("task submit schema has no command workload"))?; - let properties = schema - .get_mut("properties") - .and_then(Value::as_object_mut) - .ok_or_else(|| invalid_daemon_response("task submit schema has no properties"))?; - properties.remove("machine"); - properties.insert( - "workload".into(), - json!({ - "description": "Finite command, or a container with gpus. Container work is accepted for queued requests and for evaluation_or_next_epoch, new_background_work, and after_ended_run returns.", - "oneOf": [ - { - "title": "ResourceCommandWorkload", - "type": "object", - "additionalProperties": false, - "required": ["type", "command"], - "properties": { - "type": { "const": "task" }, - "command": command_schema - } - }, - crate::container::spec::container_workload_schema(true) - ] - }), - ); - if let Some(object) = schema.as_object_mut() { - object.insert("title".into(), Value::from("ResourceTaskSubmitSpec")); - object.insert( - "description".into(), - Value::from( - "SubmitSpec with a finite command or container workload and no execution machine.", - ), - ); - } - Ok(schema) -} - -fn resource_return_work_schema() -> Result { - let command_spec = resource_task_schema()?; - Ok(json!({ - "title": "ResourceReturnWorkSpec", - "description": "Supervisor choice bound to the saved return context. Task identities and recovery references must match the pending action.", - "oneOf": [ - { - "title": "same_run_resume", - "type": "object", - "additionalProperties": false, - "required": ["type", "stopped_task", "recovery_ref"], - "properties": { - "type": { "const": "same_run_resume" }, - "stopped_task": { "type": "string", "format": "uuid" }, - "recovery_ref": { "type": "string", "minLength": 1 } - } - }, - { - "title": "evaluation_or_next_epoch", - "type": "object", - "additionalProperties": false, - "required": ["type", "completed_task", "spec"], - "properties": { - "type": { "const": "evaluation_or_next_epoch" }, - "completed_task": { "type": "string", "format": "uuid" }, - "spec": command_spec - } - }, - { - "title": "new_background_work", - "type": "object", - "additionalProperties": false, - "required": ["type", "spec"], - "properties": { - "type": { "const": "new_background_work" }, - "spec": command_spec - } - }, - { - "title": "after_ended_run", - "type": "object", - "additionalProperties": false, - "required": ["type", "ended_task", "spec"], - "properties": { - "type": { "const": "after_ended_run" }, - "ended_task": { "type": "string", "format": "uuid" }, - "spec": command_spec - } - } - ] - })) -} - -fn resource_trainer_attempt_binding_schema() -> Value { - let identifier = json!({ - "type": "string", - "minLength": 1, - "maxLength": 128, - "pattern": "^[a-z0-9][a-z0-9._-]{0,127}$" - }); - json!({ - "title": "ResourceTrainerAttemptBinding", - "description": "Strict trainer AttemptBinding. Its task_id is the trainer identity, not the Homebased task UUID passed to resource background bind-attempt.", - "type": "object", - "additionalProperties": false, - "required": ["campaign_id", "campaign_revision_id", "task_id", "attempt_id", "attempt_number", "ownership_token"], - "properties": { - "campaign_id": identifier, - "campaign_revision_id": identifier, - "task_id": { - "description": "Trainer task identity, not a Homebased TaskId", - "type": "string", - "minLength": 1, - "maxLength": 128, - "pattern": "^[a-z0-9][a-z0-9._-]{0,127}$" - }, - "attempt_id": identifier, - "attempt_number": { "type": "integer", "minimum": 1 }, - "ownership_token": identifier - } - }) -} - -fn operator_gpu_free_attestation_schema() -> Value { - let uuid = json!({ "type": "string", "format": "uuid" }); - json!({ - "title": "OperatorGpuFreeAttestation", - "description": "Complete document for resource operator-release. It records a human confirmation after inspecting the authority GPU; it is not automatic proof that GPU work stopped. Keep operation_id and observation unchanged on retry.", - "type": "object", - "additionalProperties": false, - "required": [ - "operation_id", - "resource_id", - "authority_machine", - "task_id", - "expected_state_revision", - "state_binding", - "observation", - "confirmation" - ], - "properties": { - "operation_id": uuid, - "resource_id": uuid, - "authority_machine": uuid, - "task_id": uuid, - "expected_state_revision": { "type": "integer", "minimum": 0 }, - "state_binding": { - "oneOf": [ - { - "type": "object", - "additionalProperties": false, - "required": ["type"], - "properties": { "type": { "const": "no_loan" } } - }, - { - "type": "object", - "additionalProperties": false, - "required": ["type", "loan_id", "action_id"], - "properties": { - "type": { "const": "awaiting_release" }, - "loan_id": uuid, - "action_id": uuid - } - }, - { - "type": "object", - "additionalProperties": false, - "required": ["type", "request_id"], - "description": "No loan; task_id is the launch task of background_launch with status release_unproven", - "properties": { - "type": { "const": "first_background_launch" }, - "request_id": uuid - } - }, - { - "type": "object", - "additionalProperties": false, - "required": ["type", "loan_id", "action_id"], - "description": "Restoring loan; task_id is its resume_task_id, a direct-segment return task that ended before its confirmed start", - "properties": { - "type": { "const": "restoring_return" }, - "loan_id": uuid, - "action_id": uuid - } - }, - { - "type": "object", - "additionalProperties": false, - "required": ["type", "loan_id", "action_id"], - "description": "Restoring loan; task_id is its resume_task_id, a native foreground return task that ended or is lost without a confirmed process-group exit", - "properties": { - "type": { "const": "restoring_foreground_return" }, - "loan_id": uuid, - "action_id": uuid - } - } - ] - }, - "observation": { - "type": "string", - "minLength": 1, - "description": "Human account of what was inspected and why the GPU is free" - }, - "confirmation": { "const": "operator_confirmed_gpu_free" } - } - }) -} - -async fn operator_release(ctx: &Ctx, path: &str) -> Result { - let attestation: OperatorGpuFreeAttestation = load_json(path)?; - attestation - .validate() - .map_err(|error| invalid_resource_spec("", &error.to_string()))?; - let resource = attestation.resource_id; - let operation = attestation.operation_id.as_uuid(); - let retry_identity = operator_release_retry_identity(path, operation); - let body = OperatorReleaseBody { - api_version: API_VERSION, - attestation: attestation.clone(), - }; - let client = Client::new(ctx.home.sock_path()); - let value = client - .post( - &format!("/v1/resources/{}/operator-release", resource.as_uuid()), - &serde_json::to_value(&body)?, - ) - .await - .map_err(|error| { - operator_release_mutation_error(resource, operation, &retry_identity, error) - })?; - check_version(&value).map_err(|error| { - operator_release_unknown(resource, operation, &retry_identity, error.to_string()) - })?; - let response: OperatorReleaseResponse = serde_json::from_value(value).map_err(|error| { - operator_release_unknown( - resource, - operation, - &retry_identity, - format!("the authority returned an invalid receipt: {error}"), - ) - })?; - check_operator_release_response(&response, &attestation, &retry_identity)?; - - let replayed = response.replayed; - let value = serde_json::to_value(response).map_err(|error| { - operator_release_unknown( - resource, - operation, - &retry_identity, - format!("the saved receipt could not be rendered: {error}"), - ) - })?; - let human = if replayed { - format!("saved operator attestation receipt {operation} was replayed") - } else { - format!("operator attestation receipt {operation} was saved") - }; - emit(ctx, value, Some(&operation.to_string()), Some(&human))?; - Ok(ExitCode::SUCCESS) -} - -fn operator_release_retry_identity(path: &str, operation: Uuid) -> String { - format!( - "retry with the exact same document, operation_id {operation}, and unchanged observation using `homebased resource operator-release --spec {path}`" - ) -} - -fn operator_release_mutation_error( - resource: ResourceId, - operation: Uuid, - retry_identity: &str, - error: AppError, -) -> AppError { - match mutation_error(resource, Some(operation), retry_identity, error) { - AppError::ResourceOutcomeUnknown { message, .. } => AppError::ResourceOutcomeUnknown { - resource, - operation: Some(operation), - message: if message.contains(retry_identity) { - message - } else { - format!("{message}; {retry_identity}") - }, - }, - error => error, - } -} - -fn operator_release_unknown( - resource: ResourceId, - operation: Uuid, - retry_identity: &str, - detail: String, -) -> AppError { - AppError::ResourceOutcomeUnknown { - resource, - operation: Some(operation), - message: format!("{detail}; {retry_identity}"), - } -} - -fn check_operator_release_response( - response: &OperatorReleaseResponse, - expected: &OperatorGpuFreeAttestation, - retry_identity: &str, -) -> Result<(), AppError> { - if response.api_version == API_VERSION && response.receipt.attestation == *expected { - return Ok(()); - } - - Err(operator_release_unknown( - expected.resource_id, - expected.operation_id.as_uuid(), - retry_identity, - "the authority returned a different attestation identity or API version".into(), - )) -} - -fn initial_idle_attestation_schema() -> Value { - let uuid = json!({ "type": "string", "format": "uuid" }); - json!({ - "title": "InitialIdleAttestation", - "description": "Complete document for resource initial-idle. Use it once for a registered resource with no registered background task, loan, first background launch, or operator attestation, after inspecting the authority GPU. It records a human confirmation, not automatic proof. Keep operation_id and observation unchanged on retry.", - "type": "object", - "additionalProperties": false, - "required": [ - "operation_id", - "resource_id", - "authority_machine", - "expected_state_revision", - "observation", - "confirmation" - ], - "properties": { - "operation_id": uuid, - "resource_id": uuid, - "authority_machine": uuid, - "expected_state_revision": { "type": "integer", "minimum": 0 }, - "observation": { - "type": "string", - "minLength": 1, - "description": "Human account of what was inspected and why no work holds the GPU" - }, - "confirmation": { "const": "operator_confirmed_gpu_free" } - } - }) -} - -async fn initial_idle(ctx: &Ctx, path: &str) -> Result { - let attestation: InitialIdleAttestation = load_json(path)?; - attestation - .validate() - .map_err(|error| invalid_resource_spec("", &error.to_string()))?; - let resource = attestation.resource_id; - let operation = attestation.operation_id.as_uuid(); - let retry_identity = format!( - "retry with the exact same document and operation_id {operation} using `homebased resource initial-idle --spec {path}`" - ); - let body = InitialIdleBody { - api_version: API_VERSION, - attestation: attestation.clone(), - }; - let client = Client::new(ctx.home.sock_path()); - let value = client - .post( - &format!("/v1/resources/{}/initial-idle", resource.as_uuid()), - &serde_json::to_value(&body)?, - ) - .await - .map_err(|error| { - operator_release_mutation_error(resource, operation, &retry_identity, error) - })?; - check_version(&value).map_err(|error| { - operator_release_unknown(resource, operation, &retry_identity, error.to_string()) - })?; - let response: InitialIdleResponse = serde_json::from_value(value).map_err(|error| { - operator_release_unknown( - resource, - operation, - &retry_identity, - format!("the authority returned an invalid receipt: {error}"), - ) - })?; - if response.api_version != API_VERSION || response.receipt.attestation != attestation { - return Err(operator_release_unknown( - resource, - operation, - &retry_identity, - "the authority returned a different attestation identity or API version".into(), - )); - } - - let human = if response.replayed { - format!("saved initial idle receipt {operation} was replayed") - } else { - format!("initial idle receipt {operation} was saved") - }; - let value = serde_json::to_value(response)?; - emit(ctx, value, Some(&operation.to_string()), Some(&human))?; - Ok(ExitCode::SUCCESS) -} - -async fn register(ctx: &Ctx, path: &str) -> Result { - let spec: ResourceRegistrationSpec = load_json(path)?; - let resource_id = validate_registration(&spec)?; - let body = ResourceRegisterBody { - api_version: API_VERSION, - spec: ResourceRegistration { - id: resource_id, - display_name: spec.display_name, - supervisor: spec.supervisor, - }, - }; - let client = Client::new(ctx.home.sock_path()); - let value = post_mutation( - &client, - RESOURCE_REGISTER_PATH, - &body, - resource_id, - Some(resource_id.as_uuid()), - &format!("resource registration id {}", resource_id.as_uuid()), - ) - .await?; - check_mutation_resource( - &value, - resource_id, - Some(resource_id.as_uuid()), - &format!("resource id {}", resource_id.as_uuid()), - )?; - emit(ctx, value, Some(&resource_id.as_uuid().to_string()), None)?; - Ok(ExitCode::SUCCESS) -} - -/// Check a registration spec and return its resource identity -fn validate_registration(spec: &ResourceRegistrationSpec) -> Result { - let resource_id = ResourceId::from_uuid(spec.id) - .map_err(|_| invalid_resource_spec("/id", "resource id must not be nil"))?; - if spec.display_name.trim().is_empty() { - return Err(invalid_resource_spec( - "/display_name", - "display_name must not be empty", - )); - } - if spec.display_name.chars().count() > 120 || spec.display_name.chars().any(char::is_control) { - return Err(invalid_resource_spec( - "/display_name", - "display_name must be at most 120 characters without control characters", - )); - } - if spec.supervisor.machine.as_uuid().is_nil() || spec.supervisor.thread.0.is_nil() { - return Err(invalid_resource_spec( - "/supervisor", - "supervisor machine and thread ids must not be nil", - )); - } - Ok(resource_id) -} - -fn invalid_resource_spec(pointer: &str, message: &str) -> AppError { - AppError::InvalidSpec { - pointer: pointer.into(), - value: Value::Null, - message: message.into(), - } -} - -/// Wrap the `RESOURCE_ID` argument, refusing the nil UUID -fn resource_id_arg(value: Uuid) -> Result { - ResourceId::from_uuid(value).map_err(|_| AppError::Usage { - message: "RESOURCE_ID must not be nil".into(), - }) -} - -fn validate_uuid(field: &str, value: Uuid) -> Result<(), AppError> { - if value.is_nil() { - return Err(AppError::Usage { - message: format!("{field} must not be nil"), - }); - } - Ok(()) -} - -async fn show(ctx: &Ctx, resource_id: Uuid) -> Result { - let id = resource_id_arg(resource_id)?; - let client = Client::new(ctx.home.sock_path()); - let value = client - .get(&format!("/v1/resources/{resource_id}")) - .await - .map_err(|error| read_resource_error(id, error))?; - check_version(&value)?; - check_read_resource(&value, id)?; - emit(ctx, value, Some(&resource_id.to_string()), None)?; - Ok(ExitCode::SUCCESS) -} - -async fn supervisor(ctx: &Ctx, command: SupervisorCommand) -> Result { - match command { - SupervisorCommand::Set { - resource_id, - machine, - thread, - expected_revision, - } => { - let id = resource_id_arg(resource_id)?; - if machine.as_uuid().is_nil() || thread.0.is_nil() { - return Err(AppError::Usage { - message: "supervisor machine and thread ids must not be nil".into(), - }); - } - let body = SupervisorReplacementBody { - api_version: API_VERSION, - expected_revision: ResourceRevision::new(expected_revision), - supervisor: SupervisorAddress { machine, thread }, - }; - let client = Client::new(ctx.home.sock_path()); - let value = post_mutation( - &client, - &format!("/v1/resources/{resource_id}/supervisor"), - &body, - id, - None, - &format!("expected revision {expected_revision} and supervisor {machine}/{thread}"), - ) - .await?; - check_mutation_resource( - &value, - id, - None, - &format!("expected revision {expected_revision} and supervisor {machine}/{thread}"), - )?; - emit( - ctx, - value, - Some(&resource_id.to_string()), - Some(&format!( - "supervisor set for resource {resource_id} at expected revision {expected_revision}" - )), - )?; - Ok(ExitCode::SUCCESS) - } - } -} - -async fn background(ctx: &Ctx, command: BackgroundCommand) -> Result { - match command { - BackgroundCommand::Submit { - resource_id, - request_id, - spec: path, - allow_other_thread, - } => { - let submit = SubmitCommand { - resource_uuid: resource_id, - request_uuid: request_id, - path: &path, - background: true, - allow_other_thread, - }; - submit_command(ctx, submit).await - } - BackgroundCommand::BindAttempt { - resource_id, - task_id, - attempt_spec, - } => bind_trainer_attempt(ctx, resource_id, task_id, &attempt_spec).await, - } -} - -async fn bind_trainer_attempt( - ctx: &Ctx, - resource_uuid: Uuid, - task_id: TaskId, - path: &str, -) -> Result { - let resource_id = resource_id_arg(resource_uuid)?; - validate_uuid("--task-id", task_id.0)?; - - let attempt_binding: AttemptBinding = load_json(path)?; - let retry_identity = trainer_attempt_retry_identity(resource_uuid, task_id, &attempt_binding)?; - let body = TrainerAttemptBody { - api_version: API_VERSION, - attempt_binding: attempt_binding.clone(), - }; - let endpoint = format!("/v1/resources/{resource_uuid}/background/{task_id}/trainer-attempt"); - let client = Client::new(ctx.home.sock_path()); - let response: TrainerAttemptResponse = client - .post_json(&endpoint, &body) - .await - .map_err(|error| trainer_attempt_error(resource_id, &retry_identity, error))?; - check_trainer_attempt_response( - &response, - resource_id, - task_id, - &attempt_binding, - &retry_identity, - )?; - - let value = serde_json::to_value(response)?; - let human = format!( - "trainer attempt for resource {resource_uuid} was bound to Homebased task {task_id}" - ); - emit(ctx, value, Some(&task_id.to_string()), Some(&human))?; - Ok(ExitCode::SUCCESS) -} - -fn trainer_attempt_retry_identity( - resource_uuid: Uuid, - task_id: TaskId, - attempt_binding: &AttemptBinding, -) -> Result { - let binding_json = serde_json::to_string(attempt_binding)?; - Ok(format!( - "retry with resource UUID {resource_uuid}, Homebased task UUID {task_id}, and the identical AttemptBinding JSON: {binding_json}" - )) -} - -fn trainer_attempt_error(resource: ResourceId, retry_identity: &str, error: AppError) -> AppError { - match error { - AppError::ResourceOutcomeUnknown { message, .. } => AppError::ResourceOutcomeUnknown { - resource, - operation: None, - message: format!("{message}; {retry_identity}"), - }, - AppError::Internal { message } - | AppError::DaemonUnavailable { message } - | AppError::MachineUnavailable { message, .. } - | AppError::ResourceAuthorityUnavailable { message, .. } - | AppError::RemoteSubmissionUnavailable { message } => AppError::ResourceOutcomeUnknown { - resource, - operation: None, - message: format!("{message}; {retry_identity}"), - }, - error @ AppError::ResourceLookupIncomplete { .. } => AppError::ResourceOutcomeUnknown { - resource, - operation: None, - message: format!("{error}; {retry_identity}"), - }, - error => error, - } -} - -fn check_trainer_attempt_response( - response: &TrainerAttemptResponse, - expected_resource: ResourceId, - expected_task: TaskId, - expected_binding: &AttemptBinding, - retry_identity: &str, -) -> Result<(), AppError> { - if response.api_version == API_VERSION - && response.resource.id == expected_resource - && response.task_id == expected_task - && response.attempt_binding == *expected_binding - { - return Ok(()); - } - - Err(AppError::ResourceOutcomeUnknown { - resource: expected_resource, - operation: None, - message: format!("the trainer-attempt response has a different identity; {retry_identity}"), - }) -} - -async fn request(ctx: &Ctx, command: RequestCommand) -> Result { - match command { - RequestCommand::Submit { - resource_id, - request_id, - spec: path, - allow_other_thread, - } => { - let submit = SubmitCommand { - resource_uuid: resource_id, - request_uuid: request_id, - path: &path, - background: false, - allow_other_thread, - }; - submit_command(ctx, submit).await - } - RequestCommand::Cancel { - resource_id, - request_id, - expected_revision, - operation_id, - } => { - let id = resource_id_arg(resource_id)?; - validate_uuid("REQUEST_ID", request_id)?; - validate_uuid("--operation-id", operation_id)?; - let body = RequestCancelBody { - api_version: API_VERSION, - operation_id, - expected_revision: ResourceRevision::new(expected_revision), - }; - let client = Client::new(ctx.home.sock_path()); - let path = format!("/v1/resources/{resource_id}/requests/{request_id}/cancel"); - post_operation(ctx, &client, id, operation_id, &path, &body).await - } - RequestCommand::Move { - resource_id, - request_id, - expected_revision, - operation_id, - placement, - } => { - let id = resource_id_arg(resource_id)?; - validate_uuid("REQUEST_ID", request_id)?; - validate_uuid("--operation-id", operation_id)?; - let body = ResourceActionBody { - api_version: API_VERSION, - expected_revision: ResourceRevision::new(expected_revision), - operation_id, - action: BrowserResourceAction::MoveQueued { - request_id: RequestId(request_id), - placement: placement.placement()?, - }, - }; - let client = Client::new(ctx.home.sock_path()); - let path = format!("/v1/resources/{resource_id}/actions"); - post_operation(ctx, &client, id, operation_id, &path, &body).await - } - } -} - -/// One `resource request submit` or `resource background submit` call -struct SubmitCommand<'a> { - resource_uuid: Uuid, - request_uuid: Uuid, - path: &'a str, - background: bool, - allow_other_thread: bool, -} - -async fn submit_command(ctx: &Ctx, submit: SubmitCommand<'_>) -> Result { - let SubmitCommand { - resource_uuid, - request_uuid, - path, - background, - allow_other_thread, - } = submit; - let resource_id = resource_id_arg(resource_uuid)?; - validate_uuid("--request-id", request_uuid)?; - let spec = load_resource_task_spec(path)?; - SubmitOrigin::capture(&ctx.home)?.check(spec.thread, allow_other_thread)?; - let callback_cwd = std::env::current_dir()?; - if !callback_cwd.is_absolute() { - return Err(AppError::Usage { - message: "the callback working directory must be absolute".into(), - }); - } - let request_id = RequestId(request_uuid); - let body = ResourceSubmitBody { - api_version: API_VERSION, - request_id, - spec, - env: TaskEnv::capture(), - callback_cwd, - }; - let client = Client::new(ctx.home.sock_path()); - if background { - let path = format!("/v1/resources/{resource_uuid}/background"); - let value = client - .post(&path, &serde_json::to_value(&body)?) - .await - .map_err(|error| { - mutation_error( - resource_id, - Some(request_uuid), - &format!("request id {request_uuid}"), - error, - ) - })?; - check_version(&value).map_err(|error| AppError::ResourceOutcomeUnknown { - resource: resource_id, - operation: Some(request_uuid), - message: format!( - "the background response is invalid: {error}; retry with the same request id {request_uuid}" - ), - })?; - let response: ResourceBackgroundSubmitResponse = - decode_value(value, "resource background response").map_err(|error| { - AppError::ResourceOutcomeUnknown { - resource: resource_id, - operation: Some(request_uuid), - message: format!( - "the background response is invalid: {error}; retry with the same request id {request_uuid}" - ), - } - })?; - check_background_response(&response, resource_id, request_id)?; - let human = match &response.outcome { - ResourceBackgroundSubmitOutcome::Inserted => format!( - "background launch {request_uuid} was inserted as task {}", - response.task_id - ), - ResourceBackgroundSubmitOutcome::Existing { .. } => format!( - "background launch {request_uuid} reused existing task {}", - response.task_id - ), - }; - emit( - ctx, - serde_json::to_value(response)?, - Some(&request_uuid.to_string()), - Some(&human), - )?; - return Ok(ExitCode::SUCCESS); - } - - let path = format!("/v1/resources/{resource_uuid}/requests"); - let value = client - .post(&path, &serde_json::to_value(&body)?) - .await - .map_err(|error| { - mutation_error( - resource_id, - Some(request_uuid), - &format!("request id {request_uuid}"), - error, - ) - })?; - check_version(&value).map_err(|error| AppError::ResourceOutcomeUnknown { - resource: resource_id, - operation: Some(request_uuid), - message: format!( - "the request response is invalid: {error}; retry with the same request id {request_uuid}" - ), - })?; - let response: ResourceRequestSubmitResponse = - decode_value(value, "resource request response").map_err(|error| { - AppError::ResourceOutcomeUnknown { - resource: resource_id, - operation: Some(request_uuid), - message: format!( - "the request response is invalid: {error}; retry with the same request id {request_uuid}" - ), - } - })?; - if response.request_id != request_id - || response.resource_id != resource_id - || response.task_id.0.is_nil() - || response.task_id.0 == request_uuid - || response.authority_machine.as_uuid().is_nil() - { - return Err(AppError::ResourceOutcomeUnknown { - resource: resource_id, - operation: Some(request_uuid), - message: format!( - "the authority returned another request identity; retry with the same request id {request_uuid}" - ), - }); - } - let human = match &response.outcome { - ResourceRequestSubmitOutcome::Waiting => { - format!( - "resource request {request_uuid} is waiting as task {}", - response.task_id - ) - } - ResourceRequestSubmitOutcome::Activated => { - format!( - "resource request {request_uuid} activated task {}", - response.task_id - ) - } - ResourceRequestSubmitOutcome::Rejected { reason } => { - return Err(AppError::SubmissionRejected { - request: request_id, - task: response.task_id, - reason: reason.clone(), - }); - } - ResourceRequestSubmitOutcome::CancelledBeforeLaunch => format!( - "resource request {request_uuid} was cancelled before task {} launched", - response.task_id - ), - }; - emit( - ctx, - serde_json::to_value(response)?, - Some(&request_uuid.to_string()), - Some(&human), - )?; - Ok(ExitCode::SUCCESS) -} - -fn load_resource_task_spec(path: &str) -> Result { - let spec = spec::load_spec(path)?; - let normalized = spec::normalize(&spec)?; - CommandSpec::try_from(normalized.clone()).map_err(|error| AppError::InvalidSpec { - pointer: match error { - CommandSpecError::ExplicitMachine => "/machine".into(), - CommandSpecError::AgentWorkload => "/workload/type".into(), - CommandSpecError::ContainerWithoutGpus => "/workload/gpus".into(), - }, - value: Value::Null, - message: error.to_string(), - })?; - Ok(normalized) -} - -async fn requests(ctx: &Ctx, resource_id: Uuid) -> Result { - let id = resource_id_arg(resource_id)?; - let client = Client::new(ctx.home.sock_path()); - let detail: ResourceDetail = client - .get_json(&format!("/v1/resources/{resource_id}")) - .await - .map_err(|error| read_resource_error(id, error))?; - if detail.api_version != API_VERSION { - return Err(invalid_daemon_response("unexpected resource API version")); - } - if detail.resource.id != id { - return Err(invalid_daemon_response( - "resource request list returned another resource identity", - )); - } - let value = json!({ - "api_version": detail.api_version, - "resource_id": detail.resource.id, - "requests": detail.requests, - }); - match ctx.output { - OutputMode::Json => emit(ctx, value, None, None)?, - OutputMode::Quiet => print_ids(value.get("requests"), "request_id")?, - OutputMode::Human => emit(ctx, value, None, None)?, - } - Ok(ExitCode::SUCCESS) -} - -async fn pending(ctx: &Ctx, machine: MachineId, thread: ThreadId) -> Result { - validate_uuid("--machine", machine.as_uuid())?; - validate_uuid("--thread", thread.0)?; - let client = Client::new(ctx.home.sock_path()); - let value = pending_value(&client, machine, thread).await?; - if ctx.output == OutputMode::Quiet { - let pending: PendingActionList = decode_value(value.clone(), "pending-action list")?; - if let Some(authority) = pending.unavailable_authorities.first() { - return Err(unavailable_authority(authority)); - } - } - match ctx.output { - OutputMode::Json => emit(ctx, value, None, None)?, - OutputMode::Quiet => print_ids(value.get("actions"), "action_id")?, - OutputMode::Human => emit(ctx, value, None, None)?, - } - Ok(ExitCode::SUCCESS) -} - -async fn pending_value( - client: &Client, - machine: MachineId, - thread: ThreadId, -) -> Result { - let path = format!("{RESOURCE_PENDING_PATH}?machine={machine}&thread={thread}"); - let pending: PendingActionList = client.get_json(&path).await?; - if pending.api_version != API_VERSION { - return Err(invalid_daemon_response( - "unexpected pending-action API version", - )); - } - let address = SupervisorAddress { machine, thread }; - if pending - .actions - .iter() - .any(|action| action.supervisor != address) - { - return Err(invalid_daemon_response( - "pending-action route returned an action for another supervisor address", - )); - } - serde_json::to_value(pending).map_err(Into::into) -} - -async fn release_watch( - ctx: &Ctx, - action_uuid: Uuid, - pending_path: &str, -) -> Result { - validate_uuid("ACTION_ID", action_uuid)?; - let client = Client::new(ctx.home.sock_path()); - let context = action_context_for_current_supervisor(&client, pending_path, |action| { - action.action_id.as_uuid() == action_uuid - }) - .await?; - let observed_task = match &context.phase { - PendingActionPhase::ReleaseRequired { - observed_background_task, - } => *observed_background_task, - _ => { - return Err(AppError::ResourceActionNotAllowed { - resource: context.resource_id, - message: "the action is not waiting for a background-task release".into(), - }); - } - }; - let request = ResourceActionSubmitRequest { - api_version: API_VERSION, - authority: context.authority, - choice: ResourceActionChoice::ReleaseWatcher { - observed_background_task: observed_task, - }, - }; - let response = submit_action(&client, request, context.resource_id, context.action_id).await?; - emit_action(ctx, response, context.resource_id, context.action_id) -} - -/// Parsed `resource return` flags: one decision, or a hold of the decision window -enum ReturnChoice<'a> { - Decide { - resume_path: Option<&'a str>, - no_resume: Option<&'a str>, - request_uuid: Option, - task_id: Option, - }, - Hold(Duration), -} - -async fn return_action( - ctx: &Ctx, - action_uuid: Uuid, - pending_path: &str, - choice: ReturnChoice<'_>, -) -> Result { - validate_uuid("ACTION_ID", action_uuid)?; - let client = Client::new(ctx.home.sock_path()); - let context = action_context_for_current_supervisor(&client, pending_path, |action| { - action.action_id.as_uuid() == action_uuid - }) - .await?; - if !matches!(&context.phase, PendingActionPhase::ReturnRequired) { - return Err(AppError::ResourceActionNotAllowed { - resource: context.resource_id, - message: "the action is not waiting for a return decision".into(), - }); - } - let return_context = - context - .return_context - .as_ref() - .ok_or_else(|| AppError::ResourceActionNotAllowed { - resource: context.resource_id, - message: "the pending return action has no return context".into(), - })?; - let choice = match choice { - ReturnChoice::Hold(hold) => ResourceActionChoice::HoldReturn { hold }, - ReturnChoice::Decide { - resume_path, - no_resume, - request_uuid, - task_id, - } => { - let decision = return_decision(resume_path, no_resume, request_uuid, task_id)?; - decision - .validate_for(return_context, context.authority.supervisor.thread) - .map_err(|error| return_decision_error(context.resource_id, error))?; - ResourceActionChoice::Return { decision } - } - }; - let request = ResourceActionSubmitRequest { - api_version: API_VERSION, - authority: context.authority, - choice, - }; - let response = submit_action(&client, request, context.resource_id, context.action_id).await?; - emit_action(ctx, response, context.resource_id, context.action_id) -} - -/// Build the one return decision that the `resource return` decision flags name -fn return_decision( - resume_path: Option<&str>, - no_resume: Option<&str>, - request_uuid: Option, - task_id: Option, -) -> Result { - match (resume_path, no_resume, request_uuid, task_id) { - (Some(path), None, Some(request), Some(task)) => { - validate_uuid("--request-id", request)?; - validate_uuid("--task-id", task.0)?; - let work: ReturnWork = load_json(path)?; - Ok(ReturnDecision::Launch(Box::new(ReturnLaunch { - request_id: RequestId(request), - task_id: task, - work, - }))) - } - (None, Some(reason), None, None) if !reason.trim().is_empty() => { - Ok(ReturnDecision::NoResume { - reason: reason.to_string(), - }) - } - _ => Err(AppError::Usage { - message: "choose exactly one of --resume-spec, --no-resume, or --hold; launch needs both --request-id and --task-id".into(), - }), - } -} - -async fn resolve( - ctx: &Ctx, - loan_uuid: Uuid, - pending_path: &str, - task_id: TaskId, - reason: String, -) -> Result { - validate_uuid("LOAN_ID", loan_uuid)?; - validate_uuid("--task-id", task_id.0)?; - if reason.trim().is_empty() { - return Err(AppError::Usage { - message: "--reason must not be empty".into(), - }); - } - let client = Client::new(ctx.home.sock_path()); - let context = action_context_for_current_supervisor(&client, pending_path, |action| { - action.loan_id.as_uuid() == loan_uuid - }) - .await?; - let expected_task = match &context.phase { - PendingActionPhase::Restoring { resume_task_id } => *resume_task_id, - _ => { - return Err(AppError::ResourceActionNotAllowed { - resource: context.resource_id, - message: format!("loan {loan_uuid} is not awaiting an ended-restore resolution"), - }); - } - }; - if expected_task != task_id { - return Err(AppError::ResourceActionNotAllowed { - resource: context.resource_id, - message: format!( - "task {task_id} does not match the restoring task {expected_task} for loan {loan_uuid}" - ), - }); - } - let request = ResourceActionSubmitRequest { - api_version: API_VERSION, - authority: context.authority, - choice: ResourceActionChoice::ResolveEndedRestore { task_id, reason }, - }; - let response = submit_action(&client, request, context.resource_id, context.action_id).await?; - emit_action(ctx, response, context.resource_id, context.action_id) -} - -async fn renotify( - ctx: &Ctx, - action_uuid: Uuid, - pending_path: &str, - operation_id: Uuid, -) -> Result { - validate_uuid("ACTION_ID", action_uuid)?; - validate_uuid("--operation-id", operation_id)?; - let client = Client::new(ctx.home.sock_path()); - let context = action_context_for_current_supervisor(&client, pending_path, |action| { - action.action_id.as_uuid() == action_uuid - }) - .await?; - let notice = context - .notice - .ok_or_else(|| AppError::ResourceOperationUnavailable { - resource: context.resource_id, - message: format!( - "pending action {} has no retryable supervisor notice", - context.action_id.as_uuid() - ), - })?; - if !matches!(¬ice.delivery, SupervisorNoticeDelivery::Failed { .. }) { - return Err(AppError::ResourceActionNotAllowed { - resource: context.resource_id, - message: "the supervisor notice has no failed delivery to retry".into(), - }); - } - let body = ResourceActionBody { - api_version: API_VERSION, - expected_revision: context.authority.expected_state_revision, - operation_id, - action: BrowserResourceAction::Renotify { - notice_id: notice.id, - }, - }; - let path = format!("/v1/resources/{}/actions", context.resource_id.as_uuid()); - post_operation( - ctx, - &client, - context.resource_id, - operation_id, - &path, - &body, - ) - .await -} - -async fn action_context_for_current_supervisor( - client: &Client, - pending_path: &str, - predicate: F, -) -> Result -where - F: Fn(&PendingActionView) -> bool, -{ - let (machine, thread) = current_supervisor_address(client).await?; - let pending: PendingActionList = load_json(pending_path)?; - if pending.api_version != API_VERSION { - return Err(invalid_daemon_response( - "unexpected pending-action API version", - )); - } - let matches: Vec<_> = pending - .actions - .iter() - .filter(|action| predicate(action)) - .collect(); - if matches.len() != 1 { - if let Some(authority) = pending.unavailable_authorities.first() { - return Err(unavailable_authority(authority)); - } - return Err(AppError::Usage { - message: if matches.is_empty() { - "no matching action is present in --pending-spec; save the pending --json output before mutation and reuse it for retries".into() - } else { - "more than one pending resource action matched; refresh `resource pending` and use an exact action id".into() - }, - }); - } - let action = matches[0]; - if action.supervisor.machine != machine || action.supervisor.thread != thread { - return Err(AppError::ResourceActionNotAllowed { - resource: action.resource_id, - message: "the pending action is not assigned to the current supervisor address".into(), - }); - } - action_context(action) -} - -/// Bind a pending action to the exact authority identity a decision must present -/// -/// Revisions are compare-and-set tokens that the authority checks, and a fresh -/// resource legitimately holds revision 0, so only nil identities are rejected here -fn action_context(action: &PendingActionView) -> Result { - if action.resource_id.as_uuid().is_nil() - || action.authority_machine.as_uuid().is_nil() - || action.loan_id.as_uuid().is_nil() - || action.action_id.as_uuid().is_nil() - || action.supervisor.machine.as_uuid().is_nil() - || action.supervisor.thread.0.is_nil() - { - return Err(invalid_daemon_response( - "pending action has an invalid identity", - )); - } - let authority = SupervisorActionAuthority { - authority_machine: action.authority_machine, - resource_id: action.resource_id, - loan_id: action.loan_id, - action_id: action.action_id, - expected_state_revision: action.state_revision, - supervisor: action.supervisor, - assignment_revision: action.assignment_revision, - }; - Ok(ActionContext { - resource_id: action.resource_id, - action_id: action.action_id, - authority, - phase: action.phase.clone(), - return_context: action.return_context.clone(), - notice: action.notice.clone(), - }) -} - -fn unavailable_authority(authority: &UnavailableAuthority) -> AppError { - AppError::MachineUnavailable { - machine: authority.machine, - message: authority.message.clone(), - } -} - -async fn current_supervisor_address(client: &Client) -> Result<(MachineId, ThreadId), AppError> { - let thread = THREAD_ENV_VARS - .into_iter() - .find_map(|name| std::env::var(name).ok()) - .ok_or_else(|| AppError::Usage { - message: "set CODEX_THREAD_ID, CODEX_SESSION_ID, or CLAUDE_CODE_SESSION_ID to use a pending resource action" - .into(), - })?; - let thread = ThreadId::from_str(&thread)?; - let machines: MachinesBody = client.get_json("/v1/fleet/machines").await?; - if machines.api_version != API_VERSION { - return Err(invalid_daemon_response("unexpected fleet API version")); - } - Ok((machines.local.machine, thread)) -} - -async fn submit_action( - client: &Client, - request: ResourceActionSubmitRequest, - resource_id: ResourceId, - action_id: ActionId, -) -> Result { - let operation_id = action_id.as_uuid(); - let retry_identity = match &request.choice { - ResourceActionChoice::ReleaseWatcher { .. } => { - format!("action id {operation_id}") - } - ResourceActionChoice::Return { - decision: ReturnDecision::Launch(launch), - } => format!( - "action id {operation_id}, request id {}, and task id {}", - launch.request_id.0, launch.task_id - ), - ResourceActionChoice::Return { - decision: ReturnDecision::NoResume { .. }, - } - | ResourceActionChoice::HoldReturn { .. } => format!("action id {operation_id}"), - ResourceActionChoice::ResolveEndedRestore { task_id, .. } => { - format!("action id {operation_id} and task id {task_id}") - } - }; - let response = client - .post_json::<_, ResourceActionSubmitResponse>(RESOURCE_ACTION_SUBMIT_PATH, &request) - .await - .map_err(|error| mutation_error(resource_id, Some(operation_id), &retry_identity, error))?; - if response.api_version != API_VERSION { - return Err(AppError::ResourceOutcomeUnknown { - resource: resource_id, - operation: Some(operation_id), - message: "the resource action response used an unexpected API version; retry with the same action id".into(), - }); - } - check_action_outcome(&request, &response.outcome).map_err(|message| { - AppError::ResourceOutcomeUnknown { - resource: resource_id, - operation: Some(operation_id), - message: format!("{message}; retry with the same {retry_identity}"), - } - })?; - Ok(response) -} - -/// Check that one action outcome names the exact authority, route owner, and task -/// -/// A co-located supervisor gets a local receipt and a remote supervisor gets the -/// authority's receipt, so each form is valid only for its own placement -fn check_action_outcome( - request: &ResourceActionSubmitRequest, - outcome: &ResourceActionSubmitOutcome, -) -> Result<(), &'static str> { - let expected = request.authority; - let co_located = expected.supervisor.machine == expected.authority_machine; - let (expected_kind, expected_task) = match &request.choice { - ResourceActionChoice::ReleaseWatcher { .. } => { - (Some(ResourceActionKind::ReleaseWatcher), None) - } - ResourceActionChoice::Return { - decision: ReturnDecision::Launch(launch), - } => ( - Some(ResourceActionKind::Return), - Some((launch.request_id, launch.task_id)), - ), - ResourceActionChoice::Return { - decision: ReturnDecision::NoResume { .. }, - } - | ResourceActionChoice::ResolveEndedRestore { .. } - | ResourceActionChoice::HoldReturn { .. } => (None, None), - }; - let same_loan = |loan: &crate::resource::Loan| { - loan.id == expected.loan_id && loan.resource_id == expected.resource_id - }; - match outcome { - ResourceActionSubmitOutcome::Accepted { receipt, .. } => { - if co_located - || receipt.authority != expected - || expected_kind != Some(receipt.kind) - || expected_task.is_some_and(|ids| ids != (receipt.request_id, receipt.task_id)) - { - return Err("the action response has a different authority or task kind"); - } - } - ResourceActionSubmitOutcome::LocalReturnAccepted { - receipt, - acceptance, - } => { - if !co_located - || receipt.authority != expected - || expected_task != Some((receipt.request_id, receipt.task_id)) - { - return Err("the local return response has a different authority or task"); - } - if let LocalReturnAcceptance::Inserted { loan, .. } = acceptance - && !same_loan(loan) - { - return Err("the local return response has a different loan identity"); - } - } - ResourceActionSubmitOutcome::Closed { loan, .. } => { - if expected_kind.is_some() - || matches!(request.choice, ResourceActionChoice::HoldReturn { .. }) - || !same_loan(loan) - { - return Err("the action response has a different loan identity"); - } - } - ResourceActionSubmitOutcome::ReturnHeld { window } => { - if !matches!(request.choice, ResourceActionChoice::HoldReturn { .. }) - || window.action_id() != expected.action_id - || window.loan_id() != expected.loan_id - || window.resource_id() != expected.resource_id - { - return Err("the hold response names a different action"); - } - } - ResourceActionSubmitOutcome::Rejected { .. } => {} - } - Ok(()) -} - -fn emit_action( - ctx: &Ctx, - response: ResourceActionSubmitResponse, - resource_id: ResourceId, - action_id: ActionId, -) -> Result { - let human = match &response.outcome { - ResourceActionSubmitOutcome::Accepted { .. } => format!( - "resource action {} was accepted by its authority", - action_id.as_uuid() - ), - ResourceActionSubmitOutcome::LocalReturnAccepted { - receipt, - acceptance: LocalReturnAcceptance::Inserted { .. }, - } => format!( - "resource action {} started return task {}", - action_id.as_uuid(), - receipt.task_id - ), - ResourceActionSubmitOutcome::LocalReturnAccepted { - receipt, - acceptance: LocalReturnAcceptance::Existing { state }, - } => format!( - "resource action {} already bound return task {} ({state:?}); nothing was started again", - action_id.as_uuid(), - receipt.task_id - ), - ResourceActionSubmitOutcome::Closed { .. } => { - format!("resource action {} closed its loan", action_id.as_uuid()) - } - ResourceActionSubmitOutcome::ReturnHeld { window } => format!( - "resource action {} holds queued requests until {} (limit {})", - action_id.as_uuid(), - window - .deadline_at() - .to_rfc3339_opts(SecondsFormat::Secs, true), - window.limit_at().to_rfc3339_opts(SecondsFormat::Secs, true) - ), - ResourceActionSubmitOutcome::Rejected { reason } => { - return Err(action_rejection(resource_id, action_id, reason)); - } - }; - let value = serde_json::to_value(response)?; - emit( - ctx, - value, - Some(&action_id.as_uuid().to_string()), - Some(&human), - )?; - Ok(ExitCode::SUCCESS) -} - -fn action_rejection( - resource: ResourceId, - action_id: ActionId, - rejection: &ResourceActionRejection, -) -> AppError { - let operation = Some(action_id.as_uuid()); - let details = format!("{rejection:?}"); - match rejection { - ResourceActionRejection::NotCurrentSupervisor - | ResourceActionRejection::ActionNotPending => AppError::ResourceActionNotAllowed { - resource, - message: details, - }, - ResourceActionRejection::StaleRevision { expected, actual } => { - AppError::ResourceStaleRevision { - resource, - expected: expected.get(), - current: actual.get(), - } - } - ResourceActionRejection::SpecMismatch - | ResourceActionRejection::IdentityConflict - | ResourceActionRejection::ConflictingRetry - | ResourceActionRejection::RouteEvidenceMismatch => AppError::ResourceOperationConflict { - resource, - operation, - message: details, - }, - ResourceActionRejection::RouteEvidenceMissing - | ResourceActionRejection::WatcherUnavailable { .. } - | ResourceActionRejection::DecisionRejected { .. } - | ResourceActionRejection::RestoreNotResolvable { .. } => { - AppError::ResourceOperationUnavailable { - resource, - message: details, - } - } - } -} - -/// Post one resource control keyed by its operation id and print the confirmed detail -async fn post_operation( - ctx: &Ctx, - client: &Client, - resource: ResourceId, - operation_id: Uuid, - path: &str, - body: &T, -) -> Result { - let retry_identity = format!("operation id {operation_id}"); - let value = post_mutation( - client, - path, - body, - resource, - Some(operation_id), - &retry_identity, - ) - .await?; - check_mutation_resource(&value, resource, Some(operation_id), &retry_identity)?; - emit(ctx, value, Some(&operation_id.to_string()), None)?; - Ok(ExitCode::SUCCESS) -} - -async fn post_mutation( - client: &Client, - path: &str, - body: &T, - resource: ResourceId, - operation: Option, - retry_identity: &str, -) -> Result { - let value = client - .post(path, &serde_json::to_value(body)?) - .await - .map_err(|error| mutation_error(resource, operation, retry_identity, error))?; - check_version(&value).map_err(|error| AppError::ResourceOutcomeUnknown { - resource, - operation, - message: format!( - "the mutation response is invalid: {error}; retry with the same {retry_identity}" - ), - })?; - Ok(value) -} - -fn return_decision_error(resource: ResourceId, error: ReturnDecisionRejection) -> AppError { - AppError::ResourceActionNotAllowed { - resource, - message: format!("return choice does not match the saved context: {error}"), - } -} - -fn mutation_error( - resource: ResourceId, - operation: Option, - retry_identity: &str, - error: AppError, -) -> AppError { - match error { - error @ AppError::ResourceNotFound { .. } - | error @ AppError::ResourceLookupIncomplete { .. } - | error @ AppError::ResourceAuthorityUnavailable { .. } - | error @ AppError::ResourceOutcomeUnknown { .. } - | error @ AppError::ResourceStaleRevision { .. } - | error @ AppError::ResourceOperationConflict { .. } - | error @ AppError::ResourceActionNotAllowed { .. } - | error @ AppError::ResourceOperationUnavailable { .. } - | error @ AppError::Usage { .. } - | error @ AppError::InvalidSpec { .. } - | error @ AppError::Permission { .. } => error, - AppError::Internal { message } => { - if let Some(error) = map_resource_error(resource, operation, &message) { - error - } else if message.starts_with("http 404 ") { - AppError::ResourceOperationUnavailable { - resource, - message: format!( - "the resource route or resource is unavailable (HTTP 404): {message}" - ), - } - } else { - AppError::ResourceOutcomeUnknown { - resource, - operation, - message: format!("{message}; retry with the same {retry_identity}"), - } - } - } - error @ AppError::DaemonUnavailable { .. } - | error @ AppError::MachineUnavailable { .. } - | error @ AppError::RemoteSubmissionUnavailable { .. } => { - AppError::ResourceOutcomeUnknown { - resource, - operation, - message: format!("{error}; retry with the same {retry_identity}"), - } - } - error => error, - } -} - -fn read_resource_error(resource: ResourceId, error: AppError) -> AppError { - match error { - AppError::Internal { message } => { - if let Some(error) = map_resource_error(resource, None, &message) { - error - } else if message.starts_with("http 404 ") { - AppError::ResourceOperationUnavailable { - resource, - message: format!("the resource route is unavailable (HTTP 404): {message}"), - } - } else { - AppError::Internal { message } - } - } - error => error, - } -} - -fn map_resource_error( - resource: ResourceId, - operation: Option, - message: &str, -) -> Option { - let message = message - .strip_prefix("http ") - .and_then(|response| response.split_once(' ')) - .filter(|(status, _)| status.parse::().is_ok()) - .map_or(message, |(_, body)| body); - if message.starts_with("resource not found:") { - return Some(AppError::ResourceNotFound { resource }); - } - if let Some(authority) = message.strip_prefix("resource authority ") { - let (machine, details) = authority.split_once(" unavailable:")?; - let machine = MachineId::from_str(machine).ok()?; - return Some(AppError::ResourceAuthorityUnavailable { - resource, - machine, - message: details.trim().to_owned(), - }); - } - if message.starts_with("resource lookup incomplete for ") { - return Some(AppError::ResourceOperationUnavailable { - resource, - message: message.into(), - }); - } - if message.starts_with("resource operation outcome unknown:") { - return Some(AppError::ResourceOutcomeUnknown { - resource, - operation, - message: message.into(), - }); - } - if let Some(revision) = message.strip_prefix("resource revision is stale: expected ") { - let (expected, current) = revision.split_once(", current ")?; - let expected = expected.parse().ok()?; - let current = current.parse().ok()?; - return Some(AppError::ResourceStaleRevision { - resource, - expected, - current, - }); - } - if message.starts_with("resource operation conflict:") { - return Some(AppError::ResourceOperationConflict { - resource, - operation, - message: message.into(), - }); - } - if message.starts_with("resource action not allowed:") { - return Some(AppError::ResourceActionNotAllowed { - resource, - message: message.into(), - }); - } - if message.starts_with("resource operation unavailable:") { - return Some(AppError::ResourceOperationUnavailable { - resource, - message: message.into(), - }); - } - None -} - -fn load_json(path: &str) -> Result { - let bytes = if path == "-" { - let mut bytes = Vec::new(); - io::stdin() - .read_to_end(&mut bytes) - .map_err(|error| AppError::Internal { - message: format!("failed to read resource spec from stdin: {error}"), - })?; - bytes - } else { - std::fs::read(path).map_err(|error| AppError::Internal { - message: format!("failed to read resource spec {path}: {error}"), - })? - }; - let value: Value = serde_json::from_slice(&bytes).map_err(|error| AppError::InvalidSpec { - pointer: String::new(), - value: Value::Null, - message: format!("invalid JSON in resource spec: {error}"), - })?; - decode_spec_value(&value) -} - -fn decode_spec_value(value: &Value) -> Result { - serde_path_to_error::deserialize(value).map_err(|error| AppError::InvalidSpec { - pointer: error.path().to_string(), - value: value.clone(), - message: error.inner().to_string(), - }) -} - -fn check_version(value: &Value) -> Result<(), AppError> { - if value.get("api_version").and_then(Value::as_u64) == Some(u64::from(API_VERSION)) { - return Ok(()); - } - Err(invalid_daemon_response("unexpected resource API version")) -} - -fn check_read_resource(value: &Value, expected: ResourceId) -> Result<(), AppError> { - let found = value - .pointer("/resource/id") - .and_then(Value::as_str) - .and_then(|id| Uuid::parse_str(id).ok()) - .and_then(|id| ResourceId::from_uuid(id).ok()); - if found == Some(expected) { - return Ok(()); - } - Err(invalid_daemon_response( - "resource detail returned a missing or different resource identity", - )) -} - -fn check_mutation_resource( - value: &Value, - expected: ResourceId, - operation: Option, - retry_identity: &str, -) -> Result<(), AppError> { - let found = value - .pointer("/resource/id") - .and_then(Value::as_str) - .and_then(|id| Uuid::parse_str(id).ok()) - .and_then(|id| ResourceId::from_uuid(id).ok()); - if found == Some(expected) { - return Ok(()); - } - Err(AppError::ResourceOutcomeUnknown { - resource: expected, - operation, - message: format!( - "the mutation response has a missing or different resource identity; retry with the same {retry_identity}" - ), - }) -} - -fn check_background_response( - response: &ResourceBackgroundSubmitResponse, - expected_resource: ResourceId, - expected_request: RequestId, -) -> Result<(), AppError> { - if response.api_version == API_VERSION - && response.request_id == expected_request - && response.resource.id == expected_resource - && !response.task_id.0.is_nil() - && response.task_id.0 != expected_request.0 - { - return Ok(()); - } - Err(AppError::ResourceOutcomeUnknown { - resource: expected_resource, - operation: Some(expected_request.0), - message: format!( - "the authority returned another background request identity; retry with the same request id {}", - expected_request.0 - ), - }) -} - -fn decode_value(value: Value, description: &str) -> Result { - serde_json::from_value(value) - .map_err(|error| invalid_daemon_response(&format!("invalid {description}: {error}"))) -} - -fn invalid_daemon_response(message: &str) -> AppError { - AppError::Internal { - message: message.into(), - } -} - -fn print_ids(values: Option<&Value>, field: &str) -> Result<(), AppError> { - let rows = values.and_then(Value::as_array).ok_or_else(|| { - invalid_daemon_response(format!("resource response has no {field} list").as_str()) - })?; - for row in rows { - let id = row - .get(field) - .and_then(Value::as_str) - .ok_or_else(|| invalid_daemon_response(&format!("resource row has no {field}")))?; - println!("{id}"); - } - Ok(()) -} - -fn emit( - ctx: &Ctx, - value: Value, - quiet_id: Option<&str>, - human: Option<&str>, -) -> Result<(), AppError> { - if let Some(output) = render_output(ctx.output, value, quiet_id, human)? { - println!("{output}"); - } - Ok(()) -} - -fn render_output( - mode: OutputMode, - value: Value, - quiet_id: Option<&str>, - human: Option<&str>, -) -> Result, AppError> { - match mode { - OutputMode::Json => { - let mut value = value; - if let Some(object) = value.as_object_mut() { - object - .entry("api_version") - .or_insert(Value::from(API_VERSION)); - } - Ok(Some(serde_json::to_string_pretty(&value)?)) - } - OutputMode::Quiet => Ok(quiet_id.map(str::to_string)), - OutputMode::Human => match human { - Some(human) => Ok(Some(human.to_string())), - None => Ok(Some(serde_json::to_string_pretty(&value)?)), - }, - } -} - -#[cfg(test)] -mod tests { - use crate::resource::operator_release::OperatorObservation; - use crate::resource::trainer_publication::AttemptBinding; - use crate::resource::{AssignmentRevision, Resource}; - use clap::Parser; - use serde_json::json; - - use super::{ - BackgroundCommand, RequestCommand, ResourceCommand, ResourceSubmitBody, SupervisorCommand, - action_context, check_action_outcome, check_background_response, - check_operator_release_response, check_trainer_attempt_response, decode_spec_value, - load_json, mutation_error, operator_gpu_free_attestation_schema, render_output, - resource_registration_schema, resource_return_work_schema, resource_task_schema, - resource_trainer_attempt_binding_schema, trainer_attempt_error, - trainer_attempt_retry_identity, - }; - use crate::cli::{Cli, Command, OutputMode}; - use crate::domain::{API_VERSION, TaskEnv, TaskId, ThreadId}; - use crate::error::AppError; - use crate::machine::MachineId; - use crate::resource::api::{ - BrowserResourceAction, OperatorReleaseResponse, PendingActionPhase, PendingActionView, - QueuePlacement, RequestCancelBody, ResourceActionBody, ResourceBackgroundSubmitOutcome, - ResourceBackgroundSubmitResponse, ResourceRegisterBody, ResourceRegistration, - SupervisorReplacementBody, TrainerAttemptResponse, - }; - use crate::resource::bound_action::{ - LocalReturnAcceptance, ResourceActionChoice, ResourceActionKind, - ResourceActionSubmitOutcome, ResourceActionSubmitRequest, - }; - use crate::resource::operator_release::OperatorGpuFreeAttestation; - use crate::resource::{ - ActionId, CommandSpec, ResourceId, ResourceRevision, ReturnDecision, ReturnLaunch, - ReturnWork, SupervisorActionAuthority, SupervisorAddress, - }; - use crate::spec::{self, NormalizedSpec}; - use crate::submission::RequestId; - use serde_json::Value; - use uuid::Uuid; - - fn uuid(value: &str) -> Uuid { - Uuid::parse_str(value).unwrap() - } - - fn operator_attestation() -> OperatorGpuFreeAttestation { - OperatorGpuFreeAttestation { - operation_id: crate::resource::operator_release::OperatorAttestationId::new(), - resource_id: ResourceId::new(), - authority_machine: MachineId::new(), - task_id: TaskId::new(), - expected_state_revision: ResourceRevision::new(5), - state_binding: crate::resource::operator_release::OperatorStateBinding::NoLoan, - observation: OperatorObservation::try_from( - "checked the authority GPU and found no trainer process".to_owned(), - ) - .unwrap(), - confirmation: - crate::resource::operator_release::OperatorGpuFreeConfirmation::OperatorConfirmedGpuFree, - } - } - - fn operator_release_response( - attestation: OperatorGpuFreeAttestation, - api_version: u32, - replayed: bool, - ) -> OperatorReleaseResponse { - OperatorReleaseResponse { - api_version, - receipt: crate::resource::operator_release::OperatorGpuFreeReceipt { - attestation, - evidence: crate::resource::operator_release::OperatorGpuFreeEvidence { - trainer_end: crate::resource::operator_release::AttestedTrainerEnd::Lost, - trainer_launch: - crate::resource::operator_release::AttestedTrainerLaunch::FirstBackgroundLaunch { - request_id: RequestId::new(), - }, - normalized_spec_sha256: serde_json::from_value(json!("42".repeat(32))) - .unwrap(), - trainer_association: - crate::resource::operator_release::AttestedTrainerAssociation::Missing, - }, - state_revision: ResourceRevision::new(6), - outcome: crate::resource::operator_release::OperatorGpuFreeOutcome::IdleBoundary, - }, - replayed, - } - } - - #[test] - fn operator_release_cli_parses_a_complete_spec_path() { - let cli = Cli::try_parse_from([ - "homebased", - "resource", - "operator-release", - "--spec", - "attestation.json", - ]) - .unwrap(); - let Command::Resource { - command: ResourceCommand::OperatorRelease { spec }, - } = cli.command - else { - panic!("operator-release must keep its document path"); - }; - assert_eq!(spec, "attestation.json"); - } - - #[test] - fn operator_release_document_is_strict_and_keeps_retry_identity_and_observation() { - let attestation = operator_attestation(); - let value = serde_json::to_value(&attestation).unwrap(); - let directory = tempfile::tempdir().unwrap(); - let path = directory.path().join("attestation.json"); - std::fs::write(&path, serde_json::to_vec(&value).unwrap()).unwrap(); - let decoded: OperatorGpuFreeAttestation = load_json(path.to_str().unwrap()).unwrap(); - assert_eq!(decoded.operation_id, attestation.operation_id); - assert_eq!(decoded.observation, attestation.observation); - assert!(decoded.validate().is_ok()); - - let mut unknown_field = value.clone(); - unknown_field["unexpected"] = json!(true); - assert!(decode_spec_value::(&unknown_field).is_err()); - - let mut missing_confirmation = value.clone(); - missing_confirmation - .as_object_mut() - .unwrap() - .remove("confirmation"); - assert!(decode_spec_value::(&missing_confirmation).is_err()); - - let mut invalid_confirmation = value; - invalid_confirmation["confirmation"] = json!("confirmed"); - assert!(decode_spec_value::(&invalid_confirmation).is_err()); - } - - #[test] - fn operator_release_schema_lists_exactly_the_bindings_the_authority_decodes() { - use crate::resource::operator_release::OperatorStateBinding; - - let schema = serde_json::to_value(operator_gpu_free_attestation_schema()).unwrap(); - let variants = schema - .pointer("/properties/state_binding/oneOf") - .and_then(Value::as_array) - .unwrap(); - let (loan, action, request) = (Uuid::now_v7(), Uuid::now_v7(), Uuid::now_v7()); - let mut documented = Vec::new(); - for variant in variants { - let tag = variant - .pointer("/properties/type/const") - .and_then(Value::as_str) - .unwrap(); - documented.push(tag.to_owned()); - // build the document from the schema's own required fields - let mut binding = json!({ "type": tag }); - for field in variant["required"].as_array().unwrap() { - let field = field.as_str().unwrap(); - let value = match field { - "type" => continue, - "loan_id" => loan, - "action_id" => action, - "request_id" => request, - other => panic!("unexpected binding field {other}"), - }; - binding[field] = json!(value); - } - let mut document = serde_json::to_value(operator_attestation()).unwrap(); - document["state_binding"] = binding.clone(); - let decoded = decode_spec_value::(&document).unwrap(); - assert!(decoded.validate().is_ok(), "{tag}"); - assert_eq!( - serde_json::to_value(decoded.state_binding).unwrap(), - binding - ); - - // every binding refuses extra fields, so a typo cannot drop an identity - binding["unexpected"] = json!(true); - document["state_binding"] = binding; - assert!( - decode_spec_value::(&document).is_err(), - "{tag}" - ); - } - assert_eq!( - documented, - [ - "no_loan", - "awaiting_release", - "first_background_launch", - "restoring_return", - "restoring_foreground_return" - ] - ); - - // a nil launch or restore identity is refused before any request is sent - let mut attestation = operator_attestation(); - attestation.state_binding = OperatorStateBinding::FirstBackgroundLaunch { - request_id: RequestId(Uuid::nil()), - }; - assert!(attestation.validate().is_err()); - for binding in [ - json!({"type": "restoring_return", "loan_id": Uuid::nil(), "action_id": action}), - json!({"type": "restoring_foreground_return", "loan_id": loan, "action_id": Uuid::nil()}), - ] { - let mut document = serde_json::to_value(operator_attestation()).unwrap(); - document["state_binding"] = binding; - assert!(decode_spec_value::(&document).is_err()); - } - } - - #[test] - fn operator_release_response_requires_the_saved_document_and_version() { - let attestation = operator_attestation(); - let response = operator_release_response(attestation.clone(), API_VERSION, true); - assert!(check_operator_release_response(&response, &attestation, "retry command").is_ok()); - let response_value = serde_json::to_value(&response).unwrap(); - assert_eq!(response_value["api_version"], API_VERSION); - assert_eq!(response_value["replayed"], true); - - let mismatched_document = OperatorGpuFreeAttestation { - observation: OperatorObservation::try_from("changed observation".to_owned()).unwrap(), - ..attestation.clone() - }; - let mismatch = - check_operator_release_response(&response, &mismatched_document, "retry command") - .unwrap_err(); - assert_eq!(mismatch.code(), "resource_outcome_unknown"); - assert!(mismatch.to_string().contains("retry command")); - assert!(matches!( - mismatch, - AppError::ResourceOutcomeUnknown { - operation: Some(operation), - .. - } if operation == mismatched_document.operation_id.as_uuid() - )); - - let wrong_version = operator_release_response(attestation.clone(), API_VERSION + 1, false); - assert_eq!( - check_operator_release_response(&wrong_version, &attestation, "retry command") - .unwrap_err() - .code(), - "resource_outcome_unknown" - ); - - let mut unexpected = response_value; - unexpected["unexpected"] = json!(true); - assert!(serde_json::from_value::(unexpected).is_err()); - } - - #[test] - fn resource_commands_parse_with_distinct_identity_arguments() { - let resource_id = uuid("019b4f42-0000-7000-8000-000000000001"); - let request_id = uuid("019b4f42-0000-7000-8000-000000000002"); - let operation_id = uuid("019b4f42-0000-7000-8000-000000000003"); - let cli = Cli::try_parse_from([ - "homebased", - "resource", - "request", - "cancel", - &resource_id.to_string(), - &request_id.to_string(), - "--expected-revision", - "4", - "--operation-id", - &operation_id.to_string(), - ]) - .unwrap(); - let Command::Resource { - command: - ResourceCommand::Request { - command: - RequestCommand::Cancel { - resource_id: parsed_resource, - request_id: parsed_request, - operation_id: parsed_operation, - .. - }, - }, - } = cli.command - else { - panic!("resource request cancel must parse to its typed command"); - }; - assert_eq!(parsed_resource, resource_id); - assert_eq!(parsed_request, request_id); - assert_eq!(parsed_operation, operation_id); - } - - #[test] - fn resource_request_move_requires_one_placement() { - let resource_id = "019b4f42-0000-7000-8000-000000000011"; - let request_id = "019b4f42-0000-7000-8000-000000000012"; - let anchor_id = "019b4f42-0000-7000-8000-000000000013"; - let operation_id = "019b4f42-0000-7000-8000-000000000014"; - let parse = |args: &[&str]| { - Cli::try_parse_from( - [ - "homebased", - "resource", - "request", - "move", - resource_id, - request_id, - "--operation-id", - operation_id, - ] - .into_iter() - .chain(args.iter().copied()), - ) - }; - let placement = |args: &[&str]| { - let Command::Resource { - command: - ResourceCommand::Request { - command: RequestCommand::Move { placement, .. }, - }, - } = parse(args).unwrap().command - else { - panic!("resource request move must parse to its typed command"); - }; - placement.placement().unwrap() - }; - assert_eq!( - placement(&["--expected-revision", "0", "--before", anchor_id]), - QueuePlacement::Before { - request_id: RequestId(uuid(anchor_id)), - } - ); - assert_eq!( - placement(&["--expected-revision", "4", "--after", anchor_id]), - QueuePlacement::After { - request_id: RequestId(uuid(anchor_id)), - } - ); - assert_eq!( - placement(&["--expected-revision", "4", "--front"]), - QueuePlacement::Front - ); - assert_eq!( - placement(&["--expected-revision", "4", "--back"]), - QueuePlacement::Back - ); - - assert!( - parse(&["--expected-revision", "4"]).is_err(), - "a placement is required" - ); - assert!( - parse(&["--expected-revision", "4", "--front", "--back"]).is_err(), - "two placements must be refused" - ); - } - - #[test] - fn resource_supervisor_set_accepts_zero_expected_revision() { - let cli = Cli::try_parse_from([ - "homebased", - "resource", - "supervisor", - "set", - "019b4f42-0000-7000-8000-000000000021", - "--machine", - "019b4f42-0000-7000-8000-000000000022", - "--thread", - "019b4f42-0000-7000-8000-000000000023", - "--expected-revision", - "0", - ]) - .unwrap(); - - let Command::Resource { - command: - ResourceCommand::Supervisor { - command: - SupervisorCommand::Set { - expected_revision, .. - }, - }, - } = cli.command - else { - panic!("resource supervisor set must parse to its typed command"); - }; - - assert_eq!(expected_revision, 0); - } - - #[test] - fn resource_output_flags_parse_and_conflict() { - let resource = "019b4f42-0000-7000-8000-000000000021"; - let json = - Cli::try_parse_from(["homebased", "--json", "resource", "show", resource]).unwrap(); - assert!(json.json); - assert!(!json.quiet); - - assert!( - Cli::try_parse_from([ - "homebased", - "--json", - "--quiet", - "resource", - "show", - resource, - ]) - .is_err() - ); - } - - #[test] - fn resource_request_submit_requires_and_preserves_the_request_id() { - let resource = "019b4f42-0000-7000-8000-000000000011"; - let request = "019b4f42-0000-7000-8000-000000000012"; - let cli = Cli::try_parse_from([ - "homebased", - "resource", - "request", - "submit", - resource, - "--request-id", - request, - "--spec", - "task.json", - ]) - .unwrap(); - let Command::Resource { - command: - ResourceCommand::Request { - command: - RequestCommand::Submit { - resource_id: parsed_resource, - request_id: parsed_request, - .. - }, - }, - } = cli.command - else { - panic!("resource request submit must parse to its typed command"); - }; - assert_eq!(parsed_resource, uuid(resource)); - assert_eq!(parsed_request, uuid(request)); - assert!( - Cli::try_parse_from([ - "homebased", - "resource", - "request", - "submit", - resource, - "--spec", - "task.json", - ]) - .is_err() - ); - } - - #[test] - fn background_submit_requires_and_preserves_its_request_id() { - let resource = "019b4f42-0000-7000-8000-000000000041"; - let request = "019b4f42-0000-7000-8000-000000000042"; - let cli = Cli::try_parse_from([ - "homebased", - "resource", - "background", - "submit", - resource, - "--request-id", - request, - "--spec", - "trainer.json", - ]) - .unwrap(); - let Command::Resource { - command: - ResourceCommand::Background { - command: - BackgroundCommand::Submit { - resource_id, - request_id, - .. - }, - }, - } = cli.command - else { - panic!("resource background submit must preserve its request identity"); - }; - assert_eq!(resource_id, uuid(resource)); - assert_eq!(request_id, uuid(request)); - assert!( - Cli::try_parse_from([ - "homebased", - "resource", - "background", - "submit", - resource, - "--spec", - "trainer.json", - ]) - .is_err() - ); - } - - #[test] - fn bind_attempt_keeps_homebased_and_trainer_task_ids_separate() { - let resource = "019b4f42-0000-7000-8000-000000000051"; - let homebased_task = "019b4f42-0000-7000-8000-000000000052"; - let binding_value = json!({ - "campaign_id": "campaign-1", - "campaign_revision_id": "revision-1", - "task_id": "trainer-task-1", - "attempt_id": "attempt-1", - "attempt_number": 1, - "ownership_token": "ownership-1" - }); - let cli = Cli::try_parse_from([ - "homebased", - "resource", - "background", - "bind-attempt", - resource, - "--task-id", - homebased_task, - "--attempt-spec", - "attempt.json", - ]) - .unwrap(); - let Command::Resource { - command: - ResourceCommand::Background { - command: - BackgroundCommand::BindAttempt { - resource_id, - task_id, - attempt_spec, - }, - }, - } = cli.command - else { - panic!("bind-attempt must preserve its distinct resource and task identities"); - }; - let binding: AttemptBinding = decode_spec_value(&binding_value).unwrap(); - - assert_eq!(resource_id, uuid(resource)); - assert_eq!(task_id, TaskId(uuid(homebased_task))); - assert_eq!(attempt_spec, "attempt.json"); - assert_eq!(binding.task_id, "trainer-task-1"); - assert_ne!(binding.task_id, task_id.to_string()); - } - - #[test] - fn trainer_attempt_response_requires_exact_version_and_identities() { - let resource_id = - ResourceId::from_uuid(uuid("019b4f42-0000-7000-8000-000000000061")).unwrap(); - let other_resource = - ResourceId::from_uuid(uuid("019b4f42-0000-7000-8000-000000000062")).unwrap(); - let task_id = TaskId(uuid("019b4f42-0000-7000-8000-000000000063")); - let other_task = TaskId(uuid("019b4f42-0000-7000-8000-000000000064")); - let binding = test_attempt_binding(); - let retry_identity = - trainer_attempt_retry_identity(resource_id.as_uuid(), task_id, &binding).unwrap(); - - let accepted = trainer_attempt_response(API_VERSION, resource_id, task_id, binding.clone()); - assert!( - check_trainer_attempt_response( - &accepted, - resource_id, - task_id, - &binding, - &retry_identity, - ) - .is_ok() - ); - - let different_binding = AttemptBinding { - attempt_id: "attempt-2".into(), - ..binding.clone() - }; - let mismatches = [ - trainer_attempt_response(API_VERSION, other_resource, task_id, binding.clone()), - trainer_attempt_response(API_VERSION, resource_id, other_task, binding.clone()), - trainer_attempt_response(API_VERSION, resource_id, task_id, different_binding), - trainer_attempt_response(API_VERSION + 1, resource_id, task_id, binding.clone()), - ]; - for response in mismatches { - let error = check_trainer_attempt_response( - &response, - resource_id, - task_id, - &binding, - &retry_identity, - ) - .unwrap_err(); - let AppError::ResourceOutcomeUnknown { - resource, - operation, - message, - } = error - else { - panic!("a mismatched trainer-attempt response must be unknown"); - }; - assert_eq!(resource, resource_id); - assert_eq!(operation, None); - assert!(message.contains(&resource_id.as_uuid().to_string())); - assert!(message.contains(&task_id.to_string())); - assert!(message.contains(&serde_json::to_string(&binding).unwrap())); - } - } - - #[test] - fn trainer_attempt_binding_json_rejects_unknown_fields_and_invalid_identifiers() { - let binding = json!({ - "campaign_id": "campaign-1", - "campaign_revision_id": "revision-1", - "task_id": "trainer-task-1", - "attempt_id": "attempt-1", - "attempt_number": 1, - "ownership_token": "ownership-1" - }); - assert!(decode_spec_value::(&binding).is_ok()); - - let mut unknown_field = binding.clone(); - unknown_field["unexpected"] = json!(true); - assert!(decode_spec_value::(&unknown_field).is_err()); - - let mut invalid_identifier = binding; - invalid_identifier["task_id"] = json!("Trainer-Task-1"); - assert!(decode_spec_value::(&invalid_identifier).is_err()); - } - - #[test] - fn trainer_attempt_transport_error_recommends_the_exact_retry_identity() { - let resource_uuid = uuid("019b4f42-0000-7000-8000-000000000071"); - let resource_id = ResourceId::from_uuid(resource_uuid).unwrap(); - let task_id = TaskId(uuid("019b4f42-0000-7000-8000-000000000072")); - let binding = test_attempt_binding(); - let retry_identity = - trainer_attempt_retry_identity(resource_uuid, task_id, &binding).unwrap(); - let error = trainer_attempt_error( - resource_id, - &retry_identity, - AppError::DaemonUnavailable { - message: "socket closed".into(), - }, - ); - let AppError::ResourceOutcomeUnknown { - resource, - operation, - message, - } = error - else { - panic!("a transport error must be an unknown outcome"); - }; - assert_eq!(resource, resource_id); - assert_eq!(operation, None); - assert!(message.contains(&resource_uuid.to_string())); - assert!(message.contains(&task_id.to_string())); - assert!(message.contains(&serde_json::to_string(&binding).unwrap())); - - let rejection = AppError::ResourceActionNotAllowed { - resource: resource_id, - message: "task is not registered".into(), - }; - assert!(matches!( - trainer_attempt_error(resource_id, &retry_identity, rejection), - AppError::ResourceActionNotAllowed { .. } - )); - } - - #[test] - fn return_launch_requires_a_single_choice_and_both_stable_ids() { - let action = "019b4f42-0000-7000-8000-000000000004"; - assert!(Cli::try_parse_from(["homebased", "resource", "release-watch", action,]).is_err()); - - let no_choice = Cli::try_parse_from(["homebased", "resource", "return", action]); - assert!(no_choice.is_err()); - - let both_choices = Cli::try_parse_from([ - "homebased", - "resource", - "return", - action, - "--pending-spec", - "pending.json", - "--resume-spec", - "work.json", - "--no-resume", - "not now", - "--request-id", - "019b4f42-0000-7000-8000-000000000005", - "--task-id", - "019b4f42-0000-7000-8000-000000000006", - ]); - assert!(both_choices.is_err()); - - let missing_ids = Cli::try_parse_from([ - "homebased", - "resource", - "return", - action, - "--pending-spec", - "pending.json", - "--resume-spec", - "work.json", - ]); - assert!(missing_ids.is_err()); - - let request_id = "019b4f42-0000-7000-8000-000000000025"; - let task_id = "019b4f42-0000-7000-8000-000000000026"; - let launch = Cli::try_parse_from([ - "homebased", - "resource", - "return", - action, - "--pending-spec", - "pending.json", - "--resume-spec", - "work.json", - "--request-id", - request_id, - "--task-id", - task_id, - ]) - .unwrap(); - let Command::Resource { - command: - ResourceCommand::Return { - action_id, - request_id: Some(parsed_request), - task_id: Some(parsed_task), - .. - }, - } = launch.command - else { - panic!("resource return must preserve all launch identities"); - }; - assert_eq!(action_id, uuid(action)); - assert_eq!(parsed_request, uuid(request_id)); - assert_eq!(parsed_task, TaskId(uuid(task_id))); - - let hold = Cli::try_parse_from([ - "homebased", - "resource", - "return", - action, - "--pending-spec", - "pending.json", - "--hold", - "5m", - ]) - .unwrap(); - let Command::Resource { - command: - ResourceCommand::Return { - hold: Some(parsed_hold), - no_resume: None, - resume_spec: None, - .. - }, - } = hold.command - else { - panic!("resource return --hold must parse as a hold without a decision"); - }; - assert_eq!(parsed_hold, std::time::Duration::from_secs(5 * 60)); - let hold_and_decision = Cli::try_parse_from([ - "homebased", - "resource", - "return", - action, - "--pending-spec", - "pending.json", - "--hold", - "5m", - "--no-resume", - "not now", - ]); - assert!(hold_and_decision.is_err()); - - let loan_id = "019b4f42-0000-7000-8000-000000000027"; - let task_id = "019b4f42-0000-7000-8000-000000000028"; - let resolve = Cli::try_parse_from([ - "homebased", - "resource", - "resolve", - loan_id, - "--pending-spec", - "pending.json", - "--task-id", - task_id, - "--reason", - "confirmed ended", - ]) - .unwrap(); - let Command::Resource { - command: - ResourceCommand::Resolve { - loan_id: parsed_loan, - task_id: parsed_task, - .. - }, - } = resolve.command - else { - panic!("resource resolve must keep loan and task identities separate"); - }; - assert_eq!(parsed_loan, uuid(loan_id)); - assert_eq!(parsed_task, TaskId(uuid(task_id))); - } - - #[test] - fn resource_schema_separates_registration_and_command_specs() { - let schema = json!({ - "$defs": { - "ResourceRegistrationSpec": resource_registration_schema(), - "ResourceTaskSubmitSpec": resource_task_schema().unwrap(), - "ResourceReturnWorkSpec": resource_return_work_schema().unwrap(), - "ResourceTrainerAttemptBinding": resource_trainer_attempt_binding_schema(), - "OperatorGpuFreeAttestation": operator_gpu_free_attestation_schema(), - } - }); - assert_eq!( - schema - .pointer("/$defs/ResourceRegistrationSpec/required") - .unwrap(), - &json!(["id", "display_name", "supervisor"]) - ); - assert_eq!( - schema - .pointer( - "/$defs/ResourceTaskSubmitSpec/properties/workload/oneOf/0/properties/type/const" - ) - .unwrap(), - "task" - ); - assert_eq!( - schema - .pointer( - "/$defs/ResourceTaskSubmitSpec/properties/workload/oneOf/1/properties/type/const" - ) - .unwrap(), - "container" - ); - assert!( - schema - .pointer("/$defs/ResourceTaskSubmitSpec/properties/workload/oneOf/1/required") - .and_then(Value::as_array) - .unwrap() - .contains(&json!("gpus")), - "resource containers must name their gpus" - ); - assert!( - schema - .pointer("/$defs/ResourceTaskSubmitSpec/properties/machine") - .is_none() - ); - assert_eq!( - schema - .pointer("/$defs/ResourceReturnWorkSpec/oneOf/0/properties/type/const") - .unwrap(), - "same_run_resume" - ); - assert_eq!( - schema - .pointer("/$defs/ResourceTrainerAttemptBinding/properties/task_id/description") - .unwrap(), - "Trainer task identity, not a Homebased TaskId" - ); - assert_eq!( - schema - .pointer("/$defs/ResourceTrainerAttemptBinding/additionalProperties") - .unwrap(), - &json!(false) - ); - assert_eq!( - schema - .pointer("/$defs/OperatorGpuFreeAttestation/required") - .unwrap(), - &json!([ - "operation_id", - "resource_id", - "authority_machine", - "task_id", - "expected_state_revision", - "state_binding", - "observation", - "confirmation" - ]) - ); - assert_eq!( - schema - .pointer("/$defs/OperatorGpuFreeAttestation/additionalProperties") - .unwrap(), - &json!(false) - ); - } - - fn test_attempt_binding() -> AttemptBinding { - AttemptBinding { - campaign_id: "campaign-1".into(), - campaign_revision_id: "revision-1".into(), - task_id: "trainer-task-1".into(), - attempt_id: "attempt-1".into(), - attempt_number: 1, - ownership_token: "ownership-1".into(), - } - } - - fn trainer_attempt_response( - api_version: u32, - resource_id: ResourceId, - task_id: TaskId, - attempt_binding: AttemptBinding, - ) -> TrainerAttemptResponse { - let machine = MachineId::from_uuid(uuid("019b4f42-0000-7000-8000-000000000065")); - TrainerAttemptResponse { - api_version, - resource: Resource::new( - resource_id, - "gpu-a".into(), - machine, - SupervisorAddress { - machine, - thread: ThreadId(uuid("019b4f42-0000-7000-8000-000000000066")), - }, - AssignmentRevision::new(1), - ResourceRevision::new(1), - Some(task_id), - ), - task_id, - runtime_root: "/runtime".into(), - attempt_binding, - } - } - - #[test] - fn cancel_body_uses_operation_id_and_revision_from_the_contract() { - let request_id = uuid("019b4f42-0000-7000-8000-000000000007"); - let body = RequestCancelBody { - api_version: API_VERSION, - operation_id: request_id, - expected_revision: ResourceRevision::new(8), - }; - let value = serde_json::to_value(body).unwrap(); - assert_eq!(value["api_version"], API_VERSION); - assert_eq!(value["operation_id"], request_id.to_string()); - assert_eq!(value["expected_revision"], 8); - assert!(value.get("request_id").is_none()); - } - - #[test] - fn supervisor_replacement_body_uses_the_compare_and_set_revision() { - let body = SupervisorReplacementBody { - api_version: API_VERSION, - expected_revision: ResourceRevision::new(12), - supervisor: SupervisorAddress { - machine: MachineId::from_uuid(uuid("019b4f42-0000-7000-8000-000000000031")), - thread: ThreadId(uuid("019b4f42-0000-7000-8000-000000000032")), - }, - }; - let value = serde_json::to_value(body).unwrap(); - assert_eq!(value["api_version"], API_VERSION); - assert_eq!(value["expected_revision"], 12); - assert_eq!( - value["supervisor"]["machine"], - "019b4f42-0000-7000-8000-000000000031" - ); - assert!(value.get("operation_id").is_none()); - } - - #[test] - fn registration_and_renotify_envelopes_keep_contract_fields() { - let id = uuid("019b4f42-0000-7000-8000-000000000013"); - let body = ResourceRegisterBody { - api_version: API_VERSION, - spec: ResourceRegistration { - id: ResourceId::from_uuid(id).unwrap(), - display_name: "gpu-a".into(), - supervisor: SupervisorAddress { - machine: MachineId::from_uuid(uuid("019b4f42-0000-7000-8000-000000000014")), - thread: ThreadId(uuid("019b4f42-0000-7000-8000-000000000015")), - }, - }, - }; - let registration = serde_json::to_value(body).unwrap(); - assert_eq!(registration["api_version"], API_VERSION); - assert_eq!(registration["spec"]["id"], id.to_string()); - assert!(registration["spec"].get("authority_machine").is_none()); - - let operation_id = uuid("019b4f42-0000-7000-8000-000000000016"); - let body = ResourceActionBody { - api_version: API_VERSION, - expected_revision: ResourceRevision::new(4), - operation_id, - action: BrowserResourceAction::Renotify { - notice_id: crate::resource::NoticeId::from_uuid(uuid( - "019b4f42-0000-7000-8000-000000000017", - )) - .unwrap(), - }, - }; - let action = serde_json::to_value(body).unwrap(); - assert_eq!(action["operation_id"], operation_id.to_string()); - assert_eq!(action["expected_revision"], 4); - assert_eq!(action["action"]["type"], "renotify"); - } - - #[test] - fn submit_body_keeps_the_same_request_id_and_contract_envelope() { - let request_id = uuid("019b4f42-0000-7000-8000-00000000000a"); - let temp = tempfile::tempdir().unwrap(); - let source = json!({ - "api_version": API_VERSION, - "thread": "019b4f42-0000-7000-8000-00000000000b", - "name": "resource task", - "cwd": temp.path(), - "timeout": "1h", - "workload": { "type": "task", "command": ["true"] } - }); - let parsed = spec::parse_spec_value(&source).unwrap(); - let normalized = spec::normalize(&parsed).unwrap(); - let body = ResourceSubmitBody { - api_version: API_VERSION, - request_id: RequestId(request_id), - spec: normalized, - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - callback_cwd: temp.path().to_path_buf(), - }; - let value = serde_json::to_value(body).unwrap(); - assert_eq!(value["api_version"], API_VERSION); - assert_eq!(value["request_id"], request_id.to_string()); - assert_eq!(value["spec"]["workload"]["type"], "task"); - assert!(value.get("resource_id").is_none()); - assert_eq!(value["env"]["path"], "/bin"); - assert!(value["callback_cwd"].is_string()); - } - - #[test] - fn background_response_keeps_resource_request_and_task_identities_separate() { - let resource_id = - ResourceId::from_uuid(uuid("019b4f42-0000-7000-8000-000000000043")).unwrap(); - let request_id = RequestId(uuid("019b4f42-0000-7000-8000-000000000044")); - let task_id = TaskId(uuid("019b4f42-0000-7000-8000-000000000045")); - let machine = MachineId::from_uuid(uuid("019b4f42-0000-7000-8000-000000000046")); - let response = ResourceBackgroundSubmitResponse { - api_version: API_VERSION, - request_id, - task_id, - resource: Resource::new( - resource_id, - "gpu-a".into(), - machine, - SupervisorAddress { - machine, - thread: ThreadId(uuid("019b4f42-0000-7000-8000-000000000047")), - }, - AssignmentRevision::new(1), - ResourceRevision::new(2), - None, - ), - outcome: ResourceBackgroundSubmitOutcome::Inserted, - }; - assert!(check_background_response(&response, resource_id, request_id).is_ok()); - - let mismatch = check_background_response( - &response, - resource_id, - RequestId(uuid("019b4f42-0000-7000-8000-000000000048")), - ) - .unwrap_err(); - assert_eq!(mismatch.code(), "resource_outcome_unknown"); - - let reused_id = ResourceBackgroundSubmitResponse { - task_id: TaskId(request_id.0), - ..response - }; - let mismatch = check_background_response(&reused_id, resource_id, request_id).unwrap_err(); - assert_eq!(mismatch.code(), "resource_outcome_unknown"); - } - - #[test] - fn json_and_quiet_output_keep_api_version_and_a_stable_id() { - let value = json!({ "request_id": "request-1" }); - let rendered = render_output(OutputMode::Json, value.clone(), None, None).unwrap(); - let json_value: Value = serde_json::from_str(&rendered.unwrap()).unwrap(); - assert_eq!(json_value["api_version"], API_VERSION); - assert_eq!( - render_output(OutputMode::Quiet, value.clone(), Some("request-1"), None).unwrap(), - Some("request-1".into()) - ); - assert_eq!( - render_output(OutputMode::Quiet, value, None, None).unwrap(), - None - ); - } - - #[test] - fn mutation_errors_keep_conflicts_and_mark_transport_failures_unknown() { - let resource = ResourceId::from_uuid(uuid("019b4f42-0000-7000-8000-000000000008")).unwrap(); - let operation = uuid("019b4f42-0000-7000-8000-000000000009"); - let conflict = mutation_error( - resource, - Some(operation), - "operation id retry-this", - AppError::ResourceOperationConflict { - resource, - operation: Some(operation), - message: "changed content".into(), - }, - ); - assert_eq!(conflict.code(), "resource_operation_conflict"); - let unknown = mutation_error( - resource, - Some(operation), - "operation id retry-this", - AppError::DaemonUnavailable { - message: "socket closed after request".into(), - }, - ); - assert_eq!(unknown.code(), "resource_outcome_unknown"); - assert!( - unknown - .to_string() - .contains("retry with the same operation id retry-this") - ); - - let socket_conflict = mutation_error( - resource, - Some(operation), - "operation id retry-this", - AppError::Internal { - message: "http 409 resource operation conflict: identity already used".into(), - }, - ); - assert_eq!(socket_conflict.code(), "resource_operation_conflict"); - - let stale = mutation_error( - resource, - Some(operation), - "operation id retry-this", - AppError::Internal { - message: "http 409 resource revision is stale: expected 3, current 4".into(), - }, - ); - assert_eq!(stale.code(), "resource_stale_revision"); - } - - fn action_authority(co_located: bool) -> SupervisorActionAuthority { - let authority_machine = MachineId::new(); - SupervisorActionAuthority { - authority_machine, - resource_id: ResourceId::new(), - loan_id: crate::resource::LoanId::new(), - action_id: ActionId::new(), - expected_state_revision: ResourceRevision::new(4), - supervisor: SupervisorAddress { - machine: if co_located { - authority_machine - } else { - MachineId::new() - }, - thread: ThreadId(Uuid::now_v7()), - }, - assignment_revision: AssignmentRevision::new(1), - } - } - - #[test] - fn pending_action_on_a_fresh_resource_keeps_revision_zero() { - let authority = action_authority(true); - let action = PendingActionView { - resource_id: authority.resource_id, - authority_machine: authority.authority_machine, - loan_id: authority.loan_id, - action_id: authority.action_id, - state_revision: ResourceRevision::new(0), - supervisor: authority.supervisor, - assignment_revision: AssignmentRevision::new(0), - phase: PendingActionPhase::ReturnRequired, - return_context: None, - notice: None, - return_window: None, - }; - - let context = action_context(&action).unwrap(); - - assert_eq!( - context.authority.expected_state_revision, - ResourceRevision::new(0) - ); - assert_eq!( - context.authority.assignment_revision, - AssignmentRevision::new(0) - ); - } - - fn return_launch_request( - authority: SupervisorActionAuthority, - ) -> (ResourceActionSubmitRequest, RequestId, TaskId) { - let spec: NormalizedSpec = serde_json::from_value(json!({ - "api_version": 1, - "thread": authority.supervisor.thread, - "name": "return", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": ["/bin/echo", "resume"] } - })) - .unwrap(); - let (request_id, task_id) = (RequestId::new(), TaskId::new()); - let request = ResourceActionSubmitRequest { - api_version: API_VERSION, - authority, - choice: ResourceActionChoice::Return { - decision: ReturnDecision::Launch(Box::new(ReturnLaunch { - request_id, - task_id, - work: ReturnWork::NewBackgroundWork { - spec: CommandSpec::try_from(spec).unwrap(), - }, - })), - }, - }; - (request, request_id, task_id) - } - - #[test] - fn action_outcome_receipt_must_fit_the_supervisor_placement_and_task() { - use crate::resource::bound_action::{ActionTaskReceipt, LocalReturnReceipt}; - - let local = action_authority(true); - let (request, request_id, task_id) = return_launch_request(local); - let local_accepted = - |authority, request_id, task_id| ResourceActionSubmitOutcome::LocalReturnAccepted { - receipt: LocalReturnReceipt { - authority, - request_id, - task_id, - }, - acceptance: LocalReturnAcceptance::Existing { - state: crate::domain::ProcessStatus::Queued, - }, - }; - assert_eq!( - check_action_outcome(&request, &local_accepted(local, request_id, task_id)), - Ok(()) - ); - assert!( - check_action_outcome(&request, &local_accepted(local, request_id, TaskId::new())) - .is_err() - ); - let mut stale = local; - stale.expected_state_revision = ResourceRevision::new(5); - assert!( - check_action_outcome(&request, &local_accepted(stale, request_id, task_id)).is_err() - ); - let digest = serde_json::from_value(json!("0".repeat(64))).unwrap(); - let remote_receipt = |authority| ResourceActionSubmitOutcome::Accepted { - receipt: ActionTaskReceipt { - kind: ResourceActionKind::Return, - authority, - request_id, - task_id, - normalized_spec_sha256: digest, - }, - last_execution_state: None, - }; - // a co-located task has no remote receipt - assert!(check_action_outcome(&request, &remote_receipt(local)).is_err()); - - let remote = action_authority(false); - let (remote_request, request_id, task_id) = return_launch_request(remote); - let remote_receipt = ResourceActionSubmitOutcome::Accepted { - receipt: ActionTaskReceipt { - kind: ResourceActionKind::Return, - authority: remote, - request_id, - task_id, - normalized_spec_sha256: digest, - }, - last_execution_state: None, - }; - assert_eq!( - check_action_outcome(&remote_request, &remote_receipt), - Ok(()) - ); - // a remote supervisor owns its own route, so a local receipt is never valid - assert!( - check_action_outcome( - &remote_request, - &local_accepted(remote, request_id, task_id) - ) - .is_err() - ); - } -} diff --git a/src/client.rs b/src/client.rs index cc81c93..431d0d4 100644 --- a/src/client.rs +++ b/src/client.rs @@ -50,14 +50,6 @@ impl Client { }) } - /// Typed GET. - pub async fn get_json(&self, path: &str) -> Result { - let value = self.get(path).await?; - serde_json::from_value(value).map_err(|err| AppError::Internal { - message: err.to_string(), - }) - } - async fn request( &self, method: &str, diff --git a/src/container.rs b/src/container.rs index 79ae806..dc28ec2 100644 --- a/src/container.rs +++ b/src/container.rs @@ -2,11 +2,11 @@ //! //! Homebased never runs a `docker` command line from a submitter. It builds the //! Docker CLI calls from [`ContainerWorkload`], starts and watches the container -//! itself, and proves that the exact container stopped before it releases a -//! resource. The container runs under `dockerd`, outside the task-run process -//! group, so the container's own state is the witness, not a client process +//! itself, and proves that the exact container stopped before the task ends. +//! The container runs under `dockerd`, outside the task-run process group, so +//! the container's own state is the witness, not a client process //! -//! Version 1 targets Docker Engine on a Linux resource authority +//! Version 1 targets Docker Engine on Linux pub mod docker; pub mod lifecycle; diff --git a/src/container/docker.rs b/src/container/docker.rs index 111f128..a5a06a3 100644 --- a/src/container/docker.rs +++ b/src/container/docker.rs @@ -20,14 +20,10 @@ use super::lifecycle::{ }; use super::spec::{ContainerUser, ContainerWorkload}; use crate::domain::{ContainerId, TaskEnv, TaskId}; -use crate::resource::ResourceId; /// Label that names the Homebased task of a container pub const TASK_LABEL: &str = "homebased.task"; -/// Label that names the resource whose loan runs a container -pub const RESOURCE_LABEL: &str = "homebased.resource"; - /// Longest wait for one Docker CLI call that does not block on the container const CALL_TIMEOUT: Duration = Duration::from_secs(120); @@ -45,8 +41,6 @@ pub fn container_name(task: TaskId) -> String { pub struct CreateContext<'a> { /// Task that owns the container pub task: TaskId, - /// Resource whose loan runs the container, for resource work - pub resource: Option, /// File where Docker writes the new container ID pub cidfile: &'a Path, /// User when the workload names none: the daemon's user @@ -70,10 +64,6 @@ pub fn create_args(workload: &ContainerWorkload, context: &CreateContext<'_>) -> "--label".into(), format!("{TASK_LABEL}={}", context.task), ]; - if let Some(resource) = context.resource { - args.push("--label".into()); - args.push(format!("{RESOURCE_LABEL}={}", resource.as_uuid())); - } args.extend([ "--init".into(), "--cidfile".into(), @@ -462,14 +452,12 @@ mod tests { use crate::container::lifecycle::ContainerStatus; use crate::container::spec::{ContainerUser, ContainerWorkload}; use crate::domain::TaskId; - use crate::resource::ResourceId; const DIGEST: &str = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"; #[test] fn create_argv_comes_only_from_the_typed_spec() { let task: TaskId = "01a0ab97-a7aa-7463-a5b0-8d500e40e431".parse().unwrap(); - let resource = ResourceId::new(); let workload = ContainerWorkload::from_value(&json!({ "image": format!("eval@{DIGEST}"), "entrypoint": ["/usr/bin/python3", "-m", "eval"], @@ -488,7 +476,6 @@ mod tests { &workload, &CreateContext { task, - resource: Some(resource), cidfile: Path::new("/home/me/.homebased/tasks/t/container.cid"), default_user: ContainerUser { uid: 501, gid: 20 }, }, @@ -500,8 +487,6 @@ mod tests { "homebased-01a0ab97-a7aa-7463-a5b0-8d500e40e431", "--label", "homebased.task=01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "--label", - &format!("homebased.resource={}", resource.as_uuid()), "--init", "--cidfile", "/home/me/.homebased/tasks/t/container.cid", @@ -552,7 +537,6 @@ mod tests { &minimal, &CreateContext { task, - resource: None, cidfile: Path::new("/t/container.cid"), default_user: ContainerUser { uid: 501, gid: 20 }, }, @@ -562,11 +546,6 @@ mod tests { .iter() .any(|arg| arg == "--gpus" || arg == "--entrypoint") ); - assert!( - !args - .iter() - .any(|arg| arg.starts_with("homebased.resource=")) - ); assert_eq!(args.last().map(String::as_str), Some(DIGEST)); let user = args.iter().position(|arg| arg == "--user").unwrap(); assert_eq!(args[user + 1], "1000:1000"); diff --git a/src/container/spec.rs b/src/container/spec.rs index 5488299..5de1b67 100644 --- a/src/container/spec.rs +++ b/src/container/spec.rs @@ -460,7 +460,7 @@ pub struct ContainerWorkload { /// Arguments passed unchanged; not a shell string #[serde(default, skip_serializing_if = "Vec::is_empty")] pub args: Vec, - /// GPUs the container may use. Required for resource work + /// GPUs the container may use #[serde(default, skip_serializing_if = "Option::is_none")] pub gpus: Option, /// Memory limit, also used as the memory-plus-swap limit @@ -900,16 +900,10 @@ pub fn check_container_host(workload: &ContainerWorkload) -> Result<(), AppError } /// JSON Schema of the container workload body, including its `type` key -/// -/// `gpus_required` is set for resource work, which must name its GPUs #[must_use] -pub fn container_workload_schema(gpus_required: bool) -> Value { +pub fn container_workload_schema() -> Value { let hex = "[0-9a-f]{64}"; let component = "[a-z0-9]+([._-]+[a-z0-9]+)*"; - let mut required = vec!["type", "image", "memory"]; - if gpus_required { - required.push("gpus"); - } json!({ "title": "container", "description": "Docker container that Homebased starts, watches, stops, and removes. No shell and no Docker options beyond these fields.", @@ -943,11 +937,7 @@ pub fn container_workload_schema(gpus_required: bool) -> Value { "items": { "type": "integer", "minimum": 0, "maximum": u32::MAX } } ], - "description": if gpus_required { - "\"all\" or device indices. Required for resource work." - } else { - "\"all\" or device indices." - } + "description": "\"all\" or device indices." }, "memory": { "oneOf": [ @@ -986,7 +976,7 @@ pub fn container_workload_schema(gpus_required: bool) -> Value { "description": "Explicit environment values. Nothing passes through from the host." } }, - "required": required, + "required": ["type", "image", "memory"], "additionalProperties": false }) } diff --git a/src/daemon.rs b/src/daemon.rs index 8fcbeb8..d2b9607 100644 --- a/src/daemon.rs +++ b/src/daemon.rs @@ -16,13 +16,6 @@ pub(crate) mod message_receiver; pub(crate) mod message_sender; mod origin_submit; mod peer_read; -mod release_watcher_api; -mod resource_action; -mod resource_api; -mod resource_background; -mod resource_notice_delivery; -pub(crate) mod resource_notice_sender; -mod resource_submit; mod t3_watch; mod thread_titles; pub mod web; @@ -51,7 +44,6 @@ use crate::fleet::runtime::{FleetRuntime, FleetStart, RuntimeTimings}; use crate::home::{Home, LockMode, chmod_600, flock_exclusive}; use crate::machine::LocalIdentity; use crate::notify::Notifier; -use crate::resource::ActionId; use crate::submission::RequestId; use crate::thread_title::TitleSources; @@ -92,12 +84,6 @@ pub struct AppState { pub(crate) struct DaemonLocks { /// Origin-side remote task submissions, by caller request pub(crate) origin_submissions: KeyedLocks, - /// Origin-side resource queue submissions, by caller request - pub(crate) resource_submissions: KeyedLocks, - /// Supervisor-side resource action launches, by action - pub(crate) resource_actions: KeyedLocks, - /// Supervisor-side background launches, by launch request - pub(crate) background_launches: KeyedLocks, /// Origin-side cancellation intents, by task pub(crate) cancellation_intents: KeyedLocks, } @@ -173,11 +159,7 @@ pub async fn serve(home: Home, web_listen: WebListen, config: Config) -> Result< )); let recovery = tokio::spawn(origin_submit::recover(state.clone())); let dependency_release = tokio::spawn(dependencies::run(state.clone())); - let resource_recovery = tokio::spawn(resource_submit::recover(state.clone())); - let action_recovery = tokio::spawn(resource_action::recover(state.clone())); - let background_recovery = tokio::spawn(resource_background::recover(state.clone())); let cancellation = tokio::spawn(cancel_delivery::run(state.clone())); - let notice_delivery = tokio::spawn(resource_notice_delivery::run(state.clone())); let t3_watcher = tokio::spawn(t3_watch::run(notifier, state.machine.name.to_string())); // listeners share one shutdown: the signal task flips the flag once let (shutdown_tx, shutdown_rx) = watch::channel(false); @@ -218,11 +200,7 @@ pub async fn serve(home: Home, web_listen: WebListen, config: Config) -> Result< sender.abort(); recovery.abort(); dependency_release.abort(); - resource_recovery.abort(); - action_recovery.abort(); - background_recovery.abort(); cancellation.abort(); - notice_delivery.abort(); t3_watcher.abort(); if !supervisor_died { supervisor.stop(None); diff --git a/src/daemon/actors.rs b/src/daemon/actors.rs index 5c4b590..897c267 100644 --- a/src/daemon/actors.rs +++ b/src/daemon/actors.rs @@ -1,7 +1,6 @@ -//! Daemon ractor topology: store, callback, per-task watch, per-resource owner, supervisor +//! Daemon ractor topology: store, callback, per-task watch, supervisor pub mod callback; -pub mod resource; pub mod store; pub mod supervisor; pub mod task; @@ -16,7 +15,6 @@ use crate::error::AppError; pub const CALL_TIMEOUT: Duration = Duration::from_secs(5); pub use callback::{CallbackActor, CallbackMsg}; -pub use resource::{ResourceActorInspection, ResourceMsg}; pub(crate) use store::{StoreActor, StoreMsg}; pub(crate) use supervisor::{SupervisorActor, SupervisorArgs, SupervisorMsg}; pub use task::{TaskActor, TaskMsg}; diff --git a/src/daemon/actors/resource.rs b/src/daemon/actors/resource.rs deleted file mode 100644 index 0839161..0000000 --- a/src/daemon/actors/resource.rs +++ /dev/null @@ -1,2057 +0,0 @@ -//! Authority-owned queue reconciliation for one resource - -use std::time::Duration; - -use chrono::{DateTime, Utc}; -use ractor::{Actor, ActorId, ActorProcessingErr, ActorRef, RpcReplyPort}; -use tokio::task::AbortHandle; -use tokio::time::Instant; - -use crate::daemon::actors::supervisor::ReleaseWatcherLaunch; -use crate::daemon::actors::{StoreMsg, SupervisorMsg, call, send_reply}; -use crate::domain::{ProcessStatus, TaskEnv, TaskId}; -use crate::error::AppError; -use crate::machine::MachineId; -use crate::resource::bound_action::{PreparedActionTask, ResourceActionRejection}; -use crate::resource::release_watcher::{ReleaseWatcherCommand, release_watcher_executable}; -use crate::resource::store::{ - AssignedResourceTaskAttention, AssignedResourceTaskProgress, - AssignedResourceTaskReconcileInput, AssignedResourceTaskReconcileOutcome, CompleteReleaseError, - PreLaunchFailure, ReleaseCompletionResult, ReleaseWatcherAcceptance, - ReleaseWatcherAcceptanceError, ResourceSnapshot, ResourceTaskAcceptance, - ResourceTaskAcceptanceInput, ResourceTaskCompletionResult, ReturnDeadlineOutcome, -}; -use crate::resource::{ - ActionId, LAUNCH_CONFIRMATION_BOUND, Loan, LoanPhase, LoanState, ReleaseProofAttentionReason, - ReleaseWatcherIntent, ReleaseWatcherTaskId, Resource, ResourceId, ResourceQueueAttentionReason, - ResourceQueueReconcileOutcome, ResourceRequest, ResourceRequestState, RestoreAttentionReason, - SupervisorActionAuthority, -}; -use crate::store::{BackgroundLaunchPhase, BackgroundLaunchView, RestoreReconcileOutcome}; -use crate::submission::{RequestId, normalized_spec_sha256}; - -/// Messages for one resource actor -pub enum ResourceMsg { - /// Reconcile current authority-owned queue state and refresh this actor's snapshot - Reconcile { - /// Reply with the typed reconciliation outcome - reply: RpcReplyPort>, - }, - /// Wake only when one exact assigned task reached a task-layer terminal event - TaskTerminal { - /// Exact task whose durable terminal state is ready to reconcile - task_id: TaskId, - }, - /// Reconcile after this actor commits a queue transition - Wake, - /// Return the actor identity and the latest authority-owned snapshot - Inspect { - /// Reply with the actor's identity and current snapshot - reply: RpcReplyPort>, - }, - /// Result of one asynchronous, exact assigned-task launch request - ActivationFinished { - /// Exact request sent to the supervisor - request_id: RequestId, - /// Exact preallocated task sent to the supervisor - task_id: TaskId, - /// Result returned by the dedicated assigned-task launch path - result: ResourceTaskActivationResult, - }, - /// Wake when a durable event for one task was delivered; only the bound return task matters - TaskProgress { - /// Task whose durable state may have changed - task_id: TaskId, - }, - /// The supervisor is about to bind and launch one fixed return task - RestoreLaunchStarted { - /// Return action that owns the task - action_id: ActionId, - /// Fixed return task identity - task_id: TaskId, - }, - /// Result of the supervisor's one bind-and-launch request for a return task - RestoreLaunchFinished { - /// Return action that owns the task - action_id: ActionId, - /// Fixed return task identity - task_id: TaskId, - /// Whether that request inserted the task and started its worker - result: RestoreLaunchResult, - }, - /// The supervisor is about to bind and launch one first background task - BackgroundLaunchStarted { - /// Stable launch request identity - request_id: RequestId, - }, - /// Result of the supervisor's one bind-and-launch request for a first background task - BackgroundLaunchFinished { - /// Stable launch request identity - request_id: RequestId, - /// Whether that request inserted the task and started its worker - result: BackgroundLaunchResult, - }, - /// Bind and baseline the watcher for a remote supervisor, then return its canonical task - PrepareRemoteWatcher { - /// Exact action authority named by the supervisor machine - authority: SupervisorActionAuthority, - /// Background task named by the release action - observed_background_task: TaskId, - /// Canonical watcher task or a definitive refusal - reply: RpcReplyPort, AppError>>, - }, - /// The supervisor is about to accept a remote supervisor's bound watcher - RemoteWatcherLaunchStarted { - /// Release action that owns the watcher - action_id: ActionId, - /// Exact bound watcher task - watcher_task_id: TaskId, - }, - /// Result of one asynchronous, bound release-watcher launch request - WatcherLaunchFinished { - /// Release action that owns the watcher - action_id: ActionId, - /// Exact preallocated watcher task sent to the supervisor - watcher_task_id: TaskId, - /// Result returned by the dedicated watcher launch path - result: ReleaseWatcherLaunchResult, - }, -} - -/// Outcome of one supervisor request to bind and launch a first background task -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum BackgroundLaunchResult { - /// The launch was inserted by this request, so it was the only spawn attempt - Inserted, - /// An exact earlier launch existed and was only observed - Existing, - /// Nothing was inserted, or a committed row may have no worker - NotInserted, -} - -/// Outcome of asking the supervisor to accept one bound release watcher -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum ReleaseWatcherLaunchResult { - /// The watcher row was inserted by this request, so it was the only spawn attempt - Inserted, - /// An exact watcher acceptance already existed with this process state - Existing { - /// State retained by the task layer - state: ProcessStatus, - }, - /// The resource supervisor is not on the authority machine - UnsupportedRemoteSupervisor, - /// The authority rejected the watcher identity or command - Rejected, - /// The request may have committed, but its result was not definitive - Uncertain, -} - -/// Outcome of the supervisor's one bind-and-launch request for a return task -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum RestoreLaunchResult { - /// The request inserted the task, so its worker launch was the only spawn attempt - Inserted, - /// An exact earlier binding existed, so the request only observed it - Existing { - /// State retained by the task layer - state: ProcessStatus, - }, - /// The request wrote no task records or its result was not definitive - NotInserted, -} - -/// Launch and observation state of the bound watcher for one release action -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum ReleaseWatcherStatus { - /// The watcher is bound, and the remote supervisor machine must save its route and launch it - AwaitingRemoteSupervisor { - /// Release action that owns the watcher - action_id: ActionId, - /// Exact bound watcher task - watcher_task_id: TaskId, - /// Machine that owns the supervisor thread and the watcher callback route - supervisor_machine: MachineId, - }, - /// The one launch request for this watcher is in flight or its task has not started - Launching { - /// Release action that owns the watcher - action_id: ActionId, - /// Exact preallocated watcher task - watcher_task_id: TaskId, - }, - /// The watcher task is running - Running { - /// Release action that owns the watcher - action_id: ActionId, - /// Exact watcher task - watcher_task_id: TaskId, - }, - /// The watcher finished its part; release proof remains with the resource owner - Finished { - /// Release action that owns the watcher - action_id: ActionId, - /// Exact watcher task - watcher_task_id: TaskId, - }, - /// The loan stays reserved and the watcher needs attention - Attention { - /// Release action that owns the watcher - action_id: ActionId, - /// Typed reason the watcher cannot proceed - reason: ReleaseWatcherAttentionReason, - }, -} - -/// Why a release action cannot use or trust its bound watcher -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum ReleaseWatcherAttentionReason { - /// The resource supervisor runs on another machine, so no watcher task was written - RemoteSupervisorUnsupported { - /// Machine that owns the supervisor thread - supervisor_machine: MachineId, - }, - /// The running trainer has no saved attempt association - TrainerAssociationMissing { - /// Exact registered trainer task - task_id: TaskId, - }, - /// The daemon cannot name the executable for the canonical watcher command - ExecutableUnavailable, - /// The authority rejected binding the watcher identity to the release action - BindingRejected, - /// The authority could not capture the checkpoint baseline - BaselineUnavailable, - /// The authority rejected the watcher acceptance or its canonical command - LaunchRejected { - /// Exact watcher task - watcher_task_id: TaskId, - }, - /// The watcher task was accepted, but whether its worker started is unknown - LaunchUncertain { - /// Exact watcher task - watcher_task_id: TaskId, - }, - /// The watcher task ended without finishing its part - WatcherTaskEnded { - /// Exact watcher task - watcher_task_id: TaskId, - /// Terminal task state - state: ProcessStatus, - }, -} - -/// Outcome of asking the supervisor to accept one fixed resource task -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum ResourceTaskActivationResult { - /// The task layer inserted this task for the first time - Inserted, - /// An exact task acceptance already existed with this process state - Existing { - /// State retained by the task layer - state: ProcessStatus, - }, - /// The task layer accepted this task, and it failed before launch - FailedBeforeLaunch, - /// Cancellation prevented task acceptance - Prevented, - /// The request may have committed, but the supervisor result was not definitive - Uncertain, -} - -/// Read-only view of one resource actor -#[derive(Debug, Clone)] -pub struct ResourceActorInspection { - /// Ractor identity of the actor that returned this snapshot - pub actor_id: ActorId, - /// Durable resource state refreshed after the most recent reconciliation - pub resource: Resource, - /// Durable active or attention loan restored by the actor, if present - pub loan: Option, - /// Latest typed queue outcome, if a reconciliation has completed - pub reconcile_outcome: Option, - /// Bound watcher state for the current release action, if one applies - pub release_watcher: Option, -} - -/// State for one resource actor -pub struct ResourceActorState { - store: ActorRef, - supervisor: ActorRef, - authority_machine: MachineId, - resource: Resource, - loan: Option, - reconcile_outcome: Option, - activation_attempt: Option, - watcher_launch: Option, - release_watcher: Option, - restore_launch: Option, - background_launch: Option, - pending_background_task: Option, - return_deadline_wake: Option, - unconfirmed_launch: Option, -} - -/// Assigned launch whose outcome this actor lifetime has not confirmed -/// -/// After [`LAUNCH_CONFIRMATION_BOUND`] the task is failed before launch, and -/// again after each further bound while its outcome stays unknown -struct UnconfirmedLaunch { - key: ActivationKey, - // monotonic time of the first observation or of the last failure request - since: Instant, - wake: AbortHandle, -} - -impl Drop for UnconfirmedLaunch { - fn drop(&mut self) { - self.wake.abort(); - } -} - -/// Wait before checking a return deadline again after the store failed to answer -const RETURN_DEADLINE_RETRY: Duration = Duration::from_secs(5); - -/// Shortest wait before a deadline wake, so a clock step cannot spin the actor -const RETURN_DEADLINE_MIN_WAIT: Duration = Duration::from_secs(1); - -/// Wake armed for one pending return action -/// -/// The store decides from its saved window whether the deadline passed. The -/// timer uses the monotonic clock and the window uses wall time, so a wake can -/// fire before the saved deadline; a fired wake then no longer covers it, and -/// the next reconcile arms another one -struct ReturnDeadlineWake { - target: ReturnWakeTarget, - wake: AbortHandle, -} - -/// What one return deadline wake checks -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -enum ReturnWakeTarget { - /// The saved deadline of one return action - Deadline { - action_id: ActionId, - deadline_at: DateTime, - }, - /// Another check after the store failed to answer - Retry, -} - -impl ReturnDeadlineWake { - const fn new(target: ReturnWakeTarget, wake: AbortHandle) -> Self { - Self { target, wake } - } - - /// Whether this wake is armed for `target` and has not fired yet - fn covers(&self, target: ReturnWakeTarget) -> bool { - self.target == target && !self.wake.is_finished() - } -} - -impl Drop for ReturnDeadlineWake { - fn drop(&mut self) { - self.wake.abort(); - } -} - -// each phase launches through a different supervisor path and can outlive the -// others, so each keeps its own slot; a slot holds only this actor lifetime's -// request, so a queued row that this lifetime did not insert may have lost its -// spawn and is only observed, never respawned - -// one supervisor launch request per first background request and actor lifetime -type BackgroundLaunchAttempt = LaunchAttempt; - -// one supervisor launch request per return task and actor lifetime -type RestoreLaunchAttempt = LaunchAttempt; - -// one launch request per watcher identity and actor lifetime; a new actor after a -// restart asks again with the same saved identity and only observes what it finds -type WatcherLaunchAttempt = LaunchAttempt; - -// one supervisor launch request per assigned task and actor lifetime -type ActivationAttempt = LaunchAttempt; - -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -struct RestoreLaunchKey { - action_id: ActionId, - task_id: TaskId, -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -struct WatcherLaunchKey { - action_id: ActionId, - watcher_task_id: TaskId, -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -struct ActivationKey { - request_id: RequestId, - task_id: TaskId, -} - -/// One supervisor launch request made by this actor lifetime -#[derive(Debug, Clone, Copy)] -struct LaunchAttempt { - key: K, - progress: LaunchProgress, -} - -/// Whether a launch request has replied, and with which result -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -enum LaunchProgress { - Pending, - Finished(R), -} - -impl LaunchAttempt { - fn pending(key: K) -> Self { - Self { - key, - progress: LaunchProgress::Pending, - } - } - - /// Record the reply for this exact pending request - /// - /// A reply for another identity, or a second reply, is stale and changes nothing - fn finish(&mut self, key: K, result: R) -> bool { - if self.key != key || !matches!(self.progress, LaunchProgress::Pending) { - return false; - } - self.progress = LaunchProgress::Finished(result); - true - } - - /// Whether this request is still in flight or inserted the launch itself - fn launching(&self, key: K, inserted: impl FnOnce(R) -> bool) -> bool { - self.key == key - && match self.progress { - LaunchProgress::Pending => true, - LaunchProgress::Finished(result) => inserted(result), - } - } -} - -/// Actor that reconciles one resource from StoreActor-owned state -pub(crate) struct ResourceActor; - -impl Actor for ResourceActor { - type Msg = ResourceMsg; - type State = ResourceActorState; - type Arguments = ( - ActorRef, - ActorRef, - MachineId, - Resource, - Option, - ); - - async fn pre_start( - &self, - myself: ActorRef, - (store, supervisor, authority_machine, resource, loan): Self::Arguments, - ) -> Result { - let mut state = ResourceActorState { - store, - supervisor, - authority_machine, - resource, - loan, - reconcile_outcome: None, - activation_attempt: None, - watcher_launch: None, - release_watcher: None, - restore_launch: None, - background_launch: None, - pending_background_task: None, - return_deadline_wake: None, - unconfirmed_launch: None, - }; - reconcile_and_refresh(&myself, &mut state).await?; - - Ok(state) - } - - async fn handle( - &self, - myself: ActorRef, - message: Self::Msg, - state: &mut Self::State, - ) -> Result<(), ActorProcessingErr> { - match message { - ResourceMsg::Reconcile { reply } => { - send_reply(reply, reconcile_and_refresh(&myself, state).await); - } - ResourceMsg::TaskTerminal { task_id } => { - if restoring_task_id(state) == Some(task_id) - || state.pending_background_task == Some(task_id) - || serving_task_id(state).await? == Some(task_id) - { - reconcile_and_refresh(&myself, state).await?; - } - } - ResourceMsg::TaskProgress { task_id } => { - if restoring_task_id(state) == Some(task_id) - || state.pending_background_task == Some(task_id) - || state - .watcher_launch - .is_some_and(|attempt| attempt.key.watcher_task_id == task_id) - { - reconcile_and_refresh(&myself, state).await?; - } - } - ResourceMsg::RestoreLaunchStarted { action_id, task_id } => { - state.restore_launch = Some(RestoreLaunchAttempt::pending(RestoreLaunchKey { - action_id, - task_id, - })); - } - ResourceMsg::RestoreLaunchFinished { - action_id, - task_id, - result, - } => { - if let Some(attempt) = state.restore_launch.as_mut() { - attempt.finish(RestoreLaunchKey { action_id, task_id }, result); - } - reconcile_and_refresh(&myself, state).await?; - } - ResourceMsg::BackgroundLaunchStarted { request_id } => { - state.background_launch = Some(BackgroundLaunchAttempt::pending(request_id)); - } - ResourceMsg::BackgroundLaunchFinished { request_id, result } => { - if let Some(attempt) = state.background_launch.as_mut() { - attempt.finish(request_id, result); - } - reconcile_and_refresh(&myself, state).await?; - } - ResourceMsg::Wake => { - reconcile_and_refresh(&myself, state).await?; - } - ResourceMsg::ActivationFinished { - request_id, - task_id, - result, - } => { - handle_activation_finished(&myself, state, request_id, task_id, result).await?; - } - ResourceMsg::WatcherLaunchFinished { - action_id, - watcher_task_id, - result, - } => { - handle_watcher_launch_finished(state, action_id, watcher_task_id, result).await?; - } - ResourceMsg::PrepareRemoteWatcher { - authority, - observed_background_task, - reply, - } => { - let prepared = - prepare_remote_watcher(state, authority, observed_background_task).await; - send_reply(reply, prepared); - reconcile_and_refresh(&myself, state).await?; - } - ResourceMsg::RemoteWatcherLaunchStarted { - action_id, - watcher_task_id, - } => { - // only this lifetime's own accepted request may show a queued row as launching - state.watcher_launch = Some(WatcherLaunchAttempt::pending(WatcherLaunchKey { - action_id, - watcher_task_id, - })); - } - ResourceMsg::Inspect { reply } => send_reply( - reply, - Ok(ResourceActorInspection { - actor_id: myself.get_id(), - resource: state.resource.clone(), - loan: state.loan.clone(), - reconcile_outcome: state.reconcile_outcome.clone(), - release_watcher: state.release_watcher.clone(), - }), - ), - } - - Ok(()) - } -} - -async fn reconcile_and_refresh( - myself: &ActorRef, - state: &mut ResourceActorState, -) -> Result { - let queue_outcome = call(&state.store, |reply| StoreMsg::ReconcileResourceQueue { - authority_machine: state.authority_machine, - resource_id: state.resource.id, - reply, - }) - .await?; - - let mut snapshot = load_snapshot(state).await?; - let release = complete_release(state, &mut snapshot).await?; - state.release_watcher = match &release { - ReleaseProgress::ProofUnavailable(failure) => { - let resource = snapshot.resource.clone(); - progress_release_watcher(myself, state, &resource, failure).await? - } - ReleaseProgress::NotAwaiting | ReleaseProgress::Completed { .. } => { - state.watcher_launch = None; - None - } - }; - - let mut outcome = match release { - // the loan is still awaiting release, so no request can be serving - ReleaseProgress::ProofUnavailable(failure) => { - state.activation_attempt = None; - failure.into_outcome() - } - ReleaseProgress::NotAwaiting => { - let serving = reconcile_serving_task(myself, state, &mut snapshot).await?; - serving_outcome(serving).unwrap_or(queue_outcome) - } - ReleaseProgress::Completed { loan } => { - let serving = reconcile_serving_task(myself, state, &mut snapshot).await?; - serving_outcome(serving) - .unwrap_or(ResourceQueueReconcileOutcome::LoanAlreadyActive { loan }) - } - }; - - if let Some(restore_outcome) = reconcile_restore(myself, state, &mut snapshot).await? { - outcome = restore_outcome; - } - - if let Some(launch_outcome) = observe_background_launch(state, &snapshot, &outcome).await? { - outcome = launch_outcome; - } - - if let Some(deadline_outcome) = reconcile_return_deadline(myself, state, &mut snapshot).await? { - outcome = deadline_outcome; - } - - state.resource = snapshot.resource; - state.loan = snapshot.loan; - state.reconcile_outcome = Some(outcome.clone()); - - Ok(outcome) -} - -/// Result of trying to prove the release of a loan that awaits one -enum ReleaseProgress { - /// The loan does not await a release - NotAwaiting, - /// The store committed the release, and this loan replaced the awaiting one - Completed { loan: Loan }, - /// The release proof is not available yet, so the loan stays reserved - ProofUnavailable(ReleaseProofFailure), -} - -/// Awaiting-release loan whose release proof the store refused -struct ReleaseProofFailure { - loan: Loan, - action_id: ActionId, - trainer_task_id: TaskId, - reason: ReleaseProofAttentionReason, -} - -impl ReleaseProofFailure { - fn into_outcome(self) -> ResourceQueueReconcileOutcome { - ResourceQueueReconcileOutcome::ReleaseProofUnavailable { - loan: self.loan, - action_id: self.action_id, - task_id: self.trainer_task_id, - reason: self.reason, - } - } -} - -/// Complete the pending release action and reload the snapshot when it commits -async fn complete_release( - state: &ResourceActorState, - snapshot: &mut ResourceSnapshot, -) -> Result { - let Some((loan, action_id, trainer_task_id)) = awaiting_release(snapshot) else { - return Ok(ReleaseProgress::NotAwaiting); - }; - let completion = call(&state.store, |reply| { - StoreMsg::CompleteReleaseForAuthority { - authority_machine: state.authority_machine, - resource_id: snapshot.resource.id, - action_id, - expected_state_revision: snapshot.resource.state_revision, - reply, - } - }) - .await?; - match completion { - Ok( - ReleaseCompletionResult::Assigned { loan, .. } - | ReleaseCompletionResult::ReturnRequired { loan, .. }, - ) => { - *snapshot = load_snapshot(state).await?; - Ok(ReleaseProgress::Completed { loan }) - } - Err(error) => Ok(ReleaseProgress::ProofUnavailable(ReleaseProofFailure { - loan, - action_id, - trainer_task_id, - reason: release_proof_attention_reason(error), - })), - } -} - -/// Progress of the task assigned to the current serving loan -enum ServingProgress { - /// No request is assigned to a serving loan - Unassigned, - /// The assigned task ended, and the store committed its completion into this loan - Completed { loan: Loan }, - /// The task layer accepted the assigned task - Accepted { - loan: Loan, - // boxed because this is the common variant and both records are large - request: Box, - progress: AssignedResourceTaskProgress, - // whether this actor lifetime's own launch request inserted the task - inserted_by_this_actor: bool, - }, - /// This actor just asked the supervisor to launch the assigned task - LaunchRequested { loan: Loan }, - /// The task is not accepted, and this actor lifetime already asked to launch it - LaunchAttempted { - request: ResourceRequest, - progress: LaunchProgress, - }, - /// The assigned task needs attention - Attention { - request: ResourceRequest, - reason: ResourceQueueAttentionReason, - }, -} - -/// Reconcile the assigned task of the serving loan, launch it once when needed, -/// and fail it before launch when its outcome stays unknown past the bound -async fn reconcile_serving_task( - myself: &ActorRef, - state: &mut ResourceActorState, - snapshot: &mut ResourceSnapshot, -) -> Result { - let progress = reconcile_serving_progress(myself, state, snapshot).await?; - bound_unconfirmed_launch(myself, state, snapshot, &progress); - - Ok(progress) -} - -async fn reconcile_serving_progress( - myself: &ActorRef, - state: &mut ResourceActorState, - snapshot: &mut ResourceSnapshot, -) -> Result { - let requests = call(&state.store, |reply| StoreMsg::ResourceRequests { - authority_machine: state.authority_machine, - resource_id: snapshot.resource.id, - reply, - }) - .await?; - let serving = serving_assignment(snapshot, &requests); - let current_key = serving.as_ref().map(|(_, request)| ActivationKey { - request_id: request.request_id, - task_id: request.task_id, - }); - state.activation_attempt = state - .activation_attempt - .filter(|attempt| Some(attempt.key) == current_key); - let Some((loan, request)) = serving else { - return Ok(ServingProgress::Unassigned); - }; - - let input = AssignedResourceTaskReconcileInput { - authority_machine: state.authority_machine, - resource_id: request.resource_id, - loan_id: loan.id, - request_id: request.request_id, - task_id: request.task_id, - expected_state_revision: snapshot.resource.state_revision, - }; - let reconciled = call(&state.store, |reply| { - StoreMsg::AssignedResourceTaskReconcile { input, reply } - }) - .await?; - let task_id = request.task_id; - match reconciled { - Err(error) => { - tracing::warn!(%task_id, "assigned resource task reconciliation failed: {error}"); - Ok(ServingProgress::Attention { - request, - reason: ResourceQueueAttentionReason::AssignedTaskReconcileFailed { task_id }, - }) - } - Ok(AssignedResourceTaskReconcileOutcome::Attention(reason)) => { - Ok(ServingProgress::Attention { - request, - reason: task_completion_attention_reason(task_id, reason), - }) - } - Ok(AssignedResourceTaskReconcileOutcome::Active(progress)) => { - let inserted_by_this_actor = state.activation_attempt.is_some_and(|attempt| { - attempt.progress == LaunchProgress::Finished(ResourceTaskActivationResult::Inserted) - }); - Ok(ServingProgress::Accepted { - loan, - request: Box::new(request), - progress, - inserted_by_this_actor, - }) - } - Ok(AssignedResourceTaskReconcileOutcome::Completed(result)) => { - let next_assignment = matches!(*result, ResourceTaskCompletionResult::Assigned { .. }); - *snapshot = load_snapshot(state).await?; - let Some(loan) = snapshot.loan.clone() else { - return Err(AppError::Internal { - message: "resource task completion removed its active loan".into(), - }); - }; - if next_assignment { - myself.cast(ResourceMsg::Wake)?; - } - Ok(ServingProgress::Completed { loan }) - } - Ok(AssignedResourceTaskReconcileOutcome::NotAccepted) => { - Ok(launch_assigned_task(myself, state, snapshot, loan, request)) - } - } -} - -/// Ask the supervisor once per actor lifetime to launch the assigned task -/// -/// The supervisor is called from a detached task because it may be waiting on -/// this actor -fn launch_assigned_task( - myself: &ActorRef, - state: &mut ResourceActorState, - snapshot: &ResourceSnapshot, - loan: Loan, - request: ResourceRequest, -) -> ServingProgress { - if let Some(attempt) = state.activation_attempt { - return ServingProgress::LaunchAttempted { - request, - progress: attempt.progress, - }; - } - let input = activation_input(state, snapshot, &loan, &request); - state.activation_attempt = Some(ActivationAttempt::pending(ActivationKey { - request_id: request.request_id, - task_id: request.task_id, - })); - schedule_assigned_activation(myself.clone(), state.supervisor.clone(), input, None); - - ServingProgress::LaunchRequested { loan } -} - -fn activation_input( - state: &ResourceActorState, - snapshot: &ResourceSnapshot, - loan: &Loan, - request: &ResourceRequest, -) -> ResourceTaskAcceptanceInput { - ResourceTaskAcceptanceInput { - authority_machine: state.authority_machine, - resource_id: request.resource_id, - request_id: request.request_id, - task_id: request.task_id, - acceptance_sequence: request.acceptance_sequence, - loan_id: loan.id, - expected_state_revision: snapshot.resource.state_revision, - command_spec: request.spec().clone(), - executor_env: TaskEnv::capture(), - } -} - -/// Serving assignment whose launch has no confirmed start yet, if any -/// -/// A queued task, a launch request without a definite reply, and a task this -/// lifetime cannot see a worker for all count; a running or ended task does not -fn unconfirmed_launch( - state: &ResourceActorState, - snapshot: &ResourceSnapshot, - progress: &ServingProgress, -) -> Option<(ActivationKey, Option<(Loan, ResourceRequest)>)> { - let key = |request: &ResourceRequest| ActivationKey { - request_id: request.request_id, - task_id: request.task_id, - }; - match progress { - ServingProgress::Accepted { - loan, - request, - progress: AssignedResourceTaskProgress::Queued, - .. - } => Some((key(request), Some((loan.clone(), (**request).clone())))), - ServingProgress::LaunchAttempted { - request, - progress: - LaunchProgress::Pending - | LaunchProgress::Finished( - ResourceTaskActivationResult::Prevented - | ResourceTaskActivationResult::Uncertain - | ResourceTaskActivationResult::Existing { - state: ProcessStatus::Queued, - }, - ), - } => Some(( - key(request), - snapshot.loan.clone().map(|loan| (loan, request.clone())), - )), - // the request is only known to the pending attempt until its reply - ServingProgress::LaunchRequested { .. } => { - state.activation_attempt.map(|attempt| (attempt.key, None)) - } - ServingProgress::Unassigned - | ServingProgress::Completed { .. } - | ServingProgress::Accepted { .. } - | ServingProgress::LaunchAttempted { .. } - | ServingProgress::Attention { .. } => None, - } -} - -/// Start the confirmation clock for an unconfirmed launch, and fail the task -/// before launch once the clock passes [`LAUNCH_CONFIRMATION_BOUND`] -/// -/// The failure request cannot hide started work: the store returns an existing -/// acceptance unchanged, and the supervisor fails only a task that is still -/// queued, which no worker has claimed -fn bound_unconfirmed_launch( - myself: &ActorRef, - state: &mut ResourceActorState, - snapshot: &ResourceSnapshot, - progress: &ServingProgress, -) { - let Some((key, assignment)) = unconfirmed_launch(state, snapshot, progress) else { - state.unconfirmed_launch = None; - return; - }; - let armed = state - .unconfirmed_launch - .as_ref() - .filter(|unconfirmed| unconfirmed.key == key); - let Some(since) = armed.map(|unconfirmed| unconfirmed.since) else { - arm_unconfirmed_launch(myself, state, key, Instant::now()); - return; - }; - let remaining = LAUNCH_CONFIRMATION_BOUND.saturating_sub(since.elapsed()); - let Some((loan, request)) = assignment.filter(|_| remaining.is_zero()) else { - // a timer can fire just before the bound, and only a wake reconciles an - // idle queue, so another wake always covers the rest of the bound - if state - .unconfirmed_launch - .as_ref() - .is_some_and(|unconfirmed| unconfirmed.wake.is_finished()) - { - arm_unconfirmed_launch(myself, state, key, since); - } - return; - }; - - let task_id = key.task_id; - tracing::warn!( - %task_id, - "assigned launch has no confirmed start after the bound; failing it before launch" - ); - let input = activation_input(state, snapshot, &loan, &request); - state.activation_attempt = Some(ActivationAttempt::pending(key)); - schedule_assigned_activation( - myself.clone(), - state.supervisor.clone(), - input, - Some(PreLaunchFailure::LaunchUnconfirmed), - ); - // a failure request whose outcome is also unknown is tried again after another bound - arm_unconfirmed_launch(myself, state, key, Instant::now()); -} - -/// Track `key` from `since` and wake once the bound from `since` has passed -fn arm_unconfirmed_launch( - myself: &ActorRef, - state: &mut ResourceActorState, - key: ActivationKey, - since: Instant, -) { - // one extra millisecond keeps a timer that rounds down from landing before the bound - let wait = LAUNCH_CONFIRMATION_BOUND.saturating_sub(since.elapsed()) + Duration::from_millis(1); - let wake = myself.send_after(wait, || ResourceMsg::Wake).abort_handle(); - state.unconfirmed_launch = Some(UnconfirmedLaunch { key, since, wake }); -} - -/// Queue outcome for serving progress, or `None` to keep the earlier outcome -fn serving_outcome(progress: ServingProgress) -> Option { - let launch_uncertain = |request: ResourceRequest, reason| { - Some(ResourceQueueReconcileOutcome::AttentionRequired { request, reason }) - }; - match progress { - ServingProgress::Unassigned => None, - ServingProgress::Completed { loan } | ServingProgress::LaunchRequested { loan } => { - Some(ResourceQueueReconcileOutcome::LoanAlreadyActive { loan }) - } - ServingProgress::Attention { request, reason } => { - Some(ResourceQueueReconcileOutcome::AttentionRequired { request, reason }) - } - // only this actor's own inserted launch may still be starting; an existing - // queued row from before is never respawned here - ServingProgress::Accepted { - request, - progress: AssignedResourceTaskProgress::Queued, - inserted_by_this_actor: false, - .. - } => { - let task_id = request.task_id; - launch_uncertain( - *request, - ResourceQueueAttentionReason::AcceptedTaskLaunchUncertain { task_id }, - ) - } - ServingProgress::Accepted { loan, .. } => { - Some(ResourceQueueReconcileOutcome::LoanAlreadyActive { loan }) - } - ServingProgress::LaunchAttempted { - request, - progress: - LaunchProgress::Pending - | LaunchProgress::Finished( - ResourceTaskActivationResult::Prevented - | ResourceTaskActivationResult::Uncertain - | ResourceTaskActivationResult::Existing { - state: ProcessStatus::Queued, - }, - ), - } => { - let task_id = request.task_id; - launch_uncertain( - request, - ResourceQueueAttentionReason::AssignedTaskLaunchUncertain { task_id }, - ) - } - ServingProgress::LaunchAttempted { .. } => None, - } -} - -/// Observe the bound return task and close the loan by its saved execution mode -/// -/// A direct-segment task closes the loan on a confirmed start. A native -/// foreground task keeps it until a successful end with a confirmed -/// process-group exit. A queued row is a launch in progress only while this actor's own supervisor -/// request is pending or inserted it; otherwise its worker may never start. After -/// closure the queue is reconciled again, so requests accepted after the return -/// reservation open the next loan under the normal rules -async fn reconcile_restore( - myself: &ActorRef, - state: &mut ResourceActorState, - snapshot: &mut ResourceSnapshot, -) -> Result, AppError> { - let pending_action = snapshot.loan.as_ref().and_then(|loan| match &loan.state { - LoanState::Active { - phase: - LoanPhase::AwaitingReturn { action_id, .. } | LoanPhase::Restoring { action_id, .. }, - } => Some(*action_id), - LoanState::Active { .. } | LoanState::NeedsAttention { .. } | LoanState::Closed { .. } => { - None - } - }); - if state - .restore_launch - .is_some_and(|attempt| Some(attempt.key.action_id) != pending_action) - { - state.restore_launch = None; - } - let Some((loan, action_id, task_id)) = restoring(snapshot) else { - return Ok(None); - }; - - let reply = call(&state.store, |reply| { - StoreMsg::ReconcileRestoringLoanForAuthority { - authority_machine: state.authority_machine, - resource_id: snapshot.resource.id, - reply, - } - }) - .await?; - let attention = |loan, action_id, task_id, reason| { - Ok(Some( - ResourceQueueReconcileOutcome::RestoreAttentionRequired { - loan, - action_id, - task_id, - reason, - }, - )) - }; - match reply { - Err(error) => { - tracing::warn!(%task_id, "restoring resource task reconciliation failed: {error}"); - attention( - loan, - action_id, - task_id, - RestoreAttentionReason::ReconcileFailed, - ) - } - Ok(RestoreReconcileOutcome::NotRestoring) => Ok(None), - Ok(RestoreReconcileOutcome::Queued { - loan, - action_id, - task_id, - }) => { - let key = RestoreLaunchKey { action_id, task_id }; - let launching = state.restore_launch.is_some_and(|attempt| { - attempt.launching(key, |result| result == RestoreLaunchResult::Inserted) - }); - if launching { - return Ok(Some(ResourceQueueReconcileOutcome::LoanAlreadyActive { - loan, - })); - } - attention( - loan, - action_id, - task_id, - RestoreAttentionReason::LaunchUncertain, - ) - } - Ok(RestoreReconcileOutcome::Attention { - loan, - action_id, - task_id, - reason, - }) => attention(loan, action_id, task_id, reason), - // the running foreground task keeps the loan, so no queued request can start - Ok(RestoreReconcileOutcome::ForegroundRunning { loan, .. }) => { - Ok(Some(ResourceQueueReconcileOutcome::LoanAlreadyActive { - loan, - })) - } - Ok( - RestoreReconcileOutcome::Closed { .. } - | RestoreReconcileOutcome::ForegroundEnded { .. }, - ) => { - state.restore_launch = None; - let outcome = call(&state.store, |reply| StoreMsg::ReconcileResourceQueue { - authority_machine: state.authority_machine, - resource_id: snapshot.resource.id, - reply, - }) - .await?; - *snapshot = load_snapshot(state).await?; - // a newly opened loan still needs its watcher and assignment steps - myself.cast(ResourceMsg::Wake)?; - Ok(Some(outcome)) - } - } -} - -/// Serve queued work from an undecided return once its window closes -/// -/// While the window is open, one wake is armed for its saved deadline. The -/// store serves the next queued request from the same loan only after the -/// deadline, so the supervisor's decision and this transition cannot both win -async fn reconcile_return_deadline( - myself: &ActorRef, - state: &mut ResourceActorState, - snapshot: &mut ResourceSnapshot, -) -> Result, AppError> { - if awaiting_return_action(snapshot).is_none() { - state.return_deadline_wake = None; - return Ok(None); - } - let served = call(&state.store, |reply| StoreMsg::ServeAfterReturnDeadline { - authority_machine: state.authority_machine, - resource_id: snapshot.resource.id, - reply, - }) - .await?; - match served { - Err(error) => { - let resource = snapshot.resource.id.as_uuid(); - tracing::warn!(%resource, "return deadline check failed: {error}"); - // without another wake, queued work could wait for an unrelated event - arm_return_wake( - myself, - state, - ReturnWakeTarget::Retry, - RETURN_DEADLINE_RETRY, - ); - Ok(None) - } - Ok(ReturnDeadlineOutcome::Open { window }) => { - let target = ReturnWakeTarget::Deadline { - action_id: window.action_id(), - deadline_at: window.deadline_at(), - }; - arm_return_wake(myself, state, target, window.remaining_at(Utc::now())); - Ok(None) - } - Ok(ReturnDeadlineOutcome::NotAwaiting | ReturnDeadlineOutcome::NoQueuedRequest) => { - // a request accepted later reconciles this actor again - state.return_deadline_wake = None; - Ok(None) - } - Ok(ReturnDeadlineOutcome::Served { loan, request }) => { - state.return_deadline_wake = None; - let resource = snapshot.resource.id.as_uuid(); - let request_id = request.request_id.0; - tracing::info!( - %resource, - %request_id, - "return decision window closed; serving the next queued request" - ); - *snapshot = load_snapshot(state).await?; - // the serving loan still needs its assignment launch - myself.cast(ResourceMsg::Wake)?; - Ok(Some(ResourceQueueReconcileOutcome::LoanAlreadyActive { - loan: *loan, - })) - } - } -} - -/// Arm one wake for `target` unless an unfired wake already covers it -fn arm_return_wake( - myself: &ActorRef, - state: &mut ResourceActorState, - target: ReturnWakeTarget, - wait: Duration, -) { - if state - .return_deadline_wake - .as_ref() - .is_some_and(|armed| armed.covers(target)) - { - return; - } - let wake = myself - .send_after(wait.max(RETURN_DEADLINE_MIN_WAIT), || ResourceMsg::Wake) - .abort_handle(); - state.return_deadline_wake = Some(ReturnDeadlineWake::new(target, wake)); -} - -fn awaiting_return_action(snapshot: &ResourceSnapshot) -> Option { - match &snapshot.loan.as_ref()?.state { - LoanState::Active { - phase: LoanPhase::AwaitingReturn { action_id, .. }, - } => Some(*action_id), - LoanState::Active { .. } | LoanState::NeedsAttention { .. } | LoanState::Closed { .. } => { - None - } - } -} - -/// Track the pending first background launch and surface its unproven states -/// -/// A queued row is a launch in progress only while this actor's own supervisor -/// request is pending or inserted it. The store registers the task on its -/// confirmed start, so a registered launch needs no tracking here. A launch -/// that ended before registration with no release proof keeps the resource -/// reserved even with an empty queue, so it replaces a `NoQueuedRequest` outcome -async fn observe_background_launch( - state: &mut ResourceActorState, - snapshot: &ResourceSnapshot, - outcome: &ResourceQueueReconcileOutcome, -) -> Result, AppError> { - let view = call(&state.store, |reply| { - StoreMsg::BackgroundLaunchForAuthority { - authority_machine: state.authority_machine, - resource_id: snapshot.resource.id, - reply, - } - }) - .await?; - state.pending_background_task = view.as_ref().and_then(BackgroundLaunchView::pending_task); - if let Some(view) = &view - && view.awaits_operator_release() - && matches!(outcome, ResourceQueueReconcileOutcome::NoQueuedRequest) - { - return Ok(Some( - ResourceQueueReconcileOutcome::BackgroundLaunchReleaseUnproven { - request_id: view.request_id, - task_id: view.task_id, - }, - )); - } - let Some(view) = view.filter(|view| view.phase == BackgroundLaunchPhase::Queued) else { - return Ok(None); - }; - let launching = state.background_launch.is_some_and(|attempt| { - attempt.launching(view.request_id, |result| { - result == BackgroundLaunchResult::Inserted - }) - }); - if launching { - return Ok(None); - } - - Ok(Some( - ResourceQueueReconcileOutcome::BackgroundLaunchUncertain { - task_id: view.task_id, - }, - )) -} - -fn restoring(snapshot: &ResourceSnapshot) -> Option<(Loan, ActionId, TaskId)> { - let loan = snapshot.loan.as_ref()?; - let LoanState::Active { - phase: - LoanPhase::Restoring { - action_id, - resume_task_id, - .. - }, - } = &loan.state - else { - return None; - }; - - Some((loan.clone(), *action_id, *resume_task_id)) -} - -fn restoring_task_id(state: &ResourceActorState) -> Option { - match &state.loan.as_ref()?.state { - LoanState::Active { - phase: LoanPhase::Restoring { resume_task_id, .. }, - } => Some(*resume_task_id), - LoanState::Active { .. } | LoanState::NeedsAttention { .. } | LoanState::Closed { .. } => { - None - } - } -} - -async fn load_snapshot(state: &ResourceActorState) -> Result { - let snapshots = call(&state.store, |reply| { - StoreMsg::ResourceSnapshotsForAuthority { - authority_machine: state.authority_machine, - reply, - } - }) - .await?; - snapshots - .into_iter() - .find(|snapshot| snapshot.resource.id == state.resource.id) - .ok_or_else(|| AppError::Internal { - message: format!( - "resource {} disappeared from its authority store during reconciliation", - state.resource.id.as_uuid() - ), - }) -} - -fn awaiting_release(snapshot: &ResourceSnapshot) -> Option<(Loan, ActionId, TaskId)> { - let loan = snapshot.loan.as_ref()?; - let LoanState::Active { - phase: - LoanPhase::AwaitingRelease { - action_id, - observed_background_task, - .. - }, - } = &loan.state - else { - return None; - }; - - Some((loan.clone(), *action_id, *observed_background_task)) -} - -fn serving_assignment( - snapshot: &ResourceSnapshot, - requests: &[ResourceRequest], -) -> Option<(Loan, ResourceRequest)> { - let loan = snapshot.loan.as_ref()?; - let LoanState::Active { - phase: LoanPhase::Serving { - current_request_id, .. - }, - } = &loan.state - else { - return None; - }; - let request = requests.iter().find(|request| { - request.request_id == *current_request_id - && matches!( - request.state, - ResourceRequestState::Assigned { loan_id } if loan_id == loan.id - ) - })?; - - Some((loan.clone(), request.clone())) -} - -/// Ask the supervisor to launch the assigned task, or to fail it before launch -fn schedule_assigned_activation( - resource_actor: ActorRef, - supervisor: ActorRef, - input: ResourceTaskAcceptanceInput, - failure: Option, -) { - let request_id = input.request_id; - let task_id = input.task_id; - tokio::spawn(async move { - let input = Box::new(input); - let result = match failure { - None => { - call(&supervisor, |reply| { - SupervisorMsg::LaunchAssignedResourceTask { input, reply } - }) - .await - } - Some(failure) => { - call(&supervisor, |reply| { - SupervisorMsg::FailAssignedResourceTaskBeforeLaunch { - input, - failure, - reply, - } - }) - .await - } - }; - let result = match result { - Ok(Ok(ResourceTaskAcceptance::Inserted { task })) if task == task_id => { - ResourceTaskActivationResult::Inserted - } - Ok(Ok(ResourceTaskAcceptance::Unlaunchable { task, .. })) if task == task_id => { - ResourceTaskActivationResult::FailedBeforeLaunch - } - Ok(Ok(ResourceTaskAcceptance::Existing { task, state })) if task == task_id => { - ResourceTaskActivationResult::Existing { state } - } - Ok(Err(crate::resource::store::ResourceStoreError::Prevented)) => { - ResourceTaskActivationResult::Prevented - } - _ => ResourceTaskActivationResult::Uncertain, - }; - if let Err(error) = resource_actor.cast(ResourceMsg::ActivationFinished { - request_id, - task_id, - result, - }) { - tracing::debug!(%task_id, "resource activation result cast: {error}"); - } - }); -} - -async fn handle_activation_finished( - myself: &ActorRef, - state: &mut ResourceActorState, - request_id: RequestId, - task_id: TaskId, - result: ResourceTaskActivationResult, -) -> Result<(), ActorProcessingErr> { - let key = ActivationKey { - request_id, - task_id, - }; - let recorded = state - .activation_attempt - .as_mut() - .is_some_and(|attempt| attempt.finish(key, result)); - if !recorded { - return Ok(()); - } - reconcile_and_refresh(myself, state).await?; - - Ok(()) -} - -/// Bind, baseline, and request the one watcher for a running trainer -/// -/// Every step reuses saved identities first, so a retry or restart binds the same -/// watcher task and request instead of allocating replacements. A co-located -/// supervisor's watcher is launched here; a remote supervisor's machine saves the -/// callback route first and then asks the authority to accept the same identity -/// The supervisor is called from a detached task because it may be waiting on -/// this actor -async fn progress_release_watcher( - myself: &ActorRef, - state: &mut ResourceActorState, - resource: &Resource, - failure: &ReleaseProofFailure, -) -> Result, AppError> { - let ReleaseProofFailure { - loan, - action_id, - trainer_task_id, - .. - } = failure; - let (action_id, trainer_task_id) = (*action_id, *trainer_task_id); - if state - .watcher_launch - .is_some_and(|attempt| attempt.key.action_id != action_id) - { - state.watcher_launch = None; - } - if let Some(attempt) = state.watcher_launch { - if !launch_never_committed(state, attempt).await? { - return watcher_status(state, attempt).await.map(Some); - } - // no row exists, so the uncertain request did not accept this watcher and - // asking again with the same saved identity cannot start a second worker - state.watcher_launch = None; - } - - let (intent, executable) = - match bind_release_watcher(state, resource, loan, action_id, trainer_task_id).await? { - WatcherBinding::NotNeeded => return Ok(None), - WatcherBinding::Attention(reason) => { - return Ok(Some(ReleaseWatcherStatus::Attention { action_id, reason })); - } - WatcherBinding::Bound { intent, executable } => (intent, executable), - }; - - let watcher_task_id = intent.watcher_task_id.as_task_id(); - if resource.supervisor.machine != state.authority_machine { - return remote_watcher_status(state, resource, action_id, watcher_task_id) - .await - .map(Some); - } - let attempt = WatcherLaunchAttempt::pending(WatcherLaunchKey { - action_id, - watcher_task_id, - }); - state.watcher_launch = Some(attempt); - schedule_release_watcher_launch( - myself.clone(), - state.supervisor.clone(), - ReleaseWatcherLaunch { - authority_machine: state.authority_machine, - resource_id: resource.id, - supervisor: resource.supervisor, - intent, - executable, - }, - ); - - watcher_status(state, attempt).await.map(Some) -} - -/// Result of binding and baselining the watcher for one release action -enum WatcherBinding { - /// The trainer is not running, so no watcher is needed now - NotNeeded, - /// The watcher cannot be bound, and the loan stays reserved - Attention(ReleaseWatcherAttentionReason), - /// The saved watcher identity and a verified checkpoint baseline exist - Bound { - /// Saved watcher intent - intent: ReleaseWatcherIntent, - /// Executable placed in the canonical watcher command - executable: std::path::PathBuf, - }, -} - -/// Bind the watcher identity to the release action and capture its checkpoint baseline -/// -/// The authority allocates the watcher and request identities once and binds -/// them before any launch, so co-located and remote supervisors use one path -async fn bind_release_watcher( - state: &ResourceActorState, - resource: &Resource, - loan: &Loan, - action_id: ActionId, - trainer_task_id: TaskId, -) -> Result { - let trainer = call(&state.store, |reply| StoreMsg::GetTask { - id: trainer_task_id, - reply, - }) - .await?; - if trainer.is_none_or(|row| row.status() != ProcessStatus::Running) { - return Ok(WatcherBinding::NotNeeded); - } - let association = call(&state.store, |reply| { - StoreMsg::TrainerAttemptAssociationForTaskForAuthority { - authority_machine: state.authority_machine, - task_id: trainer_task_id, - reply, - } - }) - .await?; - if !matches!( - association, - Ok(Some(ref association)) if association.resource_id() == resource.id - ) { - return Ok(WatcherBinding::Attention( - ReleaseWatcherAttentionReason::TrainerAssociationMissing { - task_id: trainer_task_id, - }, - )); - } - let Ok(executable) = release_watcher_executable() else { - return Ok(WatcherBinding::Attention( - ReleaseWatcherAttentionReason::ExecutableUnavailable, - )); - }; - - let saved_intent = match &loan.state { - LoanState::Active { - phase: LoanPhase::AwaitingRelease { watcher_intent, .. }, - } => watcher_intent.clone(), - LoanState::Active { .. } | LoanState::NeedsAttention { .. } | LoanState::Closed { .. } => { - return Ok(WatcherBinding::NotNeeded); - } - }; - let intent = match saved_intent { - Some(intent) => intent, - None => { - let command = ReleaseWatcherCommand { - resource_id: resource.id, - action_id, - state_revision: resource.state_revision, - trainer_task_id, - watcher_task_id: ReleaseWatcherTaskId::new(TaskId::new()), - }; - let Ok(normalized_spec_sha256) = - command.normalized_spec_sha256(&executable, resource.supervisor.thread) - else { - return Ok(WatcherBinding::Attention( - ReleaseWatcherAttentionReason::ExecutableUnavailable, - )); - }; - let intent = ReleaseWatcherIntent { - action_id, - state_revision: resource.state_revision, - observed_background_task: trainer_task_id, - watcher_task_id: command.watcher_task_id, - request_id: RequestId::new(), - normalized_spec_sha256, - }; - let bound = call(&state.store, |reply| { - StoreMsg::BindReleaseWatcherForAuthority { - authority_machine: state.authority_machine, - resource_id: resource.id, - intent, - reply, - } - }) - .await?; - match bound { - Ok(intent) => intent, - Err(error) => { - tracing::warn!(resource = %resource.id.as_uuid(), action = %action_id.as_uuid(), "bind release watcher: {error}"); - return Ok(WatcherBinding::Attention( - ReleaseWatcherAttentionReason::BindingRejected, - )); - } - } - } - }; - - let baseline = call(&state.store, |reply| { - StoreMsg::CaptureReleaseCheckpointBaselineForAuthority { - authority_machine: state.authority_machine, - resource_id: resource.id, - action_id, - expected_state_revision: intent.state_revision, - reply, - } - }) - .await?; - if let Err(error) = baseline { - tracing::warn!(resource = %resource.id.as_uuid(), action = %action_id.as_uuid(), "capture release checkpoint baseline: {error}"); - return Ok(WatcherBinding::Attention( - ReleaseWatcherAttentionReason::BaselineUnavailable, - )); - } - - Ok(WatcherBinding::Bound { intent, executable }) -} - -/// Observe a remote supervisor's watcher without launching it from this actor -/// -/// No row means the supervisor machine has not launched it. A row that this actor -/// lifetime did not see accepted is observed like any existing acceptance, so a -/// queued row whose spawn may have been lost needs attention -async fn remote_watcher_status( - state: &ResourceActorState, - resource: &Resource, - action_id: ActionId, - watcher_task_id: TaskId, -) -> Result { - let row = call(&state.store, |reply| StoreMsg::GetTask { - id: watcher_task_id, - reply, - }) - .await?; - if row.is_none() { - return Ok(ReleaseWatcherStatus::AwaitingRemoteSupervisor { - action_id, - watcher_task_id, - supervisor_machine: resource.supervisor.machine, - }); - } - watcher_status( - state, - WatcherLaunchAttempt { - key: WatcherLaunchKey { - action_id, - watcher_task_id, - }, - progress: LaunchProgress::Finished(ReleaseWatcherLaunchResult::Uncertain), - }, - ) - .await -} - -/// Validate a remote supervisor's action, then return the bound canonical watcher -async fn prepare_remote_watcher( - state: &ResourceActorState, - authority: SupervisorActionAuthority, - observed_background_task: TaskId, -) -> Result, AppError> { - let snapshot = load_snapshot(state).await?; - let resource = &snapshot.resource; - if resource.supervisor != authority.supervisor - || resource.assignment_revision != authority.assignment_revision - || resource.supervisor.machine == state.authority_machine - { - return Ok(Err(ResourceActionRejection::NotCurrentSupervisor)); - } - if resource.state_revision != authority.expected_state_revision { - return Ok(Err(ResourceActionRejection::StaleRevision { - expected: authority.expected_state_revision, - actual: resource.state_revision, - })); - } - let Some((loan, action_id, trainer_task_id)) = awaiting_release(&snapshot) else { - return Ok(Err(ResourceActionRejection::ActionNotPending)); - }; - if loan.id != authority.loan_id - || action_id != authority.action_id - || trainer_task_id != observed_background_task - { - return Ok(Err(ResourceActionRejection::ActionNotPending)); - } - - let unavailable = - |reason: String| Ok(Err(ResourceActionRejection::WatcherUnavailable { reason })); - let (intent, executable) = - match bind_release_watcher(state, resource, &loan, action_id, trainer_task_id).await? { - WatcherBinding::NotNeeded => { - return unavailable("the observed background task is not running".into()); - } - WatcherBinding::Attention(reason) => return unavailable(format!("{reason:?}")), - WatcherBinding::Bound { intent, executable } => (intent, executable), - }; - let spec = ReleaseWatcherCommand::from_intent(resource.id, &intent) - .normalized_spec(&executable, resource.supervisor.thread)?; - let normalized_spec_sha256 = normalized_spec_sha256(&spec)?; - // a changed daemon executable cannot silently replace the bound command - if normalized_spec_sha256 != intent.normalized_spec_sha256 { - return unavailable("the watcher executable differs from the bound command".into()); - } - - Ok(Ok(PreparedActionTask { - request_id: intent.request_id, - task_id: intent.watcher_task_id.as_task_id(), - spec, - normalized_spec_sha256, - })) -} - -fn schedule_release_watcher_launch( - resource_actor: ActorRef, - supervisor: ActorRef, - launch: ReleaseWatcherLaunch, -) { - let action_id = launch.intent.action_id; - let watcher_task_id = launch.intent.watcher_task_id.as_task_id(); - tokio::spawn(async move { - let result = call(&supervisor, |reply| { - SupervisorMsg::LaunchBoundReleaseWatcher { - launch: Box::new(launch), - reply, - } - }) - .await; - let result = match result { - Ok(Ok(ReleaseWatcherAcceptance::Inserted { task })) if task == watcher_task_id => { - ReleaseWatcherLaunchResult::Inserted - } - Ok(Ok(ReleaseWatcherAcceptance::Existing { task, state })) - if task == watcher_task_id => - { - ReleaseWatcherLaunchResult::Existing { - state: state.status(), - } - } - Ok(Ok(ReleaseWatcherAcceptance::UnsupportedRemoteSupervisor { .. })) => { - ReleaseWatcherLaunchResult::UnsupportedRemoteSupervisor - } - Ok(Err( - ReleaseWatcherAcceptanceError::Conflict - | ReleaseWatcherAcceptanceError::Resource(_) - | ReleaseWatcherAcceptanceError::Identity(_) - | ReleaseWatcherAcceptanceError::Checkpoint(_), - )) => ReleaseWatcherLaunchResult::Rejected, - Ok(Ok(_) | Err(_)) | Err(_) => ReleaseWatcherLaunchResult::Uncertain, - }; - if let Err(error) = resource_actor.cast(ResourceMsg::WatcherLaunchFinished { - action_id, - watcher_task_id, - result, - }) { - tracing::debug!(%watcher_task_id, "release watcher launch result cast: {error}"); - } - }); -} - -async fn handle_watcher_launch_finished( - state: &mut ResourceActorState, - action_id: ActionId, - watcher_task_id: TaskId, - result: ReleaseWatcherLaunchResult, -) -> Result<(), ActorProcessingErr> { - let key = WatcherLaunchKey { - action_id, - watcher_task_id, - }; - let Some(attempt) = state.watcher_launch.as_mut() else { - return Ok(()); - }; - if !attempt.finish(key, result) { - return Ok(()); - } - let attempt = *attempt; - state.release_watcher = Some(watcher_status(state, attempt).await?); - - Ok(()) -} - -async fn launch_never_committed( - state: &ResourceActorState, - attempt: WatcherLaunchAttempt, -) -> Result { - if attempt.progress != LaunchProgress::Finished(ReleaseWatcherLaunchResult::Uncertain) { - return Ok(false); - } - let row = call(&state.store, |reply| StoreMsg::GetTask { - id: attempt.key.watcher_task_id, - reply, - }) - .await?; - - Ok(row.is_none()) -} - -/// Derive the watcher status from the launch reply and the durable task row -/// -/// A queued row is only a launch in progress when this actor's own request inserted -/// it; otherwise nobody can tell whether its worker ever started -async fn watcher_status( - state: &ResourceActorState, - attempt: WatcherLaunchAttempt, -) -> Result { - let WatcherLaunchAttempt { - key: WatcherLaunchKey { - action_id, - watcher_task_id, - }, - progress, - } = attempt; - let attention = |reason| ReleaseWatcherStatus::Attention { action_id, reason }; - let LaunchProgress::Finished(result) = progress else { - return Ok(ReleaseWatcherStatus::Launching { - action_id, - watcher_task_id, - }); - }; - let inserted = match result { - ReleaseWatcherLaunchResult::UnsupportedRemoteSupervisor => { - return Ok(attention( - ReleaseWatcherAttentionReason::RemoteSupervisorUnsupported { - supervisor_machine: state.resource.supervisor.machine, - }, - )); - } - ReleaseWatcherLaunchResult::Rejected => { - return Ok(attention(ReleaseWatcherAttentionReason::LaunchRejected { - watcher_task_id, - })); - } - ReleaseWatcherLaunchResult::Inserted => true, - ReleaseWatcherLaunchResult::Existing { .. } | ReleaseWatcherLaunchResult::Uncertain => { - false - } - }; - - let row = call(&state.store, |reply| StoreMsg::GetTask { - id: watcher_task_id, - reply, - }) - .await?; - let Some(row) = row else { - return Ok(attention(ReleaseWatcherAttentionReason::LaunchUncertain { - watcher_task_id, - })); - }; - Ok(match row.status() { - ProcessStatus::Queued if inserted => ReleaseWatcherStatus::Launching { - action_id, - watcher_task_id, - }, - ProcessStatus::Queued => { - attention(ReleaseWatcherAttentionReason::LaunchUncertain { watcher_task_id }) - } - ProcessStatus::Running => ReleaseWatcherStatus::Running { - action_id, - watcher_task_id, - }, - ProcessStatus::Succeeded => ReleaseWatcherStatus::Finished { - action_id, - watcher_task_id, - }, - status @ (ProcessStatus::Failed | ProcessStatus::Cancelled | ProcessStatus::Lost) => { - attention(ReleaseWatcherAttentionReason::WatcherTaskEnded { - watcher_task_id, - state: status, - }) - } - }) -} - -fn release_proof_attention_reason(error: CompleteReleaseError) -> ReleaseProofAttentionReason { - match error { - CompleteReleaseError::BackgroundTaskNotTerminal { .. } => { - ReleaseProofAttentionReason::TrainerNotCompleted - } - CompleteReleaseError::BackgroundTaskLost { .. } => ReleaseProofAttentionReason::TrainerLost, - CompleteReleaseError::WorkerExitUnconfirmed { .. } => { - ReleaseProofAttentionReason::WorkerExitUnconfirmed - } - CompleteReleaseError::CompletedResultChanged { .. } - | CompleteReleaseError::CompletedResultRequestMismatch { .. } - | CompleteReleaseError::Watcher(_) => { - ReleaseProofAttentionReason::CompletedResultUnavailable - } - CompleteReleaseError::StoppedCheckpointChanged { .. } => { - ReleaseProofAttentionReason::StoppedCheckpointUnavailable - } - CompleteReleaseError::OwnershipLockStillHeld { .. } - | CompleteReleaseError::OwnershipLock(_) => { - ReleaseProofAttentionReason::OwnershipLockUnverified - } - // no saved lock names the worker, so only an operator can resolve it - CompleteReleaseError::TrainerAssociationMissing { .. } => { - ReleaseProofAttentionReason::TrainerAssociationMissing - } - CompleteReleaseError::Resource(_) - | CompleteReleaseError::Notice(_) - | CompleteReleaseError::TrainerAssociation(_) - | CompleteReleaseError::Identity(_) - | CompleteReleaseError::TaskStorage(_) - | CompleteReleaseError::CommandShape(_) - | CompleteReleaseError::ActionNotFound { .. } - | CompleteReleaseError::NotAwaitingRelease { .. } - | CompleteReleaseError::InvalidReleaseNotice { .. } - | CompleteReleaseError::StaleRevision { .. } - | CompleteReleaseError::BackgroundTaskMissing { .. } - | CompleteReleaseError::TrainerAssociationMismatch { .. } - | CompleteReleaseError::TrainerIdentityMissing { .. } - | CompleteReleaseError::TrainerIdentityChanged { .. } - | CompleteReleaseError::StoppedProofUnavailable { .. } - | CompleteReleaseError::TrainerCancellationMarkerChanged { .. } - | CompleteReleaseError::Checkpoint(_) - | CompleteReleaseError::TaskStateChanged { .. } - | CompleteReleaseError::TaskCommandChanged { .. } - | CompleteReleaseError::RevisionExhausted { .. } - | CompleteReleaseError::ConflictingRetry { .. } - | CompleteReleaseError::RequestChanged { .. } - | CompleteReleaseError::LoanChanged { .. } - | CompleteReleaseError::Storage(_) => ReleaseProofAttentionReason::SavedEvidenceMismatch, - } -} - -fn task_completion_attention_reason( - task_id: TaskId, - attention: AssignedResourceTaskAttention, -) -> ResourceQueueAttentionReason { - match attention { - AssignedResourceTaskAttention::RequestChanged - | AssignedResourceTaskAttention::LoanChanged - | AssignedResourceTaskAttention::TaskIdentityMismatch => { - ResourceQueueAttentionReason::AssignedTaskIdentityMismatch { task_id } - } - AssignedResourceTaskAttention::StaleRevision => { - ResourceQueueAttentionReason::AssignedTaskStaleRevision { task_id } - } - AssignedResourceTaskAttention::ServingReleaseUnverified => { - ResourceQueueAttentionReason::UnverifiedServingRelease - } - AssignedResourceTaskAttention::TaskLost => { - ResourceQueueAttentionReason::AssignedTaskLost { task_id } - } - AssignedResourceTaskAttention::ExitWitnessUnconfirmed => { - ResourceQueueAttentionReason::AssignedTaskExitUnconfirmed { task_id } - } - AssignedResourceTaskAttention::InvalidNoChildSpawnEvidence => { - ResourceQueueAttentionReason::AssignedTaskNoChildSpawnProofInvalid { task_id } - } - AssignedResourceTaskAttention::OwnershipUncertain(risk) => { - ResourceQueueAttentionReason::AssignedTaskOwnershipUncertain { task_id, risk } - } - } -} - -async fn serving_task_id(state: &ResourceActorState) -> Result, AppError> { - let Some(loan) = state.loan.as_ref() else { - return Ok(None); - }; - let (request_id, loan_id) = match &loan.state { - LoanState::Active { - phase: LoanPhase::Serving { - current_request_id, .. - }, - } => (*current_request_id, loan.id), - LoanState::Active { .. } | LoanState::NeedsAttention { .. } | LoanState::Closed { .. } => { - return Ok(None); - } - }; - let requests = call(&state.store, |reply| StoreMsg::ResourceRequests { - authority_machine: state.authority_machine, - resource_id: state.resource.id, - reply, - }) - .await?; - Ok(requests - .into_iter() - .find(|request| { - request.request_id == request_id - && matches!( - &request.state, - ResourceRequestState::Assigned { loan_id: assigned_loan } - if *assigned_loan == loan_id - ) - }) - .map(|request| request.task_id)) -} - -/// Stable actor name for one resource identity -#[must_use] -pub(crate) fn resource_actor_name(id: ResourceId) -> String { - format!("homebased.resource.{}", id.as_uuid()) -} - -/// Parse one resource actor name into its stable resource identity -#[must_use] -pub(crate) fn resource_id_from_actor_name(name: Option) -> Option { - let name = name?; - let uuid = name.strip_prefix("homebased.resource.")?.parse().ok()?; - ResourceId::from_uuid(uuid).ok() -} - -#[cfg(test)] -mod tests; - -#[cfg(test)] -pub(crate) mod test_support; diff --git a/src/daemon/actors/resource/test_support.rs b/src/daemon/actors/resource/test_support.rs deleted file mode 100644 index 673ca5b..0000000 --- a/src/daemon/actors/resource/test_support.rs +++ /dev/null @@ -1,42 +0,0 @@ -//! Test doubles for resource actor tests - -use ractor::concurrency::JoinHandle; -use ractor::{Actor, ActorProcessingErr, ActorRef}; - -use crate::daemon::actors::SupervisorMsg; - -/// Supervisor stand-in that drops every request unanswered -/// -/// A dropped reply port reads as a failed call, so each launch the resource actor -/// asks for settles as uncertain and no task is ever started -pub(crate) struct StubSupervisor; - -impl Actor for StubSupervisor { - type Msg = SupervisorMsg; - type State = (); - type Arguments = (); - - async fn pre_start( - &self, - _myself: ActorRef, - (): Self::Arguments, - ) -> Result { - Ok(()) - } - - async fn handle( - &self, - _myself: ActorRef, - _message: Self::Msg, - _state: &mut Self::State, - ) -> Result<(), ActorProcessingErr> { - Ok(()) - } -} - -/// Spawn an unnamed stub supervisor for one resource actor test -pub(crate) async fn spawn_stub_supervisor() -> (ActorRef, JoinHandle<()>) { - StubSupervisor::spawn(None, StubSupervisor, ()) - .await - .expect("spawn stub supervisor") -} diff --git a/src/daemon/actors/resource/tests.rs b/src/daemon/actors/resource/tests.rs deleted file mode 100644 index b403603..0000000 --- a/src/daemon/actors/resource/tests.rs +++ /dev/null @@ -1,667 +0,0 @@ -use crate::resource::{Loan, SupervisorNoticePayload}; -use crate::spec::NormalizedWorkload; -use std::path::PathBuf; - -use serde_json::json; -use tempfile::tempdir; -use uuid::Uuid; - -use super::test_support::spawn_stub_supervisor; -use super::{ - ResourceActor, ResourceActorInspection, ResourceMsg, resource_actor_name, - resource_id_from_actor_name, -}; -use crate::daemon::actors::{StoreActor, StoreMsg, call}; -use crate::domain::{ExitReason, ProcessStatus, TaskEnv, TaskId, TaskWorkload, ThreadId, Workload}; -use crate::home::Home; -use crate::machine::{MachineId, load_or_create_machine_id}; -use crate::resource::store::{ResourceTaskAcceptance, ResourceTaskAcceptanceInput}; -use crate::resource::{ - AssignmentRevision, LoanPhase, LoanState, Resource, ResourceId, ResourceQueueAttentionReason, - ResourceQueueReconcileOutcome, ResourceRequestState, ResourceRevision, ReturnContext, - SupervisorAddress, -}; -use crate::spec::NormalizedSpec; -use crate::store::{NewTask, Store, new_queued_task}; -use crate::submission::RequestId; -use ractor::{Actor, ActorRef}; - -fn resource(authority: MachineId, background_task: Option) -> Resource { - Resource::new( - ResourceId::new(), - "gpu-test".into(), - authority, - SupervisorAddress { - machine: authority, - thread: ThreadId(Uuid::now_v7()), - }, - AssignmentRevision::new(0), - ResourceRevision::new(0), - background_task, - ) -} - -fn command_spec() -> NormalizedSpec { - serde_json::from_value(json!({ - "api_version": 1, - "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "resource reconcile test", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": ["/bin/echo", "hello"] } - })) - .unwrap() -} - -fn seed_queue(home: &Home, task_status: Option) -> (MachineId, Resource, RequestId) { - let authority = load_or_create_machine_id(home).unwrap(); - let background_task = task_status.map(|_| TaskId::new()); - let resource = resource(authority, background_task); - let mut store = Store::open(&home.db_path()).unwrap(); - store.register_resource(authority, &resource).unwrap(); - - if let (Some(task_id), Some(status)) = (background_task, task_status) { - let spec = command_spec(); - let NormalizedWorkload::Task(workload) = spec.workload.clone() else { - panic!("resource test must use a command workload"); - }; - let row = new_queued_task(NewTask { - id: task_id, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: Workload::Task(TaskWorkload { - command: workload.command, - }), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - binary: PathBuf::from("/bin/echo"), - }); - store.insert_task(&row).unwrap(); - match status { - ProcessStatus::Queued => {} - ProcessStatus::Running => { - store - .cas_status(task_id, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - } - ProcessStatus::Lost => { - store - .cas_status(task_id, ProcessStatus::Queued, ProcessStatus::Lost) - .unwrap() - .unwrap(); - } - other => { - let reason = match other { - ProcessStatus::Succeeded | ProcessStatus::Failed => { - ExitReason::Exit { code: 0 } - } - ProcessStatus::Cancelled => ExitReason::Cancelled, - ProcessStatus::Queued | ProcessStatus::Running | ProcessStatus::Lost => { - unreachable!() - } - }; - store - .cas_exit(task_id, ProcessStatus::Queued, &reason) - .unwrap() - .unwrap(); - } - } - } - - let request_id = RequestId::new(); - store - .accept_resource_request( - authority, - request_id, - TaskId::new(), - resource.id, - authority, - command_spec(), - ) - .unwrap(); - drop(store); - (authority, resource, request_id) -} - -fn seed_serving_task(home: &Home) -> (MachineId, Resource, RequestId, TaskId) { - let authority = load_or_create_machine_id(home).unwrap(); - let spec = command_spec(); - let background_task = TaskId::new(); - let request_id = RequestId::new(); - let task_id = TaskId::new(); - let origin = MachineId::new(); - let mut resource = resource(authority, Some(background_task)); - resource.supervisor.thread = spec.thread; - let mut store = Store::open(&home.db_path()).unwrap(); - store.register_resource(authority, &resource).unwrap(); - - let NormalizedWorkload::Task(workload) = spec.workload.clone() else { - panic!("resource task fixture must use a command workload"); - }; - store - .insert_task(&new_queued_task(NewTask { - id: background_task, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: Workload::Task(TaskWorkload { - command: workload.command, - }), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - binary: PathBuf::from("/bin/echo"), - })) - .unwrap(); - store - .cas_status( - background_task, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - let request = store - .accept_resource_request( - authority, - request_id, - task_id, - resource.id, - origin, - spec.clone(), - ) - .unwrap(); - assert!(matches!( - store - .open_release_loan_for_authority(authority, resource.id, resource.state_revision,) - .unwrap(), - crate::resource::store::OpenReleaseLoanResult::Opened { .. } - )); - store - .cas_exit( - background_task, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ) - .unwrap() - .unwrap(); - let (loan, state_revision) = store - .seed_verified_serving_loan_for_test( - authority, - resource.id, - request_id, - ReturnContext::AlreadyCompleted { - task_id: background_task, - result_ref: "actor fixture".into(), - }, - ) - .unwrap(); - assert_eq!(request.task_id, task_id); - assert!(matches!( - loan.state, - LoanState::Active { - phase: LoanPhase::Serving { current_request_id, .. } - } if current_request_id == request_id - )); - assert!(matches!( - store - .accept_assigned_resource_task(ResourceTaskAcceptanceInput { - authority_machine: authority, - resource_id: resource.id, - request_id, - task_id, - acceptance_sequence: request.acceptance_sequence, - loan_id: loan.id, - expected_state_revision: state_revision, - command_spec: crate::resource::CommandSpec::try_from(spec).unwrap(), - executor_env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - }) - .unwrap(), - ResourceTaskAcceptance::Inserted { task } if task == task_id - )); - store - .cas_status(task_id, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - - (authority, resource, request_id, task_id) -} - -async fn start_resource_actor( - home: &Home, - authority: MachineId, - resource: Resource, -) -> ( - ActorRef, - ractor::concurrency::JoinHandle<()>, - ActorRef, - ractor::concurrency::JoinHandle<()>, -) { - let (store, store_handle) = StoreActor::spawn(None, StoreActor, home.db_path()) - .await - .unwrap(); - let snapshot = call(&store, |reply| StoreMsg::ResourceSnapshotsForAuthority { - authority_machine: authority, - reply, - }) - .await - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == resource.id) - .unwrap(); - let (supervisor, _supervisor_handle) = spawn_stub_supervisor().await; - let (actor, actor_handle) = ResourceActor::spawn( - None, - ResourceActor, - ( - store.clone(), - supervisor, - authority, - snapshot.resource, - snapshot.loan, - ), - ) - .await - .unwrap(); - (actor, actor_handle, store, store_handle) -} - -async fn stop_actor(actor: ActorRef, handle: ractor::concurrency::JoinHandle<()>) { - actor.stop(None); - let _ = handle.await; -} - -async fn stop_store(store: ActorRef, handle: ractor::concurrency::JoinHandle<()>) { - store.stop(None); - let _ = handle.await; -} - -#[tokio::test] -async fn exact_terminal_wake_and_actor_restart_finish_one_assigned_task() { - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let (authority, resource, request_id, task_id) = seed_serving_task(&home); - - let (actor, actor_handle, store, store_handle) = - start_resource_actor(&home, authority, resource.clone()).await; - let before_exit = call(&actor, |reply| ResourceMsg::Inspect { reply }) - .await - .unwrap(); - assert!(matches!( - before_exit.loan, - Some(Loan { - state: LoanState::Active { - phase: LoanPhase::Serving { current_request_id: current, .. } - }, - .. - }) if current == request_id - )); - - let task_store = Store::open(&home.db_path()).unwrap(); - task_store - .cas_exit_with_evidence( - task_id, - ProcessStatus::Running, - &ExitReason::Cancelled, - crate::domain::ProcessGroupExitEvidence::ConfirmedExited, - ) - .unwrap() - .unwrap(); - drop(task_store); - - actor - .cast(ResourceMsg::TaskTerminal { - task_id: TaskId::new(), - }) - .unwrap(); - actor.cast(ResourceMsg::TaskTerminal { task_id }).unwrap(); - let after_exit = call(&actor, |reply| ResourceMsg::Inspect { reply }) - .await - .unwrap(); - assert!(matches!( - after_exit.loan, - Some(Loan { - state: LoanState::Active { - phase: LoanPhase::AwaitingReturn { .. } - }, - .. - }) - )); - assert!(matches!( - call(&store, |reply| StoreMsg::ResourceRequests { - authority_machine: authority, - resource_id: resource.id, - reply, - }) - .await - .unwrap()[0] - .state, - ResourceRequestState::Finished { - outcome: ExitReason::Cancelled - } - )); - let notices = call(&store, |reply| StoreMsg::PendingSupervisorNotices { reply }) - .await - .unwrap() - .unwrap(); - assert_eq!( - notices - .iter() - .filter(|notice| matches!( - notice.payload, - SupervisorNoticePayload::ReturnRequired { .. } - )) - .count(), - 1 - ); - let committed_revision = after_exit.resource.state_revision; - - stop_actor(actor, actor_handle).await; - stop_store(store, store_handle).await; - - let (restarted, restarted_handle, reopened_store, reopened_store_handle) = - start_resource_actor(&home, authority, resource.clone()).await; - let after_restart = call(&restarted, |reply| ResourceMsg::Inspect { reply }) - .await - .unwrap(); - assert_eq!(after_restart.resource.state_revision, committed_revision); - let notices = call(&reopened_store, |reply| { - StoreMsg::PendingSupervisorNotices { reply } - }) - .await - .unwrap() - .unwrap(); - assert_eq!( - notices - .iter() - .filter(|notice| matches!( - notice.payload, - SupervisorNoticePayload::ReturnRequired { .. } - )) - .count(), - 1 - ); - - stop_actor(restarted, restarted_handle).await; - stop_store(reopened_store, reopened_store_handle).await; -} - -#[tokio::test] -async fn actor_startup_recovers_a_queued_request_into_one_release_action() { - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let (authority, resource, request_id) = seed_queue(&home, Some(ProcessStatus::Running)); - - let (actor, actor_handle, store, store_handle) = - start_resource_actor(&home, authority, resource.clone()).await; - let inspection = call(&actor, |reply| ResourceMsg::Inspect { reply }) - .await - .unwrap(); - let ResourceQueueReconcileOutcome::ReleaseProofUnavailable { loan, .. } = - inspection.reconcile_outcome.unwrap() - else { - panic!("startup must retain the running task behind the release proof gate"); - }; - assert!(matches!( - &loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { - observed_background_task, - .. - } - } if Some(*observed_background_task) == resource.registered_background_task - )); - assert_eq!(inspection.loan, Some(loan)); - assert_eq!(inspection.resource.state_revision, ResourceRevision::new(1)); - assert_eq!( - call(&store, |reply| StoreMsg::ResourceRequests { - authority_machine: authority, - resource_id: resource.id, - reply, - }) - .await - .unwrap()[0] - .request_id, - request_id - ); - - stop_actor(actor, actor_handle).await; - stop_store(store, store_handle).await; -} - -#[tokio::test] -async fn duplicate_wakes_keep_one_loan_and_one_notice() { - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let (authority, resource, _) = seed_queue(&home, Some(ProcessStatus::Running)); - - let (actor, actor_handle, store, store_handle) = - start_resource_actor(&home, authority, resource.clone()).await; - let first = call(&actor, |reply| ResourceMsg::Inspect { reply }) - .await - .unwrap(); - let first_loan = first.loan.unwrap(); - let first_loan_id = first_loan.id; - - for _ in 0..2 { - let result = call(&actor, |reply| ResourceMsg::Reconcile { reply }) - .await - .unwrap(); - assert!(matches!( - result, - ResourceQueueReconcileOutcome::ReleaseProofUnavailable { loan, .. } - if loan.id == first_loan_id - )); - } - - let notices = call(&store, |reply| StoreMsg::PendingSupervisorNotices { reply }) - .await - .unwrap(); - let notices = notices.unwrap(); - assert_eq!(notices.len(), 1); - let snapshot = call(&actor, |reply| ResourceMsg::Inspect { reply }) - .await - .unwrap(); - assert_eq!(snapshot.loan.unwrap().id, first_loan_id); - - stop_actor(actor, actor_handle).await; - stop_store(store, store_handle).await; -} - -#[tokio::test] -async fn uncertain_background_state_stays_queued_and_is_inspectable() { - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let (authority, resource, request_id) = seed_queue(&home, Some(ProcessStatus::Lost)); - - let (actor, actor_handle, store, store_handle) = - start_resource_actor(&home, authority, resource.clone()).await; - let inspection = call(&actor, |reply| ResourceMsg::Inspect { reply }) - .await - .unwrap(); - assert!(inspection.loan.is_none()); - assert!(matches!( - inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::AttentionRequired { - request, - reason: ResourceQueueAttentionReason::BackgroundTaskNotRunning { - state, - .. - }, - }) if request.request_id == request_id && state == "lost" - )); - let requests = call(&store, |reply| StoreMsg::ResourceRequests { - authority_machine: authority, - resource_id: resource.id, - reply, - }) - .await - .unwrap(); - assert!(matches!(requests[0].state, ResourceRequestState::Queued)); - assert!( - Store::open(&home.db_path()) - .unwrap() - .next_queued_resource_request(authority, resource.id) - .unwrap() - .is_some() - ); - - stop_actor(actor, actor_handle).await; - stop_store(store, store_handle).await; -} - -#[tokio::test] -async fn ended_unregistered_first_launch_stays_reserved_across_restart_until_attested() { - use crate::resource::command_shape::test_support::FakeTrainer; - use crate::resource::operator_release::{ - OperatorAttestationId, OperatorGpuFreeAttestation, OperatorGpuFreeConfirmation, - OperatorObservation, OperatorStateBinding, - }; - use crate::store::{BackgroundLaunchAcceptance, BackgroundLaunchInput}; - use crate::submission::CallbackExecutable; - - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().join("state"))).unwrap(); - home.ensure().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let trainer = FakeTrainer::new(&root); - let authority = load_or_create_machine_id(&home).unwrap(); - let resource = resource(authority, None); - let (request_id, task_id) = (RequestId::new(), TaskId::new()); - { - // the fake runner starts and ends the launch before any owner observes its start - let mut store = Store::open(&home.db_path()).unwrap(); - store.register_resource(authority, &resource).unwrap(); - let accepted = store - .accept_background_launch_for_authority(BackgroundLaunchInput { - authority_machine: authority, - resource_id: resource.id, - request_id, - task_id, - spec: trainer.spec(resource.supervisor.thread, "direct segment trainer"), - env: trainer.env.clone(), - callback_codex: CallbackExecutable::available(PathBuf::from("/bin/echo")), - }) - .unwrap(); - assert!(matches!( - accepted, - BackgroundLaunchAcceptance::Inserted { .. } - )); - store - .cas_status(task_id, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - store - .cas_exit( - task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 1 }, - ) - .unwrap() - .unwrap(); - } - let unproven = |inspection: &ResourceActorInspection| { - inspection.resource.registered_background_task.is_none() - && inspection.loan.is_none() - && matches!( - inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::BackgroundLaunchReleaseUnproven { - request_id: request, - task_id: task, - }) if request == request_id && task == task_id - ) - }; - - // an empty queue is not an idle resource, before and after a restart - for _ in 0..2 { - let (actor, actor_handle, store, store_handle) = - start_resource_actor(&home, authority, resource.clone()).await; - let inspection = call(&actor, |reply| ResourceMsg::Inspect { reply }) - .await - .unwrap(); - assert!(unproven(&inspection), "{:?}", inspection.reconcile_outcome); - stop_actor(actor, actor_handle).await; - stop_store(store, store_handle).await; - } - - let (actor, actor_handle, store, store_handle) = - start_resource_actor(&home, authority, resource.clone()).await; - let revision = call(&actor, |reply| ResourceMsg::Inspect { reply }) - .await - .unwrap() - .resource - .state_revision; - let attestation = OperatorGpuFreeAttestation { - operation_id: OperatorAttestationId::new(), - resource_id: resource.id, - authority_machine: authority, - task_id, - expected_state_revision: revision, - state_binding: OperatorStateBinding::FirstBackgroundLaunch { request_id }, - observation: OperatorObservation::try_from( - "nvidia-smi on the authority lists no trainer process".to_owned(), - ) - .unwrap(), - confirmation: OperatorGpuFreeConfirmation::OperatorConfirmedGpuFree, - }; - call(&store, |reply| StoreMsg::AttestTrainerGpuFreeForAuthority { - authority_machine: authority, - attestation: Box::new(attestation), - reply, - }) - .await - .unwrap() - .unwrap(); - let outcome = call(&actor, |reply| ResourceMsg::Reconcile { reply }) - .await - .unwrap(); - assert!(matches!( - outcome, - ResourceQueueReconcileOutcome::NoQueuedRequest - )); - stop_actor(actor, actor_handle).await; - stop_store(store, store_handle).await; -} - -#[test] -fn resource_ids_round_trip_in_actor_names() { - let id = ResourceId::new(); - assert_eq!( - resource_id_from_actor_name(Some(resource_actor_name(id))), - Some(id) - ); - assert!(resource_id_from_actor_name(Some("homebased.resource.invalid".into())).is_none()); -} - -#[tokio::test] -async fn a_fired_deadline_wake_no_longer_covers_its_deadline() { - let target = super::ReturnWakeTarget::Deadline { - action_id: crate::resource::ActionId::new(), - deadline_at: chrono::Utc::now(), - }; - let pending = tokio::spawn(std::future::pending::<()>()); - let armed = super::ReturnDeadlineWake::new(target, pending.abort_handle()); - assert!(armed.covers(target)); - assert!(!armed.covers(super::ReturnWakeTarget::Retry)); - - // a wake that fired before the wall-clock deadline must let the next reconcile arm again - let fired = tokio::spawn(async {}); - let handle = fired.abort_handle(); - fired.await.unwrap(); - let fired = super::ReturnDeadlineWake::new(target, handle); - assert!(!fired.covers(target)); -} diff --git a/src/daemon/actors/store.rs b/src/daemon/actors/store.rs index 9c17dd7..cc19f65 100644 --- a/src/daemon/actors/store.rs +++ b/src/daemon/actors/store.rs @@ -2,18 +2,16 @@ use std::collections::HashMap; use std::path::PathBuf; -use std::time::Duration; use ractor::{Actor, ActorProcessingErr, ActorRef, RpcReplyPort}; use crate::cancellation::{ CancellationReceipt, CancellationRequest, CancellationRequestIdentity, ExecutorCancelState, - ResourceCancellationRequestIdentity, }; use crate::daemon::actors::send_reply; use crate::dependency::{DependencyLookup, HeldCancellation, TaskDependencies}; use crate::domain::{ - ExitReason, ProcessStatus, TaskEnv, TaskExitEvidence, TaskId, TaskReport, TaskRow, ThreadId, + ExitReason, ProcessStatus, TaskExitEvidence, TaskId, TaskReport, TaskRow, ThreadId, }; use crate::error::AppError; use std::num::NonZeroU64; @@ -27,87 +25,16 @@ use crate::message::{ MessageAttempt, MessageDelivery, MessageId, MessageReceipt, MessageSendRequest, OutboundMessageBinding, Recipient, }; -use crate::resource::operator_release::{OperatorGpuFreeAttestation, OperatorGpuFreeResolution}; -use crate::resource::ownership_lock::VerifiedTrainerAttempt; -use crate::resource::release_watcher::{ReleaseWatcherPollOutcome, ReleaseWatcherPollRequest}; -use crate::resource::store::{ - AcceptedResourceTask, AssignedResourceTaskReconcileInput, AssignedResourceTaskReconcileOutcome, - CompleteReleaseError, OpenReleaseLoanError, PreLaunchFailure, ReleaseCheckpointError, - ReleaseCompletionResult, ReleaseWatcherAcceptance, ReleaseWatcherAcceptanceError, - ReleaseWatcherAcceptanceInput, ResourceQueueReconcileError, ResourceSnapshot, - ResourceStoreError, ResourceTaskAcceptance, ResourceTaskAcceptanceInput, ReturnDeadlineOutcome, - SupervisorNoticeStoreError, TrainerAttemptAssociationStoreError, -}; -use crate::resource::{ - ActionId, DeliveryAttemptId, NoticeId, ReleaseCheckpointBaseline, ReleaseWatcherIntent, - Resource, ResourceId, ResourceQueueReconcileOutcome, ResourceRequest, ResourceRevision, - SupervisorActionAuthority, SupervisorAddress, SupervisorNotice, TrainerAttemptAssociation, -}; -use crate::resource::{ReturnDecisionWindow, ReturnLaunch}; use crate::spec::NormalizedSpec; use crate::store::{ - AcceptedActionTask, BackgroundLaunchAcceptance, BackgroundLaunchError, BackgroundLaunchInput, - BackgroundLaunchView, EndedRestoreResolution, PreparedReturnTask, RemoteBackgroundLaunchInput, - RemoteReleaseWatcherAcceptanceInput, ResourceActionError, RestoreReconcileOutcome, - ReturnClosure, ReturnDecisionError, ReturnTaskAcceptance, ReturnTaskAcceptanceInput, -}; -use crate::store::{ - CancelResult, HeldCancel, LocalAdmission, OperatorGpuFreeError, Store, TaskPresentation, + CancelResult, HeldCancel, IdentityError, LocalAdmission, Store, TaskPresentation, UnlaunchedTask, }; -use crate::store::{IdentityError, ResourceActionRouteResult, ResourceBackgroundRouteResult}; -use crate::store::{ - ResourceControlError, ResourceControlRequest, ResourceControlStart, ResourceReadModel, - SupervisorReplacement, -}; use crate::submission::{ CallbackExecutable, DependentRoute, ExecutorIdentity, OriginRoute, RejectionTombstone, - RequestId, ResourceCancellationReceipt, ResourceQueueReceipt, SubmissionState, + RequestId, SubmissionState, }; -fn control_error( - error: ResourceControlError, - resource: ResourceId, - operation: Option, - expected: ResourceRevision, -) -> AppError { - match error { - ResourceControlError::NotFound => AppError::ResourceNotFound { resource }, - ResourceControlError::WrongAuthority { expected, found } => { - AppError::MachineIdentityMismatch { - expected, - found: Some(found), - } - } - ResourceControlError::StaleRevision { current } => AppError::ResourceStaleRevision { - resource, - expected: expected.get(), - current: current.get(), - }, - ResourceControlError::Conflict => AppError::ResourceOperationConflict { - resource, - operation, - message: "operation identity already names different content".into(), - }, - ResourceControlError::NotAllowed(message) => { - AppError::ResourceActionNotAllowed { resource, message } - } - ResourceControlError::Notice(SupervisorNoticeStoreError::Storage(error)) - | ResourceControlError::Storage(error) => error.into(), - ResourceControlError::Notice(error @ SupervisorNoticeStoreError::CorruptRecord { .. }) => { - AppError::Internal { - message: error.to_string(), - } - } - ResourceControlError::Notice(error) => AppError::ResourceActionNotAllowed { - resource, - message: error.to_string(), - }, - ResourceControlError::Json(error) => error.into(), - ResourceControlError::Resource(error) => resource_error(error, None, None), - } -} - fn identity_error(error: IdentityError) -> AppError { match error { IdentityError::Conflict => AppError::Usage { @@ -120,97 +47,6 @@ fn identity_error(error: IdentityError) -> AppError { } } -fn resource_error( - error: ResourceStoreError, - task: Option, - request: Option, -) -> AppError { - match error { - ResourceStoreError::Conflict(reason) => task.map_or_else( - || AppError::Usage { - message: format!("resource identity conflict: {reason}"), - }, - |task| AppError::ClusterTaskConflict { task }, - ), - ResourceStoreError::RegistrationConflict { resource } => { - AppError::ResourceOperationConflict { - resource, - operation: None, - message: "resource identity is already registered with different content".into(), - } - } - ResourceStoreError::Prevented => match (request, task) { - (Some(request), Some(task)) => AppError::SubmissionRejected { - request, - task, - reason: "cancelled_before_acceptance".into(), - }, - _ => AppError::Usage { - message: "resource request was prevented before acceptance".into(), - }, - }, - ResourceStoreError::ResourceNotFound => AppError::Usage { - message: "resource not found".into(), - }, - ResourceStoreError::WrongAuthority { expected, found } => { - AppError::MachineIdentityMismatch { - expected, - found: Some(found), - } - } - ResourceStoreError::OriginRouteNotFound { task } => AppError::RouteNotFound { task }, - ResourceStoreError::ExecutorAlreadyAccepted { task } => { - AppError::ClusterTaskConflict { task } - } - ResourceStoreError::Identity(IdentityError::Conflict) => task.map_or_else( - || AppError::Usage { - message: "resource executor identity conflict".into(), - }, - |task| AppError::ClusterTaskConflict { task }, - ), - ResourceStoreError::Identity(IdentityError::RouteNotFound) => AppError::Internal { - message: "executor identity disappeared during resource cancellation".into(), - }, - ResourceStoreError::Identity(IdentityError::Storage(error)) => error, - ResourceStoreError::InvalidCommandSpec(error) => AppError::Usage { - message: error.to_string(), - }, - error @ ResourceStoreError::UnsupportedCommandOwnership { .. } => AppError::Usage { - message: error.to_string(), - }, - // a definitive refusal, so the origin saves it instead of retrying an unknown outcome - ResourceStoreError::HostInputRejected(error) => match (request, task) { - (Some(request), Some(task)) => AppError::SubmissionRejected { - request, - task, - reason: error.rejection.as_str().into(), - }, - _ => error.error, - }, - ResourceStoreError::TaskPreparation(error) => error, - ResourceStoreError::TaskRow(error) => error, - ResourceStoreError::Event(error) => event_error(error), - ResourceStoreError::ReturnDecisionRead(message) => AppError::Internal { message }, - error @ ResourceStoreError::CorruptRecord { .. } => AppError::Internal { - message: error.to_string(), - }, - ResourceStoreError::Storage(error) => error.into(), - } -} - -fn resource_queue_reconcile_error(error: ResourceQueueReconcileError) -> AppError { - match error { - ResourceQueueReconcileError::Resource(error) - | ResourceQueueReconcileError::Release(OpenReleaseLoanError::Resource(error)) => { - resource_error(error, None, None) - } - ResourceQueueReconcileError::Storage(error) => error.into(), - ResourceQueueReconcileError::Release(error) => AppError::Internal { - message: format!("resource queue reconciliation failed: {error}"), - }, - } -} - fn event_error(error: EventError) -> AppError { match error { EventError::RouteNotFound { task } => AppError::RouteNotFound { task }, @@ -228,406 +64,6 @@ pub(crate) enum StoreMsg { machine: MachineId, reply: RpcReplyPort>, }, - /// Register or reuse one resource on its fixed authority - RegisterResource { - authority_machine: MachineId, - resource: Box, - reply: RpcReplyPort>, - }, - /// Read durable resource state owned by this authority - ResourceReadModels { - /// Local authority identity - authority_machine: MachineId, - /// One resource, or every resource owned by the authority - resource_id: Option, - /// Resource, loan, request, and notice state - reply: RpcReplyPort, AppError>>, - }, - /// Commit one idempotent resource control before its external effect - BeginResourceControl { - /// Local authority identity - authority_machine: MachineId, - /// Stable caller retry identity - operation_id: uuid::Uuid, - /// Immutable operation content - request: Box, - /// Attempt identity to reserve when the control is a first renotify - attempt_id: DeliveryAttemptId, - /// Committed effect or typed refusal - reply: RpcReplyPort>, - }, - /// Replace one resource supervisor and retarget undelivered notices - ReplaceResourceSupervisor { - /// Local authority identity - authority_machine: MachineId, - /// Resource whose supervisor changes - resource_id: ResourceId, - /// Resource revision the caller observed - expected_revision: ResourceRevision, - /// Exact new supervisor thread - supervisor: SupervisorAddress, - /// New assignment and retargeted notices - reply: RpcReplyPort>, - }, - /// Load the local authority's resources and non-closed loans for startup - ResourceSnapshotsForAuthority { - authority_machine: MachineId, - reply: RpcReplyPort, AppError>>, - }, - /// Bind verified trainer attempt evidence to an authority-owned registered task - BindTrainerAttemptAssociationForAuthority { - /// Fixed authority recorded on the resource - authority_machine: MachineId, - /// Resource that registered the trainer task - resource_id: ResourceId, - /// Exact registered Homebased background task - task_id: TaskId, - /// Point-in-time trainer request and held-lock evidence - verified_attempt: Box, - /// Typed durable association result - reply: RpcReplyPort< - Result< - Result, - AppError, - >, - >, - }, - /// Read a historical trainer association by its exact task identity - TrainerAttemptAssociationForTaskForAuthority { - /// Fixed authority recorded on the associated resource - authority_machine: MachineId, - /// Exact historical task identity - task_id: TaskId, - /// Typed durable association result, if one exists - reply: RpcReplyPort< - Result< - Result, TrainerAttemptAssociationStoreError>, - AppError, - >, - >, - }, - /// Read exact accepted task identities from authority-owned resource assignments - AcceptedResourceTasksForAuthority { - authority_machine: MachineId, - reply: RpcReplyPort, AppError>>, - }, - /// Accept a command request and allocate its immutable acceptance identity - AcceptResourceRequest { - authority_machine: MachineId, - request_id: RequestId, - task_id: TaskId, - resource_id: ResourceId, - origin_machine: MachineId, - normalized_spec: Box, - reply: RpcReplyPort>, - }, - /// Atomically accept the exact request selected by a Serving loan as a queued task - AcceptAssignedResourceTask { - /// Selection identity, immutable spec, and executor runtime environment - input: Box, - /// Typed task-layer result inside actor and transport errors - reply: RpcReplyPort, AppError>>, - }, - /// Accept the exact request selected by a Serving loan as a task that must fail before launch - AcceptUnlaunchableResourceTask { - /// Selection identity, immutable spec, and executor runtime environment - input: Box, - /// Why the task ends before launch - failure: PreLaunchFailure, - /// Typed task-layer result inside actor and transport errors - reply: RpcReplyPort, AppError>>, - }, - /// Reconcile one exact assigned task and atomically advance only after its exit proof - AssignedResourceTaskReconcile { - /// Exact resource, loan, request, task, and expected revision to reconcile - input: AssignedResourceTaskReconcileInput, - /// Typed evidence result inside actor and transport errors - reply: RpcReplyPort< - Result, AppError>, - >, - }, - /// Read all requests for one resource in serving order - ResourceRequests { - authority_machine: MachineId, - resource_id: ResourceId, - reply: RpcReplyPort, AppError>>, - }, - /// Reconcile one resource queue from current authority-owned state - ReconcileResourceQueue { - /// Fixed authority machine recorded on the resource - authority_machine: MachineId, - /// Resource queue to reconcile - resource_id: ResourceId, - /// Typed durable-state result, including fail-closed attention reasons - reply: RpcReplyPort>, - }, - /// Read a durable resource cancellation receipt, checking its full identity - ResourceCancellationReceipt { - /// Stable resource cancellation identity - identity: ResourceCancellationRequestIdentity, - /// Saved authority result, if this identity was already handled - reply: RpcReplyPort, AppError>>, - }, - /// Cancel a queued or assigned resource request and retain its exact authority result - CancelResourceRequestWithReceipt { - /// Fixed authority that owns the resource queue - authority_machine: MachineId, - /// Stable resource cancellation identity - identity: ResourceCancellationRequestIdentity, - /// Exact validated origin-route proof - proof: crate::submission::ResourceRouteProof, - /// Durable resource cancellation result - reply: RpcReplyPort>, - }, - /// Bind a preallocated watcher launch identity to the saved release action - BindReleaseWatcherForAuthority { - /// Authority machine recorded on the resource - authority_machine: MachineId, - /// Resource whose release action owns this watcher - resource_id: ResourceId, - /// Exact action, revision, observed task, watcher task, request, and spec digest - intent: ReleaseWatcherIntent, - /// Storage result inside actor and transport errors - reply: RpcReplyPort, AppError>>, - }, - /// Capture the exact trainer checkpoint baseline before the watcher is accepted - CaptureReleaseCheckpointBaselineForAuthority { - /// Authority machine recorded on the resource - authority_machine: MachineId, - /// Resource whose release action owns the baseline - resource_id: ResourceId, - /// Stable release action identity - action_id: ActionId, - /// Resource revision observed with the release notice - expected_state_revision: ResourceRevision, - /// Typed checkpoint baseline or attention result - reply: RpcReplyPort< - Result, AppError>, - >, - }, - /// Atomically bind and persist the one co-located fixed-ID watcher acceptance - AcceptReleaseWatcherForAuthority { - /// Fixed task, callback, action, and owner data to accept - input: Box, - /// Typed storage outcome inside actor and transport errors - reply: RpcReplyPort< - Result, AppError>, - >, - }, - /// Validate one authority-local watcher poll and advance its saved stop decision - PollReleaseWatcherForAuthority { - /// Machine that must own the polled resource - authority_machine: MachineId, - /// Exact identities sent by the watcher command - request: ReleaseWatcherPollRequest, - /// Typed poll decision; errors are storage failures the watcher may retry - reply: RpcReplyPort< - Result, AppError>, - >, - }, - /// Complete a saved release action and keep the storage result typed - CompleteReleaseForAuthority { - /// Authority machine recorded on the resource - authority_machine: MachineId, - /// Resource whose saved release action is completed - resource_id: ResourceId, - /// Stable identity of the release action - action_id: ActionId, - /// Resource revision observed with the release notice - expected_state_revision: ResourceRevision, - /// Storage result inside actor and transport errors - reply: - RpcReplyPort, AppError>>, - }, - /// Close one exact AwaitingReturn loan with an explicit no-resume decision - RecordNoResumeForAuthority { - /// Exact supervisor authority for the pending return action - authority: SupervisorActionAuthority, - /// Supervisor's durable decision reason - reason: String, - /// Typed storage result inside actor and transport errors - reply: RpcReplyPort, AppError>>, - }, - /// Move the decision deadline of one exact AwaitingReturn action - HoldReturnForAuthority { - /// Exact supervisor authority for the pending return action - authority: SupervisorActionAuthority, - /// Requested decision time from now, capped by the window limit - hold: Duration, - /// Typed storage result inside actor and transport errors - reply: RpcReplyPort, AppError>>, - }, - /// Serve the next queued request once the pending return action's window closed - ServeAfterReturnDeadline { - /// Authority machine recorded on the resource - authority_machine: MachineId, - /// Resource whose AwaitingReturn loan is checked - resource_id: ResourceId, - /// Typed storage result inside actor and transport errors - reply: RpcReplyPort, AppError>>, - }, - /// Bind one fixed return task to its action and enter Restoring - AcceptReturnTaskForAuthority { - /// Authority, fixed identities, typed work, and executor context - input: Box, - /// Typed storage result inside actor and transport errors - reply: RpcReplyPort, AppError>>, - }, - /// Observe one Restoring loan and close it only on a confirmed start - ReconcileRestoringLoanForAuthority { - /// Authority machine recorded on the resource - authority_machine: MachineId, - /// Resource whose Restoring loan is observed - resource_id: ResourceId, - /// Typed storage result inside actor and transport errors - reply: RpcReplyPort, AppError>>, - }, - /// Close a Restoring loan after the supervisor resolves an early task end - ResolveEndedRestoreForAuthority { - /// Exact authority, bound task, and supervisor reason - resolution: Box, - /// Typed storage result inside actor and transport errors - reply: RpcReplyPort, AppError>>, - }, - /// Save one operator attestation that a resource with no history starts idle - AttestInitialIdleForAuthority { - /// Local daemon machine, which must be the resource authority - authority_machine: MachineId, - /// Exact operator attestation - attestation: Box, - /// Typed storage result inside actor and transport errors - reply: RpcReplyPort< - Result< - Result< - crate::resource::initial_idle::InitialIdleResolution, - crate::store::InitialIdleError, - >, - AppError, - >, - >, - }, - /// Commit one operator attestation that an ended registered trainer no longer holds its GPU - /// - /// The store saves the receipt and the queue or loan transition in one - /// transaction, or returns a typed refusal with no records - AttestTrainerGpuFreeForAuthority { - /// Local daemon machine, which must be the resource authority - authority_machine: MachineId, - /// Exact operator attestation - attestation: Box, - /// Typed storage result inside actor and transport errors - reply: - RpcReplyPort, AppError>>, - }, - /// Derive the canonical return task for a remote supervisor without binding it - PrepareReturnTaskForAuthority { - /// Exact supervisor authority for the pending return action - authority: SupervisorActionAuthority, - /// Supervisor-chosen fixed identities and typed work - launch: Box, - /// Executor environment for a supervisor-supplied command - executor_env: TaskEnv, - /// Canonical spec and digest inside actor and transport errors - reply: RpcReplyPort, AppError>>, - }, - /// Accept one remote-supervisor release watcher for its exact saved intent - AcceptRemoteReleaseWatcherForAuthority { - /// Authority, fixed identities, and canonical task - input: Box, - /// Saved receipt and whether this call inserted the task - reply: RpcReplyPort, AppError>>, - }, - /// Read the release-watcher task identities bound by non-closed loans on this authority - ReleaseWatcherTasksForAuthority { - /// Authority machine recorded on the resources - authority_machine: MachineId, - /// Bound watcher task identities - reply: RpcReplyPort, AppError>>, - }, - /// Read the task identities bound by Restoring loans on this authority - RestoringTasksForAuthority { - /// Authority machine recorded on the resources - authority_machine: MachineId, - /// Bound return task identities - reply: RpcReplyPort, AppError>>, - }, - /// Bind one first background launch; only an inserted launch may spawn - AcceptBackgroundLaunchForAuthority { - /// Fixed identities, full spec, and co-located executor context - input: Box, - /// Typed storage result inside actor and transport errors - reply: RpcReplyPort< - Result, AppError>, - >, - }, - /// Bind one remote supervisor's first background launch; only an insertion may spawn - AcceptRemoteBackgroundLaunchForAuthority { - /// Fixed identities, digest, supervisor assignment, spec, and authority environment - input: Box, - /// Typed storage result inside actor and transport errors - reply: RpcReplyPort< - Result, AppError>, - >, - }, - /// Read the latest first background launch of one resource - BackgroundLaunchForAuthority { - /// Authority machine recorded on the resource - authority_machine: MachineId, - /// Resource whose launch is read - resource_id: ResourceId, - /// Latest launch and its derived task-layer phase - reply: RpcReplyPort, AppError>>, - }, - /// Read the tasks of first background launches whose rows are still queued - QueuedBackgroundLaunchTasksForAuthority { - /// Authority machine recorded on the resources - authority_machine: MachineId, - /// Queued launch task identities - reply: RpcReplyPort, AppError>>, - }, - /// Read one durable supervisor notice - SupervisorNotice { - /// Stable notice identity - notice_id: NoticeId, - /// Storage result inside actor and transport errors - reply: RpcReplyPort< - Result, SupervisorNoticeStoreError>, AppError>, - >, - }, - /// List notices that can receive another delivery attempt - PendingSupervisorNotices { - /// Storage result inside actor and transport errors - reply: RpcReplyPort< - Result, SupervisorNoticeStoreError>, AppError>, - >, - }, - /// Reserve one exact supervisor notice delivery attempt - ReserveSupervisorNoticeAttempt { - /// Stable notice identity - notice_id: NoticeId, - /// Stable identity for this delivery attempt - attempt_id: DeliveryAttemptId, - /// Storage result inside actor and transport errors - reply: RpcReplyPort, AppError>>, - }, - /// Settle one exact supervisor notice delivery attempt - SettleSupervisorNoticeAttempt { - /// Stable notice identity - notice_id: NoticeId, - /// Stable identity for the in-flight delivery attempt - attempt_id: DeliveryAttemptId, - /// Result of the exact delivery attempt - result: Result<(), String>, - /// Storage result inside actor and transport errors - reply: RpcReplyPort, AppError>>, - }, - /// Recover in-flight supervisor notices after restart - RecoverSendingSupervisorNotices { - /// Storage result inside actor and transport errors - reply: RpcReplyPort< - Result, SupervisorNoticeStoreError>, AppError>, - >, - }, /// Save or reuse a caller-owned cancellation before network delivery InsertCancellationRequest { request: CancellationRequest, @@ -647,11 +83,6 @@ pub(crate) enum StoreMsg { receipt: CancellationReceipt, reply: RpcReplyPort>, }, - /// Settle one resource cancellation intent with its authority receipt - AcknowledgeResourceCancellation { - receipt: ResourceCancellationReceipt, - reply: RpcReplyPort>, - }, /// Store executor receipt and a possible pre-acceptance tombstone ReceiveCancellation { request: CancellationRequestIdentity, @@ -775,35 +206,6 @@ pub(crate) enum StoreMsg { UnknownOriginRoutes { reply: RpcReplyPort, AppError>>, }, - /// Find unresolved resource origin routes after a daemon restart - UnknownResourceOriginRoutes { - reply: RpcReplyPort, AppError>>, - }, - /// Find action-bound origin routes whose authority acceptance is unknown - UnknownResourceActionRoutes { - reply: RpcReplyPort, AppError>>, - }, - /// Find remote first background launch routes whose authority acceptance is unknown - UnknownResourceBackgroundRoutes { - reply: RpcReplyPort, AppError>>, - }, - /// Apply a definitive authority result to one remote first background launch route - ResolveResourceBackgroundRoute { - task: TaskId, - result: Box, - reply: RpcReplyPort>, - }, - /// Read the one action-bound origin route for a resource action - ResourceActionRouteByAction { - action: ActionId, - reply: RpcReplyPort, AppError>>, - }, - /// Apply a definitive authority result to one action-bound origin route - ResolveResourceActionRoute { - task: TaskId, - result: Box, - reply: RpcReplyPort>, - }, /// Durably allocate a request and its origin route before network send InsertOriginRoute { route: Box, @@ -870,16 +272,6 @@ pub(crate) enum StoreMsg { outcome: SubmissionState, reply: RpcReplyPort>, }, - /// Apply a definitive resource queue receipt to an origin route - ResolveResourceRoute { - receipt: ResourceQueueReceipt, - reply: RpcReplyPort>, - }, - /// Cancel an origin resource route after pre-activation authority confirmation - CancelResourceRouteBeforeLaunch { - receipt: ResourceCancellationReceipt, - reply: RpcReplyPort>, - }, /// Accept a sequenced origin event and return only after commit AcceptInboundEvent { event: Box, @@ -1039,382 +431,6 @@ impl Actor for StoreActor { StoreMsg::MigrateLegacyLocal { machine, reply } => { send_reply(reply, state.migrate_legacy_local(machine)); } - StoreMsg::ResourceReadModels { - authority_machine, - resource_id, - reply, - } => send_reply( - reply, - state - .resource_read_models(authority_machine, resource_id) - .map_err(|error| resource_error(error, None, None)), - ), - StoreMsg::BeginResourceControl { - authority_machine, - operation_id, - request, - attempt_id, - reply, - } => send_reply( - reply, - state - .begin_resource_control(authority_machine, operation_id, &request, attempt_id) - .map_err(|error| { - control_error( - error, - request.resource_id, - Some(operation_id), - request.expected_revision, - ) - }), - ), - StoreMsg::ReplaceResourceSupervisor { - authority_machine, - resource_id, - expected_revision, - supervisor, - reply, - } => send_reply( - reply, - state - .replace_resource_supervisor( - authority_machine, - resource_id, - expected_revision, - supervisor, - ) - .map_err(|error| control_error(error, resource_id, None, expected_revision)), - ), - StoreMsg::RegisterResource { - authority_machine, - resource, - reply, - } => send_reply( - reply, - state - .register_resource(authority_machine, &resource) - .map_err(|error| resource_error(error, None, None)), - ), - StoreMsg::ResourceSnapshotsForAuthority { - authority_machine, - reply, - } => send_reply( - reply, - state - .resource_snapshots_for_authority(authority_machine) - .map_err(|error| resource_error(error, None, None)), - ), - StoreMsg::BindTrainerAttemptAssociationForAuthority { - authority_machine, - resource_id, - task_id, - verified_attempt, - reply, - } => send_reply( - reply, - Ok(state.bind_trainer_attempt_association( - authority_machine, - resource_id, - task_id, - *verified_attempt, - )), - ), - StoreMsg::TrainerAttemptAssociationForTaskForAuthority { - authority_machine, - task_id, - reply, - } => send_reply( - reply, - Ok(state.trainer_attempt_association_for_task_for_authority( - authority_machine, - task_id, - )), - ), - StoreMsg::AcceptedResourceTasksForAuthority { - authority_machine, - reply, - } => send_reply( - reply, - state - .accepted_resource_tasks_for_authority(authority_machine) - .map_err(|error| resource_error(error, None, None)), - ), - StoreMsg::AcceptResourceRequest { - authority_machine, - request_id, - task_id, - resource_id, - origin_machine, - normalized_spec, - reply, - } => send_reply( - reply, - state - .accept_resource_request( - authority_machine, - request_id, - task_id, - resource_id, - origin_machine, - *normalized_spec, - ) - .map_err(|error| resource_error(error, Some(task_id), Some(request_id))), - ), - StoreMsg::AcceptAssignedResourceTask { input, reply } => { - send_reply(reply, Ok(state.accept_assigned_resource_task(*input))); - } - StoreMsg::AcceptUnlaunchableResourceTask { - input, - failure, - reply, - } => { - send_reply( - reply, - Ok(state.accept_unlaunchable_resource_task(*input, failure)), - ); - } - StoreMsg::AssignedResourceTaskReconcile { input, reply } => { - send_reply( - reply, - Ok(state.reconcile_assigned_resource_task_for_authority(input)), - ); - } - StoreMsg::ResourceRequests { - authority_machine, - resource_id, - reply, - } => send_reply( - reply, - state - .resource_requests(authority_machine, resource_id) - .map_err(|error| resource_error(error, None, None)), - ), - StoreMsg::ReconcileResourceQueue { - authority_machine, - resource_id, - reply, - } => send_reply( - reply, - state - .reconcile_resource_queue_for_authority(authority_machine, resource_id) - .map_err(resource_queue_reconcile_error), - ), - StoreMsg::ResourceCancellationReceipt { identity, reply } => { - let task = identity.task; - let request = identity.request; - send_reply( - reply, - state - .resource_cancellation_receipt(&identity) - .map_err(|error| resource_error(error, Some(task), Some(request))), - ); - } - StoreMsg::CancelResourceRequestWithReceipt { - authority_machine, - identity, - proof, - reply, - } => { - let task = identity.task; - let request = identity.request; - send_reply( - reply, - state - .cancel_resource_request_with_receipt(authority_machine, identity, proof) - .map_err(|error| resource_error(error, Some(task), Some(request))), - ); - } - StoreMsg::BindReleaseWatcherForAuthority { - authority_machine, - resource_id, - intent, - reply, - } => send_reply( - reply, - Ok(state.bind_release_watcher_for_authority( - authority_machine, - resource_id, - intent, - )), - ), - StoreMsg::CaptureReleaseCheckpointBaselineForAuthority { - authority_machine, - resource_id, - action_id, - expected_state_revision, - reply, - } => send_reply( - reply, - Ok(state.capture_release_checkpoint_baseline_for_authority( - authority_machine, - resource_id, - action_id, - expected_state_revision, - )), - ), - StoreMsg::AcceptReleaseWatcherForAuthority { input, reply } => send_reply( - reply, - Ok(state.accept_release_watcher_for_authority(*input)), - ), - StoreMsg::PollReleaseWatcherForAuthority { - authority_machine, - request, - reply, - } => send_reply( - reply, - Ok(state.poll_release_watcher_for_authority(authority_machine, request)), - ), - StoreMsg::RecordNoResumeForAuthority { - authority, - reason, - reply, - } => send_reply( - reply, - Ok(state.record_no_resume_for_authority(authority, reason)), - ), - StoreMsg::HoldReturnForAuthority { - authority, - hold, - reply, - } => send_reply(reply, Ok(state.hold_return_for_authority(authority, hold))), - StoreMsg::ServeAfterReturnDeadline { - authority_machine, - resource_id, - reply, - } => send_reply( - reply, - Ok(state.serve_after_return_deadline_for_authority(authority_machine, resource_id)), - ), - StoreMsg::AcceptReturnTaskForAuthority { input, reply } => { - send_reply(reply, Ok(state.accept_return_task_for_authority(*input))) - } - StoreMsg::ReconcileRestoringLoanForAuthority { - authority_machine, - resource_id, - reply, - } => send_reply( - reply, - Ok(state.reconcile_restoring_loan_for_authority(authority_machine, resource_id)), - ), - StoreMsg::ResolveEndedRestoreForAuthority { resolution, reply } => send_reply( - reply, - Ok(state.resolve_ended_restore_for_authority(*resolution)), - ), - StoreMsg::AttestInitialIdleForAuthority { - authority_machine, - attestation, - reply, - } => send_reply( - reply, - Ok(state.attest_initial_idle_for_authority(authority_machine, *attestation)), - ), - StoreMsg::AttestTrainerGpuFreeForAuthority { - authority_machine, - attestation, - reply, - } => send_reply( - reply, - Ok(state.attest_trainer_gpu_free_for_authority(authority_machine, *attestation)), - ), - StoreMsg::PrepareReturnTaskForAuthority { - authority, - launch, - executor_env, - reply, - } => send_reply( - reply, - Ok(state.prepare_return_task_for_authority(authority, *launch, executor_env)), - ), - StoreMsg::AcceptRemoteReleaseWatcherForAuthority { input, reply } => send_reply( - reply, - Ok(state.accept_remote_release_watcher_for_authority(*input)), - ), - StoreMsg::ReleaseWatcherTasksForAuthority { - authority_machine, - reply, - } => send_reply( - reply, - state.release_watcher_task_ids_for_authority(authority_machine), - ), - StoreMsg::AcceptBackgroundLaunchForAuthority { input, reply } => send_reply( - reply, - Ok(state.accept_background_launch_for_authority(*input)), - ), - StoreMsg::AcceptRemoteBackgroundLaunchForAuthority { input, reply } => send_reply( - reply, - Ok(state.accept_remote_background_launch_for_authority(*input)), - ), - StoreMsg::BackgroundLaunchForAuthority { - authority_machine, - resource_id, - reply, - } => send_reply( - reply, - state - .background_launch_for_authority(authority_machine, resource_id) - .map_err(|error| resource_error(error, None, None)), - ), - StoreMsg::QueuedBackgroundLaunchTasksForAuthority { - authority_machine, - reply, - } => send_reply( - reply, - state - .queued_background_launch_tasks_for_authority(authority_machine) - .map_err(|error| resource_error(error, None, None)), - ), - StoreMsg::RestoringTasksForAuthority { - authority_machine, - reply, - } => send_reply( - reply, - state - .restoring_task_ids_for_authority(authority_machine) - .map_err(|error| AppError::Internal { - message: format!("read restoring resource tasks: {error}"), - }), - ), - StoreMsg::CompleteReleaseForAuthority { - authority_machine, - resource_id, - action_id, - expected_state_revision, - reply, - } => send_reply( - reply, - Ok(state.complete_release_for_authority( - authority_machine, - resource_id, - action_id, - expected_state_revision, - )), - ), - StoreMsg::SupervisorNotice { notice_id, reply } => { - send_reply(reply, Ok(state.supervisor_notice(notice_id))); - } - StoreMsg::PendingSupervisorNotices { reply } => { - send_reply(reply, Ok(state.pending_supervisor_notices())); - } - StoreMsg::ReserveSupervisorNoticeAttempt { - notice_id, - attempt_id, - reply, - } => send_reply( - reply, - Ok(state.reserve_supervisor_notice_attempt(notice_id, attempt_id)), - ), - StoreMsg::SettleSupervisorNoticeAttempt { - notice_id, - attempt_id, - result, - reply, - } => send_reply( - reply, - Ok(state.settle_supervisor_notice_attempt(notice_id, attempt_id, result)), - ), - StoreMsg::RecoverSendingSupervisorNotices { reply } => { - send_reply(reply, Ok(state.recover_sending_supervisor_notices())); - } StoreMsg::InsertCancellationRequest { request, reply } => { send_reply(reply, state.insert_cancellation_request(request)); } @@ -1427,9 +443,6 @@ impl Actor for StoreActor { StoreMsg::AcknowledgeCancellation { receipt, reply } => { send_reply(reply, state.acknowledge_cancellation(&receipt)); } - StoreMsg::AcknowledgeResourceCancellation { receipt, reply } => { - send_reply(reply, state.acknowledge_resource_cancellation(&receipt)); - } StoreMsg::ReceiveCancellation { request, reply } => { send_reply(reply, state.receive_cancellation(request)); } @@ -1536,56 +549,6 @@ impl Actor for StoreActor { StoreMsg::UnknownOriginRoutes { reply } => { send_reply(reply, state.unknown_origin_routes().map_err(identity_error)) } - StoreMsg::UnknownResourceOriginRoutes { reply } => send_reply( - reply, - state - .unknown_resource_origin_routes() - .map_err(identity_error), - ), - StoreMsg::UnknownResourceActionRoutes { reply } => send_reply( - reply, - state - .unknown_resource_action_routes() - .map_err(identity_error), - ), - StoreMsg::UnknownResourceBackgroundRoutes { reply } => send_reply( - reply, - state - .unknown_resource_background_routes() - .map_err(identity_error), - ), - StoreMsg::ResolveResourceBackgroundRoute { - task, - result, - reply, - } => send_reply( - reply, - state - .resolve_resource_background_route(task, &result) - .map_err(|error| match error { - IdentityError::Conflict => AppError::ClusterTaskConflict { task }, - other => identity_error(other), - }), - ), - StoreMsg::ResourceActionRouteByAction { action, reply } => send_reply( - reply, - state - .resource_action_route_by_action(action) - .map_err(identity_error), - ), - StoreMsg::ResolveResourceActionRoute { - task, - result, - reply, - } => send_reply( - reply, - state - .resolve_resource_action_route(task, &result) - .map_err(|error| match error { - IdentityError::Conflict => AppError::ClusterTaskConflict { task }, - other => identity_error(other), - }), - ), StoreMsg::InsertOriginRoute { route, reply } => send_reply( reply, state.insert_origin_route(&route).map_err(identity_error), @@ -1630,28 +593,6 @@ impl Actor for StoreActor { .resolve_origin_route(id, outcome) .map_err(identity_error), ), - StoreMsg::ResolveResourceRoute { receipt, reply } => send_reply( - reply, - state - .resolve_resource_route(&receipt) - .map_err(|error| match error { - IdentityError::Conflict => { - AppError::ClusterTaskConflict { task: receipt.task } - } - other => identity_error(other), - }), - ), - StoreMsg::CancelResourceRouteBeforeLaunch { receipt, reply } => send_reply( - reply, - state - .cancel_resource_route_before_launch(&receipt) - .map_err(|error| match error { - IdentityError::Conflict => { - AppError::ClusterTaskConflict { task: receipt.task } - } - other => identity_error(other), - }), - ), StoreMsg::AcceptInboundEvent { event, reply } => { send_reply( reply, diff --git a/src/daemon/actors/supervisor.rs b/src/daemon/actors/supervisor.rs index ef2ec95..8c91cfc 100644 --- a/src/daemon/actors/supervisor.rs +++ b/src/daemon/actors/supervisor.rs @@ -1,4 +1,4 @@ -//! Root supervisor: store, callback, and per-task and per-resource actors +//! Root supervisor: store, callback, and per-task actors use std::collections::HashMap; use std::path::PathBuf; @@ -7,10 +7,6 @@ use std::sync::Arc; use ractor::{Actor, ActorProcessingErr, ActorRef, RpcReplyPort, SupervisionEvent}; use crate::daemon::actors::callback::{CallbackActor, CallbackArgs, CallbackMsg}; -use crate::daemon::actors::resource::{ - ResourceActor, ResourceActorInspection, ResourceMsg, resource_actor_name, - resource_id_from_actor_name, -}; use crate::daemon::actors::task::{TaskActor, TaskMsg, cancel_task}; use crate::daemon::actors::{StoreActor, StoreMsg, call, send_reply}; use crate::domain::{ @@ -21,34 +17,12 @@ use crate::home::{Home, LockMode, TaskPaths}; use crate::invocation::{persist_workload, resolve_agent_binary_with}; use crate::machine::{MachineId, load_or_create_machine_id}; use crate::notify::Notifier; -use crate::resource::background_launch::RemoteBackgroundLaunchReceipt; -use crate::resource::bound_action::{ResourceActionOutcome, ResourceActionRequest}; -use crate::resource::store::{ - PreLaunchFailure, ReleaseWatcherAcceptance, ReleaseWatcherAcceptanceError, ResourceSnapshot, - ResourceStoreError, ResourceTaskAcceptance, ResourceTaskAcceptanceInput, -}; -use crate::resource::{ - ReleaseWatcherIntent, Resource, ResourceId, ReturnDecision, SupervisorActionAuthority, - SupervisorAddress, -}; use crate::runner; use crate::spec::{NormalizedSpec, NormalizedWorkload}; -use crate::store::{ - BackgroundLaunchAcceptance, BackgroundLaunchError, CancelResult, EndedRestoreResolution, - LocalAdmission, NewTask, ReturnClosure, ReturnDecisionError, ReturnTaskAcceptance, - new_queued_task, -}; -use crate::submission::{CallbackExecutable, ExecutionRecord, ExecutorIdentity, RequestId}; +use crate::store::{CancelResult, LocalAdmission, NewTask, new_queued_task}; +use crate::submission::{CallbackExecutable, ExecutionRecord, ExecutorIdentity}; mod recovery; -use recovery::ResourceOwnedTasks; -mod resource_launch; - -pub(crate) use resource_launch::return_decision_rejection; -use resource_launch::{ - decide_return, fail_assigned_resource_task_before_launch, launch_assigned_resource_task, - launch_background, launch_bound_release_watcher, launch_remote_background, resource_action, -}; const STORE_NAME: &str = "homebased.store"; const CALLBACK_NAME: &str = "homebased.callback"; @@ -62,95 +36,6 @@ pub(crate) enum SupervisorMsg { GetStore { reply: RpcReplyPort, AppError>>, }, - /// Register or reuse one resource on this machine and ensure its actor exists - RegisterResource { - resource: Box, - reply: RpcReplyPort>, - }, - /// Inspect the restored snapshot and identity of one resource actor - InspectResource { - id: ResourceId, - reply: RpcReplyPort, AppError>>, - }, - /// Wake one resource actor after queue acceptance or an authority-side retry - ReconcileResource { - /// Resource whose authority-owned queue must be reconciled - id: ResourceId, - /// Reply after the resource actor has reconciled and refreshed its snapshot - reply: RpcReplyPort>, - }, - /// Accept one assigned request and launch only when this call inserts its task - LaunchAssignedResourceTask { - /// Exact Serving assignment and fixed task identity selected by the resource owner - input: Box, - /// Typed task acceptance, nested inside actor and storage errors - reply: RpcReplyPort, AppError>>, - }, - /// Accept one assigned request whose launch must fail, and fail its task if it has not started - FailAssignedResourceTaskBeforeLaunch { - /// Exact Serving assignment and fixed task identity selected by the resource owner - input: Box, - /// Why the task ends before launch - failure: PreLaunchFailure, - /// Typed task acceptance, nested inside actor and storage errors - reply: RpcReplyPort, AppError>>, - }, - /// Accept one authority-bound release watcher and spawn only when this call inserts it - LaunchBoundReleaseWatcher { - /// Saved watcher intent and the executable for its canonical command - launch: Box, - /// Typed watcher acceptance, nested inside actor and storage errors - reply: RpcReplyPort< - Result, AppError>, - >, - }, - /// Apply one supervisor return decision and spawn only a newly inserted return task - DecideReturn { - /// Exact supervisor authority for the pending return action - authority: SupervisorActionAuthority, - /// Typed no-resume or launch decision - decision: Box, - /// Typed decision result, nested inside actor and storage errors - reply: RpcReplyPort, AppError>>, - }, - /// Apply one validated operation from a remote supervisor's machine - /// - /// The cluster route has already checked the destination, source machine, and - /// saved callback-route evidence. Only an insertion by this call spawns a task - ResourceAction { - /// Validated request from the supervisor machine - request: Box, - /// Typed authority outcome inside actor and storage errors - reply: RpcReplyPort>, - }, - /// Close a Restoring loan after the supervisor resolves a return task that ended early - ResolveEndedRestore { - /// Exact authority, bound task, and supervisor reason - resolution: Box, - /// Typed closure result, nested inside actor and storage errors - reply: RpcReplyPort, AppError>>, - }, - /// Bind one first background launch and spawn only a newly inserted task - LaunchBackground { - /// Stable request, full spec, and co-located executor context - launch: Box, - /// Typed launch result, nested inside actor and storage errors - reply: RpcReplyPort< - Result, AppError>, - >, - }, - /// Bind one remote supervisor's first background launch and spawn only a new insertion - /// - /// The cluster route has already checked the destination, source machine, and - /// saved callback-route evidence - LaunchRemoteBackground { - /// Exact receipt that the supervisor machine saved, and the saved spec - launch: Box, - /// Typed launch result, nested inside actor and storage errors - reply: RpcReplyPort< - Result, AppError>, - >, - }, /// Persist a queued row under its admission, spawn its worker, and watch it Launch { row: Box, @@ -166,10 +51,10 @@ pub(crate) enum SupervisorMsg { cwd: PathBuf, reply: RpcReplyPort>, }, - /// Finish the launch of a saved local task; `None` when a resource flow owns it + /// Finish the launch of a saved local task ResumeLocal { id: TaskId, - reply: RpcReplyPort, AppError>>, + reply: RpcReplyPort>, }, /// Accept a remote task and launch only if this request won acceptance LaunchRemote { @@ -183,65 +68,6 @@ pub(crate) enum SupervisorMsg { }, /// Wake the ordered origin inbox worker after a receive commits DispatchInbox { id: TaskId }, - /// Wake resource owners after an executor event for a remote origin is ready to send - /// - /// The origin inbox is on another machine, so this is the authority-side hint - /// that an action-bound task may have reached its running boundary - RemoteOriginEvent { id: TaskId }, -} - -/// Authority-owned inputs for one release-watcher launch with a local callback route -/// -/// The supervisor builds the canonical command from these identities; the store -/// rejects any row or spec that differs from the saved intent's digest -pub(crate) struct ReleaseWatcherLaunch { - /// Machine that owns the resource and executes the watcher - pub(crate) authority_machine: MachineId, - /// Resource whose release action owns the watcher - pub(crate) resource_id: ResourceId, - /// Supervisor address saved on the resource - pub(crate) supervisor: SupervisorAddress, - /// Saved watcher task, request, action, and digest identities - pub(crate) intent: ReleaseWatcherIntent, - /// Executable placed in the canonical watcher command - pub(crate) executable: PathBuf, -} - -/// Caller inputs for one first background launch on this authority -/// -/// The supervisor preallocates the task identity; the store keeps the identity -/// saved by an earlier exact launch with the same request -pub(crate) struct BackgroundLaunch { - /// Resource whose background slot receives the task - pub(crate) resource_id: ResourceId, - /// Stable caller retry identity - pub(crate) request_id: RequestId, - /// Full normalized command spec - pub(crate) spec: NormalizedSpec, - /// Executor environment captured by the co-located supervisor - pub(crate) env: TaskEnv, - /// Directory used to find the callback Codex executable - pub(crate) callback_cwd: PathBuf, -} - -/// Remote supervisor inputs for one first background launch on this authority -/// -/// The supervisor machine fixed every identity before it sent the launch; the -/// authority supplies only its own executor environment -pub(crate) struct RemoteBackgroundLaunch { - /// Fixed identities, digest, supervisor assignment, and observed revision - pub(crate) receipt: RemoteBackgroundLaunchReceipt, - /// Full normalized spec saved in the supervisor's route - pub(crate) spec: NormalizedSpec, -} - -/// Durable result of one supervisor return decision -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) enum ReturnDecisionOutcome { - /// The no-resume decision closed the loan - Closed(Box), - /// The launch decision bound a task, or found the exact earlier binding - Launch(ReturnTaskAcceptance), } /// Validated executor-local inputs with the original normalized retry content @@ -270,7 +96,6 @@ pub(crate) struct SupervisorState { store: ActorRef, callback: ActorRef, tasks: HashMap>, - resources: HashMap>, } /// Root actor @@ -359,9 +184,7 @@ impl Actor for SupervisorActor { store, callback, tasks: HashMap::new(), - resources: HashMap::new(), }; - restore_resource_actors(&myself, &mut state).await?; recovery::recover_tasks(&myself, &mut state).await?; // a slow store must not fail startup; the callback actor's retry scan // picks these tasks up later @@ -384,79 +207,6 @@ impl Actor for SupervisorActor { ) -> Result<(), ActorProcessingErr> { match message { SupervisorMsg::GetStore { reply } => send_reply(reply, Ok(state.store.clone())), - SupervisorMsg::RegisterResource { resource, reply } => { - send_reply( - reply, - register_resource_actor(&myself, state, *resource).await, - ); - } - SupervisorMsg::InspectResource { id, reply } => { - let inspection = match state.resources.get(&id) { - Some(actor) => call(actor, |reply| ResourceMsg::Inspect { reply }) - .await - .map(Some), - None => Ok(None), - }; - send_reply(reply, inspection); - } - SupervisorMsg::ReconcileResource { id, reply } => { - let result = reconcile_resource_actor(&myself, state, id).await; - send_reply(reply, result); - } - SupervisorMsg::LaunchAssignedResourceTask { input, reply } => { - let result = launch_assigned_resource_task(&myself, state, *input).await; - send_reply(reply, result); - } - SupervisorMsg::FailAssignedResourceTaskBeforeLaunch { - input, - failure, - reply, - } => { - let result = - fail_assigned_resource_task_before_launch(&myself, state, *input, failure) - .await; - send_reply(reply, result); - } - SupervisorMsg::LaunchBoundReleaseWatcher { launch, reply } => { - let result = launch_bound_release_watcher(&myself, state, *launch).await; - send_reply(reply, result); - } - SupervisorMsg::DecideReturn { - authority, - decision, - reply, - } => { - let result = decide_return(&myself, state, authority, *decision).await; - send_reply(reply, result); - } - SupervisorMsg::ResourceAction { request, reply } => { - let result = resource_action(&myself, state, *request).await; - send_reply(reply, result); - } - SupervisorMsg::RemoteOriginEvent { id } => { - for resource in state.resources.values() { - resource.cast(ResourceMsg::TaskProgress { task_id: id })?; - } - } - SupervisorMsg::ResolveEndedRestore { resolution, reply } => { - let resource_id = resolution.authority.resource_id; - let result = call(&state.store, |reply| { - StoreMsg::ResolveEndedRestoreForAuthority { resolution, reply } - }) - .await; - if matches!(result, Ok(Ok(_))) { - wake_resource(state, resource_id); - } - send_reply(reply, result); - } - SupervisorMsg::LaunchBackground { launch, reply } => { - let result = launch_background(&myself, state, *launch).await; - send_reply(reply, result); - } - SupervisorMsg::LaunchRemoteBackground { launch, reply } => { - let result = launch_remote_background(&myself, state, *launch).await; - send_reply(reply, result); - } SupervisorMsg::Launch { row, spec, @@ -479,10 +229,6 @@ impl Actor for SupervisorActor { } SupervisorMsg::DispatchInbox { id } => { state.callback.cast(CallbackMsg::DispatchInbox { id })?; - // a delivered state event may confirm the start of a bound return task - for resource in state.resources.values() { - resource.cast(ResourceMsg::TaskProgress { task_id: id })?; - } } } Ok(()) @@ -505,31 +251,15 @@ impl Actor for SupervisorActor { tracing::error!(actor = ?name, "daemon actor failed; stopping serve: {err}"); return Err(err); } - if let Some(id) = resource_id_from_actor_name(name.clone()) { - tracing::error!(actor = ?name, resource = ?id, "resource actor failed; stopping serve: {err}"); - return Err(err); - } tracing::error!(actor = ?name, "actor failed: {err}"); if let Some(id) = task_id_from_name(name) { state.tasks.remove(&id); spawn_task_actor(&myself, state, id).await?; } } - SupervisionEvent::ActorTerminated(who, _, reason) => { - let name = who.get_name(); - if let Some(id) = resource_id_from_actor_name(name.clone()) { - let reason = reason.unwrap_or_else(|| "without an exit reason".into()); - tracing::error!(actor = ?name, resource = ?id, %reason, "resource actor terminated; stopping serve"); - return Err(Box::new(AppError::Internal { - message: format!( - "resource actor {} terminated unexpectedly: {reason}", - id.as_uuid() - ), - })); - } + SupervisionEvent::ActorTerminated(who, _, _) => { if let Some(id) = task_id_from_name(who.get_name()) { state.tasks.remove(&id); - reconcile_terminal_task(&myself, state, id).await?; } } _ => {} @@ -538,182 +268,6 @@ impl Actor for SupervisorActor { } } -async fn reconcile_terminal_task( - supervisor: &ActorRef, - state: &mut SupervisorState, - task_id: TaskId, -) -> Result<(), AppError> { - let Some(task) = call(&state.store, |reply| StoreMsg::GetTask { - id: task_id, - reply, - }) - .await? - else { - return Ok(()); - }; - if !task.state.is_terminal() { - return Ok(()); - } - - for resource in state.resources.values() { - resource.cast(ResourceMsg::TaskTerminal { task_id })?; - } - - let snapshots = call(&state.store, |reply| { - StoreMsg::ResourceSnapshotsForAuthority { - authority_machine: state.machine, - reply, - } - }) - .await?; - for snapshot in snapshots { - let Some(loan) = snapshot.loan else { - continue; - }; - let is_exact_release = snapshot.resource.registered_background_task == Some(task_id) - && matches!( - loan.state, - crate::resource::LoanState::Active { - phase: crate::resource::LoanPhase::AwaitingRelease { - observed_background_task, - .. - } - } if observed_background_task == task_id - ); - if is_exact_release { - reconcile_resource_actor(supervisor, state, snapshot.resource.id).await?; - } - } - - Ok(()) -} - -async fn restore_resource_actors( - supervisor: &ActorRef, - state: &mut SupervisorState, -) -> Result<(), AppError> { - let snapshots = call(&state.store, |reply| { - StoreMsg::ResourceSnapshotsForAuthority { - authority_machine: state.machine, - reply, - } - }) - .await?; - for snapshot in snapshots { - spawn_resource_actor( - supervisor, - &mut state.resources, - state.store.clone(), - state.machine, - snapshot, - ) - .await?; - } - - Ok(()) -} - -async fn register_resource_actor( - supervisor: &ActorRef, - state: &mut SupervisorState, - resource: Resource, -) -> Result { - let resource = call(&state.store, |reply| StoreMsg::RegisterResource { - authority_machine: state.machine, - resource: Box::new(resource), - reply, - }) - .await?; - ensure_resource_actor(supervisor, state, resource.id).await?; - - Ok(resource) -} - -async fn ensure_resource_actor( - supervisor: &ActorRef, - state: &mut SupervisorState, - id: ResourceId, -) -> Result<(), AppError> { - if state.resources.contains_key(&id) { - return Ok(()); - } - - let snapshots = call(&state.store, |reply| { - StoreMsg::ResourceSnapshotsForAuthority { - authority_machine: state.machine, - reply, - } - }) - .await?; - let snapshot = snapshots - .into_iter() - .find(|snapshot| snapshot.resource.id == id) - .ok_or_else(|| AppError::Internal { - message: format!( - "registered resource {} is missing from the authority store", - id.as_uuid() - ), - })?; - spawn_resource_actor( - supervisor, - &mut state.resources, - state.store.clone(), - state.machine, - snapshot, - ) - .await -} - -async fn reconcile_resource_actor( - supervisor: &ActorRef, - state: &mut SupervisorState, - id: ResourceId, -) -> Result<(), AppError> { - ensure_resource_actor(supervisor, state, id).await?; - let actor = state.resources.get(&id).ok_or_else(|| AppError::Internal { - message: format!( - "resource actor {} disappeared during reconciliation", - id.as_uuid() - ), - })?; - call(actor, |reply| ResourceMsg::Reconcile { reply }) - .await - .map(|_| ()) -} - -async fn spawn_resource_actor( - supervisor: &ActorRef, - resources: &mut HashMap>, - store: ActorRef, - authority_machine: MachineId, - snapshot: ResourceSnapshot, -) -> Result<(), AppError> { - let id = snapshot.resource.id; - if resources.contains_key(&id) { - return Ok(()); - } - - let (actor, _handle) = ResourceActor::spawn_linked( - Some(resource_actor_name(id)), - ResourceActor, - ( - store, - supervisor.clone(), - authority_machine, - snapshot.resource, - snapshot.loan, - ), - supervisor.get_cell(), - ) - .await - .map_err(|err| AppError::Internal { - message: format!("spawn resource actor {}: {err}", id.as_uuid()), - })?; - resources.insert(id, actor); - - Ok(()) -} - async fn launch_remote( supervisor: &ActorRef, state: &mut SupervisorState, @@ -856,14 +410,6 @@ fn callback_executable(resolved: Result) -> CallbackExecutabl } } -fn wake_resource(state: &SupervisorState, id: ResourceId) { - if let Some(resource) = state.resources.get(&id) - && let Err(error) = resource.cast(ResourceMsg::Wake) - { - tracing::debug!(resource = %id.as_uuid(), "resource wake cast: {error}"); - } -} - fn task_name(id: TaskId) -> String { format!("homebased.task.{id}") } @@ -966,16 +512,12 @@ async fn task_exists(state: &SupervisorState, id: TaskId) -> Result, state: &mut SupervisorState, id: TaskId, -) -> Result, AppError> { - if ResourceOwnedTasks::load(state).await?.owns(id) { - return Ok(None); - } +) -> Result { if !state.tasks.contains_key(&id) { match task_status(state, id).await? { ProcessStatus::Queued => launch_accepted(supervisor, state, id).await?, @@ -984,7 +526,7 @@ async fn resume_local( } } // a failed spawn finishes the row, so report the status after the launch - task_status(state, id).await.map(Some) + task_status(state, id).await } async fn task_status(state: &SupervisorState, id: TaskId) -> Result { @@ -1002,21 +544,6 @@ async fn record_pid(state: &SupervisorState, id: TaskId, pid: u32) -> Result<(), call(&state.store, |reply| StoreMsg::SetPid { id, pid, reply }).await } -/// Prepare the files of a task that this call inserted, then spawn its worker -/// -/// A preparation failure finishes the committed row as `SpawnFailed`, so the -/// row never waits for a worker that cannot start -async fn spawn_inserted( - supervisor: &ActorRef, - state: &mut SupervisorState, - id: TaskId, -) -> Result<(), AppError> { - match state.home.prepare_task(id) { - Ok(_) => launch_accepted(supervisor, state, id).await, - Err(error) => finish_spawn_failed(state, id, &error).await, - } -} - async fn finish_spawn_failed( state: &SupervisorState, id: TaskId, diff --git a/src/daemon/actors/supervisor/recovery.rs b/src/daemon/actors/supervisor/recovery.rs index b14824a..de0e8bb 100644 --- a/src/daemon/actors/supervisor/recovery.rs +++ b/src/daemon/actors/supervisor/recovery.rs @@ -1,133 +1,53 @@ //! Startup recovery of the non-terminal tasks that this machine owns -use std::collections::HashMap; - use ractor::ActorRef; use super::{SupervisorMsg, SupervisorState, launch_accepted, spawn_task_actor}; use crate::daemon::actors::{StoreMsg, call}; -use crate::domain::{ProcessStatus, TaskId, TaskRow}; +use crate::domain::{ProcessStatus, TaskRow}; use crate::error::AppError; -use crate::resource::store::AcceptedResourceTask; use crate::submission::ExecutorIdentity; /// Give every non-terminal task an owner after a daemon restart /// /// Only a queued remote acceptance or released local task is launched, and /// `launch_accepted` still refuses one whose runner lock is held; every other -/// row is observed or left for its resource actor +/// row is observed pub(super) async fn recover_tasks( supervisor: &ActorRef, state: &mut SupervisorState, ) -> Result<(), AppError> { - let owned = ResourceOwnedTasks::load(state).await?; let released = call(&state.store, |reply| StoreMsg::UnstartedDependentTasks { reply, }) .await?; let rows = call(&state.store, |reply| StoreMsg::NonTerminal { reply }).await?; for row in rows { - // a queued return or first background task may have lost its spawn, and a - // free runner lock cannot prove otherwise, so it stays queued for resource attention - if row.status() == ProcessStatus::Queued - && (owned.restoring.contains(&row.id) || owned.background.contains(&row.id)) - { - continue; - } - let identity = - if row.status() == ProcessStatus::Queued && !owned.accepted.contains_key(&row.id) { - call(&state.store, |reply| StoreMsg::ExecutorIdentity { - id: row.id, - reply, - }) - .await? - } else { - None - }; - match startup_recovery_action( - &row, - owned.accepted.get(&row.id), - owned.watchers.contains(&row.id), - identity.as_ref(), - released.contains(&row.id), - ) { + let identity = if row.status() == ProcessStatus::Queued { + call(&state.store, |reply| StoreMsg::ExecutorIdentity { + id: row.id, + reply, + }) + .await? + } else { + None + }; + match startup_recovery_action(&row, identity.as_ref(), released.contains(&row.id)) { StartupRecoveryAction::LaunchAccepted => { launch_accepted(supervisor, state, row.id).await?; } StartupRecoveryAction::Observe => { spawn_task_actor(supervisor, state, row.id).await?; } - StartupRecoveryAction::DeferResource => {} } } Ok(()) } -/// Tasks whose launch belongs to a resource flow on this authority -/// -/// Startup recovery and a local submit retry leave these to their resource, -/// since a free runner lock cannot prove that the first spawn never happened -pub(super) struct ResourceOwnedTasks { - accepted: HashMap, - restoring: Vec, - background: Vec, - watchers: Vec, -} - -impl ResourceOwnedTasks { - pub(super) async fn load(state: &SupervisorState) -> Result { - let authority_machine = state.machine; - let accepted = call(&state.store, |reply| { - StoreMsg::AcceptedResourceTasksForAuthority { - authority_machine, - reply, - } - }) - .await? - .into_iter() - .map(|task| (task.request.task_id, task)) - .collect(); - let restoring = call(&state.store, |reply| StoreMsg::RestoringTasksForAuthority { - authority_machine, - reply, - }) - .await?; - let background = call(&state.store, |reply| { - StoreMsg::QueuedBackgroundLaunchTasksForAuthority { - authority_machine, - reply, - } - }) - .await?; - let watchers = call(&state.store, |reply| { - StoreMsg::ReleaseWatcherTasksForAuthority { - authority_machine, - reply, - } - }) - .await?; - Ok(Self { - accepted, - restoring, - background, - watchers, - }) - } - - /// Whether a resource flow owns this task's launch - pub(super) fn owns(&self, id: TaskId) -> bool { - self.accepted.contains_key(&id) - || self.restoring.contains(&id) - || self.background.contains(&id) - || self.watchers.contains(&id) - } -} - #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub(super) enum StartupRecoveryAction { LaunchAccepted, Observe, - DeferResource, } /// Decide how startup recovery owns one non-terminal row @@ -137,23 +57,9 @@ pub(super) enum StartupRecoveryAction { /// observed into `lost` pub(super) fn startup_recovery_action( row: &TaskRow, - resource_task: Option<&AcceptedResourceTask>, - release_watcher: bool, identity: Option<&ExecutorIdentity>, released: bool, ) -> StartupRecoveryAction { - if let Some(resource_task) = resource_task { - return match (row.status(), resource_task.state) { - (ProcessStatus::Queued, ProcessStatus::Queued) => StartupRecoveryAction::DeferResource, - _ => StartupRecoveryAction::Observe, - }; - } - // a bound watcher, even one accepted for a remote supervisor, is never - // relaunched from a queued row whose first spawn may already have happened - if release_watcher { - return StartupRecoveryAction::Observe; - } - if row.status() == ProcessStatus::Queued && (released || matches!( diff --git a/src/daemon/actors/supervisor/resource_launch.rs b/src/daemon/actors/supervisor/resource_launch.rs deleted file mode 100644 index dc691b0..0000000 --- a/src/daemon/actors/supervisor/resource_launch.rs +++ /dev/null @@ -1,879 +0,0 @@ -//! Resource-owned launches: assigned tasks, release watchers, returns, and background tasks -//! -//! Every launch commits its task records before any spawn, and only the call -//! whose transaction inserted a row starts its worker - -use ractor::{ActorRef, RpcReplyPort}; - -use super::{ - BackgroundLaunch, ReleaseWatcherLaunch, RemoteBackgroundLaunch, ReturnDecisionOutcome, - SupervisorMsg, SupervisorState, ensure_resource_actor, fail_queued_task, - reconcile_resource_actor, resolve_callback_codex, spawn_inserted, spawn_task_actor, - wake_resource, -}; -use crate::daemon::actors::resource::{ - BackgroundLaunchResult, ReleaseWatcherLaunchResult, ResourceMsg, RestoreLaunchResult, -}; -use crate::daemon::actors::{StoreMsg, call}; -use crate::domain::{ProcessStatus, TaskEnv, TaskId, TaskState}; -use crate::error::AppError; -use crate::invocation::persist_workload; -use crate::resource::bound_action::{ - ActionTaskAcceptance, ActionTaskIdentity, ActionTaskReceipt, PreparedActionTask, - ResourceActionKind, ResourceActionOperation, ResourceActionOutcome, ResourceActionRejection, - ResourceActionRequest, -}; -use crate::resource::release_watcher::ReleaseWatcherCommand; -use crate::resource::store::{ - PreLaunchFailure, ReleaseWatcherAcceptance, ReleaseWatcherAcceptanceError, - ReleaseWatcherAcceptanceInput, ResourceStoreError, ResourceTaskAcceptance, - ResourceTaskAcceptanceInput, -}; -use crate::resource::{ - ReleaseWatcherIntent, ResourceId, ReturnDecision, ReturnLaunch, SupervisorActionAuthority, -}; -use crate::store::{ - AcceptedActionTask, BackgroundLaunchAcceptance, BackgroundLaunchError, BackgroundLaunchInput, - EndedRestoreResolution, NewTask, RemoteBackgroundLaunchInput, - RemoteReleaseWatcherAcceptanceInput, ResourceActionError, ReturnClosure, ReturnDecisionError, - ReturnTaskAcceptance, ReturnTaskAcceptanceInput, ReturnTaskOrigin, new_queued_task, -}; -use crate::submission::{CallbackContext, RequestId}; - -/// Resource actor that hears when one launch starts and how it finished -/// -/// The reports only speed up the actor's view; it rebuilds that view from -/// durable state, so a failed finish report is not an error -struct LaunchReporter { - resource: Option>, -} - -impl LaunchReporter { - /// Tell the resource actor, when one runs, that a launch has started - fn start( - state: &SupervisorState, - resource_id: ResourceId, - started: ResourceMsg, - ) -> Result { - let resource = state.resources.get(&resource_id).cloned(); - if let Some(resource) = &resource { - resource.cast(started)?; - } - Ok(Self { resource }) - } - - /// Tell the resource actor, when one runs, how the launch finished - fn finish(&self, finished: ResourceMsg) { - if let Some(resource) = &self.resource - && let Err(error) = resource.cast(finished) - { - tracing::debug!(resource = ?resource.get_name(), "launch result cast: {error}"); - } - } -} - -pub(super) async fn launch_assigned_resource_task( - supervisor: &ActorRef, - state: &mut SupervisorState, - input: ResourceTaskAcceptanceInput, -) -> Result, AppError> { - accept_assigned_resource_task(supervisor, state, input, None).await -} - -/// Accept an assigned request as a task that fails before launch -/// -/// A task that already has records fails only while it is still queued; a -/// started or ended task keeps its state -pub(super) async fn fail_assigned_resource_task_before_launch( - supervisor: &ActorRef, - state: &mut SupervisorState, - input: ResourceTaskAcceptanceInput, - failure: PreLaunchFailure, -) -> Result, AppError> { - accept_assigned_resource_task(supervisor, state, input, Some(failure)).await -} - -async fn accept_assigned_resource_task( - supervisor: &ActorRef, - state: &mut SupervisorState, - input: ResourceTaskAcceptanceInput, - forced_failure: Option, -) -> Result, AppError> { - let task_id = input.task_id; - let resource_id = input.resource_id; - let input = Box::new(input); - let acceptance = match forced_failure.clone() { - None => { - call(&state.store, |reply| StoreMsg::AcceptAssignedResourceTask { - input, - reply, - }) - .await? - } - Some(failure) => { - call(&state.store, |reply| { - StoreMsg::AcceptUnlaunchableResourceTask { - input, - failure, - reply, - } - }) - .await? - } - }; - let acceptance = match acceptance { - Ok(acceptance) => acceptance, - Err(error) => return Ok(Err(error)), - }; - - match &acceptance { - ResourceTaskAcceptance::Inserted { task } if *task == task_id => { - spawn_inserted(supervisor, state, task_id).await?; - } - ResourceTaskAcceptance::Unlaunchable { task, failure } if *task == task_id => { - fail_before_launch(state, task_id, failure).await?; - } - ResourceTaskAcceptance::Existing { - task, - state: status, - } if *task == task_id => match (status, &forced_failure) { - (ProcessStatus::Queued, Some(failure)) => { - fail_before_launch(state, task_id, failure).await?; - } - (ProcessStatus::Queued, None) => { - reconcile_resource_actor(supervisor, state, resource_id).await?; - } - (ProcessStatus::Running, _) => spawn_task_actor(supervisor, state, task_id).await?, - ( - ProcessStatus::Succeeded - | ProcessStatus::Failed - | ProcessStatus::Cancelled - | ProcessStatus::Lost, - _, - ) => {} - }, - _ => { - return Err(AppError::Internal { - message: format!( - "resource task acceptance returned an unexpected identity for {task_id}" - ), - }); - } - } - - Ok(Ok(acceptance)) -} - -/// Record a failure before launch on a queued task and keep its reason in the output -/// -/// The Queued-to-Failed transition wins only if no worker reached Running, so it -/// cannot hide a started child; the terminal event notifies the task's thread -async fn fail_before_launch( - state: &SupervisorState, - task_id: TaskId, - failure: &PreLaunchFailure, -) -> Result<(), AppError> { - let message = failure.message(); - if !fail_queued_task(state, task_id, message.clone()).await? { - // a worker won Running first, so the output belongs to its child - return Ok(()); - } - tracing::warn!(%task_id, "resource task failed before launch: {message}"); - - // `task output` shows the reason too; the terminal event carries it either way - let written = state.home.prepare_task(task_id).and_then(|paths| { - Ok(std::fs::write( - paths.output, - format!("homebased: {message}\n"), - )?) - }); - if let Err(error) = written { - tracing::warn!(%task_id, "cannot write the launch failure output: {error}"); - } - Ok(()) -} - -pub(super) async fn launch_bound_release_watcher( - supervisor: &ActorRef, - state: &mut SupervisorState, - launch: ReleaseWatcherLaunch, -) -> Result, AppError> { - let ReleaseWatcherLaunch { - authority_machine, - resource_id, - supervisor: owner, - intent, - executable, - } = launch; - let task_id = intent.watcher_task_id.as_task_id(); - let spec = ReleaseWatcherCommand::from_intent(resource_id, &intent) - .normalized_spec(&executable, owner.thread)?; - let (env, callback) = watcher_launch_context(state, &intent, &spec.cwd).await?; - let row = new_queued_task(NewTask { - id: task_id, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: persist_workload(&spec.workload), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env, - binary: executable, - }); - let acceptance = call(&state.store, |reply| { - StoreMsg::AcceptReleaseWatcherForAuthority { - input: Box::new(ReleaseWatcherAcceptanceInput { - authority_machine, - resource_id, - supervisor: owner, - intent, - row, - spec, - callback, - }), - reply, - } - }) - .await?; - let acceptance = match acceptance { - Ok(acceptance) => acceptance, - Err(error) => return Ok(Err(error)), - }; - - match &acceptance { - // only the transaction that inserted the row may spawn its worker - ReleaseWatcherAcceptance::Inserted { task } if *task == task_id => { - spawn_inserted(supervisor, state, task_id).await?; - } - // a queued row found again may already have a worker or a lost spawn, and a - // free runner lock does not tell them apart, so it is only observed - ReleaseWatcherAcceptance::Existing { - task, - state: task_state, - } if *task == task_id => { - if matches!(task_state, TaskState::Running { .. }) { - spawn_task_actor(supervisor, state, task_id).await?; - } - } - ReleaseWatcherAcceptance::UnsupportedRemoteSupervisor { .. } => {} - ReleaseWatcherAcceptance::Inserted { .. } | ReleaseWatcherAcceptance::Existing { .. } => { - return Err(AppError::Internal { - message: format!( - "release watcher acceptance returned an unexpected identity for {task_id}" - ), - }); - } - } - - Ok(Ok(acceptance)) -} - -/// Reuse the environment and callback saved by an earlier acceptance of this watcher -/// -/// The canonical command and spec are rebuilt and compared by the store, but the -/// daemon environment can differ after a restart and must not turn an exact retry -/// into a conflict -async fn watcher_launch_context( - state: &SupervisorState, - intent: &ReleaseWatcherIntent, - cwd: &std::path::Path, -) -> Result<(TaskEnv, CallbackContext), AppError> { - let task_id = intent.watcher_task_id.as_task_id(); - let saved_row = call(&state.store, |reply| StoreMsg::GetTask { - id: task_id, - reply, - }) - .await?; - let saved_route = call(&state.store, |reply| StoreMsg::OriginRouteByRequest { - request: intent.request_id, - reply, - }) - .await?; - if let (Some(row), Some(route)) = (saved_row, saved_route) - && route.task == task_id - { - return Ok((row.env, route.callback)); - } - - let env = TaskEnv::capture(); - let callback = CallbackContext { - codex: resolve_callback_codex(state, &env.path, cwd), - env: env.clone(), - cwd: cwd.to_path_buf(), - }; - Ok((env, callback)) -} - -pub(super) async fn decide_return( - supervisor: &ActorRef, - state: &mut SupervisorState, - authority: SupervisorActionAuthority, - decision: ReturnDecision, -) -> Result, AppError> { - let launch = match decision { - ReturnDecision::NoResume { reason } => { - let result = call(&state.store, |reply| StoreMsg::RecordNoResumeForAuthority { - authority, - reason, - reply, - }) - .await?; - if result.is_ok() { - wake_resource(state, authority.resource_id); - } - return Ok(result.map(|closure| ReturnDecisionOutcome::Closed(Box::new(closure)))); - } - ReturnDecision::Launch(launch) => *launch, - }; - - let executor_env = TaskEnv::capture(); - let callback_cwd = match launch.work.supervisor_spec() { - Some(spec) => spec.as_normalized().cwd.clone(), - None => state.home.root().to_path_buf(), - }; - let callback_codex = resolve_callback_codex(state, &executor_env.path, &callback_cwd); - launch_return_task( - supervisor, - state, - authority, - launch, - executor_env, - ReturnTaskOrigin::Local { callback_codex }, - ) - .await - .map(|result| result.map(ReturnDecisionOutcome::Launch)) -} - -/// Bind one fixed return task, and spawn it only when this call inserted it -async fn launch_return_task( - supervisor: &ActorRef, - state: &mut SupervisorState, - authority: SupervisorActionAuthority, - launch: ReturnLaunch, - executor_env: TaskEnv, - origin: ReturnTaskOrigin, -) -> Result, AppError> { - let task_id = launch.task_id; - let reporter = LaunchReporter::start( - state, - authority.resource_id, - ResourceMsg::RestoreLaunchStarted { - action_id: authority.action_id, - task_id, - }, - )?; - let report = |result| { - reporter.finish(ResourceMsg::RestoreLaunchFinished { - action_id: authority.action_id, - task_id, - result, - }); - }; - - let acceptance = call(&state.store, |reply| { - StoreMsg::AcceptReturnTaskForAuthority { - input: Box::new(ReturnTaskAcceptanceInput { - authority, - launch, - executor_env, - origin, - }), - reply, - } - }) - .await; - let acceptance = match acceptance { - Ok(Ok(acceptance)) => acceptance, - Ok(Err(error)) => { - report(RestoreLaunchResult::NotInserted); - return Ok(Err(error)); - } - Err(error) => { - report(RestoreLaunchResult::NotInserted); - return Err(error); - } - }; - - let result = match &acceptance { - // only the transaction that inserted the row may spawn its worker - ReturnTaskAcceptance::Inserted { task, .. } if *task == task_id => { - if let Err(error) = spawn_inserted(supervisor, state, task_id).await { - // the committed row may have no worker, so it must show as attention - report(RestoreLaunchResult::NotInserted); - return Err(error); - } - RestoreLaunchResult::Inserted - } - // an existing binding is observed from durable state and never respawned - ReturnTaskAcceptance::Existing { - task, - state: status, - } if *task == task_id => { - if *status == ProcessStatus::Running { - spawn_task_actor(supervisor, state, task_id).await?; - } - RestoreLaunchResult::Existing { state: *status } - } - ReturnTaskAcceptance::UnsupportedRemoteSupervisor { .. } => { - RestoreLaunchResult::NotInserted - } - ReturnTaskAcceptance::Inserted { .. } | ReturnTaskAcceptance::Existing { .. } => { - report(RestoreLaunchResult::NotInserted); - return Err(AppError::Internal { - message: format!( - "return task acceptance returned an unexpected identity for {task_id}" - ), - }); - } - }; - report(result); - - Ok(Ok(acceptance)) -} - -/// Bind one co-located first background launch, and spawn it only when this call inserted it -pub(super) async fn launch_background( - supervisor: &ActorRef, - state: &mut SupervisorState, - launch: BackgroundLaunch, -) -> Result, AppError> { - let BackgroundLaunch { - resource_id, - request_id, - spec, - env, - callback_cwd, - } = launch; - let callback_codex = resolve_callback_codex(state, &env.path, &callback_cwd); - let input = BackgroundLaunchInput { - authority_machine: state.machine, - resource_id, - request_id, - task_id: TaskId::new(), - spec, - env, - callback_codex, - }; - bind_background_launch(supervisor, state, resource_id, request_id, |reply| { - StoreMsg::AcceptBackgroundLaunchForAuthority { - input: Box::new(input), - reply, - } - }) - .await -} - -/// Bind one remote supervisor's first background launch, and spawn it only on insertion -/// -/// The task runs with this authority's executor environment. The callback -/// context stays in the route that the supervisor machine saved -pub(super) async fn launch_remote_background( - supervisor: &ActorRef, - state: &mut SupervisorState, - launch: RemoteBackgroundLaunch, -) -> Result, AppError> { - let RemoteBackgroundLaunch { receipt, spec } = launch; - let resource_id = receipt.binding.assignment.resource_id; - ensure_resource_actor(supervisor, state, resource_id).await?; - let input = RemoteBackgroundLaunchInput { - receipt, - spec, - env: TaskEnv::capture(), - }; - bind_background_launch( - supervisor, - state, - resource_id, - receipt.request_id, - |reply| StoreMsg::AcceptRemoteBackgroundLaunchForAuthority { - input: Box::new(input), - reply, - }, - ) - .await -} - -/// Apply one store launch binding, and spawn the task only when the store inserted it -/// -/// An existing launch is observed from durable state. A queued row found again -/// may already have a worker or may have lost its spawn, and a free runner lock -/// cannot tell them apart, so it is never respawned here -async fn bind_background_launch( - supervisor: &ActorRef, - state: &mut SupervisorState, - resource_id: ResourceId, - request_id: RequestId, - accept: impl FnOnce( - RpcReplyPort, AppError>>, - ) -> StoreMsg, -) -> Result, AppError> { - let reporter = LaunchReporter::start( - state, - resource_id, - ResourceMsg::BackgroundLaunchStarted { request_id }, - )?; - let report = |result| { - reporter.finish(ResourceMsg::BackgroundLaunchFinished { request_id, result }); - }; - - let acceptance = match call(&state.store, accept).await { - Ok(Ok(acceptance)) => acceptance, - Ok(Err(error)) => { - report(BackgroundLaunchResult::NotInserted); - return Ok(Err(error)); - } - Err(error) => { - // the store may have committed; the resource owner treats a queued row as uncertain - report(BackgroundLaunchResult::NotInserted); - return Err(error); - } - }; - - let result = match &acceptance { - // only the transaction that inserted the row may spawn its worker - BackgroundLaunchAcceptance::Inserted { task, .. } => { - if let Err(error) = spawn_inserted(supervisor, state, *task).await { - // the committed row may have no worker, so it must show as uncertain - report(BackgroundLaunchResult::NotInserted); - return Err(error); - } - BackgroundLaunchResult::Inserted - } - BackgroundLaunchAcceptance::Existing { - task, - state: status, - } => { - if *status == ProcessStatus::Running { - spawn_task_actor(supervisor, state, *task).await?; - } - BackgroundLaunchResult::Existing - } - BackgroundLaunchAcceptance::UnsupportedRemoteSupervisor { .. } => { - BackgroundLaunchResult::NotInserted - } - }; - report(result); - - Ok(Ok(acceptance)) -} - -/// Apply one validated remote-supervisor operation on this authority -/// -/// Prepare operations only derive or bind identities. Launch operations commit -/// the task records and exact receipt before any spawn, and only an insertion -/// by this call starts a worker -pub(super) async fn resource_action( - supervisor: &ActorRef, - state: &mut SupervisorState, - request: ResourceActionRequest, -) -> Result { - let authority = request.authority; - match request.operation { - ResourceActionOperation::PrepareReleaseWatcher { - observed_background_task, - } => { - ensure_resource_actor(supervisor, state, authority.resource_id).await?; - let actor = - state - .resources - .get(&authority.resource_id) - .ok_or_else(|| AppError::Internal { - message: format!( - "resource actor {} disappeared", - authority.resource_id.as_uuid() - ), - })?; - let prepared = call(actor, |reply| ResourceMsg::PrepareRemoteWatcher { - authority, - observed_background_task, - reply, - }) - .await?; - Ok(match prepared { - Ok(task) => ResourceActionOutcome::Prepared { task }, - Err(reason) => ResourceActionOutcome::Rejected { reason }, - }) - } - ResourceActionOperation::LaunchReleaseWatcher { - observed_background_task, - task, - } => { - launch_remote_release_watcher( - supervisor, - state, - authority, - observed_background_task, - task, - ) - .await - } - ResourceActionOperation::PrepareReturn { launch } => { - let (request_id, task_id) = (launch.request_id, launch.task_id); - let prepared = call(&state.store, |reply| { - StoreMsg::PrepareReturnTaskForAuthority { - authority, - launch: Box::new(launch), - executor_env: TaskEnv::capture(), - reply, - } - }) - .await?; - match prepared { - Ok(prepared) => Ok(ResourceActionOutcome::Prepared { - task: PreparedActionTask { - request_id, - task_id, - spec: prepared.spec, - normalized_spec_sha256: prepared.normalized_spec_sha256, - }, - }), - Err(error) => return_rejection(error), - } - } - ResourceActionOperation::LaunchReturn { - launch, - normalized_spec_sha256, - } => { - let receipt = ActionTaskReceipt { - kind: ResourceActionKind::Return, - authority, - request_id: launch.request_id, - task_id: launch.task_id, - normalized_spec_sha256, - }; - let accepted = launch_return_task( - supervisor, - state, - authority, - launch, - TaskEnv::capture(), - ReturnTaskOrigin::Remote { - normalized_spec_sha256, - }, - ) - .await?; - let acceptance = match accepted { - Ok(ReturnTaskAcceptance::Inserted { .. }) => ActionTaskAcceptance::Inserted, - Ok(ReturnTaskAcceptance::Existing { state, .. }) => { - ActionTaskAcceptance::Existing { state } - } - Ok(ReturnTaskAcceptance::UnsupportedRemoteSupervisor { .. }) => { - return Ok(ResourceActionOutcome::Rejected { - reason: ResourceActionRejection::NotCurrentSupervisor, - }); - } - Err(error) => return return_rejection(error), - }; - Ok(ResourceActionOutcome::Accepted { - receipt, - acceptance, - }) - } - ResourceActionOperation::NoResume { reason } => { - let closed = call(&state.store, |reply| StoreMsg::RecordNoResumeForAuthority { - authority, - reason, - reply, - }) - .await?; - closed_outcome(state, authority.resource_id, closed) - } - ResourceActionOperation::HoldReturn { hold } => { - let held = call(&state.store, |reply| StoreMsg::HoldReturnForAuthority { - authority, - hold, - reply, - }) - .await?; - match held { - Ok(window) => Ok(ResourceActionOutcome::ReturnHeld { window }), - Err(error) => return_rejection(error), - } - } - ResourceActionOperation::ResolveEndedRestore { task_id, reason } => { - let closed = call(&state.store, |reply| { - StoreMsg::ResolveEndedRestoreForAuthority { - resolution: Box::new(EndedRestoreResolution { - authority, - task_id, - reason, - }), - reply, - } - }) - .await?; - closed_outcome(state, authority.resource_id, closed) - } - } -} - -/// Accept a remote supervisor's bound watcher and spawn only a fresh insertion -async fn launch_remote_release_watcher( - supervisor: &ActorRef, - state: &mut SupervisorState, - authority: SupervisorActionAuthority, - observed_background_task: TaskId, - task: ActionTaskIdentity, -) -> Result { - let task_id = task.task_id; - let executable = crate::resource::release_watcher::release_watcher_executable()?; - // the authority rebuilds its own command; the store rejects any other identity - let spec = ReleaseWatcherCommand { - resource_id: authority.resource_id, - action_id: authority.action_id, - state_revision: authority.expected_state_revision, - trainer_task_id: observed_background_task, - watcher_task_id: crate::resource::ReleaseWatcherTaskId::new(task_id), - } - .normalized_spec(&executable, authority.supervisor.thread)?; - let row = new_queued_task(NewTask { - id: task_id, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: persist_workload(&spec.workload), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env: TaskEnv::capture(), - binary: executable, - }); - let reporter = LaunchReporter::start( - state, - authority.resource_id, - ResourceMsg::RemoteWatcherLaunchStarted { - action_id: authority.action_id, - watcher_task_id: task_id, - }, - )?; - let report = |result| { - reporter.finish(ResourceMsg::WatcherLaunchFinished { - action_id: authority.action_id, - watcher_task_id: task_id, - result, - }); - }; - - let accepted = call(&state.store, |reply| { - StoreMsg::AcceptRemoteReleaseWatcherForAuthority { - input: Box::new(RemoteReleaseWatcherAcceptanceInput { - authority, - observed_background_task, - task, - row, - spec, - }), - reply, - } - }) - .await; - let AcceptedActionTask { - receipt, - acceptance, - } = match accepted { - Ok(Ok(accepted)) => accepted, - Ok(Err(ResourceActionError::Rejected(reason))) => { - report(ReleaseWatcherLaunchResult::Rejected); - return Ok(ResourceActionOutcome::Rejected { reason }); - } - Ok(Err(ResourceActionError::Storage(error))) | Err(error) => { - report(ReleaseWatcherLaunchResult::Uncertain); - return Err(error); - } - }; - - let result = match acceptance { - // only the transaction that inserted the row may spawn its worker - ActionTaskAcceptance::Inserted => { - if let Err(error) = spawn_inserted(supervisor, state, task_id).await { - // the committed row may have no worker, so it must show as attention - report(ReleaseWatcherLaunchResult::Uncertain); - return Err(error); - } - ReleaseWatcherLaunchResult::Inserted - } - // an existing acceptance is observed from durable state and never respawned - ActionTaskAcceptance::Existing { state: status } => { - if status == ProcessStatus::Running { - spawn_task_actor(supervisor, state, task_id).await?; - } - ReleaseWatcherLaunchResult::Existing { state: status } - } - }; - report(result); - - Ok(ResourceActionOutcome::Accepted { - receipt, - acceptance, - }) -} - -fn closed_outcome( - state: &SupervisorState, - resource_id: ResourceId, - closed: Result, -) -> Result { - match closed { - Ok(closure) => { - wake_resource(state, resource_id); - Ok(ResourceActionOutcome::Closed { - loan: closure.loan, - state_revision: closure.state_revision, - }) - } - Err(error) => return_rejection(error), - } -} - -/// Keep storage failures retryable and turn every domain refusal into a typed rejection -fn return_rejection(error: ReturnDecisionError) -> Result { - return_decision_rejection(error).map(|reason| ResourceActionOutcome::Rejected { reason }) -} - -/// Split one return-decision error into a definitive rejection or an unknown outcome -/// -/// Storage, encoding, and internal task-record failures leave the result unknown, -/// so they stay errors and are never reported as a refusal -pub(crate) fn return_decision_rejection( - error: ReturnDecisionError, -) -> Result { - let reason = match error { - ReturnDecisionError::Storage(error) => return Err(error.into()), - ReturnDecisionError::Encoding(error) => return Err(error.into()), - ReturnDecisionError::Identity(crate::store::IdentityError::Storage(error)) - | ReturnDecisionError::TaskRecords(error @ AppError::Internal { .. }) => return Err(error), - ReturnDecisionError::Resource(ResourceStoreError::Storage(error)) => { - return Err(error.into()); - } - ReturnDecisionError::RevisionExhausted { revision } => { - return Err(AppError::Internal { - message: format!("resource revision {revision:?} cannot be incremented"), - }); - } - ReturnDecisionError::NotCurrentSupervisor | ReturnDecisionError::OriginMismatch => { - ResourceActionRejection::NotCurrentSupervisor - } - ReturnDecisionError::ActionNotPending { .. } - | ReturnDecisionError::InvalidReturnNotice { .. } - | ReturnDecisionError::Resource(_) => ResourceActionRejection::ActionNotPending, - ReturnDecisionError::StaleRevision { expected, actual } => { - ResourceActionRejection::StaleRevision { expected, actual } - } - ReturnDecisionError::SpecMismatch => ResourceActionRejection::SpecMismatch, - ReturnDecisionError::ConflictingRetry { .. } => ResourceActionRejection::ConflictingRetry, - ReturnDecisionError::IdentityConflict { .. } | ReturnDecisionError::Identity(_) => { - ResourceActionRejection::IdentityConflict - } - error @ (ReturnDecisionError::RestoreNotEnded { .. } - | ReturnDecisionError::RestoreReleaseUnproven { .. } - | ReturnDecisionError::RestoreOwnershipUnproven { .. }) => { - ResourceActionRejection::RestoreNotResolvable { - reason: error.to_string(), - } - } - error @ (ReturnDecisionError::Rejected(_) - | ReturnDecisionError::HoldRejected(_) - | ReturnDecisionError::BackgroundTaskMismatch - | ReturnDecisionError::TaskRecords(_)) => ResourceActionRejection::DecisionRejected { - reason: error.to_string(), - }, - }; - Ok(reason) -} diff --git a/src/daemon/actors/supervisor/tests.rs b/src/daemon/actors/supervisor/tests.rs index 41993e8..29ba6d8 100644 --- a/src/daemon/actors/supervisor/tests.rs +++ b/src/daemon/actors/supervisor/tests.rs @@ -1,69 +1,28 @@ -use crate::resource::{IdleBoundaryProof, RestoreAttentionReason, ReturnLaunch}; use serde_json::json; -use std::io::Write; use tempfile::tempdir; -use uuid::Uuid; use super::recovery::{StartupRecoveryAction, startup_recovery_action}; use super::{ - BackgroundLaunch, ReturnDecisionOutcome, SUPERVISOR_TEST_LOCK, SupervisorActor, SupervisorArgs, - SupervisorMsg, callback_executable, + SUPERVISOR_TEST_LOCK, SupervisorActor, SupervisorArgs, SupervisorMsg, callback_executable, }; -use crate::daemon::actors::resource::ResourceActorInspection; use crate::daemon::actors::{StoreMsg, call}; -use crate::domain::{ - ExitReason, ProcessGroupExitEvidence, ProcessStatus, TaskEnv, TaskId, TaskRow, TaskWorkload, - ThreadId, Workload, -}; +use crate::domain::{ProcessStatus, TaskEnv, TaskId, TaskRow, TaskWorkload, Workload}; use crate::error::AppError; use crate::home::{Home, LockMode, flock_exclusive}; use crate::machine::{MachineId, load_or_create_machine_id}; -use crate::resource::bound_action::{ - ActionTaskAcceptance, ResourceActionOperation, ResourceActionOutcome, ResourceActionRequest, -}; -use crate::resource::command_shape::test_support::FakeTrainer; -use crate::resource::store::{ - OpenReleaseLoanResult, ResourceStoreError, ResourceTaskAcceptance, ResourceTaskAcceptanceInput, -}; -use crate::resource::{ - AssignmentRevision, CommandSpec, Loan, LoanPhase, LoanState, Resource, ResourceId, - ResourceQueueAttentionReason, ResourceQueueReconcileOutcome, ResourceRequestState, - ResourceRevision, ReturnContext, ReturnDecision, ReturnWork, SupervisorActionAuthority, - SupervisorAddress, -}; use crate::spec::{NormalizedSpec, NormalizedWorkload}; -use crate::store::{ - BackgroundLaunchAcceptance, BackgroundLaunchError, BackgroundLaunchInput, CancelResult, - EndedRestoreResolution, NewTask, ReturnTaskAcceptance, ReturnTaskAcceptanceInput, - ReturnTaskOrigin, Store, new_queued_task, -}; +use crate::store::{NewTask, Store, new_queued_task}; use crate::submission::{ - CallbackContext, CallbackExecutable, ExecutionRecord, ExecutorIdentity, NewResourceRoute, - OriginRoute, RequestId, + CallbackContext, CallbackExecutable, ExecutionRecord, ExecutorIdentity, OriginRoute, RequestId, }; use ractor::{Actor, ActorRef}; use std::path::PathBuf; -fn resource(authority: MachineId, background_task: Option) -> Resource { - Resource::new( - ResourceId::new(), - "gpu-0".into(), - authority, - SupervisorAddress { - machine: authority, - thread: ThreadId(Uuid::now_v7()), - }, - AssignmentRevision::new(0), - ResourceRevision::new(0), - background_task, - ) -} - -fn resource_command() -> NormalizedSpec { +fn echo_command() -> NormalizedSpec { serde_json::from_value(json!({ "api_version": 1, "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "resource startup test", + "name": "supervisor test", "cwd": "/tmp", "timeout": "4h", "workload": { "type": "task", "command": ["/bin/echo", "hello"] } @@ -71,33 +30,40 @@ fn resource_command() -> NormalizedSpec { .unwrap() } -// a native foreground command prints the marker, then blocks on the FIFO -fn resource_command_waiting_on_fifo(fifo: &std::path::Path) -> NormalizedSpec { - let marker = fifo.with_extension("marker"); - std::fs::write(&marker, "launch-marker\n").unwrap(); - serde_json::from_value(json!({ - "api_version": 1, - "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "resource one-shot launch test", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": ["/bin/cat", marker, fifo] } - })) - .unwrap() +fn configure_task_runner() { + crate::runner::set_task_run_executable_for_tests(assert_cmd::cargo::cargo_bin("homebased")); +} + +async fn acquire_runner_lock_after_task_exit(home: &Home, task_id: TaskId) -> std::fs::File { + let lock_path = home.task_paths(task_id).runner_lock; + tokio::task::spawn_blocking(move || flock_exclusive(&lock_path, LockMode::Blocking)) + .await + .unwrap() + .unwrap() } -fn seed_active_loan(home: &Home, authority: MachineId) -> (Resource, Loan) { - let mut store = Store::open(&home.db_path()).unwrap(); - let background_task = TaskId::new(); - let resource = resource(authority, Some(background_task)); - store.register_resource(authority, &resource).unwrap(); +async fn stop_supervisor( + supervisor: ActorRef, + handle: ractor::concurrency::JoinHandle<()>, +) { + supervisor.stop(None); + let _ = handle.await; +} - let spec = resource_command(); +#[tokio::test] +async fn direct_launch_still_starts_its_command() { + let _guard = SUPERVISOR_TEST_LOCK.lock().await; + configure_task_runner(); + let directory = tempdir().unwrap(); + let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); + home.ensure().unwrap(); + let spec = echo_command(); let NormalizedWorkload::Task(workload) = spec.workload.clone() else { - panic!("resource startup seed must use a command workload"); + panic!("direct launch test must use a command workload"); }; + let id = TaskId::new(); let row = new_queued_task(NewTask { - id: background_task, + id, name: Some(spec.name.clone()), thread: spec.thread, workload: Workload::Task(TaskWorkload { @@ -111,62 +77,58 @@ fn seed_active_loan(home: &Home, authority: MachineId) -> (Resource, Loan) { }, binary: "/bin/echo".into(), }); - store.insert_task(&row).unwrap(); - store - .cas_status( - background_task, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource.id, - authority, - spec, - ) + home.prepare_task(id).unwrap(); + + let (supervisor, handle) = SupervisorActor::spawn( + None, + SupervisorActor, + SupervisorArgs::new(home.clone(), None), + ) + .await + .unwrap(); + call(&supervisor, |reply| SupervisorMsg::Launch { + row: Box::new(row), + spec: Box::new(spec), + admission: crate::store::LocalAdmission::Submitted { + request: RequestId::new(), + after: None, + }, + reply, + }) + .await + .unwrap(); + let runner_lock = acquire_runner_lock_after_task_exit(&home, id).await; + assert_eq!( + std::fs::read_to_string(home.task_paths(id).output).unwrap(), + "hello\n" + ); + let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) + .await .unwrap(); - let OpenReleaseLoanResult::Opened { loan, .. } = store - .open_release_loan_for_authority(authority, resource.id, resource.state_revision) - .unwrap() - else { - panic!("startup seed must open a release loan"); - }; - store - .cas_exit( - background_task, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ) + let row = call(&store, |reply| StoreMsg::GetTask { id, reply }) + .await .unwrap() .unwrap(); + assert_eq!(row.status(), ProcessStatus::Succeeded); - let saved_resource = store - .resource_snapshots_for_authority(authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == resource.id) - .unwrap() - .resource; - (saved_resource, loan) + drop(runner_lock); + stop_supervisor(supervisor, handle).await; } -fn seed_queued_resource_request(home: &Home, authority: MachineId) -> (Resource, TaskId) { - let mut store = Store::open(&home.db_path()).unwrap(); - let background_task = TaskId::new(); - let resource = resource(authority, Some(background_task)); - store.register_resource(authority, &resource).unwrap(); - - let spec = resource_command(); +#[tokio::test] +async fn resume_local_starts_a_committed_row_whose_launch_stopped() { + let _guard = SUPERVISOR_TEST_LOCK.lock().await; + configure_task_runner(); + let directory = tempdir().unwrap(); + let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); + home.ensure().unwrap(); + let spec = echo_command(); let NormalizedWorkload::Task(workload) = spec.workload.clone() else { - panic!("startup seed must use a command workload"); + panic!("resume test must use a command workload"); }; + let id = TaskId::new(); let row = new_queued_task(NewTask { - id: background_task, + id, name: Some(spec.name.clone()), thread: spec.thread, workload: Workload::Task(TaskWorkload { @@ -180,69 +142,82 @@ fn seed_queued_resource_request(home: &Home, authority: MachineId) -> (Resource, }, binary: "/bin/echo".into(), }); - store.insert_task(&row).unwrap(); - store - .cas_status( - background_task, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource.id, - authority, - spec, - ) + home.prepare_task(id).unwrap(); + let (supervisor, handle) = SupervisorActor::spawn( + None, + SupervisorActor, + SupervisorArgs::new(home.clone(), None), + ) + .await + .unwrap(); + // the row commits under its request after startup recovery ran, as when the + // submit's caller timed out and the supervisor never spawned the worker + let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) + .await .unwrap(); - home.prepare_task(background_task).unwrap(); + let machine = load_or_create_machine_id(&home).unwrap(); + call(&store, |reply| StoreMsg::InsertLocalTask { + row: Box::new(row), + spec: Box::new(spec), + machine, + admission: crate::store::LocalAdmission::Submitted { + request: RequestId::new(), + after: None, + }, + codex: CallbackExecutable::available("/bin/echo".into()), + reply, + }) + .await + .unwrap(); - (resource, background_task) -} + let status = call(&supervisor, |reply| SupervisorMsg::ResumeLocal { + id, + reply, + }) + .await + .unwrap(); + // the status is read after the launch, so the worker may already have claimed or finished the task + assert!( + matches!( + status, + ProcessStatus::Queued | ProcessStatus::Running | ProcessStatus::Succeeded + ), + "{status:?}" + ); + let runner_lock = acquire_runner_lock_after_task_exit(&home, id).await; + assert_eq!( + std::fs::read_to_string(home.task_paths(id).output).unwrap(), + "hello\n" + ); + let status = call(&supervisor, |reply| SupervisorMsg::ResumeLocal { + id, + reply, + }) + .await + .unwrap(); + assert_eq!(status, ProcessStatus::Succeeded); -fn seed_serving_assigned_resource_task( - home: &Home, - authority: MachineId, - origin: MachineId, -) -> ( - Resource, - Loan, - RequestId, - TaskId, - ResourceTaskAcceptanceInput, -) { - seed_serving_assigned_resource_task_with_spec(home, authority, origin, resource_command()) + drop(runner_lock); + stop_supervisor(supervisor, handle).await; } -fn seed_serving_assigned_resource_task_with_spec( - home: &Home, - authority: MachineId, - origin: MachineId, - spec: NormalizedSpec, -) -> ( - Resource, - Loan, - RequestId, - TaskId, - ResourceTaskAcceptanceInput, -) { - let mut store = Store::open(&home.db_path()).unwrap(); - let background_task = TaskId::new(); - let request_id = RequestId::new(); - let task_id = TaskId::new(); - let mut resource = resource(authority, Some(background_task)); - resource.supervisor.thread = spec.thread; - store.register_resource(authority, &resource).unwrap(); - +#[tokio::test] +async fn direct_launch_keeps_an_unresolved_callback_codex_unavailable() { + let _guard = SUPERVISOR_TEST_LOCK.lock().await; + configure_task_runner(); + let directory = tempdir().unwrap(); + let home = Home::resolve(Some(directory.path().join("home"))).unwrap(); + home.ensure().unwrap(); + // a PATH with no executables, so no Codex can be found for callbacks + let empty_path = directory.path().join("empty-bin"); + std::fs::create_dir(&empty_path).unwrap(); + let spec = echo_command(); let NormalizedWorkload::Task(workload) = spec.workload.clone() else { - panic!("resource acceptance seed must use a command workload"); + panic!("direct launch test must use a command workload"); }; - let background = new_queued_task(NewTask { - id: background_task, + let id = TaskId::new(); + let row = new_queued_task(NewTask { + id, name: Some(spec.name.clone()), thread: spec.thread, workload: Workload::Task(TaskWorkload { @@ -251,2399 +226,90 @@ fn seed_serving_assigned_resource_task_with_spec( cwd: spec.cwd.clone(), timeout: spec.timeout, env: TaskEnv { - path: "/bin".into(), + path: empty_path.to_string_lossy().into_owned(), home: "/tmp".into(), }, binary: "/bin/echo".into(), }); - store.insert_task(&background).unwrap(); - store - .cas_status( - background_task, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); + home.prepare_task(id).unwrap(); - if origin == authority { - store - .insert_origin_route( - &OriginRoute::new_resource_waiting(NewResourceRoute { - request: request_id, - task: task_id, - origin_machine: origin, - authority_machine: authority, - thread: spec.thread, - callback: CallbackContext { - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - cwd: "/tmp".into(), - codex: CallbackExecutable::available("/bin/echo".into()), - }, - spec: spec.clone(), - resource: resource.id, - }) - .unwrap(), - ) - .unwrap(); - } - let request = store - .accept_resource_request( - authority, - request_id, - task_id, - resource.id, - origin, - spec.clone(), - ) - .unwrap(); - let OpenReleaseLoanResult::Opened { .. } = store - .open_release_loan_for_authority(authority, resource.id, resource.state_revision) - .unwrap() - else { - panic!("resource acceptance seed must open one release loan"); - }; - store - .cas_exit( - background_task, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ) - .unwrap() - .unwrap(); - let (loan, state_revision) = store - .seed_verified_serving_loan_for_test( - authority, - resource.id, - request.request_id, - ReturnContext::AlreadyCompleted { - task_id: background_task, - result_ref: "result-1".into(), - }, - ) + let (supervisor, handle) = SupervisorActor::spawn( + None, + SupervisorActor, + SupervisorArgs::new(home.clone(), None), + ) + .await + .unwrap(); + call(&supervisor, |reply| SupervisorMsg::Launch { + row: Box::new(row), + spec: Box::new(spec), + admission: crate::store::LocalAdmission::Submitted { + request: RequestId::new(), + after: None, + }, + reply, + }) + .await + .unwrap(); + let runner_lock = acquire_runner_lock_after_task_exit(&home, id).await; + let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) + .await .unwrap(); - let input = ResourceTaskAcceptanceInput { - authority_machine: authority, - resource_id: resource.id, - request_id, - task_id, - acceptance_sequence: request.acceptance_sequence, - loan_id: loan.id, - expected_state_revision: state_revision, - command_spec: CommandSpec::try_from(spec).unwrap(), - executor_env: TaskEnv::capture(), - }; - home.prepare_task(task_id).unwrap(); - - (resource, loan, request_id, task_id, input) -} - -fn seed_accepted_resource_task( - home: &Home, - authority: MachineId, - origin: MachineId, -) -> (Resource, Loan, RequestId, TaskId) { - let (resource, loan, request_id, task_id, input) = - seed_serving_assigned_resource_task(home, authority, origin); - assert!(matches!( - Store::open(&home.db_path()) - .unwrap() - .accept_assigned_resource_task(input) - .unwrap(), - ResourceTaskAcceptance::Inserted { task } if task == task_id - )); - - (resource, loan, request_id, task_id) -} - -fn configure_task_runner() { - crate::runner::set_task_run_executable_for_tests(assert_cmd::cargo::cargo_bin("homebased")); -} - -async fn acquire_runner_lock_after_task_exit(home: &Home, task_id: TaskId) -> std::fs::File { - let lock_path = home.task_paths(task_id).runner_lock; - tokio::task::spawn_blocking(move || flock_exclusive(&lock_path, LockMode::Blocking)) + let route = call(&store, |reply| StoreMsg::OriginRoute { id, reply }) .await .unwrap() - .unwrap() -} - -async fn call_assigned_resource_launch( - supervisor: &ActorRef, - input: ResourceTaskAcceptanceInput, -) -> Result { - call(supervisor, |reply| { - SupervisorMsg::LaunchAssignedResourceTask { - input: Box::new(input), - reply, - } - }) - .await - .unwrap() -} + .expect("direct launch saves an origin route"); -async fn wait_for_running_task( - store: &ActorRef, - home: &Home, - task_id: TaskId, -) -> TaskRow { - let result = tokio::time::timeout(std::time::Duration::from_secs(10), async { - loop { - let row = call(store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .unwrap(); - if row.status() == ProcessStatus::Running && row.pid().is_some() { - return row; - } + assert_eq!(route.callback.codex.path(), None); + assert!(route.callback.codex.unavailable_reason().is_some()); - tokio::task::yield_now().await; - } - }) - .await; - match result { - Ok(row) => row, - Err(_) => { - let row = call(store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .unwrap(); - panic!( - "task {task_id} did not run: state={:?}, reason={:?}, output={:?}", - row.state, - row.exit_reason(), - std::fs::read_to_string(home.task_paths(task_id).output) - ); - } - } + drop(runner_lock); + stop_supervisor(supervisor, handle).await; } -async fn inspect_resource( - supervisor: &ActorRef, - id: ResourceId, -) -> Option { - call(supervisor, |reply| SupervisorMsg::InspectResource { +#[test] +fn direct_queued_tasks_keep_their_existing_startup_actions() { + let spec = echo_command(); + let id = TaskId::new(); + let NormalizedWorkload::Task(workload) = spec.workload.clone() else { + panic!("startup test must use a command workload"); + }; + let row = new_queued_task(NewTask { id, - reply, - }) - .await - .unwrap() -} - -async fn stop_supervisor( - supervisor: ActorRef, - handle: ractor::concurrency::JoinHandle<()>, -) { - supervisor.stop(None); - let _ = handle.await; -} - -#[tokio::test] -async fn startup_restores_resource_and_active_loan_snapshot() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (saved_resource, saved_loan) = seed_active_loan(&home, authority); - - let (supervisor, handle) = - SupervisorActor::spawn(None, SupervisorActor, SupervisorArgs::new(home, None)) - .await - .unwrap(); - let inspection = inspect_resource(&supervisor, saved_resource.id) - .await - .unwrap(); - - assert_eq!(inspection.resource, saved_resource); - assert_eq!(inspection.loan, Some(saved_loan)); - assert!(inspection.actor_id.is_local()); - - stop_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn daemon_startup_reconciles_accepted_work_without_relaunching_background_task() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, background_task) = seed_queued_resource_request(&home, authority); - let runner_lock = flock_exclusive( - &home.task_paths(background_task).runner_lock, - LockMode::NonBlocking, - ) - .unwrap(); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let inspection = inspect_resource(&supervisor, resource.id).await.unwrap(); - assert!(matches!( - inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::ReleaseProofUnavailable { .. }) - )); - assert!(matches!( - inspection.loan.map(|loan| loan.state), - Some(LoanState::Active { - phase: LoanPhase::AwaitingRelease { - observed_background_task: saved_task, - .. - } - }) if saved_task == background_task - )); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - assert_eq!( - call(&store, |reply| StoreMsg::PendingSupervisorNotices { reply }) - .await - .unwrap() - .unwrap() - .len(), - 1 - ); - - stop_supervisor(supervisor, handle).await; - drop(runner_lock); -} - -#[tokio::test] -async fn restart_defers_accepted_queued_resource_tasks_from_local_and_remote_origins() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - for local_origin in [false, true] { - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let origin = if local_origin { - authority - } else { - MachineId::new() - }; - let (resource, loan, request_id, task_id) = - seed_accepted_resource_task(&home, authority, origin); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let inspection = inspect_resource(&supervisor, resource.id).await.unwrap(); - assert!(matches!( - &inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::AttentionRequired { - request, - reason: ResourceQueueAttentionReason::AcceptedTaskLaunchUncertain { - task_id: attention_task, - }, - }) if request.request_id == request_id - && request.task_id == task_id - && *attention_task == task_id - )); - assert!(matches!( - &inspection.loan, - Some(Loan { - id, - state: LoanState::Active { - phase: LoanPhase::Serving { - current_request_id, .. - }, - }, - .. - }) if *id == loan.id && *current_request_id == request_id - )); - - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let row = call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .unwrap(); - assert_eq!(row.status(), ProcessStatus::Queued); - assert_eq!(row.pid(), None); - assert!(!home.task_paths(task_id).output.exists()); - assert!(matches!( - call(&store, |reply| StoreMsg::ResourceRequests { - authority_machine: authority, - resource_id: resource.id, - reply, - }) - .await - .unwrap()[0] - .state, - ResourceRequestState::Assigned { loan_id } if loan_id == loan.id - )); - - let runner_lock = - flock_exclusive(&home.task_paths(task_id).runner_lock, LockMode::NonBlocking).unwrap(); - drop(runner_lock); - stop_supervisor(supervisor, handle).await; - } -} - -#[tokio::test] -async fn restart_observes_accepted_running_resource_task_without_relaunching_it() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, loan, request_id, task_id) = - seed_accepted_resource_task(&home, authority, MachineId::new()); - let saved = Store::open(&home.db_path()).unwrap(); - saved - .cas_status(task_id, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - drop(saved); - let runner_lock = - flock_exclusive(&home.task_paths(task_id).runner_lock, LockMode::NonBlocking).unwrap(); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let inspection = inspect_resource(&supervisor, resource.id).await.unwrap(); - assert!(matches!( - inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::LoanAlreadyActive { loan: ref active }) - if active.id == loan.id - )); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let row = call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .unwrap(); - assert_eq!(row.status(), ProcessStatus::Running); - assert_eq!(row.pid(), None); - assert!(matches!( - inspection.loan.map(|active| active.state), - Some(LoanState::Active { - phase: LoanPhase::Serving { - current_request_id: current, - .. - }, - }) if current == request_id - )); - - stop_supervisor(supervisor, handle).await; - drop(runner_lock); -} - -#[tokio::test] -async fn fresh_serving_assignment_starts_one_command_and_retry_only_observes_it() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let fifo = directory.path().join("command-gate"); - nix::unistd::mkfifo( - &fifo, - nix::sys::stat::Mode::S_IRUSR | nix::sys::stat::Mode::S_IWUSR, - ) - .unwrap(); - let spec = resource_command_waiting_on_fifo(&fifo); - let (resource, loan, request_id, task_id, input) = - seed_serving_assigned_resource_task_with_spec(&home, authority, authority, spec); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let running = wait_for_running_task(&store, &home, task_id).await; - assert_eq!(running.id, task_id); - assert!(running.pid().is_some_and(|pid| pid > 0)); - assert!(matches!( - call_assigned_resource_launch(&supervisor, input.clone()).await, - Ok(ResourceTaskAcceptance::Existing { - task, - state: ProcessStatus::Running, - }) if task == task_id - )); - - tokio::task::spawn_blocking(move || { - let mut writer = std::fs::OpenOptions::new().write(true).open(fifo)?; - writer.write_all(b"released\n") - }) - .await - .unwrap() - .unwrap(); - let runner_lock = acquire_runner_lock_after_task_exit(&home, task_id).await; - let output = std::fs::read_to_string(home.task_paths(task_id).output).unwrap(); - assert_eq!(output, "launch-marker\nreleased\n"); - let row = call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .unwrap(); - assert_eq!(row.status(), ProcessStatus::Succeeded); - assert!(matches!( - row.exit_reason(), - Some(ExitReason::Exit { code: 0 }) - )); - assert_eq!( - Store::open(&home.db_path()) - .unwrap() - .process_group_exit_evidence(task_id) - .unwrap(), - Some(ProcessGroupExitEvidence::ConfirmedExited) - ); - assert_eq!(resource.id, input.resource_id); - assert_eq!(loan.id, input.loan_id); - assert_eq!(request_id, input.request_id); - drop(runner_lock); - stop_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn existing_queued_assignment_retries_without_launch_even_with_free_lock() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, _loan, request_id, task_id, input) = - seed_serving_assigned_resource_task(&home, authority, authority); - assert!(matches!( - Store::open(&home.db_path()) - .unwrap() - .accept_assigned_resource_task(input.clone()) - .unwrap(), - ResourceTaskAcceptance::Inserted { task } if task == task_id - )); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let inspection = inspect_resource(&supervisor, resource.id).await.unwrap(); - assert!(matches!( - inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::AttentionRequired { - request, - reason: ResourceQueueAttentionReason::AcceptedTaskLaunchUncertain { - task_id: attention_task, - }, - }) if request.request_id == request_id - && request.task_id == task_id - && attention_task == task_id - )); - let free_lock = - flock_exclusive(&home.task_paths(task_id).runner_lock, LockMode::NonBlocking).unwrap(); - drop(free_lock); - - for _ in 0..2 { - assert!(matches!( - call_assigned_resource_launch(&supervisor, input.clone()).await, - Ok(ResourceTaskAcceptance::Existing { - task, - state: ProcessStatus::Queued, - }) if task == task_id - )); - } - - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let row = call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .unwrap(); - assert_eq!(row.status(), ProcessStatus::Queued); - assert_eq!(row.pid(), None); - assert!(!home.task_paths(task_id).output.exists()); - let free_lock = - flock_exclusive(&home.task_paths(task_id).runner_lock, LockMode::NonBlocking).unwrap(); - drop(free_lock); - let inspection = inspect_resource(&supervisor, resource.id).await.unwrap(); - assert!(matches!( - inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::AttentionRequired { - reason: ResourceQueueAttentionReason::AcceptedTaskLaunchUncertain { - task_id: attention_task, - }, - .. - }) if attention_task == task_id - )); - assert!(matches!( - inspection.loan.map(|saved| saved.state), - Some(LoanState::Active { - phase: LoanPhase::Serving { - current_request_id: current, - .. - }, - }) if current == request_id - )); - - stop_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn cancellation_before_acceptance_prevents_resource_task_spawn() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, _loan, _request_id, task_id, input) = - seed_serving_assigned_resource_task(&home, authority, authority); - Store::open(&home.db_path()) - .unwrap() - .cancel_resource_request_before_activation( - authority, - input.request_id, - task_id, - resource.id, - authority, - ) - .unwrap(); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let result = call(&supervisor, |reply| { - SupervisorMsg::LaunchAssignedResourceTask { - input: Box::new(input), - reply, - } - }) - .await - .unwrap(); - assert!(matches!(result, Err(ResourceStoreError::Prevented))); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - assert!( - call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .is_none() - ); - assert!(!home.task_paths(task_id).output.exists()); - assert!(!home.task_paths(task_id).runner_lock.exists()); - - stop_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn failed_first_resource_spawn_records_no_child_proof_and_reserves_return() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, loan, _request_id, task_id, _input) = - seed_serving_assigned_resource_task(&home, authority, authority); - crate::runner::set_task_run_executable_for_tests( - directory.path().join("missing-homebased-binary"), - ); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let row = tokio::time::timeout(std::time::Duration::from_secs(10), async { - loop { - if let Some(row) = call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - && row.status() == ProcessStatus::Failed - { - return row; - } - - tokio::task::yield_now().await; - } - }) - .await - .expect("failed resource spawn must reach a durable failed state"); - assert_eq!(row.status(), ProcessStatus::Failed); - assert!(matches!( - row.exit_reason(), - Some(ExitReason::SpawnFailed { .. }) - )); - assert_eq!( - Store::open(&home.db_path()) - .unwrap() - .process_group_exit_evidence(task_id) - .unwrap(), - Some(ProcessGroupExitEvidence::NoChildSpawned) - ); - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource.id, - reply, - }) - .await - .unwrap(); - let inspection = inspect_resource(&supervisor, resource.id).await.unwrap(); - assert!(matches!( - inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::LoanAlreadyActive { - loan: Loan { - id, - state: LoanState::Active { - phase: LoanPhase::AwaitingReturn { .. }, - }, - .. - } - }) if id == loan.id - )); - let snapshot = call(&store, |reply| StoreMsg::ResourceSnapshotsForAuthority { - authority_machine: authority, - reply, - }) - .await - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == resource.id) - .unwrap(); - assert!(matches!( - snapshot.loan, - Some(Loan { - id, - state: LoanState::Active { - phase: LoanPhase::AwaitingReturn { .. }, - }, - .. - }) if id == loan.id - )); - assert!(matches!( - call(&store, |reply| StoreMsg::ResourceRequests { - authority_machine: authority, - resource_id: resource.id, - reply, - }) - .await - .unwrap()[0] - .state, - ResourceRequestState::Finished { - outcome: ExitReason::SpawnFailed { .. } - } - )); - let notices = call(&store, |reply| StoreMsg::PendingSupervisorNotices { reply }) - .await - .unwrap() - .unwrap(); - let return_notices = notices - .iter() - .filter(|notice| { - matches!( - notice.payload, - crate::resource::SupervisorNoticePayload::ReturnRequired { .. } - ) - }) - .collect::>(); - assert_eq!(return_notices.len(), 1); - assert_eq!(return_notices[0].loan_id, loan.id); - - let row = call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .unwrap(); - assert_eq!(row.id, task_id); - - stop_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn direct_launch_still_starts_its_command() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let spec = resource_command(); - let NormalizedWorkload::Task(workload) = spec.workload.clone() else { - panic!("direct launch test must use a command workload"); - }; - let id = TaskId::new(); - let row = new_queued_task(NewTask { - id, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: Workload::Task(TaskWorkload { - command: workload.command, - }), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - binary: "/bin/echo".into(), - }); - home.prepare_task(id).unwrap(); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - call(&supervisor, |reply| SupervisorMsg::Launch { - row: Box::new(row), - spec: Box::new(spec), - admission: crate::store::LocalAdmission::Submitted { - request: RequestId::new(), - after: None, - }, - reply, - }) - .await - .unwrap(); - let runner_lock = acquire_runner_lock_after_task_exit(&home, id).await; - assert_eq!( - std::fs::read_to_string(home.task_paths(id).output).unwrap(), - "hello\n" - ); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let row = call(&store, |reply| StoreMsg::GetTask { id, reply }) - .await - .unwrap() - .unwrap(); - assert_eq!(row.status(), ProcessStatus::Succeeded); - - drop(runner_lock); - stop_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn resume_local_starts_a_committed_row_whose_launch_stopped() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let spec = resource_command(); - let NormalizedWorkload::Task(workload) = spec.workload.clone() else { - panic!("resume test must use a command workload"); - }; - let id = TaskId::new(); - let row = new_queued_task(NewTask { - id, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: Workload::Task(TaskWorkload { - command: workload.command, - }), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - binary: "/bin/echo".into(), - }); - home.prepare_task(id).unwrap(); - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - // the row commits under its request after startup recovery ran, as when the - // submit's caller timed out and the supervisor never spawned the worker - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let machine = load_or_create_machine_id(&home).unwrap(); - call(&store, |reply| StoreMsg::InsertLocalTask { - row: Box::new(row), - spec: Box::new(spec), - machine, - admission: crate::store::LocalAdmission::Submitted { - request: RequestId::new(), - after: None, - }, - codex: CallbackExecutable::available("/bin/echo".into()), - reply, - }) - .await - .unwrap(); - - let status = call(&supervisor, |reply| SupervisorMsg::ResumeLocal { - id, - reply, - }) - .await - .unwrap(); - // the status is read after the launch, so the worker may already have claimed or finished the task - assert!( - matches!( - status, - Some(ProcessStatus::Queued | ProcessStatus::Running | ProcessStatus::Succeeded) - ), - "{status:?}" - ); - let runner_lock = acquire_runner_lock_after_task_exit(&home, id).await; - assert_eq!( - std::fs::read_to_string(home.task_paths(id).output).unwrap(), - "hello\n" - ); - let status = call(&supervisor, |reply| SupervisorMsg::ResumeLocal { - id, - reply, - }) - .await - .unwrap(); - assert_eq!(status, Some(ProcessStatus::Succeeded)); - - drop(runner_lock); - stop_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn direct_launch_keeps_an_unresolved_callback_codex_unavailable() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().join("home"))).unwrap(); - home.ensure().unwrap(); - // a PATH with no executables, so no Codex can be found for callbacks - let empty_path = directory.path().join("empty-bin"); - std::fs::create_dir(&empty_path).unwrap(); - let spec = resource_command(); - let NormalizedWorkload::Task(workload) = spec.workload.clone() else { - panic!("direct launch test must use a command workload"); - }; - let id = TaskId::new(); - let row = new_queued_task(NewTask { - id, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: Workload::Task(TaskWorkload { - command: workload.command, - }), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env: TaskEnv { - path: empty_path.to_string_lossy().into_owned(), - home: "/tmp".into(), - }, - binary: "/bin/echo".into(), - }); - home.prepare_task(id).unwrap(); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - call(&supervisor, |reply| SupervisorMsg::Launch { - row: Box::new(row), - spec: Box::new(spec), - admission: crate::store::LocalAdmission::Submitted { - request: RequestId::new(), - after: None, - }, - reply, - }) - .await - .unwrap(); - let runner_lock = acquire_runner_lock_after_task_exit(&home, id).await; - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let route = call(&store, |reply| StoreMsg::OriginRoute { id, reply }) - .await - .unwrap() - .expect("direct launch saves an origin route"); - - assert_eq!(route.callback.codex.path(), None); - assert!(route.callback.codex.unavailable_reason().is_some()); - - drop(runner_lock); - stop_supervisor(supervisor, handle).await; -} - -#[test] -fn direct_queued_tasks_keep_their_existing_startup_actions() { - let spec = resource_command(); - let id = TaskId::new(); - let NormalizedWorkload::Task(workload) = spec.workload.clone() else { - panic!("startup test must use a command workload"); - }; - let row = new_queued_task(NewTask { - id, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: Workload::Task(TaskWorkload { - command: workload.command, - }), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - binary: "/bin/echo".into(), - }); - assert_eq!( - startup_recovery_action(&row, None, false, None, false), - StartupRecoveryAction::Observe - ); - - let origin_machine = MachineId::new(); - let mut execution_machine = MachineId::new(); - while execution_machine == origin_machine { - execution_machine = MachineId::new(); - } - let identity = ExecutorIdentity::Accepted(ExecutionRecord { - task: id, - origin_machine, - execution_machine, - spec: spec.into(), - state: ProcessStatus::Queued, - }); - assert_eq!( - startup_recovery_action(&row, None, false, Some(&identity), false), - StartupRecoveryAction::LaunchAccepted - ); - // a remote supervisor's bound watcher may already have spawned, so it is only observed - assert_eq!( - startup_recovery_action(&row, None, true, Some(&identity), false), - StartupRecoveryAction::Observe - ); -} - -#[tokio::test] -async fn resource_registration_retry_keeps_one_actor_identity() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let resource = resource(authority, None); - - let (supervisor, handle) = - SupervisorActor::spawn(None, SupervisorActor, SupervisorArgs::new(home, None)) - .await - .unwrap(); - let registered = call(&supervisor, |reply| SupervisorMsg::RegisterResource { - resource: Box::new(resource.clone()), - reply, - }) - .await - .unwrap(); - assert_eq!(registered, resource); - let before_retry = inspect_resource(&supervisor, resource.id).await.unwrap(); - - let retried = call(&supervisor, |reply| SupervisorMsg::RegisterResource { - resource: Box::new(resource.clone()), - reply, - }) - .await - .unwrap(); - let after_retry = inspect_resource(&supervisor, resource.id).await.unwrap(); - - assert_eq!(retried, resource); - assert_eq!(before_retry.actor_id, after_retry.actor_id); - assert_eq!(before_retry.resource, after_retry.resource); - assert_eq!(before_retry.loan, None); - - stop_supervisor(supervisor, handle).await; -} - -/// Drain one accepted request so its loan awaits the supervisor return decision -fn seed_awaiting_return( - home: &Home, - authority: MachineId, -) -> (Resource, SupervisorActionAuthority) { - let (resource, _, request_id, task_id, input) = seed_serving_assigned_resource_task_with_spec( - home, - authority, - MachineId::new(), - resource_command(), - ); - let (loan_id, expected_state_revision) = (input.loan_id, input.expected_state_revision); - let mut store = Store::open(&home.db_path()).unwrap(); - assert!(matches!( - store.accept_assigned_resource_task(input).unwrap(), - ResourceTaskAcceptance::Inserted { .. } - )); - store - .cas_status(task_id, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - store - .cas_exit_with_evidence( - task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ) - .unwrap() - .unwrap(); - let crate::resource::store::AssignedResourceTaskReconcileOutcome::Completed(result) = store - .reconcile_assigned_resource_task_for_authority( - crate::resource::store::AssignedResourceTaskReconcileInput { - authority_machine: authority, - resource_id: resource.id, - loan_id, - request_id, - task_id, - expected_state_revision, - }, - ) - .unwrap() - else { - panic!("the drained request must complete"); - }; - let crate::resource::store::ResourceTaskCompletionResult::ReturnRequired { - loan, notice, .. - } = *result - else { - panic!("the drained queue must reserve the return"); - }; - let authority = SupervisorActionAuthority { - authority_machine: authority, - resource_id: resource.id, - loan_id: loan.id, - action_id: notice.action_id, - expected_state_revision: notice.state_revision, - supervisor: resource.supervisor, - assignment_revision: resource.assignment_revision, - }; - (resource, authority) -} - -/// Evaluation command that stays in its foreground process group until the gate exists -fn gated_return_launch( - root: &std::path::Path, - resource: &Resource, - marker: &std::path::Path, - gate: &std::path::Path, -) -> ReturnLaunch { - let command = crate::resource::foreground::test_support::native_fake_command(); - let spec: NormalizedSpec = serde_json::from_value(json!({ - "api_version": 1, - "thread": resource.supervisor.thread, - "name": "gated evaluation", - "cwd": root, - "timeout": "4h", - "workload": { "type": "task", "command": [command, marker, gate] } - })) - .unwrap(); - ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::EvaluationOrNextEpoch { - completed_task: resource.registered_background_task.unwrap(), - spec: CommandSpec::try_from(spec).unwrap(), - }, - } -} - -async fn decide_return_launch( - supervisor: &ActorRef, - authority: SupervisorActionAuthority, - launch: ReturnLaunch, -) -> ReturnTaskAcceptance { - let outcome = call(supervisor, |reply| SupervisorMsg::DecideReturn { - authority, - decision: Box::new(ReturnDecision::Launch(Box::new(launch))), - reply, - }) - .await - .unwrap() - .unwrap(); - let ReturnDecisionOutcome::Launch(acceptance) = outcome else { - panic!("a launch decision must return its task acceptance"); - }; - acceptance -} - -/// Native foreground return that keeps its Restoring loan, with no registration -fn foreground_restoring(inspection: &ResourceActorInspection, task_id: TaskId) -> bool { - matches!( - inspection.loan.as_ref().map(|loan| &loan.state), - Some(LoanState::Active { - phase: LoanPhase::Restoring { resume_task_id, .. } - }) if *resume_task_id == task_id - ) && inspection.resource.registered_background_task != Some(task_id) - && matches!( - inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::LoanAlreadyActive { .. }) - ) -} - -#[tokio::test] -async fn native_return_keeps_its_loan_while_running_and_its_end_serves_the_next_request() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let home = Home::resolve(Some(root.join("home"))).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, return_authority) = seed_awaiting_return(&home, authority); - // a request accepted after the return reservation waits for the next loan; - // its own gate keeps it serving until the test inspects that loan - let (later_marker, later_gate) = (root.join("later-marker"), root.join("later-gate")); - let mut later_spec = resource_command(); - later_spec.workload = NormalizedWorkload::Task(crate::spec::NormalizedTaskWorkload { - command: crate::invocation::CommandLine::try_from_argv(vec![ - crate::resource::foreground::test_support::native_fake_command() - .display() - .to_string(), - later_marker.display().to_string(), - later_gate.display().to_string(), - ]) - .unwrap(), - }); - let later = Store::open(&home.db_path()) - .unwrap() - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - later_spec, - ) - .unwrap(); - let (marker, gate) = (root.join("marker"), root.join("gate")); - let launch = gated_return_launch(&root, &resource, &marker, &gate); - let task_id = launch.task_id; - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - assert!(matches!( - decide_return_launch(&supervisor, return_authority, launch.clone()).await, - ReturnTaskAcceptance::Inserted { task, .. } if task == task_id - )); - let running = wait_for_running_task(&store, &home, task_id).await; - assert_eq!(running.thread, resource.supervisor.thread); - - // the running foreground task never opens a release for the queued request - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource.id, - reply, - }) - .await - .unwrap(); - let inspection = inspect_resource(&supervisor, resource.id).await.unwrap(); - assert!( - foreground_restoring(&inspection, task_id), - "{:?}", - (&inspection.loan, &inspection.reconcile_outcome) - ); - let requests = call(&store, |reply| StoreMsg::ResourceRequests { - authority_machine: authority, - resource_id: resource.id, - reply, - }) - .await - .unwrap(); - assert!( - requests - .iter() - .any(|request| request.request_id == later.request_id - && request.state == ResourceRequestState::Queued) - ); - - // an exact retry observes the running task and never respawns it - assert_eq!( - decide_return_launch(&supervisor, return_authority, launch).await, - ReturnTaskAcceptance::Existing { - task: task_id, - state: ProcessStatus::Running, - } - ); - std::fs::write(&gate, b"").unwrap(); - let ended = wait_for_terminal_task(&store, task_id).await; - assert_eq!(ended.status(), ProcessStatus::Succeeded); - assert_eq!(std::fs::read(&marker).unwrap(), b"x"); - - // the confirmed end closes the loan and queue reconciliation serves the request - let inspection = wait_for_inspection(&supervisor, resource.id, |inspection| { - idle_serving(inspection).is_some() - }) - .await; - assert_eq!( - idle_serving(&inspection), - Some(&IdleBoundaryProof::ForegroundReturnEnded { - loan_id: return_authority.loan_id, - task_id, - }) - ); - assert!(matches!( - inspection.loan.as_ref().map(|loan| &loan.state), - Some(LoanState::Active { - phase: LoanPhase::Serving { current_request_id, .. } - }) if *current_request_id == later.request_id - )); - assert_eq!(inspection.resource.registered_background_task, None); - std::fs::write(&later_gate, b"").unwrap(); - wait_for_terminal_task(&store, later.task_id).await; - assert_eq!(std::fs::read(&later_marker).unwrap(), b"x"); - - stop_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn startup_keeps_an_unspawned_return_task_reserved_until_explicit_resolution() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let home = Home::resolve(Some(root.join("home"))).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, return_authority) = seed_awaiting_return(&home, authority); - let (marker, gate) = (root.join("marker"), root.join("gate")); - let launch = gated_return_launch(&root, &resource, &marker, &gate); - let task_id = launch.task_id; - // the binding commits, then the daemon stops before it spawns a worker - let ReturnTaskAcceptance::Inserted { state_revision, .. } = Store::open(&home.db_path()) - .unwrap() - .accept_return_task_for_authority(ReturnTaskAcceptanceInput { - authority: return_authority, - launch: launch.clone(), - executor_env: TaskEnv::capture(), - origin: ReturnTaskOrigin::Local { - callback_codex: CallbackExecutable::available("/bin/echo".into()), - }, - }) - .unwrap() - else { - panic!("the first exact launch must insert its task"); - }; - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let attention = |inspection: &ResourceActorInspection| { - matches!( - &inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::RestoreAttentionRequired { - task_id: attention_task, - reason: RestoreAttentionReason::LaunchUncertain, - .. - }) if *attention_task == task_id - ) - }; - assert!(attention( - &inspect_resource(&supervisor, resource.id).await.unwrap() - )); - assert_eq!( - decide_return_launch(&supervisor, return_authority, launch).await, - ReturnTaskAcceptance::Existing { - task: task_id, - state: ProcessStatus::Queued, - } - ); - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource.id, - reply, - }) - .await - .unwrap(); - assert!(attention( - &inspect_resource(&supervisor, resource.id).await.unwrap() - )); - let row = call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .unwrap(); - assert_eq!(row.status(), ProcessStatus::Queued); - assert!(!marker.exists()); - - // cancelling the queued row proves no child spawned, so the supervisor can resolve it - assert!(matches!( - call(&supervisor, |reply| SupervisorMsg::Cancel { - id: task_id, - reply - }) - .await - .unwrap(), - CancelResult::CancelledQueued(_) - )); - let mut current = return_authority; - current.expected_state_revision = state_revision; - let closure = call(&supervisor, |reply| SupervisorMsg::ResolveEndedRestore { - resolution: Box::new(EndedRestoreResolution { - authority: current, - task_id, - reason: "return launch never started".into(), - }), - reply, - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - closure.loan.state, - LoanState::Closed { - result: crate::resource::LoanClosure::RestoreEnded { - outcome: ExitReason::Cancelled, - .. - } - } - )); - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource.id, - reply, - }) - .await - .unwrap(); - let inspection = inspect_resource(&supervisor, resource.id).await.unwrap(); - assert_eq!(inspection.loan, None); - assert_eq!(inspection.resource.registered_background_task, None); - assert!(!marker.exists()); - - stop_supervisor(supervisor, handle).await; -} - -/// Reassign the seeded supervisor thread to another machine -fn move_supervisor_to_another_machine( - home: &Home, - resource: &Resource, - action: &mut SupervisorActionAuthority, -) -> MachineId { - let mut remote = MachineId::new(); - while remote == action.authority_machine { - remote = MachineId::new(); - } - rusqlite::Connection::open(home.db_path()) - .unwrap() - .execute( - "UPDATE resources SET supervisor_machine = ?1 WHERE id = ?2", - rusqlite::params![remote.to_string(), resource.id.as_uuid().to_string()], - ) - .unwrap(); - action.supervisor.machine = remote; - remote -} - -async fn remote_action( - supervisor: &ActorRef, - action: SupervisorActionAuthority, - operation: ResourceActionOperation, -) -> ResourceActionOutcome { - call(supervisor, |reply| SupervisorMsg::ResourceAction { - request: Box::new(ResourceActionRequest::new(1, action, operation)), - reply, - }) - .await - .unwrap() -} - -#[tokio::test] -async fn remote_native_return_spawns_once_and_closes_after_its_confirmed_end() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let home = Home::resolve(Some(root.join("home"))).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, mut action) = seed_awaiting_return(&home, authority); - let remote = move_supervisor_to_another_machine(&home, &resource, &mut action); - let (marker, gate) = (root.join("marker"), root.join("gate")); - let launch = gated_return_launch(&root, &resource, &marker, &gate); - let task_id = launch.task_id; - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let ResourceActionOutcome::Prepared { task } = remote_action( - &supervisor, - action, - ResourceActionOperation::PrepareReturn { - launch: launch.clone(), - }, - ) - .await - else { - panic!("the authority must prepare the return task"); - }; - assert_eq!(task.task_id, task_id); - assert!(task.digest_matches()); - assert!( - call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .is_none() - ); - - let operation = ResourceActionOperation::LaunchReturn { - launch, - normalized_spec_sha256: task.normalized_spec_sha256, - }; - let ResourceActionOutcome::Accepted { - receipt, - acceptance: ActionTaskAcceptance::Inserted, - } = remote_action(&supervisor, action, operation.clone()).await - else { - panic!("the first remote launch must insert its task"); - }; - assert_eq!(receipt.origin_machine(), remote); - wait_for_running_task(&store, &home, task_id).await; - // the authority keeps no callback route for the remote supervisor's task - assert!( - call(&store, |reply| StoreMsg::OriginRoute { id: task_id, reply }) - .await - .unwrap() - .is_none() - ); - - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource.id, - reply, - }) - .await - .unwrap(); - let inspection = inspect_resource(&supervisor, resource.id).await.unwrap(); - assert!( - foreground_restoring(&inspection, task_id), - "{:?}", - (&inspection.loan, &inspection.reconcile_outcome) - ); - assert!(matches!( - remote_action(&supervisor, action, operation).await, - ResourceActionOutcome::Accepted { - acceptance: ActionTaskAcceptance::Existing { - state: ProcessStatus::Running - }, - .. - } - )); - - std::fs::write(&gate, b"").unwrap(); - wait_for_terminal_task(&store, task_id).await; - // the retry observed the running worker, so the command ran exactly once - assert_eq!(std::fs::read(&marker).unwrap(), b"x"); - // the Restoring loan was the only non-closed loan, so its closure leaves none - let inspection = wait_for_inspection(&supervisor, resource.id, |inspection| { - inspection.loan.is_none() - }) - .await; - assert_eq!(inspection.resource.registered_background_task, None); - // the remote supervisor keeps its route; the authority adds none - assert!( - call(&store, |reply| StoreMsg::OriginRoute { id: task_id, reply }) - .await - .unwrap() - .is_none() - ); - stop_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn remote_return_accepted_before_spawn_is_never_relaunched_after_restart() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let home = Home::resolve(Some(root.join("home"))).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, mut action) = seed_awaiting_return(&home, authority); - move_supervisor_to_another_machine(&home, &resource, &mut action); - let (marker, gate) = (root.join("marker"), root.join("gate")); - let launch = gated_return_launch(&root, &resource, &marker, &gate); - let task_id = launch.task_id; - // the remote acceptance commits, then the daemon stops before it spawns a worker - let mut store = Store::open(&home.db_path()).unwrap(); - let digest = store - .prepare_return_task_for_authority(action, launch.clone(), TaskEnv::capture()) - .unwrap() - .normalized_spec_sha256; - assert!(matches!( - store - .accept_return_task_for_authority(ReturnTaskAcceptanceInput { - authority: action, - launch: launch.clone(), - executor_env: TaskEnv::capture(), - origin: ReturnTaskOrigin::Remote { - normalized_spec_sha256: digest, - }, - }) - .unwrap(), - ReturnTaskAcceptance::Inserted { .. } - )); - drop(store); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let uncertain = |inspection: &ResourceActorInspection| { - matches!( - &inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::RestoreAttentionRequired { - task_id: attention_task, - reason: RestoreAttentionReason::LaunchUncertain, - .. - }) if *attention_task == task_id - ) - }; - assert!(uncertain( - &inspect_resource(&supervisor, resource.id).await.unwrap() - )); - // the lost-reply retry observes the queued row and starts nothing - assert!(matches!( - remote_action( - &supervisor, - action, - ResourceActionOperation::LaunchReturn { - launch, - normalized_spec_sha256: digest, - }, - ) - .await, - ResourceActionOutcome::Accepted { - acceptance: ActionTaskAcceptance::Existing { - state: ProcessStatus::Queued - }, - .. - } - )); - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource.id, - reply, - }) - .await - .unwrap(); - assert!(uncertain( - &inspect_resource(&supervisor, resource.id).await.unwrap() - )); - let row = call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .unwrap(); - assert_eq!(row.status(), ProcessStatus::Queued); - assert!(row.pid().is_none()); - assert!(!marker.exists()); - stop_supervisor(supervisor, handle).await; -} - -async fn launch_background_task( - supervisor: &ActorRef, - launch: BackgroundLaunch, -) -> Result { - call(supervisor, |reply| SupervisorMsg::LaunchBackground { - launch: Box::new(launch), - reply, - }) - .await - .unwrap() -} - -fn trainer_launch( - trainer: &FakeTrainer, - resource: &Resource, - request_id: RequestId, -) -> BackgroundLaunch { - BackgroundLaunch { - resource_id: resource.id, - request_id, - spec: trainer.spec(resource.supervisor.thread, "direct segment trainer"), - env: trainer.env.clone(), - callback_cwd: trainer.cwd.clone(), - } -} - -async fn wait_for_inspection( - supervisor: &ActorRef, - id: ResourceId, - predicate: impl Fn(&ResourceActorInspection) -> bool, -) -> ResourceActorInspection { - let result = tokio::time::timeout(std::time::Duration::from_secs(10), async { - loop { - let inspection = inspect_resource(supervisor, id).await.unwrap(); - if predicate(&inspection) { - return inspection; - } - tokio::time::sleep(std::time::Duration::from_millis(20)).await; - } - }) - .await; - match result { - Ok(inspection) => inspection, - Err(_) => panic!( - "resource never reached the expected state: {:?}", - inspect_resource(supervisor, id) - .await - .map(|inspection| (inspection.loan, inspection.reconcile_outcome)) - ), - } -} - -async fn wait_for_terminal_task(store: &ActorRef, task_id: TaskId) -> TaskRow { - tokio::time::timeout(std::time::Duration::from_secs(10), async { - loop { - let row = call(store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap(); - if let Some(row) = row.filter(|row| row.state.is_terminal()) { - return row; - } - tokio::time::sleep(std::time::Duration::from_millis(20)).await; - } - }) - .await - .unwrap() -} - -fn idle_serving(inspection: &ResourceActorInspection) -> Option<&IdleBoundaryProof> { - match inspection.loan.as_ref().map(|loan| &loan.state) { - Some(LoanState::Active { - phase: - LoanPhase::Serving { - return_context: ReturnContext::Idle, - release_provenance: - crate::resource::ServingReleaseProvenance::IdleBoundary { proof }, - .. - }, - }) => Some(proof), - _ => None, - } -} - -#[tokio::test] -async fn first_background_launch_spawns_once_and_registers_after_its_start_event() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let home = Home::resolve(Some(root.join("home"))).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let trainer = FakeTrainer::new(&root); - let resource = resource(authority, None); - Store::open(&home.db_path()) - .unwrap() - .register_resource(authority, &resource) - .unwrap(); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let request_id = RequestId::new(); - let BackgroundLaunchAcceptance::Inserted { task, .. } = - launch_background_task(&supervisor, trainer_launch(&trainer, &resource, request_id)) - .await - .unwrap() - else { - panic!("the first exact launch must insert and spawn its task"); - }; - wait_for_running_task(&store, &home, task).await; - - // the delivered start event wakes the owner; no caller reconcile is needed - // The event sender is not running here, so send the same inbox hint it sends - supervisor - .cast(SupervisorMsg::DispatchInbox { id: task }) - .unwrap(); - wait_for_inspection(&supervisor, resource.id, |inspection| { - inspection.resource.registered_background_task == Some(task) - }) - .await; - - // an exact retry observes the running task and never spawns a second one - assert_eq!( - launch_background_task(&supervisor, trainer_launch(&trainer, &resource, request_id)) - .await - .unwrap(), - BackgroundLaunchAcceptance::Existing { - task, - state: ProcessStatus::Running, - } - ); - // another request cannot start a competing background task - assert!(matches!( - launch_background_task( - &supervisor, - trainer_launch(&trainer, &resource, RequestId::new()) - ) - .await, - Err(BackgroundLaunchError::BackgroundTaskActive { task_id, .. }) if task_id == task - )); - - std::fs::write(&trainer.gate, b"").unwrap(); - wait_for_terminal_task(&store, task).await; - assert_eq!(trainer.starts(), 1); - - stop_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn committed_unspawned_background_launch_is_never_respawned_after_restart() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let home = Home::resolve(Some(root.join("home"))).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let trainer = FakeTrainer::new(&root); - let resource = resource(authority, None); - let request_id = RequestId::new(); - // the binding commits, then the daemon stops before it spawns a worker - let (task, later) = { - let mut store = Store::open(&home.db_path()).unwrap(); - store.register_resource(authority, &resource).unwrap(); - let launch = trainer_launch(&trainer, &resource, request_id); - let BackgroundLaunchAcceptance::Inserted { task, .. } = store - .accept_background_launch_for_authority(BackgroundLaunchInput { - authority_machine: authority, - resource_id: resource.id, - request_id, - task_id: TaskId::new(), - spec: launch.spec, - env: launch.env, - callback_codex: CallbackExecutable::available("/bin/echo".into()), - }) - .unwrap() - else { - panic!("the first exact launch must insert its task"); - }; - let later = store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - resource_command(), - ) - .unwrap(); - (task, later) - }; - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let uncertain = |inspection: &ResourceActorInspection| { - matches!( - inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::BackgroundLaunchUncertain { task_id }) - if task_id == task - ) - }; - assert!(uncertain( - &inspect_resource(&supervisor, resource.id).await.unwrap() - )); - assert_eq!( - launch_background_task(&supervisor, trainer_launch(&trainer, &resource, request_id)) - .await - .unwrap(), - BackgroundLaunchAcceptance::Existing { - task, - state: ProcessStatus::Queued, - } - ); - tokio::time::sleep(std::time::Duration::from_millis(200)).await; - let inspection = inspect_resource(&supervisor, resource.id).await.unwrap(); - assert!(uncertain(&inspection)); - assert_eq!(inspection.resource.registered_background_task, None); - assert_eq!(inspection.loan, None); - let row = call(&store, |reply| StoreMsg::GetTask { id: task, reply }) - .await - .unwrap() - .unwrap(); - assert_eq!(row.status(), ProcessStatus::Queued); - assert_eq!(trainer.starts(), 0); - - // cancelling the queued row proves no child started, which is the idle proof - assert!(matches!( - call(&supervisor, |reply| SupervisorMsg::Cancel { - id: task, - reply - }) - .await - .unwrap(), - CancelResult::CancelledQueued(_) - )); - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource.id, - reply, - }) - .await - .unwrap(); - let inspection = wait_for_inspection(&supervisor, resource.id, |inspection| { - idle_serving(inspection).is_some() - }) - .await; - assert_eq!( - idle_serving(&inspection), - Some(&IdleBoundaryProof::BackgroundLaunchNeverSpawned { - request_id, - task_id: task, - }) - ); - wait_for_terminal_task(&store, later.task_id).await; - assert_eq!(trainer.starts(), 0); - - stop_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn no_resume_closure_runs_a_later_request_from_the_idle_boundary() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let home = Home::resolve(Some(root.join("home"))).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, return_authority) = seed_awaiting_return(&home, authority); - let later = Store::open(&home.db_path()) - .unwrap() - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - resource_command(), - ) - .unwrap(); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - // the reservation stays in place until the supervisor decides - assert!(matches!( - inspect_resource(&supervisor, resource.id) - .await - .unwrap() - .loan - .map(|loan| loan.state), - Some(LoanState::Active { - phase: LoanPhase::AwaitingReturn { .. } - }) - )); - let outcome = call(&supervisor, |reply| SupervisorMsg::DecideReturn { - authority: return_authority, - decision: Box::new(ReturnDecision::NoResume { - reason: "training is complete".into(), + name: Some(spec.name.clone()), + thread: spec.thread, + workload: Workload::Task(TaskWorkload { + command: workload.command, }), - reply, - }) - .await - .unwrap() - .unwrap(); - assert!(matches!(outcome, ReturnDecisionOutcome::Closed(_))); - - let inspection = wait_for_inspection(&supervisor, resource.id, |inspection| { - idle_serving(inspection).is_some() - }) - .await; + cwd: spec.cwd.clone(), + timeout: spec.timeout, + env: TaskEnv { + path: "/bin".into(), + home: "/tmp".into(), + }, + binary: "/bin/echo".into(), + }); assert_eq!( - idle_serving(&inspection), - Some(&IdleBoundaryProof::SupervisorNoResume { - loan_id: return_authority.loan_id, - }) + startup_recovery_action(&row, None, false), + StartupRecoveryAction::Observe ); - wait_for_terminal_task(&store, later.task_id).await; - - stop_supervisor(supervisor, handle).await; -} - -/// Socket-route `submit` on a supervisor that is also the resource authority -mod co_located_action { - use super::{ - configure_task_runner, gated_return_launch, seed_awaiting_return, stop_supervisor, - wait_for_running_task, wait_for_terminal_task, - }; - use crate::daemon::AppState; - use crate::daemon::actors::supervisor::{ - SUPERVISOR_TEST_LOCK, SupervisorActor, SupervisorArgs, SupervisorMsg, - }; - use crate::daemon::actors::{StoreMsg, call}; - use crate::daemon::resource_action::submit; - use crate::domain::{ExitReason, ProcessStatus, TaskEnv, TaskId, ThreadId}; - use crate::error::AppError; - use crate::files::StreamSlots; - use crate::fleet::FleetState; - use crate::fleet::directory::LocalMachine; - use crate::fleet::protocol::SUPPORTED_PROTOCOLS; - use crate::home::Home; - use crate::machine::{LocalIdentity, MachineId, MachineName, load_or_create_machine_id}; - use crate::resource::bound_action::{ - LocalReturnAcceptance, LocalReturnReceipt, ResourceActionChoice, ResourceActionRejection, - ResourceActionSubmitOutcome, - }; - use crate::resource::{ - AssignmentRevision, LoanState, ResourceId, ResourceRevision, ReturnDecision, - SupervisorActionAuthority, SupervisorAddress, - }; - use crate::store::{ - CancelResult, ReturnTaskAcceptance, ReturnTaskAcceptanceInput, ReturnTaskOrigin, Store, - }; - use crate::submission::{CallbackExecutable, RequestId}; - use ractor::{Actor, ActorRef}; - use tempfile::tempdir; - use uuid::Uuid; - - fn app_state( - home: &Home, - supervisor: &ActorRef, - store: ActorRef, - ) -> AppState { - AppState { - home: home.clone(), - store, - supervisor: supervisor.clone(), - web: None, - content: None, - stream_slots: StreamSlots::new(), - machine: LocalMachine { - identity: LocalIdentity::start(home).unwrap(), - name: MachineName::fallback(), - protocol: SUPPORTED_PROTOCOLS, - }, - fleet: FleetState::Disabled, - message_receiver: crate::daemon::message_receiver::MessageReceiver::default(), - locks: crate::daemon::DaemonLocks::default(), - thread_titles: None, - } - } - - fn launch_choice(launch: crate::resource::ReturnLaunch) -> ResourceActionChoice { - ResourceActionChoice::Return { - decision: ReturnDecision::Launch(Box::new(launch)), - } - } - - fn no_resume(reason: &str) -> ResourceActionChoice { - ResourceActionChoice::Return { - decision: ReturnDecision::NoResume { - reason: reason.into(), - }, - } - } - - fn conflicting_retry(outcome: &ResourceActionSubmitOutcome) -> bool { - matches!( - outcome, - ResourceActionSubmitOutcome::Rejected { - reason: ResourceActionRejection::ConflictingRetry - } - ) - } - - #[tokio::test] - async fn return_launch_spawns_once_replays_and_refuses_other_content() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let home = Home::resolve(Some(root.join("home"))).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, return_authority) = seed_awaiting_return(&home, authority); - let (marker, gate) = (root.join("marker"), root.join("gate")); - let launch = gated_return_launch(&root, &resource, &marker, &gate); - let (request_id, task_id) = (launch.request_id, launch.task_id); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let state = app_state(&home, &supervisor, store.clone()); - let receipt = LocalReturnReceipt { - authority: return_authority, - request_id, - task_id, - }; - - let first = submit(&state, return_authority, launch_choice(launch.clone())) - .await - .unwrap(); - let ResourceActionSubmitOutcome::LocalReturnAccepted { - receipt: first_receipt, - acceptance: LocalReturnAcceptance::Inserted { loan, .. }, - } = first - else { - panic!("the first co-located launch must insert its task: {first:?}"); - }; - assert_eq!(first_receipt, receipt); - assert_eq!(loan.id, return_authority.loan_id); - wait_for_running_task(&store, &home, task_id).await; - - // an exact retry observes the bound task and never spawns it again - let retry = submit(&state, return_authority, launch_choice(launch.clone())) - .await - .unwrap(); - assert!(matches!( - retry, - ResourceActionSubmitOutcome::LocalReturnAccepted { - receipt: retry_receipt, - acceptance: LocalReturnAcceptance::Existing { - state: ProcessStatus::Running - }, - } if retry_receipt == receipt - )); - - let mut other = launch; - other.request_id = RequestId::new(); - other.task_id = TaskId::new(); - let other_task = other.task_id; - let conflict = submit(&state, return_authority, launch_choice(other)) - .await - .unwrap(); - assert!(conflicting_retry(&conflict), "{conflict:?}"); - let absent = call(&store, |reply| StoreMsg::GetTask { - id: other_task, - reply, - }) - .await - .unwrap(); - assert!(absent.is_none()); - - // the resource actor owns a co-located watcher, so the socket path refuses one - let watcher = submit( - &state, - return_authority, - ResourceActionChoice::ReleaseWatcher { - observed_background_task: task_id, - }, - ) - .await; - assert!(matches!( - watcher, - Err(AppError::ResourceActionNotAllowed { .. }) - )); - - std::fs::write(&gate, b"").unwrap(); - wait_for_terminal_task(&store, task_id).await; - assert_eq!(std::fs::read(&marker).unwrap(), b"x"); - - stop_supervisor(supervisor, handle).await; - } - - #[tokio::test] - async fn no_resume_closes_once_and_replays_the_saved_closure() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let home = Home::resolve(Some(root.join("home"))).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (_, return_authority) = seed_awaiting_return(&home, authority); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let state = app_state(&home, &supervisor, store); - - let first = submit(&state, return_authority, no_resume("training is complete")) - .await - .unwrap(); - let ResourceActionSubmitOutcome::Closed { - loan, - state_revision, - } = first - else { - panic!("no-resume must close the loan: {first:?}"); - }; - assert_eq!(loan.id, return_authority.loan_id); - assert!(matches!(loan.state, LoanState::Closed { .. })); - - let retry = submit(&state, return_authority, no_resume("training is complete")) - .await - .unwrap(); - assert!(matches!( - retry, - ResourceActionSubmitOutcome::Closed { - loan: ref replayed, - state_revision: replayed_revision, - } if *replayed == loan && replayed_revision == state_revision - )); - - let conflict = submit(&state, return_authority, no_resume("another reason")) - .await - .unwrap(); - assert!(conflicting_retry(&conflict), "{conflict:?}"); - - stop_supervisor(supervisor, handle).await; - } - - #[tokio::test] - async fn queued_return_after_restart_is_observed_then_resolved_without_spawn() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let home = Home::resolve(Some(root.join("home"))).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, return_authority) = seed_awaiting_return(&home, authority); - let (marker, gate) = (root.join("marker"), root.join("gate")); - let launch = gated_return_launch(&root, &resource, &marker, &gate); - let task_id = launch.task_id; - // the binding commits, then the daemon stops before it spawns a worker - let ReturnTaskAcceptance::Inserted { state_revision, .. } = Store::open(&home.db_path()) - .unwrap() - .accept_return_task_for_authority(ReturnTaskAcceptanceInput { - authority: return_authority, - launch: launch.clone(), - executor_env: TaskEnv::capture(), - origin: ReturnTaskOrigin::Local { - callback_codex: CallbackExecutable::available("/bin/echo".into()), - }, - }) - .unwrap() - else { - panic!("the first exact launch must insert its task"); - }; - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let state = app_state(&home, &supervisor, store.clone()); - - // the queued return found after restart is only observed - let observed = submit(&state, return_authority, launch_choice(launch)) - .await - .unwrap(); - assert!(matches!( - observed, - ResourceActionSubmitOutcome::LocalReturnAccepted { - receipt: LocalReturnReceipt { task_id: observed_task, .. }, - acceptance: LocalReturnAcceptance::Existing { - state: ProcessStatus::Queued - }, - } if observed_task == task_id - )); - let row = call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .unwrap(); - assert_eq!(row.status(), ProcessStatus::Queued); - - let mut current = return_authority; - current.expected_state_revision = state_revision; - let resolve = |reason: &str| ResourceActionChoice::ResolveEndedRestore { - task_id, - reason: reason.into(), - }; - // a queued task has not ended, so resolution is refused without a write - let early = submit(&state, current, resolve("launch never started")) - .await - .unwrap(); - assert!(matches!( - early, - ResourceActionSubmitOutcome::Rejected { - reason: ResourceActionRejection::RestoreNotResolvable { .. } - } - )); - - // cancelling the queued row proves no child spawned - assert!(matches!( - call(&supervisor, |reply| SupervisorMsg::Cancel { - id: task_id, - reply - }) - .await - .unwrap(), - CancelResult::CancelledQueued(_) - )); - let closed = submit(&state, current, resolve("launch never started")) - .await - .unwrap(); - let ResourceActionSubmitOutcome::Closed { loan, .. } = &closed else { - panic!("resolution must close the loan: {closed:?}"); - }; - assert_eq!(loan.id, return_authority.loan_id); - assert!(matches!( - loan.state, - LoanState::Closed { - result: crate::resource::LoanClosure::RestoreEnded { - outcome: ExitReason::Cancelled, - .. - } - } - )); - let replay = submit(&state, current, resolve("launch never started")) - .await - .unwrap(); - assert!(matches!( - &replay, - ResourceActionSubmitOutcome::Closed { loan: replayed, .. } if replayed == loan - )); - let conflict = submit(&state, current, resolve("another reason")) - .await - .unwrap(); - assert!(conflicting_retry(&conflict), "{conflict:?}"); - assert!(!marker.exists()); - - stop_supervisor(supervisor, handle).await; - } - #[tokio::test] - async fn remote_authority_keeps_the_route_path_and_foreign_supervisor_is_refused() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().join("home"))).unwrap(); - home.ensure().unwrap(); - let local = load_or_create_machine_id(&home).unwrap(); - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let state = app_state(&home, &supervisor, store); - let remote = SupervisorActionAuthority { - authority_machine: MachineId::new(), - resource_id: ResourceId::new(), - loan_id: crate::resource::LoanId::new(), - action_id: crate::resource::ActionId::new(), - expected_state_revision: ResourceRevision::new(1), - supervisor: SupervisorAddress { - machine: local, - thread: ThreadId(Uuid::now_v7()), - }, - assignment_revision: AssignmentRevision::new(1), - }; - - // a remote watcher still goes through the authority, which is unreachable here - let watcher = submit( - &state, - remote, - ResourceActionChoice::ReleaseWatcher { - observed_background_task: TaskId::new(), - }, - ) - .await; - assert!( - matches!(watcher, Err(AppError::MachineUnavailable { machine, .. }) if machine == remote.authority_machine) - ); - let closure = submit(&state, remote, no_resume("done")).await; - assert!(matches!(closure, Err(AppError::MachineUnavailable { .. }))); - - let mut foreign = remote; - foreign.supervisor.machine = MachineId::new(); - foreign.authority_machine = local; - assert!(matches!( - submit(&state, foreign, no_resume("done")).await, - Err(AppError::Usage { .. }) - )); - - stop_supervisor(supervisor, handle).await; + let origin_machine = MachineId::new(); + let mut execution_machine = MachineId::new(); + while execution_machine == origin_machine { + execution_machine = MachineId::new(); } + let identity = ExecutorIdentity::Accepted(ExecutionRecord { + task: id, + origin_machine, + execution_machine, + spec: spec.into(), + state: ProcessStatus::Queued, + }); + assert_eq!( + startup_recovery_action(&row, Some(&identity), false), + StartupRecoveryAction::LaunchAccepted + ); } #[test] @@ -2669,249 +335,6 @@ fn resolved_callback_codex_is_available() { assert_eq!(codex, CallbackExecutable::available(path)); } -fn command_in(cwd: &std::path::Path) -> NormalizedSpec { - serde_json::from_value(json!({ - "api_version": 1, - "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "resource launch failure test", - "cwd": cwd, - "timeout": "4h", - "workload": { "type": "task", "command": ["/bin/echo", "hello"] } - })) - .unwrap() -} - -async fn wait_for_failed_task(store: &ActorRef, task_id: TaskId) -> TaskRow { - tokio::time::timeout(std::time::Duration::from_secs(10), async { - loop { - if let Some(row) = call(store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - && row.status() == ProcessStatus::Failed - { - return row; - } - - tokio::time::sleep(std::time::Duration::from_millis(10)).await; - } - }) - .await - .expect("the assigned task must reach a durable failed state") -} - -async fn request_states( - store: &ActorRef, - authority: MachineId, - resource: ResourceId, -) -> Vec<(RequestId, ResourceRequestState)> { - call(store, |reply| StoreMsg::ResourceRequests { - authority_machine: authority, - resource_id: resource, - reply, - }) - .await - .unwrap() - .into_iter() - .map(|request| (request.request_id, request.state)) - .collect() -} - -fn spawn_failure_message(row: &TaskRow) -> String { - match row.exit_reason() { - Some(ExitReason::SpawnFailed { message }) => message.clone(), - other => panic!("expected a spawn failure, got {other:?}"), - } -} - -// the cwd existed when the request was queued and was removed before its turn, -// so the launch fails for a known reason and must not leave the queue blocked -#[tokio::test] -async fn known_launch_failure_fails_the_task_notifies_and_serves_the_next_request() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let work = directory.path().join("work"); - std::fs::create_dir(&work).unwrap(); - // a remote origin keeps the task's events in this executor's outbox - let origin = MachineId::new(); - let (resource, _loan, request_id, task_id, _input) = - seed_serving_assigned_resource_task_with_spec(&home, authority, origin, command_in(&work)); - let next_request = RequestId::new(); - Store::open(&home.db_path()) - .unwrap() - .accept_resource_request( - authority, - next_request, - TaskId::new(), - resource.id, - origin, - resource_command(), - ) - .unwrap(); - std::fs::remove_dir(&work).unwrap(); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - - let row = wait_for_failed_task(&store, task_id).await; - let message = spawn_failure_message(&row); - assert!(message.contains("launch failed before start"), "{message}"); - assert!(message.contains("invalid cwd"), "{message}"); - assert_eq!( - Store::open(&home.db_path()) - .unwrap() - .process_group_exit_evidence(task_id) - .unwrap(), - Some(ProcessGroupExitEvidence::NoChildSpawned) - ); - let output = std::fs::read_to_string(home.task_paths(task_id).output).unwrap(); - assert!(output.contains("invalid cwd"), "{output}"); - - // the thread hears the failure through the task's terminal event - let events = Store::open(&home.db_path()) - .unwrap() - .pending_outbound_events(task_id) - .unwrap(); - assert!( - events - .iter() - .any(|event| event.event.payload.process_state() == Some(ProcessStatus::Failed)), - "{events:?}" - ); - - // the failed request finishes and the next one owns the loan - let served = tokio::time::timeout(std::time::Duration::from_secs(10), async { - loop { - let states = request_states(&store, authority, resource.id).await; - let finished = states.iter().any(|(id, state)| { - *id == request_id - && matches!( - state, - ResourceRequestState::Finished { - outcome: ExitReason::SpawnFailed { .. } - } - ) - }); - let next_served = states.iter().any(|(id, state)| { - *id == next_request && !matches!(state, ResourceRequestState::Queued) - }); - if finished && next_served { - return; - } - - tokio::time::sleep(std::time::Duration::from_millis(10)).await; - } - }) - .await; - assert!( - served.is_ok(), - "{:?}", - request_states(&store, authority, resource.id).await - ); - - stop_supervisor(supervisor, handle).await; -} - -// a queued row this daemon lifetime did not insert has an unknown launch outcome; -// after the bound it is failed as launch_unconfirmed and the queue moves on -#[tokio::test] -async fn unknown_launch_outcome_fails_as_unconfirmed_after_the_bound() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - configure_task_runner(); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let (resource, loan, request_id, task_id, input) = - seed_serving_assigned_resource_task(&home, authority, authority); - assert!(matches!( - Store::open(&home.db_path()) - .unwrap() - .accept_assigned_resource_task(input) - .unwrap(), - ResourceTaskAcceptance::Inserted { task } if task == task_id - )); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let inspection = inspect_resource(&supervisor, resource.id).await.unwrap(); - assert!(matches!( - inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::AttentionRequired { - reason: ResourceQueueAttentionReason::AcceptedTaskLaunchUncertain { .. }, - .. - }) - )); - - // before the bound the task keeps its unknown outcome - tokio::time::pause(); - tokio::time::advance(crate::resource::LAUNCH_CONFIRMATION_BOUND / 2).await; - tokio::time::resume(); - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource.id, - reply, - }) - .await - .unwrap(); - let row = call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .unwrap(); - assert_eq!(row.status(), ProcessStatus::Queued); - - tokio::time::pause(); - tokio::time::advance(crate::resource::LAUNCH_CONFIRMATION_BOUND).await; - tokio::time::resume(); - let row = wait_for_failed_task(&store, task_id).await; - let message = spawn_failure_message(&row); - assert!(message.starts_with("launch_unconfirmed"), "{message}"); - assert_eq!(row.pid(), None); - - let released = tokio::time::timeout(std::time::Duration::from_secs(10), async { - loop { - let inspection = inspect_resource(&supervisor, resource.id).await.unwrap(); - if matches!( - inspection.loan.as_ref().map(|saved| &saved.state), - Some(LoanState::Active { - phase: LoanPhase::AwaitingReturn { .. }, - }) - ) { - return inspection; - } - - tokio::time::sleep(std::time::Duration::from_millis(10)).await; - } - }) - .await - .expect("the unconfirmed launch must release its serving turn"); - assert_eq!(released.loan.map(|saved| saved.id), Some(loan.id)); - assert!(matches!( - request_states(&store, authority, resource.id).await[..], - [(id, ResourceRequestState::Finished { .. })] if id == request_id - )); - - stop_supervisor(supervisor, handle).await; -} - /// Command task whose row and spec match, running `script` under `/bin/sh` in `/tmp` fn shell_task(script: &str) -> (TaskRow, NormalizedSpec) { let spec: NormalizedSpec = serde_json::from_value(json!({ diff --git a/src/daemon/actors/task.rs b/src/daemon/actors/task.rs index c31620c..03741bc 100644 --- a/src/daemon/actors/task.rs +++ b/src/daemon/actors/task.rs @@ -370,7 +370,7 @@ async fn apply_after_lock(actor: &TaskActor, id: TaskId) -> Result Result { let id = row.id; let claimed = call(&actor.store, |reply| StoreMsg::ClaimContainerAdoption { diff --git a/src/daemon/api.rs b/src/daemon/api.rs index 599554d..adeb9db 100644 --- a/src/daemon/api.rs +++ b/src/daemon/api.rs @@ -13,9 +13,7 @@ use serde::{Deserialize, Serialize}; use serde_json::Value; use crate::callback::last_event_for_row; -use crate::cancellation::{ - CancellationOwner, CancellationPlan, CancellationRoute, CancellationTarget, -}; +use crate::cancellation::{CancellationOwner, CancellationRoute}; use crate::daemon::actors::{StoreMsg, SupervisorMsg, call}; use crate::daemon::api::views::{ ContainerDetail, DependencyView, LogTail, StatusBody, TaskDetail, TaskFollowupSource, TaskList, @@ -51,7 +49,6 @@ impl IntoResponse for AppError { pub fn read_routes() -> Router { Router::new() .merge(crate::daemon::fleet_api::read_routes()) - .merge(crate::daemon::resource_api::read_routes()) .merge(crate::daemon::fleet_tasks::read_routes()) .merge(crate::daemon::thread_titles::read_routes()) .route("/v1/status", get(status)) @@ -78,9 +75,6 @@ pub fn socket_router(state: AppState) -> Router { .merge(write_routes()) .route("/v1/tasks/{id}/followup-source", get(followup_source)) .merge(crate::daemon::fleet_api::socket_routes()) - .merge(crate::daemon::release_watcher_api::socket_routes()) - .merge(crate::daemon::resource_action::socket_routes()) - .merge(crate::daemon::resource_api::socket_routes()) .with_state(state) } @@ -588,34 +582,19 @@ async fn cancel( } let local = call(&state.store, |reply| StoreMsg::GetTask { id, reply }).await?; - if local.is_some() && local_task_already_cancelled(&state, id).await? { - return Ok(Json(CancelResponse::status(id, ProcessStatus::Cancelled))); - } - if local.is_none() { let owner = crate::daemon::inspection::cancellation_owner(&state, id).await?; - let plan = owner - .plan(state.machine.identity.machine, id, uuid::Uuid::now_v7()) + let request = owner + .request(state.machine.identity.machine, id, uuid::Uuid::now_v7()) .map_err(|refusal| refusal.into_error(id))?; - let response = match plan { - CancellationPlan::AlreadyCancelled => { - CancelResponse::status(id, ProcessStatus::Cancelled) - } - CancellationPlan::Deliver(request) => { - let (saved, _) = call(&state.store, |reply| StoreMsg::InsertCancellationRequest { - request: *request, - reply, - }) - .await?; - CancelResponse::intent(&saved) - } - CancellationPlan::ForwardToOrigin(target) => { - crate::daemon::cluster::forward_resource_cancellation_intent(&state, &target) - .await? - } - }; - return Ok(Json(response)); + let (saved, _) = call(&state.store, |reply| StoreMsg::InsertCancellationRequest { + request, + reply, + }) + .await?; + return Ok(Json(CancelResponse::intent(&saved))); } + check_local_cancellation_owner(&state, id).await?; let result = call(&state.supervisor, |reply| SupervisorMsg::Cancel { id, reply, @@ -629,7 +608,8 @@ async fn cancel( Ok(Json(response)) } -async fn local_task_already_cancelled(state: &AppState, id: TaskId) -> Result { +/// Refuse to cancel a local row through the supervisor unless this machine executes it +async fn check_local_cancellation_owner(state: &AppState, id: TaskId) -> Result<(), AppError> { let machine = state.machine.identity.machine; let route = call(&state.store, |reply| StoreMsg::OriginRoute { id, reply }).await?; let owner = if let Some(route) = route { @@ -642,27 +622,15 @@ async fn local_task_already_cancelled(state: &AppState, id: TaskId) -> Result Ok(true), - // only an ordinary execution intent aimed at this machine may use the local row - CancellationPlan::Deliver(request) - if request.execution_machine == machine - && matches!(request.target, CancellationTarget::Execution { .. }) => - { - Ok(false) - } - CancellationPlan::Deliver(_) | CancellationPlan::ForwardToOrigin(_) => { - Err(AppError::ClusterTaskConflict { task: id }) - } + if owner.execution_machine != machine { + return Err(AppError::ClusterTaskConflict { task: id }); } + Ok(()) } #[cfg(test)] diff --git a/src/daemon/cancel_delivery.rs b/src/daemon/cancel_delivery.rs index 9b17ebd..2f0296d 100644 --- a/src/daemon/cancel_delivery.rs +++ b/src/daemon/cancel_delivery.rs @@ -12,8 +12,7 @@ use super::AppState; use super::actors::{StoreMsg, SupervisorMsg, call}; use super::cluster::{CancelBody, CancelExecution}; use crate::cancellation::{ - CancellationDelivery, CancellationReceipt, CancellationRequest, CancellationTarget, - ExecutorCancelState, ResourceCancellationRequestIdentity, + CancellationDelivery, CancellationReceipt, CancellationRequest, ExecutorCancelState, }; use crate::domain::{API_VERSION, ProcessStatus, TaskId}; use crate::error::AppError; @@ -21,9 +20,7 @@ use crate::fleet::http::ClusterClient; use crate::fleet::protocol::ClusterProtocolVersion; use crate::machine::MachineId; use crate::store::CancelResult; -use crate::submission::{ - ExecutorIdentity, ResourceCancellationOutcome, ResourceCancellationReceipt, -}; +use crate::submission::ExecutorIdentity; struct Retry { next: Instant, @@ -91,19 +88,6 @@ pub(super) async fn run(state: AppState) { } async fn deliver(state: &AppState, request: &CancellationRequest) -> Result<(), AppError> { - match &request.target { - CancellationTarget::Execution { .. } => {} - CancellationTarget::Resource(target) => { - if target.task_id != request.task - || target.origin_machine != request.origin_machine - || target.authority_machine != request.execution_machine - { - return Err(AppError::ClusterTaskConflict { task: request.task }); - } - return deliver_resource(state, request).await; - } - } - let receipt = if request.execution_machine == state.machine.identity.machine { call(&state.store, |reply| StoreMsg::ReceiveCancellation { request: request.identity(), @@ -151,122 +135,6 @@ async fn deliver(state: &AppState, request: &CancellationRequest) -> Result<(), Ok(()) } -async fn deliver_resource(state: &AppState, request: &CancellationRequest) -> Result<(), AppError> { - let identity = request - .resource_identity() - .ok_or(AppError::ClusterTaskConflict { task: request.task })?; - if identity.requester_machine != identity.origin_machine - || identity.authority_machine != request.execution_machine - { - return Err(AppError::ClusterTaskConflict { task: request.task }); - } - let receipt = if identity.authority_machine == state.machine.identity.machine { - super::cluster::process_resource_cancellation(state, identity.clone()).await? - } else { - deliver_remote_resource(state, &identity).await? - }; - verify_resource_receipt(&identity, &receipt)?; - if matches!( - &receipt.outcome, - ResourceCancellationOutcome::PreventedBeforeAcceptance - | ResourceCancellationOutcome::CancelledBeforeLaunch - ) { - call(&state.store, |reply| { - StoreMsg::CancelResourceRouteBeforeLaunch { - receipt: receipt.clone(), - reply, - } - }) - .await?; - } - call(&state.store, |reply| { - StoreMsg::AcknowledgeResourceCancellation { receipt, reply } - }) - .await?; - Ok(()) -} - -async fn deliver_remote_resource( - state: &AppState, - identity: &ResourceCancellationRequestIdentity, -) -> Result { - let fleet = state.fleet.handle().ok_or(AppError::MachineUnavailable { - machine: identity.authority_machine, - message: "fleet is disabled".into(), - })?; - let destination = fleet.connect(identity.authority_machine).await?; - let body = super::cluster::CancelResourceRequest { - api_version: API_VERSION, - protocol_version: destination.protocol.0, - destination_machine: identity.authority_machine, - request: identity.clone(), - }; - let response = ClusterClient::default() - .post_json( - &destination.address, - "/v1/cluster/resource-requests/cancel", - &body, - ) - .await - .map_err(|error| AppError::MachineUnavailable { - machine: identity.authority_machine, - message: error.to_string(), - })?; - if response.status != StatusCode::OK { - return Err(AppError::MachineUnavailable { - machine: identity.authority_machine, - message: format!("resource cancellation response status {}", response.status), - }); - } - let value: serde_json::Value = - serde_json::from_slice(&response.body).map_err(|error| AppError::MachineUnavailable { - machine: identity.authority_machine, - message: format!("invalid resource cancellation acknowledgement: {error}"), - })?; - if value.get("api_version").and_then(serde_json::Value::as_u64) != Some(u64::from(API_VERSION)) - { - return Err(AppError::MachineUnavailable { - machine: identity.authority_machine, - message: "resource cancellation acknowledgement uses an unsupported API version".into(), - }); - } - let body: super::cluster::CancelResourceBody = - serde_json::from_value(value).map_err(|error| AppError::MachineUnavailable { - machine: identity.authority_machine, - message: format!("invalid resource cancellation acknowledgement: {error}"), - })?; - if body.api_version != API_VERSION - || body.protocol_version != destination.protocol.0 - || body.destination_machine != identity.authority_machine - { - return Err(AppError::ClusterTaskConflict { - task: identity.task, - }); - } - verify_resource_receipt(identity, &body.receipt)?; - Ok(body.receipt) -} - -fn verify_resource_receipt( - identity: &ResourceCancellationRequestIdentity, - receipt: &ResourceCancellationReceipt, -) -> Result<(), AppError> { - if receipt.cancellation != identity.cancellation - || receipt.requester_machine != identity.requester_machine - || receipt.request != identity.request - || receipt.task != identity.task - || receipt.origin_machine != identity.origin_machine - || receipt.authority_machine != identity.authority_machine - || receipt.resource != identity.resource - || receipt.target_phase != identity.target_phase - { - return Err(AppError::ClusterTaskConflict { - task: identity.task, - }); - } - Ok(()) -} - fn decode_acknowledgement( value: serde_json::Value, request: &CancellationRequest, @@ -376,10 +244,6 @@ pub(super) enum CancelDeliveryResponse { Pending, /// The executor acknowledged durable receipt Delivered { executor: ExecutorCancelState }, - /// The resource authority returned a durable typed receipt - ResourceDelivered { - resource: ResourceCancellationReceipt, - }, } impl From<&CancellationDelivery> for CancelDeliveryResponse { @@ -389,9 +253,6 @@ impl From<&CancellationDelivery> for CancelDeliveryResponse { CancellationDelivery::Delivered { result } => Self::Delivered { executor: result.clone(), }, - CancellationDelivery::ResourceDelivered { result } => Self::ResourceDelivered { - resource: result.clone(), - }, } } } diff --git a/src/daemon/cluster.rs b/src/daemon/cluster.rs index f2aa61e..6155b29 100644 --- a/src/daemon/cluster.rs +++ b/src/daemon/cluster.rs @@ -1,7 +1,6 @@ //! Daemon-to-daemon identity and event routes under `/v1/cluster/*` use crate::daemon::actors::SupervisorMsg; -use crate::fleet::http::ClusterClient; use axum::extract::{Path, Query, State}; use axum::http::StatusCode; use axum::routing::{get, post}; @@ -9,15 +8,10 @@ use axum::{Json, Router}; use serde::{Deserialize, Serialize}; use std::collections::BTreeMap; -use crate::cancellation::{ - CancellationOwner, CancellationPlan, CancellationReceipt, CancellationRequest, - CancellationRequestIdentity, CancellationRoute, CancellationTarget, - ResourceCancellationRequestIdentity, ResourceCancellationTarget, -}; +use crate::cancellation::{CancellationReceipt, CancellationRequestIdentity, CancellationTarget}; use crate::daemon::AppState; use crate::daemon::actors::{StoreMsg, call}; use crate::daemon::api::views::{LogTail, TaskDetail}; -use crate::daemon::cancel_delivery::CancelResponse; use crate::domain::{API_VERSION, ProcessStatus, TaskEnv, TaskId, TaskIdentity}; use crate::error::AppError; use crate::events::{EventAcceptance, TaskEvent}; @@ -26,18 +20,10 @@ use crate::fleet::protocol::{ClusterProtocolVersion, ProtocolRange, SUPPORTED_PR use crate::fleet::runtime::FleetHandle; use crate::invocation::{StdinPolicy, invocation_from_normalized_for_identity}; use crate::machine::MachineId; -use crate::message::{MessageRequest, MessageResponse, MessageSource}; -use crate::resource::{ - ResourceQueueRequest, ResourceQueueResponse, ResourceRequest, ResourceRequestState, - SupervisorNoticeRequest, SupervisorNoticeResponse, -}; +use crate::message::{MessageRequest, MessageResponse}; use crate::spec::{self, NormalizedSpec}; use crate::store::IdentityError; -use crate::submission::{ - ExecutorIdentity, RejectionTombstone, RequestId, ResourceCancellationOutcome, - ResourceCancellationReceipt, ResourceQueueOutcome, ResourceQueueReceipt, ResourceRoutePhase, - ResourceRouteProof, SubmissionState, normalized_spec_sha256, -}; +use crate::submission::{ExecutorIdentity, RejectionTombstone, RequestId, SubmissionState}; use std::path::{Path as StdPath, PathBuf}; /// Strict destination and version for a cluster read @@ -115,79 +101,6 @@ pub struct CancelBody { pub receipt: CancellationReceipt, } -/// Destination-checked request to cancel one pre-activation resource request -#[derive(Debug, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct CancelResourceRequest { - /// Public API schema version; missing values decode as zero for a versioned usage error - #[serde(default)] - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Intended resource authority - pub destination_machine: MachineId, - /// Exact origin-owned cancellation identity and route target - pub request: ResourceCancellationRequestIdentity, -} - -/// Durable authority response to a resource cancellation request -#[derive(Debug, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct CancelResourceBody { - /// Public API schema version - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Machine that handled the request - pub destination_machine: MachineId, - /// Saved resource cancellation receipt - pub receipt: ResourceCancellationReceipt, -} - -/// Destination-checked request for the route origin to persist cancellation intent -#[derive(Debug, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct CancelOriginResourceRequest { - /// Public API schema version; missing values decode as zero for a versioned usage error - #[serde(default)] - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Intended origin daemon - pub destination_machine: MachineId, - /// Machine that invoked task cancellation - pub requester_machine: MachineId, - /// Resource-routed task whose origin owns the cancellation intent - pub task: TaskId, -} - -/// Result from the origin-owned resource cancellation-intent route -#[derive(Debug, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct CancelOriginResourceBody { - /// Public API schema version - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Machine that owns the saved origin route - pub destination_machine: MachineId, - /// Durable intent, or an already-cancelled route result - pub outcome: OriginResourceCancellationOutcome, -} - -/// Typed response from the origin-owned resource cancellation-intent route -#[derive(Debug, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum OriginResourceCancellationOutcome { - /// Origin persisted or reused this stable cancellation request - Intent { - /// Saved cancellation request and current delivery state - request: Box, - }, - /// The resource route already retained a cancellation result - AlreadyCancelled, -} - /// Retained executor identity, or an absent lookup #[derive(Debug, Serialize, Deserialize)] #[serde(deny_unknown_fields)] @@ -358,16 +271,6 @@ pub struct OriginBody { pub origin: Option, } -/// Optional safe resource-route proof retained by this origin -#[derive(Debug, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct OriginResourceRouteProofBody { - /// Public API version - pub api_version: u32, - /// Proof, if this origin retains a valid resource route for the task - pub proof: Option, -} - /// Optional local retained identity summary #[derive(Debug, Serialize, Deserialize)] #[serde(deny_unknown_fields)] @@ -396,37 +299,14 @@ pub fn routes(fleet: FleetHandle) -> Router { .route("/v1/cluster/tasks/{id}/detail", get(task_detail)) .route("/v1/cluster/tasks/{id}/log", get(task_log)) .route("/v1/cluster/origin/tasks/{id}", get(origin_summary)) - .route( - "/v1/cluster/origin/resource-routes/{task}", - get(origin_resource_route_proof), - ) - .route( - "/v1/cluster/origin/resource-routes/cancel", - post(cancel_origin_resource_request), - ) .route("/v1/cluster/identities/{id}", get(identity_summary)) .route("/v1/cluster/executions", post(submit_execution)) .route("/v1/cluster/executions/preview", post(preview_execution)) .route("/v1/cluster/executions/{id}", get(executor_identity)) .route("/v1/cluster/executions/abandon", post(abandon_execution)) .route("/v1/cluster/executions/cancel", post(cancel_execution)) - .route( - "/v1/cluster/resource-requests/cancel", - post(cancel_resource_request), - ) - .route( - "/v1/cluster/resource-requests", - post(accept_resource_request), - ) .route("/v1/cluster/events", post(receive_event)) .route("/v1/cluster/messages", post(receive_message)) - .route( - "/v1/cluster/resource-notices", - post(receive_supervisor_notice), - ) - .merge(super::resource_action::cluster_routes()) - .merge(super::resource_background::cluster_routes()) - .merge(super::resource_api::cluster_routes()) .merge(super::fleet_tasks::cluster_routes()) .merge(super::thread_titles::cluster_routes()) } @@ -442,11 +322,6 @@ async fn receive_message( check_api_version(body.api_version)?; check_protocol(body.protocol_version, body.source.machine())?; body.validate()?; - if matches!(body.source, MessageSource::ResourceNotice { .. }) { - return Err(AppError::MessageInvalid { - message: "resource notices must use the resource-notice endpoint".into(), - }); - } let protocol_version = body.protocol_version; let receipt = state.message_receiver.receive(&state, body).await?; Ok(Json(MessageResponse { @@ -457,613 +332,6 @@ async fn receive_message( })) } -async fn receive_supervisor_notice( - State(state): State, - Json(body): Json, -) -> Result, AppError> { - state - .machine - .identity - .check_destination(body.destination.machine)?; - check_api_version(body.api_version)?; - check_protocol(body.protocol_version, body.source_machine)?; - body.validate()?; - let protocol_version = body.protocol_version; - let receipt = state - .message_receiver - .receive_supervisor_notice(&state, body) - .await?; - Ok(Json(SupervisorNoticeResponse { - api_version: API_VERSION, - protocol_version, - destination_machine: state.machine.identity.machine, - receipt, - })) -} - -async fn accept_resource_request( - State(state): State, - Json(request): Json, -) -> Result, AppError> { - let authority_machine = state.machine.identity.machine; - state - .machine - .identity - .check_destination(request.destination_machine)?; - check_api_version(request.api_version)?; - check_protocol(request.protocol_version, request.origin_machine)?; - - let proof = fetch_route_proof(&state, request.origin_machine, request.task_id).await?; - let phase = validate_resource_route_proof(&request, authority_machine, proof)?; - if let Some(reason) = rejected_resource_route_reason(&phase) { - return Ok(resource_queue_response( - &request, - authority_machine, - ResourceQueueOutcome::Rejected { reason }, - )); - } - - if resource_route_requires_retained_request(&phase) { - let requests = match call(&state.store, |reply| StoreMsg::ResourceRequests { - authority_machine, - resource_id: request.resource_id, - reply, - }) - .await - { - Ok(requests) => requests, - Err(AppError::Usage { .. }) => { - return Err(resource_route_retention_conflict(&request)); - } - Err(error) => return Err(error), - }; - let stored = retained_resource_request(&request, requests)?; - reconcile_if_waiting(&state, &stored).await?; - - return resource_queue_response_from_stored(&request, authority_machine, stored); - } - - let stored = match call(&state.store, |reply| StoreMsg::AcceptResourceRequest { - authority_machine, - request_id: request.request_id, - task_id: request.task_id, - resource_id: request.resource_id, - origin_machine: request.origin_machine, - normalized_spec: Box::new(request.spec.as_normalized().clone()), - reply, - }) - .await - { - Ok(stored) => stored, - Err(AppError::SubmissionRejected { reason, .. }) => { - return Ok(resource_queue_response( - &request, - authority_machine, - ResourceQueueOutcome::Rejected { reason }, - )); - } - Err(error) => return Err(error), - }; - reconcile_if_waiting(&state, &stored).await?; - - resource_queue_response_from_stored(&request, authority_machine, stored) -} - -async fn reconcile_if_waiting(state: &AppState, request: &ResourceRequest) -> Result<(), AppError> { - if !matches!( - &request.state, - ResourceRequestState::Queued | ResourceRequestState::Assigned { .. } - ) { - return Ok(()); - } - - call(&state.supervisor, |reply| { - SupervisorMsg::ReconcileResource { - id: request.resource_id, - reply, - } - }) - .await -} - -async fn cancel_resource_request( - State(state): State, - Json(body): Json, -) -> Result, AppError> { - state - .machine - .identity - .check_destination(body.destination_machine)?; - check_api_version(body.api_version)?; - check_protocol(body.protocol_version, body.request.requester_machine)?; - if body.destination_machine != body.request.authority_machine - || body.request.requester_machine != body.request.origin_machine - { - return Err(AppError::ClusterTaskConflict { - task: body.request.task, - }); - } - - let receipt = process_resource_cancellation(&state, body.request).await?; - Ok(Json(CancelResourceBody { - api_version: API_VERSION, - protocol_version: body.protocol_version, - destination_machine: state.machine.identity.machine, - receipt, - })) -} - -/// Resolve one exact origin proof, apply the authority queue cancellation, and retain its receipt -pub(super) async fn process_resource_cancellation( - state: &AppState, - identity: ResourceCancellationRequestIdentity, -) -> Result { - let authority_machine = state.machine.identity.machine; - if identity.authority_machine != authority_machine - || identity.requester_machine != identity.origin_machine - { - return Err(AppError::ClusterTaskConflict { - task: identity.task, - }); - } - let receipt = if let Some(receipt) = call(&state.store, |reply| { - StoreMsg::ResourceCancellationReceipt { - identity: identity.clone(), - reply, - } - }) - .await? - { - receipt - } else { - let proof = fetch_route_proof(state, identity.origin_machine, identity.task).await?; - let proof = - validate_resource_cancellation_route_proof(&identity, authority_machine, proof)?; - call(&state.store, |reply| { - StoreMsg::CancelResourceRequestWithReceipt { - authority_machine, - identity: identity.clone(), - proof, - reply, - } - }) - .await? - }; - if matches!( - receipt.outcome, - ResourceCancellationOutcome::PreventedBeforeAcceptance - | ResourceCancellationOutcome::CancelledBeforeLaunch - ) { - call(&state.supervisor, |reply| { - SupervisorMsg::ReconcileResource { - id: identity.resource, - reply, - } - }) - .await?; - } - - Ok(receipt) -} - -/// Ask the route origin to create or reuse its durable resource cancellation intent -pub(super) async fn forward_resource_cancellation_intent( - state: &AppState, - target: &ResourceCancellationTarget, -) -> Result { - let fleet = state.fleet.handle().ok_or(AppError::MachineUnavailable { - machine: target.origin_machine, - message: "fleet is disabled".into(), - })?; - let destination = fleet.connect(target.origin_machine).await?; - let body = CancelOriginResourceRequest { - api_version: API_VERSION, - protocol_version: destination.protocol.0, - destination_machine: target.origin_machine, - requester_machine: state.machine.identity.machine, - task: target.task_id, - }; - let response = ClusterClient::default() - .post_json( - &destination.address, - "/v1/cluster/origin/resource-routes/cancel", - &body, - ) - .await - .map_err(|error| AppError::MachineUnavailable { - machine: target.origin_machine, - message: error.to_string(), - })?; - if response.status != StatusCode::OK { - return Err(AppError::MachineUnavailable { - machine: target.origin_machine, - message: format!("origin cancellation response status {}", response.status), - }); - } - let body: CancelOriginResourceBody = - serde_json::from_slice(&response.body).map_err(|error| AppError::MachineUnavailable { - machine: target.origin_machine, - message: format!("invalid origin cancellation response: {error}"), - })?; - if body.api_version != API_VERSION - || body.protocol_version != destination.protocol.0 - || body.destination_machine != target.origin_machine - { - return Err(AppError::ClusterTaskConflict { - task: target.task_id, - }); - } - match body.outcome { - OriginResourceCancellationOutcome::Intent { request } => { - // the requester's phase may be older than the origin's, so an intent - // that the origin saved after activation still matches - if !saved_resource_cancellation_matches(target, true, &request) { - return Err(AppError::ClusterTaskConflict { - task: target.task_id, - }); - } - Ok(CancelResponse::intent(&request)) - } - OriginResourceCancellationOutcome::AlreadyCancelled => Ok(CancelResponse::status( - target.task_id, - ProcessStatus::Cancelled, - )), - } -} - -async fn cancel_origin_resource_request( - State(state): State, - Json(body): Json, -) -> Result, AppError> { - let origin_machine = state.machine.identity.machine; - state - .machine - .identity - .check_destination(body.destination_machine)?; - check_api_version(body.api_version)?; - check_protocol(body.protocol_version, body.requester_machine)?; - let outcome = origin_resource_cancellation_intent(&state, body.task).await?; - Ok(Json(CancelOriginResourceBody { - api_version: API_VERSION, - protocol_version: body.protocol_version, - destination_machine: origin_machine, - outcome, - })) -} - -/// Save or reuse the origin-owned cancellation intent for one resource-routed task -pub(super) async fn origin_resource_cancellation_intent( - state: &AppState, - task: TaskId, -) -> Result { - let origin_machine = state.machine.identity.machine; - let _intent_guard = state.locks.cancellation_intents.lock(task).await; - let route = call(&state.store, |reply| StoreMsg::OriginRoute { - id: task, - reply, - }) - .await? - .ok_or(AppError::TaskNotFound { id: task })?; - let owner = - CancellationOwner::from_route(task, origin_machine, CancellationRoute::from(&route)) - .map_err(|refusal| refusal.into_error(task))?; - // this route only serves resource routes; an ordinary owner means the forwarder saw another route - if !matches!(owner, CancellationOwner::Resource(_)) { - return Err(AppError::ClusterTaskConflict { task }); - } - let saved = call(&state.store, |reply| StoreMsg::GetCancellationRequest { - task, - reply, - }) - .await?; - let outcome = if let Some(saved) = saved { - if !saved_origin_resource_cancellation_matches_route(&route, &saved) { - return Err(AppError::ClusterTaskConflict { task }); - } - OriginResourceCancellationOutcome::Intent { - request: Box::new(saved), - } - } else { - let plan = owner - .plan(origin_machine, task, uuid::Uuid::now_v7()) - .map_err(|refusal| refusal.into_error(task))?; - let request = match plan { - CancellationPlan::AlreadyCancelled => { - return Ok(OriginResourceCancellationOutcome::AlreadyCancelled); - } - CancellationPlan::Deliver(request) => *request, - // the origin is the requester here, so it never forwards to itself - CancellationPlan::ForwardToOrigin(_) => { - return Err(AppError::ClusterTaskConflict { task }); - } - }; - let (saved, _) = call(&state.store, |reply| StoreMsg::InsertCancellationRequest { - request, - reply, - }) - .await?; - OriginResourceCancellationOutcome::Intent { - request: Box::new(saved), - } - }; - Ok(outcome) -} - -fn saved_origin_resource_cancellation_matches_route( - route: &crate::submission::OriginRoute, - request: &CancellationRequest, -) -> bool { - let SubmissionState::Resource { resource, phase } = &route.submission else { - return false; - }; - let expected = ResourceCancellationTarget { - request_id: route.request, - task_id: route.task, - resource_id: *resource, - origin_machine: route.origin_machine, - authority_machine: route.execution_machine, - phase: phase.clone(), - }; - // only an activated route can own an execution-targeted intent - let execution_target_allowed = matches!(phase, ResourceRoutePhase::Activated); - saved_resource_cancellation_matches(&expected, execution_target_allowed, request) -} - -/// Check that a saved intent names exactly this resource route -/// -/// The target phase is not compared, since the origin advances it after the -/// intent is saved. An execution-targeted intent matches by request identity -/// only when `execution_target_allowed` -fn saved_resource_cancellation_matches( - expected: &ResourceCancellationTarget, - execution_target_allowed: bool, - request: &CancellationRequest, -) -> bool { - if request.task != expected.task_id - || request.origin_machine != expected.origin_machine - || request.requester_machine != expected.origin_machine - || request.execution_machine != expected.authority_machine - { - return false; - } - match &request.target { - CancellationTarget::Resource(saved) => { - saved.request_id == expected.request_id - && saved.task_id == expected.task_id - && saved.resource_id == expected.resource_id - && saved.origin_machine == expected.origin_machine - && saved.authority_machine == expected.authority_machine - } - CancellationTarget::Execution { request_id } => { - execution_target_allowed && *request_id == expected.request_id - } - } -} - -/// Read the origin's resource-route proof for one task -/// -/// The local store answers when this machine is the origin; otherwise the -/// origin machine is asked over the cluster route -async fn fetch_route_proof( - state: &AppState, - origin_machine: MachineId, - task: TaskId, -) -> Result, AppError> { - if origin_machine == state.machine.identity.machine { - let route = call(&state.store, |reply| StoreMsg::OriginRoute { - id: task, - reply, - }) - .await?; - return Ok(route - .as_ref() - .filter(|route| route.task == task) - .and_then(ResourceRouteProof::from_route)); - } - - let unavailable = |message: String| AppError::RemoteSubmissionUnavailable { message }; - let fleet = state - .fleet - .handle() - .ok_or_else(|| unavailable("origin resource-route proof is unavailable".into()))?; - let destination = fleet.connect(origin_machine).await.map_err(|error| { - unavailable(format!( - "origin resource-route proof is unavailable: {error}" - )) - })?; - let path = format!( - "/v1/cluster/origin/resource-routes/{task}?api_version={API_VERSION}&destination_machine={origin_machine}", - ); - let response = ClusterClient::default() - .get(&destination.address, &path) - .await - .map_err(|error| { - unavailable(format!( - "origin resource-route proof is unavailable: {error}" - )) - })?; - if response.status != StatusCode::OK { - return Err(unavailable(format!( - "origin resource-route proof is unavailable: HTTP {}", - response.status - ))); - } - let body: OriginResourceRouteProofBody = serde_json::from_slice(&response.body) - .map_err(|error| unavailable(format!("invalid origin resource-route proof: {error}")))?; - if body.api_version != API_VERSION { - return Err(unavailable( - "origin resource-route proof uses an unsupported API version".into(), - )); - } - Ok(body.proof) -} - -fn validate_resource_cancellation_route_proof( - identity: &ResourceCancellationRequestIdentity, - authority_machine: MachineId, - proof: Option, -) -> Result { - let Some(proof) = proof else { - return Err(AppError::RemoteSubmissionUnavailable { - message: "origin has no valid resource-route proof for this task".into(), - }); - }; - if proof.request != identity.request - || proof.task != identity.task - || proof.resource != identity.resource - || proof.origin_machine != identity.origin_machine - || proof.authority_machine != authority_machine - || !resource_cancellation_phase_is_forward(&identity.target_phase, &proof.phase) - { - return Err(AppError::SubmissionConflict { - request: identity.request, - task: identity.task, - message: "origin resource-route proof does not match this cancellation".into(), - }); - } - Ok(proof) -} - -fn resource_cancellation_phase_is_forward( - target: &ResourceRoutePhase, - current: &ResourceRoutePhase, -) -> bool { - match target { - ResourceRoutePhase::AcceptanceUnknown => true, - ResourceRoutePhase::Waiting => !matches!(current, ResourceRoutePhase::AcceptanceUnknown), - ResourceRoutePhase::CancelledBeforeLaunch => { - matches!(current, ResourceRoutePhase::CancelledBeforeLaunch) - } - ResourceRoutePhase::Activated | ResourceRoutePhase::Rejected { .. } => target == current, - } -} - -fn validate_resource_route_proof( - request: &ResourceQueueRequest, - authority_machine: MachineId, - proof: Option, -) -> Result { - let Some(proof) = proof else { - return Err(AppError::RemoteSubmissionUnavailable { - message: "origin has no valid resource-route proof for this task".into(), - }); - }; - let digest = normalized_spec_sha256(request.spec.as_normalized()).map_err(|error| { - AppError::Internal { - message: format!("serialize resource command specification: {error}"), - } - })?; - if proof.request != request.request_id - || proof.task != request.task_id - || proof.resource != request.resource_id - || proof.origin_machine != request.origin_machine - || proof.authority_machine != authority_machine - || proof.thread != request.spec.as_normalized().thread - || proof.normalized_spec_sha256 != digest - { - return Err(AppError::SubmissionConflict { - request: request.request_id, - task: request.task_id, - message: "origin resource-route proof does not match this request".into(), - }); - } - - Ok(proof.phase) -} - -fn rejected_resource_route_reason(phase: &ResourceRoutePhase) -> Option { - match phase { - ResourceRoutePhase::AcceptanceUnknown - | ResourceRoutePhase::Waiting - | ResourceRoutePhase::Activated => None, - ResourceRoutePhase::CancelledBeforeLaunch => Some("cancelled_before_launch".into()), - ResourceRoutePhase::Rejected { reason } => Some(reason.clone()), - } -} - -fn resource_route_requires_retained_request(phase: &ResourceRoutePhase) -> bool { - matches!( - phase, - ResourceRoutePhase::Waiting | ResourceRoutePhase::Activated - ) -} - -fn retained_resource_request( - request: &ResourceQueueRequest, - requests: Vec, -) -> Result { - requests - .into_iter() - .find(|stored| stored.request_id == request.request_id) - .ok_or_else(|| resource_route_retention_conflict(request)) -} - -fn resource_route_retention_conflict(request: &ResourceQueueRequest) -> AppError { - AppError::SubmissionConflict { - request: request.request_id, - task: request.task_id, - message: "origin route says the resource request was already accepted, but the authority has no retained request".into(), - } -} - -fn resource_queue_response_from_stored( - request: &ResourceQueueRequest, - authority_machine: MachineId, - stored: ResourceRequest, -) -> Result, AppError> { - let stored_digest = normalized_spec_sha256(stored.spec().as_normalized()).map_err(|error| { - AppError::Internal { - message: format!("serialize stored resource command specification: {error}"), - } - })?; - let request_digest = normalized_spec_sha256(request.spec.as_normalized()).map_err(|error| { - AppError::Internal { - message: format!("serialize resource command specification: {error}"), - } - })?; - if stored.request_id != request.request_id - || stored.task_id != request.task_id - || stored.resource_id != request.resource_id - || stored.origin_machine != request.origin_machine - || stored_digest != request_digest - { - return Err(AppError::SubmissionConflict { - request: request.request_id, - task: request.task_id, - message: "stored resource request does not match this retry".into(), - }); - } - - let outcome = match stored.state { - ResourceRequestState::Queued - | ResourceRequestState::Assigned { .. } - | ResourceRequestState::Finished { .. } => ResourceQueueOutcome::Waiting, - ResourceRequestState::CancelledBeforeLaunch => ResourceQueueOutcome::Rejected { - reason: "cancelled_before_launch".into(), - }, - ResourceRequestState::Rejected { reason } => ResourceQueueOutcome::Rejected { reason }, - }; - Ok(resource_queue_response(request, authority_machine, outcome)) -} - -fn resource_queue_response( - request: &ResourceQueueRequest, - authority_machine: MachineId, - outcome: ResourceQueueOutcome, -) -> Json { - Json(ResourceQueueResponse::new( - request.protocol_version, - ResourceQueueReceipt { - request: request.request_id, - task: request.task_id, - origin_machine: request.origin_machine, - authority_machine, - resource: request.resource_id, - outcome, - }, - )) -} - async fn task_detail( State(state): State, Path(id): Path, @@ -1131,28 +399,6 @@ async fn origin_summary( })) } -async fn origin_resource_route_proof( - State(state): State, - Path(task): Path, - Query(query): Query, -) -> Result, AppError> { - check(&state, query)?; - let route = call(&state.store, |reply| StoreMsg::OriginRoute { - id: task, - reply, - }) - .await?; - let proof = route - .as_ref() - .filter(|route| route.task == task) - .and_then(ResourceRouteProof::from_route); - - Ok(Json(OriginResourceRouteProofBody { - api_version: API_VERSION, - proof, - })) -} - async fn identity_summary( State(state): State, Path(id): Path, @@ -1499,7 +745,6 @@ async fn cancel_execution( .check_destination(body.request.execution_machine)?; check_api_version(body.api_version)?; check_protocol(body.protocol_version, body.request.requester_machine)?; - verify_cancellation_target(&body.request, &body.target)?; let mut receipt = call(&state.store, |reply| StoreMsg::ReceiveCancellation { request: body.request, reply, @@ -1523,22 +768,6 @@ async fn cancel_execution( })) } -fn verify_cancellation_target( - request: &CancellationRequestIdentity, - target: &CancellationTarget, -) -> Result<(), AppError> { - let CancellationTarget::Resource(target) = target else { - return Ok(()); - }; - if target.task_id != request.task - || target.origin_machine != request.origin_machine - || target.authority_machine != request.execution_machine - { - return Err(AppError::ClusterTaskConflict { task: request.task }); - } - Err(AppError::ResourceCancellationUnavailable { task: request.task }) -} - async fn machine(fleet: FleetHandle) -> Json { Json(fleet.advertisement().clone()) } @@ -1606,472 +835,3 @@ async fn task( }), )) } - -#[cfg(test)] -mod resource_queue_tests { - use crate::daemon::actors::StoreMsg; - use crate::resource::{ - IdleProofGap, ResourceQueueAttentionReason, ResourceQueueReconcileOutcome, - }; - use std::path::PathBuf; - - use super::{ - accept_resource_request, rejected_resource_route_reason, - resource_queue_response_from_stored, resource_route_requires_retained_request, - retained_resource_request, validate_resource_route_proof, - }; - use crate::daemon::actors::{SupervisorActor, SupervisorArgs, SupervisorMsg, call}; - use crate::domain::{API_VERSION, TaskEnv, TaskId, ThreadId}; - use crate::error::AppError; - use crate::files::StreamSlots; - use crate::fleet::FleetState; - use crate::fleet::directory::LocalMachine; - use crate::fleet::protocol::{CLUSTER_PROTOCOL_VERSION, SUPPORTED_PROTOCOLS}; - use crate::home::Home; - use crate::machine::{LocalIdentity, MachineId, MachineName}; - use crate::resource::{ - AssignmentRevision, CommandSpec, LoanId, Resource, ResourceId, ResourceQueueRequest, - ResourceRequest, ResourceRequestState, ResourceRevision, SupervisorAddress, - }; - use crate::spec; - use crate::submission::{ - CallbackContext, CallbackExecutable, NewResourceRoute, NormalizedSpecSha256, RequestId, - ResourceQueueOutcome, ResourceRoutePhase, ResourceRouteProof, normalized_spec_sha256, - }; - use axum::Json; - use axum::extract::State; - use ractor::Actor; - use serde::Deserialize; - use serde_json::json; - use tempfile::tempdir; - use uuid::Uuid; - - fn request() -> ResourceQueueRequest { - let thread = ThreadId(Uuid::now_v7()); - let normalized = spec::parse_normalized_value(&json!({ - "api_version": API_VERSION, - "thread": thread, - "name": "queued command", - "cwd": "/tmp", - "timeout": "30m", - "workload": { - "type": "task", - "command": ["echo", "hello"] - } - })) - .unwrap(); - let command = CommandSpec::try_from(normalized).unwrap(); - ResourceQueueRequest::new( - CLUSTER_PROTOCOL_VERSION.0, - MachineId::new(), - MachineId::new(), - RequestId::new(), - TaskId(Uuid::now_v7()), - ResourceId::new(), - command, - ) - } - - fn proof( - request: &ResourceQueueRequest, - authority_machine: MachineId, - phase: ResourceRoutePhase, - ) -> ResourceRouteProof { - ResourceRouteProof { - request: request.request_id, - task: request.task_id, - origin_machine: request.origin_machine, - authority_machine, - resource: request.resource_id, - thread: request.spec.as_normalized().thread, - normalized_spec_sha256: normalized_spec_sha256(request.spec.as_normalized()).unwrap(), - phase, - } - } - - #[test] - fn missing_or_mismatched_origin_proof_cannot_reach_acceptance() { - let request = request(); - let authority_machine = request.destination_machine; - - assert!(matches!( - validate_resource_route_proof(&request, authority_machine, None), - Err(AppError::RemoteSubmissionUnavailable { .. }) - )); - - let mut mismatched = proof( - &request, - authority_machine, - ResourceRoutePhase::AcceptanceUnknown, - ); - mismatched.request = RequestId::new(); - assert!(matches!( - validate_resource_route_proof(&request, authority_machine, Some(mismatched)), - Err(AppError::SubmissionConflict { .. }) - )); - } - - #[test] - fn route_proof_must_bind_authority_thread_and_normalized_spec() { - let request = request(); - let authority_machine = request.destination_machine; - for phase in [ - ResourceRoutePhase::AcceptanceUnknown, - ResourceRoutePhase::Waiting, - ResourceRoutePhase::Activated, - ] { - assert_eq!( - validate_resource_route_proof( - &request, - authority_machine, - Some(proof(&request, authority_machine, phase.clone())), - ) - .unwrap(), - phase - ); - } - - let mut mismatched_proofs = Vec::new(); - let mut wrong_task = proof( - &request, - authority_machine, - ResourceRoutePhase::AcceptanceUnknown, - ); - wrong_task.task = TaskId(Uuid::now_v7()); - mismatched_proofs.push(wrong_task); - - let mut wrong_resource = proof( - &request, - authority_machine, - ResourceRoutePhase::AcceptanceUnknown, - ); - wrong_resource.resource = ResourceId::new(); - mismatched_proofs.push(wrong_resource); - - let mut wrong_origin = proof( - &request, - authority_machine, - ResourceRoutePhase::AcceptanceUnknown, - ); - wrong_origin.origin_machine = MachineId::new(); - mismatched_proofs.push(wrong_origin); - - let mut wrong_authority = proof( - &request, - authority_machine, - ResourceRoutePhase::AcceptanceUnknown, - ); - wrong_authority.authority_machine = MachineId::new(); - mismatched_proofs.push(wrong_authority); - - let mut wrong_thread = proof( - &request, - authority_machine, - ResourceRoutePhase::AcceptanceUnknown, - ); - wrong_thread.thread = ThreadId(Uuid::now_v7()); - mismatched_proofs.push(wrong_thread); - - let mut wrong_spec = proof( - &request, - authority_machine, - ResourceRoutePhase::AcceptanceUnknown, - ); - wrong_spec.normalized_spec_sha256 = - NormalizedSpecSha256::deserialize(json!("00".repeat(32))).unwrap(); - mismatched_proofs.push(wrong_spec); - - for mismatched in mismatched_proofs { - assert!(matches!( - validate_resource_route_proof(&request, authority_machine, Some(mismatched)), - Err(AppError::SubmissionConflict { .. }) - )); - } - } - - #[test] - fn terminal_origin_phases_return_typed_rejections() { - assert_eq!( - rejected_resource_route_reason(&ResourceRoutePhase::CancelledBeforeLaunch), - Some("cancelled_before_launch".into()) - ); - assert_eq!( - rejected_resource_route_reason(&ResourceRoutePhase::Rejected { - reason: "origin_cancelled".into(), - }), - Some("origin_cancelled".into()) - ); - assert_eq!( - rejected_resource_route_reason(&ResourceRoutePhase::Activated), - None - ); - } - - #[test] - fn accepted_origin_phases_require_a_retained_authority_request() { - assert!(!resource_route_requires_retained_request( - &ResourceRoutePhase::AcceptanceUnknown - )); - assert!(resource_route_requires_retained_request( - &ResourceRoutePhase::Waiting - )); - assert!(resource_route_requires_retained_request( - &ResourceRoutePhase::Activated - )); - } - - #[test] - fn accepted_origin_retry_without_retained_authority_row_conflicts() { - let request = request(); - assert!(matches!( - retained_resource_request(&request, Vec::new()), - Err(AppError::SubmissionConflict { .. }) - )); - } - - #[test] - fn accepted_retry_receipt_stays_waiting_after_queue_state_changes() { - let request = request(); - let authority_machine = request.destination_machine; - let waiting = accepted_request(&request, ResourceRequestState::Queued); - let assigned = accepted_request( - &request, - ResourceRequestState::Assigned { - loan_id: LoanId::new(), - }, - ); - let finished = accepted_request( - &request, - ResourceRequestState::Finished { - outcome: crate::domain::ExitReason::Exit { code: 0 }, - }, - ); - - let first = resource_queue_response_from_stored(&request, authority_machine, waiting) - .unwrap() - .0; - let retry_assigned = - resource_queue_response_from_stored(&request, authority_machine, assigned) - .unwrap() - .0; - let retry_finished = - resource_queue_response_from_stored(&request, authority_machine, finished) - .unwrap() - .0; - - assert_eq!(first, retry_assigned); - assert_eq!(first, retry_finished); - assert_eq!(first.receipt.outcome, ResourceQueueOutcome::Waiting); - - let cancelled = accepted_request(&request, ResourceRequestState::CancelledBeforeLaunch); - let rejected = accepted_request( - &request, - ResourceRequestState::Rejected { - reason: "cancelled_before_acceptance".into(), - }, - ); - assert_eq!( - resource_queue_response_from_stored(&request, authority_machine, cancelled) - .unwrap() - .0 - .receipt - .outcome, - ResourceQueueOutcome::Rejected { - reason: "cancelled_before_launch".into(), - } - ); - assert_eq!( - resource_queue_response_from_stored(&request, authority_machine, rejected) - .unwrap() - .0 - .receipt - .outcome, - ResourceQueueOutcome::Rejected { - reason: "cancelled_before_acceptance".into(), - } - ); - } - - fn accepted_request( - request: &ResourceQueueRequest, - state: ResourceRequestState, - ) -> ResourceRequest { - let mut stored = ResourceRequest::new( - request.request_id, - request.task_id, - request.resource_id, - crate::resource::AcceptanceSequence::new(1), - request.origin_machine, - request.spec.as_normalized().clone(), - ) - .unwrap(); - stored.state = state; - stored - } - - #[tokio::test] - async fn accepted_resource_request_and_exact_retry_wake_the_authority_actor() { - let _guard = crate::daemon::actors::supervisor::SUPERVISOR_TEST_LOCK - .lock() - .await; - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().join("state"))).unwrap(); - home.ensure().unwrap(); - let (supervisor, supervisor_handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let machine = LocalMachine { - identity: LocalIdentity::start(&home).unwrap(), - name: MachineName::fallback(), - protocol: SUPPORTED_PROTOCOLS, - }; - let authority = machine.identity.machine; - let normalized = spec::parse_normalized_value(&json!({ - "api_version": API_VERSION, - "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "authority wake test", - "cwd": "/tmp", - "timeout": "30m", - "workload": { "type": "task", "command": ["echo", "hello"] } - })) - .unwrap(); - let resource_id = ResourceId::new(); - let resource = Resource::new( - resource_id, - "gpu-test".into(), - authority, - SupervisorAddress { - machine: authority, - thread: normalized.thread, - }, - AssignmentRevision::new(0), - ResourceRevision::new(0), - None, - ); - call(&supervisor, |reply| SupervisorMsg::RegisterResource { - resource: Box::new(resource), - reply, - }) - .await - .unwrap(); - - let request_id = RequestId::new(); - let task_id = TaskId::new(); - let route = crate::submission::OriginRoute::new_resource_waiting(NewResourceRoute { - request: request_id, - task: task_id, - origin_machine: authority, - authority_machine: authority, - thread: normalized.thread, - callback: CallbackContext { - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - cwd: PathBuf::from("/tmp"), - codex: CallbackExecutable::available(PathBuf::from("/bin/codex")), - }, - spec: normalized.clone(), - resource: resource_id, - }) - .unwrap(); - call(&store, |reply| StoreMsg::InsertOriginRoute { - route: Box::new(route), - reply, - }) - .await - .unwrap(); - - let request = ResourceQueueRequest::new( - CLUSTER_PROTOCOL_VERSION.0, - authority, - authority, - request_id, - task_id, - resource_id, - CommandSpec::try_from(normalized).unwrap(), - ); - let state = crate::daemon::AppState { - home: home.clone(), - store: store.clone(), - supervisor: supervisor.clone(), - web: None, - content: None, - stream_slots: StreamSlots::new(), - machine, - fleet: FleetState::Disabled, - message_receiver: crate::daemon::message_receiver::MessageReceiver::default(), - locks: crate::daemon::DaemonLocks::default(), - thread_titles: None, - }; - - let first = accept_resource_request(State(state.clone()), Json(request.clone())) - .await - .unwrap() - .0; - assert_eq!(first.receipt.outcome, ResourceQueueOutcome::Waiting); - let first_inspection = call(&supervisor, |reply| SupervisorMsg::InspectResource { - id: resource_id, - reply, - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - first_inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::AttentionRequired { - request, - reason: ResourceQueueAttentionReason::IdleNotProven { - gap: IdleProofGap::NoIdleEvidence, - }, - }) if request.request_id == request_id - )); - - call(&store, |reply| StoreMsg::ResolveResourceRoute { - receipt: first.receipt.clone(), - reply, - }) - .await - .unwrap(); - let retry = accept_resource_request(State(state), Json(request)) - .await - .unwrap() - .0; - assert_eq!(retry, first); - let retry_inspection = call(&supervisor, |reply| SupervisorMsg::InspectResource { - id: resource_id, - reply, - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - retry_inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::AttentionRequired { - request, - reason: ResourceQueueAttentionReason::IdleNotProven { - gap: IdleProofGap::NoIdleEvidence, - }, - }) if request.request_id == request_id - )); - let requests = call(&store, |reply| StoreMsg::ResourceRequests { - authority_machine: authority, - resource_id, - reply, - }) - .await - .unwrap(); - assert_eq!(requests.len(), 1); - assert!(matches!(requests[0].state, ResourceRequestState::Queued)); - - supervisor.stop(None); - let _ = supervisor_handle.await; - } -} diff --git a/src/daemon/event_sender.rs b/src/daemon/event_sender.rs index adf314d..f457420 100644 --- a/src/daemon/event_sender.rs +++ b/src/daemon/event_sender.rs @@ -227,10 +227,6 @@ impl Sender { let destination = if origin == self.local { None } else { - // the authority observes its own task row, so a remote callback outage - // cannot hide a confirmed start from the resource owner - self.supervisor - .cast(SupervisorMsg::RemoteOriginEvent { id: task })?; let FleetState::Enabled(fleet) = &self.fleet else { return Ok(DeliveryResult::Retry( "fleet is disabled for remote origin".into(), diff --git a/src/daemon/inspection.rs b/src/daemon/inspection.rs index 3cbba20..40d1fb7 100644 --- a/src/daemon/inspection.rs +++ b/src/daemon/inspection.rs @@ -25,10 +25,7 @@ use crate::fleet::probe::{VerifiedDestination, check_probed, probe}; use crate::fleet::protocol::SUPPORTED_PROTOCOLS; use crate::fleet::runtime::FleetHandle; use crate::machine::MachineId; -use crate::submission::{ - ExecutorIdentity, HeldPhase, ResourceActionRoutePhase, ResourceBackgroundRoutePhase, - SubmissionState, -}; +use crate::submission::{ExecutorIdentity, HeldPhase, SubmissionState}; /// Strict read query for a peer log, with an optional line limit #[derive(Debug, Deserialize)] @@ -657,7 +654,7 @@ pub(super) async fn cancellation_owner( return absent(id, &records); }; - let (origin, execution) = target.machines(); + let (origin, execution) = (target.origin_machine, target.execution_machine); let unchecked: Vec<_> = records .unchecked .iter() @@ -909,18 +906,7 @@ fn cached_route(records: &Records, id: TaskId) -> Result { "failed_events": route.failed_events, "last_update": route.last_updated_at, }); - let availability = if matches!( - route.submission, - SubmissionState::Rejected { .. } - | SubmissionState::ResourceAction { - phase: ResourceActionRoutePhase::Rejected { .. }, - .. - } - | SubmissionState::ResourceBackground { - phase: ResourceBackgroundRoutePhase::Rejected { .. }, - .. - } - ) { + let availability = if matches!(route.submission, SubmissionState::Rejected { .. }) { "rejected" } else if records.unchecked.contains(&route.execution_machine) { "executor_unavailable" @@ -944,17 +930,6 @@ fn cached_process_status(route: &OriginSummary) -> Option { | SubmissionState::Rejected { .. } | SubmissionState::Held { .. } => None, SubmissionState::Accepted => route.last_execution_state, - SubmissionState::Resource { .. } => route.last_execution_state, - SubmissionState::ResourceAction { - phase: ResourceActionRoutePhase::Accepted, - .. - } => route.last_execution_state, - SubmissionState::ResourceAction { .. } => None, - SubmissionState::ResourceBackground { - phase: ResourceBackgroundRoutePhase::Accepted, - .. - } => route.last_execution_state, - SubmissionState::ResourceBackground { .. } => None, } } @@ -997,8 +972,7 @@ mod tests { use crate::domain::TaskId; use crate::error::AppError; use crate::machine::MachineId; - use crate::resource::ResourceId; - use crate::submission::{RequestId, ResourceRoutePhase, SubmissionState}; + use crate::submission::{RequestId, SubmissionState}; fn origin_summary( task: TaskId, @@ -1023,38 +997,32 @@ mod tests { } #[test] - fn resource_cancellation_target_keeps_its_full_identity() { + fn execution_cancellation_target_keeps_its_full_identity() { let task = TaskId::new(); let request_id = RequestId::new(); - let resource_id = ResourceId::new(); let origin_machine = MachineId::new(); - let authority_machine = MachineId::new(); + let execution_machine = MachineId::new(); let route = origin_summary( task, request_id, origin_machine, - authority_machine, - SubmissionState::Resource { - resource: resource_id, - phase: ResourceRoutePhase::Waiting, - }, + execution_machine, + SubmissionState::Accepted, ); - let CancellationOwner::Resource(target) = - cancellation_target(task, origin_machine, &route).unwrap() - else { - panic!("resource routes must keep their authority-owned target type"); - }; - assert_eq!(target.request_id, request_id); - assert_eq!(target.task_id, task); - assert_eq!(target.resource_id, resource_id); - assert_eq!(target.origin_machine, origin_machine); - assert_eq!(target.authority_machine, authority_machine); - assert_eq!(target.phase, ResourceRoutePhase::Waiting); + assert_eq!( + cancellation_target(task, origin_machine, &route).unwrap(), + CancellationOwner { + request_id, + task, + origin_machine, + execution_machine, + } + ); } #[test] - fn resource_cancellation_target_rejects_wrong_task_or_origin_identity() { + fn cancellation_target_rejects_wrong_task_or_origin_identity() { let task = TaskId::new(); let origin_machine = MachineId::new(); let route = origin_summary( @@ -1062,10 +1030,7 @@ mod tests { RequestId::new(), origin_machine, MachineId::new(), - SubmissionState::Resource { - resource: ResourceId::new(), - phase: ResourceRoutePhase::AcceptanceUnknown, - }, + SubmissionState::AcceptanceUnknown, ); assert!(matches!( @@ -1099,60 +1064,11 @@ mod tests { } #[test] - fn unresolved_background_launch_cannot_become_a_cancellation_target() { - use crate::resource::background_launch::{ - BackgroundLaunchBinding, BackgroundSupervisorAssignment, ResourceBackgroundRejection, - }; - use crate::resource::{AssignmentRevision, ResourceRevision, SupervisorAddress}; - use crate::submission::ResourceBackgroundRoutePhase; - - let task = TaskId::new(); - let origin_machine = MachineId::new(); - let authority_machine = MachineId::new(); - let binding = BackgroundLaunchBinding { - assignment: BackgroundSupervisorAssignment { - authority_machine, - resource_id: ResourceId::new(), - supervisor: SupervisorAddress { - machine: origin_machine, - thread: ThreadId(Uuid::now_v7()), - }, - assignment_revision: AssignmentRevision::new(1), - }, - expected_state_revision: ResourceRevision::new(1), - }; - // a cancellation must never fence the fixed launch identity before acceptance - for phase in [ - ResourceBackgroundRoutePhase::AcceptanceUnknown, - ResourceBackgroundRoutePhase::Rejected { - reason: ResourceBackgroundRejection::ResourceNotFound, - }, - ] { - let route = origin_summary( - task, - RequestId::new(), - origin_machine, - authority_machine, - SubmissionState::ResourceBackground { binding, phase }, - ); - - assert!(matches!( - cancellation_target(task, origin_machine, &route), - Err(AppError::ClusterTaskConflict { task: found }) if found == task - )); - } - } - - #[test] - fn duplicate_routes_with_different_resource_request_ids_conflict() { + fn duplicate_routes_with_different_request_ids_conflict() { let task = TaskId::new(); let origin_machine = MachineId::new(); let authority_machine = MachineId::new(); - let resource_id = ResourceId::new(); - let submission = SubmissionState::Resource { - resource: resource_id, - phase: ResourceRoutePhase::Waiting, - }; + let submission = SubmissionState::Accepted; let mut records = Records::default(); records.routes.insert( MachineId::new(), diff --git a/src/daemon/local_submit.rs b/src/daemon/local_submit.rs index 3cce234..eb71d0b 100644 --- a/src/daemon/local_submit.rs +++ b/src/daemon/local_submit.rs @@ -210,19 +210,19 @@ async fn resume( reason: reason.clone(), }); } - _ => return Err(conflict(&route, "request UUID belongs to a resource route")), + // only a remote submission waits on an unknown acceptance + SubmissionState::AcceptanceUnknown => { + return Err(conflict( + &route, + "request UUID belongs to a remote submission", + )); + } } let id = route.task; - match call(&state.supervisor, |reply| SupervisorMsg::ResumeLocal { + let status = call(&state.supervisor, |reply| SupervisorMsg::ResumeLocal { id, reply, }) - .await? - { - Some(status) => Ok((id, status.into())), - None => Err(conflict( - &route, - "request UUID belongs to a resource launch", - )), - } + .await?; + Ok((id, status.into())) } diff --git a/src/daemon/message_receiver.rs b/src/daemon/message_receiver.rs index 983d879..0caf7a1 100644 --- a/src/daemon/message_receiver.rs +++ b/src/daemon/message_receiver.rs @@ -1,4 +1,4 @@ -//! Fleet endpoint work for durable direct-message and supervisor-notice delivery +//! Fleet endpoint work for durable direct-message delivery use crate::domain::API_VERSION; use std::fs::{self, File}; @@ -19,15 +19,12 @@ use crate::domain::{AgentKind, TaskEnv, ThreadId}; use crate::error::AppError; use crate::invocation::resolve_agent_binary; use crate::message::{ - MESSAGE_BODY_MAX_BYTES, MessageAttempt, MessageId, MessageReceipt, MessageRequest, - MessageSource, Recipient, + MessageAttempt, MessageId, MessageReceipt, MessageRequest, MessageSource, Recipient, }; -use crate::resource::{SupervisorNoticeReceipt, SupervisorNoticeRequest}; use crate::submission::{CallbackContext, CallbackExecutable}; const SESSION_META_MAX_BYTES: usize = 64 * 1024; const MESSAGE_PREFIX: &str = "HOMEBASED_MESSAGE "; -const RESOURCE_NOTICE_PREFIX: &str = "HOMEBASED_RESOURCE_NOTICE "; /// Per-message delivery permits prevent concurrent retries from queueing one UUID twice #[derive(Clone, Default)] @@ -94,108 +91,6 @@ impl MessageReceiver { self.deliver(state, attempt).await } - /// Resolve, persist, and deliver one exact supervisor-notice attempt - pub(crate) async fn receive_supervisor_notice( - &self, - state: &AppState, - request: SupervisorNoticeRequest, - ) -> Result { - request.validate()?; - state - .machine - .identity - .check_destination(request.destination.machine)?; - let backing_request = notice_backing_request(&request)?; - let message_id = backing_request.message_id; - let (_attempt_permit, attempt_was_contended) = self.attempt_permit(message_id).await; - let saved = call(&state.store, |reply| StoreMsg::MessageDelivery { - id: message_id, - reply, - }) - .await?; - let attempt = match saved.attempt { - Some(mut attempt) => { - if attempt.request.identity() != backing_request.identity() { - return Err(AppError::MessageConflict { id: message_id }); - } - if attempt.destination_thread != request.destination.thread { - return Err(AppError::Internal { - message: "saved supervisor notice attempt has a different thread".into(), - }); - } - if let Some(receipt) = saved.receipt { - return supervisor_notice_receipt( - &request, - receipt.for_protocol_version(request.protocol_version), - ); - } - if attempt_was_contended { - return Err(attempt_wait_failed(message_id)); - } - attempt.request = backing_request.clone(); - call(&state.store, |reply| StoreMsg::BindMessageAttempt { - attempt, - reply, - }) - .await? - } - None => { - if saved.receipt.is_some() { - return Err(AppError::Internal { - message: "message receipt has no saved attempt".into(), - }); - } - let (destination_thread, destination_cwd) = - resolve_destination(&Recipient::Thread { - thread: request.destination.thread, - }) - .await?; - if destination_thread != request.destination.thread { - return Err(AppError::Internal { - message: "supervisor notice resolved to a different thread".into(), - }); - } - call(&state.store, |reply| StoreMsg::BindMessageAttempt { - attempt: MessageAttempt { - request: backing_request, - destination_thread, - destination_cwd, - }, - reply, - }) - .await? - } - }; - - let payload = serde_json::to_string(&request) - .map_err(|error| notice_delivery_failed(message_id, error))?; - let line = format!("{RESOURCE_NOTICE_PREFIX}{payload}"); - send_queue_line( - state, - message_id, - attempt.destination_thread, - attempt.destination_cwd.clone(), - line, - ) - .await - .map_err(|error| notice_delivery_failed(message_id, error))?; - - let receipt = MessageReceipt { - api_version: API_VERSION, - protocol_version: request.protocol_version, - message_id, - destination_thread: attempt.destination_thread, - destination_cwd: attempt.destination_cwd, - delivered_at: chrono::Utc::now(), - }; - let receipt = call(&state.store, |reply| StoreMsg::CommitMessageReceipt { - receipt, - reply, - }) - .await?; - supervisor_notice_receipt(&request, receipt) - } - async fn attempt_permit(&self, id: MessageId) -> (KeyedGuard, bool) { match self.attempts.try_lock(id) { Some(permit) => (permit, false), @@ -240,59 +135,6 @@ impl MessageReceiver { } } -fn notice_backing_request(request: &SupervisorNoticeRequest) -> Result { - let message_id = MessageId::from_uuid(request.attempt_id.as_uuid())?; - let body = serde_json::to_string(request)?; - if body.len() > MESSAGE_BODY_MAX_BYTES { - return Err(AppError::MessageInvalid { - message: format!( - "supervisor notice request exceeds {MESSAGE_BODY_MAX_BYTES} UTF-8 bytes" - ), - }); - } - // keep the exact notice content in the existing durable UUID-keyed attempt ledger - Ok(MessageRequest { - api_version: request.api_version, - protocol_version: request.protocol_version, - message_id, - destination_machine: request.destination.machine, - source: MessageSource::ResourceNotice { - machine: request.source_machine, - notice_id: request.notice_id, - }, - recipient: Recipient::Thread { - thread: request.destination.thread, - }, - body, - reply_to: None, - conversation_id: request.notice_id.as_uuid(), - }) -} - -fn supervisor_notice_receipt( - request: &SupervisorNoticeRequest, - receipt: MessageReceipt, -) -> Result { - if receipt.message_id.as_uuid() != request.attempt_id.as_uuid() - || receipt.api_version != API_VERSION - || receipt.protocol_version != request.protocol_version - || receipt.destination_thread != request.destination.thread - { - return Err(AppError::Internal { - message: "saved supervisor notice receipt does not match its attempt".into(), - }); - } - Ok(SupervisorNoticeReceipt { - api_version: receipt.api_version, - protocol_version: receipt.protocol_version, - notice_id: request.notice_id, - attempt_id: request.attempt_id, - destination_machine: request.destination.machine, - destination_thread: receipt.destination_thread, - delivered_at: receipt.delivered_at, - }) -} - async fn send_queue_line( state: &AppState, message_id: MessageId, @@ -379,16 +221,6 @@ fn attempt_wait_failed(message_id: MessageId) -> AppError { } } -fn notice_delivery_failed(message_id: MessageId, error: impl std::fmt::Display) -> AppError { - tracing::warn!(attempt_id = %message_id, "supervisor notice delivery attempt failed: {error}"); - AppError::MessageDeliveryFailed { - id: message_id, - message: format!( - "delivery attempt failed: {error}; retry the same delivery attempt UUID explicitly" - ), - } -} - async fn resolve_destination(recipient: &Recipient) -> Result<(ThreadId, PathBuf), AppError> { let selector = recipient.clone(); tokio::task::spawn_blocking(move || resolve_destination_sync(&selector)) @@ -601,11 +433,10 @@ mod tests { use crate::domain::API_VERSION; use std::time::Duration; - use super::{MESSAGE_PREFIX, MessageReceiver, QueuedMessage, notice_backing_request}; + use super::{MESSAGE_PREFIX, MessageReceiver, QueuedMessage}; use crate::domain::{TaskId, ThreadId}; use crate::machine::MachineId; use crate::message::{MessageAttempt, MessageId, MessageRequest, MessageSource, Recipient}; - use crate::resource::SupervisorNoticeRequest; use serde_json::json; use std::path::PathBuf; use uuid::Uuid; @@ -680,42 +511,4 @@ mod tests { assert_eq!(body["source"]["kind"], "task"); assert!(body["source"].get("task").is_some()); } - - #[test] - fn notice_backing_request_uses_notice_identity_instead_of_a_fake_thread() { - let request = SupervisorNoticeRequest { - api_version: API_VERSION, - protocol_version: 1, - source_machine: MachineId::new(), - destination: crate::resource::SupervisorAddress { - machine: MachineId::new(), - thread: ThreadId(Uuid::now_v7()), - }, - notice_id: crate::resource::NoticeId::new(), - loan_id: crate::resource::LoanId::new(), - action_id: crate::resource::ActionId::new(), - state_revision: crate::resource::ResourceRevision::new(4), - assignment_revision: crate::resource::AssignmentRevision::new(2), - attempt_id: crate::resource::DeliveryAttemptId::new(), - payload: crate::resource::SupervisorNoticePayload::AttentionRequired { - reason: "needs supervisor review".into(), - }, - }; - - let backing = notice_backing_request(&request).unwrap(); - assert_eq!( - backing.source, - MessageSource::ResourceNotice { - machine: request.source_machine, - notice_id: request.notice_id, - } - ); - let source = serde_json::to_value(&backing.source).unwrap(); - assert_eq!(source["kind"], "resource_notice"); - assert_eq!(source["machine"], serde_json::json!(request.source_machine)); - assert_eq!(source["notice_id"], serde_json::json!(request.notice_id)); - assert!(source.get("thread").is_none()); - assert_eq!(backing.destination_machine, request.destination.machine); - assert_eq!(backing.conversation_id, request.notice_id.as_uuid()); - } } diff --git a/src/daemon/origin_submit.rs b/src/daemon/origin_submit.rs index 445861d..d51f93f 100644 --- a/src/daemon/origin_submit.rs +++ b/src/daemon/origin_submit.rs @@ -318,11 +318,6 @@ async fn finish_saved( SubmissionState::AcceptanceUnknown => reconcile(state, route).await, // the dependency release owns a held route, including its launch SubmissionState::Held { phase } => Ok((route.task, phase.status())), - SubmissionState::Resource { .. } - | SubmissionState::ResourceAction { .. } - | SubmissionState::ResourceBackground { .. } => { - Err(conflict(&route, "request UUID belongs to a resource route")) - } } } @@ -449,11 +444,6 @@ async fn resolve_identity( SubmissionState::AcceptanceUnknown | SubmissionState::Held { .. } => { Err(unknown(&route, "origin route is unresolved")) } - SubmissionState::Resource { .. } - | SubmissionState::ResourceAction { .. } - | SubmissionState::ResourceBackground { .. } => { - Err(conflict(&saved, "request UUID belongs to a resource route")) - } } } @@ -514,15 +504,6 @@ pub(super) fn conflict(route: &OriginRoute, message: impl Into) -> AppEr } fn ensure_direct_route(route: &OriginRoute) -> Result<(), AppError> { - // resource and action-bound routes never use the generic abandon path - if matches!( - route.submission, - SubmissionState::Resource { .. } - | SubmissionState::ResourceAction { .. } - | SubmissionState::ResourceBackground { .. } - ) { - return Err(conflict(route, "request UUID belongs to a resource route")); - } if route.spec.current().is_none() { return Err(conflict(route, "migrated local task has no remote request")); } @@ -624,18 +605,16 @@ fn local_dry_run( mod tests { use std::path::Path; - use super::{decode_identity, decode_preview, ensure_direct_route}; + use super::{decode_identity, decode_preview}; use crate::daemon::cluster::PreviewBody; use crate::domain::{API_VERSION, TaskEnv, TaskId}; use crate::error::AppError; use crate::fleet::http::ClusterResponse; use crate::fleet::protocol::ClusterProtocolVersion; use crate::machine::MachineId; - use crate::resource::ResourceId; use crate::spec::NormalizedSpec; use crate::submission::{ - CallbackContext, CallbackExecutable, NewResourceRoute, OriginRoute, RequestId, - ResourceRoutePhase, SubmissionState, + CallbackContext, CallbackExecutable, OriginRoute, RequestId, SubmissionState, }; use axum::http::StatusCode; use bytes::Bytes; @@ -726,47 +705,4 @@ mod tests { Err(AppError::RemoteSubmissionUnavailable { .. }) )); } - - #[test] - fn direct_retry_rejects_a_saved_resource_request_before_reconciliation() { - let spec: NormalizedSpec = serde_json::from_value(serde_json::json!({ - "api_version": 1, - "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "resource task", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": ["echo", "hello"] } - })) - .unwrap(); - let route = OriginRoute::new_resource_waiting(NewResourceRoute { - request: RequestId::new(), - task: TaskId::new(), - origin_machine: MachineId::new(), - authority_machine: MachineId::new(), - thread: spec.thread, - callback: CallbackContext { - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - cwd: Path::new("/tmp").to_path_buf(), - codex: Path::new("/bin/echo").to_path_buf().into(), - }, - spec, - resource: ResourceId::new(), - }) - .unwrap(); - - assert!(matches!( - ensure_direct_route(&route), - Err(AppError::SubmissionConflict { .. }) - )); - assert!(matches!( - route.submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::AcceptanceUnknown, - .. - } - )); - } } diff --git a/src/daemon/release_watcher_api.rs b/src/daemon/release_watcher_api.rs deleted file mode 100644 index 23a8229..0000000 --- a/src/daemon/release_watcher_api.rs +++ /dev/null @@ -1,77 +0,0 @@ -//! Socket-only internal route for release watchers running on the resource authority -//! -//! The route is merged only into the Unix-socket router. The dashboard and fleet -//! TCP listeners never serve it, so only local processes with socket access can -//! advance a release action, and only with identities the authority saved - -use axum::extract::State; -use axum::routing::post; -use axum::{Json, Router}; - -use crate::daemon::AppState; -use crate::daemon::actors::{StoreMsg, call}; -use crate::daemon::resource_api::StrictJson; -use crate::error::AppError; -use crate::resource::release_watcher::{ - RELEASE_WATCHER_POLL_PATH, RELEASE_WATCHER_PROTOCOL_VERSION, ReleaseWatcherPollOutcome, - ReleaseWatcherPollRequest, ReleaseWatcherPollResponse, -}; - -/// Routes served only on the daemon Unix socket -pub(crate) fn socket_routes() -> Router { - Router::new().route(RELEASE_WATCHER_POLL_PATH, post(poll)) -} - -async fn poll( - State(state): State, - StrictJson(request): StrictJson, -) -> Result, AppError> { - if request.protocol_version != RELEASE_WATCHER_PROTOCOL_VERSION { - return Err(AppError::Usage { - message: format!( - "unsupported release watcher protocol version {}", - request.protocol_version - ), - }); - } - - let watcher = request.watcher; - let outcome = call(&state.store, |reply| { - StoreMsg::PollReleaseWatcherForAuthority { - authority_machine: state.machine.identity.machine, - request, - reply, - } - }) - .await? - .map_err(|error| AppError::Internal { - message: format!("release watcher poll storage: {error}"), - })?; - match &outcome { - ReleaseWatcherPollOutcome::StopCommitted { generation_id, .. } => tracing::info!( - resource = %watcher.resource_id.as_uuid(), - action = %watcher.action_id.as_uuid(), - trainer = %watcher.trainer_task_id, - %generation_id, - "release watcher committed the exact trainer stop" - ), - ReleaseWatcherPollOutcome::Attention { reason } => tracing::warn!( - resource = %watcher.resource_id.as_uuid(), - action = %watcher.action_id.as_uuid(), - watcher = %watcher.watcher_task_id.as_task_id(), - ?reason, - "release watcher needs attention" - ), - ReleaseWatcherPollOutcome::WatcherNotRunning - | ReleaseWatcherPollOutcome::WaitingForTrainerStart - | ReleaseWatcherPollOutcome::WaitingForCheckpoint - | ReleaseWatcherPollOutcome::CompletedResultAwaitingTrainerExit - | ReleaseWatcherPollOutcome::TrainerCompleted - | ReleaseWatcherPollOutcome::ReleaseSettled => {} - } - - Ok(Json(ReleaseWatcherPollResponse { - protocol_version: RELEASE_WATCHER_PROTOCOL_VERSION, - outcome, - })) -} diff --git a/src/daemon/resource_action.rs b/src/daemon/resource_action.rs deleted file mode 100644 index 6c124fa..0000000 --- a/src/daemon/resource_action.rs +++ /dev/null @@ -1,941 +0,0 @@ -//! Two-machine resource actions: supervisor-owned routes and authority acceptance -//! -//! The supervisor machine asks the authority to prepare the canonical task, saves -//! a fixed-ID origin route with that exact spec, and only then sends the launch -//! The authority reads back the saved route as evidence before it accepts. A lost -//! reply or restart repeats the same launch; nothing on this path abandons the -//! identity, because abandoning could strand the resource action -//! -//! When the supervisor thread runs on the authority itself, the socket route -//! hands the choice to the supervisor actor's one-shot return and resolution -//! owners instead, and their store receipts answer retries - -use axum::extract::{Path, Query, State}; -use axum::http::StatusCode; -use axum::routing::{get, post}; -use axum::{Json, Router}; -use serde::{Deserialize, Serialize}; -use serde_json::Value; -use tracing::warn; - -use super::AppState; -use super::actors::supervisor::{ReturnDecisionOutcome, return_decision_rejection}; -use super::actors::{StoreMsg, SupervisorMsg, call}; -use super::cluster::ReadQuery; -use crate::domain::{API_VERSION, AgentKind, TaskEnv, TaskId}; -use crate::error::AppError; -use crate::fleet::http::{ClusterClient, ClusterResponse}; -use crate::fleet::protocol::{ClusterProtocolVersion, ProtocolRange, SUPPORTED_PROTOCOLS}; -use crate::invocation::resolve_agent_binary; -use crate::machine::MachineId; -use crate::resource::bound_action::{ - ActionTaskIdentity, LocalReturnAcceptance, LocalReturnReceipt, PreparedActionTask, - RESOURCE_ACTION_PATH, RESOURCE_ACTION_PROTOCOL_VERSION, RESOURCE_ACTION_ROUTE_PROOF_PATH, - RESOURCE_ACTION_SUBMIT_PATH, ResourceActionChoice, ResourceActionLaunch, - ResourceActionOperation, ResourceActionOutcome, ResourceActionRejection, ResourceActionRequest, - ResourceActionResponse, ResourceActionSubmitOutcome, ResourceActionSubmitRequest, - ResourceActionSubmitResponse, -}; -use crate::resource::{CommandSpec, ReturnDecision, ReturnLaunch, SupervisorActionAuthority}; -use crate::store::{ - EndedRestoreResolution, ResourceActionRouteResult, ReturnClosure, ReturnDecisionError, - ReturnTaskAcceptance, -}; -use crate::submission::{ - CallbackContext, CallbackExecutable, NewResourceActionRoute, OriginRoute, RequestId, - ResourceActionRouteBinding, ResourceActionRoutePhase, ResourceActionRouteProof, - SubmissionState, normalized_spec_sha256, -}; - -/// Saved route proof returned by the supervisor machine -#[derive(Debug, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceActionRouteProofBody { - /// Public API version - pub api_version: u32, - /// Proof for the named task, when this machine saved a valid action route - pub proof: Option, -} - -/// Cluster routes: authority acceptance and supervisor route evidence -pub(super) fn cluster_routes() -> Router { - Router::new() - .route(RESOURCE_ACTION_PATH, post(receive)) - .route( - &format!("{RESOURCE_ACTION_ROUTE_PROOF_PATH}/{{task}}"), - get(route_proof), - ) -} - -/// Socket-only route that starts one action from the supervisor's own machine -pub(crate) fn socket_routes() -> Router { - Router::new().route(RESOURCE_ACTION_SUBMIT_PATH, post(submit_from_socket)) -} - -// ---- authority side ---- - -async fn receive( - State(state): State, - Json(request): Json, -) -> Result, AppError> { - state - .machine - .identity - .check_destination(request.destination_machine)?; - check_protocol(request.protocol_version, request.source_machine)?; - request.validate().map_err(|error| AppError::Usage { - message: format!("invalid resource action: {error}"), - })?; - - if let Some(identity) = request.operation.launch_identity() { - let proof = fetch_route_proof(&state, request.source_machine, identity.task_id).await?; - if let Err(reason) = check_route_evidence(&request, identity, proof) { - warn!( - action = %request.authority.action_id.as_uuid(), - task = %identity.task_id, - ?reason, - "resource action route evidence refused" - ); - return Ok(Json(ResourceActionResponse::new( - &request, - ResourceActionOutcome::Rejected { reason }, - ))); - } - } - - let outcome = call(&state.supervisor, |reply| SupervisorMsg::ResourceAction { - request: Box::new(request.clone()), - reply, - }) - .await?; - Ok(Json(ResourceActionResponse::new(&request, outcome))) -} - -fn check_protocol(version: u32, machine: MachineId) -> Result<(), AppError> { - let version = ClusterProtocolVersion(version); - if SUPPORTED_PROTOCOLS.min() <= version && version <= SUPPORTED_PROTOCOLS.max() { - return Ok(()); - } - let Some(remote) = ProtocolRange::new(version, version) else { - return Err(AppError::Usage { - message: "unsupported cluster protocol version".into(), - }); - }; - Err(AppError::ClusterProtocolIncompatible { - machine, - local: SUPPORTED_PROTOCOLS, - remote, - }) -} - -/// Compare the supervisor machine's saved route with one launch request -fn check_route_evidence( - request: &ResourceActionRequest, - identity: ActionTaskIdentity, - proof: Option, -) -> Result<(), ResourceActionRejection> { - let proof = proof.ok_or(ResourceActionRejection::RouteEvidenceMissing)?; - let kind = request.operation.task_kind(); - if proof.request != identity.request_id - || proof.task != identity.task_id - || Some(proof.binding.kind) != kind - || proof.binding.authority != request.authority - || proof.normalized_spec_sha256 != identity.normalized_spec_sha256 - || matches!(proof.phase, ResourceActionRoutePhase::Rejected { .. }) - { - return Err(ResourceActionRejection::RouteEvidenceMismatch); - } - Ok(()) -} - -/// Read the action route that the supervisor machine saved for one task -async fn fetch_route_proof( - state: &AppState, - origin: MachineId, - task: TaskId, -) -> Result, AppError> { - let unavailable = |message: String| AppError::RemoteSubmissionUnavailable { - message: format!("supervisor route proof is unavailable: {message}"), - }; - let fleet = state - .fleet - .handle() - .ok_or_else(|| unavailable("fleet is disabled".into()))?; - let destination = fleet - .connect(origin) - .await - .map_err(|error| unavailable(error.to_string()))?; - let path = format!( - "{RESOURCE_ACTION_ROUTE_PROOF_PATH}/{task}?api_version={API_VERSION}&destination_machine={origin}" - ); - let response = ClusterClient::default() - .get(&destination.address, &path) - .await - .map_err(|error| unavailable(error.to_string()))?; - if response.status != StatusCode::OK { - return Err(unavailable(format!("HTTP {}", response.status))); - } - let body: ResourceActionRouteProofBody = serde_json::from_slice(&response.body) - .map_err(|error| unavailable(format!("invalid response: {error}")))?; - if body.api_version != API_VERSION { - return Err(unavailable("unsupported API version".into())); - } - Ok(body.proof.filter(|proof| proof.task == task)) -} - -/// Serve the saved action route proof for one task owned by this supervisor machine -async fn route_proof( - State(state): State, - Path(task): Path, - Query(query): Query, -) -> Result, AppError> { - state - .machine - .identity - .check_destination(query.destination_machine)?; - if query.api_version != API_VERSION { - return Err(AppError::Usage { - message: "unsupported API version".into(), - }); - } - let route = call(&state.store, |reply| StoreMsg::OriginRoute { - id: task, - reply, - }) - .await?; - let local = state.machine.identity.machine; - let proof = route - .as_ref() - .filter(|route| route.task == task && route.origin_machine == local) - .and_then(ResourceActionRouteProof::from_route); - Ok(Json(ResourceActionRouteProofBody { - api_version: API_VERSION, - proof, - })) -} - -// ---- supervisor side ---- - -async fn submit_from_socket( - State(state): State, - Json(request): Json, -) -> Result, AppError> { - if request.api_version != API_VERSION { - return Err(AppError::Usage { - message: "unsupported API version".into(), - }); - } - let outcome = submit(&state, request.authority, request.choice).await?; - Ok(Json(ResourceActionSubmitResponse { - api_version: API_VERSION, - outcome, - })) -} - -/// Apply one supervisor choice for a pending action -/// -/// On a remote authority, a launch saves the fixed-ID route before it is sent, -/// and an uncertain send leaves the route for the same retry. A saved route -/// answers repeated calls. When this machine is also the authority, the -/// supervisor actor applies the choice and its store receipt answers retries -pub(crate) async fn submit( - state: &AppState, - authority: SupervisorActionAuthority, - choice: ResourceActionChoice, -) -> Result { - let local = state.machine.identity.machine; - if authority.supervisor.machine != local { - return Err(AppError::Usage { - message: "the resource supervisor thread is not on this machine".into(), - }); - } - if authority.authority_machine == local { - return submit_co_located(state, authority, choice).await; - } - let _action_guard = state.locks.resource_actions.lock(authority.action_id).await; - match choice { - ResourceActionChoice::ReleaseWatcher { - observed_background_task, - } => { - submit_launch( - state, - authority, - None, - ResourceActionLaunch::ReleaseWatcher { - observed_background_task, - }, - ) - .await - } - ResourceActionChoice::Return { - decision: ReturnDecision::Launch(launch), - } => { - let ReturnLaunch { - request_id, - task_id, - work, - } = *launch; - submit_launch( - state, - authority, - Some((request_id, task_id)), - ResourceActionLaunch::Return { - work: Box::new(work), - }, - ) - .await - } - ResourceActionChoice::Return { - decision: ReturnDecision::NoResume { reason }, - } => { - submit_without_task( - state, - authority, - ResourceActionOperation::NoResume { reason }, - ) - .await - } - ResourceActionChoice::ResolveEndedRestore { task_id, reason } => { - submit_without_task( - state, - authority, - ResourceActionOperation::ResolveEndedRestore { task_id, reason }, - ) - .await - } - ResourceActionChoice::HoldReturn { hold } => { - submit_without_task( - state, - authority, - ResourceActionOperation::HoldReturn { hold }, - ) - .await - } - } -} - -/// Apply one choice when the supervisor thread runs on the resource authority -/// -/// The supervisor actor owns the one-shot decision: the store checks the exact -/// authority, action, revision, and assignment, saves the receipt with the task -/// records, and only an insertion by this call spawns. An actor or storage -/// failure stays an error, so an unknown outcome is never reported as a refusal -async fn submit_co_located( - state: &AppState, - authority: SupervisorActionAuthority, - choice: ResourceActionChoice, -) -> Result { - match choice { - // the resource actor starts the bound watcher itself for a local supervisor - ResourceActionChoice::ReleaseWatcher { .. } => Err(AppError::ResourceActionNotAllowed { - resource: authority.resource_id, - message: "the resource authority starts the release watcher itself when the supervisor runs on the same machine; wait for the release to finish".into(), - }), - ResourceActionChoice::Return { decision } => { - let launch_ids = match &decision { - ReturnDecision::Launch(launch) => Some((launch.request_id, launch.task_id)), - ReturnDecision::NoResume { .. } => None, - }; - let decided = call(&state.supervisor, |reply| SupervisorMsg::DecideReturn { - authority, - decision: Box::new(decision), - reply, - }) - .await?; - let outcome = match decided { - Ok(outcome) => outcome, - Err(error) => return rejected(error), - }; - co_located_return_outcome(authority, launch_ids, outcome) - } - ResourceActionChoice::ResolveEndedRestore { task_id, reason } => { - let resolved = call(&state.supervisor, |reply| SupervisorMsg::ResolveEndedRestore { - resolution: Box::new(EndedRestoreResolution { - authority, - task_id, - reason, - }), - reply, - }) - .await?; - match resolved { - Ok(closure) => Ok(ResourceActionSubmitOutcome::Closed { - loan: closure.loan, - state_revision: closure.state_revision, - }), - Err(error) => rejected(error), - } - } - // a hold changes only the saved deadline, which the resource actor reads at its wake - ResourceActionChoice::HoldReturn { hold } => { - let held = call(&state.store, |reply| StoreMsg::HoldReturnForAuthority { - authority, - hold, - reply, - }) - .await?; - match held { - Ok(window) => Ok(ResourceActionSubmitOutcome::ReturnHeld { window }), - Err(error) => rejected(error), - } - } - } -} - -/// Convert the supervisor actor's return result, checking that it names the decision -fn co_located_return_outcome( - authority: SupervisorActionAuthority, - launch_ids: Option<(RequestId, TaskId)>, - outcome: ReturnDecisionOutcome, -) -> Result { - let mismatch = |message: &str| AppError::Internal { - message: format!("co-located return decision: {message}"), - }; - let accepted = |request_id, task_id, acceptance| { - Ok(ResourceActionSubmitOutcome::LocalReturnAccepted { - receipt: LocalReturnReceipt { - authority, - request_id, - task_id, - }, - acceptance, - }) - }; - match (launch_ids, outcome) { - (None, ReturnDecisionOutcome::Closed(closure)) => { - let ReturnClosure { - loan, - state_revision, - } = *closure; - Ok(ResourceActionSubmitOutcome::Closed { - loan, - state_revision, - }) - } - ( - Some((request_id, task_id)), - ReturnDecisionOutcome::Launch(ReturnTaskAcceptance::Inserted { - loan, - task, - state_revision, - }), - ) if task == task_id => accepted( - request_id, - task_id, - LocalReturnAcceptance::Inserted { - loan, - state_revision, - }, - ), - ( - Some((request_id, task_id)), - ReturnDecisionOutcome::Launch(ReturnTaskAcceptance::Existing { task, state }), - ) if task == task_id => accepted( - request_id, - task_id, - LocalReturnAcceptance::Existing { state }, - ), - // the store answers this only when the assignment moved off this machine - ( - Some(_), - ReturnDecisionOutcome::Launch(ReturnTaskAcceptance::UnsupportedRemoteSupervisor { - .. - }), - ) => Ok(ResourceActionSubmitOutcome::Rejected { - reason: ResourceActionRejection::NotCurrentSupervisor, - }), - (Some(_), ReturnDecisionOutcome::Launch(_)) => { - Err(mismatch("the bound task differs from the decision")) - } - (None, ReturnDecisionOutcome::Launch(_)) | (Some(_), ReturnDecisionOutcome::Closed(_)) => { - Err(mismatch("the result does not fit the decision")) - } - } -} - -fn rejected(error: ReturnDecisionError) -> Result { - return_decision_rejection(error).map(|reason| ResourceActionSubmitOutcome::Rejected { reason }) -} - -/// Retry action routes whose authority acceptance was unknown at daemon startup -pub(super) async fn recover(state: AppState) { - let routes = match call(&state.store, |reply| { - StoreMsg::UnknownResourceActionRoutes { reply } - }) - .await - { - Ok(routes) => routes, - Err(error) => { - warn!("cannot scan unresolved resource action routes: {error}"); - return; - } - }; - for route in routes { - let SubmissionState::ResourceAction { binding, .. } = &route.submission else { - continue; - }; - let _action_guard = state - .locks - .resource_actions - .lock(binding.authority.action_id) - .await; - if let Err(error) = send_saved_launch(&state, &route).await { - warn!(task = %route.task, "resource action recovery: {error}"); - } - } -} - -/// Prepare, save, and send one launch, or resend the launch of a saved route -/// -/// A return decision carries supervisor-chosen request and task identities; a -/// watcher uses the identities that the authority bound to its release action -async fn submit_launch( - state: &AppState, - authority: SupervisorActionAuthority, - identities: Option<(RequestId, TaskId)>, - launch: ResourceActionLaunch, -) -> Result { - let binding = ResourceActionRouteBinding { - kind: launch.kind(), - authority, - }; - let saved = call(&state.store, |reply| { - StoreMsg::ResourceActionRouteByAction { - action: authority.action_id, - reply, - } - }) - .await?; - let route = match saved { - Some(route) => { - check_saved_choice(&route, &binding, identities, &launch)?; - route - } - None => match create_route(state, binding, identities, launch).await? { - Ok(route) => route, - // the authority refused before any route was saved, so nothing waits for a retry - Err(reason) => return Ok(ResourceActionSubmitOutcome::Rejected { reason }), - }, - }; - match &route.submission { - SubmissionState::ResourceAction { - phase: ResourceActionRoutePhase::AcceptanceUnknown, - .. - } => send_saved_launch(state, &route).await, - _ => saved_outcome(&route), - } -} - -/// A retry must repeat the saved action, identities, and supervisor choice -fn check_saved_choice( - route: &OriginRoute, - binding: &ResourceActionRouteBinding, - identities: Option<(RequestId, TaskId)>, - launch: &ResourceActionLaunch, -) -> Result<(), AppError> { - let SubmissionState::ResourceAction { - binding: saved_binding, - launch: saved_launch, - .. - } = &route.submission - else { - return Err(conflict(route, "saved route is not an action route")); - }; - if saved_binding != binding || **saved_launch != *launch { - return Err(conflict(route, "action was retried with another choice")); - } - if identities.is_some_and(|identities| identities != (route.request, route.task)) { - return Err(conflict(route, "action already binds other identities")); - } - Ok(()) -} - -/// Ask the authority to prepare the canonical task, then save the fixed-ID route -/// -/// A definitive refusal to prepare saves no route and is returned as its reason -async fn create_route( - state: &AppState, - binding: ResourceActionRouteBinding, - identities: Option<(RequestId, TaskId)>, - launch: ResourceActionLaunch, -) -> Result, AppError> { - let authority = binding.authority; - let operation = match (&launch, identities) { - ( - ResourceActionLaunch::ReleaseWatcher { - observed_background_task, - }, - None, - ) => ResourceActionOperation::PrepareReleaseWatcher { - observed_background_task: *observed_background_task, - }, - (ResourceActionLaunch::Return { work }, Some((request_id, task_id))) => { - ResourceActionOperation::PrepareReturn { - launch: ReturnLaunch { - request_id, - task_id, - work: work.as_ref().clone(), - }, - } - } - (ResourceActionLaunch::ReleaseWatcher { .. }, Some(_)) - | (ResourceActionLaunch::Return { .. }, None) => { - return Err(AppError::Usage { - message: "resource action identities do not fit the chosen launch".into(), - }); - } - }; - let prepared = match send(state, authority, operation).await? { - ResourceActionOutcome::Prepared { task } => task, - ResourceActionOutcome::Rejected { reason } => return Ok(Err(reason)), - ResourceActionOutcome::Accepted { .. } - | ResourceActionOutcome::Closed { .. } - | ResourceActionOutcome::ReturnHeld { .. } => { - return Err(AppError::RemoteSubmissionUnavailable { - message: "resource authority answered prepare with another outcome".into(), - }); - } - }; - check_prepared(&prepared, &binding, identities, &launch)?; - - let env = TaskEnv::capture(); - let cwd = state.home.root().to_path_buf(); - let codex = match resolve_agent_binary(AgentKind::Codex, &env.path, &cwd) { - Ok(path) => CallbackExecutable::available(path), - Err(error) => CallbackExecutable::Unavailable { - reason: error.to_string(), - }, - }; - let route = OriginRoute::new_resource_action(NewResourceActionRoute { - request: prepared.request_id, - task: prepared.task_id, - callback: CallbackContext { env, cwd, codex }, - spec: prepared.spec, - binding, - launch, - }) - .map_err(|error| AppError::RemoteSubmissionUnavailable { - message: format!("prepared resource action task is invalid: {error}"), - })?; - call(&state.store, |reply| StoreMsg::InsertOriginRoute { - route: Box::new(route), - reply, - }) - .await - .map(Ok) -} - -/// Check the authority's prepared task before it becomes the saved route content -fn check_prepared( - prepared: &PreparedActionTask, - binding: &ResourceActionRouteBinding, - identities: Option<(RequestId, TaskId)>, - launch: &ResourceActionLaunch, -) -> Result<(), AppError> { - let invalid = |message: &str| AppError::RemoteSubmissionUnavailable { - message: format!("resource authority prepared an invalid task: {message}"), - }; - if !prepared.digest_matches() { - return Err(invalid("digest does not cover the spec")); - } - if prepared.spec.thread != binding.authority.supervisor.thread - || prepared.spec.machine.is_some() - { - return Err(invalid("spec does not name the supervisor thread")); - } - if identities.is_some_and(|ids| ids != (prepared.request_id, prepared.task_id)) { - return Err(invalid("identities differ from the decision")); - } - let chosen = match launch { - ResourceActionLaunch::Return { work } => { - work.supervisor_spec().map(CommandSpec::as_normalized) - } - ResourceActionLaunch::ReleaseWatcher { .. } => None, - }; - if let Some(chosen) = chosen - && normalized_spec_sha256(chosen)? != prepared.normalized_spec_sha256 - { - return Err(invalid("spec differs from the supervisor's command")); - } - Ok(()) -} - -/// Send the launch for one saved route and apply a definitive answer to it -async fn send_saved_launch( - state: &AppState, - route: &OriginRoute, -) -> Result { - let SubmissionState::ResourceAction { - binding, launch, .. - } = &route.submission - else { - return Err(conflict(route, "saved route is not an action route")); - }; - let spec = route - .current_spec() - .ok_or_else(|| conflict(route, "action route has no spec"))?; - let task = ActionTaskIdentity { - request_id: route.request, - task_id: route.task, - normalized_spec_sha256: normalized_spec_sha256(spec)?, - }; - let operation = match launch.as_ref() { - ResourceActionLaunch::ReleaseWatcher { - observed_background_task, - } => ResourceActionOperation::LaunchReleaseWatcher { - observed_background_task: *observed_background_task, - task, - }, - ResourceActionLaunch::Return { work } => ResourceActionOperation::LaunchReturn { - launch: ReturnLaunch { - request_id: route.request, - task_id: route.task, - work: work.as_ref().clone(), - }, - normalized_spec_sha256: task.normalized_spec_sha256, - }, - }; - let outcome = send(state, binding.authority, operation) - .await - .map_err(|error| unknown(route, error.to_string()))?; - let result = match outcome { - ResourceActionOutcome::Accepted { receipt, .. } => { - ResourceActionRouteResult::Accepted(receipt) - } - ResourceActionOutcome::Rejected { reason } => ResourceActionRouteResult::Rejected(reason), - ResourceActionOutcome::Prepared { .. } - | ResourceActionOutcome::Closed { .. } - | ResourceActionOutcome::ReturnHeld { .. } => { - return Err(unknown( - route, - "resource authority answered a launch with another outcome", - )); - } - }; - let saved = call(&state.store, |reply| StoreMsg::ResolveResourceActionRoute { - task: route.task, - result: Box::new(result), - reply, - }) - .await - .map_err(|error| unknown(route, format!("cannot save authority result: {error}")))?; - saved_outcome(&saved) -} - -/// Send one operation that binds no task, so no route is saved before it -async fn submit_without_task( - state: &AppState, - authority: SupervisorActionAuthority, - operation: ResourceActionOperation, -) -> Result { - match send(state, authority, operation).await? { - ResourceActionOutcome::Closed { - loan, - state_revision, - } => Ok(ResourceActionSubmitOutcome::Closed { - loan, - state_revision, - }), - ResourceActionOutcome::ReturnHeld { window } => { - Ok(ResourceActionSubmitOutcome::ReturnHeld { window }) - } - ResourceActionOutcome::Rejected { reason } => { - Ok(ResourceActionSubmitOutcome::Rejected { reason }) - } - ResourceActionOutcome::Prepared { .. } | ResourceActionOutcome::Accepted { .. } => { - Err(AppError::RemoteSubmissionUnavailable { - message: "resource authority answered a taskless operation with a task outcome" - .into(), - }) - } - } -} - -fn saved_outcome(route: &OriginRoute) -> Result { - let SubmissionState::ResourceAction { binding, phase, .. } = &route.submission else { - return Err(conflict(route, "saved route is not an action route")); - }; - match phase { - ResourceActionRoutePhase::AcceptanceUnknown => { - Err(unknown(route, "authority acceptance is unresolved")) - } - ResourceActionRoutePhase::Accepted => { - let spec = route - .current_spec() - .ok_or_else(|| conflict(route, "action route has no spec"))?; - Ok(ResourceActionSubmitOutcome::Accepted { - receipt: crate::resource::bound_action::ActionTaskReceipt { - kind: binding.kind, - authority: binding.authority, - request_id: route.request, - task_id: route.task, - normalized_spec_sha256: normalized_spec_sha256(spec)?, - }, - last_execution_state: route.last_execution_state, - }) - } - ResourceActionRoutePhase::Rejected { reason } => { - Ok(ResourceActionSubmitOutcome::Rejected { - reason: reason.clone(), - }) - } - } -} - -/// Send one operation to the authority and decode its strict typed answer -async fn send( - state: &AppState, - authority: SupervisorActionAuthority, - operation: ResourceActionOperation, -) -> Result { - let unavailable = |message: String| AppError::MachineUnavailable { - machine: authority.authority_machine, - message, - }; - let fleet = state - .fleet - .handle() - .ok_or_else(|| unavailable("fleet is disabled".into()))?; - let destination = fleet - .connect(authority.authority_machine) - .await - .map_err(|error| unavailable(error.to_string()))?; - let request = ResourceActionRequest::new(destination.protocol.0, authority, operation); - let response = ClusterClient::default() - .post_json(&destination.address, RESOURCE_ACTION_PATH, &request) - .await - .map_err(|error| unavailable(error.to_string()))?; - decode_response(&request, response) -} - -fn decode_response( - request: &ResourceActionRequest, - response: ClusterResponse, -) -> Result { - if !response.status.is_success() { - return Err(crate::client::map_error(response.status, &response.body)); - } - let invalid = |message: String| AppError::RemoteSubmissionUnavailable { - message: format!("invalid resource authority response: {message}"), - }; - let value: Value = - serde_json::from_slice(&response.body).map_err(|error| invalid(error.to_string()))?; - let body: ResourceActionResponse = - serde_json::from_value(value).map_err(|error| invalid(error.to_string()))?; - if body.api_version != API_VERSION - || body.protocol_version != request.protocol_version - || body.action_protocol_version != RESOURCE_ACTION_PROTOCOL_VERSION - || body.destination_machine != request.destination_machine - || body.action_id != request.authority.action_id - { - return Err(invalid("response names another route or version".into())); - } - Ok(body.outcome) -} - -fn unknown(route: &OriginRoute, message: impl Into) -> AppError { - AppError::SubmissionOutcomeUnknown { - request: route.request, - task: route.task, - message: message.into(), - } -} - -fn conflict(route: &OriginRoute, message: impl Into) -> AppError { - AppError::SubmissionConflict { - request: route.request, - task: route.task, - message: message.into(), - } -} - -#[cfg(test)] -mod tests { - use uuid::Uuid; - - use super::{co_located_return_outcome, rejected}; - use crate::daemon::actors::supervisor::ReturnDecisionOutcome; - use crate::domain::{ProcessStatus, TaskId, ThreadId}; - use crate::machine::MachineId; - use crate::resource::bound_action::{ - LocalReturnAcceptance, LocalReturnReceipt, ResourceActionRejection, - ResourceActionSubmitOutcome, - }; - use crate::resource::{ - ActionId, AssignmentRevision, LoanId, ResourceId, ResourceRevision, - SupervisorActionAuthority, SupervisorAddress, - }; - use crate::store::{ReturnDecisionError, ReturnTaskAcceptance}; - use crate::submission::RequestId; - - fn co_located() -> SupervisorActionAuthority { - let machine = MachineId::new(); - SupervisorActionAuthority { - authority_machine: machine, - resource_id: ResourceId::new(), - loan_id: LoanId::new(), - action_id: ActionId::new(), - expected_state_revision: ResourceRevision::new(2), - supervisor: SupervisorAddress { - machine, - thread: ThreadId(Uuid::now_v7()), - }, - assignment_revision: AssignmentRevision::new(1), - } - } - - #[test] - fn storage_failure_stays_unknown_and_refusals_stay_definitive() { - let unknown = rejected(ReturnDecisionError::Storage( - rusqlite::Error::QueryReturnedNoRows, - )); - assert!(unknown.is_err()); - - let refused = rejected(ReturnDecisionError::ConflictingRetry { - action_id: ActionId::new(), - }) - .unwrap(); - assert!(matches!( - refused, - ResourceActionSubmitOutcome::Rejected { - reason: ResourceActionRejection::ConflictingRetry - } - )); - } - - #[test] - fn co_located_return_result_must_name_the_decided_task() { - let authority = co_located(); - let (request_id, task_id) = (RequestId::new(), TaskId::new()); - let existing = |task| { - ReturnDecisionOutcome::Launch(ReturnTaskAcceptance::Existing { - task, - state: ProcessStatus::Queued, - }) - }; - - let outcome = - co_located_return_outcome(authority, Some((request_id, task_id)), existing(task_id)) - .unwrap(); - assert!(matches!( - outcome, - ResourceActionSubmitOutcome::LocalReturnAccepted { - receipt, - acceptance: LocalReturnAcceptance::Existing { - state: ProcessStatus::Queued - }, - } if receipt == LocalReturnReceipt { authority, request_id, task_id } - )); - - // another bound task is an internal inconsistency, never an acceptance - assert!( - co_located_return_outcome( - authority, - Some((request_id, task_id)), - existing(TaskId::new()) - ) - .is_err() - ); - assert!(co_located_return_outcome(authority, None, existing(task_id)).is_err()); - } -} diff --git a/src/daemon/resource_api.rs b/src/daemon/resource_api.rs deleted file mode 100644 index 928ad81..0000000 --- a/src/daemon/resource_api.rs +++ /dev/null @@ -1,2292 +0,0 @@ -//! Resource reads, controls, and submissions on the socket, dashboard, and Fleet -//! -//! Every view comes from the fixed authority's durable resource, loan, request, -//! notice, and task state. A peer that cannot answer is reported as an -//! unavailable authority, never as an idle resource. Every mutation runs on the -//! authority through the StoreActor and the existing owners: origin-owned -//! cancellation, the resource queue submission path, and the notice sender - -use crate::resource::ReturnContext; -use crate::resource::trainer_publication::AttemptBinding; -use std::collections::{BTreeSet, HashMap}; -use std::path::PathBuf; -use std::time::{Duration, Instant}; - -use axum::body::Bytes; -use axum::extract::rejection::{PathRejection, QueryRejection}; -use axum::extract::{FromRequest, Path, Query, Request, State}; -use axum::http::StatusCode; -use axum::routing::{get, post}; -use axum::{Json, Router}; -use serde::Deserialize; -use serde::de::DeserializeOwned; -use serde_json::Value; -use tokio::task::JoinSet; -use tracing::warn; -use uuid::Uuid; - -use super::AppState; -use super::actors::resource::{ - ReleaseWatcherAttentionReason, ReleaseWatcherStatus, ResourceActorInspection, -}; -use super::actors::supervisor::BackgroundLaunch; -use super::actors::{StoreMsg, SupervisorMsg, call}; -use super::api::views::TaskSummary; -use super::cluster::{OriginResourceCancellationOutcome, ReadQuery}; -use super::peer_read::{READ_MAX_BODY, read_peer}; -use super::resource_submit::{ResourceSubmitInput, ResourceSubmitOutcome}; -use crate::cancellation::ResourceCancellationTarget; -use crate::domain::{API_VERSION, ProcessStatus, TaskEnv, TaskId, ThreadId}; -use crate::error::AppError; -use crate::fleet::http::{ClusterClient, ClusterResponse}; -use crate::machine::MachineId; -use crate::resource::api::{ - AttentionCode, AttentionView, BackgroundLaunchReservation, BackgroundLaunchReservationStatus, - BrowserResourceAction, CLUSTER_RESOURCE_PENDING_PATH, CLUSTER_RESOURCES_PATH, - ClusterPendingActions, ClusterResourceControl, ClusterResourceDetail, ClusterResourceList, - InitialIdleBody, InitialIdleResponse, OperatorReleaseBody, OperatorReleaseResponse, - PendingActionList, PendingActionPhase, PendingActionView, RESOURCE_PENDING_PATH, - RESOURCE_REGISTER_PATH, RequestCancelBody, ResourceActionBody, ResourceBackgroundSubmitOutcome, - ResourceBackgroundSubmitResponse, ResourceControl, ResourceDetail, ResourceList, - ResourceOverview, ResourceRegisterBody, ResourceRegistration, ResourceRequestSubmitOutcome, - ResourceRequestSubmitResponse, ResourceRequestView, ResourceTaskSummary, - SupervisorReplacementBody, TrainerAttemptBody, TrainerAttemptResponse, UnavailableAuthority, -}; -use crate::resource::initial_idle::InitialIdleRefusal; -use crate::resource::operator_release::OperatorGpuFreeRefusal; -use crate::resource::{ - AssignmentRevision, DeliveryAttemptId, IdleProofGap, Loan, LoanPhase, LoanState, Resource, - ResourceId, ResourceQueueAttentionReason, ResourceQueueReconcileOutcome, ResourceRequest, - ResourceRequestState, ResourceRevision, RestoreAttentionReason, SupervisorAddress, - SupervisorNoticeDelivery, -}; -use crate::spec::{self, NormalizedSpec}; -use crate::store::{ - BackgroundLaunchAcceptance, BackgroundLaunchError, BackgroundLaunchPhase, InitialIdleError, - OperatorGpuFreeError, ResourceControlEffect, ResourceControlRequest, ResourceReadModel, - open_action_id, -}; -use crate::submission::{RequestId, ResourceRoutePhase, SubmissionState}; - -/// Longest display name accepted for a registered resource -const DISPLAY_NAME_MAX_CHARS: usize = 120; -/// Time a control waits for its owner to apply a durable cancellation intent -const SETTLE_WAIT: Duration = Duration::from_secs(3); -const SETTLE_POLL: Duration = Duration::from_millis(100); -/// Time to read an actor snapshot before a view omits its transient attention -const INSPECT_TIMEOUT: Duration = Duration::from_secs(1); -/// Forwarded controls include the authority's settle wait and one notice attempt -const CONTROL_TIMEOUT: Duration = Duration::from_secs(30); - -/// Read routes shared by the Unix socket and the dashboard -pub(crate) fn read_routes() -> Router { - Router::new() - .route("/v1/resources", get(list)) - .route(RESOURCE_PENDING_PATH, get(pending)) - .route("/v1/resources/{id}", get(detail)) -} - -/// The three typed resource controls, shared by the socket and the dashboard origin -pub(crate) fn action_routes() -> Router { - Router::new().route("/v1/resources/{id}/actions", post(action)) -} - -/// Socket-only resource mutations for the CLI -pub(crate) fn socket_routes() -> Router { - Router::new() - .route(RESOURCE_REGISTER_PATH, post(register)) - .route("/v1/resources/{id}/supervisor", post(replace_supervisor)) - .route("/v1/resources/{id}/background", post(background)) - .route( - "/v1/resources/{id}/background/{task_id}/trainer-attempt", - post(trainer_attempt), - ) - .route("/v1/resources/{id}/requests", post(submit_request)) - .route( - "/v1/resources/{id}/requests/{request_id}/cancel", - post(cancel_request), - ) - .route( - "/v1/resources/{id}/operator-release", - post(operator_release), - ) - .route("/v1/resources/{id}/initial-idle", post(initial_idle)) - .merge(action_routes()) -} - -/// Fleet routes served by a resource authority -pub(super) fn cluster_routes() -> Router { - Router::new() - .route(CLUSTER_RESOURCES_PATH, get(cluster_list)) - .route(CLUSTER_RESOURCE_PENDING_PATH, get(cluster_pending)) - .route("/v1/cluster/resources/{id}", get(cluster_detail)) - .route("/v1/cluster/resources/{id}/control", post(cluster_control)) -} - -// ---- request parsing ---- - -/// JSON body extractor that reports malformed or unknown input in the AppError envelope -pub(crate) struct StrictJson(pub(crate) T); - -impl FromRequest for StrictJson -where - S: Send + Sync, - T: DeserializeOwned, -{ - type Rejection = AppError; - - async fn from_request(req: Request, state: &S) -> Result { - let bytes = Bytes::from_request(req, state) - .await - .map_err(|error| AppError::Usage { - message: format!("invalid request body: {error}"), - })?; - let value: Value = serde_json::from_slice(&bytes).map_err(|error| AppError::Usage { - message: format!("invalid JSON: {error}"), - })?; - serde_path_to_error::deserialize(&value) - .map(Self) - .map_err(|error| AppError::Usage { - message: format!( - "invalid request at `{}`: {}", - spec::json_pointer(error.path()), - error.inner() - ), - }) - } -} - -fn path_value(path: Result, PathRejection>) -> Result { - path.map(|Path(value)| value) - .map_err(|error| AppError::Usage { - message: format!("invalid path: {error}"), - }) -} - -fn check_version(api_version: u32) -> Result<(), AppError> { - if api_version == API_VERSION { - return Ok(()); - } - Err(AppError::Usage { - message: "unsupported API version".into(), - }) -} - -/// Thread whose pending actions are listed -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct PendingQuery { - machine: MachineId, - thread: ThreadId, -} - -impl PendingQuery { - fn address(&self) -> SupervisorAddress { - SupervisorAddress { - machine: self.machine, - thread: self.thread, - } - } -} - -/// Fleet read of pending actions, with the destination check fields -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct ClusterPendingQuery { - api_version: u32, - destination_machine: MachineId, - machine: MachineId, - thread: ThreadId, -} - -// ---- read handlers ---- - -async fn list(State(state): State) -> Result, AppError> { - let mut resources = local_overviews(&state).await?; - let mut unavailable_authorities = Vec::new(); - for (peer, result) in - read_peers::(&state, |_| format!("{CLUSTER_RESOURCES_PATH}?")).await - { - let checked = result.and_then(|body| { - let owned = body.machine == peer.machine - && body - .resources - .iter() - .all(|overview| overview.resource.authority_machine() == peer.machine); - if owned { - Ok(body.resources) - } else { - Err("peer returned resources owned by another authority".to_owned()) - } - }); - match checked { - Ok(mut remote) => resources.append(&mut remote), - Err(message) => unavailable_authorities.push(peer.unavailable(message)), - } - } - resources.sort_by(|left, right| { - left.resource - .display_name - .cmp(&right.resource.display_name) - .then(left.resource.id.as_uuid().cmp(&right.resource.id.as_uuid())) - }); - Ok(Json(ResourceList { - api_version: API_VERSION, - resources, - unavailable_authorities, - })) -} - -async fn detail( - State(state): State, - id: Result, PathRejection>, -) -> Result, AppError> { - let id = path_value(id)?; - if let Some(detail) = local_detail(&state, id).await? { - return Ok(Json(detail)); - } - remote_detail(&state, id).await.map(Json) -} - -async fn pending( - State(state): State, - query: Result, QueryRejection>, -) -> Result, AppError> { - let Query(query) = query.map_err(|error| AppError::Usage { - message: format!("invalid pending-action query: {error}"), - })?; - let address = query.address(); - let mut actions = local_pending(&state, address).await?; - let mut unavailable_authorities = Vec::new(); - for (peer, result) in read_peers::(&state, |_| { - format!( - "{CLUSTER_RESOURCE_PENDING_PATH}?machine={}&thread={}&", - address.machine, address.thread - ) - }) - .await - { - let checked = result.and_then(|body| { - let owned = body.machine == peer.machine - && body.actions.iter().all(|action| { - action.authority_machine == peer.machine && action.supervisor == address - }); - if owned { - Ok(body.actions) - } else { - Err("peer returned actions for another authority or thread".to_owned()) - } - }); - match checked { - Ok(mut remote) => actions.append(&mut remote), - Err(message) => unavailable_authorities.push(peer.unavailable(message)), - } - } - Ok(Json(PendingActionList { - api_version: API_VERSION, - actions, - unavailable_authorities, - })) -} - -// ---- mutation handlers ---- - -async fn action( - State(state): State, - id: Result, PathRejection>, - StrictJson(body): StrictJson, -) -> Result, AppError> { - let id = path_value(id)?; - check_version(body.api_version)?; - let control = ResourceControl::Action { - expected_revision: body.expected_revision, - operation_id: body.operation_id, - action: body.action, - }; - apply_control(&state, id, control).await.map(Json) -} - -async fn cancel_request( - State(state): State, - ids: Result, PathRejection>, - StrictJson(body): StrictJson, -) -> Result, AppError> { - let (id, request_id) = path_value(ids)?; - check_version(body.api_version)?; - let control = ResourceControl::Action { - expected_revision: body.expected_revision, - operation_id: body.operation_id, - action: BrowserResourceAction::CancelQueued { request_id }, - }; - apply_control(&state, id, control).await.map(Json) -} - -async fn replace_supervisor( - State(state): State, - id: Result, PathRejection>, - StrictJson(body): StrictJson, -) -> Result, AppError> { - let id = path_value(id)?; - check_version(body.api_version)?; - check_supervisor(&body.supervisor)?; - let control = ResourceControl::ReplaceSupervisor { - expected_revision: body.expected_revision, - supervisor: body.supervisor, - }; - apply_control(&state, id, control).await.map(Json) -} - -async fn register( - State(state): State, - StrictJson(body): StrictJson, -) -> Result, AppError> { - check_version(body.api_version)?; - let registration = body.spec; - let display_name = check_registration(®istration)?; - let resource = Resource::new( - registration.id, - display_name, - state.machine.identity.machine, - registration.supervisor, - AssignmentRevision::new(0), - ResourceRevision::new(0), - None, - ); - // the store compares a retry with the saved first registration, so an exact - // retry succeeds after a supervisor replacement and changed content conflicts - call(&state.supervisor, |reply| SupervisorMsg::RegisterResource { - resource: Box::new(resource), - reply, - }) - .await?; - local_detail(&state, registration.id) - .await? - .ok_or_else(|| AppError::Internal { - message: "registered resource is missing from the authority store".into(), - }) - .map(Json) -} - -fn check_registration(registration: &ResourceRegistration) -> Result { - if registration.id.as_uuid().is_nil() { - return Err(AppError::Usage { - message: "resource id must not be nil".into(), - }); - } - let name = registration.display_name.trim(); - if name.is_empty() - || name.chars().count() > DISPLAY_NAME_MAX_CHARS - || name.chars().any(char::is_control) - { - return Err(AppError::Usage { - message: format!( - "display_name must be 1 to {DISPLAY_NAME_MAX_CHARS} characters without control characters" - ), - }); - } - check_supervisor(®istration.supervisor)?; - Ok(name.to_owned()) -} - -fn check_supervisor(supervisor: &SupervisorAddress) -> Result<(), AppError> { - if supervisor.machine.as_uuid().is_nil() || supervisor.thread.0.is_nil() { - return Err(AppError::Usage { - message: "supervisor machine and thread must not be nil".into(), - }); - } - Ok(()) -} - -/// Envelope shared by background and request submissions; `spec` and `env` -/// keep their own pointer prefixes when they fail validation -#[derive(Deserialize)] -#[serde(deny_unknown_fields)] -struct ResourceSubmitEnvelope { - api_version: u32, - request_id: RequestId, - spec: Value, - env: Value, - callback_cwd: PathBuf, -} - -struct ResourceSubmitBody { - request_id: RequestId, - spec: NormalizedSpec, - env: TaskEnv, - callback_cwd: PathBuf, -} - -impl ResourceSubmitBody { - fn parse(value: &Value) -> Result { - let envelope: ResourceSubmitEnvelope = serde_path_to_error::deserialize(value) - .map_err(|error| super::api::invalid_at(value, "", &error))?; - check_version(envelope.api_version)?; - if envelope.request_id.0.is_nil() { - return Err(AppError::Usage { - message: "request_id must not be nil".into(), - }); - } - let env = serde_path_to_error::deserialize(&envelope.env) - .map_err(|error| super::api::invalid_at(&envelope.env, "/env", &error))?; - let spec = spec::parse_normalized_value(&envelope.spec) - .map_err(|error| super::api::prefix("/spec", error))?; - if spec.machine.is_some() { - return Err(AppError::InvalidSpec { - pointer: "/spec/machine".into(), - value: Value::Null, - message: "the resource authority fixes the execution machine".into(), - }); - } - if !envelope.callback_cwd.is_absolute() { - return Err(AppError::InvalidSpec { - pointer: "/callback_cwd".into(), - value: serde_json::to_value(&envelope.callback_cwd)?, - message: "callback directory must be absolute".into(), - }); - } - Ok(Self { - request_id: envelope.request_id, - spec, - env, - callback_cwd: envelope.callback_cwd, - }) - } -} - -/// Bind one first background launch through the authority's launch owner -/// -/// A co-located supervisor binds the launch directly on its own authority. A -/// supervisor on another machine saves its fixed-ID origin route here first and -/// then sends the launch to the fixed authority. A call on an authority whose -/// supervisor thread runs elsewhere writes nothing, since this machine cannot -/// own that thread's callbacks -async fn background( - State(state): State, - id: Result, PathRejection>, - StrictJson(value): StrictJson, -) -> Result, AppError> { - let resource = path_value(id)?; - let body = ResourceSubmitBody::parse(&value)?; - let request_id = body.request_id; - let local = state.machine.identity.machine; - let authority = background_authority(&state, resource, request_id).await?; - if authority != local { - let input = super::resource_background::RemoteBackgroundSubmit { - resource, - authority, - request_id, - spec: body.spec, - env: body.env, - callback_cwd: body.callback_cwd, - }; - return super::resource_background::submit(&state, input) - .await - .map(Json); - } - - let launch = BackgroundLaunch { - resource_id: resource, - request_id, - spec: body.spec, - env: body.env, - callback_cwd: body.callback_cwd, - }; - let unknown = |message: String| AppError::ResourceOutcomeUnknown { - resource, - operation: Some(request_id.0), - message: format!("{message}; retry with the same request id {}", request_id.0), - }; - let acceptance = call(&state.supervisor, |reply| SupervisorMsg::LaunchBackground { - launch: Box::new(launch), - reply, - }) - .await - .map_err(|error| unknown(format!("the launch owner did not answer: {error}")))? - .map_err(|error| background_error(resource, request_id, error))?; - let (task_id, outcome) = match acceptance { - BackgroundLaunchAcceptance::Inserted { task, .. } => { - (task, ResourceBackgroundSubmitOutcome::Inserted) - } - BackgroundLaunchAcceptance::Existing { task, state } => { - (task, ResourceBackgroundSubmitOutcome::Existing { state }) - } - BackgroundLaunchAcceptance::UnsupportedRemoteSupervisor { supervisor, .. } => { - return Err(AppError::ResourceOperationUnavailable { - resource, - message: format!( - "the supervisor thread runs on machine {}; submit the background launch \ - from that machine, which owns its callback route; nothing was written", - supervisor.machine - ), - }); - } - }; - let saved = local_resource(&state, resource).await.map_err(|error| { - unknown(format!( - "the launch committed but its resource read failed: {error}" - )) - })?; - - Ok(Json(ResourceBackgroundSubmitResponse { - api_version: API_VERSION, - request_id, - task_id, - resource: saved, - outcome, - })) -} - -/// Authority for a background launch; a retry reuses the saved remote route so a -/// lost response does not depend on Fleet discovery -async fn background_authority( - state: &AppState, - resource: ResourceId, - request: RequestId, -) -> Result { - let saved = call(&state.store, |reply| StoreMsg::OriginRouteByRequest { - request, - reply, - }) - .await?; - match saved.map(|route| route.submission) { - Some(SubmissionState::ResourceBackground { binding, .. }) - if binding.assignment.resource_id == resource => - { - Ok(binding.execution_machine()) - } - // a co-located launch saves an ordinary accepted route on its authority - Some(SubmissionState::Accepted) | None => locate_authority(state, resource).await, - Some(_) => Err(AppError::ResourceOperationConflict { - resource, - operation: Some(request.0), - message: "request id belongs to another resource or route".into(), - }), - } -} - -fn background_error( - resource: ResourceId, - request_id: RequestId, - error: BackgroundLaunchError, -) -> AppError { - use BackgroundLaunchError as Error; - let not_allowed = |message: String| AppError::ResourceActionNotAllowed { resource, message }; - let conflict = |message: String| AppError::ResourceOperationConflict { - resource, - operation: Some(request_id.0), - message, - }; - match error { - Error::Resource(crate::resource::store::ResourceStoreError::ResourceNotFound) => { - AppError::ResourceNotFound { resource } - } - error @ (Error::NotSupervisorThread { .. } - | Error::ActiveLoan { .. } - | Error::QueuedWorkAhead { .. } - | Error::BackgroundTaskActive { .. } - | Error::BackgroundTaskMissing { .. } - | Error::LaunchPending { .. } - | Error::PredecessorReleaseUnproven { .. }) => not_allowed(error.to_string()), - error @ (Error::ConflictingRetry { .. } | Error::IdentityConflict { .. }) => { - conflict(error.to_string()) - } - Error::UnsupportedCommand(error) => AppError::ResourceOperationUnavailable { - resource, - message: format!( - "a background launch accepts only the maintained direct-segment trainer, whose \ - ownership lock release proof can verify; this command does not match it: {error}" - ), - }, - // the spec or executor environment is invalid; nothing was written - Error::TaskRecords( - error @ (AppError::InvalidCwd { .. } - | AppError::ExecutableMissing { .. } - | AppError::InvalidSpec { .. } - | AppError::Usage { .. }), - ) => error, - error => AppError::ResourceOutcomeUnknown { - resource, - operation: Some(request_id.0), - message: format!( - "the launch owner failed: {error}; retry with the same request id {}", - request_id.0 - ), - }, - } -} - -pub(super) async fn local_resource( - state: &AppState, - resource: ResourceId, -) -> Result { - let snapshots = call(&state.store, |reply| { - StoreMsg::ResourceSnapshotsForAuthority { - authority_machine: state.machine.identity.machine, - reply, - } - }) - .await?; - snapshots - .into_iter() - .find(|snapshot| snapshot.resource.id == resource) - .map(|snapshot| snapshot.resource) - .ok_or(AppError::ResourceNotFound { resource }) -} - -/// Save one authority-local operator confirmation, then wake its resource owner -async fn operator_release( - State(state): State, - path: Result, PathRejection>, - StrictJson(body): StrictJson, -) -> Result, AppError> { - let resource = path_value(path)?; - check_version(body.api_version)?; - let operation_id = body.attestation.operation_id; - let operation = operation_id.as_uuid(); - if body.attestation.resource_id != resource { - return Err(AppError::Usage { - message: "path resource id does not match the attestation".into(), - }); - } - - let authority = state.machine.identity.machine; - if body.attestation.authority_machine != authority { - return Err(AppError::ResourceActionNotAllowed { - resource, - message: format!( - "operator attestation names authority {}, but this daemon is {authority}", - body.attestation.authority_machine - ), - }); - } - body.attestation - .validate() - .map_err(|error| operator_release_refusal(resource, operation, error))?; - - let resolution = call(&state.store, |reply| { - StoreMsg::AttestTrainerGpuFreeForAuthority { - authority_machine: authority, - attestation: Box::new(body.attestation), - reply, - } - }) - .await - .map_err(|error| { - operator_release_unknown( - resource, - operation, - format!("the store actor did not answer: {error}"), - ) - })? - .map_err(|error| match error { - OperatorGpuFreeError::Refused(refusal) => { - operator_release_refusal(resource, operation, refusal) - } - error => operator_release_unknown( - resource, - operation, - format!("the authority could not confirm the transaction result: {error}"), - ), - })?; - - call(&state.supervisor, |reply| { - SupervisorMsg::ReconcileResource { - id: resource, - reply, - } - }) - .await - .map_err(|error| { - operator_release_unknown( - resource, - operation, - format!("the receipt is saved, but resource reconciliation did not complete: {error}"), - ) - })?; - - Ok(Json(OperatorReleaseResponse { - api_version: API_VERSION, - receipt: resolution.receipt, - replayed: resolution.replayed, - })) -} - -/// Save one authority-local initial idle attestation, then wake its resource owner -async fn initial_idle( - State(state): State, - path: Result, PathRejection>, - StrictJson(body): StrictJson, -) -> Result, AppError> { - let resource = path_value(path)?; - check_version(body.api_version)?; - let operation = body.attestation.operation_id.as_uuid(); - if body.attestation.resource_id != resource { - return Err(AppError::Usage { - message: "path resource id does not match the attestation".into(), - }); - } - let authority = state.machine.identity.machine; - if body.attestation.authority_machine != authority { - return Err(AppError::ResourceActionNotAllowed { - resource, - message: format!( - "initial idle attestation names authority {}, but this daemon is {authority}", - body.attestation.authority_machine - ), - }); - } - body.attestation - .validate() - .map_err(|error| initial_idle_refusal(resource, operation, error))?; - - let resolution = call(&state.store, |reply| { - StoreMsg::AttestInitialIdleForAuthority { - authority_machine: authority, - attestation: Box::new(body.attestation), - reply, - } - }) - .await - .map_err(|error| { - operator_release_unknown( - resource, - operation, - format!("the store actor did not answer: {error}"), - ) - })? - .map_err(|error| match error { - InitialIdleError::Refused(refusal) => initial_idle_refusal(resource, operation, refusal), - error => operator_release_unknown( - resource, - operation, - format!("the authority could not confirm the transaction result: {error}"), - ), - })?; - - // queued work serves from the new idle boundary through the normal reconciliation - call(&state.supervisor, |reply| { - SupervisorMsg::ReconcileResource { - id: resource, - reply, - } - }) - .await - .map_err(|error| { - operator_release_unknown( - resource, - operation, - format!("the receipt is saved, but resource reconciliation did not complete: {error}"), - ) - })?; - - Ok(Json(InitialIdleResponse { - api_version: API_VERSION, - receipt: resolution.receipt, - replayed: resolution.replayed, - })) -} - -fn initial_idle_refusal( - resource: ResourceId, - operation: Uuid, - refusal: InitialIdleRefusal, -) -> AppError { - use InitialIdleRefusal as Refusal; - - match refusal { - Refusal::InvalidIdentity | Refusal::EmptyObservation => AppError::Usage { - message: refusal.to_string(), - }, - Refusal::ResourceNotFound => AppError::ResourceNotFound { resource }, - Refusal::StaleRevision { expected, actual } => AppError::ResourceStaleRevision { - resource, - expected: expected.get(), - current: actual.get(), - }, - Refusal::ConflictingRetry { .. } | Refusal::RevisionExhausted { .. } => { - AppError::ResourceOperationConflict { - resource, - operation: Some(operation), - message: refusal.to_string(), - } - } - Refusal::WrongAuthority { .. } - | Refusal::HistoryExists { .. } - | Refusal::AlreadyAttested { .. } => AppError::ResourceActionNotAllowed { - resource, - message: refusal.to_string(), - }, - } -} - -fn operator_release_refusal( - resource: ResourceId, - operation: Uuid, - refusal: OperatorGpuFreeRefusal, -) -> AppError { - use OperatorGpuFreeRefusal as Refusal; - - match refusal { - Refusal::InvalidIdentity | Refusal::EmptyObservation => AppError::Usage { - message: refusal.to_string(), - }, - Refusal::ResourceNotFound => AppError::ResourceNotFound { resource }, - Refusal::ConflictingRetry { .. } => AppError::ResourceOperationConflict { - resource, - operation: Some(operation), - message: refusal.to_string(), - }, - Refusal::StaleRevision { expected, actual } => AppError::ResourceStaleRevision { - resource, - expected: expected.get(), - current: actual.get(), - }, - Refusal::LoanStateChanged { .. } | Refusal::RevisionExhausted { .. } => { - AppError::ResourceOperationConflict { - resource, - operation: Some(operation), - message: refusal.to_string(), - } - } - refusal => AppError::ResourceActionNotAllowed { - resource, - message: refusal.to_string(), - }, - } -} - -fn operator_release_unknown(resource: ResourceId, operation: Uuid, detail: String) -> AppError { - AppError::ResourceOutcomeUnknown { - resource, - operation: Some(operation), - message: format!( - "operator attestation outcome is unknown: {detail}; retry with the same operation id {operation}" - ), - } -} - -/// Bind the supervisor-named trainer attempt to the registered background task -/// -/// The trainer creates its attempt and ownership lock only after it starts, so -/// this runs after the resource owner registered the task on its confirmed start -/// The authority reads the attempt request and probes the held lock itself. An -/// exact saved association is returned without a new probe -async fn trainer_attempt( - State(state): State, - path: Result, PathRejection>, - StrictJson(body): StrictJson, -) -> Result, AppError> { - let (resource, task_id) = path_value(path)?; - check_version(body.api_version)?; - // a remote authority verifies the attempt from its own runtime evidence - if locate_authority(&state, resource).await? != state.machine.identity.machine { - return super::resource_background::bind_trainer_attempt( - &state, - resource, - task_id, - body.attempt_binding, - ) - .await - .map(Json); - } - authority_trainer_attempt(&state, resource, task_id, body.attempt_binding) - .await - .map(Json) -} - -/// Bind or replay one trainer attempt association on this authority -/// -/// The task must be the registered background task and must hold its lock. An -/// exact saved association is returned without a new probe; another attempt -/// for the same task conflicts -pub(super) async fn authority_trainer_attempt( - state: &AppState, - resource: ResourceId, - task_id: TaskId, - attempt_binding: AttemptBinding, -) -> Result { - let authority = state.machine.identity.machine; - let not_allowed = |message: String| AppError::ResourceActionNotAllowed { resource, message }; - let saved = call(&state.store, |reply| { - StoreMsg::TrainerAttemptAssociationForTaskForAuthority { - authority_machine: authority, - task_id, - reply, - } - }) - .await? - .map_err(|error| not_allowed(format!("trainer association cannot be read: {error}")))?; - let association = match saved { - Some(saved) if saved.resource_id() == resource => { - if saved.verified_attempt().binding() != &attempt_binding { - return Err(AppError::ResourceOperationConflict { - resource, - operation: None, - message: format!("task {task_id} is associated with another trainer attempt"), - }); - } - saved - } - Some(_) => { - return Err(not_allowed(format!( - "task {task_id} belongs to another resource" - ))); - } - None => bind_trainer_attempt(state, resource, task_id, attempt_binding).await?, - }; - let evidence = association.verified_attempt(); - - Ok(TrainerAttemptResponse { - api_version: API_VERSION, - resource: local_resource(state, resource).await?, - task_id, - runtime_root: evidence.canonical_runtime_root().to_path_buf(), - attempt_binding: evidence.binding().clone(), - }) -} - -async fn bind_trainer_attempt( - state: &AppState, - resource: ResourceId, - task_id: TaskId, - binding: AttemptBinding, -) -> Result { - let authority = state.machine.identity.machine; - let not_allowed = |message: String| AppError::ResourceActionNotAllowed { resource, message }; - let identity = call(&state.store, |reply| StoreMsg::ExecutorIdentity { - id: task_id, - reply, - }) - .await?; - let Some(crate::submission::ExecutorIdentity::Accepted(record)) = identity else { - return Err(not_allowed(format!( - "task {task_id} has no accepted identity" - ))); - }; - let runtime_root = match record.current_spec().map(|spec| &spec.workload) { - Some(crate::spec::NormalizedWorkload::Task(workload)) => { - crate::resource::command_shape::direct_segment_runtime_root(&workload.command) - } - _ => None, - } - .ok_or_else(|| { - not_allowed(format!( - "task {task_id} is not a direct-segment trainer command" - )) - })?; - let verified = tokio::task::spawn_blocking(move || { - crate::resource::ownership_lock::build_trainer_attempt_registration_evidence( - &runtime_root, - &binding, - ) - }) - .await - .map_err(|error| AppError::Internal { - message: format!("trainer attempt probe failed: {error}"), - })? - .map_err(|error| { - not_allowed(format!( - "trainer attempt evidence is not available: {error}" - )) - })?; - call(&state.store, |reply| { - StoreMsg::BindTrainerAttemptAssociationForAuthority { - authority_machine: authority, - resource_id: resource, - task_id, - verified_attempt: Box::new(verified), - reply, - } - }) - .await? - .map_err(|error| not_allowed(format!("trainer attempt association refused: {error}"))) -} - -async fn submit_request( - State(state): State, - id: Result, PathRejection>, - StrictJson(value): StrictJson, -) -> Result, AppError> { - let resource = path_value(id)?; - let body = ResourceSubmitBody::parse(&value)?; - let authority = request_authority(&state, resource, body.request_id).await?; - let spec = body.spec; - let input = ResourceSubmitInput { - request: body.request_id, - resource, - authority, - spec: spec.clone(), - env: body.env, - callback_cwd: body.callback_cwd, - }; - let (task_id, outcome) = match super::resource_submit::submit(&state, input).await? { - ResourceSubmitOutcome::Rejected { reason, .. } - if let Some(rejection) = crate::spec::HostInputRejection::parse(&reason) => - { - return Err(host_input_error(&state, authority, &spec, rejection)); - } - ResourceSubmitOutcome::Waiting { task } => (task, ResourceRequestSubmitOutcome::Waiting), - ResourceSubmitOutcome::Activated { task } => { - (task, ResourceRequestSubmitOutcome::Activated) - } - ResourceSubmitOutcome::Rejected { task, reason } => { - (task, ResourceRequestSubmitOutcome::Rejected { reason }) - } - ResourceSubmitOutcome::CancelledBeforeLaunch { task } => { - (task, ResourceRequestSubmitOutcome::CancelledBeforeLaunch) - } - }; - Ok(Json(ResourceRequestSubmitResponse { - api_version: API_VERSION, - request_id: body.request_id, - task_id, - resource_id: resource, - authority_machine: authority, - outcome, - })) -} - -/// Typed error for an authority's refusal of the spec's host inputs -/// -/// A local authority checks this file system again for the precise error; a -/// remote refusal is rebuilt from the spec -fn host_input_error( - state: &AppState, - authority: MachineId, - spec: &crate::spec::NormalizedSpec, - rejection: crate::spec::HostInputRejection, -) -> AppError { - if authority == state.machine.identity.machine - && let Err(error) = crate::spec::check_spec_host(spec) - { - return error.into(); - } - rejection.into_error(spec) -} - -/// Authority for a request submission; a retry reuses the saved route so a lost -/// response does not depend on Fleet discovery -async fn request_authority( - state: &AppState, - resource: ResourceId, - request: RequestId, -) -> Result { - let saved = call(&state.store, |reply| StoreMsg::OriginRouteByRequest { - request, - reply, - }) - .await?; - if let Some(route) = saved { - return match route.submission { - SubmissionState::Resource { - resource: saved_resource, - .. - } if saved_resource == resource => Ok(route.execution_machine), - _ => Err(AppError::SubmissionConflict { - request, - task: route.task, - message: "request UUID belongs to another resource or route".into(), - }), - }; - } - locate_authority(state, resource).await -} - -// ---- authority controls ---- - -/// Apply one control on the fixed authority and return its authoritative detail -async fn apply_control( - state: &AppState, - resource: ResourceId, - control: ResourceControl, -) -> Result { - let authority = locate_authority(state, resource).await?; - if authority == state.machine.identity.machine { - return authority_control(state, resource, control).await; - } - forward_control(state, authority, resource, control).await -} - -async fn authority_control( - state: &AppState, - resource: ResourceId, - control: ResourceControl, -) -> Result { - match control { - ResourceControl::Action { - expected_revision, - operation_id, - action, - } => { - let request = ResourceControlRequest { - resource_id: resource, - expected_revision, - action, - }; - authority_action(state, operation_id, request).await - } - ResourceControl::ReplaceSupervisor { - expected_revision, - supervisor, - } => { - call(&state.store, |reply| StoreMsg::ReplaceResourceSupervisor { - authority_machine: state.machine.identity.machine, - resource_id: resource, - expected_revision, - supervisor, - reply, - }) - .await?; - // the actor refreshes its snapshot and later deliveries use the new route - if let Err(error) = call(&state.supervisor, |reply| { - SupervisorMsg::ReconcileResource { - id: resource, - reply, - } - }) - .await - { - warn!(resource = %resource.as_uuid(), "resource reconcile after supervisor replacement: {error}"); - } - required_local_detail(state, resource).await - } - } -} - -async fn authority_action( - state: &AppState, - operation_id: Uuid, - request: ResourceControlRequest, -) -> Result { - if operation_id.is_nil() { - return Err(AppError::Usage { - message: "operation_id must not be nil".into(), - }); - } - let resource = request.resource_id; - let start = call(&state.store, |reply| StoreMsg::BeginResourceControl { - authority_machine: state.machine.identity.machine, - operation_id, - request: Box::new(request), - attempt_id: DeliveryAttemptId::new(), - reply, - }) - .await?; - let unknown = |message: String| AppError::ResourceOutcomeUnknown { - resource, - operation: Some(operation_id), - message, - }; - match start.effect { - ResourceControlEffect::CancelQueued { request } => { - let mut settlement = - queued_cancel_settlement(state, resource, request.request_id).await; - if settlement == QueuedCancelSettlement::Pending { - request_origin_cancellation(state, &request, ResourceRoutePhase::Waiting) - .await - .map_err(|error| unknown(format!("origin cancellation: {error}")))?; - settlement = wait_for_queued_cancel(state, resource, request.request_id).await; - } - match settlement { - QueuedCancelSettlement::Cancelled => {} - QueuedCancelSettlement::NoLongerQueued => { - return Err(AppError::ResourceActionNotAllowed { - resource, - message: format!( - "request {} left the queue before cancellation settled; inspect its task before requesting a stop", - request.request_id.0 - ), - }); - } - QueuedCancelSettlement::Pending => { - return Err(unknown( - "queued cancellation has not settled at the resource authority".into(), - )); - } - } - } - ResourceControlEffect::Reordered => {} - ResourceControlEffect::StopActive { request } => { - request_origin_cancellation(state, &request, ResourceRoutePhase::Activated) - .await - .map_err(|error| unknown(format!("origin cancellation: {error}")))?; - if !wait_until(|| task_stop_recorded(state, request.task_id)).await { - return Err(unknown( - "active task cancellation has not settled at the executor".into(), - )); - } - } - ResourceControlEffect::Renotify { - notice_id, - attempt_id, - } if !start.replayed => { - // the spawned attempt settles even when the HTTP caller disconnects - let delivery_state = state.clone(); - let delivery = tokio::spawn(async move { - super::resource_notice_sender::deliver_reserved_attempt( - &delivery_state, - notice_id, - attempt_id, - ) - .await - }); - match delivery.await { - Ok(Ok(_)) => {} - Ok(Err(error)) => { - return Err(unknown(format!( - "renotify attempt was not settled: {error}" - ))); - } - Err(error) => { - return Err(unknown(format!("renotify attempt worker failed: {error}"))); - } - } - } - ResourceControlEffect::Renotify { .. } => {} - } - required_local_detail(state, resource).await -} - -/// Ask the request's origin to save its cancellation intent -/// -/// The origin owns the callback route and delivers the cancellation to this -/// authority through the existing restart-safe delivery path -async fn request_origin_cancellation( - state: &AppState, - request: &ResourceRequest, - phase: ResourceRoutePhase, -) -> Result<(), AppError> { - if request.origin_machine == state.machine.identity.machine { - return match super::cluster::origin_resource_cancellation_intent(state, request.task_id) - .await? - { - OriginResourceCancellationOutcome::Intent { .. } - | OriginResourceCancellationOutcome::AlreadyCancelled => Ok(()), - }; - } - let target = ResourceCancellationTarget { - request_id: request.request_id, - task_id: request.task_id, - resource_id: request.resource_id, - origin_machine: request.origin_machine, - authority_machine: state.machine.identity.machine, - phase, - }; - super::cluster::forward_resource_cancellation_intent(state, &target) - .await - .map(|_| ()) -} - -/// Poll the owner's durable result for a bounded time -async fn wait_until(mut settled: F) -> bool -where - F: FnMut() -> Fut, - Fut: Future, -{ - let deadline = Instant::now() + SETTLE_WAIT; - while Instant::now() < deadline { - if settled().await { - return true; - } - tokio::time::sleep(SETTLE_POLL).await; - } - false -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -enum QueuedCancelSettlement { - Pending, - Cancelled, - NoLongerQueued, -} - -async fn wait_for_queued_cancel( - state: &AppState, - resource: ResourceId, - request: RequestId, -) -> QueuedCancelSettlement { - let deadline = Instant::now() + SETTLE_WAIT; - while Instant::now() < deadline { - let settlement = queued_cancel_settlement(state, resource, request).await; - if settlement != QueuedCancelSettlement::Pending { - return settlement; - } - tokio::time::sleep(SETTLE_POLL).await; - } - QueuedCancelSettlement::Pending -} - -async fn queued_cancel_settlement( - state: &AppState, - resource: ResourceId, - request: RequestId, -) -> QueuedCancelSettlement { - let requests = call(&state.store, |reply| StoreMsg::ResourceRequests { - authority_machine: state.machine.identity.machine, - resource_id: resource, - reply, - }) - .await; - let Ok(requests) = requests else { - return QueuedCancelSettlement::Pending; - }; - match requests.iter().find(|saved| saved.request_id == request) { - Some(saved) => match saved.state { - ResourceRequestState::Queued => QueuedCancelSettlement::Pending, - ResourceRequestState::CancelledBeforeLaunch => QueuedCancelSettlement::Cancelled, - ResourceRequestState::Assigned { .. } - | ResourceRequestState::Finished { .. } - | ResourceRequestState::Rejected { .. } => QueuedCancelSettlement::NoLongerQueued, - }, - None => QueuedCancelSettlement::Pending, - } -} - -async fn task_stop_recorded(state: &AppState, task: TaskId) -> bool { - let row = call(&state.store, |reply| StoreMsg::GetTask { id: task, reply }).await; - row.ok() - .flatten() - .is_some_and(|row| row.cancel_requested_at.is_some() || row.status().is_terminal()) -} - -// ---- forwarding ---- - -async fn forward_control( - state: &AppState, - authority: MachineId, - resource: ResourceId, - control: ResourceControl, -) -> Result { - let operation = match &control { - ResourceControl::Action { operation_id, .. } => Some(*operation_id), - ResourceControl::ReplaceSupervisor { .. } => None, - }; - let unavailable = |message: String| AppError::ResourceAuthorityUnavailable { - resource, - machine: authority, - message, - }; - let fleet = state - .fleet - .handle() - .ok_or_else(|| unavailable("fleet is disabled".into()))?; - // nothing was sent when the authority cannot be reached - let destination = fleet - .connect(authority) - .await - .map_err(|error| unavailable(error.to_string()))?; - let body = ClusterResourceControl { - api_version: API_VERSION, - protocol_version: destination.protocol.0, - destination_machine: authority, - source_machine: state.machine.identity.machine, - resource_id: resource, - control, - }; - let path = format!("/v1/cluster/resources/{}/control", resource.as_uuid()); - let response = ClusterClient::new(CONTROL_TIMEOUT, READ_MAX_BODY) - .post_json(&destination.address, &path, &body) - .await - .map_err(|error| AppError::ResourceOutcomeUnknown { - resource, - operation, - message: error.to_string(), - })?; - decode_control_response(resource, authority, operation, &response) -} - -fn decode_control_response( - resource: ResourceId, - authority: MachineId, - operation: Option, - response: &ClusterResponse, -) -> Result { - let unknown = |message: String| AppError::ResourceOutcomeUnknown { - resource, - operation, - message, - }; - if response.status == StatusCode::OK { - let detail: ResourceDetail = serde_json::from_slice(&response.body) - .map_err(|error| unknown(format!("invalid authority response: {error}")))?; - if detail.api_version != API_VERSION - || detail.resource.id != resource - || detail.resource.authority_machine() != authority - { - return Err(unknown( - "authority response names another resource or authority".into(), - )); - } - return Ok(detail); - } - let error = remote_error(resource, operation, &response.body); - match error { - // a definitive authority refusal keeps its typed code - Some(error) if response.status.is_client_error() => Err(error), - _ => Err(unknown(format!( - "authority returned HTTP {}", - response.status - ))), - } -} - -#[derive(Deserialize)] -struct RemoteErrorEnvelope { - api_version: u32, - error: RemoteError, -} - -#[derive(Deserialize)] -struct RemoteError { - code: String, - message: String, - #[serde(default)] - input: Value, -} - -/// Rebuild a typed refusal from an authority's AppError envelope -fn remote_error(resource: ResourceId, operation: Option, body: &[u8]) -> Option { - let envelope: RemoteErrorEnvelope = serde_json::from_slice(body).ok()?; - if envelope.api_version != API_VERSION { - return None; - } - let error = envelope.error; - let detail = error - .input - .get("message") - .and_then(Value::as_str) - .map_or_else(|| error.message.clone(), str::to_owned); - let revision = |key: &str| error.input.get(key).and_then(Value::as_u64); - Some(match error.code.as_str() { - "resource_stale_revision" => AppError::ResourceStaleRevision { - resource, - expected: revision("expected_revision")?, - current: revision("current_revision")?, - }, - "resource_operation_conflict" => AppError::ResourceOperationConflict { - resource, - operation, - message: detail, - }, - "resource_not_found" => AppError::ResourceNotFound { resource }, - "resource_operation_unavailable" => AppError::ResourceOperationUnavailable { - resource, - message: detail, - }, - _ => AppError::ResourceActionNotAllowed { - resource, - message: format!("authority refused the control ({}): {detail}", error.code), - }, - }) -} - -// ---- Fleet location and reads ---- - -#[derive(Clone)] -struct PeerRef { - machine: MachineId, - name: String, -} - -impl PeerRef { - fn unavailable(&self, message: String) -> UnavailableAuthority { - UnavailableAuthority { - machine: self.machine, - name: Some(self.name.clone()), - message, - } - } -} - -async fn peers(state: &AppState) -> Vec { - let Some(fleet) = state.fleet.handle() else { - return Vec::new(); - }; - let local = state.machine.identity.machine; - let mut seen = BTreeSet::from([local]); - fleet - .peers() - .await - .into_iter() - .filter(|peer| seen.insert(peer.machine)) - .map(|peer| PeerRef { - machine: peer.machine, - name: peer.name.to_string(), - }) - .collect() -} - -/// Read one cluster route from every known peer in parallel -/// -/// `path` returns the route and any query prefix ending in `?` or `&`; the -/// destination check parameters are appended here -async fn read_peers( - state: &AppState, - path: impl Fn(&PeerRef) -> String, -) -> Vec<(PeerRef, Result)> -where - T: DeserializeOwned + Send + 'static, -{ - let Some(fleet) = state.fleet.handle() else { - return Vec::new(); - }; - let mut jobs = JoinSet::new(); - for peer in peers(state).await { - let fleet = fleet.clone(); - let path = format!( - "{}api_version={API_VERSION}&destination_machine={}", - path(&peer), - peer.machine - ); - jobs.spawn(async move { - let result = read_peer::(&fleet, peer.machine, &path).await; - (peer, result) - }); - } - let mut results = Vec::new(); - while let Some(joined) = jobs.join_next().await { - match joined { - Ok(result) => results.push(result), - Err(error) => warn!("resource peer read worker failed: {error}"), - } - } - results -} - -/// Read one resource from the peer that owns it -pub(super) async fn remote_detail( - state: &AppState, - resource: ResourceId, -) -> Result { - let mut found = Vec::new(); - let mut unchecked = Vec::new(); - for (peer, result) in read_peers::(state, |_| { - format!("/v1/cluster/resources/{}?", resource.as_uuid()) - }) - .await - { - match result { - Ok(body) if body.machine != peer.machine => unchecked.push(peer.machine), - Ok(ClusterResourceDetail { - detail: Some(detail), - .. - }) if detail.resource.id == resource - && detail.resource.authority_machine() == peer.machine => - { - found.push(detail); - } - Ok(ClusterResourceDetail { detail: None, .. }) => {} - Ok(_) | Err(_) => unchecked.push(peer.machine), - } - } - match found.len() { - 1 => Ok(found.remove(0)), - 0 if unchecked.is_empty() => Err(AppError::ResourceNotFound { resource }), - 0 => Err(AppError::ResourceLookupIncomplete { - resource, - unchecked, - }), - _ => Err(AppError::Internal { - message: format!( - "more than one authority claims resource {}", - resource.as_uuid() - ), - }), - } -} - -/// Fixed authority for a resource, from local ownership or the one peer that owns it -async fn locate_authority(state: &AppState, resource: ResourceId) -> Result { - if !local_models(state, Some(resource)).await?.is_empty() { - return Ok(state.machine.identity.machine); - } - remote_detail(state, resource) - .await - .map(|detail| detail.resource.authority_machine()) -} - -// ---- cluster handlers ---- - -fn check_read(state: &AppState, api_version: u32, destination: MachineId) -> Result<(), AppError> { - state.machine.identity.check_destination(destination)?; - check_version(api_version) -} - -async fn cluster_list( - State(state): State, - Query(query): Query, -) -> Result, AppError> { - check_read(&state, query.api_version, query.destination_machine)?; - Ok(Json(ClusterResourceList { - api_version: API_VERSION, - machine: state.machine.identity.machine, - resources: local_overviews(&state).await?, - })) -} - -async fn cluster_detail( - State(state): State, - Path(id): Path, - Query(query): Query, -) -> Result, AppError> { - check_read(&state, query.api_version, query.destination_machine)?; - Ok(Json(ClusterResourceDetail { - api_version: API_VERSION, - machine: state.machine.identity.machine, - detail: local_detail(&state, id).await?, - })) -} - -async fn cluster_pending( - State(state): State, - Query(query): Query, -) -> Result, AppError> { - check_read(&state, query.api_version, query.destination_machine)?; - let address = SupervisorAddress { - machine: query.machine, - thread: query.thread, - }; - Ok(Json(ClusterPendingActions { - api_version: API_VERSION, - machine: state.machine.identity.machine, - actions: local_pending(&state, address).await?, - })) -} - -async fn cluster_control( - State(state): State, - Path(id): Path, - Json(body): Json, -) -> Result, AppError> { - state - .machine - .identity - .check_destination(body.destination_machine)?; - super::cluster::check_api_version(body.api_version)?; - super::cluster::check_protocol(body.protocol_version, body.source_machine)?; - if body.resource_id != id { - return Err(AppError::Usage { - message: "resource path and body identities differ".into(), - }); - } - if let ResourceControl::ReplaceSupervisor { supervisor, .. } = &body.control { - check_supervisor(supervisor)?; - } - authority_control(&state, id, body.control).await.map(Json) -} - -// ---- local authority views ---- - -async fn local_models( - state: &AppState, - resource: Option, -) -> Result, AppError> { - call(&state.store, |reply| StoreMsg::ResourceReadModels { - authority_machine: state.machine.identity.machine, - resource_id: resource, - reply, - }) - .await -} - -async fn required_local_detail( - state: &AppState, - resource: ResourceId, -) -> Result { - local_detail(state, resource) - .await? - .ok_or(AppError::ResourceNotFound { resource }) -} - -async fn local_overviews(state: &AppState) -> Result, AppError> { - let models = local_models(state, None).await?; - let current: Vec = models.iter().filter_map(current_task_id).collect(); - let tasks = task_summaries(state, current).await?; - let mut overviews = Vec::with_capacity(models.len()); - for model in models { - let inspection = inspect(state, model.resource.id).await; - overviews.push(ResourceOverview { - queued_count: model - .requests - .iter() - .filter(|request| request.state == ResourceRequestState::Queued) - .count() as u64, - current_task: current_task_id(&model).and_then(|id| tasks.get(&id).cloned()), - attention: attention(&model, inspection.as_ref()), - background_launch: background_launch_reservation(&model), - return_window: model.return_window, - resource: model.resource, - loan: model.loan, - }); - } - Ok(overviews) -} - -async fn local_detail( - state: &AppState, - resource: ResourceId, -) -> Result, AppError> { - let Some(model) = local_models(state, Some(resource)).await?.pop() else { - return Ok(None); - }; - let current_task_id = current_task_id(&model); - let background_task_id = model.resource.registered_background_task; - let tasks = task_summaries( - state, - current_task_id - .into_iter() - .chain(background_task_id) - .collect(), - ) - .await?; - let inspection = inspect(state, resource).await; - let attention = attention(&model, inspection.as_ref()); - let background_launch = background_launch_reservation(&model); - Ok(Some(ResourceDetail { - background_launch, - return_execution_mode: model.return_execution_mode, - return_window: model.return_window, - api_version: API_VERSION, - requests: model.requests.iter().map(request_view).collect(), - current_task: current_task_id.and_then(|id| tasks.get(&id).cloned()), - background_task: background_task_id.and_then(|id| tasks.get(&id).cloned()), - current_task_id, - background_task_id, - attention, - resource: model.resource, - loan: model.loan, - notices: model.notices, - })) -} - -async fn local_pending( - state: &AppState, - supervisor: SupervisorAddress, -) -> Result, AppError> { - let models = local_models(state, None).await?; - Ok(models - .iter() - .filter(|model| model.resource.supervisor == supervisor) - .filter_map(pending_action) - .collect()) -} - -/// Open action of one resource, derived from its non-closed loan and notices -fn pending_action(model: &ResourceReadModel) -> Option { - let loan = model.loan.as_ref()?; - let action_id = open_action_id(loan)?; - let (phase, return_context) = match &loan.state { - LoanState::Active { phase } => pending_phase(phase)?, - LoanState::NeedsAttention { - last_safe_phase, - reason, - .. - } => ( - PendingActionPhase::AttentionRequired { - reason: reason.clone(), - last_safe_phase: Box::new(last_safe_phase.clone()), - }, - phase_return_context(last_safe_phase), - ), - LoanState::Closed { .. } => return None, - }; - Some(PendingActionView { - resource_id: model.resource.id, - authority_machine: model.resource.authority_machine(), - loan_id: loan.id, - action_id, - state_revision: model.resource.state_revision, - supervisor: model.resource.supervisor, - assignment_revision: model.resource.assignment_revision, - phase, - return_context, - notice: model - .notices - .iter() - .find(|notice| notice.action_id == action_id) - .cloned(), - return_window: model - .return_window - .filter(|window| window.action_id() == action_id), - }) -} - -fn pending_phase(phase: &LoanPhase) -> Option<(PendingActionPhase, Option)> { - match phase { - LoanPhase::AwaitingRelease { - observed_background_task, - .. - } => Some(( - PendingActionPhase::ReleaseRequired { - observed_background_task: *observed_background_task, - }, - None, - )), - LoanPhase::AwaitingReturn { return_context, .. } => Some(( - PendingActionPhase::ReturnRequired, - Some(return_context.clone()), - )), - LoanPhase::Restoring { - return_context, - resume_task_id, - .. - } => Some(( - PendingActionPhase::Restoring { - resume_task_id: *resume_task_id, - }, - Some(return_context.clone()), - )), - LoanPhase::Serving { .. } => None, - } -} - -fn phase_return_context(phase: &LoanPhase) -> Option { - match phase { - LoanPhase::Serving { return_context, .. } - | LoanPhase::AwaitingReturn { return_context, .. } - | LoanPhase::Restoring { return_context, .. } => Some(return_context.clone()), - LoanPhase::AwaitingRelease { .. } => None, - } -} - -/// First background launch that still reserves the resource, from durable launch state -fn background_launch_reservation(model: &ResourceReadModel) -> Option { - use BackgroundLaunchReservationStatus as Status; - - let launch = model.background_launch.as_ref()?; - let status = match launch.phase { - BackgroundLaunchPhase::Queued => Status::Queued, - BackgroundLaunchPhase::StartedUnregistered => Status::StartedUnregistered, - BackgroundLaunchPhase::IdentityMismatch => Status::IdentityMismatch, - _ if launch.awaits_operator_release() => Status::ReleaseUnproven, - BackgroundLaunchPhase::Superseded - | BackgroundLaunchPhase::Registered - | BackgroundLaunchPhase::EndedBeforeRegistration { .. } => return None, - }; - Some(BackgroundLaunchReservation { - request_id: launch.request_id, - task_id: launch.task_id, - status, - }) -} - -/// Task that holds the resource in the current loan phase -fn current_task_id(model: &ResourceReadModel) -> Option { - let phase = match &model.loan.as_ref()?.state { - LoanState::Active { phase } => phase, - LoanState::NeedsAttention { - last_safe_phase, .. - } => last_safe_phase, - LoanState::Closed { .. } => return None, - }; - match phase { - LoanPhase::Serving { - current_request_id, .. - } => model - .requests - .iter() - .find(|request| request.request_id == *current_request_id) - .map(|request| request.task_id), - LoanPhase::Restoring { resume_task_id, .. } => Some(*resume_task_id), - LoanPhase::AwaitingRelease { .. } | LoanPhase::AwaitingReturn { .. } => None, - } -} - -fn request_view(request: &ResourceRequest) -> ResourceRequestView { - ResourceRequestView { - request_id: request.request_id, - task_id: request.task_id, - acceptance_sequence: request.acceptance_sequence, - origin_machine: request.origin_machine, - thread: Some(request.spec().as_normalized().thread), - display_name: request.spec().as_normalized().name.as_str().to_owned(), - state: request.state.clone(), - } -} - -async fn task_summaries( - state: &AppState, - ids: Vec, -) -> Result, AppError> { - let mut rows = Vec::with_capacity(ids.len()); - for id in &ids { - if let Some(row) = call(&state.store, |reply| StoreMsg::GetTask { id: *id, reply }).await? { - rows.push(row); - } - } - if rows.is_empty() { - return Ok(HashMap::new()); - } - let presentations = call(&state.store, |reply| StoreMsg::TaskPresentations { - ids: rows.iter().map(|row| row.id).collect(), - reply, - }) - .await?; - Ok(rows - .iter() - .map(|row| { - let summary = TaskSummary::from_row(row, presentations.get(&row.id)); - (row.id, task_summary(summary, row.status())) - }) - .collect()) -} - -/// Resource view of one task row; a row always has a process status -fn task_summary(summary: TaskSummary, status: ProcessStatus) -> ResourceTaskSummary { - ResourceTaskSummary { - id: summary.id, - display_name: summary.display_name, - status, - thread: Some(summary.thread), - origin_machine: summary.origin_machine, - execution_machine: summary.execution_machine, - pid: summary.pid, - callback: summary.callback, - exit_reason: summary.exit_reason, - cancel_requested_at: summary.cancel_requested_at, - created_at: summary.created_at, - updated_at: summary.updated_at, - } -} - -async fn inspect(state: &AppState, resource: ResourceId) -> Option { - let inspection = tokio::time::timeout( - INSPECT_TIMEOUT, - call(&state.supervisor, |reply| SupervisorMsg::InspectResource { - id: resource, - reply, - }), - ) - .await; - match inspection { - Ok(Ok(inspection)) => inspection, - Ok(Err(error)) => { - warn!(resource = %resource.as_uuid(), "resource actor inspection failed: {error}"); - None - } - Err(_) => { - warn!(resource = %resource.as_uuid(), "resource actor inspection timed out"); - None - } - } -} - -// ---- attention ---- - -/// Most important attention condition, from durable state first and then the -/// actor's latest typed reconciliation for the same loan -fn attention( - model: &ResourceReadModel, - inspection: Option<&ResourceActorInspection>, -) -> Option { - if let Some(Loan { - state: LoanState::NeedsAttention { - action_id, reason, .. - }, - .. - }) = &model.loan - { - return Some(AttentionView { - code: AttentionCode::LoanNeedsAttention, - message: reason.clone(), - action_id: Some(*action_id), - task_id: None, - notice_id: None, - }); - } - // durable launch state needs no actor snapshot and no queued request to show - launch_attention(model) - .or_else(|| inspection.and_then(|inspection| actor_attention(model, inspection))) - .or_else(|| notice_attention(model)) -} - -fn launch_attention(model: &ResourceReadModel) -> Option { - let reservation = background_launch_reservation(model)?; - let (code, message) = match reservation.status { - BackgroundLaunchReservationStatus::ReleaseUnproven => ( - AttentionCode::BackgroundLaunchReleaseUnproven, - "the first background launch ended before registration, and its GPU release is not \ - proven; the resource stays reserved until an operator inspects the authority GPU \ - and attests with the first_background_launch binding" - .to_owned(), - ), - BackgroundLaunchReservationStatus::IdentityMismatch => ( - AttentionCode::QueueBlocked, - "the first background launch records do not match its receipt, so the resource \ - stays reserved" - .to_owned(), - ), - BackgroundLaunchReservationStatus::Queued - | BackgroundLaunchReservationStatus::StartedUnregistered => return None, - }; - Some(AttentionView { - code, - message, - action_id: None, - task_id: Some(reservation.task_id), - notice_id: None, - }) -} - -fn actor_attention( - model: &ResourceReadModel, - inspection: &ResourceActorInspection, -) -> Option { - let loan_id = model.loan.as_ref().map(|loan| loan.id); - let current = |loan: &Loan| Some(loan.id) == loan_id; - let reconcile = match inspection.reconcile_outcome.as_ref() { - Some(ResourceQueueReconcileOutcome::AttentionRequired { request, reason }) - if model.requests.iter().any(|saved| { - saved.request_id == request.request_id - && matches!( - saved.state, - ResourceRequestState::Queued | ResourceRequestState::Assigned { .. } - ) - }) => - { - let (message, task_id) = queue_attention(reason); - Some(AttentionView { - code: AttentionCode::QueueBlocked, - message, - action_id: None, - task_id: task_id.or(Some(request.task_id)), - notice_id: None, - }) - } - Some(ResourceQueueReconcileOutcome::BackgroundLaunchUncertain { task_id }) - if model.loan.is_none() => - { - Some(AttentionView { - code: AttentionCode::QueueBlocked, - message: "the first background launch row is queued without a proven worker \ - start; cancel the task to prove that no child started" - .into(), - action_id: None, - task_id: Some(*task_id), - notice_id: None, - }) - } - Some(ResourceQueueReconcileOutcome::RestoreAttentionRequired { - loan, - action_id, - task_id, - reason, - }) if current(loan) => Some(AttentionView { - code: AttentionCode::RestoreBlocked, - message: restore_attention(*reason), - action_id: Some(*action_id), - task_id: Some(*task_id), - notice_id: None, - }), - Some(ResourceQueueReconcileOutcome::ReleaseProofUnavailable { - loan, - action_id, - task_id, - reason, - }) if current(loan) => Some(AttentionView { - code: AttentionCode::ReleaseProofUnavailable, - message: format!("release proof is not available: {reason:?}"), - action_id: Some(*action_id), - task_id: Some(*task_id), - notice_id: None, - }), - _ => None, - }; - reconcile.or_else(|| watcher_attention(model, inspection.release_watcher.as_ref()?)) -} - -fn watcher_attention( - model: &ResourceReadModel, - status: &ReleaseWatcherStatus, -) -> Option { - let ReleaseWatcherStatus::Attention { action_id, reason } = status else { - return None; - }; - if model.loan.as_ref().and_then(open_action_id) != Some(*action_id) { - return None; - } - let task_id = match reason { - ReleaseWatcherAttentionReason::TrainerAssociationMissing { task_id } => Some(*task_id), - ReleaseWatcherAttentionReason::LaunchRejected { watcher_task_id } - | ReleaseWatcherAttentionReason::LaunchUncertain { watcher_task_id } - | ReleaseWatcherAttentionReason::WatcherTaskEnded { - watcher_task_id, .. - } => Some(*watcher_task_id), - ReleaseWatcherAttentionReason::RemoteSupervisorUnsupported { .. } - | ReleaseWatcherAttentionReason::ExecutableUnavailable - | ReleaseWatcherAttentionReason::BindingRejected - | ReleaseWatcherAttentionReason::BaselineUnavailable => None, - }; - Some(AttentionView { - code: AttentionCode::ReleaseWatcherBlocked, - message: format!("release watcher cannot proceed: {reason:?}"), - action_id: Some(*action_id), - task_id, - notice_id: None, - }) -} - -fn notice_attention(model: &ResourceReadModel) -> Option { - let notice = model - .notices - .iter() - .find(|notice| matches!(notice.delivery, SupervisorNoticeDelivery::Failed { .. }))?; - let SupervisorNoticeDelivery::Failed { - attempts, - last_error, - } = ¬ice.delivery - else { - return None; - }; - Some(AttentionView { - code: AttentionCode::NoticeDeliveryFailed, - message: format!("supervisor notice failed after {attempts} attempts: {last_error}"), - action_id: Some(notice.action_id), - task_id: None, - notice_id: Some(notice.id), - }) -} - -fn unconfirmed_launch_bound() -> String { - format!( - "if it has not started {} after this was first seen, it fails before launch as \ - launch_unconfirmed and the queue moves on", - humantime::format_duration(crate::resource::LAUNCH_CONFIRMATION_BOUND) - ) -} - -fn queue_attention(reason: &ResourceQueueAttentionReason) -> (String, Option) { - use ResourceQueueAttentionReason as Reason; - match reason { - Reason::IdleNotProven { gap } => idle_gap_attention(*gap), - Reason::BackgroundLaunchPending { task_id } => ( - "the first background launch has not reached a confirmed start".into(), - Some(*task_id), - ), - Reason::BackgroundTaskMissing { task_id } => ( - "the registered background task has no task record on the authority".into(), - Some(*task_id), - ), - Reason::BackgroundTaskNotRunning { task_id, state } => ( - format!("the registered background task is not verifiably running ({state})"), - Some(*task_id), - ), - Reason::AcceptedTaskLaunchUncertain { task_id } => ( - format!( - "an accepted command task has no proven worker start; {}", - unconfirmed_launch_bound() - ), - Some(*task_id), - ), - Reason::UnverifiedServingRelease => ( - "the serving loan has no verified trainer release proof".into(), - None, - ), - Reason::ReleaseProofUnavailable { - task_id, reason, .. - } => ( - format!("the trainer release proof is not available: {reason:?}"), - Some(*task_id), - ), - Reason::AssignedTaskLaunchUncertain { task_id } => ( - format!( - "the assigned command launch result is uncertain; {}", - unconfirmed_launch_bound() - ), - Some(*task_id), - ), - Reason::AssignedTaskLost { task_id } => ( - "the assigned command task is lost and its work may still run".into(), - Some(*task_id), - ), - Reason::AssignedTaskExitUnconfirmed { task_id } => ( - "the assigned task ended without a confirmed exit: a command needs its process-group \ - exit, and a container needs its removal" - .into(), - Some(*task_id), - ), - Reason::AssignedTaskIdentityMismatch { task_id } => ( - "the assigned command identity does not match its request or route".into(), - Some(*task_id), - ), - Reason::AssignedTaskNoChildSpawnProofInvalid { task_id } => ( - "the assigned command cannot prove that no child process started".into(), - Some(*task_id), - ), - Reason::AssignedTaskOwnershipUncertain { task_id, risk } => ( - format!("the assigned command may outlive its process group: {risk:?}"), - Some(*task_id), - ), - Reason::AssignedTaskStaleRevision { task_id } => ( - "the resource revision changed before the task completion committed".into(), - Some(*task_id), - ), - Reason::AssignedTaskReconcileFailed { task_id } => ( - "the authority could not evaluate the assigned command".into(), - Some(*task_id), - ), - } -} - -fn idle_gap_attention(gap: IdleProofGap) -> (String, Option) { - match gap { - IdleProofGap::NoIdleEvidence => ( - "no registered background task, and no saved no-resume decision or background \ - launch result proves the GPU is idle; for a resource with no history, an operator \ - who inspected the authority GPU can run resource initial-idle" - .into(), - None, - ), - IdleProofGap::BackgroundLaunchReleaseUnproven { task_id } => ( - "the first background launch ended without proof that its process released the GPU" - .into(), - Some(task_id), - ), - IdleProofGap::InconsistentHistory => ( - "the saved loan and background launch history does not match the resource \ - registration" - .into(), - None, - ), - } -} - -fn restore_attention(reason: RestoreAttentionReason) -> String { - match reason { - RestoreAttentionReason::LaunchUncertain => { - "the return task launch is not proven to have started".into() - } - RestoreAttentionReason::EndedBeforeConfirmedStart { state } => { - format!("the return task ended ({state}) before a confirmed start") - } - RestoreAttentionReason::ForegroundEnded { state } => { - format!("the native foreground return task ended ({state}) without success") - } - RestoreAttentionReason::ForegroundExitUnconfirmed { state } => format!( - "the native foreground return task ended ({state}), but its process-group exit is \ - not confirmed; the resource stays reserved until an operator inspects the authority \ - GPU and attests with the restoring_foreground_return binding" - ), - RestoreAttentionReason::ContainerEnded { state } => { - format!("the container return task ended ({state}) without success") - } - RestoreAttentionReason::ContainerExitUnconfirmed { state } => format!( - "the container return task ended ({state}), but Homebased did not confirm that its \ - container exited and was removed; the resource stays reserved until an operator \ - inspects the authority GPU and attests with the restoring_foreground_return binding" - ), - RestoreAttentionReason::Lost => "the return task is lost; the resource stays reserved \ - until an operator inspects the authority GPU and attests with the Restoring binding \ - of its execution mode" - .into(), - RestoreAttentionReason::IdentityMismatch => { - "the return task identity does not match the bound action".into() - } - RestoreAttentionReason::ReconcileFailed => { - "the authority could not evaluate the return task".into() - } - } -} - -#[cfg(test)] -mod tests; diff --git a/src/daemon/resource_api/tests.rs b/src/daemon/resource_api/tests.rs deleted file mode 100644 index 3b26249..0000000 --- a/src/daemon/resource_api/tests.rs +++ /dev/null @@ -1,1903 +0,0 @@ -use crate::daemon::actors::supervisor::SUPERVISOR_TEST_LOCK; -use crate::store::Store; -use std::net::SocketAddr; -use std::os::unix::fs::PermissionsExt; -use std::path::PathBuf; - -use axum::http::StatusCode; -use bytes::Bytes; -use http_body_util::{BodyExt, Full}; -use hyper_util::rt::TokioIo; -use ractor::{Actor, ActorProcessingErr, ActorRef}; -use serde_json::{Value, json}; -use tempfile::{TempDir, tempdir}; -use tokio::net::{TcpListener, TcpStream}; -use tokio::sync::mpsc; -use tokio::task::JoinHandle; - -use super::{attention, decode_control_response, local_models, local_resource, remote_detail}; -use crate::config::{Discovery, FleetSettings}; -use crate::daemon::AppState; -use crate::daemon::actors::{StoreMsg, SupervisorActor, SupervisorArgs, SupervisorMsg, call}; -use crate::domain::{ProcessStatus, TaskId, ThreadId}; -use crate::error::AppError; -use crate::files::StreamSlots; -use crate::fleet::FleetState; -use crate::fleet::address::MachineAddress; -use crate::fleet::directory::LocalMachine; -use crate::fleet::http::ClusterResponse; -use crate::fleet::protocol::SUPPORTED_PROTOCOLS; -use crate::fleet::runtime::{FleetRuntime, FleetStart, RuntimeTimings}; -use crate::home::Home; -use crate::machine::{LocalIdentity, MachineId, MachineName}; -use crate::resource::api::{ - AttentionCode, BrowserResourceAction, RESOURCE_PENDING_PATH, RESOURCE_REGISTER_PATH, -}; -use crate::resource::command_shape::test_support::FakeTrainer; -use crate::resource::operator_release::{ - OperatorAttestationId, OperatorGpuFreeAttestation, OperatorGpuFreeConfirmation, - OperatorObservation, OperatorStateBinding, -}; -use crate::resource::{ - ActionId, AssignmentRevision, CommandSpec, DeliveryAttemptId, Loan, LoanId, LoanPhase, - LoanState, NoticeId, ResourceId, ResourceRevision, ReturnContext, ReturnExecutionMode, - ReturnLaunch, ReturnWork, SupervisorActionAuthority, SupervisorAddress, SupervisorNotice, - SupervisorNoticeDelivery, SupervisorNoticePayload, -}; -use crate::spec::{NormalizedTaskWorkload, NormalizedWorkload}; -use crate::store::{ - BackgroundLaunchAcceptance, BackgroundLaunchInput, ResourceControlRequest, - ReturnTaskAcceptance, ReturnTaskAcceptanceInput, ReturnTaskOrigin, -}; -use crate::submission::{CallbackExecutable, RequestId}; -use uuid::Uuid; - -const THREAD: &str = "01a0ab97-a7aa-7463-a5b0-8d500e40e431"; - -/// Running daemon routes on one registered resource; callers hold the supervisor test lock -struct Fixture { - _directory: TempDir, - state: AppState, - supervisor: ActorRef, - supervisor_handle: JoinHandle<()>, - servers: Vec>, - dashboard: SocketAddr, - socket: SocketAddr, - resource: ResourceId, - bin: PathBuf, - callback_cwd: PathBuf, -} - -impl Fixture { - async fn new() -> Self { - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().join("state"))).unwrap(); - home.ensure().unwrap(); - let bin = directory.path().join("bin"); - std::fs::create_dir(&bin).unwrap(); - let codex = bin.join("codex"); - std::fs::write(&codex, "#!/bin/sh\nexit 0\n").unwrap(); - std::fs::set_permissions(&codex, std::fs::Permissions::from_mode(0o755)).unwrap(); - let callback_cwd = directory.path().join("callback"); - std::fs::create_dir(&callback_cwd).unwrap(); - - let (supervisor, supervisor_handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let state = AppState { - home: home.clone(), - store, - supervisor: supervisor.clone(), - web: None, - content: None, - stream_slots: StreamSlots::new(), - machine: LocalMachine { - identity: LocalIdentity::start(&home).unwrap(), - name: MachineName::fallback(), - protocol: SUPPORTED_PROTOCOLS, - }, - fleet: FleetState::Disabled, - message_receiver: crate::daemon::message_receiver::MessageReceiver::default(), - locks: crate::daemon::DaemonLocks::default(), - thread_titles: None, - }; - - let dashboard_listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); - let dashboard = dashboard_listener.local_addr().unwrap(); - let socket_listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); - let socket = socket_listener.local_addr().unwrap(); - let dashboard_router = crate::daemon::web::router(state.clone(), dashboard); - let socket_router = crate::daemon::api::socket_router(state.clone()); - let servers = vec![ - tokio::spawn(async move { - let _ = axum::serve(dashboard_listener, dashboard_router).await; - }), - tokio::spawn(async move { - let _ = axum::serve(socket_listener, socket_router).await; - }), - // the origin-owned cancellation intent reaches the authority through this loop - tokio::spawn(crate::daemon::cancel_delivery::run(state.clone())), - ]; - - let mut fixture = Self { - _directory: directory, - state, - supervisor, - supervisor_handle, - servers, - dashboard, - socket, - resource: ResourceId::new(), - bin, - callback_cwd, - }; - let (status, body) = fixture - .socket_post( - RESOURCE_REGISTER_PATH, - json!({ - "api_version": 1, - "spec": { - "id": fixture.resource, - "display_name": "RTX 5090", - "supervisor": { "machine": fixture.local(), "thread": THREAD }, - } - }), - ) - .await; - assert_eq!(status, StatusCode::OK, "{body}"); - fixture.resource = serde_json::from_value(body["resource"]["id"].clone()).unwrap(); - fixture - } - - fn local(&self) -> MachineId { - self.state.machine.identity.machine - } - - fn origin(&self) -> String { - format!("http://{}", self.dashboard) - } - - async fn socket_post(&self, path: &str, body: Value) -> (StatusCode, Value) { - let host = self.socket.to_string(); - send( - self.socket, - "POST", - path, - &[("host", host), ("content-type", "application/json".into())], - Some(body), - ) - .await - } - - async fn socket_get(&self, path: &str) -> (StatusCode, Value) { - send( - self.socket, - "GET", - path, - &[("host", self.socket.to_string())], - None, - ) - .await - } - - async fn browser_action(&self, body: Value) -> (StatusCode, Value) { - let path = format!("/v1/resources/{}/actions", self.resource.as_uuid()); - send( - self.dashboard, - "POST", - &path, - &[ - ("host", self.dashboard.to_string()), - ("origin", self.origin()), - ("content-type", "application/json".into()), - ], - Some(body), - ) - .await - } - - async fn detail(&self) -> Value { - let (status, body) = self - .socket_get(&format!("/v1/resources/{}", self.resource.as_uuid())) - .await; - assert_eq!(status, StatusCode::OK, "{body}"); - body - } - - async fn submit_request(&self) -> (RequestId, TaskId) { - let request_id = RequestId::new(); - let (status, body) = self - .socket_post( - &format!("/v1/resources/{}/requests", self.resource.as_uuid()), - json!({ - "api_version": 1, - "request_id": request_id, - "spec": { - "api_version": 1, - "thread": THREAD, - "name": "attention benchmark", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": ["echo", "hello"] } - }, - "env": { "path": self.bin, "home": "/tmp" }, - "callback_cwd": self.callback_cwd, - }), - ) - .await; - assert_eq!(status, StatusCode::OK, "{body}"); - assert_eq!(body["outcome"]["type"], "waiting"); - let task_id = serde_json::from_value(body["task_id"].clone()).unwrap(); - (request_id, task_id) - } - - async fn seed_registered_trainer(&self, end_task: bool) -> (TaskId, ResourceRevision) { - let root = self.callback_cwd.canonicalize().unwrap(); - let trainer = FakeTrainer::new(&root); - let request_id = RequestId::new(); - let task_id = TaskId::new(); - let acceptance = call(&self.state.store, |reply| { - StoreMsg::AcceptBackgroundLaunchForAuthority { - input: Box::new(BackgroundLaunchInput { - authority_machine: self.local(), - resource_id: self.resource, - request_id, - task_id, - spec: trainer.spec(THREAD.parse().unwrap(), "controlled trainer fixture"), - env: trainer.env.clone(), - callback_codex: CallbackExecutable::available(self.bin.join("codex")), - }), - reply, - } - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - acceptance, - BackgroundLaunchAcceptance::Inserted { task, .. } if task == task_id - )); - call(&self.state.store, |reply| StoreMsg::CasStatus { - id: task_id, - from: ProcessStatus::Queued, - to: ProcessStatus::Running, - worker_thread: None, - reply, - }) - .await - .unwrap() - .expect("the synthetic launch row must enter running state"); - call(&self.supervisor, |reply| SupervisorMsg::ReconcileResource { - id: self.resource, - reply, - }) - .await - .unwrap(); - - if end_task { - call(&self.state.store, |reply| StoreMsg::CasStatus { - id: task_id, - from: ProcessStatus::Running, - to: ProcessStatus::Lost, - worker_thread: None, - reply, - }) - .await - .unwrap() - .expect("the synthetic trainer must end before attestation"); - } - - let resource = local_resource(&self.state, self.resource).await.unwrap(); - (task_id, resource.state_revision) - } - - /// Bind a first launch whose fake task starts and is lost before its start registers - async fn seed_early_ended_launch(&self) -> (RequestId, TaskId) { - let root = self.callback_cwd.canonicalize().unwrap(); - let trainer = FakeTrainer::new(&root); - let (request_id, task_id) = (RequestId::new(), TaskId::new()); - let acceptance = call(&self.state.store, |reply| { - StoreMsg::AcceptBackgroundLaunchForAuthority { - input: Box::new(BackgroundLaunchInput { - authority_machine: self.local(), - resource_id: self.resource, - request_id, - task_id, - spec: trainer.spec(THREAD.parse().unwrap(), "controlled trainer fixture"), - env: trainer.env.clone(), - callback_codex: CallbackExecutable::available(self.bin.join("codex")), - }), - reply, - } - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - acceptance, - BackgroundLaunchAcceptance::Inserted { task, .. } if task == task_id - )); - for (from, to) in [ - (ProcessStatus::Queued, ProcessStatus::Running), - (ProcessStatus::Running, ProcessStatus::Lost), - ] { - call(&self.state.store, |reply| StoreMsg::CasStatus { - id: task_id, - from, - to, - worker_thread: None, - reply, - }) - .await - .unwrap() - .expect("the synthetic launch row must change state"); - } - (request_id, task_id) - } - - async fn socket_with_supervisor( - &self, - supervisor: ActorRef, - ) -> (SocketAddr, JoinHandle<()>) { - let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); - let address = listener.local_addr().unwrap(); - let mut state = self.state.clone(); - state.supervisor = supervisor; - let router = super::socket_routes().with_state(state); - let server = tokio::spawn(async move { - let _ = axum::serve(listener, router).await; - }); - (address, server) - } - - /// Seed an AwaitingReturn loan whose return notice used every automatic attempt - fn seed_failed_return_notice(&self, destination: SupervisorAddress) -> SupervisorNotice { - let loan_id = LoanId::new(); - let action_id = ActionId::new(); - let state = LoanState::Active { - phase: LoanPhase::AwaitingReturn { - action_id, - return_context: ReturnContext::Idle, - }, - }; - let notice = SupervisorNotice { - id: NoticeId::new(), - loan_id, - action_id, - state_revision: ResourceRevision::new(0), - destination, - assignment_revision: AssignmentRevision::new(0), - payload: SupervisorNoticePayload::ReturnRequired { - return_context: ReturnContext::Idle, - }, - delivery: SupervisorNoticeDelivery::Failed { - attempts: 3, - last_error: "receiver offline".into(), - }, - }; - let loan = Loan { - id: loan_id, - resource_id: self.resource, - state, - }; - Store::seed_loan_notice_for_test(&self.state.home.db_path(), &loan, ¬ice); - notice - } - - /// Accept a queued return task without starting its worker - async fn seed_restoring_return(&self, mode: ReturnExecutionMode) -> TaskId { - let destination = SupervisorAddress { - machine: self.local(), - thread: THREAD.parse().unwrap(), - }; - let notice = self.seed_failed_return_notice(destination); - let trainer = FakeTrainer::new(&self.callback_cwd.canonicalize().unwrap()); - let mut spec = trainer.spec(destination.thread, "resource detail return fixture"); - if mode == ReturnExecutionMode::NativeForeground { - spec.workload = NormalizedWorkload::Task(NormalizedTaskWorkload { - command: crate::invocation::CommandLine::try_from_argv(vec![ - "/bin/echo".into(), - "foreground return".into(), - ]) - .unwrap(), - }); - } - let launch = ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::NewBackgroundWork { - spec: CommandSpec::try_from(spec).unwrap(), - }, - }; - let authority = SupervisorActionAuthority { - authority_machine: self.local(), - resource_id: self.resource, - loan_id: notice.loan_id, - action_id: notice.action_id, - expected_state_revision: notice.state_revision, - supervisor: notice.destination, - assignment_revision: notice.assignment_revision, - }; - let task_id = launch.task_id; - let acceptance = call(&self.state.store, |reply| { - StoreMsg::AcceptReturnTaskForAuthority { - input: Box::new(ReturnTaskAcceptanceInput { - authority, - launch, - executor_env: trainer.env.clone(), - origin: ReturnTaskOrigin::Local { - callback_codex: CallbackExecutable::available(self.bin.join("codex")), - }, - }), - reply, - } - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - acceptance, - ReturnTaskAcceptance::Inserted { task, .. } if task == task_id - )); - task_id - } - - fn control_operation_count(&self) -> i64 { - Store::resource_control_operation_count_for_test(&self.state.home.db_path()) - } - - async fn stop(self) { - for server in &self.servers { - server.abort(); - } - self.supervisor.stop(None); - // actor names are global, so the next test waits for this tree to unregister - let _ = self.supervisor_handle.await; - } -} - -struct ReconcileProbe { - fail: bool, -} - -impl Actor for ReconcileProbe { - type Msg = SupervisorMsg; - type State = mpsc::UnboundedSender; - type Arguments = mpsc::UnboundedSender; - - async fn pre_start( - &self, - _myself: ActorRef, - requests: Self::Arguments, - ) -> Result { - Ok(requests) - } - - async fn handle( - &self, - _myself: ActorRef, - message: Self::Msg, - state: &mut Self::State, - ) -> Result<(), ActorProcessingErr> { - let SupervisorMsg::ReconcileResource { id, reply } = message else { - return Ok(()); - }; - state - .send(id) - .expect("reconciliation probe receiver is live"); - let result = if self.fail { - Err(AppError::Internal { - message: "controlled reconciliation failure".into(), - }) - } else { - Ok(()) - }; - crate::daemon::actors::send_reply(reply, result); - Ok(()) - } -} - -async fn send( - address: SocketAddr, - method: &str, - path: &str, - headers: &[(&str, String)], - body: Option, -) -> (StatusCode, Value) { - let stream = TcpStream::connect(address).await.unwrap(); - let (mut sender, connection) = hyper::client::conn::http1::handshake(TokioIo::new(stream)) - .await - .unwrap(); - tokio::spawn(async move { - let _ = connection.await; - }); - let mut builder = http::Request::builder().method(method).uri(path); - for (name, value) in headers { - builder = builder.header(*name, value); - } - let bytes = body.map_or_else(Vec::new, |body| serde_json::to_vec(&body).unwrap()); - let request = builder.body(Full::new(Bytes::from(bytes))).unwrap(); - let response = sender.send_request(request).await.unwrap(); - let status = response.status(); - let bytes = response.into_body().collect().await.unwrap().to_bytes(); - ( - status, - serde_json::from_slice(&bytes).unwrap_or(Value::Null), - ) -} - -fn error_code(body: &Value) -> &str { - body["error"]["code"].as_str().unwrap_or_default() -} - -fn operator_attestation( - resource_id: ResourceId, - authority_machine: MachineId, - task_id: TaskId, - revision: ResourceRevision, -) -> OperatorGpuFreeAttestation { - OperatorGpuFreeAttestation { - operation_id: OperatorAttestationId::new(), - resource_id, - authority_machine, - task_id, - expected_state_revision: revision, - state_binding: OperatorStateBinding::NoLoan, - observation: OperatorObservation::try_from( - "inspected the authority GPU; no trainer process remains".to_owned(), - ) - .unwrap(), - confirmation: OperatorGpuFreeConfirmation::OperatorConfirmedGpuFree, - } -} - -fn operator_release_body(attestation: &OperatorGpuFreeAttestation) -> Value { - json!({ - "api_version": 1, - "attestation": attestation, - }) -} - -async fn spawn_reconcile_probe( - fail: bool, -) -> ( - ActorRef, - JoinHandle<()>, - mpsc::UnboundedReceiver, -) { - let (sender, requests) = mpsc::unbounded_channel(); - let (actor, handle) = Actor::spawn(None, ReconcileProbe { fail }, sender) - .await - .unwrap(); - (actor, handle, requests) -} - -#[tokio::test] -async fn local_resource_detail_shows_the_saved_return_mode_only_while_restoring() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let idle = fixture.detail().await; - assert!(idle.get("return_execution_mode").is_none(), "{idle}"); - - fixture - .seed_restoring_return(ReturnExecutionMode::DirectSegmentTrainer) - .await; - let detail = fixture.detail().await; - assert_eq!(detail["return_execution_mode"], "direct_segment_trainer"); - fixture.stop().await; -} - -#[tokio::test] -async fn remote_resource_detail_preserves_the_authority_return_mode() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let authority = Fixture::new().await; - let task = authority - .seed_restoring_return(ReturnExecutionMode::NativeForeground) - .await; - - let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); - let address = listener.local_addr().unwrap(); - let authority_runtime = FleetRuntime::start(FleetStart { - local: authority.state.machine.clone(), - settings: FleetSettings { - discovery: Discovery { - mdns: false, - tailscale: None, - }, - machines: Vec::new(), - }, - listener: Some(address), - peers_path: authority._directory.path().join("authority-peers.json"), - timings: RuntimeTimings::default(), - }) - .unwrap(); - let mut authority_state = authority.state.clone(); - authority_state.fleet = FleetState::Enabled(authority_runtime.handle()); - let router = crate::daemon::web::router(authority_state, address); - let authority_server = tokio::spawn(async move { - let _ = axum::serve(listener, router).await; - }); - - let caller_machine = LocalMachine { - identity: LocalIdentity { - machine: MachineId::new(), - boot: crate::machine::BootId::new(), - }, - name: MachineName::parse("resource-reader").unwrap(), - protocol: SUPPORTED_PROTOCOLS, - }; - let peer_address = MachineAddress::from_socket(address); - let caller_runtime = FleetRuntime::start(FleetStart { - local: caller_machine.clone(), - settings: FleetSettings { - discovery: Discovery { - mdns: false, - tailscale: None, - }, - machines: vec![peer_address.clone()], - }, - listener: None, - peers_path: authority._directory.path().join("caller-peers.json"), - timings: RuntimeTimings::default(), - }) - .unwrap(); - let caller_handle = caller_runtime.handle(); - caller_handle.probe_address(&peer_address).await.unwrap(); - let mut caller_state = authority.state.clone(); - caller_state.machine = caller_machine; - caller_state.fleet = FleetState::Enabled(caller_handle); - - let detail = remote_detail(&caller_state, authority.resource) - .await - .unwrap(); - assert_eq!(detail.current_task_id, Some(task)); - assert_eq!( - detail.return_execution_mode, - Some(ReturnExecutionMode::NativeForeground) - ); - assert_eq!( - serde_json::to_value(detail).unwrap()["return_execution_mode"], - "native_foreground" - ); - - caller_runtime.shutdown().await; - authority_server.abort(); - authority_runtime.shutdown().await; - authority.stop().await; -} - -#[tokio::test] -async fn operator_release_checks_path_authority_and_confirmation_before_store() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let (task_id, revision) = fixture.seed_registered_trainer(true).await; - let attestation = operator_attestation(fixture.resource, fixture.local(), task_id, revision); - let release_path = format!( - "/v1/resources/{}/operator-release", - fixture.resource.as_uuid() - ); - - let (status, response) = send( - fixture.dashboard, - "POST", - &release_path, - &[ - ("host", fixture.dashboard.to_string()), - ("origin", fixture.origin()), - ("content-type", "application/json".into()), - ], - Some(operator_release_body(&attestation)), - ) - .await; - assert_eq!(status, StatusCode::NOT_FOUND, "{response}"); - assert_eq!(error_code(&response), "not_found"); - - let (status, response) = fixture - .socket_post( - &format!( - "/v1/resources/{}/operator-release", - ResourceId::new().as_uuid() - ), - operator_release_body(&attestation), - ) - .await; - assert_eq!(status, StatusCode::BAD_REQUEST, "{response}"); - assert_eq!(error_code(&response), "usage"); - - let wrong_authority = OperatorGpuFreeAttestation { - authority_machine: MachineId::new(), - ..attestation.clone() - }; - let (status, response) = fixture - .socket_post( - &format!( - "/v1/resources/{}/operator-release", - fixture.resource.as_uuid() - ), - operator_release_body(&wrong_authority), - ) - .await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_action_not_allowed"); - - let mut missing_confirmation = operator_release_body(&attestation); - missing_confirmation["attestation"] - .as_object_mut() - .unwrap() - .remove("confirmation"); - let (status, response) = fixture - .socket_post( - &format!( - "/v1/resources/{}/operator-release", - fixture.resource.as_uuid() - ), - missing_confirmation, - ) - .await; - assert_eq!(status, StatusCode::BAD_REQUEST, "{response}"); - assert_eq!(error_code(&response), "usage"); - - // every state binding is strict input, including the field-free no_loan tag - let loan_id = LoanId::new(); - let action_id = ActionId::new(); - for binding in [ - json!({ "type": "no_loan" }), - json!({ "type": "awaiting_release", "loan_id": loan_id, "action_id": action_id }), - json!({ "type": "first_background_launch", "request_id": RequestId::new() }), - json!({ "type": "restoring_return", "loan_id": loan_id, "action_id": action_id }), - json!({ - "type": "restoring_foreground_return", - "loan_id": loan_id, - "action_id": action_id, - }), - ] { - let mut extra_field = operator_release_body(&attestation); - extra_field["attestation"]["state_binding"] = binding; - extra_field["attestation"]["state_binding"]["unexpected"] = json!(true); - let (status, response) = fixture - .socket_post( - &format!( - "/v1/resources/{}/operator-release", - fixture.resource.as_uuid() - ), - extra_field, - ) - .await; - assert_eq!(status, StatusCode::BAD_REQUEST, "{response}"); - assert_eq!(error_code(&response), "usage"); - } - - let mut malformed_confirmation = operator_release_body(&attestation); - malformed_confirmation["attestation"]["confirmation"] = json!("automatic_proof"); - let (status, response) = fixture - .socket_post( - &format!( - "/v1/resources/{}/operator-release", - fixture.resource.as_uuid() - ), - malformed_confirmation, - ) - .await; - assert_eq!(status, StatusCode::BAD_REQUEST, "{response}"); - assert_eq!(error_code(&response), "usage"); - - for attestation in [attestation, wrong_authority] { - let receipt = Store::open(&fixture.state.home.db_path()) - .unwrap() - .operator_attestation_receipt_for_authority(fixture.local(), attestation.operation_id) - .unwrap(); - assert!(receipt.is_none()); - } - - fixture.stop().await; -} - -#[tokio::test] -async fn operator_release_refuses_a_running_trainer() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let (task_id, revision) = fixture.seed_registered_trainer(false).await; - let attestation = operator_attestation(fixture.resource, fixture.local(), task_id, revision); - let (status, response) = fixture - .socket_post( - &format!( - "/v1/resources/{}/operator-release", - fixture.resource.as_uuid() - ), - operator_release_body(&attestation), - ) - .await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_action_not_allowed"); - assert!( - response["error"]["message"] - .as_str() - .unwrap_or_default() - .contains("has not ended") - ); - fixture.stop().await; -} - -#[tokio::test] -async fn operator_release_commits_replays_and_requests_resource_reconciliation() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let (task_id, revision) = fixture.seed_registered_trainer(true).await; - let attestation = operator_attestation(fixture.resource, fixture.local(), task_id, revision); - let (probe, probe_handle, mut requests) = spawn_reconcile_probe(false).await; - let (socket, server) = fixture.socket_with_supervisor(probe.clone()).await; - let path = format!( - "/v1/resources/{}/operator-release", - fixture.resource.as_uuid() - ); - let headers = [ - ("host", socket.to_string()), - ("content-type", "application/json".into()), - ]; - - let (status, first) = send( - socket, - "POST", - &path, - &headers, - Some(operator_release_body(&attestation)), - ) - .await; - assert_eq!(status, StatusCode::OK, "{first}"); - assert_eq!(first["api_version"], 1); - assert_eq!(first["replayed"], false); - assert_eq!( - first["receipt"]["attestation"]["operation_id"], - json!(attestation.operation_id) - ); - assert_eq!(first["receipt"]["outcome"]["type"], "idle_boundary"); - - let (status, retry) = send( - socket, - "POST", - &path, - &headers, - Some(operator_release_body(&attestation)), - ) - .await; - assert_eq!(status, StatusCode::OK, "{retry}"); - assert_eq!(retry["replayed"], true); - assert_eq!(retry["receipt"], first["receipt"]); - - let changed = OperatorGpuFreeAttestation { - observation: OperatorObservation::try_from("different observation".to_owned()).unwrap(), - ..attestation.clone() - }; - let (status, response) = send( - socket, - "POST", - &path, - &headers, - Some(operator_release_body(&changed)), - ) - .await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_operation_conflict"); - - let stale = OperatorGpuFreeAttestation { - operation_id: OperatorAttestationId::new(), - ..attestation.clone() - }; - let (status, response) = send( - socket, - "POST", - &path, - &headers, - Some(operator_release_body(&stale)), - ) - .await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_stale_revision"); - - assert_eq!(requests.try_recv().unwrap(), fixture.resource); - assert_eq!(requests.try_recv().unwrap(), fixture.resource); - assert!(requests.try_recv().is_err()); - server.abort(); - probe.stop(None); - let _ = probe_handle.await; - fixture.stop().await; -} - -#[tokio::test] -async fn operator_release_keeps_its_saved_receipt_when_reconciliation_is_uncertain() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let (task_id, revision) = fixture.seed_registered_trainer(true).await; - let attestation = operator_attestation(fixture.resource, fixture.local(), task_id, revision); - let (probe, probe_handle, mut requests) = spawn_reconcile_probe(true).await; - let (socket, server) = fixture.socket_with_supervisor(probe.clone()).await; - let path = format!( - "/v1/resources/{}/operator-release", - fixture.resource.as_uuid() - ); - let headers = [ - ("host", socket.to_string()), - ("content-type", "application/json".into()), - ]; - let (status, response) = send( - socket, - "POST", - &path, - &headers, - Some(operator_release_body(&attestation)), - ) - .await; - assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{response}"); - assert_eq!(error_code(&response), "resource_outcome_unknown"); - assert_eq!( - response["error"]["input"]["operation_id"], - json!(attestation.operation_id.as_uuid()) - ); - - let receipt = Store::open(&fixture.state.home.db_path()) - .unwrap() - .operator_attestation_receipt_for_authority(fixture.local(), attestation.operation_id) - .unwrap() - .expect("the transaction receipt must remain saved after the failed wake"); - assert_eq!(receipt.attestation, attestation); - assert_eq!(requests.try_recv().unwrap(), fixture.resource); - assert!(requests.try_recv().is_err()); - - server.abort(); - probe.stop(None); - let _ = probe_handle.await; - fixture.stop().await; -} - -#[tokio::test] -async fn early_ended_first_launch_is_reserved_in_the_read_model_until_attested() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let (request_id, task_id) = fixture.seed_early_ended_launch().await; - let reserved = |body: &Value| { - body["loan"].is_null() - && body["resource"]["registered_background_task"].is_null() - && body["attention"]["code"] == "background_launch_release_unproven" - && body["attention"]["task_id"] == json!(task_id) - && body["background_launch"] - == json!({ - "request_id": request_id, - "task_id": task_id, - "status": "release_unproven", - }) - }; - - // no request is queued, and the durable state alone names the reservation - let detail = fixture.detail().await; - assert!(reserved(&detail), "{detail}"); - assert_eq!(detail["requests"], json!([])); - let (status, list) = fixture.socket_get("/v1/resources").await; - assert_eq!(status, StatusCode::OK, "{list}"); - let overview = list["resources"] - .as_array() - .unwrap() - .iter() - .find(|item| item["resource"]["id"] == json!(fixture.resource)) - .unwrap(); - assert!(reserved(overview), "{overview}"); - assert_eq!(overview["queued_count"], 0); - - // the view needs no actor snapshot, which a restarted daemon may not have yet - let model = local_models(&fixture.state, Some(fixture.resource)) - .await - .unwrap() - .remove(0); - let durable = attention(&model, None).unwrap(); - assert_eq!(durable.code, AttentionCode::BackgroundLaunchReleaseUnproven); - assert_eq!(durable.task_id, Some(task_id)); - - let revision = serde_json::from_value(detail["resource"]["state_revision"].clone()).unwrap(); - let release_path = format!( - "/v1/resources/{}/operator-release", - fixture.resource.as_uuid() - ); - // the registered-trainer binding cannot release an unregistered launch task - let registered = operator_attestation(fixture.resource, fixture.local(), task_id, revision); - let (status, response) = fixture - .socket_post(&release_path, operator_release_body(®istered)) - .await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_action_not_allowed"); - assert!(reserved(&fixture.detail().await)); - - let attestation = OperatorGpuFreeAttestation { - state_binding: OperatorStateBinding::FirstBackgroundLaunch { request_id }, - ..operator_attestation(fixture.resource, fixture.local(), task_id, revision) - }; - let (status, receipt) = fixture - .socket_post(&release_path, operator_release_body(&attestation)) - .await; - assert_eq!(status, StatusCode::OK, "{receipt}"); - assert_eq!(receipt["replayed"], false); - assert_eq!(receipt["receipt"]["outcome"]["type"], "idle_boundary"); - assert_eq!( - receipt["receipt"]["evidence"]["trainer_launch"], - json!({ "type": "first_background_launch", "request_id": request_id }) - ); - let (status, replay) = fixture - .socket_post(&release_path, operator_release_body(&attestation)) - .await; - assert_eq!(status, StatusCode::OK, "{replay}"); - assert_eq!(replay["replayed"], true); - assert_eq!(replay["receipt"], receipt["receipt"]); - - let released = fixture.detail().await; - assert!(released["attention"].is_null(), "{released}"); - assert!(released.get("background_launch").is_none(), "{released}"); - fixture.stop().await; -} - -#[tokio::test] -async fn dashboard_action_requires_exact_origin_json_and_host() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let path = format!("/v1/resources/{}/actions", fixture.resource.as_uuid()); - let body = json!({ - "api_version": 1, - "expected_revision": 0, - "operation_id": Uuid::now_v7(), - "action": { "type": "stop_active", "task_id": TaskId::new() }, - }); - let host = fixture.dashboard.to_string(); - let json_type = ("content-type", "application/json".to_owned()); - - let cases = [ - vec![("host", host.clone()), json_type.clone()], - vec![ - ("host", host.clone()), - ("origin", "http://evil.example".into()), - json_type.clone(), - ], - vec![ - ("host", host.clone()), - ("origin", format!("https://{host}")), - json_type.clone(), - ], - vec![ - ("host", host.clone()), - ("origin", fixture.origin()), - ("content-type", "text/plain".into()), - ], - vec![ - ("host", host.clone()), - ("origin", fixture.origin()), - ("sec-fetch-site", "cross-site".into()), - json_type.clone(), - ], - ]; - for headers in cases { - let (status, response) = send( - fixture.dashboard, - "POST", - &path, - &headers, - Some(body.clone()), - ) - .await; - assert_eq!(status, StatusCode::FORBIDDEN, "{headers:?} {response}"); - assert_eq!(error_code(&response), "permission"); - } - - let (status, response) = send( - fixture.dashboard, - "POST", - &path, - &[ - ("host", "attacker.example".into()), - ("origin", "http://attacker.example".into()), - json_type.clone(), - ], - Some(body.clone()), - ) - .await; - assert_eq!(status, StatusCode::BAD_REQUEST, "{response}"); - assert_eq!(response["error"]["message"], "unexpected Host header"); - - // the exact dashboard origin reaches the authority, which refuses a task that is not active - let (status, response) = fixture.browser_action(body).await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_action_not_allowed"); - assert_eq!(fixture.control_operation_count(), 0); - - // general task submit and the CLI-only resource writes stay off the dashboard - for write in [ - "/v1/tasks".to_owned(), - RESOURCE_REGISTER_PATH.to_owned(), - format!("/v1/resources/{}/requests", fixture.resource.as_uuid()), - format!("/v1/resources/{}/supervisor", fixture.resource.as_uuid()), - ] { - let (status, _) = send( - fixture.dashboard, - "POST", - &write, - &[ - ("host", host.clone()), - ("origin", fixture.origin()), - json_type.clone(), - ], - Some(json!({})), - ) - .await; - assert!( - matches!( - status, - StatusCode::METHOD_NOT_ALLOWED | StatusCode::NOT_FOUND - ), - "{write} returned {status}" - ); - } - fixture.stop().await; -} - -#[tokio::test] -async fn queued_cancel_uses_the_origin_intent_and_retries_by_operation() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let (request_id, task_id) = fixture.submit_request().await; - let detail = fixture.detail().await; - assert_eq!(detail["requests"][0]["state"]["type"], "queued"); - assert_eq!(detail["requests"][0]["display_name"], "attention benchmark"); - // no registered background task, so the queue stays reserved rather than idle - assert_eq!(detail["attention"]["code"], "queue_blocked"); - let revision = detail["resource"]["state_revision"].as_u64().unwrap(); - - let stale = json!({ - "api_version": 1, - "expected_revision": revision + 1, - "operation_id": Uuid::now_v7(), - "action": { "type": "cancel_queued", "request_id": request_id }, - }); - let (status, response) = fixture.browser_action(stale).await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_stale_revision"); - assert_eq!(response["error"]["input"]["current_revision"], revision); - - // the queued task is not the active command, so stop is refused - let stop = json!({ - "api_version": 1, - "expected_revision": revision, - "operation_id": Uuid::now_v7(), - "action": { "type": "stop_active", "task_id": task_id }, - }); - let (status, response) = fixture.browser_action(stop).await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_action_not_allowed"); - - let operation_id = Uuid::now_v7(); - let cancel = json!({ - "api_version": 1, - "expected_revision": revision, - "operation_id": operation_id, - "action": { "type": "cancel_queued", "request_id": request_id }, - }); - let (status, response) = fixture.browser_action(cancel.clone()).await; - assert_eq!(status, StatusCode::OK, "{response}"); - assert_eq!(response["api_version"], 1); - assert_eq!( - response["requests"][0]["state"]["type"], - "cancelled_before_launch" - ); - let intent = call(&fixture.state.store, |reply| { - StoreMsg::GetCancellationRequest { - task: task_id, - reply, - } - }) - .await - .unwrap(); - assert!(intent.is_some(), "the origin must own the cancellation"); - - // an exact retry replays without a revision check; changed content conflicts - let (status, response) = fixture.browser_action(cancel).await; - assert_eq!(status, StatusCode::OK, "{response}"); - let changed = json!({ - "api_version": 1, - "expected_revision": revision, - "operation_id": operation_id, - "action": { "type": "stop_active", "task_id": task_id }, - }); - let (status, response) = fixture.browser_action(changed).await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_operation_conflict"); - assert_eq!(fixture.control_operation_count(), 1); - - // the CLI cancel route uses the same operation identity rules - let (status, response) = fixture - .socket_post( - &format!( - "/v1/resources/{}/requests/{}/cancel", - fixture.resource.as_uuid(), - request_id.0 - ), - json!({ "api_version": 1, "operation_id": operation_id, "expected_revision": revision }), - ) - .await; - assert_eq!(status, StatusCode::OK, "{response}"); - fixture.stop().await; -} - -#[tokio::test] -async fn queued_move_action_route_reorders_and_replays_without_an_external_effect() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let (first_request, _) = fixture.submit_request().await; - let (second_request, _) = fixture.submit_request().await; - let detail = fixture.detail().await; - let revision = detail["resource"]["state_revision"].as_u64().unwrap(); - let operation_id = Uuid::now_v7(); - let move_action = json!({ - "api_version": 1, - "expected_revision": revision, - "operation_id": operation_id, - "action": { - "type": "move_queued", - "request_id": second_request, - "placement": { "type": "front" }, - }, - }); - - let (status, response) = fixture.browser_action(move_action.clone()).await; - assert_eq!(status, StatusCode::OK, "{response}"); - assert_eq!( - response["requests"][0]["request_id"], - second_request.0.to_string() - ); - assert_eq!( - response["requests"][1]["request_id"], - first_request.0.to_string() - ); - assert_eq!(response["resource"]["state_revision"], revision + 1); - - let (status, replayed) = fixture.browser_action(move_action).await; - assert_eq!(status, StatusCode::OK, "{replayed}"); - assert_eq!( - replayed["requests"][0]["request_id"], - second_request.0.to_string() - ); - assert_eq!(replayed["resource"]["state_revision"], revision + 1); - - let changed = json!({ - "api_version": 1, - "expected_revision": revision, - "operation_id": operation_id, - "action": { - "type": "move_queued", - "request_id": second_request, - "placement": { "type": "back" }, - }, - }); - let (status, response) = fixture.browser_action(changed).await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_operation_conflict"); - assert_eq!(fixture.control_operation_count(), 1); - fixture.stop().await; -} - -#[tokio::test] -async fn queued_cancel_race_reports_a_definite_activated_request_on_retry() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let (request_id, task_id) = fixture.submit_request().await; - let revision = fixture.detail().await["resource"]["state_revision"] - .as_u64() - .unwrap(); - let operation_id = Uuid::now_v7(); - let control = ResourceControlRequest { - resource_id: fixture.resource, - expected_revision: ResourceRevision::new(revision), - action: BrowserResourceAction::CancelQueued { request_id }, - }; - call(&fixture.state.store, |reply| { - StoreMsg::BeginResourceControl { - authority_machine: fixture.local(), - operation_id, - request: Box::new(control), - attempt_id: DeliveryAttemptId::new(), - reply, - } - }) - .await - .unwrap(); - - // activation can win after the control receipt commits but before its origin intent arrives - crate::store::mark_request_assigned_for_race(&fixture.state.home.db_path(), request_id); - - let cancel = json!({ - "api_version": 1, - "expected_revision": revision, - "operation_id": operation_id, - "action": { "type": "cancel_queued", "request_id": request_id }, - }); - for _ in 0..2 { - let (status, response) = fixture.browser_action(cancel.clone()).await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_action_not_allowed"); - } - let intent = call(&fixture.state.store, |reply| { - StoreMsg::GetCancellationRequest { - task: task_id, - reply, - } - }) - .await - .unwrap(); - assert!(intent.is_none()); - assert_eq!(fixture.control_operation_count(), 1); - fixture.stop().await; -} - -#[tokio::test] -async fn renotify_reserves_one_explicit_attempt_without_resetting_the_budget() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - // an unreachable destination fails the attempt without starting Codex - let destination = SupervisorAddress { - machine: MachineId::new(), - thread: THREAD.parse().unwrap(), - }; - let notice = fixture.seed_failed_return_notice(destination); - - let (status, pending) = fixture - .socket_get(&format!( - "{RESOURCE_PENDING_PATH}?machine={}&thread={THREAD}", - fixture.local() - )) - .await; - assert_eq!(status, StatusCode::OK, "{pending}"); - let action = &pending["actions"][0]; - assert_eq!(action["action_id"], json!(notice.action_id)); - assert_eq!(action["phase"]["type"], "return_required"); - assert_eq!(action["return_context"]["type"], "idle"); - assert_eq!(action["notice"]["delivery"]["type"], "failed"); - let (_, list) = fixture.socket_get("/v1/resources").await; - assert_eq!( - list["resources"][0]["attention"]["code"], - "notice_delivery_failed" - ); - assert_eq!(list["unavailable_authorities"], json!([])); - - let operation_id = Uuid::now_v7(); - let renotify = json!({ - "api_version": 1, - "expected_revision": 0, - "operation_id": operation_id, - "action": { "type": "renotify", "notice_id": notice.id }, - }); - let (status, response) = fixture.browser_action(renotify.clone()).await; - assert_eq!(status, StatusCode::OK, "{response}"); - let delivery = &response["notices"][0]["delivery"]; - assert_eq!(delivery["type"], "failed"); - assert_eq!(delivery["attempts"], 4); - assert!( - delivery["last_error"] - .as_str() - .unwrap() - .contains("Fleet is disabled"), - "{delivery}" - ); - - // a replay reports the settled attempt and never sends another one - let (status, response) = fixture.browser_action(renotify).await; - assert_eq!(status, StatusCode::OK, "{response}"); - assert_eq!(response["notices"][0]["delivery"]["attempts"], 4); - - // a new explicit operation is one more attempt, not a reset budget - let again = json!({ - "api_version": 1, - "expected_revision": 0, - "operation_id": Uuid::now_v7(), - "action": { "type": "renotify", "notice_id": notice.id }, - }); - let (status, response) = fixture.browser_action(again).await; - assert_eq!(status, StatusCode::OK, "{response}"); - assert_eq!(response["notices"][0]["delivery"]["attempts"], 5); - - let unknown_notice = json!({ - "api_version": 1, - "expected_revision": 0, - "operation_id": Uuid::now_v7(), - "action": { "type": "renotify", "notice_id": NoticeId::new() }, - }); - let (status, response) = fixture.browser_action(unknown_notice).await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_action_not_allowed"); - fixture.stop().await; -} - -#[tokio::test] -async fn supervisor_replacement_retargets_undelivered_notices_and_checks_revision() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let notice = fixture.seed_failed_return_notice(SupervisorAddress { - machine: fixture.local(), - thread: THREAD.parse().unwrap(), - }); - let replacement = SupervisorAddress { - machine: fixture.local(), - thread: ThreadId(Uuid::now_v7()), - }; - let path = format!("/v1/resources/{}/supervisor", fixture.resource.as_uuid()); - let body = json!({ "api_version": 1, "expected_revision": 0, "supervisor": replacement }); - - let (status, response) = fixture.socket_post(&path, body.clone()).await; - assert_eq!(status, StatusCode::OK, "{response}"); - assert_eq!(response["resource"]["supervisor"], json!(replacement)); - assert_eq!(response["resource"]["assignment_revision"], 1); - assert_eq!(response["notices"][0]["id"], json!(notice.id)); - assert_eq!(response["notices"][0]["destination"], json!(replacement)); - assert_eq!(response["notices"][0]["assignment_revision"], 1); - - // an exact retry after a lost response finds the saved assignment - let (status, response) = fixture.socket_post(&path, body).await; - assert_eq!(status, StatusCode::OK, "{response}"); - assert_eq!(response["resource"]["assignment_revision"], 1); - - let stale = json!({ - "api_version": 1, - "expected_revision": 9, - "supervisor": { "machine": fixture.local(), "thread": Uuid::now_v7() }, - }); - let (status, response) = fixture.socket_post(&path, stale).await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_stale_revision"); - fixture.stop().await; -} - -#[tokio::test] -async fn registration_retry_uses_the_first_supervisor_after_reassignment() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let replacement = SupervisorAddress { - machine: fixture.local(), - thread: ThreadId(Uuid::now_v7()), - }; - let path = format!("/v1/resources/{}/supervisor", fixture.resource.as_uuid()); - let (status, response) = fixture - .socket_post( - &path, - json!({ "api_version": 1, "expected_revision": 0, "supervisor": replacement }), - ) - .await; - assert_eq!(status, StatusCode::OK, "{response}"); - - let original = json!({ - "api_version": 1, - "spec": { - "id": fixture.resource, - "display_name": "RTX 5090", - "supervisor": { "machine": fixture.local(), "thread": THREAD }, - } - }); - let (status, response) = fixture - .socket_post(RESOURCE_REGISTER_PATH, original.clone()) - .await; - assert_eq!(status, StatusCode::OK, "{response}"); - assert_eq!(response["resource"]["supervisor"], json!(replacement)); - - let mut changed = original; - changed["spec"]["supervisor"] = json!(replacement); - let (status, response) = fixture.socket_post(RESOURCE_REGISTER_PATH, changed).await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_operation_conflict"); - fixture.stop().await; -} - -#[tokio::test] -async fn reads_are_read_only_and_unowned_writes_fail_closed() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - fixture.submit_request().await; - let before = fixture.detail().await; - let dashboard_headers = [("host", fixture.dashboard.to_string())]; - for path in [ - "/v1/resources".to_owned(), - format!("/v1/resources/{}", fixture.resource.as_uuid()), - format!( - "{RESOURCE_PENDING_PATH}?machine={}&thread={THREAD}", - fixture.local() - ), - ] { - let (status, body) = send(fixture.dashboard, "GET", &path, &dashboard_headers, None).await; - assert_eq!(status, StatusCode::OK, "{path} {body}"); - } - let (status, _) = send( - fixture.dashboard, - "GET", - &format!("/v1/resources/{}/actions", fixture.resource.as_uuid()), - &dashboard_headers, - None, - ) - .await; - assert!(status.is_client_error(), "{status}"); - let after = fixture.detail().await; - assert_eq!(before["resource"], after["resource"]); - assert_eq!(before["requests"], after["requests"]); - assert_eq!(fixture.control_operation_count(), 0); - - let (status, response) = fixture - .socket_get(&format!("/v1/resources/{}", ResourceId::new().as_uuid())) - .await; - assert_eq!(status, StatusCode::NOT_FOUND, "{response}"); - assert_eq!(error_code(&response), "resource_not_found"); - - let (status, response) = fixture - .socket_post( - &format!("/v1/resources/{}/background", fixture.resource.as_uuid()), - json!({ - "api_version": 1, - "request_id": RequestId::new(), - "spec": { - "api_version": 1, - "thread": THREAD, - "name": "trainer", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": ["/bin/echo", "train"] } - }, - "env": { "path": fixture.bin, "home": "/tmp" }, - "callback_cwd": fixture.callback_cwd, - }), - ) - .await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_operation_unavailable"); - let (_, tasks) = fixture.socket_get("/v1/tasks").await; - assert_eq!( - tasks["tasks"], - json!([]), - "a command without a verifiable ownership contract must not start a task" - ); - - let (status, response) = fixture - .socket_post( - RESOURCE_REGISTER_PATH, - json!({ - "api_version": 1, - "spec": { - "id": fixture.resource, - "display_name": "another GPU", - "supervisor": { "machine": fixture.local(), "thread": THREAD }, - } - }), - ) - .await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_operation_conflict"); - fixture.stop().await; -} - -fn response(status: StatusCode, body: &Value) -> ClusterResponse { - ClusterResponse { - status, - body: Bytes::from(serde_json::to_vec(body).unwrap()), - } -} - -#[test] -fn forwarded_control_keeps_typed_refusals_and_reports_unknown_outcomes() { - let resource = ResourceId::new(); - let authority = MachineId::new(); - let operation = Some(Uuid::now_v7()); - - let stale = AppError::ResourceStaleRevision { - resource, - expected: 3, - current: 4, - }; - let decoded = decode_control_response( - resource, - authority, - operation, - &response(stale.http_status(), &stale.to_json()), - ); - assert!(matches!( - decoded, - Err(AppError::ResourceStaleRevision { - expected: 3, - current: 4, - .. - }) - )); - - let conflict = AppError::ResourceOperationConflict { - resource, - operation, - message: "changed".into(), - }; - let decoded = decode_control_response( - resource, - authority, - operation, - &response(conflict.http_status(), &conflict.to_json()), - ); - assert!(matches!( - decoded, - Err(AppError::ResourceOperationConflict { .. }) - )); - - // a server failure or an unreadable success may follow a committed mutation - let internal = AppError::Internal { - message: "store".into(), - }; - for unknown in [ - response(internal.http_status(), &internal.to_json()), - response(StatusCode::OK, &json!({ "api_version": 1 })), - response(StatusCode::BAD_GATEWAY, &json!("proxy")), - ] { - let decoded = decode_control_response(resource, authority, operation, &unknown); - assert!( - matches!(decoded, Err(AppError::ResourceOutcomeUnknown { operation: found, .. }) if found == operation), - "{decoded:?}" - ); - } -} - -#[tokio::test] -async fn background_route_launches_one_co_located_trainer_and_refuses_remote_supervisors() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - crate::runner::set_task_run_executable_for_tests(assert_cmd::cargo::cargo_bin("homebased")); - let fixture = Fixture::new().await; - let root = fixture.callback_cwd.canonicalize().unwrap(); - let trainer = FakeTrainer::new(&root); - let body = |request_id: RequestId| { - json!({ - "api_version": 1, - "request_id": request_id, - "spec": trainer.spec(THREAD.parse().unwrap(), "direct segment trainer"), - "env": trainer.env, - "callback_cwd": root, - }) - }; - let path = format!("/v1/resources/{}/background", fixture.resource.as_uuid()); - let request_id = RequestId::new(); - - let (status, first) = fixture.socket_post(&path, body(request_id)).await; - assert_eq!(status, StatusCode::OK, "{first}"); - assert_eq!(first["outcome"]["type"], "inserted"); - assert_eq!(first["resource"]["id"], json!(fixture.resource)); - // the bound launch is not registered until its confirmed start - assert_eq!(first["resource"]["registered_background_task"], Value::Null); - let (status, retry) = fixture.socket_post(&path, body(request_id)).await; - assert_eq!(status, StatusCode::OK, "{retry}"); - assert_eq!(retry["outcome"]["type"], "existing"); - assert_eq!(retry["task_id"], first["task_id"]); - - // the trainer holds no attempt lock here, so no association can be asserted for it - let (status, response) = fixture - .socket_post( - &format!( - "{path}/{}/trainer-attempt", - first["task_id"].as_str().unwrap() - ), - json!({ - "api_version": 1, - "attempt_binding": crate::resource::ownership_lock::test_support::attempt_binding( - "attempt-1" - ), - }), - ) - .await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_action_not_allowed"); - - // a remote supervisor thread would need a saved origin route on its own machine - let remote = ResourceId::new(); - let (status, response) = fixture - .socket_post( - RESOURCE_REGISTER_PATH, - json!({ - "api_version": 1, - "spec": { - "id": remote, - "display_name": "remote-supervised GPU", - "supervisor": { "machine": MachineId::new(), "thread": THREAD }, - } - }), - ) - .await; - assert_eq!(status, StatusCode::OK, "{response}"); - let (status, response) = fixture - .socket_post( - &format!("/v1/resources/{}/background", remote.as_uuid()), - body(RequestId::new()), - ) - .await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_operation_unavailable"); - let (_, tasks) = fixture.socket_get("/v1/tasks").await; - assert_eq!(tasks["tasks"].as_array().map(Vec::len), Some(1), "{tasks}"); - - std::fs::write(&trainer.gate, b"").unwrap(); - fixture.stop().await; -} - -#[tokio::test] -async fn initial_idle_saves_one_receipt_for_a_resource_with_no_history() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let (status, detail) = fixture - .socket_get(&format!("/v1/resources/{}", fixture.resource.as_uuid())) - .await; - assert_eq!(status, StatusCode::OK, "{detail}"); - let revision = detail["resource"]["state_revision"].clone(); - let attestation = json!({ - "operation_id": OperatorAttestationId::new(), - "resource_id": fixture.resource, - "authority_machine": fixture.local(), - "expected_state_revision": revision, - "observation": "nvidia-smi on the authority shows no compute processes", - "confirmation": "operator_confirmed_gpu_free" - }); - let body = json!({ "api_version": 1, "attestation": attestation }); - let path = format!("/v1/resources/{}/initial-idle", fixture.resource.as_uuid()); - - // the browser dashboard cannot record a human attestation - let (status, response) = send( - fixture.dashboard, - "POST", - &path, - &[ - ("host", fixture.dashboard.to_string()), - ("origin", fixture.origin()), - ("content-type", "application/json".into()), - ], - Some(body.clone()), - ) - .await; - assert_eq!(status, StatusCode::NOT_FOUND, "{response}"); - - let (status, first) = fixture.socket_post(&path, body.clone()).await; - assert_eq!(status, StatusCode::OK, "{first}"); - assert_eq!(first["replayed"], false); - assert_eq!(first["receipt"]["attestation"], attestation); - let (status, retry) = fixture.socket_post(&path, body).await; - assert_eq!(status, StatusCode::OK, "{retry}"); - assert_eq!(retry["replayed"], true); - assert_eq!(retry["receipt"], first["receipt"]); - - let mut second = attestation.clone(); - second["operation_id"] = json!(OperatorAttestationId::new()); - second["expected_state_revision"] = first["receipt"]["state_revision"].clone(); - let (status, response) = fixture - .socket_post(&path, json!({ "api_version": 1, "attestation": second })) - .await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_action_not_allowed"); - fixture.stop().await; -} - -impl Fixture { - async fn submit_container( - &self, - request_id: RequestId, - cwd: &str, - source: &std::path::Path, - ) -> (StatusCode, Value) { - self.socket_post( - &format!("/v1/resources/{}/requests", self.resource.as_uuid()), - json!({ - "api_version": 1, - "request_id": request_id, - "spec": { - "api_version": 1, - "thread": THREAD, - "name": "container benchmark", - "cwd": cwd, - "timeout": "4h", - "workload": { - "type": "container", - "image": format!("bench@sha256:{}", "0".repeat(64)), - "gpus": "all", - "memory": "8g", - "mounts": [{ "source": source, "target": "/scratch" }] - } - }, - "env": { "path": self.bin, "home": "/tmp" }, - "callback_cwd": self.callback_cwd, - }), - ) - .await - } -} - -// the incident spec: `cwd` named the container mount target instead of a host path -#[tokio::test] -async fn request_submit_rejects_a_container_path_as_cwd_and_names_the_mount_source() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let source = fixture.callback_cwd.clone(); - let request_id = RequestId::new(); - - // a retry with the same request id answers from the saved rejection - for _ in 0..2 { - let (status, response) = fixture - .submit_container(request_id, "/scratch/runs", &source) - .await; - assert_eq!(status, StatusCode::BAD_REQUEST, "{response}"); - assert_eq!(error_code(&response), "invalid_cwd"); - let input = &response["error"]["input"]; - assert_eq!(input["pointer"], "/cwd"); - assert_eq!(input["value"], "/scratch/runs"); - assert_eq!(input["problem"], "not_found"); - assert_eq!(input["suggested_cwd"], json!(source.join("runs"))); - let message = response["error"]["message"].as_str().unwrap(); - assert!(message.contains("cwd is a host path"), "{message}"); - } - assert_eq!(fixture.detail().await["requests"], json!([])); - - // the mount source itself is a valid cwd - let (status, response) = fixture - .submit_container(RequestId::new(), source.to_str().unwrap(), &source) - .await; - assert_eq!(status, StatusCode::OK, "{response}"); - assert_eq!(response["outcome"]["type"], "waiting"); - fixture.stop().await; -} - -#[tokio::test] -async fn request_submit_rejects_a_missing_mount_source() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let missing = fixture.callback_cwd.join("missing-source"); - let (status, response) = fixture - .submit_container( - RequestId::new(), - fixture.callback_cwd.to_str().unwrap(), - &missing, - ) - .await; - assert_eq!(status, StatusCode::BAD_REQUEST, "{response}"); - assert_eq!(error_code(&response), "invalid_spec"); - assert_eq!( - response["error"]["input"]["pointer"], - "/workload/mounts/0/source" - ); - assert_eq!(fixture.detail().await["requests"], json!([])); - fixture.stop().await; -} - -#[tokio::test] -async fn cancelling_an_assigned_request_names_the_task_cancel_command() { - let _serial = SUPERVISOR_TEST_LOCK.lock().await; - let fixture = Fixture::new().await; - let (request_id, task_id) = fixture.submit_request().await; - crate::store::mark_request_assigned_for_race(&fixture.state.home.db_path(), request_id); - let revision = fixture.detail().await["resource"]["state_revision"] - .as_u64() - .unwrap(); - - let (status, response) = fixture - .socket_post( - &format!( - "/v1/resources/{}/requests/{}/cancel", - fixture.resource.as_uuid(), - request_id.0 - ), - json!({ - "api_version": 1, - "operation_id": Uuid::now_v7(), - "expected_revision": revision, - }), - ) - .await; - assert_eq!(status, StatusCode::CONFLICT, "{response}"); - assert_eq!(error_code(&response), "resource_action_not_allowed"); - let message = response["error"]["message"].as_str().unwrap(); - assert!( - message.contains(&format!("homebased task cancel {task_id}")), - "{message}" - ); - fixture.stop().await; -} diff --git a/src/daemon/resource_background.rs b/src/daemon/resource_background.rs deleted file mode 100644 index 2ab22b8..0000000 --- a/src/daemon/resource_background.rs +++ /dev/null @@ -1,886 +0,0 @@ -//! Two-machine first background launch: supervisor-owned route and authority acceptance -//! -//! The supervisor machine reads the resource from its fixed authority, saves one -//! fixed-ID origin route with the full spec, its callback context, and the exact -//! supervisor assignment and resource revision, and only then sends the launch -//! The authority reads that saved route back as evidence before it accepts. A -//! lost reply or restart repeats the same launch, and the authority answers an -//! exact retry from its receipt, so no second trainer starts -//! -//! The same document binds the trainer attempt of the running task. The -//! supervisor names only the attempt; the authority reads the attempt request -//! and probes the held ownership lock itself - -use std::path::PathBuf; - -use axum::extract::{Path, Query, State}; -use axum::http::StatusCode; -use axum::routing::{get, post}; -use axum::{Json, Router}; -use serde::{Deserialize, Serialize}; -use serde_json::Value; -use tracing::warn; - -use super::AppState; -use super::actors::supervisor::RemoteBackgroundLaunch; -use super::actors::{StoreMsg, SupervisorMsg, call}; -use super::cluster::ReadQuery; -use crate::domain::{API_VERSION, AgentKind, TaskEnv, TaskId}; -use crate::error::AppError; -use crate::fleet::http::{ClusterClient, ClusterResponse}; -use crate::invocation::resolve_agent_binary; -use crate::machine::MachineId; -use crate::resource::ResourceId; -use crate::resource::api::{ - ResourceBackgroundSubmitOutcome, ResourceBackgroundSubmitResponse, TrainerAttemptResponse, -}; -use crate::resource::background_launch::{ - BackgroundLaunchBinding, BackgroundSupervisorAssignment, RESOURCE_BACKGROUND_PATH, - RESOURCE_BACKGROUND_PROTOCOL_VERSION, RESOURCE_BACKGROUND_ROUTE_PROOF_PATH, - RemoteBackgroundLaunchReceipt, ResourceBackgroundOperation, ResourceBackgroundOutcome, - ResourceBackgroundRejection, ResourceBackgroundRequest, ResourceBackgroundResponse, -}; -use crate::resource::store::ResourceStoreError; -use crate::resource::trainer_publication::AttemptBinding; -use crate::spec::{self, NormalizedSpec}; -use crate::store::{ - BackgroundLaunchAcceptance, BackgroundLaunchError, ResourceBackgroundRouteResult, -}; -use crate::submission::{ - CallbackContext, NewResourceBackgroundRoute, OriginRoute, RequestId, - ResourceBackgroundRoutePhase, ResourceBackgroundRouteProof, SubmissionState, - normalized_spec_sha256, -}; - -/// Saved route proof returned by the supervisor machine -#[derive(Debug, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceBackgroundRouteProofBody { - /// Public API version - pub api_version: u32, - /// Proof for the named task, when this machine saved a valid background route - pub proof: Option, -} - -/// Socket submission of one first background launch whose authority is remote -pub(super) struct RemoteBackgroundSubmit { - /// Resource named by the socket path - pub(super) resource: ResourceId, - /// Fixed resource authority located by Fleet or the saved route - pub(super) authority: MachineId, - /// Stable caller retry identity - pub(super) request_id: RequestId, - /// Full normalized trainer spec - pub(super) spec: NormalizedSpec, - /// Environment saved for the supervisor-side callback - pub(super) env: TaskEnv, - /// Absolute directory used to resolve the supervisor-side Codex executable - pub(super) callback_cwd: PathBuf, -} - -/// Cluster routes: authority operations and supervisor route evidence -pub(super) fn cluster_routes() -> Router { - Router::new() - .route(RESOURCE_BACKGROUND_PATH, post(receive)) - .route( - &format!("{RESOURCE_BACKGROUND_ROUTE_PROOF_PATH}/{{task}}"), - get(route_proof), - ) -} - -// ---- authority side ---- - -async fn receive( - State(state): State, - Json(request): Json, -) -> Result, AppError> { - state - .machine - .identity - .check_destination(request.destination_machine)?; - super::cluster::check_protocol(request.protocol_version, request.source_machine)?; - request.validate().map_err(|error| AppError::Usage { - message: format!("invalid remote background request: {error}"), - })?; - - let outcome = match &request.operation { - ResourceBackgroundOperation::Launch { spec, .. } => { - let Some(receipt) = request.launch_receipt() else { - return Err(AppError::Internal { - message: "a launch operation has no launch receipt".into(), - }); - }; - accept_launch(&state, receipt, spec.clone()).await? - } - ResourceBackgroundOperation::BindTrainerAttempt { - task_id, - attempt_binding, - } => { - bind_attempt( - &state, - request.assignment, - *task_id, - attempt_binding.clone(), - ) - .await? - } - }; - Ok(Json(ResourceBackgroundResponse::new(&request, outcome))) -} - -/// Accept one launch after the supervisor machine proves its saved route -async fn accept_launch( - state: &AppState, - receipt: RemoteBackgroundLaunchReceipt, - spec: NormalizedSpec, -) -> Result { - let proof = fetch_route_proof(state, receipt.binding.origin_machine(), receipt.task_id).await?; - if let Err(reason) = check_route_evidence(&receipt, proof) { - warn!( - request = %receipt.request_id.0, - task = %receipt.task_id, - ?reason, - "background launch route evidence refused" - ); - return Ok(ResourceBackgroundOutcome::Rejected { reason }); - } - - let accepted = call(&state.supervisor, |reply| { - SupervisorMsg::LaunchRemoteBackground { - launch: Box::new(RemoteBackgroundLaunch { receipt, spec }), - reply, - } - }) - .await?; - let acceptance = match accepted { - Ok(BackgroundLaunchAcceptance::Inserted { .. }) => { - ResourceBackgroundSubmitOutcome::Inserted - } - Ok(BackgroundLaunchAcceptance::Existing { state, .. }) => { - ResourceBackgroundSubmitOutcome::Existing { state } - } - Ok(BackgroundLaunchAcceptance::UnsupportedRemoteSupervisor { .. }) => { - return Ok(ResourceBackgroundOutcome::Rejected { - reason: ResourceBackgroundRejection::NotCurrentSupervisor, - }); - } - Err(error) => { - return Ok(ResourceBackgroundOutcome::Rejected { - reason: launch_rejection(error)?, - }); - } - }; - let resource = - super::resource_api::local_resource(state, receipt.binding.assignment.resource_id).await?; - Ok(ResourceBackgroundOutcome::Accepted { - receipt, - acceptance, - resource, - }) -} - -/// Keep storage failures retryable and turn every domain refusal into a typed rejection -fn launch_rejection(error: BackgroundLaunchError) -> Result { - use BackgroundLaunchError as Error; - use ResourceBackgroundRejection as Rejection; - - Ok(match error { - Error::Resource( - ResourceStoreError::ResourceNotFound | ResourceStoreError::WrongAuthority { .. }, - ) => Rejection::ResourceNotFound, - // the request already names the assignment's thread, so another thread - // means the assignment changed - Error::NotCurrentSupervisor | Error::NotSupervisorThread { .. } => { - Rejection::NotCurrentSupervisor - } - Error::StaleRevision { expected, actual } => Rejection::StaleRevision { expected, actual }, - Error::ActiveLoan { loan_id } => Rejection::ActiveLoan { loan_id }, - Error::QueuedWorkAhead { request_id } => Rejection::QueuedWorkAhead { request_id }, - Error::BackgroundTaskActive { task_id, state } => { - Rejection::BackgroundTaskActive { task_id, state } - } - Error::BackgroundTaskMissing { task_id } => Rejection::BackgroundTaskMissing { task_id }, - Error::LaunchPending { task_id } => Rejection::LaunchPending { task_id }, - Error::PredecessorReleaseUnproven { task_id } - | Error::PredecessorOwnershipUnproven { task_id, .. } => { - Rejection::PredecessorReleaseUnproven { task_id } - } - Error::ConflictingRetry { .. } => Rejection::ConflictingRetry, - Error::IdentityConflict { .. } | Error::Identity(crate::store::IdentityError::Conflict) => { - Rejection::IdentityConflict - } - Error::UnsupportedCommand(error) => Rejection::UnsupportedCommand { - reason: error.to_string(), - }, - // the spec or executor environment cannot run here; nothing was written - Error::TaskRecords( - error @ (AppError::InvalidCwd { .. } - | AppError::ExecutableMissing { .. } - | AppError::InvalidSpec { .. } - | AppError::Usage { .. }), - ) => Rejection::InvalidSpec { - reason: error.to_string(), - }, - error => { - return Err(AppError::Internal { - message: format!("background launch owner failed: {error}"), - }); - } - }) -} - -/// Compare the supervisor machine's saved route with one launch request -fn check_route_evidence( - receipt: &RemoteBackgroundLaunchReceipt, - proof: Option, -) -> Result<(), ResourceBackgroundRejection> { - let proof = proof.ok_or(ResourceBackgroundRejection::RouteEvidenceMissing)?; - if proof.request != receipt.request_id - || proof.task != receipt.task_id - || proof.binding != receipt.binding - || proof.normalized_spec_sha256 != receipt.normalized_spec_sha256 - || matches!(proof.phase, ResourceBackgroundRoutePhase::Rejected { .. }) - { - return Err(ResourceBackgroundRejection::RouteEvidenceMismatch); - } - Ok(()) -} - -/// Read the background route that the supervisor machine saved for one task -async fn fetch_route_proof( - state: &AppState, - origin: MachineId, - task: TaskId, -) -> Result, AppError> { - let unavailable = |message: String| AppError::RemoteSubmissionUnavailable { - message: format!("supervisor route proof is unavailable: {message}"), - }; - let fleet = state - .fleet - .handle() - .ok_or_else(|| unavailable("fleet is disabled".into()))?; - let destination = fleet - .connect(origin) - .await - .map_err(|error| unavailable(error.to_string()))?; - let path = format!( - "{RESOURCE_BACKGROUND_ROUTE_PROOF_PATH}/{task}?api_version={API_VERSION}&destination_machine={origin}" - ); - let response = ClusterClient::default() - .get(&destination.address, &path) - .await - .map_err(|error| unavailable(error.to_string()))?; - if response.status != StatusCode::OK { - return Err(unavailable(format!("HTTP {}", response.status))); - } - let body: ResourceBackgroundRouteProofBody = serde_json::from_slice(&response.body) - .map_err(|error| unavailable(format!("invalid response: {error}")))?; - if body.api_version != API_VERSION { - return Err(unavailable("unsupported API version".into())); - } - Ok(body.proof.filter(|proof| proof.task == task)) -} - -/// Bind one trainer attempt for the current remote supervisor assignment -async fn bind_attempt( - state: &AppState, - assignment: BackgroundSupervisorAssignment, - task_id: TaskId, - attempt_binding: AttemptBinding, -) -> Result { - let resource_id = assignment.resource_id; - let resource = match super::resource_api::local_resource(state, resource_id).await { - Ok(resource) => resource, - Err(AppError::ResourceNotFound { .. }) => { - return Ok(ResourceBackgroundOutcome::Rejected { - reason: ResourceBackgroundRejection::ResourceNotFound, - }); - } - Err(error) => return Err(error), - }; - if resource.supervisor.machine == resource.authority_machine() - || !assignment.is_current(&resource) - { - return Ok(ResourceBackgroundOutcome::Rejected { - reason: ResourceBackgroundRejection::NotCurrentSupervisor, - }); - } - // the resource owner registers a confirmed start; reconcile first so a - // running launch is not refused only because its wake-up is still queued - call(&state.supervisor, |reply| { - SupervisorMsg::ReconcileResource { - id: resource_id, - reply, - } - }) - .await?; - - match super::resource_api::authority_trainer_attempt( - state, - resource_id, - task_id, - attempt_binding, - ) - .await - { - Ok(response) => { - // a release action that waited for this association can now bind its watcher - if let Err(error) = call(&state.supervisor, |reply| { - SupervisorMsg::ReconcileResource { - id: resource_id, - reply, - } - }) - .await - { - warn!(%task_id, "resource wake after trainer association: {error}"); - } - Ok(ResourceBackgroundOutcome::TrainerAttemptBound { - resource: response.resource, - task_id: response.task_id, - runtime_root: response.runtime_root, - attempt_binding: response.attempt_binding, - }) - } - Err(AppError::ResourceActionNotAllowed { message, .. }) => { - Ok(ResourceBackgroundOutcome::Rejected { - reason: ResourceBackgroundRejection::TrainerAttemptRefused { reason: message }, - }) - } - Err(AppError::ResourceOperationConflict { .. }) => { - Ok(ResourceBackgroundOutcome::Rejected { - reason: ResourceBackgroundRejection::TrainerAttemptConflict, - }) - } - Err(error) => Err(error), - } -} - -/// Serve the saved background route proof for one task owned by this supervisor machine -async fn route_proof( - State(state): State, - Path(task): Path, - Query(query): Query, -) -> Result, AppError> { - state - .machine - .identity - .check_destination(query.destination_machine)?; - super::cluster::check_api_version(query.api_version)?; - let route = call(&state.store, |reply| StoreMsg::OriginRoute { - id: task, - reply, - }) - .await?; - let local = state.machine.identity.machine; - let proof = route - .as_ref() - .filter(|route| route.task == task && route.origin_machine == local) - .and_then(ResourceBackgroundRouteProof::from_route); - Ok(Json(ResourceBackgroundRouteProofBody { - api_version: API_VERSION, - proof, - })) -} - -// ---- supervisor side ---- - -/// Submit one first background launch to a remote authority, or answer its saved route -/// -/// A new request saves its fixed-ID route before the first send. A retry with -/// the same request reuses the saved identities and spec, and different content -/// conflicts. An uncertain send leaves the route for the same retry -pub(super) async fn submit( - state: &AppState, - input: RemoteBackgroundSubmit, -) -> Result { - let _request_guard = state.locks.background_launches.lock(input.request_id).await; - let saved = call(&state.store, |reply| StoreMsg::OriginRouteByRequest { - request: input.request_id, - reply, - }) - .await?; - let route = match saved { - Some(route) => { - check_retry(&route, &input)?; - route - } - None => create_route(state, &input).await?, - }; - send_saved_launch(state, &route).await -} - -/// Retry background routes whose authority acceptance was unknown at daemon startup -pub(super) async fn recover(state: AppState) { - let routes = match call(&state.store, |reply| { - StoreMsg::UnknownResourceBackgroundRoutes { reply } - }) - .await - { - Ok(routes) => routes, - Err(error) => { - warn!("cannot scan unresolved background launch routes: {error}"); - return; - } - }; - for route in routes { - let _request_guard = state.locks.background_launches.lock(route.request).await; - if let Err(error) = send_saved_launch(&state, &route).await { - warn!(task = %route.task, "background launch recovery: {error}"); - } - } -} - -/// A retry must repeat the saved resource, authority, and full normalized spec -/// -/// The saved callback context is kept; a retry cannot move it -fn check_retry(route: &OriginRoute, input: &RemoteBackgroundSubmit) -> Result<(), AppError> { - let conflict = |message: &str| AppError::ResourceOperationConflict { - resource: input.resource, - operation: Some(input.request_id.0), - message: message.into(), - }; - let SubmissionState::ResourceBackground { binding, .. } = &route.submission else { - return Err(conflict("request id belongs to another route")); - }; - let Some(saved_spec) = route.current_spec() else { - return Err(conflict("saved background route has no normalized spec")); - }; - if binding.assignment.resource_id != input.resource - || binding.execution_machine() != input.authority - || *saved_spec != input.spec - { - return Err(conflict( - "request id was retried with a different resource, authority, or spec", - )); - } - Ok(()) -} - -/// Read the resource from its authority, then save the fixed-ID route before any send -async fn create_route( - state: &AppState, - input: &RemoteBackgroundSubmit, -) -> Result { - let resource_id = input.resource; - let unavailable = |message: String| AppError::ResourceOperationUnavailable { - resource: resource_id, - message: format!("{message}; nothing was written"), - }; - let resource = super::resource_api::remote_detail(state, resource_id) - .await? - .resource; - let local = state.machine.identity.machine; - if resource.authority_machine() != input.authority { - return Err(unavailable(format!( - "the resource authority moved from {} to {}", - input.authority, - resource.authority_machine() - ))); - } - if resource.supervisor.machine != local { - return Err(unavailable(format!( - "the supervisor thread runs on machine {}; submit the launch there", - resource.supervisor.machine - ))); - } - if input.spec.thread != resource.supervisor.thread { - return Err(AppError::ResourceActionNotAllowed { - resource: resource_id, - message: format!( - "background launch thread {} is not the supervisor thread {}", - input.spec.thread, resource.supervisor.thread - ), - }); - } - if !input.callback_cwd.is_absolute() { - return Err(AppError::InvalidSpec { - pointer: "/callback_cwd".into(), - value: serde_json::to_value(&input.callback_cwd)?, - message: "callback directory must be absolute".into(), - }); - } - spec::check_cwd(&input.callback_cwd)?; - let codex = resolve_agent_binary(AgentKind::Codex, &input.env.path, &input.callback_cwd)?; - let route = OriginRoute::new_resource_background(NewResourceBackgroundRoute { - request: input.request_id, - task: TaskId::new(), - callback: CallbackContext { - env: input.env.clone(), - cwd: input.callback_cwd.clone(), - codex: codex.into(), - }, - spec: input.spec.clone(), - binding: BackgroundLaunchBinding { - assignment: BackgroundSupervisorAssignment::of(&resource), - expected_state_revision: resource.state_revision, - }, - }) - .map_err(|error| AppError::Usage { - message: format!("background launch route is invalid: {error}"), - })?; - - match call(&state.store, |reply| StoreMsg::InsertOriginRoute { - route: Box::new(route), - reply, - }) - .await - { - Ok(saved) => Ok(saved), - Err(error) => { - // a concurrent first call with the same request may have saved it - let saved = call(&state.store, |reply| StoreMsg::OriginRouteByRequest { - request: input.request_id, - reply, - }) - .await?; - let Some(saved) = saved else { - return Err(error); - }; - check_retry(&saved, input)?; - Ok(saved) - } - } -} - -/// Send the launch of one saved route and apply a definitive answer to it -/// -/// A saved refusal is returned without a send. An accepted route is sent again -/// because the authority answers an exact retry from its receipt, which also -/// returns the current resource and task state -async fn send_saved_launch( - state: &AppState, - route: &OriginRoute, -) -> Result { - let SubmissionState::ResourceBackground { binding, phase } = &route.submission else { - return Err(AppError::ClusterTaskConflict { task: route.task }); - }; - let resource_id = binding.assignment.resource_id; - if let ResourceBackgroundRoutePhase::Rejected { reason } = phase { - return Err(rejection_error(resource_id, Some(route.request.0), reason)); - } - let spec = route - .current_spec() - .ok_or(AppError::ClusterTaskConflict { task: route.task })?; - let expected = RemoteBackgroundLaunchReceipt { - binding: *binding, - request_id: route.request, - task_id: route.task, - normalized_spec_sha256: normalized_spec_sha256(spec)?, - }; - let operation = ResourceBackgroundOperation::Launch { - expected_state_revision: binding.expected_state_revision, - request_id: route.request, - task_id: route.task, - spec: spec.clone(), - normalized_spec_sha256: expected.normalized_spec_sha256, - }; - let unresolved = |message: String| { - if matches!(phase, ResourceBackgroundRoutePhase::Accepted) { - AppError::ResourceAuthorityUnavailable { - resource: resource_id, - machine: binding.execution_machine(), - message: format!( - "the launch of task {} was accepted, but its current state is unavailable: \ - {message}", - route.task - ), - } - } else { - unknown(route, resource_id, message) - } - }; - let outcome = send(state, binding.assignment, operation) - .await - .map_err(|error| unresolved(error.to_string()))?; - - match outcome { - ResourceBackgroundOutcome::Accepted { - receipt, - acceptance, - resource, - } => { - if receipt != expected || resource.id != resource_id { - return Err(unresolved( - "the authority answered for another launch".into(), - )); - } - resolve( - state, - route, - resource_id, - ResourceBackgroundRouteResult::Accepted(receipt), - ) - .await?; - Ok(ResourceBackgroundSubmitResponse { - api_version: API_VERSION, - request_id: route.request, - task_id: route.task, - resource, - outcome: acceptance, - }) - } - ResourceBackgroundOutcome::Rejected { reason } => { - let error = rejection_error(resource_id, Some(route.request.0), &reason); - resolve( - state, - route, - resource_id, - ResourceBackgroundRouteResult::Rejected(reason), - ) - .await?; - Err(error) - } - ResourceBackgroundOutcome::TrainerAttemptBound { .. } => Err(unresolved( - "the authority answered a launch with another outcome".into(), - )), - } -} - -async fn resolve( - state: &AppState, - route: &OriginRoute, - resource: ResourceId, - result: ResourceBackgroundRouteResult, -) -> Result { - call(&state.store, |reply| { - StoreMsg::ResolveResourceBackgroundRoute { - task: route.task, - result: Box::new(result), - reply, - } - }) - .await - .map_err(|error| { - unknown( - route, - resource, - format!("cannot save the authority result: {error}"), - ) - }) -} - -/// Bind one trainer attempt on the remote authority for this supervisor machine -/// -/// The request names the current supervisor assignment read from the authority -/// An exact retry returns the saved association; another attempt conflicts -pub(super) async fn bind_trainer_attempt( - state: &AppState, - resource_id: ResourceId, - task_id: TaskId, - attempt_binding: AttemptBinding, -) -> Result { - let resource = super::resource_api::remote_detail(state, resource_id) - .await? - .resource; - if resource.supervisor.machine != state.machine.identity.machine { - return Err(AppError::ResourceOperationUnavailable { - resource: resource_id, - message: format!( - "the supervisor thread runs on machine {}; bind the trainer attempt there", - resource.supervisor.machine - ), - }); - } - let operation = ResourceBackgroundOperation::BindTrainerAttempt { - task_id, - attempt_binding: attempt_binding.clone(), - }; - let unknown = |message: String| AppError::ResourceOutcomeUnknown { - resource: resource_id, - operation: None, - message: format!("{message}; retry the same trainer attempt"), - }; - let outcome = send( - state, - BackgroundSupervisorAssignment::of(&resource), - operation, - ) - .await - .map_err(|error| unknown(error.to_string()))?; - match outcome { - ResourceBackgroundOutcome::TrainerAttemptBound { - resource, - task_id: bound_task, - runtime_root, - attempt_binding: bound_attempt, - } => { - if bound_task != task_id - || bound_attempt != attempt_binding - || resource.id != resource_id - { - return Err(unknown("the authority answered for another attempt".into())); - } - Ok(TrainerAttemptResponse { - api_version: API_VERSION, - resource, - task_id, - runtime_root, - attempt_binding, - }) - } - ResourceBackgroundOutcome::Rejected { reason } => { - Err(rejection_error(resource_id, None, &reason)) - } - ResourceBackgroundOutcome::Accepted { .. } => Err(unknown( - "the authority answered a trainer attempt with another outcome".into(), - )), - } -} - -/// Send one operation to the authority and decode its strict typed answer -async fn send( - state: &AppState, - assignment: BackgroundSupervisorAssignment, - operation: ResourceBackgroundOperation, -) -> Result { - let authority = assignment.authority_machine; - let unavailable = |message: String| AppError::MachineUnavailable { - machine: authority, - message, - }; - let fleet = state - .fleet - .handle() - .ok_or_else(|| unavailable("fleet is disabled".into()))?; - let destination = fleet - .connect(authority) - .await - .map_err(|error| unavailable(error.to_string()))?; - let request = ResourceBackgroundRequest::new(destination.protocol.0, assignment, operation); - let response = ClusterClient::default() - .post_json(&destination.address, RESOURCE_BACKGROUND_PATH, &request) - .await - .map_err(|error| unavailable(error.to_string()))?; - decode_response(&request, response) -} - -fn decode_response( - request: &ResourceBackgroundRequest, - response: ClusterResponse, -) -> Result { - if !response.status.is_success() { - return Err(crate::client::map_error(response.status, &response.body)); - } - let invalid = |message: String| AppError::RemoteSubmissionUnavailable { - message: format!("invalid resource authority response: {message}"), - }; - let value: Value = - serde_json::from_slice(&response.body).map_err(|error| invalid(error.to_string()))?; - let body: ResourceBackgroundResponse = - serde_json::from_value(value).map_err(|error| invalid(error.to_string()))?; - if body.api_version != API_VERSION - || body.protocol_version != request.protocol_version - || body.background_protocol_version != RESOURCE_BACKGROUND_PROTOCOL_VERSION - || body.destination_machine != request.destination_machine - || body.resource_id != request.assignment.resource_id - { - return Err(invalid("response names another route or version".into())); - } - Ok(body.outcome) -} - -/// Map one definitive authority refusal to the socket error for its resource -fn rejection_error( - resource: ResourceId, - operation: Option, - reason: &ResourceBackgroundRejection, -) -> AppError { - use ResourceBackgroundRejection as Rejection; - - let not_allowed = |message: String| AppError::ResourceActionNotAllowed { resource, message }; - let conflict = |message: String| AppError::ResourceOperationConflict { - resource, - operation, - message, - }; - match reason { - Rejection::ResourceNotFound => AppError::ResourceNotFound { resource }, - Rejection::StaleRevision { expected, actual } => AppError::ResourceStaleRevision { - resource, - expected: expected.get(), - current: actual.get(), - }, - Rejection::NotCurrentSupervisor => { - not_allowed("the request does not come from the current supervisor assignment".into()) - } - Rejection::ActiveLoan { loan_id } => { - not_allowed(format!("loan {} owns the resource", loan_id.as_uuid())) - } - Rejection::QueuedWorkAhead { request_id } => not_allowed(format!( - "queued resource request {} is ahead of the background launch", - request_id.0 - )), - Rejection::BackgroundTaskActive { task_id, state } => { - not_allowed(format!("registered background task {task_id} is {state}")) - } - Rejection::BackgroundTaskMissing { task_id } => not_allowed(format!( - "registered background task {task_id} has no task record" - )), - Rejection::LaunchPending { task_id } => { - not_allowed(format!("background launch task {task_id} is still pending")) - } - Rejection::PredecessorReleaseUnproven { task_id } => not_allowed(format!( - "background task {task_id} has no verified resource release" - )), - Rejection::TrainerAttemptRefused { reason } => not_allowed(reason.clone()), - Rejection::RouteEvidenceMissing | Rejection::RouteEvidenceMismatch => conflict(format!( - "the authority did not accept this machine's saved route: {reason:?}" - )), - Rejection::ConflictingRetry | Rejection::IdentityConflict => conflict(format!( - "the request or task identity is already used: {reason:?}" - )), - Rejection::TrainerAttemptConflict => { - conflict("the task is associated with another trainer attempt".into()) - } - Rejection::UnsupportedCommand { reason } => AppError::ResourceOperationUnavailable { - resource, - message: format!( - "a background launch accepts only the maintained direct-segment trainer, whose \ - ownership lock release proof can verify; this command does not match it: {reason}" - ), - }, - Rejection::InvalidSpec { reason } => AppError::Usage { - message: format!("the resource authority cannot run the spec: {reason}"), - }, - } -} - -fn unknown(route: &OriginRoute, resource: ResourceId, message: String) -> AppError { - AppError::ResourceOutcomeUnknown { - resource, - operation: Some(route.request.0), - message: format!( - "{message}; the saved route keeps task {} bound to this request; retry with the same \ - request id {}", - route.task, route.request.0 - ), - } -} - -#[cfg(test)] -mod tests { - use super::{launch_rejection, rejection_error}; - use crate::domain::TaskId; - use crate::error::AppError; - use crate::resource::ResourceId; - use crate::resource::background_launch::ResourceBackgroundRejection; - use crate::store::BackgroundLaunchError; - - #[test] - fn unproven_predecessor_is_a_definitive_remote_refusal() { - let resource = ResourceId::new(); - let task_id = TaskId::new(); - let rejection = - launch_rejection(BackgroundLaunchError::PredecessorReleaseUnproven { task_id }) - .expect("domain refusal"); - assert_eq!( - rejection, - ResourceBackgroundRejection::PredecessorReleaseUnproven { task_id } - ); - assert!(matches!( - rejection_error(resource, None, &rejection), - AppError::ResourceActionNotAllowed { resource: actual, .. } if actual == resource - )); - } -} diff --git a/src/daemon/resource_notice_delivery.rs b/src/daemon/resource_notice_delivery.rs deleted file mode 100644 index c14501b..0000000 --- a/src/daemon/resource_notice_delivery.rs +++ /dev/null @@ -1,545 +0,0 @@ -//! Daemon-owned bounded delivery for durable supervisor notices - -use std::collections::HashMap; -use std::future::Future; -use std::time::{Duration, Instant}; - -use ractor::ActorRef; -use tracing::{info, warn}; - -use super::AppState; -use super::actors::{StoreMsg, call}; -use super::keyed_locks::KeyedLocks; -use super::resource_notice_sender::{ - ResourceNoticeDeliveryOutcome, ResourceNoticeSendError, deliver_one, -}; -use crate::resource::store::SupervisorNoticeStoreError; -use crate::resource::{NoticeId, SupervisorNoticeDelivery}; - -const SCAN_INTERVAL: Duration = Duration::from_secs(15); -const INITIAL_RETRY_DELAY: Duration = Duration::from_secs(15); -const MAX_RETRY_DELAY: Duration = Duration::from_secs(5 * 60); - -/// Recover durable in-flight attempts, then scan and deliver eligible notices -pub(super) async fn run(state: AppState) { - let mut worker = DeliveryWorker::new(state.store.clone(), RetryPolicy::default()); - - loop { - let delivery_state = state.clone(); - worker - .tick_with(|notice_id| { - let delivery_state = delivery_state.clone(); - async move { deliver_one(&delivery_state, notice_id).await } - }) - .await; - tokio::time::sleep(SCAN_INTERVAL).await; - } -} - -struct DeliveryWorker { - store: ActorRef, - retry_policy: RetryPolicy, - retries: HashMap, - in_flight: InFlight, - recovered: bool, -} - -impl DeliveryWorker { - fn new(store: ActorRef, retry_policy: RetryPolicy) -> Self { - Self { - store, - retry_policy, - retries: HashMap::new(), - in_flight: InFlight::default(), - recovered: false, - } - } - - async fn tick_with(&mut self, deliver: F) - where - F: FnMut(NoticeId) -> Fut, - Fut: Future>, - { - if !self.recovered && !self.recover().await { - return; - } - - self.scan_with(deliver).await; - } - - async fn recover(&mut self) -> bool { - match call(&self.store, |reply| { - StoreMsg::RecoverSendingSupervisorNotices { reply } - }) - .await - { - Ok(Ok(recovered)) => { - if !recovered.is_empty() { - info!( - count = recovered.len(), - "recovered in-flight supervisor notices" - ); - } - self.recovered = true; - true - } - Ok(Err(error)) => { - warn!("supervisor notice startup recovery failed: {error}"); - false - } - Err(error) => { - warn!("supervisor notice startup recovery call failed: {error}"); - false - } - } - } - - async fn scan_with(&mut self, mut deliver: F) - where - F: FnMut(NoticeId) -> Fut, - Fut: Future>, - { - let pending = match call(&self.store, |reply| StoreMsg::PendingSupervisorNotices { - reply, - }) - .await - { - Ok(Ok(pending)) => pending, - Ok(Err(error)) => { - warn!("supervisor notice pending scan failed: {error}"); - return; - } - Err(error) => { - warn!("supervisor notice pending scan call failed: {error}"); - return; - } - }; - - // a notice drops out of the scan once its action resolves, so forget its backoff - self.retries - .retain(|notice_id, _| pending.iter().any(|notice| notice.id == *notice_id)); - - for notice in pending { - let ready = self - .retries - .entry(notice.id) - .or_insert_with(|| Retry::new(self.retry_policy)) - .ready(); - if !ready { - continue; - } - - let notice_id = notice.id; - let Some(result) = self - .in_flight - .deliver_if_idle(notice_id, || deliver(notice_id)) - .await - else { - continue; - }; - - match result { - Ok(outcome) => self.record_outcome(notice_id, outcome), - Err(ResourceNoticeSendError::Store( - SupervisorNoticeStoreError::ActionNoLongerAwaited, - )) => { - info!(notice_id = %notice_id.as_uuid(), "supervisor notice action resolved before delivery"); - self.retries.remove(¬ice_id); - } - Err(error) => { - warn!(notice_id = %notice_id.as_uuid(), "supervisor notice delivery call failed: {error}"); - self.retry_later(notice_id); - } - } - } - } - - fn record_outcome(&mut self, notice_id: NoticeId, outcome: ResourceNoticeDeliveryOutcome) { - match outcome.notice.delivery { - SupervisorNoticeDelivery::Delivered { attempts } => { - info!(notice_id = %notice_id.as_uuid(), attempts, "supervisor notice delivered"); - self.retries.remove(¬ice_id); - } - SupervisorNoticeDelivery::RetryPending { - attempts, - last_error, - } => { - warn!( - notice_id = %notice_id.as_uuid(), - attempts, - error = %last_error, - "supervisor notice delivery failed; retry scheduled" - ); - self.retry_later(notice_id); - } - SupervisorNoticeDelivery::Failed { - attempts, - last_error, - } => { - warn!( - notice_id = %notice_id.as_uuid(), - attempts, - error = %last_error, - "supervisor notice delivery failed; automatic delivery stopped" - ); - self.retries.remove(¬ice_id); - } - SupervisorNoticeDelivery::Pending { .. } | SupervisorNoticeDelivery::Sending { .. } => { - warn!( - notice_id = %notice_id.as_uuid(), - "supervisor notice delivery returned an unsettled state" - ); - self.retry_later(notice_id); - } - } - } - - fn retry_later(&mut self, notice_id: NoticeId) { - self.retries - .entry(notice_id) - .or_insert_with(|| Retry::new(self.retry_policy)) - .failed(self.retry_policy); - } -} - -#[derive(Clone, Copy)] -struct RetryPolicy { - initial: Duration, - maximum: Duration, -} - -impl RetryPolicy { - const fn new(initial: Duration, maximum: Duration) -> Self { - Self { initial, maximum } - } -} - -impl Default for RetryPolicy { - fn default() -> Self { - Self::new(INITIAL_RETRY_DELAY, MAX_RETRY_DELAY) - } -} - -struct Retry { - next: Instant, - delay: Duration, -} - -impl Retry { - fn new(policy: RetryPolicy) -> Self { - Self { - next: Instant::now(), - delay: policy.initial, - } - } - - fn ready(&self) -> bool { - Instant::now() >= self.next - } - - fn failed(&mut self, policy: RetryPolicy) { - self.next = Instant::now() + self.delay; - self.delay = (self.delay * 2).min(policy.maximum); - } -} - -/// Notices with a delivery in progress -#[derive(Default)] -struct InFlight(KeyedLocks); - -impl InFlight { - /// Run `deliver` unless the same notice is already being delivered - async fn deliver_if_idle(&self, notice_id: NoticeId, deliver: F) -> Option - where - F: FnOnce() -> Fut, - Fut: Future, - { - let _permit = self.0.try_lock(notice_id)?; - Some(deliver().await) - } -} - -#[cfg(test)] -mod tests { - use std::sync::Arc; - use std::sync::atomic::{AtomicUsize, Ordering}; - - use ractor::{Actor, ActorRef}; - use rusqlite::{Connection, params}; - use tempfile::TempDir; - use uuid::Uuid; - - use super::{DeliveryWorker, InFlight, RetryPolicy}; - use crate::daemon::actors::{StoreActor, StoreMsg, call}; - use crate::daemon::resource_notice_sender::{ - ResourceNoticeDeliveryOutcome, ResourceNoticeSendError, - }; - use crate::domain::{TaskId, ThreadId}; - use crate::machine::MachineId; - use crate::resource::store::insert_supervisor_notice_in_transaction; - use crate::resource::{ - ActionId, AssignmentRevision, DeliveryAttemptId, LoanId, LoanPhase, LoanState, NoticeId, - ResourceId, ResourceRevision, SupervisorAddress, SupervisorNotice, - SupervisorNoticeDelivery, SupervisorNoticePayload, - }; - use crate::store::Store; - use std::time::Duration; - - #[tokio::test] - async fn startup_recovery_runs_before_dispatch() { - let (store, notice, _directory) = seeded_store().await; - reserve_attempt(&store, notice.id).await; - let mut worker = DeliveryWorker::new(store.clone(), zero_retry_policy()); - let dispatches = Arc::new(AtomicUsize::new(0)); - let dispatch_count = Arc::clone(&dispatches); - let delivery_store = store.clone(); - - worker - .tick_with(move |notice_id| { - let store = delivery_store.clone(); - let dispatch_count = Arc::clone(&dispatch_count); - async move { - let recovered = read_notice(&store, notice_id).await; - assert!(matches!( - recovered.delivery, - SupervisorNoticeDelivery::RetryPending { attempts: 1, .. } - )); - dispatch_count.fetch_add(1, Ordering::SeqCst); - fail_attempt(&store, notice_id, "fake delivery failure").await - } - }) - .await; - - assert_eq!(dispatches.load(Ordering::SeqCst), 1); - assert!(matches!( - read_notice(&store, notice.id).await.delivery, - SupervisorNoticeDelivery::RetryPending { attempts: 2, .. } - )); - store.stop(None); - } - - #[tokio::test] - async fn persisted_attempt_budget_stops_delivery_after_three_failures() { - let (store, notice, _directory) = seeded_store().await; - let first_worker = DeliveryWorker::new(store.clone(), zero_retry_policy()); - let calls = Arc::new(AtomicUsize::new(0)); - let first_calls = Arc::clone(&calls); - let first_store = store.clone(); - let mut first_worker = first_worker; - - for _ in 0..2 { - let store = first_store.clone(); - let calls = Arc::clone(&first_calls); - first_worker - .tick_with(move |notice_id| { - let store = store.clone(); - let calls = Arc::clone(&calls); - async move { - calls.fetch_add(1, Ordering::SeqCst); - fail_attempt(&store, notice_id, "temporary failure").await - } - }) - .await; - } - assert!(matches!( - read_notice(&store, notice.id).await.delivery, - SupervisorNoticeDelivery::RetryPending { attempts: 2, .. } - )); - - let mut restarted_worker = DeliveryWorker::new(store.clone(), zero_retry_policy()); - let store_for_retry = store.clone(); - let calls_for_retry = Arc::clone(&calls); - restarted_worker - .tick_with(move |notice_id| { - let store = store_for_retry.clone(); - let calls = Arc::clone(&calls_for_retry); - async move { - calls.fetch_add(1, Ordering::SeqCst); - fail_attempt(&store, notice_id, "final failure").await - } - }) - .await; - assert!(matches!( - read_notice(&store, notice.id).await.delivery, - SupervisorNoticeDelivery::Failed { - attempts: 3, - ref last_error, - } if last_error == "final failure" - )); - - let not_retried = Arc::new(AtomicUsize::new(0)); - let not_retried_counter = Arc::clone(¬_retried); - restarted_worker - .tick_with(move |_| { - let not_retried_counter = Arc::clone(¬_retried_counter); - async move { - not_retried_counter.fetch_add(1, Ordering::SeqCst); - Err(ResourceNoticeSendError::ReservationMismatch) - } - }) - .await; - assert_eq!(calls.load(Ordering::SeqCst), 3); - assert_eq!(not_retried.load(Ordering::SeqCst), 0); - assert!(pending_notices(&store).await.is_empty()); - store.stop(None); - } - - #[tokio::test] - async fn concurrent_dispatches_share_one_notice_guard() { - let guard = InFlight::default(); - let notice_id = NoticeId::new(); - let calls = Arc::new(AtomicUsize::new(0)); - let first_calls = Arc::clone(&calls); - let second_calls = Arc::clone(&calls); - - let (first, second) = tokio::join!( - guard.deliver_if_idle(notice_id, || async { - first_calls.fetch_add(1, Ordering::SeqCst); - tokio::time::sleep(Duration::from_millis(20)).await; - }), - guard.deliver_if_idle(notice_id, || async { - second_calls.fetch_add(1, Ordering::SeqCst); - }) - ); - - assert!(first.is_some()); - assert!(second.is_none()); - assert_eq!(calls.load(Ordering::SeqCst), 1); - } - - fn zero_retry_policy() -> RetryPolicy { - RetryPolicy::new(Duration::ZERO, Duration::ZERO) - } - - async fn fail_attempt( - store: &ActorRef, - notice_id: NoticeId, - error: &str, - ) -> Result { - let attempt_id = DeliveryAttemptId::new(); - let reserved = call(store, |reply| StoreMsg::ReserveSupervisorNoticeAttempt { - notice_id, - attempt_id, - reply, - }) - .await??; - if !matches!( - reserved.delivery, - SupervisorNoticeDelivery::Sending { - attempt_id: reserved_attempt, - .. - } if reserved_attempt == attempt_id - ) { - return Err(ResourceNoticeSendError::ReservationMismatch); - } - - let settled = call(store, |reply| StoreMsg::SettleSupervisorNoticeAttempt { - notice_id, - attempt_id, - result: Err(error.to_owned()), - reply, - }) - .await??; - Ok(ResourceNoticeDeliveryOutcome { - notice: settled, - receipt: None, - }) - } - - async fn reserve_attempt(store: &ActorRef, notice_id: NoticeId) { - let attempt_id = DeliveryAttemptId::new(); - call(store, |reply| StoreMsg::ReserveSupervisorNoticeAttempt { - notice_id, - attempt_id, - reply, - }) - .await - .unwrap() - .unwrap(); - } - - async fn read_notice(store: &ActorRef, notice_id: NoticeId) -> SupervisorNotice { - call(store, |reply| StoreMsg::SupervisorNotice { - notice_id, - reply, - }) - .await - .unwrap() - .unwrap() - .unwrap() - } - - async fn pending_notices(store: &ActorRef) -> Vec { - call(store, |reply| StoreMsg::PendingSupervisorNotices { reply }) - .await - .unwrap() - .unwrap() - } - - async fn seeded_store() -> (ActorRef, SupervisorNotice, TempDir) { - let directory = tempfile::tempdir().unwrap(); - let path = directory.path().join("homebased.sqlite"); - drop(Store::open(&path).unwrap()); - - let authority = MachineId::new(); - let resource_id = ResourceId::new(); - let notice = SupervisorNotice { - id: NoticeId::new(), - loan_id: LoanId::new(), - action_id: ActionId::new(), - state_revision: ResourceRevision::new(2), - destination: SupervisorAddress { - machine: authority, - thread: ThreadId(Uuid::now_v7()), - }, - assignment_revision: AssignmentRevision::new(1), - payload: SupervisorNoticePayload::AttentionRequired { - reason: "test notice".into(), - }, - delivery: SupervisorNoticeDelivery::Pending { attempts: 0 }, - }; - let mut connection = Connection::open(&path).unwrap(); - connection - .execute( - "INSERT INTO resources ( - id, display_name, authority_machine, supervisor_machine, - supervisor_thread, assignment_revision, state_revision, - registered_background_task - ) VALUES (?1, 'test resource', ?2, ?2, ?3, 1, 2, NULL)", - params![ - resource_id.as_uuid().to_string(), - authority.as_uuid().to_string(), - notice.destination.thread.to_string(), - ], - ) - .unwrap(); - connection - .execute( - "INSERT INTO loans (id, resource_id, state_json) VALUES (?1, ?2, ?3)", - params![ - notice.loan_id.as_uuid().to_string(), - resource_id.as_uuid().to_string(), - serde_json::to_string(&LoanState::NeedsAttention { - action_id: notice.action_id, - last_safe_phase: LoanPhase::AwaitingRelease { - action_id: ActionId::new(), - observed_background_task: TaskId::new(), - watcher_intent: None, - }, - reason: "test notice".into(), - }) - .unwrap(), - ], - ) - .unwrap(); - let tx = connection.transaction().unwrap(); - insert_supervisor_notice_in_transaction(&tx, ¬ice).unwrap(); - tx.commit().unwrap(); - drop(connection); - - let (store, _handle) = Actor::spawn(None, StoreActor, path).await.unwrap(); - assert_eq!(read_notice(&store, notice.id).await, notice); - - (store, notice, directory) - } -} diff --git a/src/daemon/resource_notice_sender.rs b/src/daemon/resource_notice_sender.rs deleted file mode 100644 index 6e56f0a..0000000 --- a/src/daemon/resource_notice_sender.rs +++ /dev/null @@ -1,536 +0,0 @@ -//! One authority-side delivery attempt for a durable supervisor notice - -use axum::http::StatusCode; -use std::future::Future; - -use crate::daemon::AppState; -use crate::daemon::actors::{StoreMsg, call}; -use crate::domain::API_VERSION; -use crate::error::AppError; -use crate::fleet::http::ClusterClient; -use crate::fleet::protocol::CLUSTER_PROTOCOL_VERSION; -use crate::machine::MachineId; -use crate::resource::store::SupervisorNoticeStoreError; -use crate::resource::{ - DeliveryAttemptId, NoticeId, SupervisorNotice, SupervisorNoticeDelivery, - SupervisorNoticeReceipt, SupervisorNoticeRequest, SupervisorNoticeResponse, -}; - -const RESOURCE_NOTICE_PATH: &str = "/v1/cluster/resource-notices"; - -/// Settled delivery state and the verified receiver receipt, when delivery succeeded -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) struct ResourceNoticeDeliveryOutcome { - /// Notice after this attempt was settled by the StoreActor - pub notice: SupervisorNotice, - /// Receipt returned by the exact local or remote destination - pub receipt: Option, -} - -/// The attempt could not be reserved or its settlement could not be persisted -#[derive(Debug, thiserror::Error)] -pub(crate) enum ResourceNoticeSendError { - /// Actor transport failed - #[error(transparent)] - Actor(#[from] AppError), - /// Durable notice state rejected the operation - #[error(transparent)] - Store(#[from] SupervisorNoticeStoreError), - /// The StoreActor did not return the attempt reserved by this call - #[error("reserved supervisor notice attempt does not match its identity")] - ReservationMismatch, -} - -/// Reserve and settle at most one supervisor-notice delivery attempt -pub(crate) async fn deliver_one( - state: &AppState, - notice_id: NoticeId, -) -> Result { - reserve_deliver_settle(&state.store, notice_id, |notice, attempt_id| async move { - deliver_reserved(state, ¬ice, attempt_id).await - }) - .await -} - -/// Deliver and settle one attempt that an explicit renotify already reserved -/// -/// Returns `None` without sending when the notice no longer holds that exact -/// reservation, so a replayed operation never sends a second copy -pub(crate) async fn deliver_reserved_attempt( - state: &AppState, - notice_id: NoticeId, - attempt_id: DeliveryAttemptId, -) -> Result, ResourceNoticeSendError> { - let notice = call(&state.store, |reply| StoreMsg::SupervisorNotice { - notice_id, - reply, - }) - .await?? - .ok_or(ResourceNoticeSendError::Store( - SupervisorNoticeStoreError::NotFound, - ))?; - if !matches!( - notice.delivery, - SupervisorNoticeDelivery::Sending { - attempt_id: reserved, - .. - } if reserved == attempt_id - ) { - return Ok(None); - } - let delivery = deliver_reserved(state, ¬ice, attempt_id).await; - let (receipt, result) = match delivery { - Ok(receipt) => (Some(receipt), Ok(())), - Err(error) => (None, Err(error)), - }; - let notice = call(&state.store, |reply| { - StoreMsg::SettleSupervisorNoticeAttempt { - notice_id, - attempt_id, - result, - reply, - } - }) - .await??; - Ok(Some(ResourceNoticeDeliveryOutcome { notice, receipt })) -} - -async fn reserve_deliver_settle( - store: &ractor::ActorRef, - notice_id: NoticeId, - deliver: F, -) -> Result -where - F: FnOnce(SupervisorNotice, DeliveryAttemptId) -> Fut, - Fut: Future>, -{ - let attempt_id = DeliveryAttemptId::new(); - let notice = call(store, |reply| StoreMsg::ReserveSupervisorNoticeAttempt { - notice_id, - attempt_id, - reply, - }) - .await??; - if !matches!( - notice.delivery, - SupervisorNoticeDelivery::Sending { - attempt_id: reserved, - .. - } if reserved == attempt_id - ) { - return Err(ResourceNoticeSendError::ReservationMismatch); - } - - let delivery = deliver(notice, attempt_id).await; - let (receipt, result) = match delivery { - Ok(receipt) => (Some(receipt), Ok(())), - Err(error) => (None, Err(error)), - }; - let notice = call(store, |reply| StoreMsg::SettleSupervisorNoticeAttempt { - notice_id, - attempt_id, - result, - reply, - }) - .await??; - - Ok(ResourceNoticeDeliveryOutcome { notice, receipt }) -} - -async fn deliver_reserved( - state: &AppState, - notice: &SupervisorNotice, - attempt_id: DeliveryAttemptId, -) -> Result { - let destination = notice.destination.machine; - let local = state.machine.identity.machine; - if destination == local { - let request = notice_request(notice, attempt_id, local, CLUSTER_PROTOCOL_VERSION.0) - .map_err(|error| error.to_string())?; - let receipt = state - .message_receiver - .receive_supervisor_notice(state, request.clone()) - .await - .map_err(|error| error.to_string())?; - return verify_receipt(receipt, &request); - } - - let fleet = state - .fleet - .handle() - .ok_or_else(|| "Fleet is disabled for the remote supervisor destination".to_owned())?; - let verified = fleet - .connect(destination) - .await - .map_err(|error| error.to_string())?; - let request = notice_request(notice, attempt_id, local, verified.protocol.0) - .map_err(|error| error.to_string())?; - post_remote_notice(&ClusterClient::default(), &verified.address, &request).await -} - -fn notice_request( - notice: &SupervisorNotice, - attempt_id: DeliveryAttemptId, - source_machine: MachineId, - protocol_version: u32, -) -> Result { - let request = SupervisorNoticeRequest { - api_version: API_VERSION, - protocol_version, - source_machine, - destination: notice.destination, - notice_id: notice.id, - loan_id: notice.loan_id, - action_id: notice.action_id, - state_revision: notice.state_revision, - assignment_revision: notice.assignment_revision, - attempt_id, - payload: notice.payload.clone(), - }; - request.validate()?; - Ok(request) -} - -fn decode_remote_receipt( - status: StatusCode, - body: &[u8], - request: &SupervisorNoticeRequest, -) -> Result { - if status != StatusCode::OK { - return Err(format!("resource notice receiver returned HTTP {status}")); - } - let response: SupervisorNoticeResponse = serde_json::from_slice(body).map_err(|error| { - format!("resource notice receiver returned an invalid response: {error}") - })?; - if response.api_version != API_VERSION - || response.protocol_version != request.protocol_version - || response.destination_machine != request.destination.machine - { - return Err( - "resource notice response does not match the requested destination or version".into(), - ); - } - verify_receipt(response.receipt, request) -} - -async fn post_remote_notice( - client: &ClusterClient, - address: &crate::fleet::address::MachineAddress, - request: &SupervisorNoticeRequest, -) -> Result { - let response = client - .post_json(address, RESOURCE_NOTICE_PATH, request) - .await - .map_err(|error| error.to_string())?; - decode_remote_receipt(response.status, &response.body, request) -} - -fn verify_receipt( - receipt: SupervisorNoticeReceipt, - request: &SupervisorNoticeRequest, -) -> Result { - if receipt.api_version != API_VERSION - || receipt.protocol_version != request.protocol_version - || receipt.destination_machine != request.destination.machine - || receipt.notice_id != request.notice_id - || receipt.attempt_id != request.attempt_id - || receipt.destination_thread != request.destination.thread - { - return Err("resource notice receipt does not match the reserved attempt".into()); - } - Ok(receipt) -} - -#[cfg(test)] -mod tests { - use super::{ - RESOURCE_NOTICE_PATH, decode_remote_receipt, notice_request, post_remote_notice, - reserve_deliver_settle, verify_receipt, - }; - use crate::daemon::actors::{StoreActor, StoreMsg, call}; - use crate::domain::{API_VERSION, TaskId, ThreadId}; - use crate::fleet::address::MachineAddress; - use crate::fleet::http::ClusterClient; - use crate::fleet::protocol::CLUSTER_PROTOCOL_VERSION; - use crate::machine::MachineId; - use crate::resource::store::insert_supervisor_notice_in_transaction; - use crate::resource::{ - ActionId, AssignmentRevision, DeliveryAttemptId, LoanId, LoanPhase, LoanState, NoticeId, - ResourceId, ResourceRevision, SupervisorAddress, SupervisorNotice, - SupervisorNoticeDelivery, SupervisorNoticePayload, SupervisorNoticeReceipt, - SupervisorNoticeRequest, SupervisorNoticeResponse, - }; - use crate::store::Store; - use axum::Json; - use axum::http::StatusCode; - use axum::routing::post; - use chrono::Utc; - use ractor::Actor; - use rusqlite::{Connection, params}; - use tempfile::TempDir; - use uuid::Uuid; - - fn request() -> SupervisorNoticeRequest { - SupervisorNoticeRequest { - api_version: API_VERSION, - protocol_version: CLUSTER_PROTOCOL_VERSION.0, - source_machine: MachineId::new(), - destination: SupervisorAddress { - machine: MachineId::new(), - thread: ThreadId(Uuid::now_v7()), - }, - notice_id: NoticeId::new(), - loan_id: LoanId::new(), - action_id: ActionId::new(), - state_revision: ResourceRevision::new(2), - assignment_revision: AssignmentRevision::new(1), - attempt_id: DeliveryAttemptId::new(), - payload: SupervisorNoticePayload::AttentionRequired { - reason: "test notice".into(), - }, - } - } - - fn receipt(request: &SupervisorNoticeRequest) -> SupervisorNoticeReceipt { - SupervisorNoticeReceipt { - api_version: API_VERSION, - protocol_version: request.protocol_version, - notice_id: request.notice_id, - attempt_id: request.attempt_id, - destination_machine: request.destination.machine, - destination_thread: request.destination.thread, - delivered_at: Utc::now(), - } - } - - fn response(request: &SupervisorNoticeRequest) -> SupervisorNoticeResponse { - SupervisorNoticeResponse { - api_version: API_VERSION, - protocol_version: request.protocol_version, - destination_machine: request.destination.machine, - receipt: receipt(request), - } - } - - fn assert_same_receipt(actual: &SupervisorNoticeReceipt, expected: &SupervisorNoticeReceipt) { - assert_eq!(actual.api_version, expected.api_version); - assert_eq!(actual.protocol_version, expected.protocol_version); - assert_eq!(actual.notice_id, expected.notice_id); - assert_eq!(actual.attempt_id, expected.attempt_id); - assert_eq!(actual.destination_machine, expected.destination_machine); - assert_eq!(actual.destination_thread, expected.destination_thread); - } - - #[test] - fn accepts_a_matching_remote_receipt() { - let request = request(); - let response = serde_json::to_vec(&response(&request)).unwrap(); - let decoded = decode_remote_receipt(StatusCode::OK, &response, &request).unwrap(); - assert_same_receipt(&decoded, &receipt(&request)); - } - - #[test] - fn rejects_wrong_machine_thread_notice_and_attempt_receipts() { - let request = request(); - let mut cases = Vec::new(); - - let mut wrong_machine = response(&request); - wrong_machine.receipt.destination_machine = MachineId::new(); - cases.push(wrong_machine); - - let mut wrong_thread = response(&request); - wrong_thread.receipt.destination_thread = ThreadId(Uuid::now_v7()); - cases.push(wrong_thread); - - let mut wrong_notice = response(&request); - wrong_notice.receipt.notice_id = NoticeId::new(); - cases.push(wrong_notice); - - let mut wrong_attempt = response(&request); - wrong_attempt.receipt.attempt_id = DeliveryAttemptId::new(); - cases.push(wrong_attempt); - - for response in cases { - let body = serde_json::to_vec(&response).unwrap(); - assert!(decode_remote_receipt(StatusCode::OK, &body, &request).is_err()); - } - } - - #[test] - fn rejects_invalid_remote_responses_as_uncertain_delivery() { - let request = request(); - assert!(decode_remote_receipt(StatusCode::OK, b"{}", &request).is_err()); - assert!(decode_remote_receipt(StatusCode::SERVICE_UNAVAILABLE, b"{}", &request).is_err()); - } - - #[tokio::test] - async fn posts_to_a_fake_remote_notice_endpoint_and_accepts_its_receipt() { - async fn fake_receiver( - Json(request): Json, - ) -> Json { - Json(response(&request)) - } - - let request = request(); - let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); - let address: MachineAddress = format!("http://{}", listener.local_addr().unwrap()) - .parse() - .unwrap(); - let server = tokio::spawn(async move { - axum::serve( - listener, - axum::Router::new().route(RESOURCE_NOTICE_PATH, post(fake_receiver)), - ) - .await - .unwrap(); - }); - - let delivered = post_remote_notice(&ClusterClient::default(), &address, &request) - .await - .unwrap(); - assert_same_receipt(&delivered, &receipt(&request)); - server.abort(); - } - - #[tokio::test] - async fn local_queue_receipt_settles_the_reserved_attempt_as_delivered() { - let (store, notice, _directory) = seeded_store().await; - let local = notice.destination.machine; - let expected_notice = notice.clone(); - - let outcome = reserve_deliver_settle(&store, notice.id, |saved, attempt_id| async move { - assert_eq!(saved.id, expected_notice.id); - assert_eq!(saved.loan_id, expected_notice.loan_id); - assert_eq!(saved.action_id, expected_notice.action_id); - assert_eq!(saved.state_revision, expected_notice.state_revision); - assert_eq!(saved.destination, expected_notice.destination); - assert_eq!( - saved.assignment_revision, - expected_notice.assignment_revision - ); - assert_eq!(saved.payload, expected_notice.payload); - assert!(matches!( - saved.delivery, - SupervisorNoticeDelivery::Sending { attempt_id: reserved, .. } - if reserved == attempt_id - )); - let request = notice_request(&saved, attempt_id, local, CLUSTER_PROTOCOL_VERSION.0) - .map_err(|error| error.to_string())?; - let fake_queue_receipt = receipt(&request); - verify_receipt(fake_queue_receipt, &request) - }) - .await - .unwrap(); - - let receipt = outcome.receipt.as_ref().unwrap(); - assert_eq!(receipt.notice_id, notice.id); - assert_eq!(receipt.destination_machine, local); - assert_eq!(receipt.destination_thread, notice.destination.thread); - assert!(matches!( - outcome.notice.delivery, - SupervisorNoticeDelivery::Delivered { attempts: 1 } - )); - store.stop(None); - } - - #[tokio::test] - async fn delivery_failure_is_settled_without_a_receipt() { - let (store, notice, _directory) = seeded_store().await; - - let outcome = reserve_deliver_settle(&store, notice.id, |saved, attempt_id| async move { - assert!(matches!( - saved.delivery, - SupervisorNoticeDelivery::Sending { attempt_id: reserved, .. } - if reserved == attempt_id - )); - Err("fake queue endpoint unavailable".into()) - }) - .await - .unwrap(); - - assert!(outcome.receipt.is_none()); - assert!(matches!( - outcome.notice.delivery, - SupervisorNoticeDelivery::RetryPending { - attempts: 1, - ref last_error, - } if last_error == "fake queue endpoint unavailable" - )); - store.stop(None); - } - - async fn seeded_store() -> (ractor::ActorRef, SupervisorNotice, TempDir) { - let directory = tempfile::tempdir().unwrap(); - let path = directory.path().join("homebased.sqlite"); - drop(Store::open(&path).unwrap()); - - let authority = MachineId::new(); - let resource_id = ResourceId::new(); - let notice = SupervisorNotice { - id: NoticeId::new(), - loan_id: LoanId::new(), - action_id: ActionId::new(), - state_revision: ResourceRevision::new(2), - destination: SupervisorAddress { - machine: authority, - thread: ThreadId(Uuid::now_v7()), - }, - assignment_revision: AssignmentRevision::new(1), - payload: SupervisorNoticePayload::AttentionRequired { - reason: "test notice".into(), - }, - delivery: SupervisorNoticeDelivery::Pending { attempts: 0 }, - }; - let mut connection = Connection::open(&path).unwrap(); - connection - .execute( - "INSERT INTO resources ( - id, display_name, authority_machine, supervisor_machine, - supervisor_thread, assignment_revision, state_revision, - registered_background_task - ) VALUES (?1, 'test resource', ?2, ?2, ?3, 1, 2, NULL)", - params![ - resource_id.as_uuid().to_string(), - authority.as_uuid().to_string(), - notice.destination.thread.to_string(), - ], - ) - .unwrap(); - connection - .execute( - "INSERT INTO loans (id, resource_id, state_json) VALUES (?1, ?2, ?3)", - params![ - notice.loan_id.as_uuid().to_string(), - resource_id.as_uuid().to_string(), - serde_json::to_string(&LoanState::NeedsAttention { - action_id: notice.action_id, - last_safe_phase: LoanPhase::AwaitingRelease { - action_id: ActionId::new(), - observed_background_task: TaskId::new(), - watcher_intent: None, - }, - reason: "test notice".into(), - }) - .unwrap(), - ], - ) - .unwrap(); - let tx = connection.transaction().unwrap(); - insert_supervisor_notice_in_transaction(&tx, ¬ice).unwrap(); - tx.commit().unwrap(); - drop(connection); - - let (store, _handle) = Actor::spawn(None, StoreActor, path).await.unwrap(); - let saved = call(&store, |reply| StoreMsg::SupervisorNotice { - notice_id: notice.id, - reply, - }) - .await - .unwrap() - .unwrap() - .unwrap(); - assert_eq!(saved, notice); - - (store, notice, directory) - } -} diff --git a/src/daemon/resource_submit.rs b/src/daemon/resource_submit.rs deleted file mode 100644 index 2924e8d..0000000 --- a/src/daemon/resource_submit.rs +++ /dev/null @@ -1,978 +0,0 @@ -//! Origin-owned submission of command requests to resource authorities - -use std::path::PathBuf; - -use serde_json::Value; -use tracing::warn; - -use super::AppState; -use super::actors::{StoreMsg, SupervisorMsg, call}; -use crate::domain::{API_VERSION, AgentKind, TaskEnv, TaskId}; -use crate::error::AppError; -use crate::fleet::http::{ClusterClient, ClusterResponse}; -use crate::invocation::resolve_agent_binary; -use crate::machine::MachineId; -use crate::resource::{ - CommandSpec, ResourceId, ResourceQueueRequest, ResourceQueueResponse, ResourceRequest, - ResourceRequestState, -}; -use crate::spec::{self, NormalizedSpec}; -use crate::submission::{ - CallbackContext, NewResourceRoute, OriginRoute, RequestId, ResourceQueueOutcome, - ResourceQueueReceipt, ResourceRoutePhase, SubmissionState, -}; - -/// Input that identifies one resource-backed command submission -#[derive(Debug, Clone)] -pub(super) struct ResourceSubmitInput { - /// Caller retry identity - pub(super) request: RequestId, - /// Fixed resource queue to use - pub(super) resource: ResourceId, - /// Machine that owns the resource queue - pub(super) authority: MachineId, - /// Normalized command specification - pub(super) spec: NormalizedSpec, - /// Environment saved for the origin-side callback - pub(super) env: TaskEnv, - /// Absolute directory used to resolve the origin-side Codex executable - pub(super) callback_cwd: PathBuf, -} - -/// Saved result of a resource-backed command submission -#[derive(Debug, Clone, PartialEq, Eq)] -pub(super) enum ResourceSubmitOutcome { - /// The authority accepted the command into its serving queue - Waiting { task: TaskId }, - /// The authority activated the command on a resource - Activated { task: TaskId }, - /// The authority rejected the command before activation - Rejected { task: TaskId, reason: String }, - /// The authority cancelled the command before activation - CancelledBeforeLaunch { task: TaskId }, -} - -/// Submit a resource-backed command or return its durable saved result -pub(super) async fn submit( - state: &AppState, - input: ResourceSubmitInput, -) -> Result { - let _request_guard = state.locks.resource_submissions.lock(input.request).await; - let saved = call(&state.store, |reply| StoreMsg::OriginRouteByRequest { - request: input.request, - reply, - }) - .await?; - - let route = match saved { - Some(route) => { - validate_retry(&route, &input)?; - route - } - None => create_route(state, &input).await?, - }; - submit_route(state, &route).await -} - -async fn retry_saved_route( - state: &AppState, - expected: &OriginRoute, -) -> Result { - let _request_guard = state - .locks - .resource_submissions - .lock(expected.request) - .await; - let saved = call(&state.store, |reply| StoreMsg::OriginRouteByRequest { - request: expected.request, - reply, - }) - .await?; - let Some(saved) = saved else { - return Err(unknown( - expected, - "saved resource route disappeared during startup recovery", - )); - }; - ensure_same_recovery_identity(expected, &saved)?; - submit_route(state, &saved).await -} - -async fn submit_route( - state: &AppState, - route: &OriginRoute, -) -> Result { - if !route_acceptance_is_unknown(route) { - return outcome_from_route(route); - } - - let receipt = if route.execution_machine == state.machine.identity.machine { - submit_local(state, route).await? - } else { - submit_remote(state, route).await? - }; - let saved = call(&state.store, |reply| StoreMsg::ResolveResourceRoute { - receipt, - reply, - }) - .await - .map_err(|error| { - unknown( - route, - format!("cannot save resource authority result: {error}"), - ) - })?; - - outcome_from_route(&saved) -} - -fn ensure_same_recovery_identity( - expected: &OriginRoute, - saved: &OriginRoute, -) -> Result<(), AppError> { - let Some(expected_resource) = resource_from_route(expected) else { - return Err(conflict( - expected, - "startup recovery scan returned a non-resource route", - )); - }; - let Some(saved_resource) = resource_from_route(saved) else { - return Err(conflict( - expected, - "saved route is no longer a resource route", - )); - }; - let same_spec = expected.current_spec() == saved.current_spec(); - if expected.request != saved.request - || expected.task != saved.task - || expected.origin_machine != saved.origin_machine - || expected.execution_machine != saved.execution_machine - || expected.thread != saved.thread - || expected.callback != saved.callback - || expected_resource != saved_resource - || !same_spec - { - return Err(conflict( - expected, - "saved resource route identity changed during startup recovery", - )); - } - Ok(()) -} - -/// Retry origin routes whose authority acceptance was unknown at daemon startup -pub(super) async fn recover(state: AppState) { - let routes = match call(&state.store, |reply| { - StoreMsg::UnknownResourceOriginRoutes { reply } - }) - .await - { - Ok(routes) => routes, - Err(error) => { - warn!("cannot scan unresolved resource submissions: {error}"); - return; - } - }; - for route in routes { - if let Err(error) = retry_saved_route(&state, &route).await { - warn!("resource submission recovery: {error}"); - } - } -} - -async fn create_route( - state: &AppState, - input: &ResourceSubmitInput, -) -> Result { - CommandSpec::try_from(input.spec.clone()).map_err(|error| AppError::Usage { - message: error.to_string(), - })?; - if !input.callback_cwd.is_absolute() { - return Err(AppError::InvalidSpec { - pointer: "/callback_cwd".into(), - value: serde_json::to_value(&input.callback_cwd)?, - message: "callback directory must be absolute".into(), - }); - } - spec::check_cwd(&input.callback_cwd)?; - let codex = resolve_agent_binary(AgentKind::Codex, &input.env.path, &input.callback_cwd)?; - let task = TaskId::new(); - let route = OriginRoute::new_resource_waiting(NewResourceRoute { - request: input.request, - task, - origin_machine: state.machine.identity.machine, - authority_machine: input.authority, - thread: input.spec.thread, - callback: CallbackContext { - env: input.env.clone(), - cwd: input.callback_cwd.clone(), - codex: codex.into(), - }, - spec: input.spec.clone(), - resource: input.resource, - }) - .map_err(|error| AppError::Usage { - message: error.to_string(), - })?; - - match call(&state.store, |reply| StoreMsg::InsertOriginRoute { - route: Box::new(route), - reply, - }) - .await - { - Ok(saved) => Ok(saved), - Err(error) => { - let saved = call(&state.store, |reply| StoreMsg::OriginRouteByRequest { - request: input.request, - reply, - }) - .await?; - let Some(saved) = saved else { - return Err(error); - }; - validate_retry(&saved, input)?; - Ok(saved) - } - } -} - -fn validate_retry(route: &OriginRoute, input: &ResourceSubmitInput) -> Result<(), AppError> { - let SubmissionState::Resource { resource, .. } = &route.submission else { - return Err(conflict( - route, - "request UUID belongs to a non-resource route", - )); - }; - let Some(saved_spec) = route.current_spec() else { - return Err(conflict( - route, - "saved resource route has no normalized specification", - )); - }; - if route.execution_machine != input.authority - || *resource != input.resource - || *saved_spec != input.spec - { - return Err(conflict( - route, - "request UUID has different resource authority, resource, or normalized content", - )); - } - Ok(()) -} - -async fn submit_local( - state: &AppState, - route: &OriginRoute, -) -> Result { - let resource = resource_id(route)?; - let spec = route.current_spec().ok_or_else(|| { - conflict( - route, - "saved resource route has no normalized specification", - ) - })?; - let result = call(&state.store, |reply| StoreMsg::AcceptResourceRequest { - authority_machine: route.execution_machine, - request_id: route.request, - task_id: route.task, - resource_id: resource, - origin_machine: route.origin_machine, - normalized_spec: Box::new(spec.clone()), - reply, - }) - .await; - match result { - Ok(stored) => { - reconcile_local_request_if_waiting(state, route, &stored).await?; - local_receipt(route, &stored) - } - Err(AppError::SubmissionRejected { - request, - task, - reason, - }) if request == route.request && task == route.task => { - Ok(receipt(route, ResourceQueueOutcome::Rejected { reason })?) - } - Err(error) => Err(unknown( - route, - format!("local resource authority did not return a definitive queue result: {error}"), - )), - } -} - -async fn reconcile_local_request_if_waiting( - state: &AppState, - route: &OriginRoute, - request: &ResourceRequest, -) -> Result<(), AppError> { - if !matches!( - &request.state, - ResourceRequestState::Queued | ResourceRequestState::Assigned { .. } - ) { - return Ok(()); - } - - call(&state.supervisor, |reply| { - SupervisorMsg::ReconcileResource { - id: request.resource_id, - reply, - } - }) - .await - .map_err(|error| { - unknown( - route, - format!( - "local resource queue accepted the request but could not reconcile it: {error}" - ), - ) - }) -} - -fn local_receipt( - route: &OriginRoute, - stored: &ResourceRequest, -) -> Result { - let resource = resource_id(route)?; - let spec = route.current_spec().ok_or_else(|| { - conflict( - route, - "saved resource route has no normalized specification", - ) - })?; - if stored.request_id != route.request - || stored.task_id != route.task - || stored.resource_id != resource - || stored.origin_machine != route.origin_machine - || stored.spec().as_normalized() != spec - { - return Err(unknown( - route, - "local resource authority returned a different request identity", - )); - } - let outcome = match &stored.state { - ResourceRequestState::Queued - | ResourceRequestState::Assigned { .. } - | ResourceRequestState::Finished { .. } => ResourceQueueOutcome::Waiting, - ResourceRequestState::CancelledBeforeLaunch => ResourceQueueOutcome::Rejected { - reason: "cancelled_before_launch".into(), - }, - ResourceRequestState::Rejected { reason } => ResourceQueueOutcome::Rejected { - reason: reason.clone(), - }, - }; - receipt(route, outcome) -} - -async fn submit_remote( - state: &AppState, - route: &OriginRoute, -) -> Result { - let fleet = state - .fleet - .handle() - .ok_or_else(|| unknown(route, "fleet is disabled"))?; - let destination = fleet - .connect(route.execution_machine) - .await - .map_err(|error| unknown(route, error.to_string()))?; - let spec = route.current_spec().ok_or_else(|| { - conflict( - route, - "saved resource route has no normalized specification", - ) - })?; - let command = CommandSpec::try_from(spec.clone()) - .map_err(|error| unknown(route, format!("saved resource command is invalid: {error}")))?; - let request = ResourceQueueRequest::new( - destination.protocol.0, - route.execution_machine, - route.origin_machine, - route.request, - route.task, - resource_id(route)?, - command, - ); - let response = ClusterClient::default() - .post_json( - &destination.address, - "/v1/cluster/resource-requests", - &request, - ) - .await - .map_err(|error| unknown(route, error.to_string()))?; - - decode_resource_response(route, response, destination.protocol.0) -} - -fn decode_resource_response( - route: &OriginRoute, - response: ClusterResponse, - expected_protocol: u32, -) -> Result { - if !response.status.is_success() { - return Err(unknown( - route, - format!("resource authority returned HTTP {}", response.status), - )); - } - let value: Value = serde_json::from_slice(&response.body).map_err(|error| { - unknown( - route, - format!("invalid resource authority response: {error}"), - ) - })?; - if value.get("api_version").and_then(Value::as_u64) != Some(u64::from(API_VERSION)) { - return Err(unknown( - route, - "resource authority returned an unsupported API version", - )); - } - let response: ResourceQueueResponse = serde_json::from_value(value).map_err(|error| { - unknown( - route, - format!("invalid resource authority response: {error}"), - ) - })?; - if response.api_version != API_VERSION { - return Err(unknown( - route, - "resource authority returned an unsupported API version", - )); - } - if response.protocol_version != expected_protocol { - return Err(unknown( - route, - "resource authority returned another protocol version", - )); - } - if response.destination_machine != route.execution_machine { - return Err(unknown( - route, - "resource authority response names another destination", - )); - } - let receipt = response.receipt; - if receipt.request != route.request - || receipt.task != route.task - || receipt.origin_machine != route.origin_machine - || receipt.authority_machine != route.execution_machine - || Some(receipt.resource) != resource_from_route(route) - { - return Err(unknown( - route, - "resource authority returned a receipt for another request identity", - )); - } - Ok(receipt) -} - -fn receipt( - route: &OriginRoute, - outcome: ResourceQueueOutcome, -) -> Result { - Ok(ResourceQueueReceipt { - request: route.request, - task: route.task, - origin_machine: route.origin_machine, - authority_machine: route.execution_machine, - resource: resource_id(route)?, - outcome, - }) -} - -fn resource_id(route: &OriginRoute) -> Result { - resource_from_route(route) - .ok_or_else(|| conflict(route, "request UUID belongs to a non-resource route")) -} - -fn resource_from_route(route: &OriginRoute) -> Option { - match &route.submission { - SubmissionState::Resource { resource, .. } => Some(*resource), - SubmissionState::AcceptanceUnknown - | SubmissionState::Accepted - | SubmissionState::Rejected { .. } - | SubmissionState::ResourceAction { .. } - | SubmissionState::ResourceBackground { .. } - | SubmissionState::Held { .. } => None, - } -} - -fn route_acceptance_is_unknown(route: &OriginRoute) -> bool { - matches!( - &route.submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::AcceptanceUnknown, - .. - } - ) -} - -fn outcome_from_route(route: &OriginRoute) -> Result { - let SubmissionState::Resource { phase, .. } = &route.submission else { - return Err(conflict( - route, - "request UUID belongs to a non-resource route", - )); - }; - match phase { - ResourceRoutePhase::AcceptanceUnknown => Err(unknown( - route, - "resource queue acceptance is unresolved; retry the same request UUID", - )), - ResourceRoutePhase::Waiting => Ok(ResourceSubmitOutcome::Waiting { task: route.task }), - ResourceRoutePhase::Activated => Ok(ResourceSubmitOutcome::Activated { task: route.task }), - ResourceRoutePhase::Rejected { reason } => Ok(ResourceSubmitOutcome::Rejected { - task: route.task, - reason: reason.clone(), - }), - ResourceRoutePhase::CancelledBeforeLaunch => { - Ok(ResourceSubmitOutcome::CancelledBeforeLaunch { task: route.task }) - } - } -} - -fn unknown(route: &OriginRoute, message: impl Into) -> AppError { - AppError::SubmissionOutcomeUnknown { - request: route.request, - task: route.task, - message: message.into(), - } -} - -fn conflict(route: &OriginRoute, message: impl Into) -> AppError { - AppError::SubmissionConflict { - request: route.request, - task: route.task, - message: message.into(), - } -} - -#[cfg(test)] -mod tests { - use std::path::Path; - - use axum::http::StatusCode; - use bytes::Bytes; - use ractor::Actor; - use tempfile::tempdir; - - use super::{ - ResourceSubmitInput, ResourceSubmitOutcome, decode_resource_response, - ensure_same_recovery_identity, outcome_from_route, resource_from_route, - route_acceptance_is_unknown, submit_local, validate_retry, - }; - use crate::daemon::AppState; - use crate::daemon::actors::{StoreMsg, SupervisorActor, SupervisorArgs, SupervisorMsg, call}; - use crate::domain::{TaskEnv, TaskId}; - use crate::error::AppError; - use crate::files::StreamSlots; - use crate::fleet::FleetState; - use crate::fleet::directory::LocalMachine; - use crate::fleet::http::ClusterResponse; - use crate::fleet::protocol::SUPPORTED_PROTOCOLS; - use crate::home::Home; - use crate::machine::{LocalIdentity, MachineId, MachineName}; - use crate::resource::{ - AssignmentRevision, Resource, ResourceId, ResourceQueueAttentionReason, - ResourceQueueReconcileOutcome, ResourceQueueResponse, ResourceRevision, SupervisorAddress, - }; - use crate::spec::NormalizedSpec; - use crate::submission::{ - CallbackContext, CallbackExecutable, NewResourceRoute, OriginRoute, RequestId, - ResourceQueueOutcome, ResourceQueueReceipt, ResourceRoutePhase, SubmissionState, - }; - use serde_json::Value; - - fn spec() -> NormalizedSpec { - serde_json::from_value(serde_json::json!({ - "api_version": 1, - "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "resource task", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": ["echo", "hello"] } - })) - .unwrap() - } - - fn route() -> OriginRoute { - let spec = spec(); - OriginRoute::new_resource_waiting(NewResourceRoute { - request: RequestId::new(), - task: TaskId::new(), - origin_machine: MachineId::new(), - authority_machine: MachineId::new(), - thread: spec.thread, - callback: CallbackContext { - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - cwd: Path::new("/tmp").to_path_buf(), - codex: CallbackExecutable::available(Path::new("/bin/echo").to_path_buf()), - }, - spec, - resource: ResourceId::new(), - }) - .unwrap() - } - - fn local_route(authority: MachineId, resource: ResourceId) -> OriginRoute { - let spec = spec(); - OriginRoute::new_resource_waiting(NewResourceRoute { - request: RequestId::new(), - task: TaskId::new(), - origin_machine: authority, - authority_machine: authority, - thread: spec.thread, - callback: CallbackContext { - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - cwd: Path::new("/tmp").to_path_buf(), - codex: CallbackExecutable::available(Path::new("/bin/echo").to_path_buf()), - }, - spec, - resource, - }) - .unwrap() - } - - fn response(route: &OriginRoute, protocol_version: u32) -> ClusterResponse { - let receipt = ResourceQueueReceipt { - request: route.request, - task: route.task, - origin_machine: route.origin_machine, - authority_machine: route.execution_machine, - resource: resource_from_route(route).unwrap(), - outcome: ResourceQueueOutcome::Waiting, - }; - ClusterResponse { - status: StatusCode::OK, - body: Bytes::from( - serde_json::to_vec(&ResourceQueueResponse::new(protocol_version, receipt)).unwrap(), - ), - } - } - - fn input(route: &OriginRoute) -> ResourceSubmitInput { - ResourceSubmitInput { - request: route.request, - resource: resource_from_route(route).unwrap(), - authority: route.execution_machine, - spec: route.current_spec().unwrap().clone(), - env: TaskEnv { - path: "/different/bin".into(), - home: "/different/home".into(), - }, - callback_cwd: Path::new("/different/callback").to_path_buf(), - } - } - - #[tokio::test] - async fn local_acceptance_and_exact_retry_wake_the_resource_actor() { - let _guard = crate::daemon::actors::supervisor::SUPERVISOR_TEST_LOCK - .lock() - .await; - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().join("state"))).unwrap(); - home.ensure().unwrap(); - let (supervisor, supervisor_handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let machine = LocalMachine { - identity: LocalIdentity::start(&home).unwrap(), - name: MachineName::fallback(), - protocol: SUPPORTED_PROTOCOLS, - }; - let authority = machine.identity.machine; - let resource_id = ResourceId::new(); - let resource = Resource::new( - resource_id, - "gpu-test".into(), - authority, - SupervisorAddress { - machine: authority, - thread: spec().thread, - }, - AssignmentRevision::new(0), - ResourceRevision::new(0), - None, - ); - call(&supervisor, |reply| SupervisorMsg::RegisterResource { - resource: Box::new(resource), - reply, - }) - .await - .unwrap(); - let route = local_route(authority, resource_id); - call(&store, |reply| StoreMsg::InsertOriginRoute { - route: Box::new(route.clone()), - reply, - }) - .await - .unwrap(); - let state = AppState { - home: home.clone(), - store: store.clone(), - supervisor: supervisor.clone(), - web: None, - content: None, - stream_slots: StreamSlots::new(), - machine, - fleet: FleetState::Disabled, - message_receiver: crate::daemon::message_receiver::MessageReceiver::default(), - locks: crate::daemon::DaemonLocks::default(), - thread_titles: None, - }; - - let first = submit_local(&state, &route).await.unwrap(); - assert_eq!(first.outcome, ResourceQueueOutcome::Waiting); - let inspection = call(&supervisor, |reply| SupervisorMsg::InspectResource { - id: resource_id, - reply, - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::AttentionRequired { - request, - reason: ResourceQueueAttentionReason::IdleNotProven { - gap: crate::resource::IdleProofGap::NoIdleEvidence, - }, - }) if request.request_id == route.request - )); - - assert_eq!(submit_local(&state, &route).await.unwrap(), first); - let requests = call(&store, |reply| StoreMsg::ResourceRequests { - authority_machine: authority, - resource_id, - reply, - }) - .await - .unwrap(); - assert_eq!(requests.len(), 1); - - supervisor.stop(None); - let _ = supervisor_handle.await; - } - - #[test] - fn response_requires_protocol_and_exact_receipt_identity() { - let route = route(); - - assert!(decode_resource_response(&route, response(&route, 7), 7).is_ok()); - assert!(matches!( - decode_resource_response(&route, response(&route, 6), 7), - Err(AppError::SubmissionOutcomeUnknown { .. }) - )); - - let mut wrong_identity: Value = serde_json::from_slice(&response(&route, 7).body).unwrap(); - wrong_identity["receipt"]["task"] = serde_json::to_value(TaskId::new()).unwrap(); - let invalid = ClusterResponse { - status: StatusCode::OK, - body: Bytes::from(serde_json::to_vec(&wrong_identity).unwrap()), - }; - assert!(matches!( - decode_resource_response(&route, invalid, 7), - Err(AppError::SubmissionOutcomeUnknown { .. }) - )); - assert!(route_acceptance_is_unknown(&route)); - } - - #[test] - fn failed_response_stays_unknown() { - let route = route(); - let failed = ClusterResponse { - status: StatusCode::BAD_GATEWAY, - body: Bytes::new(), - }; - - assert!(matches!( - decode_resource_response(&route, failed, 7), - Err(AppError::SubmissionOutcomeUnknown { .. }) - )); - assert!(route_acceptance_is_unknown(&route)); - } - - #[test] - fn retry_compares_authority_resource_and_spec_but_reuses_saved_callback() { - let route = route(); - let retry = input(&route); - - assert!(validate_retry(&route, &retry).is_ok()); - - let mut changed_authority = retry.clone(); - changed_authority.authority = MachineId::new(); - assert!(matches!( - validate_retry(&route, &changed_authority), - Err(AppError::SubmissionConflict { .. }) - )); - - let mut changed_resource = retry.clone(); - changed_resource.resource = ResourceId::new(); - assert!(matches!( - validate_retry(&route, &changed_resource), - Err(AppError::SubmissionConflict { .. }) - )); - - let mut changed_spec = retry; - changed_spec.spec.name = serde_json::from_value(serde_json::json!("changed")).unwrap(); - assert!(matches!( - validate_retry(&route, &changed_spec), - Err(AppError::SubmissionConflict { .. }) - )); - } - - #[test] - fn saved_route_phase_maps_to_typed_outcomes() { - let mut route = route(); - route.submission = SubmissionState::Resource { - resource: resource_from_route(&route).unwrap(), - phase: ResourceRoutePhase::Rejected { - reason: "unavailable".into(), - }, - }; - - assert_eq!( - outcome_from_route(&route).unwrap(), - ResourceSubmitOutcome::Rejected { - task: route.task, - reason: "unavailable".into(), - } - ); - } - - #[test] - fn recovery_retry_requires_exact_saved_identity_and_skips_later_phases() { - let expected = route(); - let saved = expected.clone(); - - assert!(ensure_same_recovery_identity(&expected, &saved).is_ok()); - assert_eq!(saved.request, expected.request); - assert_eq!(saved.task, expected.task); - assert_eq!(resource_from_route(&saved), resource_from_route(&expected)); - assert_eq!(saved.execution_machine, expected.execution_machine); - assert_eq!(saved.callback, expected.callback); - assert_eq!(saved.current_spec(), expected.current_spec()); - - let mut changed_request = saved.clone(); - changed_request.request = RequestId::new(); - assert!(matches!( - ensure_same_recovery_identity(&expected, &changed_request), - Err(AppError::SubmissionConflict { .. }) - )); - - let mut changed_task = saved.clone(); - changed_task.task = TaskId::new(); - assert!(matches!( - ensure_same_recovery_identity(&expected, &changed_task), - Err(AppError::SubmissionConflict { .. }) - )); - - let mut changed_authority = saved.clone(); - changed_authority.execution_machine = MachineId::new(); - assert!(matches!( - ensure_same_recovery_identity(&expected, &changed_authority), - Err(AppError::SubmissionConflict { .. }) - )); - - let mut changed_resource = saved.clone(); - let SubmissionState::Resource { phase, .. } = &saved.submission else { - unreachable!(); - }; - changed_resource.submission = SubmissionState::Resource { - resource: ResourceId::new(), - phase: phase.clone(), - }; - assert!(matches!( - ensure_same_recovery_identity(&expected, &changed_resource), - Err(AppError::SubmissionConflict { .. }) - )); - - let mut changed_spec = saved.clone(); - changed_spec.spec.current_mut().unwrap().name = - serde_json::from_value(serde_json::json!("changed")).unwrap(); - assert!(matches!( - ensure_same_recovery_identity(&expected, &changed_spec), - Err(AppError::SubmissionConflict { .. }) - )); - - let mut changed_callback = saved.clone(); - changed_callback.callback.cwd = Path::new("/changed/callback").to_path_buf(); - assert!(matches!( - ensure_same_recovery_identity(&expected, &changed_callback), - Err(AppError::SubmissionConflict { .. }) - )); - - let resource = resource_from_route(&saved).unwrap(); - let mut activated = saved.clone(); - activated.submission = SubmissionState::Resource { - resource, - phase: ResourceRoutePhase::Activated, - }; - assert!(ensure_same_recovery_identity(&expected, &activated).is_ok()); - assert!(!route_acceptance_is_unknown(&activated)); - assert_eq!( - outcome_from_route(&activated).unwrap(), - ResourceSubmitOutcome::Activated { - task: expected.task, - } - ); - - let mut cancelled = saved.clone(); - cancelled.submission = SubmissionState::Resource { - resource, - phase: ResourceRoutePhase::CancelledBeforeLaunch, - }; - assert!(ensure_same_recovery_identity(&expected, &cancelled).is_ok()); - assert!(!route_acceptance_is_unknown(&cancelled)); - assert_eq!( - outcome_from_route(&cancelled).unwrap(), - ResourceSubmitOutcome::CancelledBeforeLaunch { - task: expected.task, - } - ); - } - - #[test] - fn response_identity_includes_authority_and_resource() { - let route = route(); - let mut wrong_resource: Value = serde_json::from_slice(&response(&route, 7).body).unwrap(); - wrong_resource["receipt"]["resource"] = serde_json::to_value(ResourceId::new()).unwrap(); - let invalid = ClusterResponse { - status: StatusCode::OK, - body: Bytes::from(serde_json::to_vec(&wrong_resource).unwrap()), - }; - assert!(matches!( - decode_resource_response(&route, invalid, 7), - Err(AppError::SubmissionOutcomeUnknown { .. }) - )); - - let mut wrong_destination: Value = - serde_json::from_slice(&response(&route, 7).body).unwrap(); - wrong_destination["destination_machine"] = serde_json::to_value(MachineId::new()).unwrap(); - let invalid = ClusterResponse { - status: StatusCode::OK, - body: Bytes::from(serde_json::to_vec(&wrong_destination).unwrap()), - }; - assert!(matches!( - decode_resource_response(&route, invalid, 7), - Err(AppError::SubmissionOutcomeUnknown { .. }) - )); - } -} diff --git a/src/daemon/web.rs b/src/daemon/web.rs index 7235d74..ee48af1 100644 --- a/src/daemon/web.rs +++ b/src/daemon/web.rs @@ -1,18 +1,16 @@ -//! Optional TCP listener: dashboard assets, the read-only API, the three typed -//! resource controls, and, when the fleet is enabled, the `/v1/cluster/*` routes +//! Optional TCP listener: dashboard assets, the read-only API, and, when the +//! fleet is enabled, the `/v1/cluster/*` routes //! //! Off unless `--web-listen` / `HOMEBASED_WEB_LISTEN` is a host:port. A TCP port //! is reachable from any web page the user has open, so this router never -//! exposes task submit or cancel. The only browser write is -//! `POST /v1/resources/{id}/actions`, which requires the exact dashboard -//! `Origin`, a JSON body within a small limit, and no CORS response headers +//! exposes task submit or cancel use std::fmt; use std::net::SocketAddr; use std::str::FromStr; -use axum::extract::{DefaultBodyLimit, Request}; -use axum::http::{HeaderMap, StatusCode, Uri, header}; +use axum::extract::Request; +use axum::http::{StatusCode, Uri, header}; use axum::middleware::{self, Next}; use axum::response::{IntoResponse, Response}; use axum::{Json, Router}; @@ -21,15 +19,12 @@ use tokio::net::TcpListener; use tracing::warn; use crate::daemon::AppState; -use crate::daemon::{api, cluster, resource_api}; +use crate::daemon::{api, cluster}; use crate::domain::API_VERSION; use crate::error::AppError; use crate::files::{HostPolicy, host_guard}; use serde_json::json; -/// Largest JSON body accepted by the dashboard resource-action route -const BROWSER_ACTION_MAX_BYTES: usize = 16 * 1024; - /// Usual dashboard port when an operator opts in pub const DEFAULT_PORT: u16 = 7677; @@ -90,7 +85,7 @@ pub fn url_for(addr: SocketAddr) -> String { format!("http://{addr}") } -/// Read-only API, the typed resource controls, and the embedded single-page app +/// Read-only API and the embedded single-page app pub fn router(state: AppState, bind: SocketAddr) -> Router { let policy = HostPolicy { bind }; let routes = match state.fleet.handle() { @@ -98,7 +93,6 @@ pub fn router(state: AppState, bind: SocketAddr) -> Router { None => api::read_routes(), }; routes - .merge(browser_action_routes()) .fallback(asset) .layer(middleware::from_fn(move |request: Request, next: Next| { let policy = policy.clone(); @@ -107,53 +101,6 @@ pub fn router(state: AppState, bind: SocketAddr) -> Router { .with_state(state) } -/// The resource-action route with its browser-only request checks -fn browser_action_routes() -> Router { - resource_api::action_routes() - .layer(DefaultBodyLimit::max(BROWSER_ACTION_MAX_BYTES)) - .layer(middleware::from_fn(same_origin_guard)) -} - -/// Refuse a browser write unless it comes from this dashboard's own origin -/// -/// The Host guard runs first, so the Host value is one this listener serves -/// Browsers always send `Origin` on a cross-origin POST, and a missing value is -/// refused too, so a page on another origin cannot use this route -async fn same_origin_guard(request: Request, next: Next) -> Response { - match check_same_origin(request.headers()) { - Ok(()) => next.run(request).await, - Err(message) => AppError::Permission { - message: message.into(), - } - .into_response(), - } -} - -fn check_same_origin(headers: &HeaderMap) -> Result<(), &'static str> { - let header_text = |name| headers.get(name).and_then(|value| value.to_str().ok()); - let host = header_text(header::HOST).ok_or("resource actions require a Host header")?; - let origin = header_text(header::ORIGIN).ok_or("resource actions require an Origin header")?; - if origin != format!("http://{host}") { - return Err("resource actions require the exact dashboard Origin"); - } - let fetch_site = headers - .get("sec-fetch-site") - .and_then(|value| value.to_str().ok()); - if fetch_site.is_some_and(|site| site != "same-origin") { - return Err("resource actions require a same-origin browser request"); - } - let json = header_text(header::CONTENT_TYPE).is_some_and(|value| { - value - .split(';') - .next() - .is_some_and(|media| media.trim().eq_ignore_ascii_case("application/json")) - }); - if !json { - return Err("resource actions require an application/json body"); - } - Ok(()) -} - /// Output of `npm run build` in `web/`. Empty until the dashboard is built; /// `build.rs` creates the directory so the crate always compiles #[derive(RustEmbed)] diff --git a/src/digest.rs b/src/digest.rs deleted file mode 100644 index 6991a4e..0000000 --- a/src/digest.rs +++ /dev/null @@ -1,172 +0,0 @@ -//! SHA-256 digests with one lowercase hexadecimal text form - -use std::fmt; -use std::str::FromStr; - -use serde::{Deserialize, Deserializer, Serialize, Serializer}; -use sha2::{Digest, Sha256}; - -/// Number of bytes in one SHA-256 digest -const SHA256_LEN: usize = 32; - -/// Raw SHA-256 digest -/// -/// Its text, JSON, and database form is exactly 64 lowercase hexadecimal -/// characters. Decoding refuses any other text, so a saved or received digest -/// is always in canonical form and two equal digests always have equal text -#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)] -pub struct Sha256Digest([u8; SHA256_LEN]); - -/// Text that is not 64 lowercase hexadecimal characters -#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] -#[error("SHA-256 digest must be 64 lowercase hexadecimal characters")] -pub struct InvalidSha256Hex; - -impl Sha256Digest { - /// Hash the exact bytes - #[must_use] - pub fn of(bytes: impl AsRef<[u8]>) -> Self { - Self(Sha256::digest(bytes).into()) - } - - /// Wrap a finished 32-byte digest - #[must_use] - pub const fn from_bytes(bytes: [u8; SHA256_LEN]) -> Self { - Self(bytes) - } - - /// Borrow the raw 32-byte digest - #[must_use] - pub const fn as_bytes(&self) -> &[u8; SHA256_LEN] { - &self.0 - } - - /// Return the digest as 64 lowercase hexadecimal characters - #[must_use] - pub fn to_hex(self) -> String { - self.to_string() - } - - /// Parse exactly 64 lowercase hexadecimal characters - pub fn from_hex(text: &str) -> Result { - let (pairs, remainder) = text.as_bytes().as_chunks::<2>(); - if pairs.len() != SHA256_LEN || !remainder.is_empty() { - return Err(InvalidSha256Hex); - } - - let mut digest = [0_u8; SHA256_LEN]; - for (byte, [high, low]) in digest.iter_mut().zip(pairs) { - *byte = (hex_value(*high)? << 4) | hex_value(*low)?; - } - Ok(Self(digest)) - } -} - -impl From<[u8; SHA256_LEN]> for Sha256Digest { - fn from(bytes: [u8; SHA256_LEN]) -> Self { - Self(bytes) - } -} - -impl From for Sha256Digest { - fn from(hasher: Sha256) -> Self { - Self(hasher.finalize().into()) - } -} - -impl fmt::Display for Sha256Digest { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - const HEX: &[u8; 16] = b"0123456789abcdef"; - - for byte in self.0 { - let high = char::from(HEX[usize::from(byte >> 4)]); - let low = char::from(HEX[usize::from(byte & 0x0f)]); - write!(formatter, "{high}{low}")?; - } - Ok(()) - } -} - -// debug output shows the same hex text as logs and JSON instead of a byte array -impl fmt::Debug for Sha256Digest { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - write!(formatter, "Sha256Digest({self})") - } -} - -impl FromStr for Sha256Digest { - type Err = InvalidSha256Hex; - - fn from_str(text: &str) -> Result { - Self::from_hex(text) - } -} - -impl Serialize for Sha256Digest { - fn serialize(&self, serializer: S) -> Result - where - S: Serializer, - { - serializer.collect_str(self) - } -} - -impl<'de> Deserialize<'de> for Sha256Digest { - fn deserialize(deserializer: D) -> Result - where - D: Deserializer<'de>, - { - // a borrowed str is not always available, for example from a `Value` - let text = String::deserialize(deserializer)?; - Self::from_hex(&text).map_err(serde::de::Error::custom) - } -} - -fn hex_value(digit: u8) -> Result { - match digit { - b'0'..=b'9' => Ok(digit - b'0'), - b'a'..=b'f' => Ok(digit - b'a' + 10), - _ => Err(InvalidSha256Hex), - } -} - -#[cfg(test)] -mod tests { - use serde_json::json; - - use super::{InvalidSha256Hex, Sha256Digest}; - - #[test] - fn hex_text_round_trips_through_json() { - let digest = Sha256Digest::of(b"abc"); - let text = "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"; - - assert_eq!(digest.to_hex(), text); - assert_eq!(serde_json::to_value(digest).unwrap(), json!(text)); - assert_eq!( - serde_json::from_value::(json!(text)).unwrap(), - digest - ); - } - - #[test] - fn decoding_refuses_noncanonical_text() { - let upper = "BA7816BF8F01CFEA414140DE5DAE2223B00361A396177A9CB410FF61F20015AD"; - for text in [ - upper, - "", - "ab", - &"a".repeat(63), - &"a".repeat(65), - &"g".repeat(64), - &format!("{}é", "a".repeat(62)), - ] { - assert_eq!( - Sha256Digest::from_hex(text), - Err(InvalidSha256Hex), - "{text}" - ); - assert!(serde_json::from_value::(json!(text)).is_err()); - } - } -} diff --git a/src/domain.rs b/src/domain.rs index 81ec38c..64b8481 100644 --- a/src/domain.rs +++ b/src/domain.rs @@ -16,9 +16,9 @@ pub const API_VERSION: u32 = 1; /// SQLite `user_version` /// -/// Released versions 1, 2, and 27 through 33 migrate in place to this version +/// Released versions 1, 2, and 27 through 34 migrate in place to this version /// Unreleased development versions 3 through 26 are refused -pub const SCHEMA_VERSION: i64 = 34; +pub const SCHEMA_VERSION: i64 = 35; /// Maximum Unicode scalar values in a submitted task name pub const TASK_NAME_MAX_CHARS: usize = 120; diff --git a/src/error.rs b/src/error.rs index 7c5e109..f5cfa24 100644 --- a/src/error.rs +++ b/src/error.rs @@ -6,14 +6,12 @@ use std::path::PathBuf; use std::process::ExitCode; use serde_json::{Value, json}; -use uuid::Uuid; use crate::callback::destination::ThreadSuggestion; use crate::dependency::DependencyOutcome; use crate::domain::{AgentKind, ProcessStatus, TaskId, ThreadId}; use crate::fleet::protocol::ProtocolRange; use crate::machine::{MachineId, MachineName}; -use crate::resource::ResourceId; use crate::spec::CwdProblem; use crate::submission::RequestId; @@ -380,12 +378,6 @@ pub enum AppError { /// Task with conflicting ownership task: TaskId, }, - /// Resource-owned work needs an authority-specific cancellation route - #[error("resource cancellation is unavailable for task {task}")] - ResourceCancellationUnavailable { - /// Task whose resource authority must own cancellation - task: TaskId, - }, /// A previously accepted event sequence has different content #[error("event content conflict: {task} sequence {seq}")] EventContentConflict { @@ -454,81 +446,11 @@ pub enum AppError { /// Remote accepted range remote: ProtocolRange, }, - /// Unexpected internal failure - /// No checked authority owns this resource - #[error("resource not found: {}", resource.as_uuid())] - ResourceNotFound { - /// Resource that no checked authority owns - resource: ResourceId, - }, - /// Some machines could not be checked, so the resource may still exist - #[error("resource lookup incomplete for {}", resource.as_uuid())] - ResourceLookupIncomplete { - /// Resource being located - resource: ResourceId, - /// Machines that did not answer - unchecked: Vec, - }, - /// The fixed resource authority could not be reached before a mutation was sent - #[error("resource authority {machine} unavailable: {message}")] - ResourceAuthorityUnavailable { - /// Resource owned by the authority - resource: ResourceId, - /// Fixed resource authority - machine: MachineId, - /// Why the authority is unavailable - message: String, - }, - /// A resource mutation may have committed; retry the same operation identity - #[error("resource operation outcome unknown: {message}")] - ResourceOutcomeUnknown { - /// Resource that the mutation targets - resource: ResourceId, - /// Stable operation identity, when the mutation has one - operation: Option, - /// Why the outcome is unknown - message: String, - }, - /// The caller observed an older resource revision - #[error("resource revision is stale: expected {expected}, current {current}")] - ResourceStaleRevision { - /// Resource whose revision changed - resource: ResourceId, - /// Revision that the caller observed - expected: u64, - /// Current authority revision - current: u64, - }, - /// A resource operation identity was reused with different content - #[error("resource operation conflict: {message}")] - ResourceOperationConflict { - /// Resource that the operation targets - resource: ResourceId, - /// Reused operation identity, when the mutation has one - operation: Option, - /// Conflict detail - message: String, - }, - /// The action does not apply to the current resource phase - #[error("resource action not allowed: {message}")] - ResourceActionNotAllowed { - /// Resource that the action targets - resource: ResourceId, - /// Why the current phase refuses the action - message: String, - }, - /// The operation needs an owner proof that this daemon cannot supply yet - #[error("resource operation unavailable: {message}")] - ResourceOperationUnavailable { - /// Resource that the operation targets - resource: ResourceId, - /// Missing owner or proof - message: String, - }, /// A daemon actor did not answer within the call timeout. The operation may /// still complete after the caller stops waiting #[error("daemon busy: an internal call timed out; the operation may still complete")] DaemonBusy, + /// Unexpected internal failure #[error("{message}")] Internal { /// Underlying failure text @@ -622,7 +544,6 @@ impl AppError { Self::SubmissionConflict { .. } => "submission_conflict", Self::RouteNotFound { .. } => "route_not_found", Self::ClusterTaskConflict { .. } => "cluster_task_conflict", - Self::ResourceCancellationUnavailable { .. } => "resource_cancellation_unavailable", Self::EventContentConflict { .. } => "event_content_conflict", Self::MessageInvalid { .. } => "message_invalid", Self::MessageToSelf { .. } => "message_to_self", @@ -632,14 +553,6 @@ impl AppError { Self::MessageOutcomeUnknown { .. } => "message_outcome_unknown", Self::MessageUnavailable { .. } => "message_receiver_unavailable", Self::ClusterProtocolIncompatible { .. } => "cluster_protocol_incompatible", - Self::ResourceNotFound { .. } => "resource_not_found", - Self::ResourceLookupIncomplete { .. } => "resource_lookup_incomplete", - Self::ResourceAuthorityUnavailable { .. } => "resource_authority_unavailable", - Self::ResourceOutcomeUnknown { .. } => "resource_outcome_unknown", - Self::ResourceStaleRevision { .. } => "resource_stale_revision", - Self::ResourceOperationConflict { .. } => "resource_operation_conflict", - Self::ResourceActionNotAllowed { .. } => "resource_action_not_allowed", - Self::ResourceOperationUnavailable { .. } => "resource_operation_unavailable", Self::DaemonBusy => "daemon_busy", Self::Internal { .. } => "internal", } @@ -657,9 +570,6 @@ impl AppError { | Self::SubmissionOutcomeUnknown { .. } | Self::ClusterLookupIncomplete { .. } | Self::TaskUnavailable { .. } - | Self::ResourceLookupIncomplete { .. } - | Self::ResourceAuthorityUnavailable { .. } - | Self::ResourceOutcomeUnknown { .. } | Self::NotifyFailed { .. } | Self::DaemonBusy | Self::Internal { .. } => 1, @@ -683,8 +593,7 @@ impl AppError { | Self::FileNotFound { .. } | Self::NotDirectory { .. } | Self::UnsupportedFile { .. } - | Self::MachineNotFound { .. } - | Self::ResourceNotFound { .. } => 3, + | Self::MachineNotFound { .. } => 3, Self::AgentThreadNotFound { .. } => 3, Self::Permission { .. } => 4, Self::TooManyReports { .. } @@ -700,18 +609,13 @@ impl AppError { | Self::StreamLimit { .. } | Self::MachineIdentityMismatch { .. } | Self::ClusterTaskConflict { .. } - | Self::ResourceCancellationUnavailable { .. } | Self::EventContentConflict { .. } | Self::DuplicateMachineIdentity { .. } | Self::DuplicateMachineName { .. } | Self::ClusterProtocolIncompatible { .. } => 5, Self::SubmissionRejected { .. } | Self::SubmissionConflict { .. } - | Self::MessageConflict { .. } - | Self::ResourceStaleRevision { .. } - | Self::ResourceOperationConflict { .. } - | Self::ResourceActionNotAllowed { .. } - | Self::ResourceOperationUnavailable { .. } => 5, + | Self::MessageConflict { .. } => 5, Self::MessageDeliveryFailed { .. } | Self::MessageOutcomeUnknown { .. } | Self::MessageUnavailable { .. } => 1, @@ -741,8 +645,7 @@ impl AppError { | Self::UnknownDependency { .. } | Self::ExecutableMissing { .. } | Self::FileNotFound { .. } - | Self::MachineNotFound { .. } - | Self::ResourceNotFound { .. } => http::StatusCode::NOT_FOUND, + | Self::MachineNotFound { .. } => http::StatusCode::NOT_FOUND, Self::AgentThreadNotFound { .. } => http::StatusCode::NOT_FOUND, Self::Permission { .. } => http::StatusCode::FORBIDDEN, Self::TooManyReports { .. } @@ -758,18 +661,13 @@ impl AppError { | Self::StreamLimit { .. } | Self::MachineIdentityMismatch { .. } | Self::ClusterTaskConflict { .. } - | Self::ResourceCancellationUnavailable { .. } | Self::EventContentConflict { .. } | Self::DuplicateMachineIdentity { .. } | Self::DuplicateMachineName { .. } | Self::ClusterProtocolIncompatible { .. } => http::StatusCode::CONFLICT, Self::SubmissionRejected { .. } | Self::SubmissionConflict { .. } - | Self::MessageConflict { .. } - | Self::ResourceStaleRevision { .. } - | Self::ResourceOperationConflict { .. } - | Self::ResourceActionNotAllowed { .. } - | Self::ResourceOperationUnavailable { .. } => http::StatusCode::CONFLICT, + | Self::MessageConflict { .. } => http::StatusCode::CONFLICT, Self::MachineUnavailable { .. } | Self::RemoteSubmissionUnavailable { .. } | Self::ClusterLookupIncomplete { .. } @@ -778,9 +676,6 @@ impl AppError { | Self::MessageOutcomeUnknown { .. } | Self::MessageUnavailable { .. } | Self::SubmissionOutcomeUnknown { .. } - | Self::ResourceLookupIncomplete { .. } - | Self::ResourceAuthorityUnavailable { .. } - | Self::ResourceOutcomeUnknown { .. } | Self::DaemonBusy => http::StatusCode::SERVICE_UNAVAILABLE, Self::DaemonUnavailable { .. } | Self::ConfigInvalid { .. } @@ -805,9 +700,6 @@ impl AppError { | Self::MessageDeliveryFailed { .. } | Self::MessageOutcomeUnknown { .. } | Self::MessageUnavailable { .. } - | Self::ResourceLookupIncomplete { .. } - | Self::ResourceAuthorityUnavailable { .. } - | Self::ResourceOutcomeUnknown { .. } ) } @@ -843,9 +735,7 @@ impl AppError { Self::DependencyFailed { task, outcome } => { json!({ "pointer": "/after", "task": task, "outcome": outcome }) } - Self::RouteNotFound { task } - | Self::ClusterTaskConflict { task } - | Self::ResourceCancellationUnavailable { task } => { + Self::RouteNotFound { task } | Self::ClusterTaskConflict { task } => { json!({ "task": task }) } Self::EventContentConflict { task, seq } => json!({ "task": task, "seq": seq }), @@ -916,39 +806,6 @@ impl AppError { Self::NotifyFailed { message } => json!({ "message": message }), Self::DaemonBusy => json!({}), Self::Internal { message } => json!({ "message": message }), - Self::ResourceNotFound { resource } => json!({ "resource_id": resource }), - Self::ResourceLookupIncomplete { - resource, - unchecked, - } => json!({ "resource_id": resource, "unchecked": unchecked }), - Self::ResourceAuthorityUnavailable { - resource, - machine, - message, - } => json!({ "resource_id": resource, "machine": machine, "message": message }), - Self::ResourceOutcomeUnknown { - resource, - operation, - message, - } => json!({ "resource_id": resource, "operation_id": operation, "message": message }), - Self::ResourceStaleRevision { - resource, - expected, - current, - } => json!({ - "resource_id": resource, - "expected_revision": expected, - "current_revision": current, - }), - Self::ResourceOperationConflict { - resource, - operation, - message, - } => json!({ "resource_id": resource, "operation_id": operation, "message": message }), - Self::ResourceActionNotAllowed { resource, message } - | Self::ResourceOperationUnavailable { resource, message } => { - json!({ "resource_id": resource, "message": message }) - } Self::ConfigInvalid { path, .. } => json!({ "path": path }), Self::MachineNotFound { machine } => json!({ "machine": machine }), Self::MachineIdentityMismatch { expected, found } => { diff --git a/src/invocation.rs b/src/invocation.rs index f8fb6c2..024fd5f 100644 --- a/src/invocation.rs +++ b/src/invocation.rs @@ -410,7 +410,6 @@ pub fn invocation_from_normalized_for_identity( container, &CreateContext { task, - resource: None, cidfile: &cidfile, default_user: ContainerUser::current(), }, diff --git a/src/lib.rs b/src/lib.rs index ee8f227..ea2b896 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -11,7 +11,6 @@ pub mod container; pub(crate) mod curl; pub mod daemon; pub mod dependency; -pub mod digest; pub mod domain; pub mod error; pub mod events; @@ -25,7 +24,6 @@ pub mod message; pub mod notify; pub mod power; pub mod report; -pub mod resource; pub mod runner; pub mod spec; pub mod store; diff --git a/src/message.rs b/src/message.rs index d345a9b..9e90850 100644 --- a/src/message.rs +++ b/src/message.rs @@ -12,7 +12,6 @@ use uuid::Uuid; use crate::domain::{TaskId, ThreadId}; use crate::error::AppError; use crate::machine::MachineId; -use crate::resource::NoticeId; /// Maximum message body size in UTF-8 bytes pub const MESSAGE_BODY_MAX_BYTES: usize = 16 * 1024; @@ -82,7 +81,7 @@ impl<'de> Deserialize<'de> for MessageId { } } -/// Source identity for a message from a thread, task, or resource notice +/// Source identity for a message from a thread or task #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(tag = "kind", rename_all = "snake_case", deny_unknown_fields)] pub enum MessageSource { @@ -100,13 +99,6 @@ pub enum MessageSource { /// Source task UUID task: TaskId, }, - /// Reply route for a durable resource notice - ResourceNotice { - /// Machine that owns the resource authority - machine: MachineId, - /// Stable identity of the notice - notice_id: NoticeId, - }, } impl MessageSource { @@ -114,9 +106,7 @@ impl MessageSource { #[must_use] pub const fn machine(&self) -> MachineId { match self { - Self::Thread { machine, .. } - | Self::Task { machine, .. } - | Self::ResourceNotice { machine, .. } => *machine, + Self::Thread { machine, .. } | Self::Task { machine, .. } => *machine, } } } @@ -415,7 +405,7 @@ impl MessageRequest { destination_machine: self.destination_machine, source: self.source.clone(), recipient: self.recipient.clone(), - body: semantic_body(&self.source, &self.body), + body: self.body.clone(), reply_to: self.reply_to, conversation_id: self.conversation_id, } @@ -464,21 +454,6 @@ impl MessageRequest { } } -fn semantic_body(source: &MessageSource, body: &str) -> String { - if !matches!(source, MessageSource::ResourceNotice { .. }) { - return body.to_string(); - } - - // supervisor notices embed the wire version in their serialized backing body - let Ok(mut value) = serde_json::from_str::(body) else { - return body.to_string(); - }; - if let Some(object) = value.as_object_mut() { - object.remove("protocol_version"); - } - serde_json::to_string(&value).unwrap_or_else(|_| body.to_string()) -} - fn validate_body(body: &str) -> Result<(), AppError> { if body.trim().is_empty() { return Err(AppError::MessageInvalid { @@ -522,69 +497,6 @@ fn validate_cwd_selector(cwd: &std::path::Path) -> Result<(), AppError> { Ok(()) } -#[cfg(test)] -mod tests { - use super::{MessageId, MessageRequest, MessageSource, Recipient}; - use crate::domain::{API_VERSION, ThreadId}; - use crate::machine::MachineId; - use crate::resource::NoticeId; - use uuid::Uuid; - - #[test] - fn resource_notice_source_requires_a_non_nil_notice_identity() { - let notice_id = NoticeId::new(); - let request = MessageRequest { - api_version: API_VERSION, - protocol_version: 1, - message_id: MessageId::new(), - destination_machine: MachineId::new(), - source: MessageSource::ResourceNotice { - machine: MachineId::new(), - notice_id, - }, - recipient: Recipient::Thread { - thread: ThreadId(Uuid::now_v7()), - }, - body: "Resource notice source validation".into(), - reply_to: None, - conversation_id: Uuid::now_v7(), - }; - let mut wire = serde_json::to_value(&request).unwrap(); - assert!(serde_json::from_value::(wire.clone()).is_ok()); - - wire["source"]["notice_id"] = serde_json::json!(Uuid::nil()); - assert!(serde_json::from_value::(wire).is_err()); - } - - #[test] - fn semantic_identity_ignores_the_notice_wire_version_only() { - let mut request = MessageRequest { - api_version: API_VERSION, - protocol_version: 1, - message_id: MessageId::new(), - destination_machine: MachineId::new(), - source: MessageSource::ResourceNotice { - machine: MachineId::new(), - notice_id: NoticeId::new(), - }, - recipient: Recipient::Thread { - thread: ThreadId(Uuid::now_v7()), - }, - body: serde_json::json!({"protocol_version": 1, "reason": "review"}).to_string(), - reply_to: None, - conversation_id: Uuid::now_v7(), - }; - let identity = request.identity(); - - request.protocol_version = 2; - request.body = serde_json::json!({"protocol_version": 2, "reason": "review"}).to_string(); - assert_eq!(request.identity(), identity); - - request.body = serde_json::json!({"protocol_version": 2, "reason": "changed"}).to_string(); - assert_ne!(request.identity(), identity); - } -} - /// Durable binding between a message request and its resolved local destination #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(deny_unknown_fields)] diff --git a/src/resource.rs b/src/resource.rs deleted file mode 100644 index bfb91fe..0000000 --- a/src/resource.rs +++ /dev/null @@ -1,2255 +0,0 @@ -//! Typed resource, request, loan, and supervisor-notice domain models - -use crate::error::AppError; -use std::path::PathBuf; - -use serde::{Deserialize, Deserializer, Serialize}; - -use crate::digest::Sha256Digest; -use crate::domain::{API_VERSION, ExitReason, ProcessStatus, TaskId, ThreadId}; -use crate::machine::MachineId; -use crate::spec::{NormalizedSpec, NormalizedWorkload}; -use crate::submission::{NormalizedSpecSha256, RequestId, ResourceQueueReceipt}; - -pub mod api; -pub mod background_launch; -pub mod bound_action; -pub mod command_shape; -pub mod foreground; -mod id; -pub mod initial_idle; -pub mod operator_release; -pub mod ownership_lock; -mod release_checkpoint; -pub mod release_watcher; -pub mod return_window; -pub(crate) mod store; -pub mod trainer_publication; - -/// Longest time an assigned command may sit with an unknown launch outcome -/// -/// The resource actor starts the clock when it first sees the outcome as -/// unknown, such as after a daemon restart mid-launch or a supervisor reply -/// that never arrived. After the bound, a task that has not started is failed -/// before launch with `launch_unconfirmed`, its thread is told, and the queue -/// serves the next request. A worker must win the task's Queued-to-Running -/// transition before it spawns anything, so failing a still-queued task cannot -/// race a child that already started -pub const LAUNCH_CONFIRMATION_BOUND: std::time::Duration = std::time::Duration::from_secs(120); - -pub use id::{ - ActionId, DeliveryAttemptId, IdentityParseError, LoanId, NilIdentity, NoticeId, - ReleaseStopReservationId, ResourceId, -}; -pub(crate) use release_checkpoint::{ - ReleaseCheckpointAction, ReleaseCheckpointBaseline, ReleaseCheckpointBinding, - ReleaseCheckpointCancellation, ReleaseCheckpointPhase, ReleaseCheckpointState, - ReleaseCheckpointStopDecision, ReleaseCheckpointStopOutcome, -}; -pub use return_window::{ - InvalidReturnDecisionWindow, RETURN_DECISION_GRACE, RETURN_DECISION_LIMIT, - ReturnDecisionWindow, ReturnHoldRejection, -}; - -/// Immutable authority-assigned acceptance identity for a resource request -#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -#[serde(transparent)] -pub struct AcceptanceSequence(u64); - -impl AcceptanceSequence { - /// Wrap an authority-assigned sequence value - #[must_use] - pub const fn new(value: u64) -> Self { - Self(value) - } - - /// Return the underlying sequence value - #[must_use] - pub const fn get(self) -> u64 { - self.0 - } -} - -/// Revision of resource-owned state used to reject stale actions -#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -#[serde(transparent)] -pub struct ResourceRevision(u64); - -impl ResourceRevision { - /// Wrap a resource state revision - #[must_use] - pub const fn new(value: u64) -> Self { - Self(value) - } - - /// Return the underlying revision value - #[must_use] - pub const fn get(self) -> u64 { - self.0 - } - - /// Return the revision that follows this one, or `None` at the maximum value - #[must_use] - pub const fn next(self) -> Option { - match self.0.checked_add(1) { - Some(value) => Some(Self(value)), - None => None, - } - } -} - -/// Revision of the resource supervisor assignment -#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -#[serde(transparent)] -pub struct AssignmentRevision(u64); - -impl AssignmentRevision { - /// Wrap a supervisor assignment revision - #[must_use] - pub const fn new(value: u64) -> Self { - Self(value) - } - - /// Return the underlying revision value - #[must_use] - pub const fn get(self) -> u64 { - self.0 - } -} - -/// Exact fleet address of the supervisor assigned to a resource -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct SupervisorAddress { - /// Machine that owns the supervisor thread - pub machine: MachineId, - /// Exact Codex thread that receives supervisor notices - pub thread: ThreadId, -} - -/// One exclusive physical GPU and its fixed authority machine -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct Resource { - /// Stable resource identity - pub id: ResourceId, - /// Human-readable resource name - pub display_name: String, - authority_machine: MachineId, - /// Exact assigned supervisor - pub supervisor: SupervisorAddress, - /// Revision of the supervisor assignment - pub assignment_revision: AssignmentRevision, - /// Revision of resource-owned state - pub state_revision: ResourceRevision, - /// Registered background task identity, if one exists - pub registered_background_task: Option, -} - -impl Resource { - /// Create a resource with an authority that cannot be changed through this model - #[must_use] - pub fn new( - id: ResourceId, - display_name: String, - authority_machine: MachineId, - supervisor: SupervisorAddress, - assignment_revision: AssignmentRevision, - state_revision: ResourceRevision, - registered_background_task: Option, - ) -> Self { - Self { - id, - display_name, - authority_machine, - supervisor, - assignment_revision, - state_revision, - registered_background_task, - } - } - - /// Return the fixed authority machine for this resource - #[must_use] - pub const fn authority_machine(&self) -> MachineId { - self.authority_machine - } -} - -/// Immutable content of the first registration of one resource -/// -/// A later replacement changes the current supervisor, so a registration retry -/// compares this saved content and never the mutable resource row -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(crate) struct ResourceRegistrationReceipt { - /// Registered resource identity - pub(crate) resource_id: ResourceId, - /// Display name named by the first registration - pub(crate) display_name: String, - /// Fixed authority that accepted the first registration - pub(crate) authority_machine: MachineId, - /// Supervisor named by the first registration - pub(crate) initial_supervisor: SupervisorAddress, -} - -impl ResourceRegistrationReceipt { - /// Registration content named by a resource before its first insert - pub(crate) fn initial(resource: &Resource) -> Self { - Self { - resource_id: resource.id, - display_name: resource.display_name.clone(), - authority_machine: resource.authority_machine, - initial_supervisor: resource.supervisor, - } - } -} - -/// Durable association between one authority-owned resource and one trainer task -/// -/// This immutable record keeps exact point-in-time trainer request and lock evidence -/// with the normalized spec digest from the accepted Homebased task identity -/// The Store checks the direct-segment command shape before its first insert -/// This record does not prove that the live process used the lock or that the GPU worker exited -/// It cannot permit release completion or `Serving` -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct TrainerAttemptAssociation { - resource_id: ResourceId, - authority_machine: MachineId, - task_id: TaskId, - verified_attempt: ownership_lock::VerifiedTrainerAttempt, - normalized_spec_sha256: NormalizedSpecSha256, -} - -impl TrainerAttemptAssociation { - /// Return the associated authority-owned resource - #[must_use] - pub const fn resource_id(&self) -> ResourceId { - self.resource_id - } - - /// Return the machine that owns the associated resource - #[must_use] - pub const fn authority_machine(&self) -> MachineId { - self.authority_machine - } - - /// Return the exact registered Homebased background task - #[must_use] - pub const fn task_id(&self) -> TaskId { - self.task_id - } - - /// Return the exact trainer attempt evidence captured at registration time - #[must_use] - pub const fn verified_attempt(&self) -> &ownership_lock::VerifiedTrainerAttempt { - &self.verified_attempt - } - - /// Return the digest derived from the accepted executor identity's normalized spec - #[must_use] - pub const fn normalized_spec_sha256(&self) -> NormalizedSpecSha256 { - self.normalized_spec_sha256 - } - - pub(crate) fn from_components( - resource_id: ResourceId, - authority_machine: MachineId, - task_id: TaskId, - verified_attempt: ownership_lock::VerifiedTrainerAttempt, - normalized_spec_sha256: NormalizedSpecSha256, - ) -> Result { - if authority_machine.as_uuid().is_nil() || task_id.0.is_nil() { - return Err("trainer attempt association identities must not be nil"); - } - - Ok(Self { - resource_id, - authority_machine, - task_id, - verified_attempt, - normalized_spec_sha256, - }) - } -} - -/// A normalized specification restricted to finite command or container workloads -#[derive(Debug, Clone, PartialEq, Eq, Serialize)] -#[serde(transparent)] -pub struct CommandSpec(NormalizedSpec); - -impl CommandSpec { - /// Borrow the immutable normalized specification - #[must_use] - pub const fn as_normalized(&self) -> &NormalizedSpec { - &self.0 - } -} - -impl TryFrom for CommandSpec { - type Error = CommandSpecError; - - fn try_from(spec: NormalizedSpec) -> Result { - if spec.machine.is_some() { - return Err(CommandSpecError::ExplicitMachine); - } - - match &spec.workload { - NormalizedWorkload::Task(_) => Ok(Self(spec)), - // resource work holds one exact GPU, so the container must name it - NormalizedWorkload::Container(container) if container.gpus.is_none() => { - Err(CommandSpecError::ContainerWithoutGpus) - } - NormalizedWorkload::Container(_) => Ok(Self(spec)), - NormalizedWorkload::Agent(_) => Err(CommandSpecError::AgentWorkload), - } - } -} - -impl<'de> Deserialize<'de> for CommandSpec { - fn deserialize(deserializer: D) -> Result - where - D: Deserializer<'de>, - { - let spec = NormalizedSpec::deserialize(deserializer)?; - Self::try_from(spec).map_err(serde::de::Error::custom) - } -} - -/// Why a normalized specification cannot enter the resource queue -#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] -pub enum CommandSpecError { - /// Resource command execution is routed by the resource authority - #[error("resource command specifications cannot specify a machine")] - ExplicitMachine, - /// Agent workloads are open-ended and cannot enter the finite command queue - #[error("resource requests require a command or container workload")] - AgentWorkload, - /// A container on a resource must request its GPUs - #[error("resource container workloads require gpus")] - ContainerWithoutGpus, -} - -/// Queue and task-result state for a resource request -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceRequestState { - /// Ready for selection in the authority's serving order - Queued, - /// Reserved by one loan and awaiting task-layer execution - Assigned { - /// Loan that owns this request - loan_id: LoanId, - }, - /// The task ended and its process outcome is known - Finished { - /// Existing task-layer process outcome - outcome: ExitReason, - }, - /// Cancelled before the command task was activated - CancelledBeforeLaunch, - /// Rejected before command activation - Rejected { - /// Durable rejection reason - reason: String, - }, -} - -/// Accepted command request waiting for one exclusive resource -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceRequest { - /// Caller-provided retry identity - pub request_id: RequestId, - /// Preallocated task identity retained when the command is selected - pub task_id: TaskId, - /// Resource authority that accepted the request - pub resource_id: ResourceId, - /// Immutable authority-assigned acceptance identity - pub acceptance_sequence: AcceptanceSequence, - /// Machine that owns the requesting thread and callback route - pub origin_machine: MachineId, - /// Immutable normalized command specification - spec: CommandSpec, - /// Resource queue state, separate from task process state - pub state: ResourceRequestState, -} - -impl ResourceRequest { - /// Create a queued request, rejecting agent workloads before acceptance - pub fn new( - request_id: RequestId, - task_id: TaskId, - resource_id: ResourceId, - acceptance_sequence: AcceptanceSequence, - origin_machine: MachineId, - normalized_spec: NormalizedSpec, - ) -> Result { - Ok(Self { - request_id, - task_id, - resource_id, - acceptance_sequence, - origin_machine, - spec: CommandSpec::try_from(normalized_spec)?, - state: ResourceRequestState::Queued, - }) - } - - /// Borrow the immutable command specification - #[must_use] - pub const fn spec(&self) -> &CommandSpec { - &self.spec - } -} - -/// Strict versioned request to queue one command on a resource authority -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceQueueRequest { - /// Public API schema version - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Intended resource authority - pub destination_machine: MachineId, - /// Machine that owns the requesting thread and callback route - pub origin_machine: MachineId, - /// Stable caller retry identity - pub request_id: RequestId, - /// Preallocated global task identity - pub task_id: TaskId, - /// Resource whose serving queue will receive the command - pub resource_id: ResourceId, - /// Command workload with no explicit execution machine - pub spec: CommandSpec, -} - -impl ResourceQueueRequest { - /// Build a queue request with the current public API version - #[must_use] - pub fn new( - protocol_version: u32, - destination_machine: MachineId, - origin_machine: MachineId, - request_id: RequestId, - task_id: TaskId, - resource_id: ResourceId, - spec: CommandSpec, - ) -> Self { - Self { - api_version: API_VERSION, - protocol_version, - destination_machine, - origin_machine, - request_id, - task_id, - resource_id, - spec, - } - } -} - -/// Strict versioned response from a resource queue authority -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceQueueResponse { - /// Public API schema version - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Machine that handled the queue request - pub destination_machine: MachineId, - /// Definitive receipt bound to the resource route identity - pub receipt: ResourceQueueReceipt, -} - -impl ResourceQueueResponse { - /// Build a response whose destination matches the receipt authority - #[must_use] - pub fn new(protocol_version: u32, receipt: ResourceQueueReceipt) -> Self { - Self { - api_version: API_VERSION, - protocol_version, - destination_machine: receipt.authority_machine, - receipt, - } - } -} - -/// Opaque training and result references retained for the return decision -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ReturnContext { - /// A training task stopped after publishing a newer checkpoint - Stopped { - /// Exact background task that was stopped - task_id: TaskId, - /// Opaque identity of the complete checkpoint publication - checkpoint_ref: String, - /// Opaque immutable recovery instructions or reference - recovery_ref: String, - }, - /// A training task finished before it was stopped - AlreadyCompleted { - /// Exact background task that completed - task_id: TaskId, - /// Opaque final result reference - result_ref: String, - }, - /// A training task ended with no usable result and no saved checkpoint stop - /// - /// No evidence names a checkpoint that the run can resume from. The release - /// basis is saved beside this context: either the authority proved that the - /// exact trainer worker released its saved ownership lock, or an operator - /// attested that its GPU work is gone. The supervisor can start new work or - /// record no-resume; a same-run resume is never valid for this context - EndedWithoutResult { - /// Exact background task that ended - task_id: TaskId, - /// Task-layer outcome of the ended run - outcome: ExitReason, - }, - /// A training task was lost with no exit reason, no result, and no checkpoint stop - /// - /// Only an operator attestation releases a lost trainer, and its receipt is - /// the release basis. The supervisor can start new work or record - /// no-resume; a same-run resume is never valid for this context - LostWithoutResult { - /// Exact background task that was lost - task_id: TaskId, - }, - /// No registered live background task occupied the resource - Idle, -} - -impl ReturnContext { - /// Return the released background task named by this context, if any - #[must_use] - pub const fn released_task(&self) -> Option { - match self { - Self::Stopped { task_id, .. } - | Self::AlreadyCompleted { task_id, .. } - | Self::EndedWithoutResult { task_id, .. } - | Self::LostWithoutResult { task_id } => Some(*task_id), - Self::Idle => None, - } - } -} - -/// Foreground ownership contract accepted for one resource background command -/// -/// Homebased proves release only for a command whose GPU ownership has a -/// witness that the authority can check after the process exits. Command names -/// alone cannot detect a shebang wrapper or a self-daemonizing executable, so -/// every other shape is refused before a task record exists -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum BackgroundCommandContract { - /// The maintained `python -m ops.run_segment run` trainer - /// - /// Its witness is the segment ownership lock under the runtime root. A - /// release needs a trainer-attempt association that the supervisor binds - /// after the running trainer holds that lock - DirectSegmentTrainer { - /// Runtime root named by `--runtime-root` - runtime_root: PathBuf, - }, -} - -/// Saved evidence that no background work holds an unregistered resource -/// -/// Each variant names the latest resource history record that cleared or never -/// created the background registration. A missing task row is never evidence -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum IdleBoundaryProof { - /// The latest loan closed with the supervisor's explicit no-resume decision - /// - /// That decision required the prior background task to be terminal with a - /// verified or operator-attested release, or no background task at all - SupervisorNoResume { - /// Closed loan that holds the decision - loan_id: LoanId, - }, - /// The latest loan closed after the supervisor resolved a return task that - /// ended with proven process-group release - RestoreEndedWithProvenRelease { - /// Closed loan that holds the resolution - loan_id: LoanId, - /// Return task that ended - task_id: TaskId, - }, - /// The latest loan closed after a native foreground or container return - /// task ended successfully with its confirmed exit witness - ForegroundReturnEnded { - /// Closed loan that holds the foreground-end evidence - loan_id: LoanId, - /// Return task that ended - task_id: TaskId, - }, - /// The latest first background launch ended before a child process started - BackgroundLaunchNeverSpawned { - /// Stable launch request identity - request_id: RequestId, - /// Launch task that recorded no child spawn - task_id: TaskId, - }, - /// An operator attested that a resource with no history started with a free GPU - /// - /// The resource had no registered task, loan, or first background launch - /// when the attestation committed, and still has none. This is a human - /// trust decision saved with its receipt, not a process or lock proof - OperatorAttestedInitialIdle { - /// Attestation whose receipt holds the observation - operation_id: operator_release::OperatorAttestationId, - }, - /// An operator attested that an ended trainer no longer holds the GPU - /// - /// The trainer was the registered trainer, a first background launch task, - /// or a Restoring loan's return task. This is a human trust decision saved - /// with its receipt, not a process or lock proof. The attestation cleared - /// the registration - OperatorAttestedGpuFree { - /// Attestation whose receipt holds the observation and evidence - operation_id: operator_release::OperatorAttestationId, - /// Trainer task that the operator resolved - task_id: TaskId, - }, -} - -/// Missing evidence that keeps an unregistered resource from serving queued work -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum IdleProofGap { - /// No loan closure or background launch records why the GPU is free - NoIdleEvidence, - /// The latest first background launch ended, but its process release is not proven - BackgroundLaunchReleaseUnproven { - /// Launch task that ended without a no-child-spawned record - task_id: TaskId, - }, - /// The latest closure or launch record does not match the resource registration - InconsistentHistory, -} - -/// Authority decision at the idle boundary of an unregistered resource -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum IdleBoundaryDecision { - /// Saved history proves that no background work holds the resource - Proven(IdleBoundaryProof), - /// The exact proof that is missing, so the resource stays reserved - Unproven(IdleProofGap), -} - -/// Stable Homebased task identity reserved for one release watcher -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] -#[serde(transparent)] -pub struct ReleaseWatcherTaskId(TaskId); - -impl ReleaseWatcherTaskId { - /// Wrap a preallocated Homebased task identity - #[must_use] - pub const fn new(task_id: TaskId) -> Self { - Self(task_id) - } - - /// Return the underlying Homebased task identity - #[must_use] - pub const fn as_task_id(self) -> TaskId { - self.0 - } -} - -/// Durable identity intent linking one watcher task to one release action -/// -/// This record reserves the exact task identity before launch. It does not -/// establish that the Homebased task is running or that the resource is free -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ReleaseWatcherIntent { - /// Stable release decision identity - pub action_id: ActionId, - /// Resource revision associated with the release action - pub state_revision: ResourceRevision, - /// Exact background task observed when release was requested - pub observed_background_task: TaskId, - /// Preallocated identity of the Homebased release watcher task - pub watcher_task_id: ReleaseWatcherTaskId, - /// Stable request identity distinct from the preallocated task identity - pub request_id: RequestId, - /// Digest of the watcher's immutable normalized command specification - pub normalized_spec_sha256: NormalizedSpecSha256, -} - -impl ReleaseWatcherIntent { - /// Check that the watcher identities are usable for the observed trainer - pub fn validate(&self) -> Result<(), InvalidReleaseWatcherIdentity> { - validate_release_watcher_identity( - self.observed_background_task, - self.request_id, - self.watcher_task_id.as_task_id(), - ) - } -} - -/// Release watcher identities that are nil or that share one UUID between two roles -#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] -#[error("release watcher identities must be non-nil and distinct from each other and the trainer")] -pub struct InvalidReleaseWatcherIdentity; - -/// Check the identities of one release watcher task before it is saved or launched -/// -/// The retry identity, the watcher task, and the observed trainer are three -/// different records, so no two of them may share a UUID -pub fn validate_release_watcher_identity( - observed_background_task: TaskId, - request_id: RequestId, - watcher_task_id: TaskId, -) -> Result<(), InvalidReleaseWatcherIdentity> { - if request_id.0.is_nil() - || watcher_task_id.0.is_nil() - || request_id.0 == watcher_task_id.0 - || watcher_task_id == observed_background_task - { - return Err(InvalidReleaseWatcherIdentity); - } - Ok(()) -} - -/// Saved form of one trainer-attempt association -/// -/// The association row and every checkpoint binding store this exact shape -/// Decoding checks only the shape; [`TrainerAttemptAssociation::try_from`] -/// checks the identities and the attempt evidence -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(crate) struct TrainerAttemptAssociationProof { - pub(crate) resource_id: ResourceId, - pub(crate) authority_machine: MachineId, - pub(crate) task_id: TaskId, - pub(crate) canonical_runtime_root: PathBuf, - pub(crate) attempt_binding: trainer_publication::AttemptBinding, - pub(crate) request_sha256: ownership_lock::TrainerRequestDigest, - pub(crate) ownership_lock_identity: OwnershipLockProof, - pub(crate) normalized_spec_sha256: NormalizedSpecSha256, -} - -impl From<&TrainerAttemptAssociation> for TrainerAttemptAssociationProof { - fn from(association: &TrainerAttemptAssociation) -> Self { - let evidence = association.verified_attempt(); - let lock = evidence.ownership_lock_identity(); - Self { - resource_id: association.resource_id(), - authority_machine: association.authority_machine(), - task_id: association.task_id(), - canonical_runtime_root: evidence.canonical_runtime_root().to_path_buf(), - attempt_binding: evidence.binding().clone(), - request_sha256: evidence.request_digest(), - ownership_lock_identity: OwnershipLockProof { - device: lock.device(), - inode: lock.inode(), - }, - normalized_spec_sha256: association.normalized_spec_sha256(), - } - } -} - -impl TryFrom for TrainerAttemptAssociation { - type Error = &'static str; - - fn try_from(proof: TrainerAttemptAssociationProof) -> Result { - let lock = proof.ownership_lock_identity; - let verified_attempt = ownership_lock::VerifiedTrainerAttempt::from_persisted( - proof.canonical_runtime_root, - proof.attempt_binding, - proof.request_sha256, - ownership_lock::OwnershipLockIdentity::new(lock.device, lock.inode), - )?; - Self::from_components( - proof.resource_id, - proof.authority_machine, - proof.task_id, - verified_attempt, - proof.normalized_spec_sha256, - ) - } -} - -/// Saved device and inode of the trainer ownership lock -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(crate) struct OwnershipLockProof { - pub(crate) device: u64, - pub(crate) inode: u64, -} - -/// Proof provenance for the trainer release that opened a serving loan -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ServingReleaseProvenance { - /// The authority verified the exact trainer's completed-result publication - CompletedTrainerResult { - /// Release action whose authority-built proof was accepted - action_id: ActionId, - /// Exact registered trainer task in the proof - task_id: TaskId, - /// SHA-256 digest of the published completed result - publication_sha256: Sha256Digest, - }, - /// The authority derived a saved idle boundary before opening the loan - /// - /// No background task was registered, and the latest resource history - /// records why no background work holds the GPU - IdleBoundary { - /// Evidence named by the loan-opening receipt - proof: IdleBoundaryProof, - }, - /// The authority verified a stopped trainer's exact checkpoint publication - StoppedTrainerCheckpoint { - /// Release action whose committed stop decision was proved - action_id: ActionId, - /// Exact registered trainer task in the proof - task_id: TaskId, - /// Generation accepted by the trainer's `--resume` contract - generation_id: String, - /// SHA-256 digest of the selected checkpoint record - record_sha256: Sha256Digest, - /// SHA-256 digest of the selected checkpoint inventory - inventory_sha256: Sha256Digest, - }, - /// The authority proved that an ended trainer released its exact saved lock - /// - /// The trainer had no usable completed result and no committed checkpoint - /// stop, so this release carries no resume evidence - EndedTrainerLockReleased { - /// Release action whose authority-built proof was accepted - action_id: ActionId, - /// Exact registered trainer task in the proof - task_id: TaskId, - /// Task-layer outcome of the ended trainer - outcome: ExitReason, - /// Request digest of the trainer-attempt association that named the lock - attempt_request_sha256: ownership_lock::TrainerRequestDigest, - }, - /// The supervisor left a return action undecided past its decision window - /// - /// The loan entered AwaitingReturn only after its own release basis was - /// proven, and nothing ran on the resource while it waited. The deadline - /// receipt of the expired action holds the loan, return context, and - /// registration that the authority saw when it served the queue - ReturnDeadlinePassed { - /// Expired return action whose deadline receipt permits activation - action_id: ActionId, - }, - /// An operator attested that the ended trainer of this release action no longer holds the GPU - /// - /// No process-group exit or trainer lock proves the release. The attestation - /// receipt holds the operator observation and the authority evidence snapshot - OperatorAttestedGpuFree { - /// Attestation whose receipt permits activation - operation_id: operator_release::OperatorAttestationId, - /// Release action that the attestation resolved - action_id: ActionId, - /// Exact registered trainer task that the operator resolved - task_id: TaskId, - }, -} - -/// Phase-specific identities and return data for every active loan phase -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum LoanPhase { - /// A watcher must release the observed background task - AwaitingRelease { - /// Stable release decision identity - action_id: ActionId, - /// Exact background task observed when release was requested - observed_background_task: TaskId, - /// Optional durable watcher task identity reserved for this action - #[serde(default, skip_serializing_if = "Option::is_none")] - watcher_intent: Option, - }, - /// One selected request is using the resource - Serving { - /// Return obligation retained for the whole interruption - return_context: ReturnContext, - /// Selected request identity - current_request_id: RequestId, - /// Durable proof that permits activation once its receipt matches - release_provenance: ServingReleaseProvenance, - }, - /// The queue is drained and the supervisor must decide what runs next - AwaitingReturn { - /// Stable return decision identity - action_id: ActionId, - /// Return obligation retained for the whole interruption - return_context: ReturnContext, - }, - /// A supervisor-bound task submission is unresolved or starting - Restoring { - /// Stable return decision identity - action_id: ActionId, - /// Return obligation retained until the task is reconciled - return_context: ReturnContext, - /// Preallocated task identity bound to the return action - resume_task_id: TaskId, - }, -} - -/// Result retained after a loan is closed -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum LoanClosure { - /// Background work resumed or a new background task started - Resumed { - /// Return evidence retained after closure - return_context: ReturnContext, - /// Task identity registered for background work - task_id: TaskId, - }, - /// Supervisor explicitly chose not to resume background work - NoResume { - /// Return evidence retained after closure - return_context: ReturnContext, - /// Supervisor's durable decision reason - reason: String, - }, - /// The release watcher confirmed that the observed task was not stopped - NotStopped { - /// Exact background task that remained active - background_task: TaskId, - /// Evidence that the watcher cannot stop the task later - reason: String, - }, - /// A return task that held the loan while it ran ended successfully with its - /// confirmed exit witness - /// - /// A native foreground task needs a confirmed process-group exit. A container - /// task needs its exited container removed and confirmed absent. The task was - /// never registered as background training, so the closure leaves the - /// resource unregistered and names the terminal evidence - ForegroundReturnEnded { - /// Return evidence retained after closure - return_context: ReturnContext, - /// Bound return task that ended - task_id: TaskId, - /// Successful task-layer outcome - outcome: ExitReason, - }, - /// The bound return task ended without a successful closure, and the supervisor accepted that end - /// - /// A direct-segment task ended before a confirmed start. A native foreground - /// task ended without success. Both needed proven process release - RestoreEnded { - /// Return evidence retained after closure - return_context: ReturnContext, - /// Bound return task that ended - task_id: TaskId, - /// Task-layer outcome with proven process release - outcome: ExitReason, - /// Supervisor's durable resolution reason - reason: String, - }, - /// The bound return task ended without release proof, and an operator - /// attested that its GPU work is gone - /// - /// A direct-segment task ended before a confirmed start, or a native - /// foreground task ended or was lost while its loan stayed reserved. The - /// receipt's binding names which. This is a human trust decision saved - /// with its receipt, not a process or lock proof. The closure cleared the - /// background registration - OperatorAttestedRestoreEnded { - /// Return evidence retained after closure - return_context: ReturnContext, - /// Bound return task that ended - task_id: TaskId, - /// Attestation whose receipt holds the observation and evidence - operation_id: operator_release::OperatorAttestationId, - }, -} - -/// Exact supervisor authority presented with one release or return action -/// -/// The deciding transaction compares every field with the saved resource, loan, -/// action, and supervisor assignment. A stale or different value cannot decide -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct SupervisorActionAuthority { - /// Resource authority that owns the loan - pub authority_machine: MachineId, - /// Resource whose loan awaits the decision - pub resource_id: ResourceId, - /// Loan that owns the return action - pub loan_id: LoanId, - /// Stable return action identity - pub action_id: ActionId, - /// Resource revision that the supervisor observed with the action - pub expected_state_revision: ResourceRevision, - /// Supervisor that makes the decision - pub supervisor: SupervisorAddress, - /// Supervisor assignment revision that the decision uses - pub assignment_revision: AssignmentRevision, -} - -/// Supervisor's explicit choice after the ready queue drained -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ReturnDecision { - /// Close the interruption without starting background work - NoResume { - /// Supervisor's durable decision reason - reason: String, - }, - /// Bind one fixed task identity as the returning background work - Launch(Box), -} - -/// Fixed task identity and typed work for one return launch -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ReturnLaunch { - /// Stable caller retry identity for the return task - pub request_id: RequestId, - /// Preallocated task identity for the return task - pub task_id: TaskId, - /// Kind of background work, checked against the saved return context - pub work: ReturnWork, -} - -/// Kind of background work that the supervisor chose -/// -/// Each kind is valid for exactly one return context. Homebased does not infer -/// the kind from a checkpoint or an optimization result -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ReturnWork { - /// Resume the stopped run from its selected checkpoint - /// - /// The authority derives the command from the saved run records, so the - /// caller cannot supply command text for this choice - SameRunResume { - /// Stopped task named by the return context - stopped_task: TaskId, - /// Recovery reference named by the return context - recovery_ref: String, - }, - /// Run evaluation or the next epoch after the previous run completed - EvaluationOrNextEpoch { - /// Completed task named by the return context - completed_task: TaskId, - /// Command chosen by the supervisor - spec: CommandSpec, - }, - /// Start new background work when no background task was registered - NewBackgroundWork { - /// Command chosen by the supervisor - spec: CommandSpec, - }, - /// Start new background work after the previous run ended or was lost without a usable result - /// - /// The ended run has no resume evidence, so the command is a new choice by - /// the supervisor and never a continuation of that run - AfterEndedRun { - /// Ended task named by the return context - ended_task: TaskId, - /// Command chosen by the supervisor - spec: CommandSpec, - }, -} - -impl ReturnWork { - /// Return the command chosen by the supervisor, or `None` for a same-run resume - /// - /// The authority derives a same-run resume command from saved run records - #[must_use] - pub const fn supervisor_spec(&self) -> Option<&CommandSpec> { - match self { - Self::SameRunResume { .. } => None, - Self::EvaluationOrNextEpoch { spec, .. } - | Self::NewBackgroundWork { spec } - | Self::AfterEndedRun { spec, .. } => Some(spec), - } - } -} - -/// How one accepted return task holds the resource, fixed when the task is accepted -/// -/// The authority derives the mode from the validated ownership contract in the -/// accepting transaction and saves it with the decision receipt. Later -/// reconciliation, restart, and retry read the saved mode; they never classify -/// the command again -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum ReturnExecutionMode { - /// Maintained direct-segment trainer - /// - /// A confirmed start closes the Restoring loan and registers the task as the - /// background trainer. Its segment ownership lock is the release witness - DirectSegmentTrainer, - /// Native foreground executable - /// - /// The task never becomes registered background training. The Restoring - /// loan keeps the resource reserved while the task runs and closes only - /// after a successful end with a confirmed process-group exit - NativeForeground, - /// Typed container workload - /// - /// Like a native foreground task, it never becomes registered background - /// training and keeps the Restoring loan while it runs. The loan closes only - /// after exit code 0 with confirmed container evidence: the exact container - /// exited, Homebased removed it, and its ID is absent - Container, -} - -impl ReturnExecutionMode { - /// Whether the return task holds the Restoring loan until it ends - #[must_use] - pub const fn holds_loan_while_running(self) -> bool { - match self { - Self::NativeForeground | Self::Container => true, - Self::DirectSegmentTrainer => false, - } - } -} - -impl From for ReturnExecutionMode { - fn from(contract: foreground::CommandOwnershipContract) -> Self { - match contract { - foreground::CommandOwnershipContract::ForegroundExecutable => Self::NativeForeground, - foreground::CommandOwnershipContract::DirectSegmentTrainer => { - Self::DirectSegmentTrainer - } - foreground::CommandOwnershipContract::Container => Self::Container, - } - } -} - -/// Why a return decision cannot apply to the saved action -#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] -pub enum ReturnDecisionRejection { - /// A decision identity is nil or reuses one UUID for two roles - #[error("return decision identities are invalid")] - InvalidIdentity, - /// A no-resume decision has no reason - #[error("no-resume reason must not be empty")] - EmptyReason, - /// Same-run resume needs the stopped context with the same task and recovery reference - #[error("same-run resume requires the matching stopped return context")] - ResumeRequiresStoppedContext, - /// Evaluation or next epoch needs the completed context with the same task - #[error("evaluation or next epoch requires the matching completed return context")] - EvaluationRequiresCompletedContext, - /// New background work needs the idle context - #[error("new background work requires the idle return context")] - NewWorkRequiresIdleContext, - /// Work after an ended run needs the ended or lost context with the same task - #[error("work after an ended run requires the matching ended return context")] - AfterEndedRunRequiresEndedContext, - /// The command callback thread is not the assigned supervisor thread - #[error("return command thread must be the assigned supervisor thread")] - ThreadMismatch, - /// The command shape can keep GPU work alive outside its task process group - #[error("return command can outlive its task process group ({risk:?})")] - UnsupportedCommandOwnership { - /// Recognized command shape that can outlive the task-run process group - risk: ResourceTaskOwnershipRisk, - }, - /// The saved records cannot prove the stopped run's immutable resume inputs - #[error("saved records cannot prove same-run resume ({gap:?})")] - ResumeUnproven { - /// First missing or changed proof element - gap: SameRunResumeGap, - }, -} - -/// Missing or changed evidence that prevents a same-run resume -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum SameRunResumeGap { - /// No verified release receipt produced this loan's stopped context - ReleaseReceiptMissing, - /// The committed checkpoint stop decision is missing or names another checkpoint - CheckpointDecisionMismatch, - /// The selected checkpoint publication is missing or changed on disk - CheckpointUnavailable, - /// The stopped run has no matching trainer-attempt association - AssociationMissing, - /// The accepted identity, task row, or spec digest of the stopped run changed - RunRecordsChanged, - /// The saved command does not have the maintained direct-segment shape - CommandShapeInvalid, - /// The saved interpreter no longer resolves from the saved environment - InterpreterChanged, -} - -impl ReturnDecision { - /// Check the typed choice against the saved return context and supervisor thread - /// - /// Storage and command-derivation checks run later in the deciding transaction - pub fn validate_for( - &self, - context: &ReturnContext, - supervisor_thread: ThreadId, - ) -> Result<(), ReturnDecisionRejection> { - let launch = match self { - Self::NoResume { reason } if reason.trim().is_empty() => { - return Err(ReturnDecisionRejection::EmptyReason); - } - Self::NoResume { .. } => return Ok(()), - Self::Launch(launch) => launch, - }; - if launch.request_id.0.is_nil() - || launch.task_id.0.is_nil() - || launch.request_id.0 == launch.task_id.0 - { - return Err(ReturnDecisionRejection::InvalidIdentity); - } - - let spec = match (&launch.work, context) { - ( - ReturnWork::SameRunResume { - stopped_task, - recovery_ref, - }, - ReturnContext::Stopped { - task_id, - recovery_ref: saved_recovery, - .. - }, - ) if stopped_task == task_id - && recovery_ref == saved_recovery - && *task_id != launch.task_id => - { - return Ok(()); - } - (ReturnWork::SameRunResume { .. }, _) => { - return Err(ReturnDecisionRejection::ResumeRequiresStoppedContext); - } - ( - ReturnWork::EvaluationOrNextEpoch { - completed_task, - spec, - }, - ReturnContext::AlreadyCompleted { task_id, .. }, - ) if completed_task == task_id && *task_id != launch.task_id => spec, - (ReturnWork::EvaluationOrNextEpoch { .. }, _) => { - return Err(ReturnDecisionRejection::EvaluationRequiresCompletedContext); - } - (ReturnWork::NewBackgroundWork { spec }, ReturnContext::Idle) => spec, - (ReturnWork::NewBackgroundWork { .. }, _) => { - return Err(ReturnDecisionRejection::NewWorkRequiresIdleContext); - } - ( - ReturnWork::AfterEndedRun { ended_task, spec }, - ReturnContext::EndedWithoutResult { task_id, .. } - | ReturnContext::LostWithoutResult { task_id }, - ) if ended_task == task_id && *task_id != launch.task_id => spec, - (ReturnWork::AfterEndedRun { .. }, _) => { - return Err(ReturnDecisionRejection::AfterEndedRunRequiresEndedContext); - } - }; - - if spec.as_normalized().thread != supervisor_thread { - return Err(ReturnDecisionRejection::ThreadMismatch); - } - foreground::CommandOwnershipContract::for_return_work(&spec.as_normalized().workload) - .map_err(|risk| ReturnDecisionRejection::UnsupportedCommandOwnership { risk })?; - - Ok(()) - } -} - -/// Lifecycle of one resource interruption and its return obligation -/// -/// Active phases are carried by [`LoanState::Active`] and shared with the -/// typed `last_safe_phase` in [`LoanState::NeedsAttention`] -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum LoanState { - /// One active phase with its required identities and return data - Active { - /// Shared phase definition for active state and attention fallback - phase: LoanPhase, - }, - /// An unresolved resource condition requires a supervisor decision - NeedsAttention { - /// Stable identity of the required attention decision - action_id: ActionId, - /// Last safe phase with all of its required identity and return data - last_safe_phase: LoanPhase, - /// Durable explanation of the unresolved condition - reason: String, - }, - /// The interruption and its return decision are resolved - Closed { - /// Resume, no-resume, or confirmed non-interruption result - result: LoanClosure, - }, -} - -impl LoanState { - /// Whether this state still waits for the exact decision that `notice` asks for - /// - /// Notice delivery state is independent of action completion, so a notice - /// with attempts left must stop once the loan moves past its action. - /// `Restoring` keeps the return action identity but already has its decision - #[must_use] - pub fn awaits_notice(&self, notice: &SupervisorNotice) -> bool { - let awaited = match (self, ¬ice.payload) { - ( - Self::Active { - phase: LoanPhase::AwaitingRelease { action_id, .. }, - }, - SupervisorNoticePayload::ReleaseRequired { .. }, - ) - | ( - Self::Active { - phase: LoanPhase::AwaitingReturn { action_id, .. }, - }, - SupervisorNoticePayload::ReturnRequired { .. }, - ) - | ( - Self::NeedsAttention { action_id, .. }, - SupervisorNoticePayload::AttentionRequired { .. }, - ) => action_id, - _ => return false, - }; - *awaited == notice.action_id - } -} - -/// One interruption spanning all optimization requests before training returns -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct Loan { - /// Stable loan identity - pub id: LoanId, - /// Exclusive resource held for this interruption - pub resource_id: ResourceId, - /// Active phase, attention fallback, or closed result - pub state: LoanState, -} - -/// Action presented to the exact assigned supervisor -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum SupervisorNoticePayload { - /// Release the exact background task observed by the resource authority - ReleaseRequired { - /// Background task that must release the resource - task_id: TaskId, - }, - /// Decide whether and how to return to background work - ReturnRequired { - /// Context that the supervisor must use for its decision - return_context: ReturnContext, - }, - /// Resolve an unsafe or uncertain loan condition - AttentionRequired { - /// Durable explanation of the condition - reason: String, - }, -} - -/// Independent delivery state for one supervisor notice -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum SupervisorNoticeDelivery { - /// Notice has not had a delivery attempt - Pending { - /// Number of attempts already reserved - attempts: u8, - }, - /// A prior attempt failed and another attempt may be reserved - RetryPending { - /// Number of attempts already reserved - attempts: u8, - /// Error from the most recent failed attempt - last_error: String, - }, - /// Delivery attempt is reserved and may be in flight - Sending { - /// Stable identity of this attempt - attempt_id: DeliveryAttemptId, - /// One-based attempt number - attempt: u8, - }, - /// Notice was delivered - Delivered { - /// Number of attempts made - attempts: u8, - }, - /// Bounded delivery attempts were exhausted - Failed { - /// Number of attempts made - attempts: u8, - /// Error from the final failed attempt - last_error: String, - }, -} - -/// Durable notice to one exact supervisor destination -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct SupervisorNotice { - /// Stable deduplication identity for this notice - pub id: NoticeId, - /// Loan that owns the required decision - pub loan_id: LoanId, - /// Stable action identity shared with loan state - pub action_id: ActionId, - /// Resource state revision associated with this notice - pub state_revision: ResourceRevision, - /// Exact machine and thread destination - pub destination: SupervisorAddress, - /// Supervisor assignment revision used for routing - pub assignment_revision: AssignmentRevision, - /// Typed action content - pub payload: SupervisorNoticePayload, - /// Delivery status, independent of action completion - pub delivery: SupervisorNoticeDelivery, -} - -/// Authority-owned result of reconciling queued work for one resource -#[derive(Debug, Clone)] -pub enum ResourceQueueReconcileOutcome { - /// No request is ready for selection - NoQueuedRequest, - /// A non-closed loan already reserves the resource - LoanAlreadyActive { - /// Existing loan that prevents a second loan from opening - loan: Loan, - }, - /// Saved idle evidence opened a loan that serves the next request in serving order - IdleServing { - /// Serving loan with an idle return context - loan: Loan, - /// Request selected by the opening - request: ResourceRequest, - /// Evidence named by the loan-opening receipt - proof: IdleBoundaryProof, - }, - /// A queued first background launch row may have lost its worker spawn - BackgroundLaunchUncertain { - /// Launch task that keeps the resource reserved - task_id: TaskId, - }, - /// A first background launch ended before registration with no release proof - /// - /// No request is queued, but the resource stays reserved until an operator - /// attests that the launch task's GPU work is gone - BackgroundLaunchReleaseUnproven { - /// Stable launch request identity - request_id: RequestId, - /// Launch task that keeps the resource reserved - task_id: TaskId, - }, - /// A running registered task now has one durable release action - ReleaseRequired { - /// Loan created for the resource interruption - loan: Loan, - /// Notice saved in the same transaction as the loan - notice: SupervisorNotice, - }, - /// Work needs owner attention because assignment or resource safety is uncertain - AttentionRequired { - /// Request that remains reserved or queued - request: ResourceRequest, - /// Authoritative reason why the resource cannot be assigned - reason: ResourceQueueAttentionReason, - }, - /// The bound return task keeps the loan reserved because its start or release is not proven - RestoreAttentionRequired { - /// Restoring loan that keeps the resource reserved - loan: Loan, - /// Return action bound to the task - action_id: ActionId, - /// Bound return task - task_id: TaskId, - /// Authority-classified reason - reason: RestoreAttentionReason, - }, - /// The exact registered trainer remains reserved because its release proof failed - ReleaseProofUnavailable { - /// Loan that remains in its existing AwaitingRelease phase - loan: Loan, - /// Exact release action that still needs proof - action_id: ActionId, - /// Exact registered trainer task that still owns the release obligation - task_id: TaskId, - /// Authority-classified reason why the proof did not pass - reason: ReleaseProofAttentionReason, - }, -} - -/// Why an authority cannot safely select a queued request -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum ResourceQueueAttentionReason { - /// No registered task exists, and the store has no proof that the GPU is idle - IdleNotProven { - /// Exact proof that is missing - gap: IdleProofGap, - }, - /// A first background launch has not reached a confirmed start - BackgroundLaunchPending { - /// Launch task that keeps the resource reserved - task_id: TaskId, - }, - /// A registered background task has no authority-owned task row - BackgroundTaskMissing { - /// Exact task registered on the resource - task_id: TaskId, - }, - /// A registered background task is not verifiably running - BackgroundTaskNotRunning { - /// Exact task registered on the resource - task_id: TaskId, - /// Durable process state observed by the authority - state: String, - }, - /// A resource task was accepted, but startup cannot prove its worker started - AcceptedTaskLaunchUncertain { - /// Exact accepted command task that needs an owner decision - task_id: TaskId, - }, - /// The saved Serving loan's release provenance does not match its saved receipts - UnverifiedServingRelease, - /// The exact trainer release proof is missing or did not pass validation - ReleaseProofUnavailable { - /// Exact release action retained by the active loan - action_id: ActionId, - /// Exact registered trainer task retained by the active loan - task_id: TaskId, - /// Authority-classified reason why the proof is not sufficient - reason: ReleaseProofAttentionReason, - }, - /// Task acceptance may have committed, but its launch result is uncertain - AssignedTaskLaunchUncertain { - /// Exact assigned command task that needs an owner decision - task_id: TaskId, - }, - /// The assigned command task is lost and cannot prove that its work stopped - AssignedTaskLost { - /// Exact assigned command task that needs an owner decision - task_id: TaskId, - }, - /// The assigned task is terminal but its exit witness is unconfirmed: a - /// process-group exit for a command, or a removed container for a container - AssignedTaskExitUnconfirmed { - /// Exact assigned command task that needs an owner decision - task_id: TaskId, - }, - /// The assigned task, request, executor, loan, or route identity does not match - AssignedTaskIdentityMismatch { - /// Exact assigned command task that needs an owner decision - task_id: TaskId, - }, - /// The assigned task could not prove that no child was spawned for its terminal result - AssignedTaskNoChildSpawnProofInvalid { - /// Exact assigned command task that needs an owner decision - task_id: TaskId, - }, - /// The assigned command may outlive its task-run process group - AssignedTaskOwnershipUncertain { - /// Exact assigned command task that needs an owner decision - task_id: TaskId, - /// Recognized command shape that can outlive its local process group - risk: ResourceTaskOwnershipRisk, - }, - /// The resource revision changed before the task completion transition committed - AssignedTaskStaleRevision { - /// Exact assigned command task that needs an owner decision - task_id: TaskId, - }, - /// The authority store could not evaluate the assigned task, so the loan stays reserved - AssignedTaskReconcileFailed { - /// Exact assigned command task that needs an owner decision - task_id: TaskId, - }, -} - -/// Why a Restoring loan cannot close or advance -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum RestoreAttentionReason { - /// The task row is queued, and this actor did not insert it, so its worker may never start - LaunchUncertain, - /// The task ended before a confirmed start; the supervisor must resolve it explicitly - EndedBeforeConfirmedStart { - /// Terminal task state - state: ProcessStatus, - }, - /// The native foreground task ended without success; the supervisor must resolve it explicitly - ForegroundEnded { - /// Terminal task state - state: ProcessStatus, - }, - /// The native foreground task ended, but its process-group exit is not confirmed - ForegroundExitUnconfirmed { - /// Terminal task state - state: ProcessStatus, - }, - /// The container task ended without success; the supervisor must resolve it explicitly - ContainerEnded { - /// Terminal task state - state: ProcessStatus, - }, - /// The container task ended, but its container evidence is not confirmed - ContainerExitUnconfirmed { - /// Terminal task state - state: ProcessStatus, - }, - /// The task is lost, so its process release is not proven - Lost, - /// The task, route, identity, or receipt does not match the bound return action - IdentityMismatch, - /// The authority store could not evaluate the restore - ReconcileFailed, -} - -/// Why a command falls outside the foreground ownership contract -/// -/// See [`foreground`] for the contract. Each variant names a shape whose GPU work -/// can outlive, or hide from, the task-run process group that release proof observes -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum ResourceTaskOwnershipRisk { - /// A remote execution client can exit while its remote command continues - RemoteShell, - /// A container client can exit while a container process continues - ContainerClient, - /// A shell wrapper can start work outside its foreground process group - ShellWrapper, - /// A command explicitly starts or manages detached work - DetachedLauncher, - /// An interpreter runs code that Homebased does not inspect - Interpreter, - /// A launcher runs another program named in its arguments - ProgramLauncher, - /// The entry point is a script whose interpreter and code Homebased does not inspect - ScriptEntryPoint, - /// The entry point is missing, unreadable, or not a recognized native executable - UninspectableEntryPoint, -} - -/// Safe, inspectable classification for a failed release proof -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum ReleaseProofAttentionReason { - /// The registered trainer has not reached a successful terminal state - TrainerNotCompleted, - /// The registered trainer is lost - TrainerLost, - /// Homebased has not confirmed that the trainer worker exited - WorkerExitUnconfirmed, - /// No exact completed-result publication is available - CompletedResultUnavailable, - /// No exact stopped-checkpoint publication is available - StoppedCheckpointUnavailable, - /// The saved ownership lock is held or cannot be verified - OwnershipLockUnverified, - /// The registered trainer has no trainer-attempt association, so no saved - /// lock can prove its release; only a separate operator resolution applies - TrainerAssociationMissing, - /// Saved task, authority, route, or proof identities do not match - SavedEvidenceMismatch, -} - -/// Strict Fleet request for one exact supervisor-notice delivery attempt -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct SupervisorNoticeRequest { - /// Public API schema version - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Machine that sent this delivery attempt - pub source_machine: MachineId, - /// Exact destination machine and Codex thread - pub destination: SupervisorAddress, - /// Stable identity of the notice - pub notice_id: NoticeId, - /// Loan that owns the required decision - pub loan_id: LoanId, - /// Stable identity of the required supervisor action - pub action_id: ActionId, - /// Resource state revision associated with this notice - pub state_revision: ResourceRevision, - /// Supervisor assignment revision used for routing - pub assignment_revision: AssignmentRevision, - /// Stable identity of this delivery attempt - pub attempt_id: DeliveryAttemptId, - /// Typed action content without mutable delivery state - pub payload: SupervisorNoticePayload, -} - -impl SupervisorNoticeRequest { - /// Validate the strict versioned route and all stable identities - pub fn validate(&self) -> Result<(), AppError> { - if self.api_version != API_VERSION { - return Err(AppError::Usage { - message: "unsupported API version".into(), - }); - } - - if self.source_machine.as_uuid().is_nil() - || self.destination.machine.as_uuid().is_nil() - || self.destination.thread.0.is_nil() - { - return Err(AppError::MessageInvalid { - message: "supervisor notice identities must not be nil".into(), - }); - } - - let payload_task = match &self.payload { - SupervisorNoticePayload::ReleaseRequired { task_id } => Some(*task_id), - SupervisorNoticePayload::ReturnRequired { return_context } => { - return_context.released_task() - } - SupervisorNoticePayload::AttentionRequired { .. } => None, - }; - if payload_task.is_some_and(|task_id| task_id.0.is_nil()) { - return Err(AppError::MessageInvalid { - message: "supervisor notice task identity must not be nil".into(), - }); - } - - Ok(()) - } -} - -/// Durable receipt returned after the exact notice attempt reaches Codex -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct SupervisorNoticeReceipt { - /// Public API schema version - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Stable identity of the delivered notice - pub notice_id: NoticeId, - /// Stable identity of the delivered attempt - pub attempt_id: DeliveryAttemptId, - /// Machine that accepted the queue command - pub destination_machine: MachineId, - /// Exact thread that received the queue command - pub destination_thread: ThreadId, - /// Time when the receiver committed this receipt - pub delivered_at: chrono::DateTime, -} - -/// Versioned response from a remote supervisor-notice receiver -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct SupervisorNoticeResponse { - /// Public API schema version - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Receiver machine identity - pub destination_machine: MachineId, - /// Receiver's durable success receipt - pub receipt: SupervisorNoticeReceipt, -} - -#[cfg(test)] -mod tests { - use std::path::PathBuf; - use std::time::Duration; - - use serde_json::json; - use uuid::Uuid; - - use crate::domain::{API_VERSION, AgentKind, TaskId, TaskName, ThreadId}; - use crate::invocation::CommandLine; - use crate::machine::{MachineId, MachineName}; - use crate::spec::{ - NormalizedAgentWorkload, NormalizedSpec, NormalizedTaskWorkload, NormalizedWorkload, - }; - use crate::submission::{RequestId, ResourceQueueOutcome, ResourceQueueReceipt}; - - use super::{ - AcceptanceSequence, ActionId, AssignmentRevision, CommandSpec, CommandSpecError, - DeliveryAttemptId, LoanClosure, LoanId, LoanPhase, LoanState, NoticeId, ResourceId, - ResourceQueueRequest, ResourceQueueResponse, ResourceRequest, ResourceRequestState, - ResourceRevision, ReturnContext, ReturnDecision, ReturnDecisionRejection, ReturnLaunch, - ReturnWork, SupervisorAddress, SupervisorNoticePayload, SupervisorNoticeRequest, - }; - use crate::domain::ExitReason; - - fn task_spec() -> NormalizedSpec { - NormalizedSpec { - api_version: API_VERSION, - thread: ThreadId(Uuid::now_v7()), - name: TaskName::parse("resource command").unwrap(), - cwd: PathBuf::from("/tmp"), - machine: None, - timeout: Duration::from_secs(1800), - workload: NormalizedWorkload::Task(NormalizedTaskWorkload { - command: CommandLine::try_from_argv(vec!["echo".into(), "gpu".into()]).unwrap(), - }), - } - } - - fn agent_spec() -> NormalizedSpec { - NormalizedSpec { - api_version: API_VERSION, - thread: ThreadId(Uuid::now_v7()), - name: TaskName::parse("resource agent").unwrap(), - cwd: PathBuf::from("/tmp"), - machine: None, - timeout: Duration::from_secs(1800), - workload: NormalizedWorkload::Agent(NormalizedAgentWorkload { - agent: AgentKind::Codex, - model: None, - prompt: "do work".into(), - extra_args: Vec::new(), - report_trailer: true, - resume_thread: None, - }), - } - } - - fn request(spec: NormalizedSpec) -> Result { - ResourceRequest::new( - RequestId::new(), - TaskId::new(), - ResourceId::new(), - AcceptanceSequence::new(1), - MachineId::new(), - spec, - ) - } - - fn queue_request(spec: NormalizedSpec) -> Result { - Ok(ResourceQueueRequest::new( - 1, - MachineId::new(), - MachineId::new(), - RequestId::new(), - TaskId::new(), - ResourceId::new(), - CommandSpec::try_from(spec)?, - )) - } - - fn queue_receipt() -> ResourceQueueReceipt { - ResourceQueueReceipt { - request: RequestId::new(), - task: TaskId::new(), - origin_machine: MachineId::new(), - authority_machine: MachineId::new(), - resource: ResourceId::new(), - outcome: ResourceQueueOutcome::Waiting, - } - } - - #[test] - fn request_construction_only_accepts_command_workloads() { - let queued = request(task_spec()).unwrap(); - assert!(matches!(queued.state, ResourceRequestState::Queued)); - assert!(matches!( - request(agent_spec()), - Err(CommandSpecError::AgentWorkload) - )); - } - - #[test] - fn command_spec_rejects_an_explicit_machine() { - let mut spec = task_spec(); - spec.machine = Some(MachineName::parse("other").unwrap()); - - assert!(matches!( - CommandSpec::try_from(spec), - Err(CommandSpecError::ExplicitMachine) - )); - } - - #[test] - fn request_deserialization_rejects_agent_workloads() { - let mut value = serde_json::to_value(request(task_spec()).unwrap()).unwrap(); - value["spec"] = serde_json::to_value(agent_spec()).unwrap(); - - assert!(serde_json::from_value::(value).is_err()); - } - - #[test] - fn request_deserialization_rejects_an_explicit_machine() { - let mut value = serde_json::to_value(request(task_spec()).unwrap()).unwrap(); - value["spec"]["machine"] = json!("other"); - - assert!(serde_json::from_value::(value).is_err()); - } - - #[test] - fn resource_queue_request_round_trips_command_without_explicit_machine() { - let request = queue_request(task_spec()).unwrap(); - let encoded = serde_json::to_value(&request).unwrap(); - - assert_eq!(encoded["api_version"], API_VERSION); - assert_eq!(encoded["spec"]["workload"]["type"], "task"); - assert!(encoded["spec"].get("machine").is_none()); - - let decoded = serde_json::from_value::(encoded.clone()).unwrap(); - assert_eq!(decoded.request_id, request.request_id); - assert_eq!(decoded.task_id, request.task_id); - assert_eq!(decoded.resource_id, request.resource_id); - assert!(decoded.spec.as_normalized().machine.is_none()); - assert!(matches!( - &decoded.spec.as_normalized().workload, - NormalizedWorkload::Task(_) - )); - assert_eq!(serde_json::to_value(decoded).unwrap(), encoded); - } - - #[test] - fn resource_queue_request_rejects_agents_and_explicit_machines() { - assert!(matches!( - queue_request(agent_spec()), - Err(CommandSpecError::AgentWorkload) - )); - - let mut explicit_machine = task_spec(); - explicit_machine.machine = Some(MachineName::parse("other").unwrap()); - assert!(matches!( - queue_request(explicit_machine), - Err(CommandSpecError::ExplicitMachine) - )); - - let valid = serde_json::to_value(queue_request(task_spec()).unwrap()).unwrap(); - let mut agent = valid.clone(); - agent["spec"] = serde_json::to_value(agent_spec()).unwrap(); - assert!(serde_json::from_value::(agent).is_err()); - - let mut explicit_machine = valid; - explicit_machine["spec"]["machine"] = json!("other"); - assert!(serde_json::from_value::(explicit_machine).is_err()); - } - - #[test] - fn resource_queue_response_binds_destination_to_receipt_authority() { - let receipt = queue_receipt(); - let response = ResourceQueueResponse::new(1, receipt.clone()); - let encoded = serde_json::to_value(&response).unwrap(); - - assert_eq!(response.destination_machine, receipt.authority_machine); - assert_eq!(encoded["api_version"], API_VERSION); - assert_eq!( - serde_json::from_value::(encoded.clone()).unwrap(), - response - ); - } - - #[test] - fn resource_queue_request_and_response_reject_unknown_fields() { - let mut request = serde_json::to_value(queue_request(task_spec()).unwrap()).unwrap(); - request["unexpected"] = json!(true); - assert!(serde_json::from_value::(request).is_err()); - - let mut response = - serde_json::to_value(ResourceQueueResponse::new(1, queue_receipt())).unwrap(); - response["unexpected"] = json!(true); - assert!(serde_json::from_value::(response).is_err()); - } - - #[test] - fn stopped_return_context_preserves_task_and_opaque_references() { - let context = ReturnContext::Stopped { - task_id: TaskId::new(), - checkpoint_ref: "trainer://generation/204".into(), - recovery_ref: "run-config:immutable-77".into(), - }; - let encoded = serde_json::to_value(&context).unwrap(); - - assert_eq!(encoded["type"], "stopped"); - assert_eq!( - serde_json::from_value::(encoded).unwrap(), - context - ); - } - - #[test] - fn loan_attention_retains_the_last_safe_phase_data() { - let state = LoanState::NeedsAttention { - action_id: ActionId::new(), - last_safe_phase: LoanPhase::AwaitingRelease { - action_id: ActionId::new(), - observed_background_task: TaskId::new(), - watcher_intent: None, - }, - reason: "watcher could not establish process exit".into(), - }; - let encoded = serde_json::to_value(&state).unwrap(); - - assert_eq!(encoded["type"], "needs_attention"); - assert_eq!(encoded["last_safe_phase"]["type"], "awaiting_release"); - assert_eq!(serde_json::from_value::(encoded).unwrap(), state); - } - - #[test] - fn awaiting_release_without_watcher_intent_remains_compatible() { - let action_id = ActionId::new(); - let background_task = TaskId::new(); - let legacy = json!({ - "type": "active", - "phase": { - "type": "awaiting_release", - "action_id": action_id, - "observed_background_task": background_task, - }, - }); - - assert!(matches!( - serde_json::from_value::(legacy).unwrap(), - LoanState::Active { - phase: LoanPhase::AwaitingRelease { - action_id: saved_action, - observed_background_task: saved_task, - watcher_intent: None, - } - } if saved_action == action_id && saved_task == background_task - )); - } - - #[test] - fn loan_state_deserialization_requires_active_phase_data() { - let incomplete = json!({ - "type": "active", - "phase": { - "type": "awaiting_release", - "action_id": ActionId::new(), - }, - }); - - assert!(serde_json::from_value::(incomplete).is_err()); - } - - #[test] - fn loan_attention_rejects_attention_or_closed_last_safe_phase() { - for phase_type in ["needs_attention", "closed"] { - let invalid = json!({ - "type": "needs_attention", - "action_id": ActionId::new(), - "last_safe_phase": { "type": phase_type }, - "reason": "unresolved condition", - }); - - assert!(serde_json::from_value::(invalid).is_err()); - } - } - - #[test] - fn return_decisions_are_strict_and_older_closures_still_decode() { - let launch = ReturnDecision::Launch(Box::new(ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::NewBackgroundWork { - spec: CommandSpec::try_from(task_spec()).unwrap(), - }, - })); - let mut encoded = serde_json::to_value(&launch).unwrap(); - assert_eq!(encoded["type"], "launch"); - assert_eq!(encoded["work"]["type"], "new_background_work"); - serde_json::from_value::(encoded.clone()).unwrap(); - encoded["work"]["command"] = json!(["/bin/sh", "-c", "resume"]); - assert!(serde_json::from_value::(encoded).is_err()); - - let legacy = json!({ - "type": "closed", - "result": { - "type": "no_resume", - "return_context": { "type": "idle" }, - "reason": "legacy", - }, - }); - assert!(matches!( - serde_json::from_value::(legacy).unwrap(), - LoanState::Closed { - result: LoanClosure::NoResume { .. } - } - )); - let ended = LoanState::Closed { - result: LoanClosure::RestoreEnded { - return_context: ReturnContext::Idle, - task_id: TaskId::new(), - outcome: ExitReason::Cancelled, - reason: "never started".into(), - }, - }; - assert_eq!( - serde_json::from_value::(serde_json::to_value(&ended).unwrap()).unwrap(), - ended - ); - } - - #[test] - fn return_decision_choice_must_match_its_context_and_supervisor_thread() { - let spec = task_spec(); - let thread = spec.thread; - let new_work = |spec: NormalizedSpec| { - ReturnDecision::Launch(Box::new(ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::NewBackgroundWork { - spec: CommandSpec::try_from(spec).unwrap(), - }, - })) - }; - - assert_eq!( - new_work(spec.clone()).validate_for(&ReturnContext::Idle, thread), - Ok(()) - ); - assert_eq!( - new_work(spec.clone()).validate_for( - &ReturnContext::AlreadyCompleted { - task_id: TaskId::new(), - result_ref: "result".into(), - }, - thread - ), - Err(ReturnDecisionRejection::NewWorkRequiresIdleContext) - ); - assert_eq!( - new_work(spec.clone()).validate_for(&ReturnContext::Idle, ThreadId(Uuid::now_v7())), - Err(ReturnDecisionRejection::ThreadMismatch) - ); - let mut detached = spec; - detached.workload = NormalizedWorkload::Task(NormalizedTaskWorkload { - command: CommandLine::try_from_argv(vec!["setsid".into(), "trainer".into()]).unwrap(), - }); - assert_eq!( - new_work(detached).validate_for(&ReturnContext::Idle, thread), - Err(ReturnDecisionRejection::UnsupportedCommandOwnership { - risk: super::ResourceTaskOwnershipRisk::DetachedLauncher, - }) - ); - assert_eq!( - ReturnDecision::NoResume { reason: " ".into() } - .validate_for(&ReturnContext::Idle, thread), - Err(ReturnDecisionRejection::EmptyReason) - ); - } - - #[test] - fn ended_context_permits_new_work_or_no_resume_but_never_same_run_resume() { - let spec = task_spec(); - let thread = spec.thread; - let ended_task = TaskId::new(); - let ended = ReturnContext::EndedWithoutResult { - task_id: ended_task, - outcome: ExitReason::Exit { code: 3 }, - }; - let launch = |work| { - ReturnDecision::Launch(Box::new(ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work, - })) - }; - let after_ended = |ended_task| ReturnWork::AfterEndedRun { - ended_task, - spec: CommandSpec::try_from(spec.clone()).unwrap(), - }; - - assert_eq!( - launch(after_ended(ended_task)).validate_for(&ended, thread), - Ok(()) - ); - assert_eq!( - ReturnDecision::NoResume { - reason: "failed run".into() - } - .validate_for(&ended, thread), - Ok(()) - ); - assert_eq!( - launch(ReturnWork::SameRunResume { - stopped_task: ended_task, - recovery_ref: "generation-after-failure".into(), - }) - .validate_for(&ended, thread), - Err(ReturnDecisionRejection::ResumeRequiresStoppedContext) - ); - assert_eq!( - launch(after_ended(TaskId::new())).validate_for(&ended, thread), - Err(ReturnDecisionRejection::AfterEndedRunRequiresEndedContext) - ); - assert_eq!( - launch(after_ended(ended_task)).validate_for(&ReturnContext::Idle, thread), - Err(ReturnDecisionRejection::AfterEndedRunRequiresEndedContext) - ); - // a lost run released by an operator attestation has the same choices - let lost = ReturnContext::LostWithoutResult { - task_id: ended_task, - }; - assert_eq!( - launch(after_ended(ended_task)).validate_for(&lost, thread), - Ok(()) - ); - assert_eq!( - launch(after_ended(TaskId::new())).validate_for(&lost, thread), - Err(ReturnDecisionRejection::AfterEndedRunRequiresEndedContext) - ); - assert_eq!( - launch(ReturnWork::SameRunResume { - stopped_task: ended_task, - recovery_ref: "generation-before-loss".into(), - }) - .validate_for(&lost, thread), - Err(ReturnDecisionRejection::ResumeRequiresStoppedContext) - ); - assert_eq!( - launch(ReturnWork::NewBackgroundWork { - spec: CommandSpec::try_from(spec.clone()).unwrap(), - }) - .validate_for(&ended, thread), - Err(ReturnDecisionRejection::NewWorkRequiresIdleContext) - ); - } - - fn supervisor_notice_request() -> SupervisorNoticeRequest { - SupervisorNoticeRequest { - api_version: API_VERSION, - protocol_version: 1, - source_machine: MachineId::new(), - destination: SupervisorAddress { - machine: MachineId::new(), - thread: ThreadId(Uuid::now_v7()), - }, - notice_id: NoticeId::new(), - loan_id: LoanId::new(), - action_id: ActionId::new(), - state_revision: ResourceRevision::new(4), - assignment_revision: AssignmentRevision::new(2), - attempt_id: DeliveryAttemptId::new(), - payload: SupervisorNoticePayload::ReleaseRequired { - task_id: TaskId::new(), - }, - } - } - - #[test] - fn supervisor_notice_request_is_strict_and_excludes_delivery_state() { - let request = supervisor_notice_request(); - request.validate().unwrap(); - let value = serde_json::to_value(&request).unwrap(); - - assert!(value.get("delivery").is_none()); - assert!(value.get("attempt_id").is_some()); - let mut unknown = value; - unknown["unexpected"] = json!(true); - assert!(serde_json::from_value::(unknown).is_err()); - } - - #[test] - fn supervisor_notice_request_rejects_nil_identity_and_unsupported_version() { - let mut wire = serde_json::to_value(supervisor_notice_request()).unwrap(); - wire["attempt_id"] = json!(Uuid::nil()); - assert!(serde_json::from_value::(wire).is_err()); - - let mut request = supervisor_notice_request(); - request.destination.thread = ThreadId(Uuid::nil()); - assert!(request.validate().is_err()); - - let mut request = supervisor_notice_request(); - request.api_version += 1; - assert!(request.validate().is_err()); - } -} diff --git a/src/resource/api.rs b/src/resource/api.rs deleted file mode 100644 index 58adcbb..0000000 --- a/src/resource/api.rs +++ /dev/null @@ -1,629 +0,0 @@ -//! Versioned read and control bodies for resource routes on the socket, dashboard, and Fleet -//! -//! The resource authority builds every view from durable resource, loan, -//! request, notice, and task state. No view stores a second occupancy state - -use crate::resource::trainer_publication::AttemptBinding; -use chrono::{DateTime, Utc}; -use serde::{Deserialize, Serialize}; -use uuid::Uuid; - -use crate::domain::{CallbackStatus, ExitReason, ProcessStatus, TaskId, ThreadId}; -use crate::machine::MachineId; -use crate::resource::operator_release::{OperatorGpuFreeAttestation, OperatorGpuFreeReceipt}; -use crate::resource::{ - AcceptanceSequence, ActionId, AssignmentRevision, Loan, LoanId, LoanPhase, NoticeId, Resource, - ResourceId, ResourceRequestState, ResourceRevision, ReturnContext, ReturnDecisionWindow, - ReturnExecutionMode, SupervisorAddress, SupervisorNotice, -}; -use crate::submission::RequestId; - -/// Socket route that registers a resource on this daemon as its fixed authority -pub const RESOURCE_REGISTER_PATH: &str = "/v1/resources/register"; -/// Socket and dashboard route that lists pending supervisor actions for one thread -pub const RESOURCE_PENDING_PATH: &str = "/v1/resources/pending"; -/// Fleet route that lists resources owned by the receiving authority -pub const CLUSTER_RESOURCES_PATH: &str = "/v1/cluster/resources"; -/// Fleet route that lists pending actions owned by the receiving authority -pub const CLUSTER_RESOURCE_PENDING_PATH: &str = "/v1/cluster/resources/pending"; - -/// `GET /v1/resources` -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceList { - /// Public API version - pub api_version: u32, - /// Resources read from reachable authorities - pub resources: Vec, - /// Authorities whose resources could not be read; their GPUs are unknown, not idle - pub unavailable_authorities: Vec, -} - -/// Summary of one resource derived from its durable state -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceOverview { - /// Resource with its fixed authority, supervisor, and revisions - pub resource: Resource, - /// Non-closed loan that reserves the resource, if one exists - pub loan: Option, - /// Requests still waiting for selection in serving order - pub queued_count: u64, - /// Task that holds the resource for the current loan phase - pub current_task: Option, - /// Most important condition that needs an operator or supervisor - pub attention: Option, - /// First background launch that reserves the unregistered resource, if one does - #[serde(default, skip_serializing_if = "Option::is_none")] - pub background_launch: Option, - /// Decision window of the pending return action, when the loan awaits one - #[serde(default, skip_serializing_if = "Option::is_none")] - pub return_window: Option, -} - -/// First background launch that reserves a resource before its task is registered -/// -/// The authority derives it from the launch receipt and the task row. It clears -/// only when the task registers on its confirmed start, when saved evidence or -/// an operator attestation releases it, or when a later loan supersedes it -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct BackgroundLaunchReservation { - /// Stable launch request identity, used by an operator attestation binding - pub request_id: RequestId, - /// Launch task that keeps the resource reserved - pub task_id: TaskId, - /// Why the launch still reserves the resource - pub status: BackgroundLaunchReservationStatus, -} - -/// Why a first background launch still reserves its resource -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum BackgroundLaunchReservationStatus { - /// The task row is queued, and its worker may still be starting - Queued, - /// The task started and waits for the resource owner to register it - StartedUnregistered, - /// The task ended before registration, and no proof shows that its GPU work stopped - /// - /// Only an operator attestation after inspecting the authority GPU releases it - ReleaseUnproven, - /// The launch records do not match the launch receipt - IdentityMismatch, -} - -/// One authority that could not answer a resource read -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct UnavailableAuthority { - /// Machine that did not answer - pub machine: MachineId, - /// Last known machine name, when discovery has one - pub name: Option, - /// Why the read failed - pub message: String, -} - -/// `GET /v1/resources/{id}` and the result of a settled resource mutation -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceDetail { - /// Public API version - pub api_version: u32, - /// Resource with its fixed authority, supervisor, and revisions - pub resource: Resource, - /// Non-closed loan that reserves the resource, if one exists - pub loan: Option, - /// Every request for this resource in serving order - pub requests: Vec, - /// Supervisor notices for the non-closed loan with their delivery state - pub notices: Vec, - /// Task that holds the resource for the current loan phase - pub current_task: Option, - /// Registered background task - pub background_task: Option, - /// Most important condition that needs an operator or supervisor - pub attention: Option, - /// Identity of `current_task`, present even when its task row is unavailable - pub current_task_id: Option, - /// Identity of `background_task`, present even when its task row is unavailable - pub background_task_id: Option, - /// First background launch that reserves the unregistered resource, if one does - #[serde(default, skip_serializing_if = "Option::is_none")] - pub background_launch: Option, - /// Accepted execution mode of the exact current Restoring loan, when proven - #[serde(default, skip_serializing_if = "Option::is_none")] - pub return_execution_mode: Option, - /// Decision window of the pending return action, when the loan awaits one - #[serde(default, skip_serializing_if = "Option::is_none")] - pub return_window: Option, -} - -/// One queued, assigned, or retained request without its private command content -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceRequestView { - /// Caller retry identity - pub request_id: RequestId, - /// Preallocated task identity - pub task_id: TaskId, - /// Immutable authority-assigned acceptance identity - pub acceptance_sequence: AcceptanceSequence, - /// Machine that owns the requesting thread and callback route - pub origin_machine: MachineId, - /// Requesting thread, which runs on `origin_machine` - #[serde(default, skip_serializing_if = "Option::is_none")] - pub thread: Option, - /// Submitted task name - pub display_name: String, - /// Queue state, separate from task process state - pub state: ResourceRequestState, -} - -/// Task fields a resource page needs, with the same names as the task list view -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceTaskSummary { - /// Task identity - pub id: TaskId, - /// Non-empty server-derived label - pub display_name: String, - /// Process status - pub status: ProcessStatus, - /// Submitting thread, which runs on `origin_machine`, or on the executor - /// when `origin_machine` is unknown - #[serde(default, skip_serializing_if = "Option::is_none")] - pub thread: Option, - /// Machine that owns callbacks, when known - pub origin_machine: Option, - /// Machine that runs the task, when known - pub execution_machine: Option, - /// Worker pid while running - pub pid: Option, - /// Terminal callback delivery state - pub callback: CallbackStatus, - /// Why the process ended, once known - pub exit_reason: Option, - /// When cancellation was requested - pub cancel_requested_at: Option>, - /// Insert time - pub created_at: DateTime, - /// Last row update - pub updated_at: DateTime, -} - -/// Typed reason a resource needs attention -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum AttentionCode { - /// The loan is in its durable attention state - LoanNeedsAttention, - /// The authority cannot safely select or advance a queued request - QueueBlocked, - /// The bound return task cannot close or advance the loan - RestoreBlocked, - /// The exact trainer release proof is missing or failed validation - ReleaseProofUnavailable, - /// A first background launch ended before registration with no release proof - /// - /// The resource stays reserved until an operator attests that its GPU work is gone - BackgroundLaunchReleaseUnproven, - /// The bound release watcher cannot proceed - ReleaseWatcherBlocked, - /// A supervisor notice used its automatic delivery attempts - NoticeDeliveryFailed, -} - -/// Condition that needs an operator or supervisor -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct AttentionView { - /// Typed reason - pub code: AttentionCode, - /// Human-readable explanation - pub message: String, - /// Action that owns the condition, when one exists - pub action_id: Option, - /// Task that the condition names, when one exists - pub task_id: Option, - /// Notice that the condition names, when one exists - pub notice_id: Option, -} - -/// `GET /v1/resources/pending?machine=&thread=` -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct PendingActionList { - /// Public API version - pub api_version: u32, - /// Open supervisor actions for the exact thread - pub actions: Vec, - /// Authorities that could not answer; their actions are unknown, not absent - pub unavailable_authorities: Vec, -} - -/// One open supervisor action read from durable loan and notice state -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct PendingActionView { - /// Resource that the loan reserves - pub resource_id: ResourceId, - /// Fixed resource authority - pub authority_machine: MachineId, - /// Loan that owns the action - pub loan_id: LoanId, - /// Stable action identity - pub action_id: ActionId, - /// Resource revision a decision must present - pub state_revision: ResourceRevision, - /// Exact assigned supervisor - pub supervisor: SupervisorAddress, - /// Supervisor assignment revision a decision must present - pub assignment_revision: AssignmentRevision, - /// Required decision - pub phase: PendingActionPhase, - /// Return evidence retained by the loan, when the phase has one - pub return_context: Option, - /// Durable notice for the action, when one exists - pub notice: Option, - /// Decision window of the pending return action, when the loan awaits one - #[serde(default, skip_serializing_if = "Option::is_none")] - pub return_window: Option, -} - -/// Decision that one open action requires -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum PendingActionPhase { - /// Release the exact observed background task - ReleaseRequired { - /// Background task that must release the resource - observed_background_task: TaskId, - }, - /// Choose same-task resume, new background work, or no resume - ReturnRequired, - /// A bound return task has not closed the loan yet - Restoring { - /// Fixed return task identity - resume_task_id: TaskId, - }, - /// Resolve an unsafe or uncertain loan condition - AttentionRequired { - /// Durable explanation of the condition - reason: String, - /// Last safe phase with its identities and return data - last_safe_phase: Box, - }, -} - -/// Fields that register one resource on the receiving daemon -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceRegistration { - /// Caller-allocated stable resource identity - pub id: ResourceId, - /// Human-readable resource name - pub display_name: String, - /// Exact supervisor thread for release and return notices - pub supervisor: SupervisorAddress, -} - -/// `POST /v1/resources/register` -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceRegisterBody { - /// Public API version - pub api_version: u32, - /// Registration fields - pub spec: ResourceRegistration, -} - -/// `POST /v1/resources/{id}/supervisor` -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct SupervisorReplacementBody { - /// Public API version - pub api_version: u32, - /// Resource revision the caller observed - pub expected_revision: ResourceRevision, - /// Exact new supervisor thread - pub supervisor: SupervisorAddress, -} - -/// `POST /v1/resources/{id}/requests/{request_id}/cancel` -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct RequestCancelBody { - /// Public API version - pub api_version: u32, - /// Stable retry identity for this cancellation - pub operation_id: Uuid, - /// Resource revision the caller observed - pub expected_revision: ResourceRevision, -} - -/// `POST /v1/resources/{id}/actions` -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceActionBody { - /// Public API version - pub api_version: u32, - /// Resource revision the caller observed - pub expected_revision: ResourceRevision, - /// Stable retry identity; a changed body for the same identity is a conflict - pub operation_id: Uuid, - /// Requested control - pub action: BrowserResourceAction, -} - -/// One place for a moved queued request, relative to other queued requests -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum QueuePlacement { - /// Place the request first in the queue - Front, - /// Place the request last in the queue - Back, - /// Place the request directly before another queued request - Before { - /// Anchor request identity - request_id: RequestId, - }, - /// Place the request directly after another queued request - After { - /// Anchor request identity - request_id: RequestId, - }, -} - -/// The resource controls exposed to operators and the dashboard -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum BrowserResourceAction { - /// Remove one queued request through its origin-owned cancellation - CancelQueued { - /// Queued request identity - request_id: RequestId, - }, - /// Move one queued request to a new place in the serving order - MoveQueued { - /// Queued request identity - request_id: RequestId, - /// New place in the queue - placement: QueuePlacement, - }, - /// Cancel the active command task through its origin-owned cancellation - StopActive { - /// Exact active task identity - task_id: TaskId, - }, - /// Start one explicit delivery attempt for a notice whose automatic attempts failed - Renotify { - /// Notice identity - notice_id: NoticeId, - }, -} - -/// Response to `POST /v1/resources/{id}/requests` -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceRequestSubmitResponse { - /// Public API version - pub api_version: u32, - /// Caller retry identity - pub request_id: RequestId, - /// Preallocated task identity - pub task_id: TaskId, - /// Resource queue that owns the request - pub resource_id: ResourceId, - /// Fixed resource authority - pub authority_machine: MachineId, - /// Saved authority result - pub outcome: ResourceRequestSubmitOutcome, -} - -/// Response to `POST /v1/resources/{id}/background` -/// -/// The launch binds a task but does not register it; the resource owner -/// registers the task only after the task layer records a confirmed start -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceBackgroundSubmitResponse { - /// Public API version - pub api_version: u32, - /// Caller retry identity - pub request_id: RequestId, - /// Task identity bound to the request - pub task_id: TaskId, - /// Authoritative resource after the launch binding - pub resource: Resource, - /// Saved authority result - pub outcome: ResourceBackgroundSubmitOutcome, -} - -/// Definitive authority result for one first background launch -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceBackgroundSubmitOutcome { - /// This request bound the task and started its one worker spawn - Inserted, - /// An exact earlier launch exists; its task was observed and not respawned - Existing { - /// State retained by the task layer - state: ProcessStatus, - }, -} - -/// `POST /v1/resources/{id}/background/{task_id}/trainer-attempt` -/// -/// The supervisor names the trainer attempt after the registered trainer holds -/// its ownership lock. The authority reads the attempt request and probes the -/// lock itself; the body carries no evidence -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct TrainerAttemptBody { - /// Public API version - pub api_version: u32, - /// Trainer identity of the running attempt - pub attempt_binding: AttemptBinding, -} - -/// Response to `POST /v1/resources/{id}/background/{task_id}/trainer-attempt` -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct TrainerAttemptResponse { - /// Public API version - pub api_version: u32, - /// Authoritative resource after the association - pub resource: Resource, - /// Registered task bound to the attempt - pub task_id: TaskId, - /// Canonical runtime root whose lock the trainer held - pub runtime_root: std::path::PathBuf, - /// Trainer identity of the associated attempt - pub attempt_binding: AttemptBinding, -} - -/// `POST /v1/resources/{id}/initial-idle` -/// -/// The attestation is a human confirmation after inspecting the authority GPU -/// It is not an automatic proof that GPU work stopped -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct InitialIdleBody { - /// Public API version - pub api_version: u32, - /// Complete, immutable initial idle attestation - pub attestation: crate::resource::initial_idle::InitialIdleAttestation, -} - -/// Saved receipt from `POST /v1/resources/{id}/initial-idle` -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct InitialIdleResponse { - /// Public API version - pub api_version: u32, - /// Durable receipt saved by the authority - pub receipt: crate::resource::initial_idle::InitialIdleReceipt, - /// Whether the authority returned an exact earlier receipt - pub replayed: bool, -} - -/// `POST /v1/resources/{id}/operator-release` -/// -/// The attestation is a human confirmation after inspecting the authority GPU -/// It is not an automatic proof that GPU work stopped -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct OperatorReleaseBody { - /// Public API version - pub api_version: u32, - /// Complete, immutable operator attestation - pub attestation: OperatorGpuFreeAttestation, -} - -/// Saved receipt from `POST /v1/resources/{id}/operator-release` -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct OperatorReleaseResponse { - /// Public API version - pub api_version: u32, - /// Durable receipt saved by the authority - pub receipt: OperatorGpuFreeReceipt, - /// Whether the authority returned an exact earlier receipt - pub replayed: bool, -} - -/// Definitive authority result for one request submission -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceRequestSubmitOutcome { - /// The request waits in serving order or runs under a loan - Waiting, - /// The authority activated the command task - Activated, - /// The authority rejected the request before activation - Rejected { - /// Durable rejection reason - reason: String, - }, - /// The request was cancelled before activation - CancelledBeforeLaunch, -} - -/// Fleet read of one authority's resources -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ClusterResourceList { - /// Public API version - pub api_version: u32, - /// Authority that answered - pub machine: MachineId, - /// Resources owned by that authority - pub resources: Vec, -} - -/// Fleet read of one resource detail, absent when the receiver is not its authority -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ClusterResourceDetail { - /// Public API version - pub api_version: u32, - /// Authority that answered - pub machine: MachineId, - /// Detail, when the receiver owns the resource - pub detail: Option, -} - -/// Fleet read of pending actions owned by one authority -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ClusterPendingActions { - /// Public API version - pub api_version: u32, - /// Authority that answered - pub machine: MachineId, - /// Open actions for the requested thread - pub actions: Vec, -} - -/// Destination-checked control forwarded to the resource authority -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ClusterResourceControl { - /// Public API version - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Intended resource authority - pub destination_machine: MachineId, - /// Daemon that forwarded the control - pub source_machine: MachineId, - /// Resource that the control targets - pub resource_id: ResourceId, - /// Requested control - pub control: ResourceControl, -} - -/// One authority-owned resource mutation -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceControl { - /// A dashboard or CLI action with a stable operation identity - Action { - /// Resource revision the caller observed - expected_revision: ResourceRevision, - /// Stable retry identity - operation_id: Uuid, - /// Requested action - action: BrowserResourceAction, - }, - /// Explicit supervisor replacement - ReplaceSupervisor { - /// Resource revision the caller observed - expected_revision: ResourceRevision, - /// Exact new supervisor thread - supervisor: SupervisorAddress, - }, -} diff --git a/src/resource/background_launch.rs b/src/resource/background_launch.rs deleted file mode 100644 index 945a455..0000000 --- a/src/resource/background_launch.rs +++ /dev/null @@ -1,590 +0,0 @@ -//! Versioned Fleet protocol for a first background launch from a remote supervisor -//! -//! A first background launch has no loan or action, so it cannot use the -//! action-bound protocol. The supervisor machine owns the callback route and -//! saves it with a fixed request, task, spec, and supervisor assignment before -//! it sends the launch. The resource authority owns the GPU, the task row, and -//! the only spawn. The same versioned document binds the trainer attempt of the -//! running task, which the authority verifies from its own runtime evidence - -use serde::{Deserialize, Serialize}; - -use super::api::ResourceBackgroundSubmitOutcome; -use super::trainer_publication::AttemptBinding; -use super::{ - AssignmentRevision, LoanId, Resource, ResourceId, ResourceRevision, SupervisorAddress, -}; -use crate::domain::{API_VERSION, ProcessStatus, TaskId}; -use crate::machine::MachineId; -use crate::spec::NormalizedSpec; -use crate::submission::{NormalizedSpecSha256, RequestId, normalized_spec_sha256}; - -/// Version of the remote background request and response documents -pub const RESOURCE_BACKGROUND_PROTOCOL_VERSION: u32 = 1; - -/// Authority route that serves every remote background operation -pub const RESOURCE_BACKGROUND_PATH: &str = "/v1/cluster/resource-background"; - -/// Origin route that proves the supervisor saved one background launch route -pub const RESOURCE_BACKGROUND_ROUTE_PROOF_PATH: &str = - "/v1/cluster/origin/resource-background-routes"; - -/// Supervisor assignment that one remote background operation acts for -/// -/// The authority accepts an operation only while the resource still names this -/// exact supervisor machine, thread, and assignment revision -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct BackgroundSupervisorAssignment { - /// Fixed resource authority and execution machine - pub authority_machine: MachineId, - /// Resource whose background slot the operation uses - pub resource_id: ResourceId, - /// Assigned supervisor machine and thread - pub supervisor: SupervisorAddress, - /// Supervisor assignment revision - pub assignment_revision: AssignmentRevision, -} - -impl BackgroundSupervisorAssignment { - /// Read the current assignment from one resource - #[must_use] - pub fn of(resource: &Resource) -> Self { - Self { - authority_machine: resource.authority_machine(), - resource_id: resource.id, - supervisor: resource.supervisor, - assignment_revision: resource.assignment_revision, - } - } - - /// Whether the resource still names this exact assignment - #[must_use] - pub fn is_current(&self, resource: &Resource) -> bool { - *self == Self::of(resource) - } -} - -/// Fixed binding that one remote first background launch saves on both machines -/// -/// The expected revision is the resource revision that the supervisor machine -/// read before it saved the route. The authority refuses a launch after any -/// later resource transition -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct BackgroundLaunchBinding { - /// Supervisor assignment that chose the launch - pub assignment: BackgroundSupervisorAssignment, - /// Resource revision observed before the route was saved - pub expected_state_revision: ResourceRevision, -} - -impl BackgroundLaunchBinding { - /// Machine that owns the task's callback route - #[must_use] - pub const fn origin_machine(&self) -> MachineId { - self.assignment.supervisor.machine - } - - /// Machine that executes the task - #[must_use] - pub const fn execution_machine(&self) -> MachineId { - self.assignment.authority_machine - } -} - -/// Exact binding that the authority saved with one accepted remote launch -/// -/// A retry must match every field -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct RemoteBackgroundLaunchReceipt { - /// Supervisor assignment and resource revision that the launch bound - pub binding: BackgroundLaunchBinding, - /// Stable retry identity, distinct from the task identity - pub request_id: RequestId, - /// Preallocated global task identity - pub task_id: TaskId, - /// Digest of the full normalized spec - pub normalized_spec_sha256: NormalizedSpecSha256, -} - -/// One operation that a remote supervisor asks the authority to apply -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceBackgroundOperation { - /// Accept the exact saved launch once and start it only on first insertion - Launch { - /// Resource revision observed before the route was saved - expected_state_revision: ResourceRevision, - /// Stable retry identity - request_id: RequestId, - /// Preallocated global task identity - task_id: TaskId, - /// Full normalized trainer spec saved in the route - spec: NormalizedSpec, - /// Digest of `spec` - normalized_spec_sha256: NormalizedSpecSha256, - }, - /// Bind the named trainer attempt to the running registered background task - /// - /// The request names only the attempt identity. The authority reads the - /// attempt request and probes the held ownership lock itself - BindTrainerAttempt { - /// Registered background task - task_id: TaskId, - /// Trainer identity of the running attempt - attempt_binding: AttemptBinding, - }, -} - -/// Strict versioned request from a remote supervisor to the resource authority -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceBackgroundRequest { - /// Public API schema version - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Remote background document version - pub background_protocol_version: u32, - /// Intended resource authority - pub destination_machine: MachineId, - /// Supervisor machine that sent the request and owns the callback route - pub source_machine: MachineId, - /// Exact resource and supervisor assignment - pub assignment: BackgroundSupervisorAssignment, - /// Requested operation - pub operation: ResourceBackgroundOperation, -} - -/// Why a remote background request is malformed before any saved state is read -#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] -pub enum ResourceBackgroundRequestError { - /// The API or document version is not supported - #[error("unsupported remote background version")] - UnsupportedVersion, - /// The destination is not the authority named by the assignment - #[error("remote background destination is not the named authority")] - DestinationMismatch, - /// The sender is not the assigned supervisor machine - #[error("remote background source is not the supervisor machine")] - SourceMismatch, - /// The supervisor runs on the authority and must use the local path - #[error("a co-located supervisor must use the local background path")] - CoLocatedSupervisor, - /// An identity is nil or one UUID has two roles - #[error("remote background identities are invalid")] - InvalidIdentity, - /// The spec does not fit a remote background launch or its digest - #[error("remote background spec is invalid")] - InvalidSpec, -} - -impl ResourceBackgroundRequest { - /// Build a current-version request from the supervisor machine - #[must_use] - pub fn new( - protocol_version: u32, - assignment: BackgroundSupervisorAssignment, - operation: ResourceBackgroundOperation, - ) -> Self { - Self { - api_version: API_VERSION, - protocol_version, - background_protocol_version: RESOURCE_BACKGROUND_PROTOCOL_VERSION, - destination_machine: assignment.authority_machine, - source_machine: assignment.supervisor.machine, - assignment, - operation, - } - } - - /// Validate versions, route owners, identity shape, and spec digest without saved state - pub fn validate(&self) -> Result<(), ResourceBackgroundRequestError> { - use ResourceBackgroundRequestError as Error; - - if self.api_version != API_VERSION - || self.background_protocol_version != RESOURCE_BACKGROUND_PROTOCOL_VERSION - { - return Err(Error::UnsupportedVersion); - } - let assignment = &self.assignment; - if self.destination_machine != assignment.authority_machine { - return Err(Error::DestinationMismatch); - } - if self.source_machine != assignment.supervisor.machine { - return Err(Error::SourceMismatch); - } - if self.source_machine == self.destination_machine { - return Err(Error::CoLocatedSupervisor); - } - if self.source_machine.as_uuid().is_nil() - || self.destination_machine.as_uuid().is_nil() - || assignment.supervisor.thread.0.is_nil() - { - return Err(Error::InvalidIdentity); - } - - match &self.operation { - ResourceBackgroundOperation::Launch { - request_id, - task_id, - spec, - normalized_spec_sha256: digest, - .. - } => { - if request_id.0.is_nil() || task_id.0.is_nil() || request_id.0 == task_id.0 { - return Err(Error::InvalidIdentity); - } - let matches = normalized_spec_sha256(spec).is_ok_and(|saved| saved == *digest); - if !matches - || spec.machine.is_some() - || spec.thread != assignment.supervisor.thread - || !matches!(spec.workload, crate::spec::NormalizedWorkload::Task(_)) - { - return Err(Error::InvalidSpec); - } - Ok(()) - } - ResourceBackgroundOperation::BindTrainerAttempt { task_id, .. } - if task_id.0.is_nil() => - { - Err(Error::InvalidIdentity) - } - ResourceBackgroundOperation::BindTrainerAttempt { .. } => Ok(()), - } - } - - /// Fixed launch receipt that a launch operation asks the authority to accept - #[must_use] - pub fn launch_receipt(&self) -> Option { - let ResourceBackgroundOperation::Launch { - expected_state_revision, - request_id, - task_id, - normalized_spec_sha256, - .. - } = &self.operation - else { - return None; - }; - Some(RemoteBackgroundLaunchReceipt { - binding: BackgroundLaunchBinding { - assignment: self.assignment, - expected_state_revision: *expected_state_revision, - }, - request_id: *request_id, - task_id: *task_id, - normalized_spec_sha256: *normalized_spec_sha256, - }) - } -} - -/// Definitive authority reason for refusing one remote background operation -/// -/// Every refusal writes nothing on the authority -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceBackgroundRejection { - /// The authority does not own this resource - ResourceNotFound, - /// The request does not come from the current supervisor assignment - NotCurrentSupervisor, - /// The resource changed after the supervisor machine read it - StaleRevision { - /// Revision named by the request - expected: ResourceRevision, - /// Revision saved by the authority - actual: ResourceRevision, - }, - /// The supervisor machine has no saved callback route for this task - RouteEvidenceMissing, - /// The supervisor machine's saved route names other identities or content - RouteEvidenceMismatch, - /// The request identity belongs to a different launch or task - ConflictingRetry, - /// A fixed request or task identity already belongs to other records - IdentityConflict, - /// A non-closed loan owns the resource; background work returns through its action - ActiveLoan { - /// Non-closed loan - loan_id: LoanId, - }, - /// Queued resource work blocks a new background launch - QueuedWorkAhead { - /// Queued request that blocks the launch - request_id: RequestId, - }, - /// The registered background task has not ended - BackgroundTaskActive { - /// Registered task - task_id: TaskId, - /// Task-layer state - state: ProcessStatus, - }, - /// The registered background task has no task record on the authority - BackgroundTaskMissing { - /// Registered task - task_id: TaskId, - }, - /// An earlier first background launch has not reached a confirmed start or an end - LaunchPending { - /// Pending launch task - task_id: TaskId, - }, - /// An earlier background task has no verified release evidence - PredecessorReleaseUnproven { - /// Earlier task that may still own the resource - task_id: TaskId, - }, - /// The command has no ownership contract that release proof can verify - UnsupportedCommand { - /// Authority-side reason - reason: String, - }, - /// The spec cannot run on the authority as written - InvalidSpec { - /// Authority-side reason - reason: String, - }, - /// The authority cannot verify the named trainer attempt for this task now - TrainerAttemptRefused { - /// Authority-side reason - reason: String, - }, - /// The task is already associated with a different trainer attempt - TrainerAttemptConflict, -} - -/// Typed authority result for one remote background request -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceBackgroundOutcome { - /// The authority accepted the exact launch - Accepted { - /// Saved exact launch binding - receipt: RemoteBackgroundLaunchReceipt, - /// Whether this request inserted the task or observed an earlier insertion - acceptance: ResourceBackgroundSubmitOutcome, - /// Authoritative resource after the acceptance - resource: Resource, - }, - /// The authority associated the exact trainer attempt with the task - TrainerAttemptBound { - /// Authoritative resource after the association - resource: Resource, - /// Registered task bound to the attempt - task_id: TaskId, - /// Canonical runtime root whose lock the trainer held - runtime_root: std::path::PathBuf, - /// Trainer identity of the associated attempt - attempt_binding: AttemptBinding, - }, - /// The authority refused the operation and wrote nothing - Rejected { - /// Definitive reason - reason: ResourceBackgroundRejection, - }, -} - -/// Strict versioned response from the resource authority -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceBackgroundResponse { - /// Public API schema version - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Remote background document version - pub background_protocol_version: u32, - /// Authority that handled the request - pub destination_machine: MachineId, - /// Resource named by the request - pub resource_id: ResourceId, - /// Typed result - pub outcome: ResourceBackgroundOutcome, -} - -impl ResourceBackgroundResponse { - /// Build the current-version response to one validated request - #[must_use] - pub fn new(request: &ResourceBackgroundRequest, outcome: ResourceBackgroundOutcome) -> Self { - Self { - api_version: API_VERSION, - protocol_version: request.protocol_version, - background_protocol_version: RESOURCE_BACKGROUND_PROTOCOL_VERSION, - destination_machine: request.destination_machine, - resource_id: request.assignment.resource_id, - outcome, - } - } -} - -#[cfg(test)] -mod tests { - use serde_json::json; - use uuid::Uuid; - - use super::{ - BackgroundSupervisorAssignment, RESOURCE_BACKGROUND_PROTOCOL_VERSION, - ResourceBackgroundOperation, ResourceBackgroundRequest, ResourceBackgroundRequestError, - }; - use crate::domain::{TaskId, ThreadId}; - use crate::machine::MachineId; - use crate::resource::trainer_publication::AttemptBinding; - use crate::resource::{AssignmentRevision, ResourceId, ResourceRevision, SupervisorAddress}; - use crate::spec::NormalizedSpec; - use crate::submission::{RequestId, normalized_spec_sha256}; - - fn assignment() -> BackgroundSupervisorAssignment { - BackgroundSupervisorAssignment { - authority_machine: MachineId::new(), - resource_id: ResourceId::new(), - supervisor: SupervisorAddress { - machine: MachineId::new(), - thread: ThreadId(Uuid::now_v7()), - }, - assignment_revision: AssignmentRevision::new(2), - } - } - - fn spec(thread: ThreadId) -> NormalizedSpec { - serde_json::from_value(json!({ - "api_version": 1, - "thread": thread, - "name": "trainer", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": ["python3", "-m", "ops.run_segment", "run"] } - })) - .unwrap() - } - - fn launch(assignment: BackgroundSupervisorAssignment) -> ResourceBackgroundRequest { - let spec = spec(assignment.supervisor.thread); - ResourceBackgroundRequest::new( - 1, - assignment, - ResourceBackgroundOperation::Launch { - expected_state_revision: ResourceRevision::new(4), - request_id: RequestId::new(), - task_id: TaskId::new(), - normalized_spec_sha256: normalized_spec_sha256(&spec).unwrap(), - spec, - }, - ) - } - - #[test] - fn request_rejects_unknown_fields_and_callback_context() { - let request = launch(assignment()); - let mut wire = serde_json::to_value(&request).unwrap(); - assert_eq!(wire["operation"]["type"], "launch"); - serde_json::from_value::(wire.clone()).unwrap(); - - wire["operation"]["callback"] = json!({"cwd": "/tmp"}); - assert!(serde_json::from_value::(wire.clone()).is_err()); - wire["operation"] - .as_object_mut() - .unwrap() - .remove("callback"); - wire["env"] = json!({"path": "/bin", "home": "/tmp"}); - assert!(serde_json::from_value::(wire).is_err()); - } - - #[test] - fn validation_checks_owners_identities_and_the_carried_digest() { - use ResourceBackgroundRequestError as Error; - - let request = launch(assignment()); - assert_eq!(request.validate(), Ok(())); - - let mut wrong_destination = request.clone(); - wrong_destination.destination_machine = MachineId::new(); - assert_eq!( - wrong_destination.validate(), - Err(Error::DestinationMismatch) - ); - let mut wrong_source = request.clone(); - wrong_source.source_machine = MachineId::new(); - assert_eq!(wrong_source.validate(), Err(Error::SourceMismatch)); - - let mut co_located = assignment(); - co_located.supervisor.machine = co_located.authority_machine; - assert_eq!( - launch(co_located).validate(), - Err(Error::CoLocatedSupervisor) - ); - - let mut reused = request.clone(); - let ResourceBackgroundOperation::Launch { - request_id, - task_id, - .. - } = &mut reused.operation - else { - unreachable!(); - }; - *request_id = RequestId(task_id.0); - assert_eq!(reused.validate(), Err(Error::InvalidIdentity)); - - let mut changed = request.clone(); - let ResourceBackgroundOperation::Launch { spec, .. } = &mut changed.operation else { - unreachable!(); - }; - spec.name = crate::domain::TaskName::parse("changed").unwrap(); - assert_eq!(changed.validate(), Err(Error::InvalidSpec)); - - let mut other_thread = request.clone(); - let ResourceBackgroundOperation::Launch { - spec, - normalized_spec_sha256: digest, - .. - } = &mut other_thread.operation - else { - unreachable!(); - }; - *spec = super::tests::spec(ThreadId(Uuid::now_v7())); - *digest = normalized_spec_sha256(spec).unwrap(); - assert_eq!(other_thread.validate(), Err(Error::InvalidSpec)); - - let mut version = request; - version.background_protocol_version = RESOURCE_BACKGROUND_PROTOCOL_VERSION + 1; - assert_eq!(version.validate(), Err(Error::UnsupportedVersion)); - } - - #[test] - fn launch_receipt_comes_only_from_a_launch() { - let request = launch(assignment()); - let receipt = request.launch_receipt().unwrap(); - assert_eq!(receipt.binding.assignment, request.assignment); - assert_eq!( - receipt.binding.expected_state_revision, - ResourceRevision::new(4) - ); - assert_eq!(receipt.binding.origin_machine(), request.source_machine); - assert_eq!( - receipt.binding.execution_machine(), - request.destination_machine - ); - - let bind = ResourceBackgroundRequest::new( - 1, - request.assignment, - ResourceBackgroundOperation::BindTrainerAttempt { - task_id: TaskId::new(), - attempt_binding: AttemptBinding { - campaign_id: "campaign".into(), - campaign_revision_id: "revision".into(), - task_id: "task".into(), - attempt_id: "attempt".into(), - attempt_number: 1, - ownership_token: "owner".into(), - }, - }, - ); - assert_eq!(bind.validate(), Ok(())); - assert!(bind.launch_receipt().is_none()); - } -} diff --git a/src/resource/bound_action.rs b/src/resource/bound_action.rs deleted file mode 100644 index 9fbf44d..0000000 --- a/src/resource/bound_action.rs +++ /dev/null @@ -1,767 +0,0 @@ -//! Versioned Fleet protocol for tasks bound to one resource action -//! -//! A supervisor on another machine owns the callback route for its release watcher -//! or return task. The resource authority owns the loan, the action, the canonical -//! command, and the only spawn. The supervisor saves its fixed-ID origin route -//! before it sends a launch, and every retry repeats the same identities - -use std::time::Duration; - -use serde::{Deserialize, Serialize}; - -use super::{ - ActionId, Loan, ResourceRevision, ReturnDecision, ReturnDecisionWindow, ReturnLaunch, - ReturnWork, SupervisorActionAuthority, validate_release_watcher_identity, -}; -use crate::domain::{API_VERSION, ProcessStatus, TaskId}; -use crate::machine::MachineId; -use crate::spec::NormalizedSpec; -use crate::submission::{NormalizedSpecSha256, RequestId, normalized_spec_sha256}; - -/// Version of the resource-action request and response documents -pub const RESOURCE_ACTION_PROTOCOL_VERSION: u32 = 1; - -/// Authority route that serves every resource-action operation -pub const RESOURCE_ACTION_PATH: &str = "/v1/cluster/resource-actions"; - -/// Origin route that proves the supervisor saved one action-bound callback route -pub const RESOURCE_ACTION_ROUTE_PROOF_PATH: &str = "/v1/cluster/origin/resource-action-routes"; - -/// Socket-only route on the supervisor machine that starts one resource action -pub const RESOURCE_ACTION_SUBMIT_PATH: &str = "/v1/internal/resource-actions"; - -/// Kind of task that one resource action binds -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum ResourceActionKind { - /// The authority-built watcher for a release action - ReleaseWatcher, - /// The returning background task for a return action - Return, -} - -/// Supervisor choice that one action-bound route launches -/// -/// The route keeps it so a retry after a lost reply or restart resends exactly -/// the same launch without asking the caller again -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceActionLaunch { - /// The watcher bound to the release action for this background task - ReleaseWatcher { - /// Background task named by the release action - observed_background_task: TaskId, - }, - /// The returning background task for the return action - Return { - /// Typed work chosen by the supervisor - work: Box, - }, -} - -impl ResourceActionLaunch { - /// Kind of task this choice launches - #[must_use] - pub const fn kind(&self) -> ResourceActionKind { - match self { - Self::ReleaseWatcher { .. } => ResourceActionKind::ReleaseWatcher, - Self::Return { .. } => ResourceActionKind::Return, - } - } -} - -/// One operation that a remote supervisor asks the authority to apply -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceActionOperation { - /// Return the watcher identity and canonical command that the authority bound - PrepareReleaseWatcher { - /// Background task named by the release action - observed_background_task: TaskId, - }, - /// Accept the exact prepared watcher once and start it only on first insertion - LaunchReleaseWatcher { - /// Background task named by the release action - observed_background_task: TaskId, - /// Fixed identities and digest from the prepared watcher - task: ActionTaskIdentity, - }, - /// Derive the canonical return task for one launch decision without binding it - PrepareReturn { - /// Supervisor-chosen fixed identities and typed work - launch: ReturnLaunch, - }, - /// Bind the exact prepared return task once and start it only on first insertion - LaunchReturn { - /// Supervisor-chosen fixed identities and typed work - launch: ReturnLaunch, - /// Digest of the prepared canonical spec - normalized_spec_sha256: NormalizedSpecSha256, - }, - /// Close the return action without starting background work - NoResume { - /// Supervisor's durable decision reason - reason: String, - }, - /// Close a Restoring loan whose bound task ended before a confirmed start - ResolveEndedRestore { - /// Bound return task that ended - task_id: TaskId, - /// Supervisor's durable resolution reason - reason: String, - }, - /// Keep queued work waiting longer for the return decision - HoldReturn { - /// Requested decision time from now, capped by the window limit - #[serde(with = "humantime_serde")] - hold: Duration, - }, -} - -impl ResourceActionOperation { - /// Kind of task this operation prepares or launches, if any - #[must_use] - pub fn task_kind(&self) -> Option { - match self { - Self::PrepareReleaseWatcher { .. } | Self::LaunchReleaseWatcher { .. } => { - Some(ResourceActionKind::ReleaseWatcher) - } - Self::PrepareReturn { .. } | Self::LaunchReturn { .. } => { - Some(ResourceActionKind::Return) - } - Self::NoResume { .. } | Self::ResolveEndedRestore { .. } | Self::HoldReturn { .. } => { - None - } - } - } - - /// Fixed task identity that a launch operation asks the authority to accept - #[must_use] - pub fn launch_identity(&self) -> Option { - match self { - Self::LaunchReleaseWatcher { task, .. } => Some(*task), - Self::LaunchReturn { - launch, - normalized_spec_sha256, - } => Some(ActionTaskIdentity { - request_id: launch.request_id, - task_id: launch.task_id, - normalized_spec_sha256: *normalized_spec_sha256, - }), - Self::PrepareReleaseWatcher { .. } - | Self::PrepareReturn { .. } - | Self::NoResume { .. } - | Self::ResolveEndedRestore { .. } - | Self::HoldReturn { .. } => None, - } - } -} - -/// Fixed request and task identities of one action-bound task with its spec digest -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ActionTaskIdentity { - /// Stable retry identity, distinct from the task identity - pub request_id: RequestId, - /// Preallocated global task identity - pub task_id: TaskId, - /// Digest of the canonical normalized spec - pub normalized_spec_sha256: NormalizedSpecSha256, -} - -/// Strict versioned request from a remote supervisor to the resource authority -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceActionRequest { - /// Public API schema version - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Resource-action document version - pub action_protocol_version: u32, - /// Intended resource authority - pub destination_machine: MachineId, - /// Supervisor machine that sent the request and owns the callback route - pub source_machine: MachineId, - /// Exact resource, loan, action, revision, and supervisor assignment - pub authority: SupervisorActionAuthority, - /// Requested operation - pub operation: ResourceActionOperation, -} - -/// Why a resource-action request is malformed before any saved state is read -#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] -pub enum ResourceActionRequestError { - /// The API or document version is not supported - #[error("unsupported resource-action version")] - UnsupportedVersion, - /// The destination is not the authority named by the action - #[error("resource-action destination is not the named authority")] - DestinationMismatch, - /// The sender is not the assigned supervisor machine - #[error("resource-action source is not the supervisor machine")] - SourceMismatch, - /// The supervisor runs on the authority and must use the local path - #[error("a co-located supervisor must use the local action path")] - CoLocatedSupervisor, - /// An identity is nil or one UUID has two roles - #[error("resource-action identities are invalid")] - InvalidIdentity, - /// A decision or resolution reason is empty - #[error("resource-action reason must not be empty")] - EmptyReason, -} - -impl ResourceActionRequest { - /// Build a current-version request from the supervisor machine - #[must_use] - pub fn new( - protocol_version: u32, - authority: SupervisorActionAuthority, - operation: ResourceActionOperation, - ) -> Self { - Self { - api_version: API_VERSION, - protocol_version, - action_protocol_version: RESOURCE_ACTION_PROTOCOL_VERSION, - destination_machine: authority.authority_machine, - source_machine: authority.supervisor.machine, - authority, - operation, - } - } - - /// Validate versions, route owners, and identity shape without saved state - pub fn validate(&self) -> Result<(), ResourceActionRequestError> { - if self.api_version != API_VERSION - || self.action_protocol_version != RESOURCE_ACTION_PROTOCOL_VERSION - { - return Err(ResourceActionRequestError::UnsupportedVersion); - } - let authority = &self.authority; - if self.destination_machine != authority.authority_machine { - return Err(ResourceActionRequestError::DestinationMismatch); - } - if self.source_machine != authority.supervisor.machine { - return Err(ResourceActionRequestError::SourceMismatch); - } - if self.source_machine == self.destination_machine { - return Err(ResourceActionRequestError::CoLocatedSupervisor); - } - if self.source_machine.as_uuid().is_nil() - || self.destination_machine.as_uuid().is_nil() - || authority.supervisor.thread.0.is_nil() - { - return Err(ResourceActionRequestError::InvalidIdentity); - } - - match &self.operation { - ResourceActionOperation::PrepareReleaseWatcher { - observed_background_task, - } if observed_background_task.0.is_nil() => { - Err(ResourceActionRequestError::InvalidIdentity) - } - ResourceActionOperation::LaunchReleaseWatcher { - observed_background_task, - task, - } if observed_background_task.0.is_nil() - || validate_release_watcher_identity( - *observed_background_task, - task.request_id, - task.task_id, - ) - .is_err() => - { - Err(ResourceActionRequestError::InvalidIdentity) - } - ResourceActionOperation::PrepareReturn { launch } - | ResourceActionOperation::LaunchReturn { launch, .. } - if !distinct_identity(launch.request_id, launch.task_id) => - { - Err(ResourceActionRequestError::InvalidIdentity) - } - ResourceActionOperation::NoResume { reason } - | ResourceActionOperation::ResolveEndedRestore { reason, .. } - if reason.trim().is_empty() => - { - Err(ResourceActionRequestError::EmptyReason) - } - ResourceActionOperation::ResolveEndedRestore { task_id, .. } if task_id.0.is_nil() => { - Err(ResourceActionRequestError::InvalidIdentity) - } - _ => Ok(()), - } - } -} - -fn distinct_identity(request_id: RequestId, task_id: TaskId) -> bool { - !request_id.0.is_nil() && !task_id.0.is_nil() && request_id.0 != task_id.0 -} - -/// Canonical task that the authority derived for one pending action -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct PreparedActionTask { - /// Stable retry identity for the task - pub request_id: RequestId, - /// Preallocated global task identity - pub task_id: TaskId, - /// Canonical normalized spec that the supervisor saves in its route - pub spec: NormalizedSpec, - /// Digest of `spec` - pub normalized_spec_sha256: NormalizedSpecSha256, -} - -impl PreparedActionTask { - /// Whether the digest covers the carried spec exactly - #[must_use] - pub fn digest_matches(&self) -> bool { - normalized_spec_sha256(&self.spec).is_ok_and(|digest| digest == self.normalized_spec_sha256) - } -} - -/// Exact action binding that the authority saved with one accepted task -/// -/// The supervisor machine is the callback origin and the authority is the -/// execution machine. A retry must match every field -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ActionTaskReceipt { - /// Kind of bound task - pub kind: ResourceActionKind, - /// Exact action authority that accepted the task - pub authority: SupervisorActionAuthority, - /// Stable retry identity for the task - pub request_id: RequestId, - /// Preallocated global task identity - pub task_id: TaskId, - /// Digest of the canonical normalized spec - pub normalized_spec_sha256: NormalizedSpecSha256, -} - -impl ActionTaskReceipt { - /// Machine that owns the task's callback route - #[must_use] - pub const fn origin_machine(&self) -> MachineId { - self.authority.supervisor.machine - } - - /// Machine that executes the task - #[must_use] - pub const fn execution_machine(&self) -> MachineId { - self.authority.authority_machine - } -} - -/// Whether one launch request inserted its task or only observed an earlier insertion -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ActionTaskAcceptance { - /// This request committed the task records, so it made the only spawn attempt - Inserted, - /// An exact earlier acceptance existed; the authority did not spawn again - Existing { - /// State retained by the task layer - state: ProcessStatus, - }, -} - -/// Definitive authority reason for refusing one resource action -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceActionRejection { - /// The request does not come from the current supervisor assignment - NotCurrentSupervisor, - /// The loan does not have this pending action in the required phase - ActionNotPending, - /// The request names a stale resource revision - StaleRevision { - /// Revision named by the request - expected: ResourceRevision, - /// Revision saved by the authority - actual: ResourceRevision, - }, - /// The spec digest differs from the canonical spec that the authority derived - SpecMismatch, - /// A fixed request or task identity already belongs to other records - IdentityConflict, - /// A saved receipt holds a different request for the same action - ConflictingRetry, - /// The supervisor machine has no saved callback route for this task - RouteEvidenceMissing, - /// The supervisor machine's saved callback route names other identities or content - RouteEvidenceMismatch, - /// The authority cannot bind a watcher for this release action now - WatcherUnavailable { - /// Authority-side reason - reason: String, - }, - /// The typed return decision does not fit the saved action - DecisionRejected { - /// Authority-side reason - reason: String, - }, - /// The return task has not ended with proven process release - RestoreNotResolvable { - /// Authority-side reason - reason: String, - }, -} - -/// Typed authority result for one resource-action request -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceActionOutcome { - /// The canonical task for the requested launch - Prepared { - /// Identities, spec, and digest to save in the origin route - task: PreparedActionTask, - }, - /// The authority accepted the fixed task for the action - Accepted { - /// Saved exact action binding - receipt: ActionTaskReceipt, - /// Whether this request inserted the task - acceptance: ActionTaskAcceptance, - }, - /// The decision closed the loan without another task - Closed { - /// Closed loan with its retained return context - loan: Loan, - /// Resource revision committed with the closure - state_revision: ResourceRevision, - }, - /// The authority saved a later deadline for the return decision - ReturnHeld { - /// Saved decision window of the action - window: ReturnDecisionWindow, - }, - /// The authority refused the request and wrote nothing - Rejected { - /// Definitive reason - reason: ResourceActionRejection, - }, -} - -/// Strict versioned response from the resource authority -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceActionResponse { - /// Public API schema version - pub api_version: u32, - /// Cluster protocol version - pub protocol_version: u32, - /// Resource-action document version - pub action_protocol_version: u32, - /// Authority that handled the request - pub destination_machine: MachineId, - /// Action named by the request - pub action_id: ActionId, - /// Typed result - pub outcome: ResourceActionOutcome, -} - -impl ResourceActionResponse { - /// Build the current-version response to one validated request - #[must_use] - pub fn new(request: &ResourceActionRequest, outcome: ResourceActionOutcome) -> Self { - Self { - api_version: API_VERSION, - protocol_version: request.protocol_version, - action_protocol_version: RESOURCE_ACTION_PROTOCOL_VERSION, - destination_machine: request.destination_machine, - action_id: request.authority.action_id, - outcome, - } - } -} - -/// Supervisor choice sent to its own daemon for one pending resource action -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceActionChoice { - /// Launch the watcher that the authority bound to a release action - ReleaseWatcher { - /// Background task named by the release notice - observed_background_task: TaskId, - }, - /// Apply one return decision - Return { - /// No-resume or fixed-identity launch decision - decision: ReturnDecision, - }, - /// Close a Restoring loan whose bound task ended before a confirmed start - ResolveEndedRestore { - /// Bound return task that ended - task_id: TaskId, - /// Supervisor's durable resolution reason - reason: String, - }, - /// Keep queued work waiting longer for the return decision - HoldReturn { - /// Requested decision time from now, capped by the window limit - #[serde(with = "humantime_serde")] - hold: Duration, - }, -} - -/// Socket request from the supervisor thread's machine to its own daemon -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceActionSubmitRequest { - /// Public API schema version - pub api_version: u32, - /// Exact action authority from the supervisor notice - pub authority: SupervisorActionAuthority, - /// Supervisor choice - pub choice: ResourceActionChoice, -} - -/// Durable result of one supervisor-side resource action submission -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceActionSubmitOutcome { - /// The authority accepted the bound task, and this machine owns its callback route - Accepted { - /// Exact receipt saved by the authority - receipt: ActionTaskReceipt, - /// Last task state learned by this machine, if any - last_execution_state: Option, - }, - /// This machine is the authority, and it bound the return task with its own callback route - /// - /// No remote receipt exists for this task: the route committed with the task - /// records in the authority's return transaction - LocalReturnAccepted { - /// Exact action authority and fixed identities of the bound task - receipt: LocalReturnReceipt, - /// Whether this request inserted the task or observed the earlier binding - acceptance: LocalReturnAcceptance, - }, - /// The decision closed the loan without another task - Closed { - /// Closed loan with its retained return context - loan: Loan, - /// Resource revision committed with the closure - state_revision: ResourceRevision, - }, - /// The authority saved a later deadline for the return decision - ReturnHeld { - /// Saved decision window of the action - window: ReturnDecisionWindow, - }, - /// The authority refused the action and wrote nothing - Rejected { - /// Definitive reason - reason: ResourceActionRejection, - }, -} - -/// Exact binding of one return task on the machine that is both supervisor and authority -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct LocalReturnReceipt { - /// Exact action authority that bound the task - pub authority: SupervisorActionAuthority, - /// Stable retry identity for the task - pub request_id: RequestId, - /// Preallocated global task identity - pub task_id: TaskId, -} - -/// Whether one co-located return decision inserted its task or found the earlier binding -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum LocalReturnAcceptance { - /// This request committed the task records and made the only spawn attempt - Inserted { - /// Restoring loan that keeps the resource reserved - loan: Box, - /// Resource revision committed with the binding - state_revision: ResourceRevision, - }, - /// An exact earlier decision bound the task; nothing was spawned again - Existing { - /// State retained by the task layer - state: ProcessStatus, - }, -} - -/// Socket response for one supervisor-side resource action submission -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceActionSubmitResponse { - /// Public API schema version - pub api_version: u32, - /// Durable result - pub outcome: ResourceActionSubmitOutcome, -} - -#[cfg(test)] -mod tests { - use serde_json::json; - use uuid::Uuid; - - use super::{ - ActionTaskIdentity, PreparedActionTask, RESOURCE_ACTION_PROTOCOL_VERSION, - ResourceActionKind, ResourceActionOperation, ResourceActionRequest, - ResourceActionRequestError, - }; - use crate::domain::{TaskId, ThreadId}; - use crate::machine::MachineId; - use crate::resource::{ - ActionId, AssignmentRevision, LoanId, ResourceId, ResourceRevision, ReturnLaunch, - ReturnWork, SupervisorActionAuthority, SupervisorAddress, - }; - use crate::spec::NormalizedSpec; - use crate::submission::{RequestId, normalized_spec_sha256}; - - fn authority() -> SupervisorActionAuthority { - SupervisorActionAuthority { - authority_machine: MachineId::new(), - resource_id: ResourceId::new(), - loan_id: LoanId::new(), - action_id: ActionId::new(), - expected_state_revision: ResourceRevision::new(3), - supervisor: SupervisorAddress { - machine: MachineId::new(), - thread: ThreadId(Uuid::now_v7()), - }, - assignment_revision: AssignmentRevision::new(1), - } - } - - fn launch_watcher(authority: SupervisorActionAuthority) -> ResourceActionRequest { - ResourceActionRequest::new( - 1, - authority, - ResourceActionOperation::LaunchReleaseWatcher { - observed_background_task: TaskId::new(), - task: ActionTaskIdentity { - request_id: RequestId::new(), - task_id: TaskId::new(), - normalized_spec_sha256: serde_json::from_value(json!("0".repeat(64))).unwrap(), - }, - }, - ) - } - - #[test] - fn request_rejects_unknown_fields_and_keeps_its_wire_shape() { - let request = launch_watcher(authority()); - let mut wire = serde_json::to_value(&request).unwrap(); - assert_eq!(wire["operation"]["type"], "launch_release_watcher"); - serde_json::from_value::(wire.clone()).unwrap(); - - wire["operation"]["command"] = json!(["/bin/sh", "-c", "anything"]); - assert!(serde_json::from_value::(wire.clone()).is_err()); - wire["operation"].as_object_mut().unwrap().remove("command"); - wire["callback"] = json!({"cwd": "/tmp"}); - assert!(serde_json::from_value::(wire).is_err()); - } - - #[test] - fn request_validation_checks_route_owners_and_identities() { - let request = launch_watcher(authority()); - assert_eq!(request.validate(), Ok(())); - - let mut wrong_destination = request.clone(); - wrong_destination.destination_machine = MachineId::new(); - assert_eq!( - wrong_destination.validate(), - Err(ResourceActionRequestError::DestinationMismatch) - ); - - let mut wrong_source = request.clone(); - wrong_source.source_machine = MachineId::new(); - assert_eq!( - wrong_source.validate(), - Err(ResourceActionRequestError::SourceMismatch) - ); - - let mut co_located = authority(); - co_located.supervisor.machine = co_located.authority_machine; - assert_eq!( - launch_watcher(co_located).validate(), - Err(ResourceActionRequestError::CoLocatedSupervisor) - ); - - let mut reused = request.clone(); - let ResourceActionOperation::LaunchReleaseWatcher { task, .. } = &mut reused.operation - else { - unreachable!(); - }; - task.request_id = RequestId(task.task_id.0); - assert_eq!( - reused.validate(), - Err(ResourceActionRequestError::InvalidIdentity) - ); - - let mut version = request; - version.action_protocol_version = RESOURCE_ACTION_PROTOCOL_VERSION + 1; - assert_eq!( - version.validate(), - Err(ResourceActionRequestError::UnsupportedVersion) - ); - - let no_resume = ResourceActionRequest::new( - 1, - authority(), - ResourceActionOperation::NoResume { reason: " ".into() }, - ); - assert_eq!( - no_resume.validate(), - Err(ResourceActionRequestError::EmptyReason) - ); - } - - #[test] - fn launch_identity_comes_only_from_launch_operations() { - let spec: NormalizedSpec = serde_json::from_value(json!({ - "api_version": 1, - "thread": ThreadId(Uuid::now_v7()), - "name": "return", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": ["/bin/echo", "resume"] } - })) - .unwrap(); - let launch = ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::NewBackgroundWork { - spec: crate::resource::CommandSpec::try_from(spec.clone()).unwrap(), - }, - }; - let digest = normalized_spec_sha256(&spec).unwrap(); - let operation = ResourceActionOperation::LaunchReturn { - launch: launch.clone(), - normalized_spec_sha256: digest, - }; - assert_eq!( - operation.launch_identity(), - Some(ActionTaskIdentity { - request_id: launch.request_id, - task_id: launch.task_id, - normalized_spec_sha256: digest, - }) - ); - assert_eq!(operation.task_kind(), Some(ResourceActionKind::Return)); - assert!( - ResourceActionOperation::PrepareReturn { launch } - .launch_identity() - .is_none() - ); - - let prepared = PreparedActionTask { - request_id: RequestId::new(), - task_id: TaskId::new(), - spec: spec.clone(), - normalized_spec_sha256: digest, - }; - assert!(prepared.digest_matches()); - let mut changed = prepared; - changed.spec.name = crate::domain::TaskName::parse("other").unwrap(); - assert!(!changed.digest_matches()); - } -} diff --git a/src/resource/command_shape.rs b/src/resource/command_shape.rs deleted file mode 100644 index 499663a..0000000 --- a/src/resource/command_shape.rs +++ /dev/null @@ -1,1613 +0,0 @@ -//! Strict validation of the maintained direct-segment trainer invocation - -use std::fs; -use std::io; -use std::path::{Path, PathBuf}; -use std::time::Duration; - -use thiserror::Error; - -use crate::domain::{TaskId, TaskRow}; -use crate::invocation::{CommandLine, resolve_executable}; -use crate::resource::TrainerAttemptAssociation; -use crate::resource::ownership_lock::VerifiedTrainerAttempt; -use crate::spec::{NormalizedSpec, NormalizedWorkload}; -use crate::submission::{NormalizedSpecSha256, normalized_spec_sha256}; - -const DEFAULT_TIMEOUT: Duration = Duration::from_secs(86_400); -const MAX_TIMEOUT_SECONDS: f64 = 7.0 * 24.0 * 60.0 * 60.0; -const REQUIRED_MODULE_ARGS: [&str; 3] = ["-m", "ops.run_segment", "run"]; - -/// A path role checked while binding an accepted direct-segment invocation -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum DirectSegmentPathRole { - /// The trainer task metadata file passed with `--task` - TaskFile, - /// The prepared trainer input directory passed with `--input-root` - InputRoot, - /// The optional input-verification receipt passed to the trainer - InputVerificationReceipt, - /// The accepted task's effective working directory - WorkingDirectory, - /// The trainer runtime root passed with `--runtime-root` - RuntimeRoot, -} - -/// A validated argv and filesystem binding for `python -m ops.run_segment run` -/// -/// This proves only that the accepted normalized specification and task row have -/// the maintained invocation shape and bind its paths to the saved canonical -/// runtime root. It does not prove which interpreter ran or that the live process -/// used the ownership lock. Release still requires exact task-exit evidence and -/// an exact held-lock guard through the release transaction -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct DirectSegmentCommandShape { - task_id: TaskId, - python_executable: PathBuf, - working_directory: PathBuf, - task_file: PathBuf, - input_root: PathBuf, - runtime_root: PathBuf, - image_argument: String, - input_verification_receipt: Option, - resume_generation: Option, - timeout: Duration, -} - -impl DirectSegmentCommandShape { - /// Validate an accepted specification and task row against one saved attempt association - /// - /// The normalized-spec digest and task row must match the association's exact task. The - /// returned shape proves invocation and path binding, not interpreter identity or live lock - /// use. Do not use it alone to release a resource - pub fn validate( - spec: &NormalizedSpec, - task: &TaskRow, - association: &TrainerAttemptAssociation, - ) -> Result { - Self::validate_binding( - spec, - task, - association.task_id(), - association.normalized_spec_sha256(), - association.verified_attempt(), - ) - } - - /// Validate an accepted specification and task row before creating its association - /// - /// The task ID and normalized-spec digest must come from the accepted executor identity - /// The verified attempt carries the exact runtime root that the command must name - pub fn validate_binding( - spec: &NormalizedSpec, - task: &TaskRow, - expected_task_id: TaskId, - expected_spec_sha256: NormalizedSpecSha256, - verified_attempt: &VerifiedTrainerAttempt, - ) -> Result { - if task.id != expected_task_id { - return Err(DirectSegmentCommandShapeError::TaskIdentityMismatch { - expected: expected_task_id, - found: task.id, - }); - } - - let spec_digest = normalized_spec_sha256(spec) - .map_err(DirectSegmentCommandShapeError::NormalizedSpecEncoding)?; - if spec_digest != expected_spec_sha256 { - return Err( - DirectSegmentCommandShapeError::NormalizedSpecDigestMismatch { task_id: task.id }, - ); - } - - Self::validate_invocation(spec, task, Some(verified_attempt.canonical_runtime_root())) - } - - /// Validate a first background launch before its task record exists - /// - /// The trainer creates its attempt and lock only after it starts, so this - /// check binds no runtime evidence. The runtime root must be absolute; the - /// later association compares it with the canonical root that the running - /// trainer locked - pub fn validate_launch( - spec: &NormalizedSpec, - task: &TaskRow, - ) -> Result { - Self::validate_invocation(spec, task, None) - } - - fn validate_invocation( - spec: &NormalizedSpec, - task: &TaskRow, - associated_runtime_root: Option<&Path>, - ) -> Result { - let NormalizedWorkload::Task(spec_workload) = &spec.workload else { - return Err(DirectSegmentCommandShapeError::NotCommandTask { task_id: task.id }); - }; - if task.name.as_ref() != Some(&spec.name) - || task.thread != spec.thread - || task.timeout != spec.timeout - || task.workload != crate::invocation::persist_workload(&spec.workload) - { - return Err(DirectSegmentCommandShapeError::TaskRowMismatch { task_id: task.id }); - } - - let working_directory = accepted_working_directory(spec, task)?; - check_trainer_layout(&working_directory)?; - - if !task.binary.is_absolute() { - return Err(DirectSegmentCommandShapeError::ExecutableNotAbsolute { - path: task.binary.clone(), - }); - } - let resolved_executable = resolve_executable( - spec_workload.command.program(), - &task.env.path, - &working_directory, - ) - .map_err(DirectSegmentCommandShapeError::ExecutableResolution)?; - if resolved_executable != task.binary { - return Err(DirectSegmentCommandShapeError::ExecutableMismatch { - resolved: resolved_executable, - accepted: task.binary.clone(), - }); - } - if !is_python_executable(&task.binary) { - return Err(DirectSegmentCommandShapeError::NotPythonExecutable { - path: task.binary.clone(), - }); - } - - let command = &spec_workload.command; - check_module_invocation(command)?; - let parsed = parse_flags(command)?; - - let task_file_arg = required_flag(parsed.task_file, "--task")?; - let task_file = canonical_file(DirectSegmentPathRole::TaskFile, Path::new(&task_file_arg))?; - let input_root_arg = required_flag(parsed.input_root, "--input-root")?; - let input_root = - canonical_directory(DirectSegmentPathRole::InputRoot, Path::new(&input_root_arg))?; - let runtime_root = PathBuf::from(required_flag(parsed.runtime_root, "--runtime-root")?); - match associated_runtime_root { - Some(expected) if runtime_root.as_os_str() != expected.as_os_str() => { - return Err(DirectSegmentCommandShapeError::RuntimeRootMismatch { - provided: runtime_root, - expected: expected.to_path_buf(), - }); - } - Some(_) => {} - None if !runtime_root.is_absolute() => { - return Err(DirectSegmentCommandShapeError::PathNotAbsolute { - role: DirectSegmentPathRole::RuntimeRoot, - path: runtime_root, - }); - } - None => {} - } - let image_argument = required_flag(parsed.image_argument, "--image-digest")?.to_owned(); - let input_verification_receipt = parsed - .input_verification_receipt - .map(|path| { - canonical_file( - DirectSegmentPathRole::InputVerificationReceipt, - Path::new(&path), - ) - }) - .transpose()?; - let timeout = parsed - .timeout_seconds - .map_or(Ok(DEFAULT_TIMEOUT), parse_timeout)?; - - Ok(Self { - task_id: task.id, - python_executable: task.binary.clone(), - working_directory, - task_file, - input_root, - runtime_root, - image_argument, - input_verification_receipt, - resume_generation: parsed.resume_generation, - timeout, - }) - } - - /// Return the exact Homebased task identity bound by this shape - #[must_use] - pub const fn task_id(&self) -> TaskId { - self.task_id - } - - /// Return the absolute executable path resolved and stored by the task row - #[must_use] - pub fn python_executable(&self) -> &Path { - &self.python_executable - } - - /// Return the canonical working directory used by the task row - #[must_use] - pub fn working_directory(&self) -> &Path { - &self.working_directory - } - - /// Return the canonical trainer task metadata file path - #[must_use] - pub fn task_file(&self) -> &Path { - &self.task_file - } - - /// Return the canonical prepared input root - #[must_use] - pub fn input_root(&self) -> &Path { - &self.input_root - } - - /// Return the exact saved canonical runtime root passed to the command - #[must_use] - pub fn runtime_root(&self) -> &Path { - &self.runtime_root - } - - /// Return the opaque `--image-digest` argument without scientific validation - #[must_use] - pub fn image_argument(&self) -> &str { - &self.image_argument - } - - /// Return the canonical input-verification receipt path, when supplied - #[must_use] - pub fn input_verification_receipt(&self) -> Option<&Path> { - self.input_verification_receipt.as_deref() - } - - /// Return the optional resume generation argument without checkpoint validation - #[must_use] - pub fn resume_generation(&self) -> Option<&str> { - self.resume_generation.as_deref() - } - - /// Return the parsed timeout, including the trainer's default when omitted - #[must_use] - pub const fn timeout(&self) -> Duration { - self.timeout - } -} - -/// A typed reason that an accepted task does not match the direct-segment invocation contract -#[derive(Debug, Error)] -pub enum DirectSegmentCommandShapeError { - /// The accepted task row has another task identity than its trainer-attempt binding - #[error("binding task {expected} does not match row task {found}")] - TaskIdentityMismatch { - /// Exact task in the supplied trainer-attempt binding - expected: TaskId, - /// Exact task in the supplied row - found: TaskId, - }, - /// The normalized specification cannot be serialized for digest comparison - #[error("cannot encode normalized trainer specification: {0}")] - NormalizedSpecEncoding(#[source] serde_json::Error), - /// The normalized specification differs from the accepted task digest - #[error("normalized trainer specification differs from task {task_id} binding")] - NormalizedSpecDigestMismatch { - /// Exact task whose accepted specification was checked - task_id: TaskId, - }, - /// The accepted specification is not a command task - #[error("trainer task {task_id} is not a command workload")] - NotCommandTask { - /// Exact task whose workload was checked - task_id: TaskId, - }, - /// The accepted task row does not match the normalized command workload - #[error("task row {task_id} differs from its normalized command specification")] - TaskRowMismatch { - /// Exact task whose row was checked - task_id: TaskId, - }, - /// A working directory path is not absolute - #[error("{role:?} path is not absolute: {path}")] - PathNotAbsolute { - /// Path role being checked - role: DirectSegmentPathRole, - /// Path supplied by the accepted task or command - path: PathBuf, - }, - /// An accepted remote working directory cannot be bound to the task row - #[error("accepted working directory {accepted} does not match task row directory {row}")] - WorkingDirectoryMismatch { - /// Directory resolved from the normalized specification - accepted: PathBuf, - /// Directory saved on the accepted task row - row: PathBuf, - }, - /// A path cannot be inspected - #[error("cannot inspect {role:?} path {path}: {source}")] - PathInspection { - /// Path role being checked - role: DirectSegmentPathRole, - /// Path that could not be inspected - path: PathBuf, - /// Filesystem error - #[source] - source: io::Error, - }, - /// A path is not its exact canonical absolute spelling - #[error("{role:?} path is not canonical: supplied {path}, canonical {canonical}")] - PathNotCanonical { - /// Path role being checked - role: DirectSegmentPathRole, - /// Path supplied by the accepted task or command - path: PathBuf, - /// Canonical path observed on disk - canonical: PathBuf, - }, - /// A checked path has the wrong filesystem type - #[error("{role:?} path {path} is not a {expected}")] - PathTypeMismatch { - /// Path role being checked - role: DirectSegmentPathRole, - /// Path whose type was checked - path: PathBuf, - /// Required filesystem type - expected: &'static str, - }, - /// The maintained trainer files or their `ops` directory are missing or unsafe - #[error("maintained trainer path is missing or unsafe: {path}")] - TrainerLayoutInvalid { - /// Trainer path that failed the layout check - path: PathBuf, - }, - /// The task row has no absolute resolved executable path - #[error("resolved task executable is not absolute: {path}")] - ExecutableNotAbsolute { - /// Executable path saved on the task row - path: PathBuf, - }, - /// The command's program cannot be resolved with the accepted task environment - #[error("cannot resolve accepted task executable: {0}")] - ExecutableResolution(#[source] crate::error::AppError), - /// The resolved command program differs from the executable saved on the task row - #[error("command resolves to {resolved}, but task row records {accepted}")] - ExecutableMismatch { - /// Executable resolved from the command, environment, and working directory - resolved: PathBuf, - /// Executable saved on the task row - accepted: PathBuf, - }, - /// The task row executable does not have a Python executable name - #[error("resolved executable is not a Python executable: {path}")] - NotPythonExecutable { - /// Executable path saved on the task row - path: PathBuf, - }, - /// Python argv does not have the exact module and `run` prefix - #[error("Python argv position {position} must be {expected:?}, found {found:?}")] - InvalidModuleInvocation { - /// Position in argv after the executable - position: usize, - /// Required argument at that position - expected: &'static str, - /// Supplied argument at that position, when present - found: Option, - }, - /// A token is a positional argument or an unsupported short option - #[error("unexpected positional or short-option argument: {value}")] - UnexpectedArgument { - /// Unsupported argv token - value: String, - }, - /// A command flag is not one of the exact maintained trainer flags - #[error("unknown direct-segment flag: {flag}")] - UnknownFlag { - /// Unknown flag spelling - flag: String, - }, - /// One exact flag appears more than once - #[error("direct-segment flag appears more than once: {flag}")] - DuplicateFlag { - /// Repeated flag spelling - flag: &'static str, - }, - /// A flag has no value or has an empty inline value - #[error("direct-segment flag has no value: {flag}")] - MissingFlagValue { - /// Flag that requires a value - flag: &'static str, - }, - /// A flag value starts with a dash and could be parsed as another option - #[error("ambiguous value for direct-segment flag {flag}: {value}")] - AmbiguousFlagValue { - /// Flag whose value is ambiguous - flag: &'static str, - /// Ambiguous value - value: String, - }, - /// A required direct-segment flag is absent - #[error("required direct-segment flag is missing: {flag}")] - MissingRequiredFlag { - /// Missing flag spelling - flag: &'static str, - }, - /// The command runtime root differs from the verified attempt's canonical runtime root - #[error("runtime root {provided} does not match verified attempt root {expected}")] - RuntimeRootMismatch { - /// Runtime path passed in command argv - provided: PathBuf, - /// Canonical runtime path saved in the verified attempt - expected: PathBuf, - }, - /// The timeout is not a finite positive value within the trainer's seven-day limit - #[error("invalid direct-segment timeout in seconds: {value}")] - InvalidTimeout { - /// Raw timeout argument - value: String, - }, -} - -fn accepted_working_directory( - spec: &NormalizedSpec, - task: &TaskRow, -) -> Result { - let accepted = if spec.cwd.is_absolute() { - spec.cwd.clone() - } else if spec.machine.is_some() && spec.cwd.to_str().is_some_and(|path| path.starts_with("~/")) - { - let home = PathBuf::from(&task.env.home); - if !home.is_absolute() { - return Err(DirectSegmentCommandShapeError::PathNotAbsolute { - role: DirectSegmentPathRole::WorkingDirectory, - path: home, - }); - } - let relative = spec.cwd.strip_prefix(Path::new("~/")).map_err(|_| { - DirectSegmentCommandShapeError::PathNotAbsolute { - role: DirectSegmentPathRole::WorkingDirectory, - path: spec.cwd.clone(), - } - })?; - if relative.is_absolute() - || relative - .components() - .any(|component| matches!(component, std::path::Component::ParentDir)) - { - return Err(DirectSegmentCommandShapeError::PathNotAbsolute { - role: DirectSegmentPathRole::WorkingDirectory, - path: spec.cwd.clone(), - }); - } - home.join(relative) - } else { - return Err(DirectSegmentCommandShapeError::PathNotAbsolute { - role: DirectSegmentPathRole::WorkingDirectory, - path: spec.cwd.clone(), - }); - }; - - if accepted.as_os_str() != task.cwd.as_os_str() { - return Err(DirectSegmentCommandShapeError::WorkingDirectoryMismatch { - accepted, - row: task.cwd.clone(), - }); - } - - canonical_directory(DirectSegmentPathRole::WorkingDirectory, &task.cwd) -} - -fn check_trainer_layout(cwd: &Path) -> Result<(), DirectSegmentCommandShapeError> { - let ops = cwd.join("ops"); - let ops_metadata = fs::symlink_metadata(&ops) - .map_err(|_| DirectSegmentCommandShapeError::TrainerLayoutInvalid { path: ops.clone() })?; - if !ops_metadata.file_type().is_dir() { - return Err(DirectSegmentCommandShapeError::TrainerLayoutInvalid { path: ops }); - } - - for file in ["run_segment.py", "segment_artifacts.py"] { - let path = cwd.join("ops").join(file); - let metadata = fs::symlink_metadata(&path).map_err(|_| { - DirectSegmentCommandShapeError::TrainerLayoutInvalid { path: path.clone() } - })?; - if !metadata.file_type().is_file() { - return Err(DirectSegmentCommandShapeError::TrainerLayoutInvalid { path }); - } - } - - Ok(()) -} - -fn canonical_directory( - role: DirectSegmentPathRole, - path: &Path, -) -> Result { - let canonical = canonical_path(role, path)?; - let metadata = fs::symlink_metadata(path).map_err(|source| { - DirectSegmentCommandShapeError::PathInspection { - role, - path: path.to_path_buf(), - source, - } - })?; - if !metadata.file_type().is_dir() { - return Err(DirectSegmentCommandShapeError::PathTypeMismatch { - role, - path: path.to_path_buf(), - expected: "directory", - }); - } - Ok(canonical) -} - -fn canonical_file( - role: DirectSegmentPathRole, - path: &Path, -) -> Result { - let canonical = canonical_path(role, path)?; - let metadata = fs::symlink_metadata(path).map_err(|source| { - DirectSegmentCommandShapeError::PathInspection { - role, - path: path.to_path_buf(), - source, - } - })?; - if !metadata.file_type().is_file() { - return Err(DirectSegmentCommandShapeError::PathTypeMismatch { - role, - path: path.to_path_buf(), - expected: "regular file", - }); - } - Ok(canonical) -} - -fn canonical_path( - role: DirectSegmentPathRole, - path: &Path, -) -> Result { - if !path.is_absolute() { - return Err(DirectSegmentCommandShapeError::PathNotAbsolute { - role, - path: path.to_path_buf(), - }); - } - let canonical = fs::canonicalize(path).map_err(|source| { - DirectSegmentCommandShapeError::PathInspection { - role, - path: path.to_path_buf(), - source, - } - })?; - if canonical.as_os_str() != path.as_os_str() { - return Err(DirectSegmentCommandShapeError::PathNotCanonical { - role, - path: path.to_path_buf(), - canonical, - }); - } - Ok(canonical) -} - -fn is_python_executable(path: &Path) -> bool { - let Some(name) = path.file_name().and_then(|name| name.to_str()) else { - return false; - }; - if name == "python" { - return true; - } - - let Some(version) = name.strip_prefix("python") else { - return false; - }; - let mut has_digit = false; - let mut has_digit_after_dot = true; - for character in version.chars() { - match character { - '0'..='9' => { - has_digit = true; - has_digit_after_dot = true; - } - '.' if has_digit && has_digit_after_dot => has_digit_after_dot = false, - _ => return false, - } - } - has_digit && has_digit_after_dot -} - -fn check_module_invocation(command: &CommandLine) -> Result<(), DirectSegmentCommandShapeError> { - for (position, expected) in REQUIRED_MODULE_ARGS.iter().enumerate() { - let found = command.args().get(position).cloned(); - if found.as_deref() != Some(expected) { - return Err(DirectSegmentCommandShapeError::InvalidModuleInvocation { - position, - expected, - found, - }); - } - } - Ok(()) -} - -#[derive(Debug, Clone, Copy)] -enum DirectSegmentFlag { - TaskFile, - InputRoot, - RuntimeRoot, - ImageArgument, - InputVerificationReceipt, - ResumeGeneration, - TimeoutSeconds, -} - -impl DirectSegmentFlag { - const fn name(self) -> &'static str { - match self { - Self::TaskFile => "--task", - Self::InputRoot => "--input-root", - Self::RuntimeRoot => "--runtime-root", - Self::ImageArgument => "--image-digest", - Self::InputVerificationReceipt => "--input-verification-receipt", - Self::ResumeGeneration => "--resume", - Self::TimeoutSeconds => "--timeout-seconds", - } - } - - fn parse(name: &str) -> Option { - match name { - "--task" => Some(Self::TaskFile), - "--input-root" => Some(Self::InputRoot), - "--runtime-root" => Some(Self::RuntimeRoot), - "--image-digest" => Some(Self::ImageArgument), - "--input-verification-receipt" => Some(Self::InputVerificationReceipt), - "--resume" => Some(Self::ResumeGeneration), - "--timeout-seconds" => Some(Self::TimeoutSeconds), - _ => None, - } - } -} - -#[derive(Default)] -struct ParsedFlags { - task_file: Option, - input_root: Option, - runtime_root: Option, - image_argument: Option, - input_verification_receipt: Option, - resume_generation: Option, - timeout_seconds: Option, -} - -impl ParsedFlags { - fn contains(&self, flag: DirectSegmentFlag) -> bool { - self.value(flag).is_some() - } - - fn value(&self, flag: DirectSegmentFlag) -> Option<&String> { - match flag { - DirectSegmentFlag::TaskFile => self.task_file.as_ref(), - DirectSegmentFlag::InputRoot => self.input_root.as_ref(), - DirectSegmentFlag::RuntimeRoot => self.runtime_root.as_ref(), - DirectSegmentFlag::ImageArgument => self.image_argument.as_ref(), - DirectSegmentFlag::InputVerificationReceipt => self.input_verification_receipt.as_ref(), - DirectSegmentFlag::ResumeGeneration => self.resume_generation.as_ref(), - DirectSegmentFlag::TimeoutSeconds => self.timeout_seconds.as_ref(), - } - } - - fn set(&mut self, flag: DirectSegmentFlag, value: String) { - match flag { - DirectSegmentFlag::TaskFile => self.task_file = Some(value), - DirectSegmentFlag::InputRoot => self.input_root = Some(value), - DirectSegmentFlag::RuntimeRoot => self.runtime_root = Some(value), - DirectSegmentFlag::ImageArgument => self.image_argument = Some(value), - DirectSegmentFlag::InputVerificationReceipt => { - self.input_verification_receipt = Some(value); - } - DirectSegmentFlag::ResumeGeneration => self.resume_generation = Some(value), - DirectSegmentFlag::TimeoutSeconds => self.timeout_seconds = Some(value), - } - } -} - -fn parse_flags(command: &CommandLine) -> Result { - let args = command.args(); - let mut parsed = ParsedFlags::default(); - let mut index = REQUIRED_MODULE_ARGS.len(); - - while index < args.len() { - let argument = &args[index]; - let Some(option) = argument.strip_prefix("--") else { - return Err(DirectSegmentCommandShapeError::UnexpectedArgument { - value: argument.clone(), - }); - }; - let (name, inline_value) = option - .split_once('=') - .map_or((option, None), |(name, value)| (name, Some(value))); - let flag = DirectSegmentFlag::parse(&format!("--{name}")).ok_or_else(|| { - DirectSegmentCommandShapeError::UnknownFlag { - flag: if name.is_empty() { - "--".into() - } else { - format!("--{name}") - }, - } - })?; - - if parsed.contains(flag) { - return Err(DirectSegmentCommandShapeError::DuplicateFlag { flag: flag.name() }); - } - - let value = match inline_value { - Some("") => { - return Err(DirectSegmentCommandShapeError::MissingFlagValue { flag: flag.name() }); - } - Some(value) => value.to_owned(), - None => { - let Some(value) = args.get(index + 1) else { - return Err(DirectSegmentCommandShapeError::MissingFlagValue { - flag: flag.name(), - }); - }; - if value.starts_with('-') { - return Err(DirectSegmentCommandShapeError::AmbiguousFlagValue { - flag: flag.name(), - value: value.clone(), - }); - } - index += 1; - value.clone() - } - }; - if value.is_empty() { - return Err(DirectSegmentCommandShapeError::MissingFlagValue { flag: flag.name() }); - } - if value.starts_with('-') { - return Err(DirectSegmentCommandShapeError::AmbiguousFlagValue { - flag: flag.name(), - value, - }); - } - - parsed.set(flag, value); - index += 1; - } - - Ok(parsed) -} - -/// Return the `--runtime-root` value of one maintained direct-segment command -/// -/// The argv must have the module invocation and valid flags. Path and shape -/// checks against a task row stay with the caller -pub(crate) fn direct_segment_runtime_root(command: &CommandLine) -> Option { - check_module_invocation(command).ok()?; - parse_flags(command).ok()?.runtime_root.map(PathBuf::from) -} - -/// Build the same-run resume argv from one saved direct-segment command -/// -/// The saved command must have the maintained module invocation and all required -/// flags. The result keeps the program and every saved flag value, and sets only -/// `--resume` to the selected generation. Path checks and the choice to resume -/// stay with the caller -pub(crate) fn same_run_resume_command( - command: &CommandLine, - generation: &str, -) -> Result { - check_module_invocation(command)?; - let mut parsed = parse_flags(command)?; - for flag in [ - DirectSegmentFlag::TaskFile, - DirectSegmentFlag::InputRoot, - DirectSegmentFlag::RuntimeRoot, - DirectSegmentFlag::ImageArgument, - ] { - required_flag(parsed.value(flag).cloned(), flag.name())?; - } - let resume = DirectSegmentFlag::ResumeGeneration.name(); - if generation.is_empty() { - return Err(DirectSegmentCommandShapeError::MissingFlagValue { flag: resume }); - } - if generation.starts_with('-') { - return Err(DirectSegmentCommandShapeError::AmbiguousFlagValue { - flag: resume, - value: generation.to_owned(), - }); - } - parsed.set(DirectSegmentFlag::ResumeGeneration, generation.to_owned()); - - let mut argv = vec![command.program().to_owned()]; - argv.extend( - REQUIRED_MODULE_ARGS - .iter() - .map(|argument| (*argument).to_owned()), - ); - for flag in [ - DirectSegmentFlag::TaskFile, - DirectSegmentFlag::InputRoot, - DirectSegmentFlag::RuntimeRoot, - DirectSegmentFlag::ImageArgument, - DirectSegmentFlag::InputVerificationReceipt, - DirectSegmentFlag::ResumeGeneration, - DirectSegmentFlag::TimeoutSeconds, - ] { - if let Some(value) = parsed.value(flag) { - argv.push(flag.name().to_owned()); - argv.push(value.clone()); - } - } - - CommandLine::try_from_argv(argv).map_err(|_| { - DirectSegmentCommandShapeError::UnexpectedArgument { - value: generation.to_owned(), - } - }) -} - -fn required_flag( - value: Option, - flag: &'static str, -) -> Result { - value.ok_or(DirectSegmentCommandShapeError::MissingRequiredFlag { flag }) -} - -fn parse_timeout(value: String) -> Result { - let seconds = value - .parse::() - .ok() - .filter(|seconds| seconds.is_finite() && 0.0 < *seconds && *seconds <= MAX_TIMEOUT_SECONDS) - .ok_or_else(|| DirectSegmentCommandShapeError::InvalidTimeout { - value: value.clone(), - })?; - Duration::try_from_secs_f64(seconds) - .map_err(|_| DirectSegmentCommandShapeError::InvalidTimeout { value }) -} - -#[cfg(test)] -pub(crate) mod test_support { - use std::fs; - use std::os::unix::fs::PermissionsExt; - use std::path::{Path, PathBuf}; - - use crate::domain::{TaskEnv, ThreadId}; - use crate::spec::NormalizedSpec; - - /// Fake maintained trainer layout whose `python3` is a gated shell script - /// - /// The script appends one byte to `marker` for every start and exits only - /// after `gate` exists, so a test can count spawns without a real GPU - pub(crate) struct FakeTrainer { - /// Canonical trainer working directory - pub(crate) cwd: PathBuf, - /// Runtime root passed with `--runtime-root` - pub(crate) runtime_root: PathBuf, - /// File that records each start - pub(crate) marker: PathBuf, - /// File whose creation lets the fake trainer exit - pub(crate) gate: PathBuf, - /// Executor environment whose PATH resolves the fake `python3` first - pub(crate) env: TaskEnv, - task_file: PathBuf, - input_root: PathBuf, - } - - impl FakeTrainer { - /// Build the layout under an existing canonical directory - pub(crate) fn new(root: &Path) -> Self { - let cwd = root.join("trainer"); - fs::create_dir_all(cwd.join("ops")).unwrap(); - fs::write(cwd.join("ops/run_segment.py"), b"# maintained runner\n").unwrap(); - fs::write(cwd.join("ops/segment_artifacts.py"), b"# artifacts\n").unwrap(); - let runtime_root = root.join("runtime"); - fs::create_dir_all(&runtime_root).unwrap(); - let task_file = root.join("task.json"); - fs::write(&task_file, b"{}\n").unwrap(); - let input_root = root.join("inputs"); - fs::create_dir_all(&input_root).unwrap(); - let (marker, gate) = (root.join("trainer-starts"), root.join("trainer-gate")); - let bin = root.join("bin"); - fs::create_dir_all(&bin).unwrap(); - let python = bin.join("python3"); - fs::write( - &python, - format!( - "#!/bin/sh\nprintf x >> '{}'\nwhile [ ! -e '{}' ]; do sleep 0.02; done\n", - marker.display(), - gate.display() - ), - ) - .unwrap(); - fs::set_permissions(&python, fs::Permissions::from_mode(0o755)).unwrap(); - - Self { - cwd, - runtime_root, - marker, - gate, - env: TaskEnv { - path: format!("{}:/bin:/usr/bin", bin.display()), - home: root.to_string_lossy().into_owned(), - }, - task_file, - input_root, - } - } - - /// Normalized trainer spec whose callbacks go to `thread` - pub(crate) fn spec(&self, thread: ThreadId, name: &str) -> NormalizedSpec { - serde_json::from_value(serde_json::json!({ - "api_version": 1, - "thread": thread, - "name": name, - "cwd": self.cwd, - "timeout": "4h", - "workload": { - "type": "task", - "command": [ - "python3", "-m", "ops.run_segment", "run", - "--task", self.task_file, - "--input-root", self.input_root, - "--runtime-root", self.runtime_root, - "--image-digest", "not-validated-by-shape" - ] - } - })) - .unwrap() - } - - /// Number of times the fake trainer started - pub(crate) fn starts(&self) -> usize { - fs::read(&self.marker).map_or(0, |bytes| bytes.len()) - } - } -} - -#[cfg(test)] -mod tests { - use std::os::unix::fs::OpenOptionsExt; - use std::path::PathBuf; - - use super::{ - DEFAULT_TIMEOUT, DirectSegmentCommandShape, DirectSegmentCommandShapeError, - DirectSegmentPathRole, same_run_resume_command, - }; - use crate::domain::{AgentKind, TaskEnv, TaskId, TaskName, TaskRow, TaskWorkload, Workload}; - use crate::invocation::CommandLine; - use crate::machine::{MachineId, MachineName}; - use crate::resource::ownership_lock::{ - OwnershipLockIdentity, TrainerRequestDigest, VerifiedTrainerAttempt, - }; - use crate::resource::trainer_publication::AttemptBinding; - use crate::resource::{ResourceId, TrainerAttemptAssociation}; - use crate::spec::{NormalizedSpec, NormalizedWorkload}; - use crate::store::{NewTask, new_queued_task}; - use crate::submission::normalized_spec_sha256; - use std::fs; - use std::path::Path; - use std::time::Duration; - use tempfile::{TempDir, tempdir}; - use uuid::Uuid; - - struct Fixture { - _directory: TempDir, - root: PathBuf, - spec: NormalizedSpec, - task: TaskRow, - association: TrainerAttemptAssociation, - trainer_root: PathBuf, - runtime_root: PathBuf, - task_file: PathBuf, - input_root: PathBuf, - python: PathBuf, - } - - impl Fixture { - fn new() -> Self { - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let trainer_root = root.join("trainer"); - let ops = trainer_root.join("ops"); - fs::create_dir_all(&ops).unwrap(); - fs::write(ops.join("run_segment.py"), b"# maintained runner\n").unwrap(); - fs::write( - ops.join("segment_artifacts.py"), - b"# maintained artifacts\n", - ) - .unwrap(); - - let runtime_root = root.join("runtime"); - fs::create_dir(&runtime_root).unwrap(); - let task_file = root.join("task.json"); - fs::write(&task_file, b"{}\n").unwrap(); - let input_root = root.join("inputs"); - fs::create_dir(&input_root).unwrap(); - - let bin = root.join("bin"); - fs::create_dir(&bin).unwrap(); - let python = bin.join("python3"); - fs::OpenOptions::new() - .write(true) - .create_new(true) - .mode(0o755) - .open(&python) - .unwrap(); - - let command = vec![ - "python3".into(), - "-m".into(), - "ops.run_segment".into(), - "run".into(), - "--task".into(), - task_file.to_string_lossy().into_owned(), - "--input-root".into(), - input_root.to_string_lossy().into_owned(), - "--runtime-root".into(), - runtime_root.to_string_lossy().into_owned(), - "--image-digest".into(), - "not-validated-by-shape".into(), - ]; - let spec: NormalizedSpec = serde_json::from_value(serde_json::json!({ - "api_version": 1, - "thread": Uuid::now_v7(), - "name": "direct segment trainer", - "cwd": trainer_root, - "machine": null, - "timeout": "4h", - "workload": { "type": "task", "command": command } - })) - .unwrap(); - let task_id = TaskId::new(); - let row = task_row(task_id, &spec, &trainer_root, &python, &bin); - let association = association(task_id, &spec, &runtime_root); - - Self { - _directory: directory, - root, - spec, - task: row, - association, - trainer_root, - runtime_root, - task_file, - input_root, - python, - } - } - - fn set_command(&mut self, command: Vec) { - let NormalizedWorkload::Task(workload) = &mut self.spec.workload else { - panic!("fixture workload must be a task"); - }; - workload.command = CommandLine::try_from_argv(command).unwrap(); - self.task.workload = crate::invocation::persist_workload(&self.spec.workload); - self.association = association(self.task.id, &self.spec, &self.runtime_root); - } - - fn argv(&self) -> Vec { - let NormalizedWorkload::Task(workload) = &self.spec.workload else { - panic!("fixture workload must be a task"); - }; - workload.command.to_vec() - } - - fn set_remote_cwd(&mut self) { - self.spec.cwd = PathBuf::from("~/trainer"); - self.spec.machine = Some(MachineName::parse("remote").unwrap()); - self.task.cwd = self.trainer_root.clone(); - self.task.env.home = self.root.to_string_lossy().into_owned(); - self.association = association(self.task.id, &self.spec, &self.runtime_root); - } - } - - fn task_row( - task_id: TaskId, - spec: &NormalizedSpec, - cwd: &Path, - python: &Path, - bin: &Path, - ) -> TaskRow { - let NormalizedWorkload::Task(workload) = &spec.workload else { - panic!("fixture workload must be a task"); - }; - new_queued_task(NewTask { - id: task_id, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: Workload::Task(TaskWorkload { - command: workload.command.clone(), - }), - cwd: cwd.to_path_buf(), - timeout: spec.timeout, - env: TaskEnv { - path: bin.to_string_lossy().into_owned(), - home: cwd.parent().unwrap().to_string_lossy().into_owned(), - }, - binary: python.to_path_buf(), - }) - } - - fn association( - task_id: TaskId, - spec: &NormalizedSpec, - runtime_root: &Path, - ) -> TrainerAttemptAssociation { - let digest = TrainerRequestDigest::from_hex(&"00".repeat(32)).unwrap(); - let verified = VerifiedTrainerAttempt::from_persisted( - runtime_root.to_path_buf(), - AttemptBinding { - campaign_id: "campaign-1".into(), - campaign_revision_id: "revision-1".into(), - task_id: "trainer-task-1".into(), - attempt_id: "attempt-1".into(), - attempt_number: 1, - ownership_token: "token-1".into(), - }, - digest, - OwnershipLockIdentity::new(1, 2), - ) - .unwrap(); - TrainerAttemptAssociation::from_components( - ResourceId::new(), - MachineId::new(), - task_id, - verified, - normalized_spec_sha256(spec).unwrap(), - ) - .unwrap() - } - - fn append_options(fixture: &Fixture, options: &[&str]) -> Vec { - let mut argv = fixture.argv(); - argv.extend(options.iter().map(|value| (*value).to_owned())); - argv - } - - #[test] - fn validates_local_accepted_task_shape_and_keeps_image_opaque() { - let fixture = Fixture::new(); - - let shape = - DirectSegmentCommandShape::validate(&fixture.spec, &fixture.task, &fixture.association) - .unwrap(); - - assert_eq!(shape.task_id(), fixture.task.id); - assert_eq!(shape.python_executable(), fixture.python); - assert_eq!(shape.working_directory(), fixture.trainer_root); - assert_eq!(shape.task_file(), fixture.task_file); - assert_eq!(shape.input_root(), fixture.input_root); - assert_eq!(shape.runtime_root(), fixture.runtime_root); - assert_eq!(shape.image_argument(), "not-validated-by-shape"); - assert_eq!(shape.timeout(), DEFAULT_TIMEOUT); - assert_eq!(shape.input_verification_receipt(), None); - assert_eq!(shape.resume_generation(), None); - } - - #[test] - fn validates_remote_accepted_task_with_executor_home_cwd_expansion() { - let mut fixture = Fixture::new(); - fixture.set_remote_cwd(); - - let shape = - DirectSegmentCommandShape::validate(&fixture.spec, &fixture.task, &fixture.association) - .unwrap(); - - assert_eq!(shape.working_directory(), fixture.trainer_root); - } - - #[test] - fn accepts_exact_flags_in_either_value_form_and_preserves_optional_values() { - let mut fixture = Fixture::new(); - let receipt = fixture.root.join("receipt.json"); - fs::write(&receipt, b"{}\n").unwrap(); - let mut argv = vec![ - "python3".into(), - "-m".into(), - "ops.run_segment".into(), - "run".into(), - format!("--task={}", fixture.task_file.display()), - "--input-root".into(), - fixture.input_root.to_string_lossy().into_owned(), - "--runtime-root".into(), - fixture.runtime_root.to_string_lossy().into_owned(), - "--image-digest=opaque".into(), - "--resume".into(), - "generation-1".into(), - "--input-verification-receipt".into(), - receipt.to_string_lossy().into_owned(), - "--timeout-seconds=120.5".into(), - ]; - argv.shrink_to_fit(); - fixture.set_command(argv); - - let shape = - DirectSegmentCommandShape::validate(&fixture.spec, &fixture.task, &fixture.association) - .unwrap(); - - assert_eq!(shape.image_argument(), "opaque"); - assert_eq!(shape.resume_generation(), Some("generation-1")); - assert_eq!(shape.input_verification_receipt(), Some(receipt.as_path())); - assert_eq!(shape.timeout(), Duration::from_millis(120_500)); - } - - #[test] - fn rejects_shell_and_wrapper_executables() { - let mut fixture = Fixture::new(); - let shell = fixture.root.join("bin/bash"); - fs::OpenOptions::new() - .write(true) - .create_new(true) - .mode(0o755) - .open(&shell) - .unwrap(); - let mut argv = fixture.argv(); - argv[0] = "bash".into(); - argv.splice( - 1..4, - ["-lc".into(), "python3 -m ops.run_segment run".into()], - ); - fixture.set_command(argv); - fixture.task.binary = shell; - - assert!(matches!( - DirectSegmentCommandShape::validate(&fixture.spec, &fixture.task, &fixture.association,), - Err(DirectSegmentCommandShapeError::NotPythonExecutable { .. }) - )); - } - - #[test] - fn rejects_python_c_invocation() { - let mut fixture = Fixture::new(); - let mut argv = fixture.argv(); - argv.splice(1.., ["-c".into(), "print(1)".into()]); - fixture.set_command(argv); - - assert!(matches!( - DirectSegmentCommandShape::validate( - &fixture.spec, - &fixture.task, - &fixture.association, - ), - Err(DirectSegmentCommandShapeError::InvalidModuleInvocation { - position: 0, - expected: "-m", - found: Some(found), - }) if found == "-c" - )); - } - - #[test] - fn rejects_import_subcommand() { - let mut fixture = Fixture::new(); - let mut argv = fixture.argv(); - argv[3] = "import".into(); - fixture.set_command(argv); - - assert!(matches!( - DirectSegmentCommandShape::validate( - &fixture.spec, - &fixture.task, - &fixture.association, - ), - Err(DirectSegmentCommandShapeError::InvalidModuleInvocation { - position: 2, - expected: "run", - found: Some(found), - }) if found == "import" - )); - } - - #[test] - fn rejects_unknown_duplicate_missing_and_positional_arguments() { - let mut unknown = Fixture::new(); - let argv = append_options(&unknown, &["--unknown", "value"]); - unknown.set_command(argv); - assert!(matches!( - DirectSegmentCommandShape::validate(&unknown.spec, &unknown.task, &unknown.association,), - Err(DirectSegmentCommandShapeError::UnknownFlag { .. }) - )); - - let mut duplicate = Fixture::new(); - let argv = append_options(&duplicate, &["--task", "again"]); - duplicate.set_command(argv); - assert!(matches!( - DirectSegmentCommandShape::validate( - &duplicate.spec, - &duplicate.task, - &duplicate.association, - ), - Err(DirectSegmentCommandShapeError::DuplicateFlag { flag: "--task" }) - )); - - let mut missing = Fixture::new(); - let mut argv = missing.argv(); - let runtime_index = argv.iter().position(|arg| arg == "--runtime-root").unwrap(); - argv.drain(runtime_index..=runtime_index + 1); - missing.set_command(argv); - assert!(matches!( - DirectSegmentCommandShape::validate(&missing.spec, &missing.task, &missing.association,), - Err(DirectSegmentCommandShapeError::MissingRequiredFlag { - flag: "--runtime-root" - }) - )); - - let mut missing_value = Fixture::new(); - let argv = append_options(&missing_value, &["--resume"]); - missing_value.set_command(argv); - assert!(matches!( - DirectSegmentCommandShape::validate( - &missing_value.spec, - &missing_value.task, - &missing_value.association, - ), - Err(DirectSegmentCommandShapeError::MissingFlagValue { flag: "--resume" }) - )); - - let mut positional = Fixture::new(); - let argv = append_options(&positional, &["free-argument"]); - positional.set_command(argv); - assert!(matches!( - DirectSegmentCommandShape::validate( - &positional.spec, - &positional.task, - &positional.association, - ), - Err(DirectSegmentCommandShapeError::UnexpectedArgument { .. }) - )); - } - - #[test] - fn rejects_ambiguous_flag_values() { - let mut fixture = Fixture::new(); - let argv = append_options(&fixture, &["--resume", "--timeout-seconds", "4"]); - fixture.set_command(argv); - - assert!(matches!( - DirectSegmentCommandShape::validate(&fixture.spec, &fixture.task, &fixture.association,), - Err(DirectSegmentCommandShapeError::AmbiguousFlagValue { - flag: "--resume", - .. - }) - )); - } - - #[test] - fn rejects_runtime_root_mismatch() { - let mut fixture = Fixture::new(); - let other_runtime = fixture.root.join("other-runtime"); - fs::create_dir(&other_runtime).unwrap(); - let mut argv = fixture.argv(); - let runtime_index = argv.iter().position(|arg| arg == "--runtime-root").unwrap(); - argv[runtime_index + 1] = other_runtime.to_string_lossy().into_owned(); - fixture.set_command(argv); - - assert!(matches!( - DirectSegmentCommandShape::validate(&fixture.spec, &fixture.task, &fixture.association,), - Err(DirectSegmentCommandShapeError::RuntimeRootMismatch { .. }) - )); - - let mut noncanonical = Fixture::new(); - let mut argv = noncanonical.argv(); - let runtime_index = argv.iter().position(|arg| arg == "--runtime-root").unwrap(); - argv[runtime_index + 1] = format!("{}/", noncanonical.runtime_root.display()); - noncanonical.set_command(argv); - assert!(matches!( - DirectSegmentCommandShape::validate( - &noncanonical.spec, - &noncanonical.task, - &noncanonical.association, - ), - Err(DirectSegmentCommandShapeError::RuntimeRootMismatch { .. }) - )); - } - - #[test] - fn rejects_relative_and_noncanonical_required_paths() { - let mut relative = Fixture::new(); - let mut argv = relative.argv(); - let task_index = argv.iter().position(|arg| arg == "--task").unwrap(); - argv[task_index + 1] = "task.json".into(); - relative.set_command(argv); - assert!(matches!( - DirectSegmentCommandShape::validate( - &relative.spec, - &relative.task, - &relative.association, - ), - Err(DirectSegmentCommandShapeError::PathNotAbsolute { - role: DirectSegmentPathRole::TaskFile, - .. - }) - )); - - let mut noncanonical = Fixture::new(); - fs::create_dir(noncanonical.root.join("nested")).unwrap(); - let aliased_task = noncanonical.root.join("nested/../task.json"); - let mut argv = noncanonical.argv(); - let task_index = argv.iter().position(|arg| arg == "--task").unwrap(); - argv[task_index + 1] = aliased_task.to_string_lossy().into_owned(); - noncanonical.set_command(argv); - assert!(matches!( - DirectSegmentCommandShape::validate( - &noncanonical.spec, - &noncanonical.task, - &noncanonical.association, - ), - Err(DirectSegmentCommandShapeError::PathNotCanonical { - role: DirectSegmentPathRole::TaskFile, - .. - }) - )); - - let mut relative_input = Fixture::new(); - let mut argv = relative_input.argv(); - let input_index = argv.iter().position(|arg| arg == "--input-root").unwrap(); - argv[input_index + 1] = "inputs".into(); - relative_input.set_command(argv); - assert!(matches!( - DirectSegmentCommandShape::validate( - &relative_input.spec, - &relative_input.task, - &relative_input.association, - ), - Err(DirectSegmentCommandShapeError::PathNotAbsolute { - role: DirectSegmentPathRole::InputRoot, - .. - }) - )); - - let mut noncanonical_input = Fixture::new(); - fs::create_dir(noncanonical_input.root.join("nested")).unwrap(); - let aliased_input = noncanonical_input.root.join("nested/../inputs"); - let mut argv = noncanonical_input.argv(); - let input_index = argv.iter().position(|arg| arg == "--input-root").unwrap(); - argv[input_index + 1] = aliased_input.to_string_lossy().into_owned(); - noncanonical_input.set_command(argv); - assert!(matches!( - DirectSegmentCommandShape::validate( - &noncanonical_input.spec, - &noncanonical_input.task, - &noncanonical_input.association, - ), - Err(DirectSegmentCommandShapeError::PathNotCanonical { - role: DirectSegmentPathRole::InputRoot, - .. - }) - )); - } - - #[test] - fn rejects_missing_trainer_file_and_symlinked_final_component() { - let missing = Fixture::new(); - fs::remove_file(missing.trainer_root.join("ops/segment_artifacts.py")).unwrap(); - assert!(matches!( - DirectSegmentCommandShape::validate(&missing.spec, &missing.task, &missing.association,), - Err(DirectSegmentCommandShapeError::TrainerLayoutInvalid { .. }) - )); - - let symlinked = Fixture::new(); - let original = symlinked.trainer_root.join("ops/run_segment.py"); - fs::remove_file(&original).unwrap(); - std::os::unix::fs::symlink("segment_artifacts.py", &original).unwrap(); - assert!(matches!( - DirectSegmentCommandShape::validate( - &symlinked.spec, - &symlinked.task, - &symlinked.association, - ), - Err(DirectSegmentCommandShapeError::TrainerLayoutInvalid { path }) - if path == original - )); - } - - #[test] - fn rejects_non_python_resolved_executable_and_bad_timeout() { - let mut wrapper = Fixture::new(); - let wrapper_path = wrapper.root.join("bin/env"); - fs::OpenOptions::new() - .write(true) - .create_new(true) - .mode(0o755) - .open(&wrapper_path) - .unwrap(); - let mut argv = wrapper.argv(); - argv[0] = "env".into(); - wrapper.set_command(argv); - wrapper.task.binary = wrapper_path; - assert!(matches!( - DirectSegmentCommandShape::validate(&wrapper.spec, &wrapper.task, &wrapper.association,), - Err(DirectSegmentCommandShapeError::NotPythonExecutable { .. }) - )); - - let mut bad_timeout = Fixture::new(); - let argv = append_options(&bad_timeout, &["--timeout-seconds", "604800.5"]); - bad_timeout.set_command(argv); - assert!(matches!( - DirectSegmentCommandShape::validate( - &bad_timeout.spec, - &bad_timeout.task, - &bad_timeout.association, - ), - Err(DirectSegmentCommandShapeError::InvalidTimeout { .. }) - )); - } - - #[test] - fn rejects_changed_normalized_spec_digest_and_task_row_identity() { - let mut fixture = Fixture::new(); - fixture.spec.name = TaskName::parse("changed task").unwrap(); - assert!(matches!( - DirectSegmentCommandShape::validate(&fixture.spec, &fixture.task, &fixture.association,), - Err(DirectSegmentCommandShapeError::NormalizedSpecDigestMismatch { .. }) - )); - - let fixture = Fixture::new(); - let other_task = task_row( - TaskId::new(), - &fixture.spec, - &fixture.trainer_root, - &fixture.python, - &fixture.root.join("bin"), - ); - assert!(matches!( - DirectSegmentCommandShape::validate(&fixture.spec, &other_task, &fixture.association,), - Err(DirectSegmentCommandShapeError::TaskIdentityMismatch { .. }) - )); - } - - #[test] - fn rejects_non_command_workloads() { - let mut fixture = Fixture::new(); - fixture.spec.workload = NormalizedWorkload::Agent(crate::spec::NormalizedAgentWorkload { - agent: AgentKind::Codex, - model: None, - prompt: "not a task command".into(), - extra_args: Vec::new(), - report_trailer: true, - resume_thread: None, - }); - fixture.association = association(fixture.task.id, &fixture.spec, &fixture.runtime_root); - - assert!(matches!( - DirectSegmentCommandShape::validate(&fixture.spec, &fixture.task, &fixture.association,), - Err(DirectSegmentCommandShapeError::NotCommandTask { .. }) - )); - } - - #[test] - fn rejects_explicit_python_script_and_import_wrappers() { - let mut fixture = Fixture::new(); - let mut argv = fixture.argv(); - argv.splice(1..4, ["ops/run_segment.py".into(), "run".into()]); - fixture.set_command(argv); - assert!(matches!( - DirectSegmentCommandShape::validate(&fixture.spec, &fixture.task, &fixture.association,), - Err(DirectSegmentCommandShapeError::InvalidModuleInvocation { .. }) - )); - } - - #[test] - fn same_run_resume_keeps_saved_flags_and_sets_only_the_selected_generation() { - let fixture = Fixture::new(); - let mut argv = fixture.argv(); - argv.extend(["--resume".into(), "generation-0".into()]); - let saved = CommandLine::try_from_argv(argv).unwrap(); - - let resumed = same_run_resume_command(&saved, "generation-9").unwrap(); - let mut expected = fixture.argv(); - expected.extend(["--resume".into(), "generation-9".into()]); - assert_eq!(resumed.to_vec(), expected); - - let mut wrapped = fixture.argv(); - wrapped.splice(1..4, ["ops/run_segment.py".into(), "run".into()]); - assert!(matches!( - same_run_resume_command( - &CommandLine::try_from_argv(wrapped).unwrap(), - "generation-9" - ), - Err(DirectSegmentCommandShapeError::InvalidModuleInvocation { .. }) - )); - assert!(matches!( - same_run_resume_command(&saved, "--task"), - Err(DirectSegmentCommandShapeError::AmbiguousFlagValue { .. }) - )); - } -} diff --git a/src/resource/foreground.rs b/src/resource/foreground.rs deleted file mode 100644 index 059499a..0000000 --- a/src/resource/foreground.rs +++ /dev/null @@ -1,596 +0,0 @@ -//! Ownership contract for resource work -//! -//! After resource work ends, the authority can release the GPU only from -//! task-layer evidence that the work stopped. Process-group evidence covers GPU -//! work only while the work stays in that process group, so a command must name -//! its real workload directly. The contract accepts one of three shapes: -//! -//! - A native foreground executable. Its entry point is an ELF or Mach-O file, -//! and neither its name nor its resolved target names a known shell, -//! interpreter, program launcher, detach tool, container client, or remote -//! client. Those programs hide the real workload in their arguments or in code -//! that Homebased does not inspect, or they can move it out of the group -//! - The maintained direct-segment trainer, only for background return work -//! Its ownership lock, not its process group, is the release witness -//! - A typed container workload. Homebased starts and removes the container -//! itself, so the container's own state is the witness. A `docker` command -//! line is still refused, because its client can exit while the container runs -//! -//! The rule is conservative, not complete. A native executable can still start -//! work in another session, and a copied or hard-linked launcher keeps no -//! recognizable name. The contract does not prove physical GPU exclusion - -use std::fs::File; -use std::io::Read; -use std::path::Path; - -use crate::invocation::CommandLine; -use crate::resource::ResourceTaskOwnershipRisk; -use crate::resource::ResourceTaskOwnershipRisk::{ - ContainerClient, DetachedLauncher, Interpreter, ProgramLauncher, RemoteShell, ShellWrapper, -}; -use crate::resource::command_shape::direct_segment_runtime_root; -use crate::spec::{NormalizedSpec, NormalizedWorkload}; - -/// Release witness supported by one accepted command shape -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum CommandOwnershipContract { - /// Native foreground executable; the task-run process-group exit is the witness - ForegroundExecutable, - /// Maintained direct-segment trainer; its segment ownership lock is the witness - /// - /// The static argv shape only selects this contract. The launch must still - /// pass the full direct-segment check against its task row - DirectSegmentTrainer, - /// Typed container workload; the exited, removed container is the witness - Container, -} - -impl CommandOwnershipContract { - /// Classify queued work: a foreground command or a container - pub fn for_queued_work( - workload: &NormalizedWorkload, - ) -> Result { - match workload { - NormalizedWorkload::Task(task) => { - foreground_argv(&task.command).map(|()| Self::ForegroundExecutable) - } - NormalizedWorkload::Container(_) => Ok(Self::Container), - NormalizedWorkload::Agent(_) => Err(ResourceTaskOwnershipRisk::UninspectableEntryPoint), - } - } - - /// Classify return work, which may also be the maintained trainer - pub fn for_return_work( - workload: &NormalizedWorkload, - ) -> Result { - if let NormalizedWorkload::Task(task) = workload - && direct_segment_runtime_root(&task.command).is_some() - { - return Ok(Self::DirectSegmentTrainer); - } - Self::for_queued_work(workload) - } -} - -/// Return the command line of a task workload, or `None` for other workloads -#[must_use] -pub fn task_command(spec: &NormalizedSpec) -> Option<&CommandLine> { - match &spec.workload { - NormalizedWorkload::Task(task) => Some(&task.command), - NormalizedWorkload::Agent(_) | NormalizedWorkload::Container(_) => None, - } -} - -/// Check the argv of a foreground command without reading the file system -/// -/// The program must not be a known launcher of other work. A later argument -/// must not name a shell, interpreter, detach tool, container client, or remote -/// client, because an unrecognized launcher can run it -pub fn foreground_argv(command: &CommandLine) -> Result<(), ResourceTaskOwnershipRisk> { - if let Some(risk) = program_risk(command.program()) { - return Err(risk); - } - match command - .args() - .iter() - .find_map(|arg| nested_program_risk(arg)) - { - Some(risk) => Err(risk), - None => Ok(()), - } -} - -/// Inspect the entry point of a queued command when its program names a path -/// -/// A bare program name resolves from the executor `PATH` only when its task binds, -/// so [`inspect_foreground_entry_point`] checks it then. A container has no -/// entry point on the host; its host inputs are checked instead -pub fn inspect_path_qualified_entry_point( - spec: &NormalizedSpec, -) -> Result<(), ResourceTaskOwnershipRisk> { - if let NormalizedWorkload::Container(_) = &spec.workload { - return Ok(()); - } - let Some(command) = task_command(spec) else { - return Err(ResourceTaskOwnershipRisk::UninspectableEntryPoint); - }; - if !command.program().contains('/') { - return Ok(()); - } - inspect_foreground_entry_point(&spec.cwd.join(command.program())) -} - -/// Inspect one resolved foreground entry point -/// -/// The canonical target must not be a known launcher, which catches a renamed -/// symbolic link, and its first bytes must be a native executable header. A -/// script, an unknown format, or an unreadable file fails closed -pub fn inspect_foreground_entry_point(binary: &Path) -> Result<(), ResourceTaskOwnershipRisk> { - let canonical = std::fs::canonicalize(binary) - .map_err(|_| ResourceTaskOwnershipRisk::UninspectableEntryPoint)?; - if let Some(risk) = canonical.to_str().and_then(program_risk) { - return Err(risk); - } - - let mut header = [0_u8; 4]; - File::open(&canonical) - .and_then(|mut file| file.read_exact(&mut header)) - .map_err(|_| ResourceTaskOwnershipRisk::UninspectableEntryPoint)?; - if header.starts_with(b"#!") { - return Err(ResourceTaskOwnershipRisk::ScriptEntryPoint); - } - if NATIVE_EXECUTABLE_MAGIC.contains(&header) { - return Ok(()); - } - - Err(ResourceTaskOwnershipRisk::UninspectableEntryPoint) -} - -// ELF, then 32-bit and 64-bit Mach-O in both byte orders, then universal Mach-O -const NATIVE_EXECUTABLE_MAGIC: [[u8; 4]; 6] = [ - [0x7f, b'E', b'L', b'F'], - [0xfe, 0xed, 0xfa, 0xce], - [0xce, 0xfa, 0xed, 0xfe], - [0xfe, 0xed, 0xfa, 0xcf], - [0xcf, 0xfa, 0xed, 0xfe], - [0xca, 0xfe, 0xba, 0xbe], -]; - -/// Where a known program name is refused -#[derive(Clone, Copy, PartialEq, Eq)] -enum Scope { - /// The name is refused as the program and as any later argument - Anywhere, - /// The name is refused only as the program, because it is also a common word - ProgramOnly, -} - -use Scope::{Anywhere, ProgramOnly}; - -const KNOWN_PROGRAMS: &[(&str, ResourceTaskOwnershipRisk, Scope)] = &[ - // remote execution and cluster schedulers run the work on another host - ("ssh", RemoteShell, Anywhere), - ("scp", RemoteShell, Anywhere), - ("sftp", RemoteShell, Anywhere), - ("mosh", RemoteShell, Anywhere), - ("rsh", RemoteShell, Anywhere), - ("autossh", RemoteShell, Anywhere), - ("sshpass", RemoteShell, Anywhere), - ("srun", RemoteShell, Anywhere), - ("sbatch", RemoteShell, Anywhere), - ("salloc", RemoteShell, Anywhere), - ("qsub", RemoteShell, Anywhere), - // container clients ask a daemon to run the work - ("docker", ContainerClient, Anywhere), - ("docker-compose", ContainerClient, Anywhere), - ("nvidia-docker", ContainerClient, Anywhere), - ("podman", ContainerClient, Anywhere), - ("nerdctl", ContainerClient, Anywhere), - ("ctr", ContainerClient, Anywhere), - ("crictl", ContainerClient, Anywhere), - ("kubectl", ContainerClient, Anywhere), - ("apptainer", ContainerClient, Anywhere), - ("singularity", ContainerClient, Anywhere), - ("lxc", ContainerClient, Anywhere), - ("incus", ContainerClient, Anywhere), - ("runc", ContainerClient, Anywhere), - ("crun", ContainerClient, Anywhere), - ("enroot", ContainerClient, Anywhere), - ("finch", ContainerClient, Anywhere), - ("systemd-nspawn", ContainerClient, Anywhere), - // shells run a command string that can background or detach work - ("sh", ShellWrapper, Anywhere), - ("bash", ShellWrapper, Anywhere), - ("dash", ShellWrapper, Anywhere), - ("ash", ShellWrapper, Anywhere), - ("zsh", ShellWrapper, Anywhere), - ("fish", ShellWrapper, Anywhere), - ("ksh", ShellWrapper, Anywhere), - ("mksh", ShellWrapper, Anywhere), - ("csh", ShellWrapper, Anywhere), - ("tcsh", ShellWrapper, Anywhere), - ("busybox", ShellWrapper, Anywhere), - ("pwsh", ShellWrapper, Anywhere), - ("powershell", ShellWrapper, Anywhere), - ("cmd", ShellWrapper, Anywhere), - ("xonsh", ShellWrapper, Anywhere), - ("nu", ShellWrapper, ProgramOnly), - ("elvish", ShellWrapper, Anywhere), - // detach tools start work that outlives the caller - ("nohup", DetachedLauncher, Anywhere), - ("setsid", DetachedLauncher, Anywhere), - ("daemon", DetachedLauncher, Anywhere), - ("daemonize", DetachedLauncher, Anywhere), - ("start-stop-daemon", DetachedLauncher, Anywhere), - ("screen", DetachedLauncher, Anywhere), - ("tmux", DetachedLauncher, Anywhere), - ("zellij", DetachedLauncher, Anywhere), - ("dtach", DetachedLauncher, Anywhere), - ("abduco", DetachedLauncher, Anywhere), - ("systemd-run", DetachedLauncher, Anywhere), - ("launchctl", DetachedLauncher, Anywhere), - ("pm2", DetachedLauncher, Anywhere), - ("supervisorctl", DetachedLauncher, Anywhere), - ("at", DetachedLauncher, ProgramOnly), - ("batch", DetachedLauncher, ProgramOnly), - ("open", DetachedLauncher, ProgramOnly), - ("xdg-open", DetachedLauncher, Anywhere), - // interpreters run code that Homebased does not inspect - ("python", Interpreter, Anywhere), - ("pythonw", Interpreter, Anywhere), - ("pypy", Interpreter, Anywhere), - ("ipython", Interpreter, Anywhere), - ("jupyter", Interpreter, Anywhere), - ("node", Interpreter, Anywhere), - ("nodejs", Interpreter, Anywhere), - ("deno", Interpreter, Anywhere), - ("bun", Interpreter, Anywhere), - ("perl", Interpreter, Anywhere), - ("ruby", Interpreter, Anywhere), - ("php", Interpreter, Anywhere), - ("lua", Interpreter, Anywhere), - ("luajit", Interpreter, Anywhere), - ("rscript", Interpreter, Anywhere), - ("julia", Interpreter, Anywhere), - ("java", Interpreter, Anywhere), - ("osascript", Interpreter, Anywhere), - ("tclsh", Interpreter, Anywhere), - ("awk", Interpreter, ProgramOnly), - ("gawk", Interpreter, Anywhere), - ("mawk", Interpreter, Anywhere), - ("nawk", Interpreter, Anywhere), - // launchers run another program named in their arguments - ("env", ProgramLauncher, ProgramOnly), - ("sudo", ProgramLauncher, Anywhere), - ("doas", ProgramLauncher, Anywhere), - ("su", ProgramLauncher, ProgramOnly), - ("runuser", ProgramLauncher, Anywhere), - ("pkexec", ProgramLauncher, Anywhere), - ("sg", ProgramLauncher, ProgramOnly), - ("newgrp", ProgramLauncher, ProgramOnly), - ("timeout", ProgramLauncher, ProgramOnly), - ("gtimeout", ProgramLauncher, ProgramOnly), - ("nice", ProgramLauncher, ProgramOnly), - ("ionice", ProgramLauncher, ProgramOnly), - ("chrt", ProgramLauncher, ProgramOnly), - ("taskset", ProgramLauncher, ProgramOnly), - ("numactl", ProgramLauncher, ProgramOnly), - ("stdbuf", ProgramLauncher, ProgramOnly), - ("unbuffer", ProgramLauncher, ProgramOnly), - ("time", ProgramLauncher, ProgramOnly), - ("strace", ProgramLauncher, ProgramOnly), - ("ltrace", ProgramLauncher, ProgramOnly), - ("perf", ProgramLauncher, ProgramOnly), - ("valgrind", ProgramLauncher, ProgramOnly), - ("gdb", ProgramLauncher, ProgramOnly), - ("lldb", ProgramLauncher, ProgramOnly), - ("nsys", ProgramLauncher, ProgramOnly), - ("ncu", ProgramLauncher, ProgramOnly), - ("nvprof", ProgramLauncher, ProgramOnly), - ("flock", ProgramLauncher, ProgramOnly), - ("xargs", ProgramLauncher, ProgramOnly), - ("parallel", ProgramLauncher, ProgramOnly), - ("watch", ProgramLauncher, ProgramOnly), - ("chroot", ProgramLauncher, ProgramOnly), - ("unshare", ProgramLauncher, ProgramOnly), - ("nsenter", ProgramLauncher, ProgramOnly), - ("firejail", ProgramLauncher, ProgramOnly), - ("bwrap", ProgramLauncher, ProgramOnly), - ("setpriv", ProgramLauncher, ProgramOnly), - ("capsh", ProgramLauncher, ProgramOnly), - ("prlimit", ProgramLauncher, ProgramOnly), - ("cgexec", ProgramLauncher, ProgramOnly), - ("sandbox-exec", ProgramLauncher, ProgramOnly), - ("script", ProgramLauncher, ProgramOnly), - ("expect", ProgramLauncher, ProgramOnly), - ("caffeinate", ProgramLauncher, ProgramOnly), - ("arch", ProgramLauncher, ProgramOnly), - ("exec", ProgramLauncher, ProgramOnly), - ("command", ProgramLauncher, ProgramOnly), - ("eatmydata", ProgramLauncher, ProgramOnly), - ("proxychains", ProgramLauncher, ProgramOnly), - ("direnv", ProgramLauncher, ProgramOnly), - ("entr", ProgramLauncher, ProgramOnly), - ("chronic", ProgramLauncher, ProgramOnly), - ("torchrun", ProgramLauncher, ProgramOnly), - ("accelerate", ProgramLauncher, ProgramOnly), - ("deepspeed", ProgramLauncher, ProgramOnly), - ("mpirun", ProgramLauncher, ProgramOnly), - ("mpiexec", ProgramLauncher, ProgramOnly), - ("horovodrun", ProgramLauncher, ProgramOnly), - // task and package runners run configured commands - ("make", ProgramLauncher, ProgramOnly), - ("gmake", ProgramLauncher, ProgramOnly), - ("just", ProgramLauncher, ProgramOnly), - ("cargo", ProgramLauncher, ProgramOnly), - ("go", ProgramLauncher, ProgramOnly), - ("npm", ProgramLauncher, ProgramOnly), - ("npx", ProgramLauncher, ProgramOnly), - ("pnpm", ProgramLauncher, ProgramOnly), - ("yarn", ProgramLauncher, ProgramOnly), - ("uv", ProgramLauncher, ProgramOnly), - ("uvx", ProgramLauncher, ProgramOnly), - ("pipx", ProgramLauncher, ProgramOnly), - ("poetry", ProgramLauncher, ProgramOnly), - ("pdm", ProgramLauncher, ProgramOnly), - ("hatch", ProgramLauncher, ProgramOnly), - ("conda", ProgramLauncher, ProgramOnly), - ("mamba", ProgramLauncher, ProgramOnly), - ("micromamba", ProgramLauncher, ProgramOnly), - ("pixi", ProgramLauncher, ProgramOnly), - ("tox", ProgramLauncher, ProgramOnly), - ("nox", ProgramLauncher, ProgramOnly), -]; - -/// Name families whose members carry a variable suffix, such as `lxc-start` -const KNOWN_PREFIXES: &[(&str, ResourceTaskOwnershipRisk)] = &[ - ("lxc-", ContainerClient), - ("docker-", ContainerClient), - ("python", Interpreter), - ("pypy", Interpreter), -]; - -fn program_risk(token: &str) -> Option { - known_program(token).map(|(risk, _)| risk) -} - -fn nested_program_risk(token: &str) -> Option { - known_program(token) - .filter(|(_, scope)| *scope == Anywhere) - .map(|(risk, _)| risk) -} - -/// Match the file name of one argv token against the known programs -/// -/// The match ignores case, a `.exe` suffix, and a trailing version such as the -/// `3.11` in `python3.11` -fn known_program(token: &str) -> Option<(ResourceTaskOwnershipRisk, Scope)> { - let name = Path::new(token).file_name()?.to_str()?.to_ascii_lowercase(); - let name = name.strip_suffix(".exe").unwrap_or(&name); - let unversioned = name.trim_end_matches(|c: char| c.is_ascii_digit() || c == '.' || c == '-'); - for candidate in [name, unversioned] { - if let Some((_, risk, scope)) = KNOWN_PROGRAMS - .iter() - .find(|(known, _, _)| *known == candidate) - { - return Some((*risk, *scope)); - } - } - - KNOWN_PREFIXES - .iter() - .find(|(prefix, _)| name.starts_with(prefix)) - .map(|(_, risk)| (*risk, Anywhere)) -} - -#[cfg(test)] -mod tests { - use std::fs; - use std::os::unix::fs::{PermissionsExt, symlink}; - - use tempfile::tempdir; - - use super::{CommandOwnershipContract, foreground_argv, inspect_foreground_entry_point}; - use crate::container::ContainerWorkload; - use crate::invocation::CommandLine; - use crate::resource::ResourceTaskOwnershipRisk::{ - self, ContainerClient, DetachedLauncher, Interpreter, ProgramLauncher, RemoteShell, - ShellWrapper, - }; - use crate::spec::{NormalizedTaskWorkload, NormalizedWorkload}; - use std::path::Path; - - fn argv(parts: &[&str]) -> CommandLine { - CommandLine::try_from_argv(parts.iter().map(|part| (*part).to_owned()).collect()).unwrap() - } - - fn task(parts: &[&str]) -> NormalizedWorkload { - NormalizedWorkload::Task(NormalizedTaskWorkload { - command: argv(parts), - }) - } - - #[test] - fn wrapped_or_nested_launch_shapes_are_refused() { - for (parts, expected) in [ - // a wrapper hides the detached container behind its own exit - ( - &["env", "docker", "run", "-d", "trainer"][..], - ProgramLauncher, - ), - ( - &["/usr/bin/sudo", "-n", "/opt/gpu/bench"][..], - ProgramLauncher, - ), - (&["timeout", "1h", "/opt/gpu/bench"][..], ProgramLauncher), - (&["python3.11", "bench.py"][..], Interpreter), - (&["/bin/bash", "-c", "bench &"][..], ShellWrapper), - (&["setsid", "/opt/gpu/bench"][..], DetachedLauncher), - (&["lxc-execute", "-n", "gpu"][..], ContainerClient), - // an unknown launcher still exposes the nested client or shell - ( - &["/opt/tools/renamed-env", "docker", "run", "-d", "x"][..], - ContainerClient, - ), - ( - &["/opt/tools/profiler", "--", "/usr/bin/ssh", "gpu-host"][..], - RemoteShell, - ), - ( - &["/opt/tools/runner", "/usr/bin/nohup", "/opt/gpu/bench"][..], - DetachedLauncher, - ), - (&["/opt/tools/runner", "Python3", "x.py"][..], Interpreter), - ] { - assert_eq!(foreground_argv(&argv(parts)), Err(expected), "{parts:?}"); - } - } - - #[test] - fn direct_commands_keep_ordinary_arguments() { - for parts in [ - &["/opt/gpu/bench", "--iterations", "10"][..], - &["./bin/optimize", "--time", "30", "--env", "prod"][..], - &["nvidia-smi", "--query-gpu=name", "--format=csv"][..], - ] { - assert_eq!(foreground_argv(&argv(parts)), Ok(()), "{parts:?}"); - assert_eq!( - CommandOwnershipContract::for_queued_work(&task(parts)), - Ok(CommandOwnershipContract::ForegroundExecutable) - ); - } - } - - #[test] - fn only_return_work_may_select_the_trainer_contract() { - let trainer = task(&[ - "python3", - "-m", - "ops.run_segment", - "run", - "--task", - "/t/task.json", - "--input-root", - "/t/inputs", - "--runtime-root", - "/t/runtime", - "--image-digest", - "sha256:00", - ]); - assert_eq!( - CommandOwnershipContract::for_return_work(&trainer), - Ok(CommandOwnershipContract::DirectSegmentTrainer) - ); - assert_eq!( - CommandOwnershipContract::for_queued_work(&trainer), - Err(Interpreter) - ); - assert_eq!( - CommandOwnershipContract::for_return_work(&task(&["python3", "evaluate.py"])), - Err(Interpreter) - ); - } - - #[test] - fn a_typed_container_selects_the_container_contract_but_a_docker_command_does_not() { - let container = NormalizedWorkload::Container(Box::new( - ContainerWorkload::from_value(&serde_json::json!({ - "image": format!("sha256:{}", "0".repeat(64)), - "memory": "1g", - "gpus": "all" - })) - .unwrap(), - )); - for classify in [ - CommandOwnershipContract::for_queued_work, - CommandOwnershipContract::for_return_work, - ] { - assert_eq!( - classify(&container), - Ok(CommandOwnershipContract::Container) - ); - assert_eq!( - classify(&task(&["docker", "run", "--gpus", "all", "eval"])), - Err(ContainerClient) - ); - } - } - - #[test] - fn entry_point_must_be_an_inspectable_native_executable() { - let directory = tempdir().unwrap(); - let root = directory.path(); - let script = root.join("prepared-command"); - fs::write(&script, "#!/bin/sh\nexec /opt/gpu/bench &\n").unwrap(); - fs::set_permissions(&script, fs::Permissions::from_mode(0o755)).unwrap(); - let unknown = root.join("unknown-format"); - fs::write(&unknown, b"\0\0\0\0payload").unwrap(); - // a renamed link to a shell is judged by its target - let renamed = root.join("bench"); - symlink("/bin/sh", &renamed).unwrap(); - - assert_eq!( - inspect_foreground_entry_point(&script), - Err(ResourceTaskOwnershipRisk::ScriptEntryPoint) - ); - assert_eq!( - inspect_foreground_entry_point(&unknown), - Err(ResourceTaskOwnershipRisk::UninspectableEntryPoint) - ); - assert_eq!( - inspect_foreground_entry_point(&root.join("missing")), - Err(ResourceTaskOwnershipRisk::UninspectableEntryPoint) - ); - assert!(inspect_foreground_entry_point(&renamed).is_err()); - assert_eq!( - inspect_foreground_entry_point(Path::new("/bin/echo")), - Ok(()) - ); - } -} - -#[cfg(test)] -pub(crate) mod test_support { - use std::path::{Path, PathBuf}; - use std::process::Command; - use std::sync::OnceLock; - - // appends one `x` to its first argument, then waits until its optional second - // argument exists, so a test can count launches and hold a task running - const SOURCE: &str = r#" -#include -#include -int main(int argc, char **argv) { - if (argc < 2) return 2; - FILE *marker = fopen(argv[1], "a"); - if (marker == NULL || fputs("x", marker) < 0 || fclose(marker) != 0) return 1; - while (argc > 2 && access(argv[2], F_OK) != 0) usleep(20000); - return 0; -} -"#; - - /// Native stand-in for a prepared GPU command, compiled once per test process - /// - /// A shell script is refused by the foreground contract, and macOS kills a - /// copied system binary, so the tests build their own native executable - pub(crate) fn native_fake_command() -> &'static Path { - static COMMAND: OnceLock = OnceLock::new(); - COMMAND.get_or_init(|| { - let directory = std::env::temp_dir() - .join(format!("homebased-fake-gpu-command-{}", std::process::id())); - std::fs::create_dir_all(&directory).unwrap(); - let source = directory.join("fake-gpu-command.c"); - std::fs::write(&source, SOURCE).unwrap(); - let binary = directory.join("fake-gpu-command"); - let status = Command::new("cc") - .arg(&source) - .arg("-o") - .arg(&binary) - .status() - .unwrap(); - assert!(status.success(), "cc must build the fake GPU command"); - binary - }) - } -} diff --git a/src/resource/id.rs b/src/resource/id.rs deleted file mode 100644 index a5a234a..0000000 --- a/src/resource/id.rs +++ /dev/null @@ -1,149 +0,0 @@ -//! UUID identities of resource records that are never nil -//! -//! Every identity is allocated as a UUIDv7 or decoded from a saved or received -//! value. Construction and decoding both refuse the nil UUID, so code that holds -//! one of these identities never has to check for nil again - -use std::str::FromStr; - -use serde::{Deserialize, Deserializer, Serialize}; -use uuid::Uuid; - -/// The nil UUID was offered as a resource record identity -#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] -#[error("resource identities must not be the nil UUID")] -pub struct NilIdentity; - -/// Text that does not name a non-nil resource record identity -#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] -pub enum IdentityParseError { - /// The text is not a UUID - #[error("invalid resource identity: {0}")] - Uuid(#[from] uuid::Error), - /// The text is the nil UUID - #[error(transparent)] - Nil(#[from] NilIdentity), -} - -macro_rules! resource_uuid_id { - ($(#[$meta:meta])* pub struct $name:ident;) => { - $(#[$meta])* - #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize)] - #[serde(transparent)] - pub struct $name(Uuid); - - impl $name { - /// Allocate a new time-ordered identity - #[must_use] - pub fn new() -> Self { - Self(Uuid::now_v7()) - } - - /// Wrap an existing UUID, refusing the nil UUID - pub const fn from_uuid(uuid: Uuid) -> Result { - if uuid.is_nil() { - return Err(NilIdentity); - } - Ok(Self(uuid)) - } - - /// Return the underlying UUID value - #[must_use] - pub const fn as_uuid(&self) -> Uuid { - self.0 - } - } - - impl Default for $name { - fn default() -> Self { - Self::new() - } - } - - impl TryFrom for $name { - type Error = NilIdentity; - - fn try_from(uuid: Uuid) -> Result { - Self::from_uuid(uuid) - } - } - - impl FromStr for $name { - type Err = IdentityParseError; - - fn from_str(text: &str) -> Result { - Ok(Self::from_uuid(Uuid::parse_str(text)?)?) - } - } - - impl<'de> Deserialize<'de> for $name { - fn deserialize(deserializer: D) -> Result - where - D: Deserializer<'de>, - { - Self::from_uuid(Uuid::deserialize(deserializer)?).map_err(serde::de::Error::custom) - } - } - }; -} - -resource_uuid_id! { - /// Stable identity of one physical GPU resource - pub struct ResourceId; -} - -resource_uuid_id! { - /// Stable identity of one interruption loan - pub struct LoanId; -} - -resource_uuid_id! { - /// Stable identity of one required supervisor decision - pub struct ActionId; -} - -resource_uuid_id! { - /// Stable identity of one durable supervisor notice - pub struct NoticeId; -} - -resource_uuid_id! { - /// Stable identity of one notice delivery attempt - pub struct DeliveryAttemptId; -} - -resource_uuid_id! { - /// Stable identity of one reserved exact-task stop request - pub struct ReleaseStopReservationId; -} - -resource_uuid_id! { - /// Stable caller identity of one operator attestation - /// - /// The caller allocates it before the first send. An exact retry returns the - /// saved receipt, and different content under the same identity conflicts - pub struct OperatorAttestationId; -} - -#[cfg(test)] -mod tests { - use serde_json::json; - use uuid::Uuid; - - use super::{NilIdentity, ResourceId}; - - #[test] - fn nil_identity_is_refused_by_construction_and_decoding() { - assert_eq!(ResourceId::from_uuid(Uuid::nil()), Err(NilIdentity)); - assert!(serde_json::from_value::(json!(Uuid::nil())).is_err()); - } - - #[test] - fn a_saved_identity_decodes_to_the_same_uuid_text() { - let text = "019b4f42-0000-7000-8000-000000000061"; - let id = serde_json::from_value::(json!(text)).unwrap(); - - assert_eq!(id.as_uuid(), Uuid::parse_str(text).unwrap()); - assert_eq!(serde_json::to_value(id).unwrap(), json!(text)); - } -} diff --git a/src/resource/initial_idle.rs b/src/resource/initial_idle.rs deleted file mode 100644 index d75f9c7..0000000 --- a/src/resource/initial_idle.rs +++ /dev/null @@ -1,184 +0,0 @@ -//! Operator attestation that a resource with no history starts with a free GPU -//! -//! Queued work serves an unregistered resource only from a saved idle -//! boundary: a closed loan, a first background launch that never spawned, or -//! an operator attestation about one ended task. A resource that was just -//! registered has none of these, so its first request waits. An operator who -//! inspected the authority GPU can record one attestation that the resource -//! starts idle. The authority accepts it only while the resource has no -//! registered task, no loan, and no first background launch, and it becomes the -//! idle boundary only while that is still true -//! -//! The attestation is a human trust decision. It is never proof that a process -//! exited or that a lock was released - -use serde::{Deserialize, Serialize}; - -use super::operator_release::{ - OperatorAttestationId, OperatorGpuFreeConfirmation, OperatorObservation, -}; -use super::{ResourceId, ResourceRevision}; -use crate::machine::MachineId; - -/// Immutable operator request that a resource with no history starts idle -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct InitialIdleAttestation { - /// Stable caller retry identity - pub operation_id: OperatorAttestationId, - /// Resource that starts idle - pub resource_id: ResourceId, - /// Authority machine that the operator inspected - pub authority_machine: MachineId, - /// Resource revision that the operator observed - pub expected_state_revision: ResourceRevision, - /// What the operator inspected and why no work holds the GPU - pub observation: OperatorObservation, - /// Explicit human confirmation - pub confirmation: OperatorGpuFreeConfirmation, -} - -impl InitialIdleAttestation { - /// Check the identities that a well-formed attestation needs - pub fn validate(&self) -> Result<(), InitialIdleRefusal> { - if self.authority_machine.as_uuid().is_nil() { - return Err(InitialIdleRefusal::InvalidIdentity); - } - if self.observation.as_str().trim().is_empty() { - return Err(InitialIdleRefusal::EmptyObservation); - } - Ok(()) - } -} - -/// Saved result of one initial idle attestation -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct InitialIdleReceipt { - /// Exact attestation content - pub attestation: InitialIdleAttestation, - /// Resource revision committed with the attestation - pub state_revision: ResourceRevision, -} - -/// Result of one attestation call -#[derive(Debug, Clone)] -pub struct InitialIdleResolution { - /// Saved receipt - pub receipt: InitialIdleReceipt, - /// Whether an earlier call committed this receipt - pub replayed: bool, -} - -/// Resource history that already decides whether the GPU is idle -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum ResourceHistory { - /// A background task is registered - RegisteredTask, - /// A loan exists or existed - Loan, - /// A first background launch exists or existed - BackgroundLaunch, - /// An operator attestation about an ended task exists - OperatorAttestation, -} - -/// Why the authority refused an initial idle attestation without writing any record -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, thiserror::Error)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum InitialIdleRefusal { - /// An identity is nil - #[error("initial idle attestation identities must not be nil")] - InvalidIdentity, - /// The observation has no text - #[error("operator observation must not be empty")] - EmptyObservation, - /// The resource does not exist on this authority - #[error("resource not found")] - ResourceNotFound, - /// The attestation or daemon names another authority - #[error("resource authority is {expected}, not {found}")] - WrongAuthority { - /// Authority saved on the resource - expected: MachineId, - /// Authority named by the attestation or the daemon - found: MachineId, - }, - /// The operation identity already names different content - #[error( - "initial idle attestation {} was retried with different content", - operation_id.as_uuid() - )] - ConflictingRetry { - /// Reused operation identity - operation_id: OperatorAttestationId, - }, - /// The resource changed after the operator read it - #[error("stale resource revision: expected {expected:?}, found {actual:?}")] - StaleRevision { - /// Revision that the operator observed - expected: ResourceRevision, - /// Current revision - actual: ResourceRevision, - }, - /// The resource already has history, which decides its idle state instead - #[error("resource already has history ({history:?}), which decides whether its GPU is idle")] - HistoryExists { - /// First history record found - history: ResourceHistory, - }, - /// Another initial idle attestation is saved for this resource - #[error("resource already has initial idle attestation {}", operation_id.as_uuid())] - AlreadyAttested { - /// Saved attestation - operation_id: OperatorAttestationId, - }, - /// The resource revision cannot be incremented - #[error("resource revision {revision:?} cannot be incremented")] - RevisionExhausted { - /// Current revision - revision: ResourceRevision, - }, -} - -#[cfg(test)] -mod tests { - use serde_json::json; - - use super::InitialIdleAttestation; - use crate::machine::MachineId; - use crate::resource::ResourceId; - - fn document() -> serde_json::Value { - json!({ - "operation_id": uuid::Uuid::now_v7(), - "resource_id": ResourceId::new().as_uuid(), - "authority_machine": MachineId::new(), - "expected_state_revision": 0, - "observation": "nvidia-smi shows no compute processes", - "confirmation": "operator_confirmed_gpu_free" - }) - } - - #[test] - fn the_document_is_strict_and_needs_a_confirmation_and_observation() { - let parsed: InitialIdleAttestation = serde_json::from_value(document()).unwrap(); - parsed.validate().unwrap(); - for (key, value) in [ - ("confirmation", json!("yes")), - ("observation", json!(" ")), - ("task_id", json!(uuid::Uuid::now_v7())), - ] { - let mut changed = document(); - changed[key] = value; - assert!( - serde_json::from_value::(changed).is_err(), - "{key}" - ); - } - let mut missing = document(); - missing.as_object_mut().unwrap().remove("confirmation"); - assert!(serde_json::from_value::(missing).is_err()); - } -} diff --git a/src/resource/operator_release.rs b/src/resource/operator_release.rs deleted file mode 100644 index d8c55d2..0000000 --- a/src/resource/operator_release.rs +++ /dev/null @@ -1,524 +0,0 @@ -//! Operator attestation that unprovable ended GPU work no longer holds its GPU -//! -//! A direct-segment trainer can end before the supervisor binds its trainer -//! attempt, so no saved lock can prove that its detached worker exited. This -//! includes a first background launch or a return task that ends before its -//! confirmed start registers it, because only a registered trainer can bind an -//! attempt. A native foreground return task can be lost, or end without a -//! confirmed process-group exit, so the task layer cannot prove that its process -//! released the GPU. The automatic release proof stays fail-closed for all of -//! them. An operator who inspected the authority machine can instead record one -//! explicit, auditable attestation. The authority saves it with an evidence -//! snapshot and the resulting queue or loan transition in one transaction -//! -//! An attestation is a human trust decision. It is never a confirmed -//! process-group exit, a trainer-lock release, or a resumable checkpoint, and -//! every record that it produces says so through its own typed variant - -use serde::{Deserialize, Serialize}; - -pub use super::id::OperatorAttestationId; -use super::ownership_lock::TrainerRequestDigest; -use super::{ActionId, Loan, LoanId, ResourceId, ResourceRequest, ResourceRevision}; -use super::{SupervisorNotice, TaskId}; -use crate::domain::{ContainerExitEvidence, ExitReason, ProcessGroupExitEvidence, ProcessStatus}; -use crate::machine::MachineId; -use crate::submission::{NormalizedSpecSha256, RequestId}; - -/// Non-empty operator account of what was inspected and why the GPU is free -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(try_from = "String", into = "String")] -pub struct OperatorObservation(String); - -impl OperatorObservation { - /// Borrow the saved observation text - #[must_use] - pub fn as_str(&self) -> &str { - &self.0 - } -} - -impl TryFrom for OperatorObservation { - type Error = OperatorGpuFreeRefusal; - - fn try_from(value: String) -> Result { - if value.trim().is_empty() { - return Err(OperatorGpuFreeRefusal::EmptyObservation); - } - Ok(Self(value)) - } -} - -impl From for String { - fn from(value: OperatorObservation) -> Self { - value.0 - } -} - -/// Explicit human confirmation that the operator checked the GPU on the authority -/// -/// The field has one value and no default, so a request that omits it cannot -/// decode or be built -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum OperatorGpuFreeConfirmation { - /// The operator confirmed that no GPU work of the named trainer remains - OperatorConfirmedGpuFree, -} - -/// Resource state that the operator observed and attests against -/// -/// The authority compares it with the current state in the deciding -/// transaction, so an attestation never applies to a state it did not name -/// `NoLoan` and `AwaitingRelease` name the registered trainer -/// `FirstBackgroundLaunch` and `RestoringReturn` name a trainer that ended -/// before its confirmed start registered it. `RestoringForegroundReturn` names -/// a native foreground return task, which is never registered -/// -/// Decoding refuses unknown fields on every variant, including `NoLoan` -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde( - tag = "type", - rename_all = "snake_case", - deny_unknown_fields, - from = "StrictStateBinding" -)] -pub enum OperatorStateBinding { - /// No non-closed loan reserves the resource, and the task is the registered trainer - NoLoan, - /// One exact release action waits for the registered trainer - AwaitingRelease { - /// Loan that owns the release action - loan_id: LoanId, - /// Release action that the attestation resolves - action_id: ActionId, - }, - /// The latest first background launch ended before its confirmed start, with no loan - /// - /// The task is the launch task, which was never registered - FirstBackgroundLaunch { - /// Stable launch request identity - request_id: RequestId, - }, - /// The direct-segment return task of one Restoring loan ended before its confirmed start - /// - /// The task is the loan's bound return task, which was never registered - RestoringReturn { - /// Restoring loan that keeps the resource reserved - loan_id: LoanId, - /// Return action that bound the task - action_id: ActionId, - }, - /// The native foreground return task of one Restoring loan ended or was lost - /// without proof that its process group released the GPU - /// - /// The task is the loan's bound return task. It is terminal or lost, and - /// its saved decision accepted it as native foreground work - RestoringForegroundReturn { - /// Restoring loan that keeps the resource reserved - loan_id: LoanId, - /// Return action that bound the task - action_id: ActionId, - }, -} - -/// Decoding shape of [`OperatorStateBinding`] -/// -/// Serde ignores extra fields on a unit variant of an internally tagged enum, -/// so `no_loan` decodes through an empty struct variant that refuses them. The -/// JSON shape is the same as the public enum -#[derive(Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -enum StrictStateBinding { - NoLoan {}, - AwaitingRelease { - loan_id: LoanId, - action_id: ActionId, - }, - FirstBackgroundLaunch { - request_id: RequestId, - }, - RestoringReturn { - loan_id: LoanId, - action_id: ActionId, - }, - RestoringForegroundReturn { - loan_id: LoanId, - action_id: ActionId, - }, -} - -impl From for OperatorStateBinding { - fn from(value: StrictStateBinding) -> Self { - match value { - StrictStateBinding::NoLoan {} => Self::NoLoan, - StrictStateBinding::AwaitingRelease { loan_id, action_id } => { - Self::AwaitingRelease { loan_id, action_id } - } - StrictStateBinding::FirstBackgroundLaunch { request_id } => { - Self::FirstBackgroundLaunch { request_id } - } - StrictStateBinding::RestoringReturn { loan_id, action_id } => { - Self::RestoringReturn { loan_id, action_id } - } - StrictStateBinding::RestoringForegroundReturn { loan_id, action_id } => { - Self::RestoringForegroundReturn { loan_id, action_id } - } - } - } -} - -impl OperatorStateBinding { - /// Whether the attested task must be the registered trainer - #[must_use] - pub const fn names_registered_trainer(&self) -> bool { - match self { - Self::NoLoan | Self::AwaitingRelease { .. } => true, - Self::FirstBackgroundLaunch { .. } - | Self::RestoringReturn { .. } - | Self::RestoringForegroundReturn { .. } => false, - } - } - - /// Restoring loan and return action that the binding names, if any - #[must_use] - pub const fn restoring_action(&self) -> Option<(LoanId, ActionId)> { - match *self { - Self::RestoringReturn { loan_id, action_id } - | Self::RestoringForegroundReturn { loan_id, action_id } => Some((loan_id, action_id)), - Self::NoLoan | Self::AwaitingRelease { .. } | Self::FirstBackgroundLaunch { .. } => { - None - } - } - } - - fn has_nil_identity(&self) -> bool { - matches!(self, Self::FirstBackgroundLaunch { request_id } if request_id.0.is_nil()) - } -} - -/// Immutable operator request that one ended trainer no longer holds its GPU -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct OperatorGpuFreeAttestation { - /// Stable caller retry identity - pub operation_id: OperatorAttestationId, - /// Resource whose trainer ended - pub resource_id: ResourceId, - /// Authority machine that the operator inspected - pub authority_machine: MachineId, - /// Exact task whose GPU work the operator attests is gone - /// - /// It is the registered trainer, the launch task, or the bound return task, - /// as `state_binding` requires - pub task_id: TaskId, - /// Resource revision that the operator observed - pub expected_state_revision: ResourceRevision, - /// Loan state that the operator observed - pub state_binding: OperatorStateBinding, - /// What the operator inspected and why the GPU work is gone - pub observation: OperatorObservation, - /// Explicit human confirmation - pub confirmation: OperatorGpuFreeConfirmation, -} - -impl OperatorGpuFreeAttestation { - /// Check the identities that a well-formed attestation needs - pub fn validate(&self) -> Result<(), OperatorGpuFreeRefusal> { - if self.authority_machine.as_uuid().is_nil() - || self.task_id.0.is_nil() - || self.state_binding.has_nil_identity() - { - return Err(OperatorGpuFreeRefusal::InvalidIdentity); - } - if self.observation.as_str().trim().is_empty() { - return Err(OperatorGpuFreeRefusal::EmptyObservation); - } - Ok(()) - } -} - -/// Task-layer end of the attested task when the attestation committed -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum AttestedTrainerEnd { - /// The wrapper wrote an exit reason - Finished { - /// Task-layer outcome - outcome: ExitReason, - /// Wrapper process-group evidence as saved - /// - /// It never covers a detached trainer worker. For a native foreground - /// return it is the saved task-layer fact, often `unconfirmed`; the - /// attestation does not upgrade it to a confirmed exit - process_group_exit: ProcessGroupExitEvidence, - /// Container evidence as saved, for a container task only - /// - /// The attestation does not upgrade it to confirmed evidence - #[serde(default, skip_serializing_if = "Option::is_none")] - container_exit: Option, - }, - /// The wrapper was lost with no exit reason - Lost, -} - -/// Resource launch that bound the attested task -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum AttestedTrainerLaunch { - /// First background launch that bound the task - FirstBackgroundLaunch { - /// Stable launch request identity - request_id: RequestId, - }, - /// Direct-segment return decision that bound the task - DirectSegmentReturn { - /// Return action that bound the task - action_id: ActionId, - /// Stable return request identity - request_id: RequestId, - }, - /// Native foreground return decision that bound the task - /// - /// The saved decision accepted the task as native foreground work, which - /// holds the GPU only through its own process group - NativeForegroundReturn { - /// Return action that bound the task - action_id: ActionId, - /// Stable return request identity - request_id: RequestId, - }, - /// Container return decision that bound the task - /// - /// The saved decision accepted the task as a container, which holds the - /// GPU through the container that Homebased started - ContainerReturn { - /// Return action that bound the task - action_id: ActionId, - /// Stable return request identity - request_id: RequestId, - }, -} - -/// Trainer-attempt association of the attested task when the attestation committed -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum AttestedTrainerAssociation { - /// No attempt was bound, so no saved lock names the worker - Missing, - /// An attempt was bound, but its release proof was not used - Saved { - /// Request digest of the saved attempt - attempt_request_sha256: TrainerRequestDigest, - }, -} - -/// Authority evidence snapshot saved with the attestation -/// -/// These facts show which run the operator resolved. None of them proves that -/// the trainer worker exited -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct OperatorGpuFreeEvidence { - /// Task-layer end of the trainer - pub trainer_end: AttestedTrainerEnd, - /// Resource launch that bound the trainer - pub trainer_launch: AttestedTrainerLaunch, - /// Digest of the accepted executor identity's normalized spec - pub normalized_spec_sha256: NormalizedSpecSha256, - /// Trainer-attempt association state - pub trainer_association: AttestedTrainerAssociation, -} - -/// Queue or loan transition committed with the attestation -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum OperatorGpuFreeOutcome { - /// The release action closed and the next queued request now serves - ReleaseResolvedServing { - /// Loan moved from AwaitingRelease to Serving - loan: Loan, - /// Request selected in the same transaction - request: ResourceRequest, - }, - /// The release action closed with an empty queue; the supervisor owns the return - ReleaseResolvedReturnRequired { - /// Loan moved from AwaitingRelease to AwaitingReturn - loan: Loan, - /// Return notice saved in the same transaction - notice: SupervisorNotice, - }, - /// The registration cleared and an idle loan serves the next queued request - IdleServing { - /// New Serving loan that names this attestation as its idle boundary - loan: Loan, - /// Request selected in the same transaction - request: ResourceRequest, - }, - /// The registration cleared and this attestation is the saved idle boundary - /// - /// A later queue reconciliation or first background launch reads it - IdleBoundary, - /// The Restoring loan closed and an idle loan serves the next queued request - RestoreClosedServing { - /// Restoring loan closed with the operator-attested end - closed: Box, - /// New Serving loan that names this attestation as its idle boundary - loan: Loan, - /// Request selected in the same transaction - request: ResourceRequest, - }, - /// The Restoring loan closed with an empty queue, and its closure is the idle boundary - RestoreClosedIdleBoundary { - /// Restoring loan closed with the operator-attested end - closed: Loan, - }, -} - -/// Durable receipt of one committed operator attestation -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct OperatorGpuFreeReceipt { - /// Exact attestation content - pub attestation: OperatorGpuFreeAttestation, - /// Authority evidence snapshot - pub evidence: OperatorGpuFreeEvidence, - /// Resource revision committed with the transition - pub state_revision: ResourceRevision, - /// Committed transition - pub outcome: OperatorGpuFreeOutcome, -} - -/// Result of one attestation call -#[derive(Debug, Clone)] -pub struct OperatorGpuFreeResolution { - /// Saved receipt - pub receipt: OperatorGpuFreeReceipt, - /// Whether an earlier call committed this receipt - pub replayed: bool, -} - -/// Why the authority refused an attestation without writing any record -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, thiserror::Error)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum OperatorGpuFreeRefusal { - /// An identity is nil - #[error("operator attestation identities must not be nil")] - InvalidIdentity, - /// The observation has no text - #[error("operator observation must not be empty")] - EmptyObservation, - /// The resource does not exist on this authority - #[error("resource not found")] - ResourceNotFound, - /// The attestation or daemon names another authority - #[error("resource authority is {expected}, not {found}")] - WrongAuthority { - /// Authority saved on the resource - expected: MachineId, - /// Authority named by the attestation or the daemon - found: MachineId, - }, - /// The operation identity already names different content - #[error("operator attestation {operation_id:?} was retried with different content")] - ConflictingRetry { - /// Reused operation identity - operation_id: OperatorAttestationId, - }, - /// The resource changed after the operator read it - #[error("stale resource revision: expected {expected:?}, found {actual:?}")] - StaleRevision { - /// Revision that the operator observed - expected: ResourceRevision, - /// Current revision - actual: ResourceRevision, - }, - /// The task is not the current registered trainer - #[error("task {task_id} is not the registered background task")] - NotRegisteredTrainer { - /// Task named by the attestation - task_id: TaskId, - /// Current registration - registered: Option, - }, - /// The task is not the task that the named launch or return action bound - #[error("task {task_id} is not the task bound by the attested launch or action")] - NotBoundTask { - /// Task named by the attestation - task_id: TaskId, - /// Task bound by the launch or return action - bound: TaskId, - }, - /// The latest first background launch is not the named launch awaiting release - /// - /// It is another launch, it was registered or superseded, or it already has - /// automatic proof that no child started - #[error("first background launch {request_id:?} does not await an operator release")] - LaunchNotAwaitingRelease { - /// Launch named by the attestation - request_id: RequestId, - /// Latest first background launch, if any - current_launch: Option, - }, - /// The named loan is not in its Restoring phase - #[error("loan {loan_id:?} is not restoring")] - LoanNotRestoring { - /// Current non-closed loan - loan_id: LoanId, - }, - /// The current loan state differs from the named binding - #[error("the current loan state differs from the attested binding")] - LoanStateChanged { - /// Current non-closed loan, if any - current_loan: Option, - }, - /// The loan is past its release phase, so its work may already use the GPU - #[error("loan {loan_id:?} is not awaiting release")] - LoanNotAwaitingRelease { - /// Current non-closed loan - loan_id: LoanId, - }, - /// The release action notice is missing or names another task or assignment - #[error("release action {action_id:?} has an invalid durable notice")] - InvalidReleaseNotice { - /// Release action named by the attestation - action_id: ActionId, - }, - /// The task row is missing on this authority - #[error("task {task_id} is missing")] - TaskMissing { - /// Task named by the attestation - task_id: TaskId, - }, - /// The task has not ended, so its wrapper still owns the GPU - #[error("task {task_id} has not ended (state {state})")] - TaskNotEnded { - /// Task named by the attestation - task_id: TaskId, - /// Current task-layer state - state: ProcessStatus, - }, - /// No saved resource launch of the binding's execution mode and matching accepted identity name the task - /// - /// A direct-segment binding needs direct-segment work, and a foreground - /// binding needs native foreground work - #[error( - "task {task_id} has no matching resource launch, execution mode, and accepted identity" - )] - TrainerLaunchUnproven { - /// Task named by the attestation - task_id: TaskId, - }, - /// Saved launch history does not match the registration - #[error("resource launch history is inconsistent with task {task_id}")] - InconsistentHistory { - /// Task named by the attestation - task_id: TaskId, - }, - /// The resource revision cannot be incremented - #[error("resource revision {revision:?} cannot be incremented")] - RevisionExhausted { - /// Current revision - revision: ResourceRevision, - }, -} diff --git a/src/resource/ownership_lock.rs b/src/resource/ownership_lock.rs deleted file mode 100644 index b2888f1..0000000 --- a/src/resource/ownership_lock.rs +++ /dev/null @@ -1,220 +0,0 @@ -//! Ownership-lock evidence for direct-segment trainers -//! -//! A trainer holds `.segment.lock` under its runtime root while its GPU worker -//! lives. [`attempt_evidence`] captures the lock and exact request of one live -//! attempt at registration time, and [`lock_probe`] later checks whether that -//! exact lock was released - -mod attempt_evidence; -mod lock_probe; - -pub use attempt_evidence::{ - TrainerAttemptEvidenceError, TrainerAttemptUnsafePathReason, TrainerRequestDigest, - VerifiedTrainerAttempt, build_trainer_attempt_registration_evidence, -}; -pub use lock_probe::{ - OwnershipLockGuard, OwnershipLockIdentity, OwnershipLockIdentityMismatchReason, - OwnershipLockProbe, OwnershipLockProbeError, OwnershipLockProbeOperation, - probe_segment_ownership_lock, verify_ownership_lock_guard, -}; - -/// Lock file that the trainer holds directly under its runtime root -const SEGMENT_LOCK_FILE: &str = ".segment.lock"; - -#[cfg(test)] -pub(crate) mod test_support { - use std::env; - use std::io::{BufRead, BufReader, Read, Write}; - use std::path::{Path, PathBuf}; - use std::process::{Child, ChildStdout, Command, Stdio}; - - use std::fs::{self, File, OpenOptions}; - use std::time::SystemTime; - - use super::{OwnershipLockIdentity, TrainerRequestDigest, VerifiedTrainerAttempt}; - use crate::digest::Sha256Digest; - use crate::resource::trainer_publication::AttemptBinding; - - pub(crate) fn attempt_binding(attempt_id: &str) -> AttemptBinding { - AttemptBinding { - campaign_id: "campaign-1".into(), - campaign_revision_id: "revision-1".into(), - task_id: "trainer-task-1".into(), - attempt_id: attempt_id.into(), - attempt_number: 1, - ownership_token: "token-1".into(), - } - } - - pub(crate) fn verified_attempt( - binding: AttemptBinding, - request_sha256: [u8; 32], - lock_identity: OwnershipLockIdentity, - ) -> VerifiedTrainerAttempt { - VerifiedTrainerAttempt::from_persisted( - PathBuf::from("/trainer/runtime"), - binding, - TrainerRequestDigest::from_hex(&Sha256Digest::from_bytes(request_sha256).to_hex()) - .expect("test request digest is canonical"), - lock_identity, - ) - .expect("test attempt evidence is valid") - } - - pub(crate) const CHILD_PATH_ENV: &str = "HOMEBASED_OWNERSHIP_LOCK_TEST_PATH"; - pub(crate) const CHILD_MODE_ENV: &str = "HOMEBASED_OWNERSHIP_LOCK_TEST_MODE"; - pub(crate) const HELPER_TEST: &str = - "resource::ownership_lock::lock_probe::tests::fake_process_lock_helper"; - pub(crate) const LOCK_ACQUIRED: &str = "FAKE_LOCK_ACQUIRED"; - pub(crate) const LOCK_CONTENDED: &str = "FAKE_LOCK_CONTENDED"; - - /// Separate test process that holds one exact lock file until released or dropped - /// - /// A `flock` lock is owned by the open file description, so a holder in another process - /// behaves like the trainer worker that outlives its wrapper - pub(crate) struct FakeLockProcess { - child: Child, - output: BufReader, - released: bool, - } - - impl FakeLockProcess { - /// Let the holder release the lock and wait for it to exit - pub(crate) fn release(&mut self) { - self.child - .stdin - .as_mut() - .expect("fake lock process stdin is piped") - .write_all(b"x") - .expect("fake lock process accepts release input"); - self.child.stdin.take(); - let status = self.child.wait().expect("fake lock process exits"); - let mut trailing = String::new(); - self.output - .read_to_string(&mut trailing) - .expect("read fake lock process output"); - assert!(status.success(), "fake lock process failed: {trailing}"); - self.released = true; - } - } - - impl Drop for FakeLockProcess { - fn drop(&mut self) { - if self.released { - return; - } - - if let Some(stdin) = self.child.stdin.as_mut() { - let _ = stdin.write_all(b"x"); - } - self.child.stdin.take(); - let _ = self.child.wait(); - } - } - - /// Start a separate process that holds the existing lock file at `path` - pub(crate) fn start_fake_lock_process(path: &Path) -> FakeLockProcess { - let mut child = Command::new(env::current_exe().expect("find current test binary")) - .arg("--exact") - .arg(HELPER_TEST) - .arg("--nocapture") - .env(CHILD_PATH_ENV, path) - .env(CHILD_MODE_ENV, "hold") - .stdin(Stdio::piped()) - .stdout(Stdio::piped()) - .stderr(Stdio::piped()) - .spawn() - .expect("start fake lock-holder process"); - let mut process = FakeLockProcess { - output: BufReader::new(child.stdout.take().expect("capture helper stdout")), - child, - released: false, - }; - - let mut line = String::new(); - loop { - line.clear(); - let bytes = process - .output - .read_line(&mut line) - .expect("read fake lock-holder readiness"); - assert_ne!(bytes, 0, "fake lock-holder exited before acquiring lock"); - if line.trim() == LOCK_ACQUIRED { - break; - } - } - - process - } - - #[derive(Debug, PartialEq, Eq)] - pub(crate) struct FileSnapshot { - identity: OwnershipLockIdentity, - contents: Vec, - length: u64, - modified: Option, - } - - pub(crate) fn lock_path(runtime_root: &Path) -> PathBuf { - runtime_root.join(".segment.lock") - } - - pub(crate) fn create_lock(runtime_root: &Path, contents: &[u8]) -> File { - fs::create_dir_all(runtime_root).expect("create temporary runtime root"); - let mut file = OpenOptions::new() - .create_new(true) - .read(true) - .write(true) - .open(lock_path(runtime_root)) - .expect("create fake lock file"); - file.write_all(contents).expect("write fake lock file"); - file - } - - pub(crate) fn identity(file: &File) -> OwnershipLockIdentity { - OwnershipLockIdentity::from_metadata(&file.metadata().expect("read file metadata")) - } - - pub(crate) fn snapshot(path: &Path) -> FileSnapshot { - let file = File::open(path).expect("open file for a read-only snapshot"); - let metadata = file.metadata().expect("read file snapshot metadata"); - let contents = fs::read(path).expect("read file snapshot contents"); - - FileSnapshot { - identity: OwnershipLockIdentity::from_metadata(&metadata), - contents, - length: metadata.len(), - modified: metadata.modified().ok(), - } - } - - pub(crate) fn directory_entries(path: &Path) -> Vec { - let mut entries = fs::read_dir(path) - .expect("read runtime root entries") - .map(|entry| entry.expect("read runtime root entry").path()) - .collect::>(); - entries.sort(); - entries - } - - pub(crate) fn try_fake_lock_process(path: &Path) -> String { - let output = Command::new(env::current_exe().expect("find current test binary")) - .arg("--exact") - .arg(HELPER_TEST) - .arg("--nocapture") - .env(CHILD_PATH_ENV, path) - .env(CHILD_MODE_ENV, "try") - .stdin(Stdio::null()) - .stdout(Stdio::piped()) - .stderr(Stdio::piped()) - .output() - .expect("run fake nonblocking lock attempt"); - assert!( - output.status.success(), - "fake lock attempt failed: {}", - String::from_utf8_lossy(&output.stderr) - ); - - String::from_utf8(output.stdout).expect("fake lock helper output is UTF-8") - } -} diff --git a/src/resource/ownership_lock/attempt_evidence.rs b/src/resource/ownership_lock/attempt_evidence.rs deleted file mode 100644 index 68885ef..0000000 --- a/src/resource/ownership_lock/attempt_evidence.rs +++ /dev/null @@ -1,1071 +0,0 @@ -//! Point-in-time evidence that one live direct-segment trainer attempt holds its lock - -use std::fs::{self, File, OpenOptions}; -use std::io::{self, Read}; -use std::os::fd::AsFd; -use std::os::unix::fs::{MetadataExt, OpenOptionsExt}; -use std::path::{Path, PathBuf}; - -use serde::{Deserialize, Serialize}; -use thiserror::Error; - -use super::SEGMENT_LOCK_FILE; -use super::lock_probe::{ - OwnershipLockIdentity, OwnershipLockProbe, OwnershipLockProbeError, - probe_segment_ownership_lock, -}; -use crate::digest::Sha256Digest; -use crate::resource::trainer_publication::{ - AttemptBinding, AttemptRequestValidationError, WatcherError, validate_attempt_request, -}; - -/// Why a trainer attempt path is not safe to use as registration evidence -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum TrainerAttemptUnsafePathReason { - /// A path component is a symbolic link - SymbolicLink, - /// A path expected to be a directory is not a directory - NotDirectory, - /// A path expected to be a regular file is not a regular file - NotRegularFile, -} - -/// A typed failure while collecting read-only trainer attempt evidence -#[derive(Debug, Error)] -pub enum TrainerAttemptEvidenceError { - /// A required runtime, attempt, request, or lock path is missing - #[error("trainer attempt evidence path is missing: {path}")] - Missing { - /// Missing path - path: PathBuf, - }, - /// The supplied runtime path is not its exact canonical absolute path - #[error("runtime root is not canonical: supplied {provided}, canonical {canonical}")] - NonCanonicalRuntimeRoot { - /// Runtime root supplied by the caller - provided: PathBuf, - /// Canonical runtime root observed on disk - canonical: PathBuf, - }, - /// A path has an unsafe filesystem type - #[error("unsafe trainer attempt path {path}: {reason:?}")] - UnsafePath { - /// Unsafe path - path: PathBuf, - /// Filesystem type that makes the path unsafe - reason: TrainerAttemptUnsafePathReason, - }, - /// A filesystem operation failed while collecting evidence - #[error("cannot inspect trainer attempt evidence path {path}: {source}")] - Io { - /// Path involved in the failed operation - path: PathBuf, - /// Underlying filesystem error - #[source] - source: io::Error, - }, - /// The expected trainer binding is invalid - #[error("invalid expected trainer attempt binding: {reason}")] - InvalidExpectedBinding { - /// Binding validation reason - reason: &'static str, - }, - /// The request file does not match the maintained trainer request contract - #[error("malformed trainer request at {path}: {reason}")] - MalformedRequest { - /// Exact request path - path: PathBuf, - /// Projection or schema validation reason - reason: String, - }, - /// The request is valid but belongs to a different trainer attempt - #[error("trainer request binding at {path} does not match the expected attempt")] - BindingMismatch { - /// Exact request path - path: PathBuf, - /// Expected attempt identity - expected: Box, - /// Attempt identity persisted in the request - observed: Box, - }, - /// The request changed while evidence was being collected - #[error("trainer request changed during evidence collection: {path}")] - RequestChanged { - /// Exact request path - path: PathBuf, - }, - /// The ownership lock was free while evidence was collected - #[error("trainer ownership lock is not held at probe time: {path}")] - OwnershipLockFree { - /// Exact runtime lock path - path: PathBuf, - }, - /// The lock path changed after the held-lock observation - #[error("trainer ownership lock changed during evidence collection: {path}")] - OwnershipLockChanged { - /// Exact runtime lock path - path: PathBuf, - }, - /// The existing ownership-lock probe could not establish a safe result - #[error("cannot verify trainer ownership lock: {source}")] - OwnershipLock { - /// Typed error from the existing lock probe - #[source] - source: OwnershipLockProbeError, - }, -} - -/// SHA-256 digest of the exact bytes in one validated trainer request -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] -#[serde(transparent)] -pub struct TrainerRequestDigest(Sha256Digest); - -impl TrainerRequestDigest { - /// Hash the exact persisted request bytes - pub(crate) fn of(request_bytes: &[u8]) -> Self { - Self(Sha256Digest::of(request_bytes)) - } - - /// Return the digest as 64 lowercase hexadecimal characters - #[must_use] - pub fn to_hex(self) -> String { - self.0.to_hex() - } - - pub(crate) fn from_hex(value: &str) -> Option { - Sha256Digest::from_hex(value).ok().map(Self) - } -} - -/// Point-in-time evidence for registering one active direct-segment trainer attempt -/// -/// This binds the canonical runtime root, strict trainer `AttemptBinding`, exact -/// request bytes, and the device/inode of a lock observed as held. It does not bind -/// the exact Homebased `TaskId` or normalized command spec; the later authority-owned -/// registration must validate both. This evidence is not proof that the GPU has -/// been released and must not be used to mark a resource free. The lock observation -/// is point-in-time; the lock may be released after this value is returned -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct VerifiedTrainerAttempt { - canonical_runtime_root: PathBuf, - binding: AttemptBinding, - request_digest: TrainerRequestDigest, - ownership_lock_identity: OwnershipLockIdentity, -} - -impl VerifiedTrainerAttempt { - /// Return the canonical direct-segment runtime root - #[must_use] - pub fn canonical_runtime_root(&self) -> &Path { - &self.canonical_runtime_root - } - - /// Return the exact trainer attempt binding from the validated request - #[must_use] - pub const fn binding(&self) -> &AttemptBinding { - &self.binding - } - - /// Return the digest of the exact persisted request bytes - #[must_use] - pub const fn request_digest(&self) -> TrainerRequestDigest { - self.request_digest - } - - /// Return the device and inode of the lock observed as held - #[must_use] - pub const fn ownership_lock_identity(&self) -> OwnershipLockIdentity { - self.ownership_lock_identity - } - - pub(crate) fn from_persisted( - canonical_runtime_root: PathBuf, - binding: AttemptBinding, - request_digest: TrainerRequestDigest, - ownership_lock_identity: OwnershipLockIdentity, - ) -> Result { - if !canonical_runtime_root.is_absolute() - || canonical_runtime_root.components().any(|component| { - matches!( - component, - std::path::Component::CurDir | std::path::Component::ParentDir - ) - }) - { - return Err("saved trainer runtime root is not a canonical absolute path"); - } - binding - .validate() - .map_err(|_| "saved trainer attempt binding is invalid")?; - - Ok(Self { - canonical_runtime_root, - binding, - request_digest, - ownership_lock_identity, - }) - } -} - -/// Build point-in-time registration evidence for one existing active trainer attempt -/// -/// The runtime root must be an absolute, exact canonical path. The builder reads -/// only `attempts//request.json` and probes the existing `.segment.lock` -/// It does not create or write trainer artifacts. It returns evidence only if the -/// request is stable and valid and the exact lock is held at probe time. The result -/// does not prove GPU release. Later resource-aware registration must also bind the -/// exact Homebased `TaskId` and normalized command spec. The lock API reports -/// contention but does not identify its owner, so the caller must not hold this lock -/// itself. The intended lock holder is the trainer process -pub fn build_trainer_attempt_registration_evidence( - runtime_root: &Path, - expected_binding: &AttemptBinding, -) -> Result { - build_trainer_attempt_registration_evidence_inner(runtime_root, expected_binding, || {}) -} - -fn build_trainer_attempt_registration_evidence_inner( - runtime_root: &Path, - expected_binding: &AttemptBinding, - after_request_read: F, -) -> Result -where - F: FnOnce(), -{ - expected_binding.validate().map_err(|error| match error { - WatcherError::InvalidAttemptBinding { reason } => { - TrainerAttemptEvidenceError::InvalidExpectedBinding { reason } - } - _ => TrainerAttemptEvidenceError::InvalidExpectedBinding { - reason: "trainer attempt binding is invalid", - }, - })?; - - let (canonical_runtime_root, root_directory, root_identity) = - open_canonical_runtime_root(runtime_root)?; - let attempts_path = canonical_runtime_root.join("attempts"); - let attempts_directory = open_directory_at(&root_directory, "attempts", &attempts_path)?; - let attempt_path = attempts_path.join(&expected_binding.attempt_id); - let attempt_directory = open_directory_at( - &attempts_directory, - &expected_binding.attempt_id, - &attempt_path, - )?; - let request_path = attempt_path.join("request.json"); - let request_file = open_request_at(&attempt_directory, &request_path)?; - let request_state = stable_file_state(&request_file, &request_path)?; - - let mut request_bytes = Vec::new(); - let mut reader = &request_file; - reader - .read_to_end(&mut request_bytes) - .map_err(|source| TrainerAttemptEvidenceError::Io { - path: request_path.clone(), - source, - })?; - if stable_file_state(&request_file, &request_path)? != request_state { - return Err(TrainerAttemptEvidenceError::RequestChanged { path: request_path }); - } - - after_request_read(); - verify_request_path(&attempt_directory, &request_path, request_state)?; - - match validate_attempt_request(&request_bytes, expected_binding) { - Ok(()) => {} - Err(AttemptRequestValidationError::InvalidExpectedBinding(reason)) => { - return Err(TrainerAttemptEvidenceError::InvalidExpectedBinding { reason }); - } - Err(AttemptRequestValidationError::Malformed(reason)) => { - return Err(TrainerAttemptEvidenceError::MalformedRequest { - path: request_path, - reason, - }); - } - Err(AttemptRequestValidationError::BindingMismatch(observed)) => { - return Err(TrainerAttemptEvidenceError::BindingMismatch { - path: request_path, - expected: Box::new(expected_binding.clone()), - observed, - }); - } - } - - let lock_path = canonical_runtime_root.join(SEGMENT_LOCK_FILE); - let lock_metadata = fs::symlink_metadata(&lock_path).map_err(|source| { - if source.kind() == io::ErrorKind::NotFound { - return TrainerAttemptEvidenceError::Missing { - path: lock_path.clone(), - }; - } - - TrainerAttemptEvidenceError::Io { - path: lock_path.clone(), - source, - } - })?; - if lock_metadata.file_type().is_symlink() { - return Err(TrainerAttemptEvidenceError::UnsafePath { - path: lock_path, - reason: TrainerAttemptUnsafePathReason::SymbolicLink, - }); - } - if !lock_metadata.is_file() { - return Err(TrainerAttemptEvidenceError::UnsafePath { - path: lock_path, - reason: TrainerAttemptUnsafePathReason::NotRegularFile, - }); - } - - let lock_identity = OwnershipLockIdentity::from_metadata(&lock_metadata); - match probe_segment_ownership_lock(&canonical_runtime_root, lock_identity) { - OwnershipLockProbe::OwnershipHeld => {} - OwnershipLockProbe::ExactOwnershipReleased(guard) => { - drop(guard); - return Err(TrainerAttemptEvidenceError::OwnershipLockFree { path: lock_path }); - } - OwnershipLockProbe::Attention(source) => { - return Err(TrainerAttemptEvidenceError::OwnershipLock { source }); - } - } - - verify_request_path(&attempt_directory, &request_path, request_state)?; - verify_directory_name( - &root_directory, - "attempts", - &attempts_path, - &attempts_directory, - )?; - verify_directory_name( - &attempts_directory, - &expected_binding.attempt_id, - &attempt_path, - &attempt_directory, - )?; - verify_runtime_root_identity(&canonical_runtime_root, &root_directory, root_identity)?; - verify_lock_path_identity(&lock_path, lock_identity)?; - - Ok(VerifiedTrainerAttempt { - canonical_runtime_root, - binding: expected_binding.clone(), - request_digest: TrainerRequestDigest::of(&request_bytes), - ownership_lock_identity: lock_identity, - }) -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -struct StableFileState { - identity: OwnershipLockIdentity, - length: u64, - mode: u32, - modified_seconds: i64, - modified_nanoseconds: i64, - changed_seconds: i64, - changed_nanoseconds: i64, -} - -fn open_canonical_runtime_root( - runtime_root: &Path, -) -> Result<(PathBuf, File, OwnershipLockIdentity), TrainerAttemptEvidenceError> { - if !runtime_root.is_absolute() { - let canonical_runtime_root = runtime_root.canonicalize().map_err(|source| { - if source.kind() == io::ErrorKind::NotFound { - return TrainerAttemptEvidenceError::Missing { - path: runtime_root.to_path_buf(), - }; - } - - TrainerAttemptEvidenceError::Io { - path: runtime_root.to_path_buf(), - source, - } - })?; - return Err(TrainerAttemptEvidenceError::NonCanonicalRuntimeRoot { - provided: runtime_root.to_path_buf(), - canonical: canonical_runtime_root, - }); - } - - let root_metadata = fs::symlink_metadata(runtime_root).map_err(|source| { - if source.kind() == io::ErrorKind::NotFound { - return TrainerAttemptEvidenceError::Missing { - path: runtime_root.to_path_buf(), - }; - } - - TrainerAttemptEvidenceError::Io { - path: runtime_root.to_path_buf(), - source, - } - })?; - if root_metadata.file_type().is_symlink() { - return Err(TrainerAttemptEvidenceError::UnsafePath { - path: runtime_root.to_path_buf(), - reason: TrainerAttemptUnsafePathReason::SymbolicLink, - }); - } - if !root_metadata.is_dir() { - return Err(TrainerAttemptEvidenceError::UnsafePath { - path: runtime_root.to_path_buf(), - reason: TrainerAttemptUnsafePathReason::NotDirectory, - }); - } - - let canonical_runtime_root = - runtime_root - .canonicalize() - .map_err(|source| TrainerAttemptEvidenceError::Io { - path: runtime_root.to_path_buf(), - source, - })?; - if canonical_runtime_root.as_os_str() != runtime_root.as_os_str() { - return Err(TrainerAttemptEvidenceError::NonCanonicalRuntimeRoot { - provided: runtime_root.to_path_buf(), - canonical: canonical_runtime_root, - }); - } - - let root_directory = OpenOptions::new() - .read(true) - .custom_flags(nix::libc::O_CLOEXEC | nix::libc::O_DIRECTORY | nix::libc::O_NOFOLLOW) - .open(runtime_root) - .map_err(|source| TrainerAttemptEvidenceError::Io { - path: runtime_root.to_path_buf(), - source, - })?; - let opened_metadata = - root_directory - .metadata() - .map_err(|source| TrainerAttemptEvidenceError::Io { - path: runtime_root.to_path_buf(), - source, - })?; - let root_identity = OwnershipLockIdentity::from_metadata(&root_metadata); - if OwnershipLockIdentity::from_metadata(&opened_metadata) != root_identity { - return Err(TrainerAttemptEvidenceError::RequestChanged { - path: runtime_root.to_path_buf(), - }); - } - - Ok((canonical_runtime_root, root_directory, root_identity)) -} - -fn open_directory_at( - parent: &File, - name: &str, - path: &Path, -) -> Result { - let descriptor = nix::fcntl::openat( - parent.as_fd(), - name, - nix::fcntl::OFlag::O_CLOEXEC - | nix::fcntl::OFlag::O_DIRECTORY - | nix::fcntl::OFlag::O_NOFOLLOW - | nix::fcntl::OFlag::O_RDONLY, - nix::sys::stat::Mode::empty(), - ) - .map_err(|error| open_path_error(path, error, TrainerAttemptUnsafePathReason::NotDirectory))?; - let directory = File::from(descriptor); - let metadata = directory - .metadata() - .map_err(|source| TrainerAttemptEvidenceError::Io { - path: path.to_path_buf(), - source, - })?; - if !metadata.is_dir() { - return Err(TrainerAttemptEvidenceError::UnsafePath { - path: path.to_path_buf(), - reason: TrainerAttemptUnsafePathReason::NotDirectory, - }); - } - - Ok(directory) -} - -fn open_request_at( - attempt_directory: &File, - request_path: &Path, -) -> Result { - let descriptor = nix::fcntl::openat( - attempt_directory.as_fd(), - "request.json", - nix::fcntl::OFlag::O_CLOEXEC - | nix::fcntl::OFlag::O_NOFOLLOW - | nix::fcntl::OFlag::O_NONBLOCK - | nix::fcntl::OFlag::O_RDONLY, - nix::sys::stat::Mode::empty(), - ) - .map_err(|error| { - open_path_error( - request_path, - error, - TrainerAttemptUnsafePathReason::NotRegularFile, - ) - })?; - let request_file = File::from(descriptor); - if !request_file - .metadata() - .map_err(|source| TrainerAttemptEvidenceError::Io { - path: request_path.to_path_buf(), - source, - })? - .is_file() - { - return Err(TrainerAttemptEvidenceError::UnsafePath { - path: request_path.to_path_buf(), - reason: TrainerAttemptUnsafePathReason::NotRegularFile, - }); - } - - Ok(request_file) -} - -fn stable_file_state( - file: &File, - path: &Path, -) -> Result { - let metadata = file - .metadata() - .map_err(|source| TrainerAttemptEvidenceError::Io { - path: path.to_path_buf(), - source, - })?; - if !metadata.is_file() { - return Err(TrainerAttemptEvidenceError::UnsafePath { - path: path.to_path_buf(), - reason: TrainerAttemptUnsafePathReason::NotRegularFile, - }); - } - - Ok(StableFileState { - identity: OwnershipLockIdentity::from_metadata(&metadata), - length: metadata.len(), - mode: metadata.mode(), - modified_seconds: metadata.mtime(), - modified_nanoseconds: metadata.mtime_nsec(), - changed_seconds: metadata.ctime(), - changed_nanoseconds: metadata.ctime_nsec(), - }) -} - -fn verify_request_path( - attempt_directory: &File, - request_path: &Path, - expected_state: StableFileState, -) -> Result<(), TrainerAttemptEvidenceError> { - let current_request = open_request_at(attempt_directory, request_path)?; - if stable_file_state(¤t_request, request_path)? != expected_state { - return Err(TrainerAttemptEvidenceError::RequestChanged { - path: request_path.to_path_buf(), - }); - } - - Ok(()) -} - -fn verify_directory_name( - parent: &File, - name: &str, - path: &Path, - expected_directory: &File, -) -> Result<(), TrainerAttemptEvidenceError> { - let current_directory = open_directory_at(parent, name, path)?; - let current_metadata = - current_directory - .metadata() - .map_err(|source| TrainerAttemptEvidenceError::Io { - path: path.to_path_buf(), - source, - })?; - let expected_metadata = - expected_directory - .metadata() - .map_err(|source| TrainerAttemptEvidenceError::Io { - path: path.to_path_buf(), - source, - })?; - if OwnershipLockIdentity::from_metadata(¤t_metadata) - != OwnershipLockIdentity::from_metadata(&expected_metadata) - { - return Err(TrainerAttemptEvidenceError::RequestChanged { - path: path.to_path_buf(), - }); - } - - Ok(()) -} - -fn verify_runtime_root_identity( - runtime_root: &Path, - root_directory: &File, - expected_identity: OwnershipLockIdentity, -) -> Result<(), TrainerAttemptEvidenceError> { - let root_metadata = fs::symlink_metadata(runtime_root).map_err(|source| { - if source.kind() == io::ErrorKind::NotFound { - return TrainerAttemptEvidenceError::Missing { - path: runtime_root.to_path_buf(), - }; - } - - TrainerAttemptEvidenceError::Io { - path: runtime_root.to_path_buf(), - source, - } - })?; - if root_metadata.file_type().is_symlink() { - return Err(TrainerAttemptEvidenceError::UnsafePath { - path: runtime_root.to_path_buf(), - reason: TrainerAttemptUnsafePathReason::SymbolicLink, - }); - } - if !root_metadata.is_dir() { - return Err(TrainerAttemptEvidenceError::UnsafePath { - path: runtime_root.to_path_buf(), - reason: TrainerAttemptUnsafePathReason::NotDirectory, - }); - } - let opened_metadata = - root_directory - .metadata() - .map_err(|source| TrainerAttemptEvidenceError::Io { - path: runtime_root.to_path_buf(), - source, - })?; - let canonical_runtime_root = - runtime_root - .canonicalize() - .map_err(|source| TrainerAttemptEvidenceError::Io { - path: runtime_root.to_path_buf(), - source, - })?; - if canonical_runtime_root.as_os_str() != runtime_root.as_os_str() { - return Err(TrainerAttemptEvidenceError::NonCanonicalRuntimeRoot { - provided: runtime_root.to_path_buf(), - canonical: canonical_runtime_root, - }); - } - if OwnershipLockIdentity::from_metadata(&root_metadata) != expected_identity - || OwnershipLockIdentity::from_metadata(&opened_metadata) != expected_identity - { - return Err(TrainerAttemptEvidenceError::RequestChanged { - path: runtime_root.to_path_buf(), - }); - } - - Ok(()) -} - -fn verify_lock_path_identity( - lock_path: &Path, - expected_identity: OwnershipLockIdentity, -) -> Result<(), TrainerAttemptEvidenceError> { - let metadata = fs::symlink_metadata(lock_path).map_err(|source| { - if source.kind() == io::ErrorKind::NotFound { - return TrainerAttemptEvidenceError::Missing { - path: lock_path.to_path_buf(), - }; - } - - TrainerAttemptEvidenceError::Io { - path: lock_path.to_path_buf(), - source, - } - })?; - if metadata.file_type().is_symlink() { - return Err(TrainerAttemptEvidenceError::UnsafePath { - path: lock_path.to_path_buf(), - reason: TrainerAttemptUnsafePathReason::SymbolicLink, - }); - } - if !metadata.is_file() { - return Err(TrainerAttemptEvidenceError::UnsafePath { - path: lock_path.to_path_buf(), - reason: TrainerAttemptUnsafePathReason::NotRegularFile, - }); - } - if OwnershipLockIdentity::from_metadata(&metadata) != expected_identity { - return Err(TrainerAttemptEvidenceError::OwnershipLockChanged { - path: lock_path.to_path_buf(), - }); - } - - Ok(()) -} - -fn open_path_error( - path: &Path, - error: nix::errno::Errno, - not_regular_reason: TrainerAttemptUnsafePathReason, -) -> TrainerAttemptEvidenceError { - let raw_error = error as i32; - if error == nix::errno::Errno::ENOENT { - return TrainerAttemptEvidenceError::Missing { - path: path.to_path_buf(), - }; - } - if error == nix::errno::Errno::ELOOP { - return TrainerAttemptEvidenceError::UnsafePath { - path: path.to_path_buf(), - reason: TrainerAttemptUnsafePathReason::SymbolicLink, - }; - } - if error == nix::errno::Errno::ENOTDIR || error == nix::errno::Errno::EISDIR { - return TrainerAttemptEvidenceError::UnsafePath { - path: path.to_path_buf(), - reason: not_regular_reason, - }; - } - - TrainerAttemptEvidenceError::Io { - path: path.to_path_buf(), - source: io::Error::from_raw_os_error(raw_error), - } -} - -#[cfg(test)] -mod tests { - use std::fs::{self, File}; - use std::path::{Path, PathBuf}; - - use serde_json::{Value, json}; - use tempfile::TempDir; - - use super::super::test_support::{ - FileSnapshot, LOCK_ACQUIRED, create_lock, identity, lock_path, snapshot, - start_fake_lock_process, try_fake_lock_process, - }; - use super::{ - AttemptBinding, OwnershipLockIdentity, TrainerAttemptEvidenceError, - TrainerAttemptUnsafePathReason, VerifiedTrainerAttempt, - build_trainer_attempt_registration_evidence, - build_trainer_attempt_registration_evidence_inner, - }; - use crate::digest::Sha256Digest; - - #[derive(Debug, PartialEq, Eq)] - enum TreeSnapshotEntry { - Directory(OwnershipLockIdentity), - File(FileSnapshot), - Symlink(PathBuf), - } - - fn expected_binding() -> AttemptBinding { - AttemptBinding { - campaign_id: "campaign-a".into(), - campaign_revision_id: "revision-a".into(), - task_id: "trainer-task-a".into(), - attempt_id: "attempt-a".into(), - attempt_number: 1, - ownership_token: "owner-a".into(), - } - } - - fn request_value(binding: &AttemptBinding, epochs: u64, schema_version: u64) -> Value { - json!({ - "schema_version": schema_version, - "binding": binding, - "adapter": {"kind": "speakrs"}, - "start_mode": {"kind": "fresh"}, - "payload": {"training": {"epochs": epochs}}, - "input_view": { - "schema_version": 1, - "revision_id": binding.campaign_revision_id, - "root": "/trainer/input", - }, - "inputs": [], - "expected_outputs": [{"path": "checkpoint.bin", "kind": "file"}], - "resources": {"accelerator": "cuda"}, - "worker": { - "worker_id": "trainer-worker", - "executable_digest": "a".repeat(64), - "source_digest": "b".repeat(64), - "environment_digest": "c".repeat(64), - }, - }) - } - - fn request_bytes(binding: &AttemptBinding) -> Vec { - serde_json::to_vec(&request_value(binding, 4, 1)).expect("serialize request fixture") - } - - fn create_attempt_runtime( - temp: &TempDir, - binding: &AttemptBinding, - request: Option<&[u8]>, - ) -> (PathBuf, PathBuf, File) { - let runtime_root = temp.path().join("runtime"); - let attempt_path = runtime_root.join("attempts").join(&binding.attempt_id); - fs::create_dir_all(&attempt_path).expect("create trainer attempt directory"); - let lock = create_lock(&runtime_root, b"preserve trainer lock bytes"); - if let Some(request) = request { - fs::write(attempt_path.join("request.json"), request) - .expect("write trainer request fixture"); - } - - let runtime_root = fs::canonicalize(runtime_root).expect("canonicalize runtime root"); - let attempt_path = runtime_root.join("attempts").join(&binding.attempt_id); - (runtime_root, attempt_path, lock) - } - - fn tree_snapshot(root: &Path) -> Vec<(PathBuf, TreeSnapshotEntry)> { - fn visit(root: &Path, directory: &Path, entries: &mut Vec<(PathBuf, TreeSnapshotEntry)>) { - for entry in fs::read_dir(directory).expect("read runtime tree") { - let path = entry.expect("read runtime entry").path(); - let relative = path - .strip_prefix(root) - .expect("runtime member is contained") - .to_path_buf(); - let metadata = fs::symlink_metadata(&path).expect("inspect runtime entry"); - if metadata.file_type().is_symlink() { - entries.push(( - relative, - TreeSnapshotEntry::Symlink( - fs::read_link(&path).expect("read runtime symlink target"), - ), - )); - } else if metadata.is_dir() { - entries.push(( - relative, - TreeSnapshotEntry::Directory(OwnershipLockIdentity::from_metadata( - &metadata, - )), - )); - visit(root, &path, entries); - } else { - entries.push((relative, TreeSnapshotEntry::File(snapshot(&path)))); - } - } - } - - let mut entries = Vec::new(); - visit(root, root, &mut entries); - entries.sort_by(|left, right| left.0.cmp(&right.0)); - entries - } - - fn assert_attempt_evidence( - evidence: &VerifiedTrainerAttempt, - runtime_root: &Path, - binding: &AttemptBinding, - request: &[u8], - lock_identity: OwnershipLockIdentity, - ) { - assert_eq!(evidence.canonical_runtime_root(), runtime_root); - assert_eq!(evidence.binding(), binding); - assert_eq!( - evidence.request_digest().to_hex(), - Sha256Digest::of(request).to_hex() - ); - assert_eq!(evidence.ownership_lock_identity(), lock_identity); - } - - #[test] - fn exact_live_attempt_returns_immutable_read_only_evidence() { - let temp = tempfile::tempdir().expect("create temporary directory"); - let binding = expected_binding(); - let request = request_bytes(&binding); - let (runtime_root, attempt_path, lock) = - create_attempt_runtime(&temp, &binding, Some(&request)); - let expected_identity = identity(&lock); - let before = tree_snapshot(&runtime_root); - let mut owner = start_fake_lock_process(&lock_path(&runtime_root)); - - let evidence = build_trainer_attempt_registration_evidence(&runtime_root, &binding) - .expect("collect exact active attempt evidence"); - - assert_attempt_evidence( - &evidence, - &runtime_root, - &binding, - &request, - expected_identity, - ); - assert_eq!( - fs::read(attempt_path.join("request.json")).expect("read unchanged request"), - request - ); - assert_eq!(tree_snapshot(&runtime_root), before); - - owner.release(); - } - - #[test] - fn wrong_attempt_and_wrong_token_need_attention() { - for observed in [ - AttemptBinding { - attempt_id: "attempt-other".into(), - ..expected_binding() - }, - AttemptBinding { - ownership_token: "owner-other".into(), - ..expected_binding() - }, - ] { - let temp = tempfile::tempdir().expect("create temporary directory"); - let expected = expected_binding(); - let request = request_bytes(&observed); - let (runtime_root, _attempt_path, _lock) = - create_attempt_runtime(&temp, &expected, Some(&request)); - let before = tree_snapshot(&runtime_root); - let mut owner = start_fake_lock_process(&lock_path(&runtime_root)); - - assert!(matches!( - build_trainer_attempt_registration_evidence(&runtime_root, &expected), - Err(TrainerAttemptEvidenceError::BindingMismatch { - expected: actual_expected, - observed: actual_observed, - .. - }) if actual_expected.as_ref() == &expected - && actual_observed.as_ref() == &observed - )); - assert_eq!(tree_snapshot(&runtime_root), before); - - owner.release(); - } - } - - #[test] - fn malformed_json_and_unsupported_request_schema_need_attention() { - let binding = expected_binding(); - for request in [ - b"{not valid JSON".to_vec(), - serde_json::to_vec(&request_value(&binding, 4, 2)) - .expect("serialize unsupported schema fixture"), - ] { - let temp = tempfile::tempdir().expect("create temporary directory"); - let (runtime_root, _attempt_path, _lock) = - create_attempt_runtime(&temp, &binding, Some(&request)); - let before = tree_snapshot(&runtime_root); - - assert!(matches!( - build_trainer_attempt_registration_evidence(&runtime_root, &binding), - Err(TrainerAttemptEvidenceError::MalformedRequest { .. }) - )); - assert_eq!(tree_snapshot(&runtime_root), before); - } - } - - #[test] - fn missing_request_needs_attention_without_creating_it() { - let temp = tempfile::tempdir().expect("create temporary directory"); - let binding = expected_binding(); - let (runtime_root, attempt_path, _lock) = create_attempt_runtime(&temp, &binding, None); - let request_path = attempt_path.join("request.json"); - let before = tree_snapshot(&runtime_root); - - assert!(matches!( - build_trainer_attempt_registration_evidence(&runtime_root, &binding), - Err(TrainerAttemptEvidenceError::Missing { path }) if path == request_path - )); - assert_eq!(tree_snapshot(&runtime_root), before); - assert!(!request_path.exists()); - } - - #[test] - fn symlinked_request_needs_attention_without_touching_its_target() { - use std::os::unix::fs::symlink; - - let temp = tempfile::tempdir().expect("create temporary directory"); - let binding = expected_binding(); - let (runtime_root, attempt_path, _lock) = create_attempt_runtime(&temp, &binding, None); - let request_target = temp.path().join("request-target.json"); - fs::write(&request_target, request_bytes(&binding)).expect("write target fixture"); - let request_target_before = snapshot(&request_target); - let request_path = attempt_path.join("request.json"); - symlink(&request_target, &request_path).expect("create request symlink"); - let before = tree_snapshot(&runtime_root); - - assert!(matches!( - build_trainer_attempt_registration_evidence(&runtime_root, &binding), - Err(TrainerAttemptEvidenceError::UnsafePath { - path, - reason: TrainerAttemptUnsafePathReason::SymbolicLink, - }) if path == request_path - )); - assert_eq!(tree_snapshot(&runtime_root), before); - assert_eq!(snapshot(&request_target), request_target_before); - } - - #[test] - fn free_lock_needs_attention_without_creating_or_holding_a_lock() { - let temp = tempfile::tempdir().expect("create temporary directory"); - let binding = expected_binding(); - let request = request_bytes(&binding); - let (runtime_root, _attempt_path, _lock) = - create_attempt_runtime(&temp, &binding, Some(&request)); - let lock_path = lock_path(&runtime_root); - let before = tree_snapshot(&runtime_root); - - assert!(matches!( - build_trainer_attempt_registration_evidence(&runtime_root, &binding), - Err(TrainerAttemptEvidenceError::OwnershipLockFree { path }) if path == lock_path - )); - assert_eq!(tree_snapshot(&runtime_root), before); - assert!(try_fake_lock_process(&lock_path).contains(LOCK_ACQUIRED)); - assert_eq!(tree_snapshot(&runtime_root), before); - } - - #[test] - fn noncanonical_runtime_root_needs_attention() { - let temp = tempfile::tempdir().expect("create temporary directory"); - let binding = expected_binding(); - let (runtime_root, _attempt_path, _lock) = create_attempt_runtime(&temp, &binding, None); - let noncanonical_root = runtime_root.join("..").join("runtime"); - let before = tree_snapshot(&runtime_root); - - assert!(matches!( - build_trainer_attempt_registration_evidence(&noncanonical_root, &binding), - Err(TrainerAttemptEvidenceError::NonCanonicalRuntimeRoot { - provided, - canonical, - }) if provided == noncanonical_root && canonical == runtime_root - )); - assert_eq!(tree_snapshot(&runtime_root), before); - } - - #[test] - fn relative_runtime_root_reports_its_canonical_path() { - let relative_root = PathBuf::from("."); - let canonical_root = fs::canonicalize(&relative_root).expect("canonicalize current dir"); - - assert!(matches!( - build_trainer_attempt_registration_evidence(&relative_root, &expected_binding()), - Err(TrainerAttemptEvidenceError::NonCanonicalRuntimeRoot { - provided, - canonical, - }) if provided == relative_root && canonical == canonical_root - )); - } - - #[test] - fn request_replacement_during_collection_needs_attention() { - let temp = tempfile::tempdir().expect("create temporary directory"); - let binding = expected_binding(); - let original_request = request_bytes(&binding); - let (runtime_root, attempt_path, lock) = - create_attempt_runtime(&temp, &binding, Some(&original_request)); - let mut owner = start_fake_lock_process(&lock_path(&runtime_root)); - let request_path = attempt_path.join("request.json"); - let replacement_path = attempt_path.join("request.replacement"); - let replacement_request = serde_json::to_vec(&request_value(&binding, 5, 1)) - .expect("serialize replacement request"); - - let result = - build_trainer_attempt_registration_evidence_inner(&runtime_root, &binding, || { - fs::write(&replacement_path, &replacement_request) - .expect("write replacement fixture"); - fs::rename(&replacement_path, &request_path) - .expect("replace request during evidence collection"); - }); - - assert!(matches!( - result, - Err(TrainerAttemptEvidenceError::RequestChanged { path }) if path == request_path - )); - assert_eq!( - fs::read(&request_path).expect("read replacement request"), - replacement_request - ); - owner.release(); - drop(lock); - } -} diff --git a/src/resource/ownership_lock/lock_probe.rs b/src/resource/ownership_lock/lock_probe.rs deleted file mode 100644 index 6eb7849..0000000 --- a/src/resource/ownership_lock/lock_probe.rs +++ /dev/null @@ -1,546 +0,0 @@ -//! Read-only probing of the exact direct-segment ownership lock - -use std::fs::{self, File, Metadata, OpenOptions, TryLockError}; -use std::io; -use std::os::unix::fs::{MetadataExt, OpenOptionsExt}; -use std::path::{Path, PathBuf}; - -use thiserror::Error; - -use super::SEGMENT_LOCK_FILE; - -/// Device and inode identity of one trainer ownership lock -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] -pub struct OwnershipLockIdentity { - device: u64, - inode: u64, -} - -impl OwnershipLockIdentity { - /// Create an identity from Unix device and inode values - #[must_use] - pub const fn new(device: u64, inode: u64) -> Self { - Self { device, inode } - } - - /// Capture the device and inode from an opened trainer lock file - #[must_use] - pub fn from_metadata(metadata: &Metadata) -> Self { - Self::new(metadata.dev(), metadata.ino()) - } - - /// Return the device number on Unix - #[must_use] - pub const fn device(self) -> u64 { - self.device - } - - /// Return the inode number on Unix - #[must_use] - pub const fn inode(self) -> u64 { - self.inode - } -} - -/// Reason an existing runtime lock path cannot prove the expected identity -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum OwnershipLockIdentityMismatchReason { - /// The `.segment.lock` path does not exist - Missing, - /// The `.segment.lock` path is a symbolic link - SymbolicLink, - /// The path does not name a regular file - NotRegularFile, - /// The path names a different device and inode - DifferentFile, - /// The supplied runtime root is a symbolic link - RuntimeRootSymbolicLink, - /// The supplied runtime root is not a directory - RuntimeRootNotDirectory, -} - -/// Filesystem or platform operation involved in an ownership-lock probe -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum OwnershipLockProbeOperation { - /// Inspect the supplied runtime root without following its final component - InspectRuntimeRoot, - /// Inspect the lock path without following its final component - InspectLockPath, - /// Open the existing lock file - OpenLockFile, - /// Inspect the opened lock descriptor - InspectOpenedFile, - /// Attempt the exclusive nonblocking lock - AcquireExclusiveLock, -} - -/// A typed reason that an ownership-lock probe needs attention -#[derive(Debug, Error)] -pub enum OwnershipLockProbeError { - /// A filesystem operation failed while probing the runtime lock - #[error("cannot {operation:?} for runtime ownership lock {path}: {source}")] - Io { - /// Operation that failed - operation: OwnershipLockProbeOperation, - /// Path involved in the failed operation - path: PathBuf, - /// Underlying operating system error - #[source] - source: io::Error, - }, - /// A lock path does not match the identity captured for this trainer attempt - #[error( - "runtime ownership lock identity mismatch at {path}: expected {expected:?}, observed {observed:?} ({reason:?})" - )] - IdentityMismatch { - /// Path that failed the identity check - path: PathBuf, - /// Exact lock identity saved for this trainer attempt - expected: OwnershipLockIdentity, - /// Identity observed at the path, when it names a filesystem object - observed: Option, - /// Why the path cannot be trusted as the saved lock - reason: OwnershipLockIdentityMismatchReason, - }, -} - -/// A guard that keeps the exact released lock exclusively held until it is dropped -#[must_use = "keep the guard alive through the release transaction"] -#[derive(Debug)] -pub struct OwnershipLockGuard { - file: File, -} - -impl OwnershipLockGuard { - /// Return the identity of the lock held by this guard - pub fn identity(&self) -> io::Result { - Ok(OwnershipLockIdentity::from_metadata(&self.file.metadata()?)) - } -} - -/// Conclusion from probing one exact runtime ownership lock -#[derive(Debug)] -pub enum OwnershipLockProbe { - /// The exact lock file is still held by a process - OwnershipHeld, - /// The exact lock was released and is now held exclusively by this result's guard - ExactOwnershipReleased(OwnershipLockGuard), - /// The probe could not establish ownership state safely - Attention(OwnershipLockProbeError), -} - -enum LockAttempt { - Held, - Acquired(File), -} - -/// Probe the existing `.segment.lock` for one previously bound trainer attempt -/// -/// The runtime root must be the canonical path saved with that attempt. The probe -/// does not create or write files. It returns a held guard only after the file's -/// device and inode match `expected` both before and after lock acquisition. This -/// proves only the state of this trainer lock, not that the physical GPU is free -#[must_use] -pub fn probe_segment_ownership_lock( - runtime_root: &Path, - expected: OwnershipLockIdentity, -) -> OwnershipLockProbe { - match probe_lock(runtime_root, expected) { - Ok(LockAttempt::Held) => OwnershipLockProbe::OwnershipHeld, - Ok(LockAttempt::Acquired(file)) => { - OwnershipLockProbe::ExactOwnershipReleased(OwnershipLockGuard { file }) - } - Err(error) => OwnershipLockProbe::Attention(error), - } -} - -/// Recheck that an exclusive guard still holds the exact lock named by its saved runtime root -/// -/// The check rejects a replaced, missing, or symbolic-link path while retaining the guard -pub fn verify_ownership_lock_guard( - runtime_root: &Path, - expected: OwnershipLockIdentity, - guard: &OwnershipLockGuard, -) -> Result<(), OwnershipLockProbeError> { - let lock_path = runtime_root.join(SEGMENT_LOCK_FILE); - let opened_identity = guard - .identity() - .map_err(|source| OwnershipLockProbeError::Io { - operation: OwnershipLockProbeOperation::InspectOpenedFile, - path: lock_path.clone(), - source, - })?; - if opened_identity != expected { - return Err(identity_mismatch( - &lock_path, - expected, - Some(opened_identity), - OwnershipLockIdentityMismatchReason::DifferentFile, - )); - } - - verify_runtime_root(runtime_root, expected)?; - verify_path_identity(&lock_path, expected) -} - -fn probe_lock( - runtime_root: &Path, - expected: OwnershipLockIdentity, -) -> Result { - let lock_path = runtime_root.join(SEGMENT_LOCK_FILE); - verify_runtime_root(runtime_root, expected)?; - verify_path_identity(&lock_path, expected)?; - - let file = OpenOptions::new() - .read(true) - .write(true) - .custom_flags(nix::libc::O_CLOEXEC | nix::libc::O_NOFOLLOW | nix::libc::O_NONBLOCK) - .open(&lock_path) - .map_err(|source| classify_open_error(&lock_path, expected, source))?; - - let opened_metadata = file - .metadata() - .map_err(|source| OwnershipLockProbeError::Io { - operation: OwnershipLockProbeOperation::InspectOpenedFile, - path: lock_path.clone(), - source, - })?; - let opened_identity = verify_metadata_identity(&lock_path, expected, &opened_metadata)?; - - // std file locks are flock(2) on Unix, the same lock the trainer takes - match file.try_lock() { - Ok(()) => { - verify_path_identity(&lock_path, opened_identity)?; - Ok(LockAttempt::Acquired(file)) - } - Err(TryLockError::WouldBlock) => { - verify_path_identity(&lock_path, opened_identity)?; - Ok(LockAttempt::Held) - } - Err(TryLockError::Error(source)) => Err(OwnershipLockProbeError::Io { - operation: OwnershipLockProbeOperation::AcquireExclusiveLock, - path: lock_path, - source, - }), - } -} - -fn verify_runtime_root( - runtime_root: &Path, - expected: OwnershipLockIdentity, -) -> Result<(), OwnershipLockProbeError> { - let metadata = - fs::symlink_metadata(runtime_root).map_err(|source| OwnershipLockProbeError::Io { - operation: OwnershipLockProbeOperation::InspectRuntimeRoot, - path: runtime_root.to_path_buf(), - source, - })?; - - if metadata.file_type().is_symlink() { - return Err(identity_mismatch( - runtime_root, - expected, - None, - OwnershipLockIdentityMismatchReason::RuntimeRootSymbolicLink, - )); - } - if !metadata.is_dir() { - return Err(identity_mismatch( - runtime_root, - expected, - None, - OwnershipLockIdentityMismatchReason::RuntimeRootNotDirectory, - )); - } - - Ok(()) -} - -fn verify_path_identity( - path: &Path, - expected: OwnershipLockIdentity, -) -> Result<(), OwnershipLockProbeError> { - let metadata = fs::symlink_metadata(path).map_err(|source| { - if source.kind() == io::ErrorKind::NotFound { - return identity_mismatch( - path, - expected, - None, - OwnershipLockIdentityMismatchReason::Missing, - ); - } - - OwnershipLockProbeError::Io { - operation: OwnershipLockProbeOperation::InspectLockPath, - path: path.to_path_buf(), - source, - } - })?; - - if metadata.file_type().is_symlink() { - return Err(identity_mismatch( - path, - expected, - None, - OwnershipLockIdentityMismatchReason::SymbolicLink, - )); - } - - verify_metadata_identity(path, expected, &metadata).map(|_| ()) -} - -fn verify_metadata_identity( - path: &Path, - expected: OwnershipLockIdentity, - metadata: &Metadata, -) -> Result { - let observed = OwnershipLockIdentity::from_metadata(metadata); - - if !metadata.is_file() { - return Err(identity_mismatch( - path, - expected, - Some(observed), - OwnershipLockIdentityMismatchReason::NotRegularFile, - )); - } - - if observed != expected { - return Err(identity_mismatch( - path, - expected, - Some(observed), - OwnershipLockIdentityMismatchReason::DifferentFile, - )); - } - - Ok(observed) -} - -fn classify_open_error( - path: &Path, - expected: OwnershipLockIdentity, - source: io::Error, -) -> OwnershipLockProbeError { - let reason = match source.kind() { - io::ErrorKind::NotFound => Some(OwnershipLockIdentityMismatchReason::Missing), - io::ErrorKind::IsADirectory => Some(OwnershipLockIdentityMismatchReason::NotRegularFile), - _ if source.raw_os_error() == Some(nix::libc::ELOOP) => { - Some(OwnershipLockIdentityMismatchReason::SymbolicLink) - } - _ => None, - }; - - if let Some(reason) = reason { - return identity_mismatch(path, expected, None, reason); - } - - OwnershipLockProbeError::Io { - operation: OwnershipLockProbeOperation::OpenLockFile, - path: path.to_path_buf(), - source, - } -} - -fn identity_mismatch( - path: &Path, - expected: OwnershipLockIdentity, - observed: Option, - reason: OwnershipLockIdentityMismatchReason, -) -> OwnershipLockProbeError { - OwnershipLockProbeError::IdentityMismatch { - path: path.to_path_buf(), - expected, - observed, - reason, - } -} - -#[cfg(test)] -mod tests { - use std::env; - use std::fs::{self, OpenOptions, TryLockError}; - use std::io::{Read, Write}; - - use super::super::test_support::{ - CHILD_MODE_ENV, CHILD_PATH_ENV, LOCK_ACQUIRED, LOCK_CONTENDED, create_lock, - directory_entries, identity, lock_path, snapshot, start_fake_lock_process, - try_fake_lock_process, - }; - use super::{ - OwnershipLockIdentity, OwnershipLockIdentityMismatchReason, OwnershipLockProbe, - OwnershipLockProbeError, probe_segment_ownership_lock, - }; - - #[test] - fn fake_process_lock_helper() -> Result<(), Box> { - let (Ok(path), Ok(mode)) = (env::var(CHILD_PATH_ENV), env::var(CHILD_MODE_ENV)) else { - return Ok(()); - }; - - let file = OpenOptions::new().read(true).write(true).open(path)?; - match file.try_lock() { - Ok(()) if mode == "hold" => { - writeln!(std::io::stdout(), "{LOCK_ACQUIRED}")?; - std::io::stdout().flush()?; - let mut release = [0_u8; 1]; - std::io::stdin().read_exact(&mut release)?; - } - Ok(()) => writeln!(std::io::stdout(), "{LOCK_ACQUIRED}")?, - Err(TryLockError::WouldBlock) => { - writeln!(std::io::stdout(), "{LOCK_CONTENDED}")?; - } - Err(TryLockError::Error(error)) => return Err(Box::new(error)), - } - - Ok(()) - } - - #[test] - fn separate_process_ownership_is_detected_and_released_guard_stays_exclusive() { - let temp = tempfile::tempdir().expect("create temporary directory"); - let runtime_root = temp.path().join("runtime"); - let file = create_lock(&runtime_root, b"preserve lock bytes"); - let expected = identity(&file); - let path = lock_path(&runtime_root); - let before_file = snapshot(&path); - let before_entries = directory_entries(&runtime_root); - let mut owner = start_fake_lock_process(&path); - - assert!(matches!( - probe_segment_ownership_lock(&runtime_root, expected), - OwnershipLockProbe::OwnershipHeld - )); - assert_eq!(snapshot(&path), before_file); - assert_eq!(directory_entries(&runtime_root), before_entries); - - owner.release(); - let guard = match probe_segment_ownership_lock(&runtime_root, expected) { - OwnershipLockProbe::ExactOwnershipReleased(guard) => guard, - outcome => panic!("expected an exclusive released-lock guard, got {outcome:?}"), - }; - assert_eq!(guard.identity().expect("read guard identity"), expected); - assert!(try_fake_lock_process(&path).contains(LOCK_CONTENDED)); - assert_eq!(snapshot(&path), before_file); - assert_eq!(directory_entries(&runtime_root), before_entries); - - drop(guard); - assert!(try_fake_lock_process(&path).contains(LOCK_ACQUIRED)); - assert_eq!(snapshot(&path), before_file); - assert_eq!(directory_entries(&runtime_root), before_entries); - } - - #[test] - fn replaced_lock_path_needs_attention_without_mutating_the_replacement() { - let temp = tempfile::tempdir().expect("create temporary directory"); - let runtime_root = temp.path().join("runtime"); - let old_file = create_lock(&runtime_root, b"old lock"); - let expected = identity(&old_file); - let path = lock_path(&runtime_root); - let replacement_path = runtime_root.join("replacement"); - let mut replacement = OpenOptions::new() - .create_new(true) - .read(true) - .write(true) - .open(&replacement_path) - .expect("create replacement file"); - replacement - .write_all(b"replacement bytes") - .expect("write replacement file"); - let replacement_identity = identity(&replacement); - assert_ne!(replacement_identity, expected); - fs::rename(&replacement_path, &path).expect("replace lock path atomically"); - let before_file = snapshot(&path); - let before_entries = directory_entries(&runtime_root); - - assert!(matches!( - probe_segment_ownership_lock(&runtime_root, expected), - OwnershipLockProbe::Attention(OwnershipLockProbeError::IdentityMismatch { - reason: OwnershipLockIdentityMismatchReason::DifferentFile, - .. - }) - )); - assert_eq!(snapshot(&path), before_file); - assert_eq!(directory_entries(&runtime_root), before_entries); - drop(old_file); - } - - #[test] - fn missing_lock_needs_attention_and_is_not_created() { - let temp = tempfile::tempdir().expect("create temporary directory"); - let runtime_root = temp.path().join("runtime"); - fs::create_dir(&runtime_root).expect("create runtime root"); - let expected = OwnershipLockIdentity::new(u64::MAX, u64::MAX); - let path = lock_path(&runtime_root); - - assert!(matches!( - probe_segment_ownership_lock(&runtime_root, expected), - OwnershipLockProbe::Attention(OwnershipLockProbeError::IdentityMismatch { - reason: OwnershipLockIdentityMismatchReason::Missing, - .. - }) - )); - assert!(!path.exists()); - assert!(directory_entries(&runtime_root).is_empty()); - } - - #[test] - fn symlinked_lock_needs_attention_without_opening_or_mutating_its_target() { - use std::os::unix::fs::symlink; - - let temp = tempfile::tempdir().expect("create temporary directory"); - let runtime_root = temp.path().join("runtime"); - fs::create_dir(&runtime_root).expect("create runtime root"); - let target = temp.path().join("target.lock"); - let mut target_file = OpenOptions::new() - .create_new(true) - .read(true) - .write(true) - .open(&target) - .expect("create symlink target"); - target_file - .write_all(b"target must not be touched") - .expect("write symlink target"); - let expected = identity(&target_file); - let path = lock_path(&runtime_root); - symlink(&target, &path).expect("create lock symlink"); - let before_target = snapshot(&target); - let before_entries = directory_entries(&runtime_root); - - assert!(matches!( - probe_segment_ownership_lock(&runtime_root, expected), - OwnershipLockProbe::Attention(OwnershipLockProbeError::IdentityMismatch { - reason: OwnershipLockIdentityMismatchReason::SymbolicLink, - .. - }) - )); - assert_eq!(snapshot(&target), before_target); - assert_eq!(directory_entries(&runtime_root), before_entries); - assert!( - fs::symlink_metadata(&path) - .expect("inspect original lock symlink") - .file_type() - .is_symlink() - ); - } - - #[test] - fn nonregular_lock_path_needs_attention() { - let temp = tempfile::tempdir().expect("create temporary directory"); - let runtime_root = temp.path().join("runtime"); - fs::create_dir(&runtime_root).expect("create runtime root"); - let path = lock_path(&runtime_root); - fs::create_dir(&path).expect("create nonregular lock path"); - let expected = OwnershipLockIdentity::new(u64::MAX, u64::MAX); - let before_entries = directory_entries(&runtime_root); - - assert!(matches!( - probe_segment_ownership_lock(&runtime_root, expected), - OwnershipLockProbe::Attention(OwnershipLockProbeError::IdentityMismatch { - reason: OwnershipLockIdentityMismatchReason::NotRegularFile, - .. - }) - )); - assert_eq!(directory_entries(&runtime_root), before_entries); - } -} diff --git a/src/resource/release_checkpoint.rs b/src/resource/release_checkpoint.rs deleted file mode 100644 index 3c34b17..0000000 --- a/src/resource/release_checkpoint.rs +++ /dev/null @@ -1,204 +0,0 @@ -//! Durable checkpoint evidence for one release action -//! -//! The authority saves a baseline of trainer publications when it binds the -//! release watcher, then a stop decision for one new verified checkpoint, then -//! the exact trainer cancellation. Each phase keeps every owner identity so a -//! restart or retry can check it against the saved release action - -use chrono::{DateTime, Utc}; -use serde::{Deserialize, Serialize}; - -use super::trainer_publication::{ - AttemptBinding, RecoverySnapshot, VerifiedCheckpointPublication, WatcherAttention, -}; -use super::{ - ActionId, ReleaseStopReservationId, ReleaseWatcherIntent, ResourceId, ResourceRevision, - TrainerAttemptAssociationProof, -}; -use crate::domain::TaskId; - -/// Durable release identity captured before the watcher task can act -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(crate) struct ReleaseCheckpointAction { - pub(crate) resource_id: ResourceId, - pub(crate) action_id: ActionId, - pub(crate) state_revision: ResourceRevision, - pub(crate) observed_background_task: TaskId, -} - -/// Full authority-owned identity that scopes one checkpoint baseline and stop decision -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(crate) struct ReleaseCheckpointBinding { - pub(crate) action: ReleaseCheckpointAction, - pub(crate) association: TrainerAttemptAssociationProof, - pub(crate) attempt_binding: AttemptBinding, - pub(crate) watcher_intent: ReleaseWatcherIntent, -} - -impl ReleaseCheckpointBinding { - pub(crate) fn validate_for( - &self, - action: &ReleaseCheckpointAction, - ) -> Result<(), &'static str> { - validate_release_checkpoint_binding(action, self) - } -} - -/// Checkpoint baseline captured from the saved trainer association -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(crate) struct ReleaseCheckpointBaseline { - pub(crate) binding: ReleaseCheckpointBinding, - pub(crate) snapshot: RecoverySnapshot, -} - -/// Stop decision reserved after the authority verified its exact checkpoint -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(crate) struct ReleaseCheckpointStopDecision { - pub(crate) binding: ReleaseCheckpointBinding, - pub(crate) reservation_id: ReleaseStopReservationId, - pub(crate) selected_checkpoint: VerifiedCheckpointPublication, -} - -/// Exact trainer task cancellation committed with its saved stop decision -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(crate) struct ReleaseCheckpointCancellation { - pub(crate) task_id: TaskId, - pub(crate) cancel_requested_at: DateTime, -} - -/// Durable checkpoint-evidence phase stored beside the resource loan -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub(crate) enum ReleaseCheckpointPhase { - /// New release action with no watcher identity yet - WatcherBindingPending, - /// Exact watcher identity is bound to a persisted pre-action baseline - BaselineCaptured { - /// Saved baseline and all immutable owner identities - baseline: ReleaseCheckpointBaseline, - }, - /// Exact-task stop request is durably reserved after checkpoint verification - StopReserved { - /// Baseline used to reject pre-existing publications - baseline: ReleaseCheckpointBaseline, - /// Fixed checkpoint selected by the authority - decision: Box, - }, - /// Exact trainer cancellation committed with the saved stop decision - CancellationCommitted { - /// Baseline used to reject pre-existing publications - baseline: ReleaseCheckpointBaseline, - /// Fixed checkpoint selected by the authority - decision: Box, - /// Exact task cancellation marker committed in the same transaction - cancellation: ReleaseCheckpointCancellation, - }, -} - -/// Durable evidence state for one exact AwaitingRelease action -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(crate) struct ReleaseCheckpointState { - pub(crate) action: ReleaseCheckpointAction, - pub(crate) phase: ReleaseCheckpointPhase, -} - -impl ReleaseCheckpointState { - pub(crate) fn validate(&self) -> Result<(), &'static str> { - if self.action.observed_background_task.0.is_nil() || self.action.state_revision.get() == 0 - { - return Err("release checkpoint action identity is invalid"); - } - - let baseline = match &self.phase { - ReleaseCheckpointPhase::WatcherBindingPending => return Ok(()), - ReleaseCheckpointPhase::BaselineCaptured { baseline } - | ReleaseCheckpointPhase::StopReserved { baseline, .. } - | ReleaseCheckpointPhase::CancellationCommitted { baseline, .. } => baseline, - }; - validate_release_checkpoint_binding(&self.action, &baseline.binding)?; - if !baseline.snapshot.is_for(&baseline.binding.attempt_binding) { - return Err("checkpoint baseline belongs to a different trainer attempt"); - } - - let decision = match &self.phase { - ReleaseCheckpointPhase::StopReserved { decision, .. } - | ReleaseCheckpointPhase::CancellationCommitted { decision, .. } => decision, - ReleaseCheckpointPhase::WatcherBindingPending - | ReleaseCheckpointPhase::BaselineCaptured { .. } => return Ok(()), - }; - if let ReleaseCheckpointPhase::CancellationCommitted { cancellation, .. } = &self.phase - && cancellation.task_id != self.action.observed_background_task - { - return Err("checkpoint cancellation differs from its observed trainer task"); - } - validate_release_checkpoint_binding(&self.action, &decision.binding)?; - if decision.binding != baseline.binding - || decision.selected_checkpoint.binding != baseline.binding.attempt_binding - || baseline - .snapshot - .contains_generation(&decision.selected_checkpoint.generation_id) - || decision.selected_checkpoint.path - != baseline - .binding - .association - .canonical_runtime_root - .join("published") - .join(&decision.selected_checkpoint.generation_id) - { - return Err("checkpoint stop decision differs from its verified baseline"); - } - - Ok(()) - } -} - -fn validate_release_checkpoint_binding( - action: &ReleaseCheckpointAction, - binding: &ReleaseCheckpointBinding, -) -> Result<(), &'static str> { - let intent = &binding.watcher_intent; - let association = &binding.association; - if binding.action != *action - || association.resource_id != action.resource_id - || association.task_id != action.observed_background_task - || association.authority_machine.as_uuid().is_nil() - || association.attempt_binding != binding.attempt_binding - || !association.canonical_runtime_root.is_absolute() - || intent.action_id != action.action_id - || intent.state_revision != action.state_revision - || intent.observed_background_task != action.observed_background_task - || intent.validate().is_err() - { - return Err("release checkpoint binding does not match the saved action"); - } - - binding - .attempt_binding - .validate() - .map_err(|_| "release checkpoint attempt binding is invalid") -} - -/// Typed result of checking whether an exact-task stop can be reserved -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) enum ReleaseCheckpointStopOutcome { - /// A new stop decision and checkpoint identity were persisted - Reserved(ReleaseCheckpointStopDecision), - /// An exact retry returned the previously persisted decision - AlreadyReserved(ReleaseCheckpointStopDecision), - /// The exact task has not started yet - WaitingForTaskStart, - /// No complete new checkpoint is available yet - WaitingForCheckpoint, - /// A complete final result takes precedence over a checkpoint stop - CompletedResultAwaitingTaskExit, - /// The task already published its successful final result - AlreadyCompleted, - /// The saved task or publication state needs attention - Attention(WatcherAttention), -} diff --git a/src/resource/release_watcher.rs b/src/resource/release_watcher.rs deleted file mode 100644 index cd4246d..0000000 --- a/src/resource/release_watcher.rs +++ /dev/null @@ -1,338 +0,0 @@ -//! Canonical release-watcher command and its socket-only poll protocol -//! -//! The resource authority builds the only accepted watcher command from typed release -//! identities. The watcher always runs on the authority, even when its supervisor -//! thread and callback route are on another machine. The hidden watcher subcommand -//! sends the same identities back through the local daemon socket; neither side -//! accepts caller-authored command text - -use std::path::{Path, PathBuf}; -use std::time::Duration; - -use chrono::{DateTime, Utc}; -use serde::{Deserialize, Serialize}; - -use super::{ActionId, ReleaseWatcherIntent, ReleaseWatcherTaskId, ResourceId, ResourceRevision}; -use crate::domain::{API_VERSION, TaskId, TaskName, ThreadId}; -use crate::error::AppError; -use crate::invocation::CommandLine; -use crate::spec::{NormalizedSpec, NormalizedTaskWorkload, NormalizedWorkload}; -use crate::submission::{NormalizedSpecSha256, normalized_spec_sha256}; - -/// Hidden Homebased subcommand that runs one bound release watcher -pub const RELEASE_WATCHER_SUBCOMMAND: &str = "resource-release-watcher"; - -/// Socket-only route that serves one watcher poll -pub const RELEASE_WATCHER_POLL_PATH: &str = "/v1/internal/resource-release-watcher/poll"; - -/// Version of the watcher poll request and response documents -pub const RELEASE_WATCHER_PROTOCOL_VERSION: u32 = 1; - -const RELEASE_WATCHER_TASK_NAME: &str = "resource release watcher"; -const RELEASE_WATCHER_CWD: &str = "/"; - -// checkpoints were measured about 47-49 minutes apart, so the ordinary -// inactivity reminder must not fire while the watcher waits silently for one -const RELEASE_WATCHER_TIMEOUT: Duration = Duration::from_secs(2 * 60 * 60); - -/// Exact release identities carried by one watcher command and every poll -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ReleaseWatcherCommand { - /// Resource whose release action owns the watcher - pub resource_id: ResourceId, - /// Stable release action identity - pub action_id: ActionId, - /// Resource revision saved with the release action - pub state_revision: ResourceRevision, - /// Exact registered trainer task observed by the release action - pub trainer_task_id: TaskId, - /// Preallocated Homebased task identity of this watcher - pub watcher_task_id: ReleaseWatcherTaskId, -} - -impl ReleaseWatcherCommand { - /// Build the command identities from one saved watcher intent - #[must_use] - pub fn from_intent(resource_id: ResourceId, intent: &ReleaseWatcherIntent) -> Self { - Self { - resource_id, - action_id: intent.action_id, - state_revision: intent.state_revision, - trainer_task_id: intent.observed_background_task, - watcher_task_id: intent.watcher_task_id, - } - } - - /// Return the canonical argv for `executable` - pub fn argv(&self, executable: &Path) -> Result, AppError> { - if !executable.is_absolute() { - return Err(AppError::Internal { - message: format!( - "release watcher executable {} is not absolute", - executable.display() - ), - }); - } - let Some(program) = executable.to_str() else { - return Err(AppError::Internal { - message: format!( - "release watcher executable {} is not UTF-8", - executable.display() - ), - }); - }; - - Ok(vec![ - program.to_owned(), - RELEASE_WATCHER_SUBCOMMAND.to_owned(), - "--resource-id".to_owned(), - self.resource_id.as_uuid().to_string(), - "--action-id".to_owned(), - self.action_id.as_uuid().to_string(), - "--state-revision".to_owned(), - self.state_revision.get().to_string(), - "--trainer-task-id".to_owned(), - self.trainer_task_id.to_string(), - "--watcher-task-id".to_owned(), - self.watcher_task_id.as_task_id().to_string(), - ]) - } - - /// Build the one normalized spec the authority accepts for this watcher - /// - /// The supervisor thread receives the watcher's ordinary task callbacks - pub fn normalized_spec( - &self, - executable: &Path, - supervisor_thread: ThreadId, - ) -> Result { - let command = CommandLine::try_from_argv(self.argv(executable)?).map_err(|error| { - AppError::Internal { - message: format!("release watcher command: {error}"), - } - })?; - let name = - TaskName::parse(RELEASE_WATCHER_TASK_NAME).map_err(|error| AppError::Internal { - message: format!("release watcher name: {error}"), - })?; - - Ok(NormalizedSpec { - api_version: API_VERSION, - thread: supervisor_thread, - name, - cwd: PathBuf::from(RELEASE_WATCHER_CWD), - machine: None, - timeout: RELEASE_WATCHER_TIMEOUT, - workload: NormalizedWorkload::Task(NormalizedTaskWorkload { command }), - }) - } - - /// Return the digest of the canonical normalized spec - pub fn normalized_spec_sha256( - &self, - executable: &Path, - supervisor_thread: ThreadId, - ) -> Result { - Ok(normalized_spec_sha256( - &self.normalized_spec(executable, supervisor_thread)?, - )?) - } -} - -/// Resolve the executable that the authority places in every watcher command -/// -/// The watcher is a hidden subcommand of the same binary that runs `task-run` -pub(crate) fn release_watcher_executable() -> Result { - crate::runner::task_run_executable() -} - -/// One strict watcher poll sent over the local daemon socket -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ReleaseWatcherPollRequest { - /// Poll protocol version, always [`RELEASE_WATCHER_PROTOCOL_VERSION`] - pub protocol_version: u32, - /// Exact identities from the watcher's canonical command - pub watcher: ReleaseWatcherCommand, -} - -impl ReleaseWatcherPollRequest { - /// Build a current-version poll for one watcher command - #[must_use] - pub const fn new(watcher: ReleaseWatcherCommand) -> Self { - Self { - protocol_version: RELEASE_WATCHER_PROTOCOL_VERSION, - watcher, - } - } -} - -/// Versioned authority reply to one watcher poll -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ReleaseWatcherPollResponse { - /// Poll protocol version, always [`RELEASE_WATCHER_PROTOCOL_VERSION`] - pub protocol_version: u32, - /// Typed authority decision for this poll - pub outcome: ReleaseWatcherPollOutcome, -} - -/// Authority decision for one release-watcher poll -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ReleaseWatcherPollOutcome { - /// The accepted watcher task has not reached its saved running identity yet - WatcherNotRunning, - /// The exact trainer task has not started - WaitingForTrainerStart, - /// No complete checkpoint outside the saved baseline is available yet - WaitingForCheckpoint, - /// A matching final result exists, so the watcher waits for trainer exit without cancelling - CompletedResultAwaitingTrainerExit, - /// The trainer finished with its final result; the resource owner proves release - TrainerCompleted, - /// The saved stop decision and exact trainer cancellation marker are committed - StopCommitted { - /// Checkpoint generation fixed by the saved stop decision - generation_id: String, - /// Durable cancellation marker time on the trainer task - cancel_requested_at: DateTime, - }, - /// The release action already completed with an authority-built proof - ReleaseSettled, - /// The loan stays reserved and the release action needs attention - Attention { - /// Typed reason the watcher cannot advance the release action - reason: ReleaseWatcherPollAttention, - }, -} - -impl ReleaseWatcherPollOutcome { - /// Whether the watcher has finished its part of the release action - #[must_use] - pub const fn is_final(&self) -> bool { - matches!( - self, - Self::TrainerCompleted - | Self::StopCommitted { .. } - | Self::ReleaseSettled - | Self::Attention { .. } - ) - } -} - -/// Why a watcher poll cannot advance its release action -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum ReleaseWatcherPollAttention { - /// This daemon is not the resource authority - WrongAuthority, - /// The poll names an action, revision, or trainer that is not the saved release action - ActionNotCurrent, - /// The poll names a watcher task other than the saved watcher identity - WrongWatcher, - /// The release action has no saved watcher identity - WatcherIntentMissing, - /// The accepted watcher task no longer matches its saved launch identity - WatcherIdentityConflict, - /// The registered trainer task, command, association, or cancel marker changed - TrainerChanged, - /// Trainer publications changed or cannot be verified - PublicationChanged, - /// Homebased lost the trainer task, so process state is unknown - TrainerLost, - /// The trainer ended without a successful result or saved stop decision - TrainerFailed, - /// Trainer publications appeared before its task started - PublicationBeforeTrainerStart, - /// The trainer exited successfully without a matching final result - TrainerCompletedWithoutResult, - /// A saved authority record cannot be decoded, so a retry cannot succeed - CorruptRecord, -} - -#[cfg(test)] -mod tests { - use super::{ReleaseWatcherCommand, ReleaseWatcherPollRequest}; - use crate::domain::{TaskId, ThreadId}; - use crate::resource::{ActionId, ReleaseWatcherTaskId, ResourceId, ResourceRevision}; - use std::path::Path; - - fn command() -> ReleaseWatcherCommand { - ReleaseWatcherCommand { - resource_id: ResourceId::new(), - action_id: ActionId::new(), - state_revision: ResourceRevision::new(3), - trainer_task_id: TaskId::new(), - watcher_task_id: ReleaseWatcherTaskId::new(TaskId::new()), - } - } - - #[test] - fn every_release_identity_changes_the_canonical_digest() { - let executable = Path::new("/usr/local/bin/homebased"); - let thread = ThreadId(uuid::Uuid::now_v7()); - let base = command(); - let digest = base.normalized_spec_sha256(executable, thread).unwrap(); - let variants = [ - ReleaseWatcherCommand { - resource_id: ResourceId::new(), - ..base - }, - ReleaseWatcherCommand { - action_id: ActionId::new(), - ..base - }, - ReleaseWatcherCommand { - state_revision: ResourceRevision::new(4), - ..base - }, - ReleaseWatcherCommand { - trainer_task_id: TaskId::new(), - ..base - }, - ReleaseWatcherCommand { - watcher_task_id: ReleaseWatcherTaskId::new(TaskId::new()), - ..base - }, - ]; - for variant in variants { - assert_ne!( - variant.normalized_spec_sha256(executable, thread).unwrap(), - digest - ); - } - assert_ne!( - base.normalized_spec_sha256(Path::new("/tmp/homebased"), thread) - .unwrap(), - digest - ); - assert_ne!( - base.normalized_spec_sha256(executable, ThreadId(uuid::Uuid::now_v7())) - .unwrap(), - digest - ); - assert_eq!( - base.normalized_spec_sha256(executable, thread).unwrap(), - digest - ); - } - - #[test] - fn relative_executable_is_rejected() { - assert!(command().argv(Path::new("homebased")).is_err()); - } - - #[test] - fn poll_documents_reject_unknown_fields() { - let request = ReleaseWatcherPollRequest::new(command()); - let mut value = serde_json::to_value(request).unwrap(); - assert_eq!( - serde_json::from_value::(value.clone()).unwrap(), - request - ); - value["watcher"]["command"] = serde_json::json!(["/bin/sh"]); - assert!(serde_json::from_value::(value).is_err()); - } -} diff --git a/src/resource/return_window.rs b/src/resource/return_window.rs deleted file mode 100644 index 9dd4e8c..0000000 --- a/src/resource/return_window.rs +++ /dev/null @@ -1,269 +0,0 @@ -//! Time the supervisor has to decide a return before queued work takes the resource -//! -//! A drained queue reserves the resource for the supervisor's return decision. -//! The reservation must not hold queued GPU work for a supervisor that does not -//! answer, so every return action opens a decision window. When the window -//! closes with a request queued, the authority serves that request from the -//! same loan and keeps the return obligation for the next drained queue. The -//! supervisor can hold the window open, but never past its hard limit - -use std::time::Duration; - -use chrono::{DateTime, Utc}; -use serde::{Deserialize, Serialize}; - -use super::{ActionId, LoanId, ResourceId}; - -/// Decision time a return action gets before queued work can take the resource -pub const RETURN_DECISION_GRACE: Duration = Duration::from_secs(2 * 60); - -/// Most decision time any hold can give a return action, counted from its opening -pub const RETURN_DECISION_LIMIT: Duration = Duration::from_secs(10 * 60); - -/// Saved decision window of one return action -/// -/// The fields are private so every window keeps its deadline between its -/// opening and its hard limit, including a window decoded from saved JSON -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(try_from = "SavedReturnDecisionWindow")] -pub struct ReturnDecisionWindow { - action_id: ActionId, - loan_id: LoanId, - resource_id: ResourceId, - opened_at: DateTime, - deadline_at: DateTime, -} - -/// Unchecked JSON shape of a saved window -#[derive(Deserialize)] -#[serde(deny_unknown_fields)] -struct SavedReturnDecisionWindow { - action_id: ActionId, - loan_id: LoanId, - resource_id: ResourceId, - opened_at: DateTime, - deadline_at: DateTime, -} - -/// A saved window whose deadline is outside its opening and hard limit -#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] -#[error("return decision window deadline is outside its opening and {limit_minutes} minute limit")] -pub struct InvalidReturnDecisionWindow { - limit_minutes: u64, -} - -impl TryFrom for ReturnDecisionWindow { - type Error = InvalidReturnDecisionWindow; - - fn try_from(saved: SavedReturnDecisionWindow) -> Result { - let window = Self { - action_id: saved.action_id, - loan_id: saved.loan_id, - resource_id: saved.resource_id, - opened_at: saved.opened_at, - deadline_at: saved.deadline_at, - }; - if window.deadline_at < window.opened_at || window.deadline_at > window.limit_at() { - return Err(InvalidReturnDecisionWindow { - limit_minutes: limit_minutes(), - }); - } - Ok(window) - } -} - -/// Why a hold cannot move the deadline of a return action -#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] -pub enum ReturnHoldRejection { - /// A hold must ask for a positive duration - #[error("a return hold must be longer than zero")] - Empty, - /// The window already reached its hard limit - #[error("the return decision window already reached its {limit_minutes} minute limit")] - LimitReached { - /// Hard limit of every decision window, in minutes - limit_minutes: u64, - }, -} - -impl ReturnDecisionWindow { - /// Open the window of a return action at `opened_at` with the default grace - #[must_use] - pub fn open( - action_id: ActionId, - loan_id: LoanId, - resource_id: ResourceId, - opened_at: DateTime, - ) -> Self { - Self { - action_id, - loan_id, - resource_id, - opened_at, - deadline_at: opened_at + RETURN_DECISION_GRACE, - } - } - - /// Return action that this window belongs to - #[must_use] - pub const fn action_id(&self) -> ActionId { - self.action_id - } - - /// Loan that awaits the return decision - #[must_use] - pub const fn loan_id(&self) -> LoanId { - self.loan_id - } - - /// Resource reserved for the decision - #[must_use] - pub const fn resource_id(&self) -> ResourceId { - self.resource_id - } - - /// When the queue drained and the return action opened - #[must_use] - pub const fn opened_at(&self) -> DateTime { - self.opened_at - } - - /// When queued work may take the resource if no decision exists - #[must_use] - pub const fn deadline_at(&self) -> DateTime { - self.deadline_at - } - - /// Latest deadline that any hold can set - #[must_use] - pub fn limit_at(&self) -> DateTime { - self.opened_at + RETURN_DECISION_LIMIT - } - - /// Whether queued work may take the resource at `now` - #[must_use] - pub fn expired_at(&self, now: DateTime) -> bool { - now >= self.deadline_at - } - - /// Time left before the deadline, or zero when it passed - #[must_use] - pub fn remaining_at(&self, now: DateTime) -> Duration { - (self.deadline_at - now).to_std().unwrap_or(Duration::ZERO) - } - - /// Move the deadline to `now + hold`, capped at the hard limit - /// - /// A hold never shortens the window, so a short hold after a longer one - /// keeps the longer deadline - pub fn hold(self, now: DateTime, hold: Duration) -> Result { - if hold.is_zero() { - return Err(ReturnHoldRejection::Empty); - } - let limit_at = self.limit_at(); - if self.deadline_at >= limit_at || now >= limit_at { - return Err(ReturnHoldRejection::LimitReached { - limit_minutes: limit_minutes(), - }); - } - // a hold longer than the chrono range is capped by the limit anyway - let requested = chrono::Duration::from_std(hold) - .ok() - .and_then(|hold| now.checked_add_signed(hold)) - .unwrap_or(limit_at); - Ok(Self { - deadline_at: requested.min(limit_at).max(self.deadline_at), - ..self - }) - } -} - -fn limit_minutes() -> u64 { - RETURN_DECISION_LIMIT.as_secs() / 60 -} - -#[cfg(test)] -mod tests { - use super::*; - - fn window(opened_at: DateTime) -> ReturnDecisionWindow { - ReturnDecisionWindow::open(ActionId::new(), LoanId::new(), ResourceId::new(), opened_at) - } - - #[test] - fn a_new_window_expires_after_the_default_grace() { - let opened_at = Utc::now(); - let window = window(opened_at); - - assert!(!window.expired_at(opened_at + chrono::Duration::seconds(119))); - assert!(window.expired_at(opened_at + chrono::Duration::seconds(120))); - } - - #[test] - fn a_hold_extends_from_now_but_never_past_the_limit() { - let opened_at = Utc::now(); - let now = opened_at + chrono::Duration::minutes(1); - - let held = window(opened_at) - .hold(now, Duration::from_secs(5 * 60)) - .unwrap(); - assert_eq!(held.deadline_at, now + chrono::Duration::minutes(5)); - - let capped = held.hold(now, Duration::from_secs(60 * 60)).unwrap(); - assert_eq!( - capped.deadline_at, - opened_at + chrono::Duration::minutes(10) - ); - assert_eq!( - capped.hold(now, Duration::from_secs(60)), - Err(ReturnHoldRejection::LimitReached { limit_minutes: 10 }) - ); - } - - #[test] - fn a_short_hold_keeps_a_longer_deadline() { - let opened_at = Utc::now(); - let held = window(opened_at) - .hold(opened_at, Duration::from_secs(8 * 60)) - .unwrap(); - - let shorter = held.hold(opened_at, Duration::from_secs(60)).unwrap(); - - assert_eq!(shorter.deadline_at, held.deadline_at); - } - - #[test] - fn a_saved_window_with_a_deadline_past_its_limit_is_refused() { - let saved = serde_json::to_value(window(Utc::now())).unwrap(); - let opened_at = saved["opened_at"] - .as_str() - .unwrap() - .parse::>() - .unwrap(); - let mut past_limit = saved.clone(); - past_limit["deadline_at"] = serde_json::json!(opened_at + chrono::Duration::minutes(11)); - let mut before_opening = saved.clone(); - before_opening["deadline_at"] = serde_json::json!(opened_at - chrono::Duration::seconds(1)); - - assert!(serde_json::from_value::(saved).is_ok()); - assert!(serde_json::from_value::(past_limit).is_err()); - assert!(serde_json::from_value::(before_opening).is_err()); - } - - #[test] - fn a_hold_after_the_limit_is_refused() { - let opened_at = Utc::now(); - - assert_eq!( - window(opened_at).hold( - opened_at + chrono::Duration::minutes(11), - Duration::from_secs(60) - ), - Err(ReturnHoldRejection::LimitReached { limit_minutes: 10 }) - ); - assert_eq!( - window(opened_at).hold(opened_at, Duration::ZERO), - Err(ReturnHoldRejection::Empty) - ); - } -} diff --git a/src/resource/store.rs b/src/resource/store.rs deleted file mode 100644 index 0a68041..0000000 --- a/src/resource/store.rs +++ /dev/null @@ -1,77 +0,0 @@ -//! SQLite operations for authority-local resource state - -mod acceptance; -mod assigned_task; -mod cancellation; -mod checkpoint; -mod codec; -mod error; -mod notice; -mod provenance; -mod queue; -mod release_completion; -mod release_loan; -mod release_watcher; -mod resources; -mod return_window; -mod revision; -mod rows; -mod schema; -#[cfg(test)] -pub(crate) mod test_support; -#[cfg(test)] -mod tests; - -pub(crate) use acceptance::{ - AcceptedResourceTask, PreLaunchFailure, ResourceTaskAcceptance, ResourceTaskAcceptanceInput, - assigned_resource_request_for_acceptance, -}; -pub(crate) use assigned_task::{ - AssignedResourceTaskAttention, AssignedResourceTaskProgress, - AssignedResourceTaskReconcileInput, AssignedResourceTaskReconcileOutcome, - ResourceTaskCompletionResult, reconcile_assigned_resource_task_for_authority, -}; -pub(crate) use cancellation::{QueueCancellationResult, cancel_request_before_activation_on}; -pub(crate) use checkpoint::{ - ReleaseCheckpointCancellationOutcome, ReleaseCheckpointCancellationResult, - ReleaseCheckpointError, release_checkpoint_state_for_action, update_release_checkpoint_state, -}; -pub(crate) use error::{ConflictReason, ResourceStoreError, TrainerAttemptAssociationStoreError}; -pub(crate) use notice::{ - SupervisorNoticeStoreError, decode_supervisor_notice_record, - insert_supervisor_notice_in_transaction, pending_supervisor_notices, - recover_sending_supervisor_notices, reserve_supervisor_notice_attempt, - retarget_supervisor_notice_in_transaction, select_supervisor_notice_record, - select_supervisor_notice_record_by_action, settle_supervisor_notice_attempt, supervisor_notice, - update_supervisor_notice_cas, -}; -pub(crate) use queue::{ - accept_request_for_authority, assigned_resource_requests_for_authority, - next_queued_request_for_authority, requests_for_resource_for_authority, - rewrite_queued_request_ranks, -}; -pub(crate) use release_completion::{ - CompleteReleaseError, ReleaseCompletionResult, complete_release_for_authority, - release_completion_for_loan, release_completion_for_retry, -}; -pub(crate) use release_loan::{ - OpenReleaseLoanError, ResourceQueueReconcileError, reconcile_resource_queue_for_authority, -}; -pub(crate) use return_window::{ - ReturnDeadlineOutcome, open_missing_return_windows_on, return_window_on, - save_held_return_window_on, serve_after_return_deadline_for_authority, -}; -// production opens release loans only inside queue reconciliation -#[cfg(test)] -pub(crate) use release_loan::OpenReleaseLoanResult; -pub(crate) use release_watcher::{ - ReleaseWatcherAcceptance, ReleaseWatcherAcceptanceError, ReleaseWatcherAcceptanceInput, - bind_release_watcher_for_authority, bind_release_watcher_intent_on, - validate_release_watcher_for_local_acceptance, validate_release_watcher_intent_on, -}; -pub(crate) use resources::{ - ResourceSnapshot, register_resource_for_authority, resources_for_authority, -}; -pub(crate) use revision::swap_resource_revision; -pub(crate) use rows::{select_non_closed_loan, select_request_by_id, select_resource}; -pub(crate) use schema::RESOURCE_SCHEMA; diff --git a/src/resource/store/acceptance.rs b/src/resource/store/acceptance.rs deleted file mode 100644 index 7078c90..0000000 --- a/src/resource/store/acceptance.rs +++ /dev/null @@ -1,225 +0,0 @@ -//! Task-layer acceptance of the assigned resource request - -use rusqlite::Connection; - -use super::error::{ConflictReason, ResourceStoreError}; -use super::provenance::serving_release_provenance_matches; -use super::queue::{ - RequestIdentity, earlier_active_request_exists, prevention_exists, request_matches_identity, - select_executor_identity, -}; -use super::rows::{ - check_resource_authority, select_non_closed_loan, select_request_by_id, select_resource, -}; -use crate::domain::{ProcessStatus, TaskEnv, TaskId}; -use crate::machine::MachineId; -use crate::resource::{ - AcceptanceSequence, CommandSpec, LoanId, LoanPhase, LoanState, ResourceId, ResourceRequest, - ResourceRequestState, ResourceRevision, -}; -use crate::submission::{ExecutorIdentity, PreAcceptanceRejection, RequestId}; - -/// Exact authority selection and executor context for accepting an assigned request -#[derive(Debug, Clone)] -pub(crate) struct ResourceTaskAcceptanceInput { - /// Fixed authority machine recorded on the resource - pub(crate) authority_machine: MachineId, - /// Resource and serving loan that selected this request - pub(crate) resource_id: ResourceId, - /// Caller retry identity saved with the request - pub(crate) request_id: RequestId, - /// Preallocated task identity saved with the request - pub(crate) task_id: TaskId, - /// Immutable authority-assigned acceptance identity of the selected request - pub(crate) acceptance_sequence: AcceptanceSequence, - /// Serving loan that owns the selected request - pub(crate) loan_id: LoanId, - /// Resource revision observed when the request was selected - pub(crate) expected_state_revision: ResourceRevision, - /// Immutable command spec observed with the selected request - pub(crate) command_spec: CommandSpec, - /// Executor-machine environment used to create the task row - pub(crate) executor_env: TaskEnv, -} - -/// Result of the atomic task-layer acceptance for one assigned resource request -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) enum ResourceTaskAcceptance { - /// The task row, accepted identity, and first queued event committed together - Inserted { - /// Preallocated task identity accepted by this authority - task: TaskId, - }, - /// This call committed the task records, but the task must fail before it - /// launches; the caller records the failure instead of spawning a worker - Unlaunchable { - /// Preallocated task identity accepted by this authority - task: TaskId, - /// Why the task cannot launch - failure: PreLaunchFailure, - }, - /// An exact acceptance already exists and retains its current task state - Existing { - /// Preallocated task identity accepted by this authority - task: TaskId, - /// State retained by the task layer - state: ProcessStatus, - }, -} - -/// Why an assigned task ends before it launches -/// -/// The task still gets its records and a failed terminal event, so its thread -/// hears the reason and the queue serves the next request -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) enum PreLaunchFailure { - /// The executor cannot prepare the saved command here, such as a missing - /// `cwd`, mount source, or executable - Preparation { - /// Preparation error shown to the thread - message: String, - }, - /// The launch outcome stayed unknown past [`crate::resource::LAUNCH_CONFIRMATION_BOUND`] - /// with no started worker - LaunchUnconfirmed, -} - -impl PreLaunchFailure { - /// Spawn-failure message saved with the failed task - #[must_use] - pub(crate) fn message(&self) -> String { - match self { - Self::Preparation { message } => format!("launch failed before start: {message}"), - Self::LaunchUnconfirmed => format!( - "launch_unconfirmed: the launch had no confirmed start after {}, so it was \ - failed before launch", - humantime::format_duration(crate::resource::LAUNCH_CONFIRMATION_BOUND) - ), - } - } -} - -/// Accepted task identity retained by an authority-owned assigned request -#[derive(Debug, Clone)] -pub(crate) struct AcceptedResourceTask { - /// Exact assigned resource request that owns this task identity and names its loan - pub(crate) request: ResourceRequest, - /// Durable executor state, cross-checked against the task row and first event - pub(crate) state: ProcessStatus, -} - -/// Validate the exact assigned request that is eligible for task-layer acceptance -pub(crate) fn assigned_resource_request_for_acceptance( - conn: &Connection, - input: &ResourceTaskAcceptanceInput, -) -> Result { - let authority = input.authority_machine; - check_resource_authority(conn, input.resource_id, authority)?; - let saved = select_request_by_id(conn, input.request_id)? - .ok_or(ResourceStoreError::Conflict(ConflictReason::RequestMissing))?; - let identity = RequestIdentity { - request_id: input.request_id, - task_id: input.task_id, - resource_id: input.resource_id, - origin_machine: saved.origin_machine, - }; - if !request_matches_identity(&saved, identity) - || saved.acceptance_sequence != input.acceptance_sequence - || saved.spec().as_normalized() != input.command_spec.as_normalized() - { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestIdentityMismatch, - )); - } - - if prevention_exists(conn, identity)? { - return Err(ResourceStoreError::Prevented); - } - if let Some(executor) = select_executor_identity(conn, input.task_id)? { - match executor { - ExecutorIdentity::Rejected(rejection) => { - if rejection.origin_machine == saved.origin_machine - && rejection.execution_machine == authority - && rejection.reason == PreAcceptanceRejection::Cancelled.as_str() - { - return Err(ResourceStoreError::Prevented); - } - return Err(ResourceStoreError::Conflict( - ConflictReason::ExecutorIdentityMismatch, - )); - } - ExecutorIdentity::Accepted(record) => { - let same_saved_spec = record.current_spec() == Some(saved.spec().as_normalized()); - if record.task != input.task_id - || record.origin_machine != saved.origin_machine - || record.execution_machine != authority - || !record.has_valid_spec_owners() - || !same_saved_spec - { - return Err(ResourceStoreError::Conflict( - ConflictReason::ExecutorIdentityMismatch, - )); - } - } - } - } - if matches!(&saved.state, ResourceRequestState::CancelledBeforeLaunch) { - return Err(ResourceStoreError::Prevented); - } - - let resource = - select_resource(conn, input.resource_id)?.ok_or(ResourceStoreError::ResourceNotFound)?; - if resource.state_revision != input.expected_state_revision { - return Err(ResourceStoreError::Conflict( - ConflictReason::ResourceRevisionChanged, - )); - } - - let ResourceRequestState::Assigned { loan_id } = &saved.state else { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestStateChanged, - )); - }; - if *loan_id != input.loan_id { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - } - let loan = select_non_closed_loan(conn, input.resource_id)? - .ok_or(ResourceStoreError::Conflict(ConflictReason::LoanChanged))?; - if loan.id != input.loan_id || loan.resource_id != input.resource_id { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - } - let LoanState::Active { - phase: - LoanPhase::Serving { - current_request_id, - return_context, - release_provenance, - }, - } = &loan.state - else { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - }; - if *current_request_id != input.request_id { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - } - if !serving_release_provenance_matches( - conn, - authority, - &resource, - &loan, - return_context, - release_provenance, - )? { - return Err(ResourceStoreError::Conflict( - ConflictReason::ServingReleaseUnverified, - )); - } - - if earlier_active_request_exists(conn, input.resource_id, input.request_id)? { - return Err(ResourceStoreError::Conflict( - ConflictReason::EarlierRequestActive, - )); - } - - Ok(saved) -} diff --git a/src/resource/store/assigned_task.rs b/src/resource/store/assigned_task.rs deleted file mode 100644 index 04b66d0..0000000 --- a/src/resource/store/assigned_task.rs +++ /dev/null @@ -1,674 +0,0 @@ -//! Proof-gated completion of an accepted resource task - -use rusqlite::{Connection, OptionalExtension, Transaction, TransactionBehavior, params}; -use serde::Serialize; - -use super::codec::{encode_json, sqlite_integer, stored_json}; -use super::error::{ConflictReason, ResourceStoreError}; -use super::notice::{SupervisorNoticeStoreError, insert_supervisor_notice_in_transaction}; -use super::provenance::serving_release_provenance_matches; -use super::queue::{earlier_active_request_exists, next_queued_request_for_authority}; -use super::revision::swap_resource_revision; -use super::rows::{check_authority, select_non_closed_loan, select_request_by_id, select_resource}; -use crate::domain::{ExitReason, TaskId, TaskState, WorkExitEvidence}; -use crate::machine::MachineId; -use crate::resource::foreground::CommandOwnershipContract; -use crate::resource::{ - ActionId, CommandSpec, Loan, LoanId, LoanPhase, LoanState, NoticeId, ResourceId, - ResourceRequest, ResourceRequestState, ResourceRevision, ResourceTaskOwnershipRisk, - SupervisorNotice, SupervisorNoticeDelivery, SupervisorNoticePayload, -}; -use crate::submission::{ExecutorIdentity, RequestId}; - -/// Exact Serving assignment to reconcile against task-layer ownership evidence -#[derive(Debug, Clone, Copy)] -pub(crate) struct AssignedResourceTaskReconcileInput { - /// Fixed authority machine recorded on the resource - pub(crate) authority_machine: MachineId, - /// Resource whose active loan owns the command - pub(crate) resource_id: ResourceId, - /// Serving loan that reserved the command - pub(crate) loan_id: LoanId, - /// Exact accepted resource request - pub(crate) request_id: RequestId, - /// Preallocated task identity bound to the accepted request - pub(crate) task_id: TaskId, - /// Resource revision observed before this reconciliation - pub(crate) expected_state_revision: ResourceRevision, -} - -/// Task-layer observation and proof-gated resource completion result -#[derive(Debug, Clone)] -pub(crate) enum AssignedResourceTaskReconcileOutcome { - /// No task-layer acceptance exists, so the one-shot launch path may proceed - NotAccepted, - /// The exact accepted task is not terminal yet - Active(AssignedResourceTaskProgress), - /// The exact terminal task was durably finished and the loan advanced once - // boxed because the result carries two requests and a loan, while the other - // variants are a few bytes - Completed(Box), - /// The task remains assigned because ownership or identity needs attention - Attention(AssignedResourceTaskAttention), -} - -/// Non-terminal task-layer state of one accepted resource task -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(crate) enum AssignedResourceTaskProgress { - /// The task row exists, but its worker has not reached the running boundary - Queued, - /// The task-run worker owns a running child process group - Running, -} - -/// Typed failure that keeps an assigned resource request and its loan reserved -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(crate) enum AssignedResourceTaskAttention { - /// The request row changed or no longer belongs to this task - RequestChanged, - /// The serving loan changed or no longer owns this request - LoanChanged, - /// The resource revision changed before the transaction could commit - StaleRevision, - /// The loan lacks durable release provenance for its Serving phase - ServingReleaseUnverified, - /// The exact task row is missing or disagrees with accepted identity data - TaskIdentityMismatch, - /// The task reached Lost without proving ownership exit - TaskLost, - /// The task is terminal but its exit witness is not confirmed - ExitWitnessUnconfirmed, - /// No-child evidence came with an outcome that no pre-spawn path records - InvalidNoChildSpawnEvidence, - /// The command can outlive the task-run process group - OwnershipUncertain(ResourceTaskOwnershipRisk), -} - -/// Durable result of completing one exact accepted resource task -#[derive(Debug, Clone, Serialize, serde::Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub(crate) enum ResourceTaskCompletionResult { - /// The next queued request now owns this loan - Assigned { - /// Request whose exact task result was recorded - finished_request: ResourceRequest, - /// Loan that remains reserved for the same return context - loan: Loan, - /// Next queued request selected for the next serving turn - next_request: ResourceRequest, - /// Resource revision committed with the assignment - state_revision: ResourceRevision, - }, - /// No queued request remained, so the loan now awaits its return decision - ReturnRequired { - /// Request whose exact task result was recorded - finished_request: ResourceRequest, - /// Loan that retains the same return context - loan: Loan, - /// Durable notice for the exact current supervisor assignment - notice: SupervisorNotice, - }, -} - -/// Proof kind that authorized one resource task to release its loan turn -#[derive(Debug, Clone, PartialEq, Eq, Serialize, serde::Deserialize)] -#[serde(rename_all = "snake_case")] -pub(super) enum ResourceTaskReleaseProof { - /// The task-run worker confirmed that its owned process group exited - ConfirmedProcessGroupExit, - /// The worker read the container's exit code, removed it, and saw its ID absent - ConfirmedContainerRemoved { - /// Removed container - container_id: crate::domain::ContainerId, - }, - /// The task failed before it spawned a child process - NoChildSpawnedAfterSpawnFailure, - /// Cancellation won the Queued CAS, so the worker can never reach its spawn - NoChildSpawnedAfterQueuedCancel, -} - -/// Durable idempotency receipt for one exact task, request, and loan transition -#[derive(Debug, Clone, Serialize, serde::Deserialize)] -#[serde(deny_unknown_fields)] -struct ResourceTaskCompletionReceipt { - authority_machine: MachineId, - resource_id: ResourceId, - loan_id: LoanId, - request_id: RequestId, - task_id: TaskId, - expected_state_revision: ResourceRevision, - outcome: ExitReason, - release_proof: ResourceTaskReleaseProof, - result: ResourceTaskCompletionResult, -} - -/// Reconcile one exact assigned task and atomically finish it before selecting more work -pub(crate) fn reconcile_assigned_resource_task_for_authority( - conn: &mut Connection, - input: AssignedResourceTaskReconcileInput, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - if let Some(receipt) = select_resource_task_completion_receipt(&tx, input.task_id)? { - return retry_resource_task_completion(&receipt, input); - } - - let Some(resource) = select_resource(&tx, input.resource_id)? else { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::LoanChanged, - )); - }; - check_authority(resource.authority_machine(), input.authority_machine)?; - if resource.state_revision != input.expected_state_revision { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::StaleRevision, - )); - } - - let Some(mut request) = select_request_by_id(&tx, input.request_id)? else { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::RequestChanged, - )); - }; - if request.resource_id != input.resource_id || request.task_id != input.task_id { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::TaskIdentityMismatch, - )); - } - if !matches!( - &request.state, - ResourceRequestState::Assigned { loan_id } if *loan_id == input.loan_id - ) { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::RequestChanged, - )); - } - - let Some(loan) = select_non_closed_loan(&tx, input.resource_id)? else { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::LoanChanged, - )); - }; - if loan.id != input.loan_id || loan.resource_id != input.resource_id { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::LoanChanged, - )); - } - let LoanState::Active { - phase: - LoanPhase::Serving { - return_context, - current_request_id, - release_provenance, - }, - } = &loan.state - else { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::LoanChanged, - )); - }; - if *current_request_id != input.request_id - || !serving_release_provenance_matches( - &tx, - input.authority_machine, - &resource, - &loan, - return_context, - release_provenance, - )? - { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - if *current_request_id != input.request_id { - AssignedResourceTaskAttention::LoanChanged - } else { - AssignedResourceTaskAttention::ServingReleaseUnverified - }, - )); - } - - let (outcome, release_proof) = - match match_resource_task_layer(&tx, &request, input.authority_machine)? { - ResourceTaskLayerObservation::NotAccepted => { - tx.commit()?; - return Ok(AssignedResourceTaskReconcileOutcome::NotAccepted); - } - ResourceTaskLayerObservation::Active(progress) => { - tx.commit()?; - return Ok(AssignedResourceTaskReconcileOutcome::Active(progress)); - } - ResourceTaskLayerObservation::Finished { - outcome, - release_proof, - } => (outcome, release_proof), - ResourceTaskLayerObservation::Lost => { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::TaskLost, - )); - } - ResourceTaskLayerObservation::Attention(reason) => { - return Ok(AssignedResourceTaskReconcileOutcome::Attention(reason)); - } - }; - - if earlier_active_request_exists(&tx, input.resource_id, request.request_id)? { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::RequestChanged, - )); - } - - request.state = ResourceRequestState::Finished { - outcome: outcome.clone(), - }; - let request_state_json = encode_json(&request.state)?; - let changed = tx.execute( - "UPDATE resource_requests SET state_json = ?1 - WHERE request_id = ?2 AND task_id = ?3 AND resource_id = ?4 - AND acceptance_sequence = ?5 - AND json_extract(state_json, '$.type') = 'assigned' - AND json_extract(state_json, '$.loan_id') = ?6", - params![ - request_state_json, - request.request_id.0.to_string(), - request.task_id.to_string(), - request.resource_id.as_uuid().to_string(), - sqlite_integer(request.acceptance_sequence.get())?, - input.loan_id.as_uuid().to_string(), - ], - )?; - if changed != 1 { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::RequestChanged, - )); - } - - let next_revision = - input - .expected_state_revision - .next() - .ok_or(ResourceStoreError::Conflict( - ConflictReason::RevisionExhausted, - ))?; - let result = if let Some(mut next_request) = - next_queued_request_for_authority(&tx, input.authority_machine, input.resource_id)? - { - next_request.state = ResourceRequestState::Assigned { - loan_id: input.loan_id, - }; - let next_request_state_json = encode_json(&next_request.state)?; - let changed = tx.execute( - "UPDATE resource_requests SET state_json = ?1 - WHERE request_id = ?2 AND resource_id = ?3 AND acceptance_sequence = ?4 - AND json_extract(state_json, '$.type') = 'queued'", - params![ - next_request_state_json, - next_request.request_id.0.to_string(), - input.resource_id.as_uuid().to_string(), - sqlite_integer(next_request.acceptance_sequence.get())?, - ], - )?; - if changed != 1 { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::RequestChanged, - )); - } - - let updated_loan = Loan { - id: loan.id, - resource_id: input.resource_id, - state: LoanState::Active { - phase: LoanPhase::Serving { - return_context: return_context.clone(), - current_request_id: next_request.request_id, - release_provenance: release_provenance.clone(), - }, - }, - }; - update_serving_loan_for_task_completion(&tx, &loan, input.request_id, &updated_loan)?; - update_resource_revision_for_task_completion( - &tx, - input.authority_machine, - input.resource_id, - input.expected_state_revision, - next_revision, - )?; - - ResourceTaskCompletionResult::Assigned { - finished_request: request.clone(), - loan: updated_loan, - next_request, - state_revision: next_revision, - } - } else { - let action_id = ActionId::new(); - let updated_loan = Loan { - id: loan.id, - resource_id: input.resource_id, - state: LoanState::Active { - phase: LoanPhase::AwaitingReturn { - action_id, - return_context: return_context.clone(), - }, - }, - }; - update_serving_loan_for_task_completion(&tx, &loan, input.request_id, &updated_loan)?; - update_resource_revision_for_task_completion( - &tx, - input.authority_machine, - input.resource_id, - input.expected_state_revision, - next_revision, - )?; - - let notice = SupervisorNotice { - id: NoticeId::new(), - loan_id: loan.id, - action_id, - state_revision: next_revision, - destination: resource.supervisor, - assignment_revision: resource.assignment_revision, - payload: SupervisorNoticePayload::ReturnRequired { - return_context: return_context.clone(), - }, - delivery: SupervisorNoticeDelivery::Pending { attempts: 0 }, - }; - let notice = - insert_supervisor_notice_in_transaction(&tx, ¬ice).map_err(|error| match error { - SupervisorNoticeStoreError::Storage(error) => ResourceStoreError::Storage(error), - SupervisorNoticeStoreError::CorruptRecord { what, reason } => { - ResourceStoreError::CorruptRecord { what, reason } - } - _ => ResourceStoreError::Conflict(ConflictReason::SupervisorNoticeRejected), - })?; - ResourceTaskCompletionResult::ReturnRequired { - finished_request: request.clone(), - loan: updated_loan, - notice, - } - }; - - let receipt = ResourceTaskCompletionReceipt { - authority_machine: input.authority_machine, - resource_id: input.resource_id, - loan_id: input.loan_id, - request_id: input.request_id, - task_id: input.task_id, - expected_state_revision: input.expected_state_revision, - outcome, - release_proof, - result: result.clone(), - }; - tx.execute( - "INSERT INTO resource_task_completions (task_id, request_id, receipt_json) - VALUES (?1, ?2, ?3)", - params![ - input.task_id.to_string(), - input.request_id.0.to_string(), - encode_json(&receipt)?, - ], - )?; - tx.commit()?; - - Ok(AssignedResourceTaskReconcileOutcome::Completed(Box::new( - result, - ))) -} - -enum ResourceTaskLayerObservation { - NotAccepted, - Active(AssignedResourceTaskProgress), - Finished { - outcome: ExitReason, - release_proof: ResourceTaskReleaseProof, - }, - Lost, - Attention(AssignedResourceTaskAttention), -} - -fn match_resource_task_layer( - conn: &Connection, - request: &ResourceRequest, - authority_machine: MachineId, -) -> Result { - let identity = crate::store::executor_identity_for_resource_task_on(conn, request.task_id)?; - let task = - crate::store::task_by_id_on(conn, request.task_id).map_err(ResourceStoreError::TaskRow)?; - let has_event: bool = conn.query_row( - "SELECT EXISTS( - SELECT 1 FROM executor_outbox WHERE task_id=?1 - UNION ALL SELECT 1 FROM executor_event_receipts WHERE task_id=?1 - UNION ALL SELECT 1 FROM executor_event_cursors WHERE task_id=?1 - UNION ALL SELECT 1 FROM executor_event_routes WHERE task_id=?1 - )", - [request.task_id.to_string()], - |row| row.get(0), - )?; - let Some(identity) = identity else { - return Ok(if task.is_none() && !has_event { - ResourceTaskLayerObservation::NotAccepted - } else { - ResourceTaskLayerObservation::Attention( - AssignedResourceTaskAttention::TaskIdentityMismatch, - ) - }); - }; - let ExecutorIdentity::Accepted(record) = identity else { - return Ok(ResourceTaskLayerObservation::Attention( - AssignedResourceTaskAttention::TaskIdentityMismatch, - )); - }; - let Some(task) = task else { - return Ok(ResourceTaskLayerObservation::Attention( - AssignedResourceTaskAttention::TaskIdentityMismatch, - )); - }; - let saved_origin: Option = conn - .query_row( - "SELECT origin_machine FROM executor_identities WHERE task_id=?1", - [request.task_id.to_string()], - |row| row.get(0), - ) - .optional()?; - let event_matches = match crate::store::initial_queued_event_matches_on( - conn, - request.task_id, - request.origin_machine, - authority_machine, - ) { - Ok(matches) => matches, - Err(_) => { - return Ok(ResourceTaskLayerObservation::Attention( - AssignedResourceTaskAttention::TaskIdentityMismatch, - )); - } - }; - let spec_matches = record.current_spec() == Some(request.spec().as_normalized()); - let expected_origin = request.origin_machine.as_uuid().to_string(); - if record.task != request.task_id - || record.origin_machine != request.origin_machine - || record.execution_machine != authority_machine - || !record.has_valid_spec_owners() - || record.state != task.status() - || !spec_matches - || saved_origin.as_deref() != Some(expected_origin.as_str()) - || !has_event - || !event_matches - || !resource_task_row_matches(&task, request) - { - return Ok(ResourceTaskLayerObservation::Attention( - AssignedResourceTaskAttention::TaskIdentityMismatch, - )); - } - - match &task.state { - TaskState::Queued => Ok(ResourceTaskLayerObservation::Active( - AssignedResourceTaskProgress::Queued, - )), - TaskState::Running { .. } => Ok(ResourceTaskLayerObservation::Active( - AssignedResourceTaskProgress::Running, - )), - TaskState::Lost => Ok(ResourceTaskLayerObservation::Lost), - TaskState::Finished { reason } => { - match resource_task_release_proof(request, reason, task.work_exit_evidence()) { - Ok(release_proof) => Ok(ResourceTaskLayerObservation::Finished { - outcome: reason.clone(), - release_proof, - }), - Err(reason) => Ok(ResourceTaskLayerObservation::Attention(reason)), - } - } - } -} - -fn resource_task_row_matches(task: &crate::domain::TaskRow, request: &ResourceRequest) -> bool { - let spec = request.spec().as_normalized(); - task.id == request.task_id - && task.name.as_ref() == Some(&spec.name) - && task.thread == spec.thread - && task.workload == crate::invocation::persist_workload(&spec.workload) - && task.cwd == spec.cwd - && task.timeout == spec.timeout - // an empty binary never resolved, so its task failed before launch - && (task.binary.is_absolute() || task.binary.as_os_str().is_empty()) -} - -pub(super) fn resource_task_release_proof( - request: &ResourceRequest, - outcome: &ExitReason, - evidence: WorkExitEvidence, -) -> Result { - let confirmed = match evidence { - // the task layer records no started work only before the worker spawns its - // child or starts its container, or when a store CAS leaves Queued, which - // the worker needs to win first; any other outcome contradicts those paths - WorkExitEvidence::NoWorkStarted => { - return match outcome { - ExitReason::SpawnFailed { .. } => { - Ok(ResourceTaskReleaseProof::NoChildSpawnedAfterSpawnFailure) - } - ExitReason::Cancelled => { - Ok(ResourceTaskReleaseProof::NoChildSpawnedAfterQueuedCancel) - } - ExitReason::Exit { .. } | ExitReason::Signal { .. } => { - Err(AssignedResourceTaskAttention::InvalidNoChildSpawnEvidence) - } - }; - } - WorkExitEvidence::Unconfirmed => { - return Err(AssignedResourceTaskAttention::ExitWitnessUnconfirmed); - } - confirmed @ (WorkExitEvidence::ProcessGroupExited - | WorkExitEvidence::ContainerRemoved { .. }) => confirmed, - }; - let contract = - CommandOwnershipContract::for_queued_work(&request.spec().as_normalized().workload) - .map_err(AssignedResourceTaskAttention::OwnershipUncertain)?; - match (confirmed, contract) { - (WorkExitEvidence::ProcessGroupExited, CommandOwnershipContract::ForegroundExecutable) => { - Ok(ResourceTaskReleaseProof::ConfirmedProcessGroupExit) - } - ( - WorkExitEvidence::ContainerRemoved { container_id, .. }, - CommandOwnershipContract::Container, - ) => Ok(ResourceTaskReleaseProof::ConfirmedContainerRemoved { container_id }), - // a witness that the queued contract does not name proves nothing - _ => Err(AssignedResourceTaskAttention::ExitWitnessUnconfirmed), - } -} - -/// Classify queued work against the ownership contract -/// -/// A foreground command releases the resource on its process-group exit and a -/// container on its removal. Every other shape is refused -pub(super) fn resource_task_ownership_risk( - command: &CommandSpec, -) -> Option { - CommandOwnershipContract::for_queued_work(&command.as_normalized().workload).err() -} - -fn select_resource_task_completion_receipt( - conn: &Connection, - task_id: TaskId, -) -> Result, ResourceStoreError> { - let receipt_json: Option = conn - .query_row( - "SELECT receipt_json FROM resource_task_completions WHERE task_id=?1", - [task_id.to_string()], - |row| row.get(0), - ) - .optional()?; - let receipt = receipt_json - .map(|json| stored_json("resource task completion receipt", &json)) - .transpose()?; - Ok(receipt) -} - -// the receipt is the committed decision for this exact task, so a retry returns -// it even after later transitions; re-reading task rows here would let retention -// or later edits turn a settled completion into attention -fn retry_resource_task_completion( - receipt: &ResourceTaskCompletionReceipt, - input: AssignedResourceTaskReconcileInput, -) -> Result { - check_authority(receipt.authority_machine, input.authority_machine)?; - if receipt.resource_id != input.resource_id - || receipt.loan_id != input.loan_id - || receipt.request_id != input.request_id - || receipt.task_id != input.task_id - { - return Ok(AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::TaskIdentityMismatch, - )); - } - - Ok(AssignedResourceTaskReconcileOutcome::Completed(Box::new( - receipt.result.clone(), - ))) -} - -fn update_serving_loan_for_task_completion( - tx: &Transaction<'_>, - original: &Loan, - request_id: RequestId, - updated: &Loan, -) -> Result<(), ResourceStoreError> { - let state_json = encode_json(&updated.state)?; - let changed = tx.execute( - "UPDATE loans SET state_json = ?1 - WHERE id = ?2 AND resource_id = ?3 - AND json_extract(state_json, '$.type') = 'active' - AND json_extract(state_json, '$.phase.type') = 'serving' - AND json_extract(state_json, '$.phase.current_request_id') = ?4", - params![ - state_json, - original.id.as_uuid().to_string(), - original.resource_id.as_uuid().to_string(), - request_id.0.to_string(), - ], - )?; - if changed != 1 { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - } - - Ok(()) -} - -fn update_resource_revision_for_task_completion( - tx: &Transaction<'_>, - authority_machine: MachineId, - resource_id: ResourceId, - expected_revision: ResourceRevision, - next_revision: ResourceRevision, -) -> Result<(), ResourceStoreError> { - if !swap_resource_revision::( - tx, - authority_machine, - resource_id, - expected_revision, - next_revision, - )? { - return Err(ResourceStoreError::Conflict( - ConflictReason::ResourceRevisionChanged, - )); - } - - Ok(()) -} diff --git a/src/resource/store/cancellation.rs b/src/resource/store/cancellation.rs deleted file mode 100644 index 2009d6a..0000000 --- a/src/resource/store/cancellation.rs +++ /dev/null @@ -1,325 +0,0 @@ -//! Cancellation of resource requests before task activation - -use rusqlite::{Transaction, params}; - -use super::codec::{encode_json, sqlite_integer}; -use super::error::{ConflictReason, ResourceStoreError}; -use super::notice::{SupervisorNoticeStoreError, insert_supervisor_notice_in_transaction}; -use super::queue::{ - RequestIdentity, local_task_exists, next_queued_request_for_authority, prevention_exists, - request_matches_identity, select_executor_identity, task_id_exists, -}; -use super::revision::swap_resource_revision; -use super::rows::{ - check_resource_authority, select_non_closed_loan, select_request_by_id, select_resource, -}; -use crate::domain::TaskId; -use crate::machine::MachineId; -use crate::resource::{ - ActionId, Loan, LoanId, LoanPhase, LoanState, NoticeId, ResourceId, ResourceRequest, - ResourceRequestState, SupervisorNotice, SupervisorNoticeDelivery, SupervisorNoticePayload, -}; -use crate::store::Store; -use crate::submission::{ExecutorIdentity, PreAcceptanceRejection, RejectionTombstone, RequestId}; - -/// Result of cancelling a resource request before task activation -#[derive(Debug, Clone)] -pub enum QueueCancellationResult { - /// The exact request identity is retained to prevent delayed acceptance - PreventedBeforeAcceptance, - /// The saved request state after cancellation or a terminal-state retry - Request(Box), -} - -/// Cancel a request before activation inside a caller-owned transaction -/// -/// Nothing is committed here, so the caller can save its own record atomically -pub(crate) fn cancel_request_before_activation_on( - tx: &Transaction<'_>, - authority_machine: MachineId, - request_id: RequestId, - task_id: TaskId, - resource_id: ResourceId, - origin_machine: MachineId, -) -> Result { - let identity = RequestIdentity { - request_id, - task_id, - resource_id, - origin_machine, - }; - - if let Some(mut saved) = select_request_by_id(tx, request_id)? { - if !request_matches_identity(&saved, identity) { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestIdentityMismatch, - )); - } - check_resource_authority(tx, saved.resource_id, authority_machine)?; - - match saved.state.clone() { - ResourceRequestState::Queued => { - cancel_queued_request(tx, &mut saved, authority_machine)?; - } - ResourceRequestState::Assigned { loan_id } => { - cancel_assigned_request(tx, &mut saved, loan_id, authority_machine)?; - } - _ => {} - } - - return Ok(QueueCancellationResult::Request(Box::new(saved))); - } - - if task_id_exists(tx, task_id)? { - return Err(ResourceStoreError::Conflict( - ConflictReason::TaskIdentityInUse, - )); - } - - let prevented = prevention_exists(tx, identity)?; - check_resource_authority(tx, resource_id, authority_machine)?; - if !prevented { - tx.execute( - "INSERT INTO resource_request_preventions ( - request_id, task_id, resource_id, origin_machine - ) VALUES (?1, ?2, ?3, ?4)", - params![ - request_id.0.to_string(), - task_id.to_string(), - resource_id.as_uuid().to_string(), - origin_machine.as_uuid().to_string(), - ], - )?; - } - let executor_identity = Store::reject_execution_in( - tx, - &cancellation_tombstone(task_id, origin_machine, authority_machine), - )?; - if matches!(executor_identity, ExecutorIdentity::Accepted(_)) { - return Err(ResourceStoreError::ExecutorAlreadyAccepted { task: task_id }); - } - Ok(QueueCancellationResult::PreventedBeforeAcceptance) -} - -fn cancel_queued_request( - tx: &Transaction<'_>, - saved: &mut ResourceRequest, - authority: MachineId, -) -> Result<(), ResourceStoreError> { - let cancelled = ResourceRequestState::CancelledBeforeLaunch; - let state_json = encode_json(&cancelled)?; - let updated = tx.execute( - "UPDATE resource_requests SET state_json = ?1 - WHERE request_id = ?2 - AND json_extract(state_json, '$.type') = 'queued'", - params![state_json, saved.request_id.0.to_string()], - )?; - if updated != 1 { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestStateChanged, - )); - } - - let executor_identity = Store::reject_execution_in( - tx, - &cancellation_tombstone(saved.task_id, saved.origin_machine, authority), - )?; - if matches!(executor_identity, ExecutorIdentity::Accepted(_)) { - return Err(ResourceStoreError::ExecutorAlreadyAccepted { - task: saved.task_id, - }); - } - - saved.state = cancelled; - Ok(()) -} - -fn cancel_assigned_request( - tx: &Transaction<'_>, - saved: &mut ResourceRequest, - loan_id: LoanId, - authority: MachineId, -) -> Result<(), ResourceStoreError> { - if matches!( - select_executor_identity(tx, saved.task_id)?, - Some(ExecutorIdentity::Accepted(_)) - ) { - return Err(ResourceStoreError::ExecutorAlreadyAccepted { - task: saved.task_id, - }); - } - if local_task_exists(tx, saved.task_id)? { - return Err(ResourceStoreError::Conflict( - ConflictReason::TaskIdentityInUse, - )); - } - - let executor_identity = Store::reject_execution_in( - tx, - &cancellation_tombstone(saved.task_id, saved.origin_machine, authority), - )?; - if matches!(executor_identity, ExecutorIdentity::Accepted(_)) { - return Err(ResourceStoreError::ExecutorAlreadyAccepted { - task: saved.task_id, - }); - } - - let resource = - select_resource(tx, saved.resource_id)?.ok_or(ResourceStoreError::ResourceNotFound)?; - let loan = select_non_closed_loan(tx, saved.resource_id)? - .ok_or(ResourceStoreError::Conflict(ConflictReason::LoanChanged))?; - if loan.id != loan_id || loan.resource_id != saved.resource_id { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - } - let (return_context, release_provenance) = match &loan.state { - LoanState::Active { - phase: - LoanPhase::Serving { - return_context, - current_request_id, - release_provenance, - }, - } if *current_request_id == saved.request_id => { - (return_context.clone(), release_provenance.clone()) - } - _ => return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)), - }; - - let next_revision = resource - .state_revision - .next() - .ok_or(ResourceStoreError::Conflict( - ConflictReason::RevisionExhausted, - ))?; - let cancelled = ResourceRequestState::CancelledBeforeLaunch; - let state_json = encode_json(&cancelled)?; - let changed = tx.execute( - "UPDATE resource_requests SET state_json = ?1 - WHERE request_id = ?2 AND task_id = ?3 AND resource_id = ?4 - AND json_extract(state_json, '$.type') = 'assigned' - AND json_extract(state_json, '$.loan_id') = ?5", - params![ - state_json, - saved.request_id.0.to_string(), - saved.task_id.to_string(), - saved.resource_id.as_uuid().to_string(), - loan_id.as_uuid().to_string(), - ], - )?; - if changed != 1 { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestStateChanged, - )); - } - saved.state = cancelled; - - let updated_loan = if let Some(mut next_request) = - next_queued_request_for_authority(tx, authority, saved.resource_id)? - { - next_request.state = ResourceRequestState::Assigned { loan_id }; - let next_state_json = encode_json(&next_request.state)?; - let changed = tx.execute( - "UPDATE resource_requests SET state_json = ?1 - WHERE request_id = ?2 AND resource_id = ?3 AND acceptance_sequence = ?4 - AND json_extract(state_json, '$.type') = 'queued'", - params![ - next_state_json, - next_request.request_id.0.to_string(), - saved.resource_id.as_uuid().to_string(), - sqlite_integer(next_request.acceptance_sequence.get())?, - ], - )?; - if changed != 1 { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestStateChanged, - )); - } - - Loan { - id: loan.id, - resource_id: saved.resource_id, - state: LoanState::Active { - phase: LoanPhase::Serving { - return_context, - current_request_id: next_request.request_id, - release_provenance, - }, - }, - } - } else { - let action_id = ActionId::new(); - let updated_loan = Loan { - id: loan.id, - resource_id: saved.resource_id, - state: LoanState::Active { - phase: LoanPhase::AwaitingReturn { - action_id, - return_context: return_context.clone(), - }, - }, - }; - let notice = SupervisorNotice { - id: NoticeId::new(), - loan_id: loan.id, - action_id, - state_revision: next_revision, - destination: resource.supervisor, - assignment_revision: resource.assignment_revision, - payload: SupervisorNoticePayload::ReturnRequired { return_context }, - delivery: SupervisorNoticeDelivery::Pending { attempts: 0 }, - }; - insert_supervisor_notice_in_transaction(tx, ¬ice).map_err(|error| match error { - SupervisorNoticeStoreError::Storage(error) => ResourceStoreError::Storage(error), - SupervisorNoticeStoreError::CorruptRecord { what, reason } => { - ResourceStoreError::CorruptRecord { what, reason } - } - _ => ResourceStoreError::Conflict(ConflictReason::SupervisorNoticeRejected), - })?; - updated_loan - }; - - let loan_state_json = encode_json(&updated_loan.state)?; - let changed = tx.execute( - "UPDATE loans SET state_json = ?1 - WHERE id = ?2 AND resource_id = ?3 - AND json_extract(state_json, '$.type') = 'active' - AND json_extract(state_json, '$.phase.type') = 'serving' - AND json_extract(state_json, '$.phase.current_request_id') = ?4", - params![ - loan_state_json, - loan.id.as_uuid().to_string(), - saved.resource_id.as_uuid().to_string(), - saved.request_id.0.to_string(), - ], - )?; - if changed != 1 { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - } - - if !swap_resource_revision::( - tx, - authority, - saved.resource_id, - resource.state_revision, - next_revision, - )? { - return Err(ResourceStoreError::Conflict( - ConflictReason::ResourceRevisionChanged, - )); - } - - Ok(()) -} - -fn cancellation_tombstone( - task: TaskId, - origin_machine: MachineId, - execution_machine: MachineId, -) -> RejectionTombstone { - RejectionTombstone { - task, - origin_machine, - execution_machine, - reason: PreAcceptanceRejection::Cancelled.as_str().into(), - } -} diff --git a/src/resource/store/checkpoint.rs b/src/resource/store/checkpoint.rs deleted file mode 100644 index 5451bf9..0000000 --- a/src/resource/store/checkpoint.rs +++ /dev/null @@ -1,215 +0,0 @@ -//! Durable checkpoint-stop evidence for one release action - -use rusqlite::{Connection, OptionalExtension, Transaction, params}; - -use super::error::{ResourceStoreError, TrainerAttemptAssociationStoreError}; -use crate::domain::{ProcessStatus, TaskId}; -use crate::error::AppError; -use crate::resource::trainer_publication::WatcherError; -use crate::resource::{ActionId, ReleaseCheckpointState, ResourceId}; - -/// Failure to capture or reserve exact checkpoint-stop evidence -#[derive(Debug, thiserror::Error)] -pub(crate) enum ReleaseCheckpointError { - /// Resource authority or stored resource data failed validation - #[error(transparent)] - Resource(#[from] ResourceStoreError), - /// Saved trainer association could not be read or does not match the action - #[error(transparent)] - TrainerAssociation(#[from] TrainerAttemptAssociationStoreError), - /// Trainer publication evidence could not be verified - #[error(transparent)] - Watcher(#[from] WatcherError), - /// The exact task row could not be read - #[error(transparent)] - Task(#[from] AppError), - /// The exact action is not in the expected AwaitingRelease phase - #[error("release action {action_id:?} is not awaiting checkpoint evidence")] - NotAwaitingRelease { - /// Action required by the caller - action_id: ActionId, - }, - /// A watcher identity must be bound before baseline capture - #[error("release action {action_id:?} has no saved watcher identity")] - WatcherIntentMissing { - /// Action that still needs its preallocated watcher identity - action_id: ActionId, - }, - /// The baseline has not been durably captured for this watcher - #[error("release action {action_id:?} has no durable checkpoint baseline")] - BaselineMissing { - /// Action whose baseline is absent - action_id: ActionId, - }, - /// No exact stop decision is reserved for the requested release action - #[error("release action {action_id:?} has no reserved stop decision")] - StopDecisionMissing { - /// Action whose stop decision is absent - action_id: ActionId, - }, - /// The associated trainer task is missing on the authority - #[error("observed trainer task {task_id} is missing locally")] - TaskMissing { - /// Exact task observed by the release action - task_id: TaskId, - }, - /// The observed trainer task is not running on the authority - #[error("observed trainer task {task_id} is not running (state {state:?})")] - TrainerTaskNotRunning { - /// Exact task observed by the release action - task_id: TaskId, - /// Current persisted process state - state: ProcessStatus, - }, - /// The accepted trainer command or its task row changed after association - #[error("accepted trainer command binding changed for task {task_id}")] - TrainerCommandBindingChanged { - /// Exact task observed by the release action - task_id: TaskId, - }, - /// Another cancellation was already recorded for the trainer task - #[error("trainer task {task_id} already has a cancellation marker")] - TrainerCancellationConflict { - /// Exact task observed by the release action - task_id: TaskId, - }, - /// The accepted watcher identity no longer matches its saved launch intent - #[error("release watcher task {task_id} conflicts with its saved launch identity")] - WatcherIdentityConflict { - /// Exact watcher task reserved for this release action - task_id: TaskId, - }, - /// The supplied stop decision is not the saved action decision - #[error("stop decision for release action {action_id:?} does not match its reservation")] - StopDecisionMismatch { - /// Action whose decision did not match - action_id: ActionId, - }, - /// The release action has no saved trainer-attempt association - #[error("release action has no trainer association for task {task_id}")] - TrainerAssociationMissing { - /// Exact trainer task observed by the release action - task_id: TaskId, - }, - /// The action or its saved checkpoint evidence changed - #[error("release checkpoint evidence conflicts with the saved action")] - Conflict, - /// The selected publication no longer matches its saved immutable identity - #[error("selected checkpoint for release action {action_id:?} changed")] - SelectedCheckpointChanged { - /// Action whose saved publication changed - action_id: ActionId, - }, - /// Persisted evidence does not satisfy its typed invariants - #[error("invalid saved release checkpoint evidence: {0}")] - InvalidStoredEvidence(&'static str), - /// SQLite or stored data failed - #[error("release checkpoint evidence storage error: {0}")] - Storage(#[from] rusqlite::Error), - /// A typed evidence document could not be serialized - #[error("release checkpoint evidence encoding failed: {0}")] - Serialization(#[from] serde_json::Error), -} - -/// Result data retained after the trainer cancellation marker commits -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) struct ReleaseCheckpointCancellationResult { - /// Fixed stop decision that authorized cancellation - pub(crate) decision: crate::resource::ReleaseCheckpointStopDecision, - /// Exact trainer task and durable marker time - pub(crate) cancellation: crate::resource::ReleaseCheckpointCancellation, -} - -/// Typed result of committing a reserved exact-task cancellation -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) enum ReleaseCheckpointCancellationOutcome { - /// Decision and task marker committed in one transaction - Committed(ReleaseCheckpointCancellationResult), - /// An exact retry returned the existing committed reservation - AlreadyCommitted(ReleaseCheckpointCancellationResult), - /// The reserved watcher task has not reached its accepted running boundary - WatcherNotReady { - /// Exact watcher task reserved for this release action - watcher_task_id: TaskId, - }, -} - -pub(crate) fn release_checkpoint_state_for_action( - conn: &Connection, - resource_id: ResourceId, - action_id: ActionId, -) -> Result, ReleaseCheckpointError> { - let saved = conn - .query_row( - "SELECT resource_id, state_json FROM resource_release_checkpoint_states - WHERE action_id = ?1", - [action_id.as_uuid().to_string()], - |row| Ok((row.get::<_, String>(0)?, row.get::<_, String>(1)?)), - ) - .optional()?; - let Some((saved_resource, state_json)) = saved else { - return Ok(None); - }; - let saved_resource = saved_resource - .parse::() - .map_err(|_| ReleaseCheckpointError::InvalidStoredEvidence("resource id is invalid"))?; - let state: ReleaseCheckpointState = serde_json::from_str(&state_json) - .map_err(|error| ResourceStoreError::corrupt("release checkpoint state", error))?; - state - .validate() - .map_err(ReleaseCheckpointError::InvalidStoredEvidence)?; - if saved_resource != resource_id - || state.action.resource_id != resource_id - || state.action.action_id != action_id - { - return Err(ReleaseCheckpointError::InvalidStoredEvidence( - "row identities do not match the typed state", - )); - } - - Ok(Some((state, state_json))) -} - -pub(super) fn insert_release_checkpoint_state( - tx: &Transaction<'_>, - state: &ReleaseCheckpointState, -) -> Result<(), ReleaseCheckpointError> { - state - .validate() - .map_err(ReleaseCheckpointError::InvalidStoredEvidence)?; - tx.execute( - "INSERT INTO resource_release_checkpoint_states (action_id, resource_id, state_json) - VALUES (?1, ?2, ?3)", - params![ - state.action.action_id.as_uuid().to_string(), - state.action.resource_id.as_uuid().to_string(), - serde_json::to_string(state)?, - ], - )?; - Ok(()) -} - -pub(crate) fn update_release_checkpoint_state( - tx: &Transaction<'_>, - previous_json: &str, - state: &ReleaseCheckpointState, -) -> Result<(), ReleaseCheckpointError> { - state - .validate() - .map_err(ReleaseCheckpointError::InvalidStoredEvidence)?; - let changed = tx.execute( - "UPDATE resource_release_checkpoint_states SET state_json = ?1 - WHERE action_id = ?2 AND resource_id = ?3 AND state_json = ?4", - params![ - serde_json::to_string(state)?, - state.action.action_id.as_uuid().to_string(), - state.action.resource_id.as_uuid().to_string(), - previous_json, - ], - )?; - if changed != 1 { - return Err(ReleaseCheckpointError::Conflict); - } - - Ok(()) -} diff --git a/src/resource/store/codec.rs b/src/resource/store/codec.rs deleted file mode 100644 index 0031969..0000000 --- a/src/resource/store/codec.rs +++ /dev/null @@ -1,107 +0,0 @@ -//! Column and JSON conversion helpers - -use std::fmt; - -use rusqlite::Row; -use rusqlite::types::FromSql; -use serde::Serialize; -use serde::de::DeserializeOwned; -use uuid::Uuid; - -use super::error::ResourceStoreError; -use crate::resource::NilIdentity; - -pub(super) fn sqlite_integer(value: u64) -> Result { - i64::try_from(value).map_err(|err| { - ResourceStoreError::Storage(rusqlite::Error::ToSqlConversionFailure(Box::new(err))) - }) -} - -pub(super) fn encode_json(value: &T) -> Result { - serde_json::to_string(value).map_err(|err| { - ResourceStoreError::Storage(rusqlite::Error::ToSqlConversionFailure(Box::new(err))) - }) -} - -/// Failure to read one saved value -/// -/// SQLite failures stay retryable, while a value that cannot decode cannot -/// succeed on retry and needs attention -#[derive(Debug)] -pub(crate) enum StoredReadError { - /// SQLite failed while reading the value - Storage(rusqlite::Error), - /// The saved value cannot decode or fails its integrity check - Corrupt { - /// Kind of record that failed to decode - what: &'static str, - /// Decoder or integrity failure - reason: String, - }, -} - -impl StoredReadError { - /// Classify one saved value that cannot be decoded or verified - pub(crate) fn corrupt(what: &'static str, reason: impl fmt::Display) -> Self { - Self::Corrupt { - what, - reason: reason.to_string(), - } - } -} - -impl From for StoredReadError { - fn from(error: rusqlite::Error) -> Self { - Self::Storage(error) - } -} - -/// Read one column of a saved record, classifying a type mismatch as corruption -pub(super) fn stored_column( - row: &Row<'_>, - index: usize, - what: &'static str, -) -> Result { - row.get(index).map_err(|error| match error { - rusqlite::Error::InvalidColumnType(..) - | rusqlite::Error::FromSqlConversionFailure(..) - | rusqlite::Error::IntegralValueOutOfRange(..) => StoredReadError::corrupt(what, error), - error => StoredReadError::Storage(error), - }) -} - -/// Decode a saved JSON record, classifying a decode failure as corruption -pub(super) fn stored_json( - what: &'static str, - value: &str, -) -> Result { - serde_json::from_str(value).map_err(|error| StoredReadError::corrupt(what, error)) -} - -/// Parse a saved UUID, classifying a parse failure as corruption -pub(super) fn stored_uuid(what: &'static str, value: &str) -> Result { - Uuid::parse_str(value).map_err(|error| StoredReadError::corrupt(what, error)) -} - -/// Parse a saved non-nil record identity, classifying a nil or malformed value as corruption -pub(super) fn stored_id(what: &'static str, value: &str) -> Result -where - T: TryFrom, -{ - T::try_from(stored_uuid(what, value)?).map_err(|error| StoredReadError::corrupt(what, error)) -} - -/// Convert a saved counter, classifying a negative value as corruption -pub(super) fn stored_count(what: &'static str, value: i64) -> Result { - u64::try_from(value).map_err(|error| StoredReadError::corrupt(what, error)) -} - -/// Collect rows whose decoder separates SQLite failures from corrupt values -pub(super) fn collect_decoded( - rows: impl Iterator>>, -) -> Result, E> -where - E: From, -{ - rows.map(|row| row?).collect() -} diff --git a/src/resource/store/error.rs b/src/resource/store/error.rs deleted file mode 100644 index fb695b1..0000000 --- a/src/resource/store/error.rs +++ /dev/null @@ -1,308 +0,0 @@ -//! Shared failures for authority-local resource storage - -use crate::error::AppError; -use std::fmt; - -use super::codec::StoredReadError; -use crate::domain::TaskId; -use crate::machine::MachineId; -use crate::resource::{CommandSpecError, ResourceId, ResourceTaskOwnershipRisk}; -use crate::store::IdentityError; - -/// Failure to accept or read authority-local resource data -#[derive(Debug, thiserror::Error)] -pub(crate) enum ResourceStoreError { - /// Saved authority state contradicts the requested identity or transition - #[error("resource identity conflict: {0}")] - Conflict(ConflictReason), - /// The resource identity was first registered with different content - #[error("resource {resource:?} is already registered with different content")] - RegistrationConflict { - /// Resource whose saved registration differs - resource: ResourceId, - }, - /// The request was cancelled before it entered the resource queue - #[error("resource request was prevented before acceptance")] - Prevented, - /// The referenced resource does not exist - #[error("resource not found")] - ResourceNotFound, - /// The local daemon does not own the resource authority - #[error("resource authority mismatch: expected {expected}, found {found}")] - WrongAuthority { - /// Fixed authority recorded for this resource - expected: MachineId, - /// Machine identity supplied by the daemon - found: MachineId, - }, - /// The local origin route required for callback delivery is missing - #[error("resource callback route for task {task} is missing")] - OriginRouteNotFound { - /// Preallocated task identity whose saved origin route is missing - task: TaskId, - }, - /// A prior ordinary execution acceptance won before queue cancellation - #[error("task identity {task} was accepted before resource cancellation")] - ExecutorAlreadyAccepted { - /// Task identity that was already accepted - task: TaskId, - }, - /// The executor identity table rejected this atomic resource transition - #[error(transparent)] - Identity(#[from] IdentityError), - /// An agent workload cannot enter the finite command queue - #[error(transparent)] - InvalidCommandSpec(#[from] CommandSpecError), - /// The command falls outside the foreground ownership contract, so its end cannot release the resource - #[error("resource command can outlive or hide from its task process group ({risk:?})")] - UnsupportedCommandOwnership { - /// First contract violation found in the command or its entry point - risk: ResourceTaskOwnershipRisk, - }, - /// The spec's host inputs cannot be used on this authority, so it can never launch here - #[error("{}", .0.error)] - HostInputRejected(crate::spec::HostInputError), - /// The saved command cannot be prepared on the executor machine - #[error("resource command cannot be prepared: {0}")] - TaskPreparation(#[from] AppError), - /// The accepted resource task is inconsistent with its durable task row - #[error("accepted resource task storage is inconsistent: {0}")] - TaskRow(AppError), - /// A saved return decision could not be read for a resource view - #[error("saved return decision could not be read: {0}")] - ReturnDecisionRead(String), - /// The first durable task event is missing or conflicts with its identity - #[error(transparent)] - Event(#[from] crate::events::EventError), - /// A saved record cannot be decoded or fails its integrity check, so a retry cannot succeed - #[error("corrupt stored {what}: {reason}")] - CorruptRecord { - /// Kind of record that failed to decode - what: &'static str, - /// Decoder or integrity failure - reason: String, - }, - /// SQLite failed - #[error("resource storage error: {0}")] - Storage(#[from] rusqlite::Error), -} - -impl ResourceStoreError { - /// Classify one saved record that cannot be decoded or verified - pub(crate) fn corrupt(what: &'static str, reason: impl fmt::Display) -> Self { - Self::CorruptRecord { - what, - reason: reason.to_string(), - } - } -} - -impl From for ResourceStoreError { - fn from(error: StoredReadError) -> Self { - match error { - StoredReadError::Storage(error) => Self::Storage(error), - StoredReadError::Corrupt { what, reason } => Self::CorruptRecord { what, reason }, - } - } -} - -/// Failure to bind or read durable trainer-attempt evidence on its resource authority -#[derive(Debug, thiserror::Error)] -pub(crate) enum TrainerAttemptAssociationStoreError { - /// Resource existence or authority ownership failed - #[error(transparent)] - Resource(#[from] ResourceStoreError), - /// The task does not match the resource's registered background task - #[error("task {task_id} is not the registered background task")] - TaskNotRegistered { - /// Task supplied for the association - task_id: TaskId, - }, - /// The registered task row is missing on this authority - #[error("registered background task {task_id} is missing locally")] - TaskMissing { - /// Exact task registered on the resource - task_id: TaskId, - }, - /// The registered task row is not running - #[error("registered background task {task_id} is not running (state {state})")] - TaskNotRunning { - /// Exact task registered on the resource - task_id: TaskId, - /// Process state saved on this authority - state: String, - }, - /// No accepted executor identity exists for the registered task - #[error("accepted executor identity for task {task_id} is missing")] - IdentityMissing { - /// Exact task registered on the resource - task_id: TaskId, - }, - /// The executor identity is a rejection, not an accepted task - #[error("executor identity for task {task_id} is not accepted")] - IdentityNotAccepted { - /// Exact task registered on the resource - task_id: TaskId, - }, - /// The accepted identity does not belong to this task or authority - #[error("accepted executor identity for task {task_id} has a different owner")] - IdentityMismatch { - /// Exact task registered on the resource - task_id: TaskId, - }, - /// The accepted executor identity is not running - #[error("accepted executor identity for task {task_id} is not running")] - IdentityNotRunning { - /// Exact task registered on the resource - task_id: TaskId, - }, - /// The accepted identity has no current normalized request spec - #[error("accepted executor identity for task {task_id} has no normalized spec")] - NormalizedSpecMissing { - /// Exact task registered on the resource - task_id: TaskId, - }, - /// The accepted normalized spec does not match the local task row - #[error("accepted normalized spec does not match task row {task_id}")] - NormalizedSpecMismatch { - /// Exact task registered on the resource - task_id: TaskId, - }, - /// The accepted task binding changed after its command shape was checked - #[error("accepted task binding for task {task_id} changed during command-shape validation")] - BindingChanged { - /// Exact task whose accepted binding changed - task_id: TaskId, - }, - /// The accepted command does not match the maintained direct-segment contract - #[error(transparent)] - DirectSegmentCommandShape( - #[from] crate::resource::command_shape::DirectSegmentCommandShapeError, - ), - /// A non-closed loan does not await release of this exact task - #[error("resource {resource_id:?} has an incompatible active loan")] - ActiveLoanConflict { - /// Resource whose active loan blocks this association - resource_id: ResourceId, - }, - /// The resource already has a different immutable association - #[error( - "trainer attempt association for resource {resource_id:?} conflicts with saved evidence" - )] - Conflict { - /// Resource whose immutable association conflicts with this request - resource_id: ResourceId, - }, - /// The task is already associated with another resource - #[error("trainer task {task_id} is already associated with another resource")] - TaskAlreadyAssociated { - /// Task already owned by a trainer association - task_id: TaskId, - }, - /// Persisted association data is malformed or disagrees with its row identity - #[error("invalid saved trainer attempt association: {0}")] - InvalidStoredAssociation(String), - /// A validated task row or JSON conversion failed - #[error(transparent)] - App(#[from] AppError), - /// Executor identity storage failed - #[error(transparent)] - Identity(#[from] IdentityError), - /// SQLite or stored data failed - #[error("trainer attempt association storage error: {0}")] - Storage(#[from] rusqlite::Error), -} - -/// Saved authority state that a resource transition contradicts -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(crate) enum ConflictReason { - /// A request identity was reused with a different task, resource, or origin - RequestIdentityMismatch, - /// An exact request retry carries a different command - RequestSpecMismatch, - /// The selected request is missing from the authority queue - RequestMissing, - /// The request left the state that this transition expects - RequestStateChanged, - /// A request before the selected request in serving order is still queued or assigned - EarlierRequestActive, - /// The next queue rank would overflow the stored integer - QueueRankExhausted, - /// A prevention record holds a different identity for this request or task - PreventionIdentityMismatch, - /// The task identity already belongs to a request, task row, or executor identity - TaskIdentityInUse, - /// The task row or its first event disagrees with the accepted request - TaskRowMismatch, - /// The executor identity is a rejection or disagrees with the request - ExecutorIdentityMismatch, - /// The saved origin route disagrees with the request - OriginRouteMismatch, - /// The resource revision changed before the transition committed - ResourceRevisionChanged, - /// The resource revision cannot advance past its maximum value - RevisionExhausted, - /// The resource supervisor or registered background task changed - ResourceAssignmentChanged, - /// The loan changed or no longer owns this request - LoanChanged, - /// The loan is not in the release action that this transition expects - ReleaseActionChanged, - /// The release notice is missing or disagrees with its action - ReleaseNoticeMismatch, - /// The release watcher identity is nil or reuses the observed task or request - WatcherIdentityInvalid, - /// Another loan, request, or task already claims the release watcher identity - WatcherIdentityClaimed, - /// A different release watcher is already bound to this action - WatcherIntentMismatch, - /// The Serving loan has no verified release provenance - ServingReleaseUnverified, - /// The supervisor notice for this transition conflicts with a saved notice - SupervisorNoticeRejected, - /// The saved registration receipt names a different resource - RegistrationReceiptMismatch, - /// A cancellation disagrees with its saved receipt or route proof - CancellationReceiptMismatch, -} - -impl ConflictReason { - /// Stable description for logs and API error messages - #[must_use] - pub(crate) const fn as_str(self) -> &'static str { - match self { - Self::RequestIdentityMismatch => "request identity belongs to different content", - Self::RequestSpecMismatch => "request retry carries a different command", - Self::RequestMissing => "request is missing from the authority queue", - Self::RequestStateChanged => "request state changed", - Self::EarlierRequestActive => { - "an earlier request in serving order is still queued or assigned" - } - Self::QueueRankExhausted => "resource queue rank cannot advance", - Self::PreventionIdentityMismatch => "prevention record holds a different identity", - Self::TaskIdentityInUse => "task identity is already in use", - Self::TaskRowMismatch => "task row disagrees with the accepted request", - Self::ExecutorIdentityMismatch => "executor identity disagrees with the request", - Self::OriginRouteMismatch => "origin route disagrees with the request", - Self::ResourceRevisionChanged => "resource revision changed", - Self::RevisionExhausted => "resource revision cannot advance", - Self::ResourceAssignmentChanged => "resource assignment changed", - Self::LoanChanged => "loan changed", - Self::ReleaseActionChanged => "release action changed", - Self::ReleaseNoticeMismatch => "release notice is missing or differs", - Self::WatcherIdentityInvalid => "release watcher identity is invalid", - Self::WatcherIdentityClaimed => "release watcher identity is already claimed", - Self::WatcherIntentMismatch => "a different release watcher is bound", - Self::ServingReleaseUnverified => "serving release provenance is unverified", - Self::SupervisorNoticeRejected => "supervisor notice conflicts with saved state", - Self::RegistrationReceiptMismatch => "registration receipt names another resource", - Self::CancellationReceiptMismatch => "cancellation disagrees with its saved receipt", - } - } -} - -impl fmt::Display for ConflictReason { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - formatter.write_str(self.as_str()) - } -} diff --git a/src/resource/store/notice.rs b/src/resource/store/notice.rs deleted file mode 100644 index 5a1f42e..0000000 --- a/src/resource/store/notice.rs +++ /dev/null @@ -1,443 +0,0 @@ -//! Durable supervisor notices and their bounded delivery attempts - -use chrono::Utc; -use rusqlite::{Connection, OptionalExtension, Row, Transaction, TransactionBehavior, params}; - -use super::codec::{StoredReadError, collect_decoded, stored_column, stored_id, stored_json}; -use super::return_window::open_return_window_on; -use crate::resource::{ - ActionId, AssignmentRevision, DeliveryAttemptId, LoanId, LoanState, NoticeId, - SupervisorAddress, SupervisorNotice, SupervisorNoticeDelivery, SupervisorNoticePayload, -}; - -const SUPERVISOR_NOTICE_MAX_ATTEMPTS: u8 = 3; -pub(super) const INTERRUPTED_DELIVERY_ERROR: &str = - "delivery attempt interrupted before settlement"; -pub(super) const RETARGETED_DELIVERY_ERROR: &str = - "delivery attempt invalidated by supervisor reassignment"; - -/// Failure to persist, query, or transition a supervisor notice -#[derive(Debug, thiserror::Error)] -pub(crate) enum SupervisorNoticeStoreError { - /// The notice identity or action already belongs to different content - #[error("supervisor notice identity conflict")] - Conflict, - /// The requested notice does not exist - #[error("supervisor notice not found")] - NotFound, - /// The notice has already been delivered and cannot be retargeted - #[error("delivered supervisor notice cannot be retargeted")] - AlreadyDelivered, - /// The loan no longer waits for the decision this notice asks for - #[error("supervisor notice action is no longer awaited")] - ActionNoLongerAwaited, - /// A delivery attempt is already in flight - #[error("supervisor notice delivery attempt is already in flight")] - AttemptInFlight, - /// The supplied delivery attempt is not the current in-flight attempt - #[error("supervisor notice delivery attempt is stale")] - StaleAttempt, - /// The notice has exhausted its bounded delivery attempts - #[error("supervisor notice delivery attempt budget is exhausted")] - AttemptBudgetExhausted, - /// The notice assignment changed since the caller read it - #[error("supervisor notice assignment revision is stale")] - StaleAssignmentRevision, - /// A retarget revision must be newer than its expected revision - #[error("supervisor notice assignment revision must increase")] - AssignmentRevisionMustIncrease, - /// A new notice must begin in the untouched pending state - #[error("new supervisor notice must have zero pending attempts")] - InvalidInitialDelivery, - /// A saved notice cannot be decoded or fails its integrity check, so a retry cannot succeed - #[error("corrupt stored {what}: {reason}")] - CorruptRecord { - /// Kind of record that failed to decode - what: &'static str, - /// Decoder or integrity failure - reason: String, - }, - /// SQLite failed or a notice could not be serialized - #[error("supervisor notice storage error: {0}")] - Storage(#[from] rusqlite::Error), -} - -impl From for SupervisorNoticeStoreError { - fn from(error: StoredReadError) -> Self { - match error { - StoredReadError::Storage(error) => Self::Storage(error), - StoredReadError::Corrupt { what, reason } => Self::CorruptRecord { what, reason }, - } - } -} - -/// Insert a notice without committing the caller's loan transaction -pub(crate) fn insert_supervisor_notice_in_transaction( - tx: &Transaction<'_>, - notice: &SupervisorNotice, -) -> Result { - let existing = { - let mut statement = tx.prepare( - "SELECT id, loan_id, action_id, notice_json - FROM resource_supervisor_notices - WHERE id = ?1 OR action_id = ?2", - )?; - let rows = statement.query_map( - params![ - notice.id.as_uuid().to_string(), - notice.action_id.as_uuid().to_string() - ], - |row| Ok(decode_supervisor_notice_record(row)), - )?; - collect_decoded::<_, StoredReadError>(rows)? - }; - - if let [(saved, _)] = existing.as_slice() - && same_supervisor_notice_content(saved, notice) - { - return Ok(saved.clone()); - } - if !existing.is_empty() { - return Err(SupervisorNoticeStoreError::Conflict); - } - - if notice.delivery != (SupervisorNoticeDelivery::Pending { attempts: 0 }) { - return Err(SupervisorNoticeStoreError::InvalidInitialDelivery); - } - - tx.execute( - "INSERT INTO resource_supervisor_notices (id, loan_id, action_id, notice_json) - VALUES (?1, ?2, ?3, ?4)", - params![ - notice.id.as_uuid().to_string(), - notice.loan_id.as_uuid().to_string(), - notice.action_id.as_uuid().to_string(), - encode_supervisor_notice(notice)?, - ], - )?; - // a return notice opens its action, so the decision window starts with it - if matches!( - notice.payload, - SupervisorNoticePayload::ReturnRequired { .. } - ) { - open_return_window_on(tx, notice.action_id, notice.loan_id, Utc::now())?; - } - Ok(notice.clone()) -} - -/// Read one notice by its stable deduplication identity -pub(crate) fn supervisor_notice( - conn: &Connection, - notice_id: NoticeId, -) -> Result, SupervisorNoticeStoreError> { - Ok(select_supervisor_notice_record(conn, notice_id)?.map(|(notice, _)| notice)) -} - -/// Read notices that can receive another delivery attempt in stable ID order -/// -/// A notice whose loan has moved past its action keeps its delivery state but -/// is not listed, so a late copy never reaches the supervisor -pub(crate) fn pending_supervisor_notices( - conn: &Connection, -) -> Result, SupervisorNoticeStoreError> { - let mut statement = conn.prepare( - "SELECT notice.id, notice.loan_id, notice.action_id, notice.notice_json, loan.state_json - FROM resource_supervisor_notices AS notice - JOIN loans AS loan ON loan.id = notice.loan_id - WHERE json_extract(notice.notice_json, '$.delivery.type') IN ('pending', 'retry_pending') - ORDER BY notice.id ASC", - )?; - let rows = statement.query_map([], |row| { - Ok( - decode_supervisor_notice_record(row).and_then(|(notice, _)| { - let loan_state = decode_loan_state(row, 4)?; - Ok((notice, loan_state)) - }), - ) - })?; - let records = collect_decoded::<_, StoredReadError>(rows)?; - Ok(records - .into_iter() - .filter(|(notice, loan_state)| loan_state.awaits_notice(notice)) - .map(|(notice, _)| notice) - .collect()) -} - -/// Reserve one bounded attempt, or return an existing reservation for an identical retry -pub(crate) fn reserve_supervisor_notice_attempt( - conn: &mut Connection, - notice_id: NoticeId, - attempt_id: DeliveryAttemptId, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let (mut notice, old_json) = select_supervisor_notice_record(&tx, notice_id)? - .ok_or(SupervisorNoticeStoreError::NotFound)?; - // the loan can close between the pending scan and this reservation - if !select_loan_state(&tx, notice.loan_id)?.awaits_notice(¬ice) { - return Err(SupervisorNoticeStoreError::ActionNoLongerAwaited); - } - - let attempts = match notice.delivery { - SupervisorNoticeDelivery::Pending { attempts } - | SupervisorNoticeDelivery::RetryPending { attempts, .. } => attempts, - SupervisorNoticeDelivery::Sending { - attempt_id: reserved, - .. - } if reserved == attempt_id => { - tx.commit()?; - return Ok(notice); - } - SupervisorNoticeDelivery::Sending { .. } => { - return Err(SupervisorNoticeStoreError::AttemptInFlight); - } - SupervisorNoticeDelivery::Delivered { .. } => { - return Err(SupervisorNoticeStoreError::AlreadyDelivered); - } - SupervisorNoticeDelivery::Failed { .. } => { - return Err(SupervisorNoticeStoreError::AttemptBudgetExhausted); - } - }; - - if attempts >= SUPERVISOR_NOTICE_MAX_ATTEMPTS { - return Err(SupervisorNoticeStoreError::AttemptBudgetExhausted); - } - - let attempt = attempts + 1; - notice.delivery = SupervisorNoticeDelivery::Sending { - attempt_id, - attempt, - }; - update_supervisor_notice_cas(&tx, ¬ice, &old_json)?; - tx.commit()?; - Ok(notice) -} - -/// Settle only the exact attempt that is still in flight -pub(crate) fn settle_supervisor_notice_attempt( - conn: &mut Connection, - notice_id: NoticeId, - attempt_id: DeliveryAttemptId, - result: Result<(), String>, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let (mut notice, old_json) = select_supervisor_notice_record(&tx, notice_id)? - .ok_or(SupervisorNoticeStoreError::NotFound)?; - let attempt = match notice.delivery { - SupervisorNoticeDelivery::Sending { - attempt_id: active_attempt, - attempt, - } if active_attempt == attempt_id => attempt, - _ => return Err(SupervisorNoticeStoreError::StaleAttempt), - }; - - notice.delivery = match result { - Ok(()) => SupervisorNoticeDelivery::Delivered { attempts: attempt }, - Err(last_error) if attempt >= SUPERVISOR_NOTICE_MAX_ATTEMPTS => { - SupervisorNoticeDelivery::Failed { - attempts: attempt, - last_error, - } - } - Err(last_error) => SupervisorNoticeDelivery::RetryPending { - attempts: attempt, - last_error, - }, - }; - - update_supervisor_notice_cas(&tx, ¬ice, &old_json)?; - tx.commit()?; - Ok(notice) -} - -/// Recover in-flight notices after startup without treating them as delivered -pub(crate) fn recover_sending_supervisor_notices( - conn: &mut Connection, -) -> Result, SupervisorNoticeStoreError> { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let sending = { - let mut statement = tx.prepare( - "SELECT id, loan_id, action_id, notice_json - FROM resource_supervisor_notices - WHERE json_extract(notice_json, '$.delivery.type') = 'sending' - ORDER BY id ASC", - )?; - let rows = statement.query_map([], |row| Ok(decode_supervisor_notice_record(row)))?; - collect_decoded::<_, StoredReadError>(rows)? - }; - - let mut recovered = Vec::with_capacity(sending.len()); - for (mut notice, old_json) in sending { - let SupervisorNoticeDelivery::Sending { attempt, .. } = notice.delivery else { - unreachable!("query selected only sending notices") - }; - notice.delivery = if attempt >= SUPERVISOR_NOTICE_MAX_ATTEMPTS { - SupervisorNoticeDelivery::Failed { - attempts: attempt, - last_error: INTERRUPTED_DELIVERY_ERROR.into(), - } - } else { - SupervisorNoticeDelivery::RetryPending { - attempts: attempt, - last_error: INTERRUPTED_DELIVERY_ERROR.into(), - } - }; - update_supervisor_notice_cas(&tx, ¬ice, &old_json)?; - recovered.push(notice); - } - - tx.commit()?; - Ok(recovered) -} - -/// Retarget an undelivered notice inside the caller's transaction -pub(crate) fn retarget_supervisor_notice_in_transaction( - tx: &Transaction<'_>, - notice_id: NoticeId, - expected_assignment_revision: AssignmentRevision, - destination: SupervisorAddress, - new_assignment_revision: AssignmentRevision, -) -> Result { - if new_assignment_revision.get() <= expected_assignment_revision.get() { - return Err(SupervisorNoticeStoreError::AssignmentRevisionMustIncrease); - } - - let (mut notice, old_json) = select_supervisor_notice_record(tx, notice_id)? - .ok_or(SupervisorNoticeStoreError::NotFound)?; - if notice.assignment_revision != expected_assignment_revision { - return Err(SupervisorNoticeStoreError::StaleAssignmentRevision); - } - if matches!(notice.delivery, SupervisorNoticeDelivery::Delivered { .. }) { - return Err(SupervisorNoticeStoreError::AlreadyDelivered); - } - - if let SupervisorNoticeDelivery::Sending { attempt, .. } = notice.delivery { - notice.delivery = if attempt >= SUPERVISOR_NOTICE_MAX_ATTEMPTS { - SupervisorNoticeDelivery::Failed { - attempts: attempt, - last_error: RETARGETED_DELIVERY_ERROR.into(), - } - } else { - SupervisorNoticeDelivery::RetryPending { - attempts: attempt, - last_error: RETARGETED_DELIVERY_ERROR.into(), - } - }; - } - notice.destination = destination; - notice.assignment_revision = new_assignment_revision; - - update_supervisor_notice_cas(tx, ¬ice, &old_json)?; - Ok(notice) -} - -fn same_supervisor_notice_content(left: &SupervisorNotice, right: &SupervisorNotice) -> bool { - left.id == right.id - && left.loan_id == right.loan_id - && left.action_id == right.action_id - && left.state_revision == right.state_revision - && left.destination == right.destination - && left.assignment_revision == right.assignment_revision - && left.payload == right.payload -} - -pub(crate) fn select_supervisor_notice_record( - conn: &Connection, - notice_id: NoticeId, -) -> Result, StoredReadError> { - conn.query_row( - "SELECT id, loan_id, action_id, notice_json - FROM resource_supervisor_notices WHERE id = ?1", - [notice_id.as_uuid().to_string()], - |row| Ok(decode_supervisor_notice_record(row)), - ) - .optional()? - .transpose() -} - -pub(crate) fn select_supervisor_notice_record_by_action( - conn: &Connection, - action_id: ActionId, -) -> Result, StoredReadError> { - conn.query_row( - "SELECT id, loan_id, action_id, notice_json - FROM resource_supervisor_notices WHERE action_id = ?1", - [action_id.as_uuid().to_string()], - |row| Ok(decode_supervisor_notice_record(row)), - ) - .optional()? - .transpose() -} - -fn select_loan_state(conn: &Connection, loan_id: LoanId) -> Result { - conn.query_row( - "SELECT state_json FROM loans WHERE id = ?1", - [loan_id.as_uuid().to_string()], - |row| Ok(decode_loan_state(row, 0)), - )? -} - -fn decode_loan_state(row: &Row<'_>, index: usize) -> Result { - stored_json( - "loan state", - &stored_column::(row, index, "loan state")?, - ) -} - -/// Decode `id, loan_id, action_id, notice_json` columns and check that they agree -pub(crate) fn decode_supervisor_notice_record( - row: &Row<'_>, -) -> Result<(SupervisorNotice, String), StoredReadError> { - let text = |index, what| stored_column::(row, index, what); - let id: NoticeId = stored_id( - "supervisor notice identity", - &text(0, "supervisor notice identity")?, - )?; - let loan_id: LoanId = stored_id( - "supervisor notice loan identity", - &text(1, "supervisor notice loan identity")?, - )?; - let action_id: ActionId = stored_id( - "supervisor notice action identity", - &text(2, "supervisor notice action identity")?, - )?; - let notice_json = text(3, "supervisor notice")?; - let notice: SupervisorNotice = stored_json("supervisor notice", ¬ice_json)?; - if notice.id != id || notice.loan_id != loan_id || notice.action_id != action_id { - return Err(StoredReadError::corrupt( - "supervisor notice", - "identity columns do not match the typed notice", - )); - } - - Ok((notice, notice_json)) -} - -pub(crate) fn update_supervisor_notice_cas( - tx: &Transaction<'_>, - notice: &SupervisorNotice, - expected_json: &str, -) -> Result<(), SupervisorNoticeStoreError> { - let updated = tx.execute( - "UPDATE resource_supervisor_notices SET notice_json = ?1 - WHERE id = ?2 AND action_id = ?3 AND notice_json = ?4", - params![ - encode_supervisor_notice(notice)?, - notice.id.as_uuid().to_string(), - notice.action_id.as_uuid().to_string(), - expected_json, - ], - )?; - if updated != 1 { - return Err(SupervisorNoticeStoreError::StaleAttempt); - } - - Ok(()) -} - -fn encode_supervisor_notice( - notice: &SupervisorNotice, -) -> Result { - serde_json::to_string(notice).map_err(|err| { - SupervisorNoticeStoreError::Storage(rusqlite::Error::ToSqlConversionFailure(Box::new(err))) - }) -} diff --git a/src/resource/store/provenance.rs b/src/resource/store/provenance.rs deleted file mode 100644 index 674afc9..0000000 --- a/src/resource/store/provenance.rs +++ /dev/null @@ -1,302 +0,0 @@ -//! Verification of the release provenance saved on a Serving loan - -use rusqlite::{Connection, OptionalExtension}; - -use super::codec::stored_json; -use super::error::ResourceStoreError; -use super::release_completion::{ReleaseCompletionReceipt, ReleaseCompletionResult}; -use super::return_window::return_deadline_serving_matches_on; -use crate::domain::{ProcessGroupExitEvidence, ProcessStatus, TaskState}; -use crate::machine::MachineId; -use crate::resource::{ - Loan, LoanPhase, LoanState, ReleaseCheckpointPhase, ReleaseCheckpointState, Resource, - ResourceRequestState, ResourceRevision, ReturnContext, ServingReleaseProvenance, -}; -use crate::submission::{ExecutorIdentity, normalized_spec_sha256}; - -/// Whether saved authority evidence still proves the release provenance of a Serving loan -pub(super) fn serving_release_provenance_matches( - conn: &Connection, - authority: MachineId, - resource: &Resource, - loan: &Loan, - return_context: &ReturnContext, - provenance: &ServingReleaseProvenance, -) -> Result { - let (action_id, task_id) = match provenance { - ServingReleaseProvenance::IdleBoundary { proof } => { - return crate::store::idle_opening_matches_on( - conn, - authority, - resource, - loan, - return_context, - proof, - ); - } - ServingReleaseProvenance::OperatorAttestedGpuFree { .. } => { - return crate::store::operator_serving_release_matches_on( - conn, - authority, - resource, - loan, - return_context, - provenance, - ); - } - ServingReleaseProvenance::ReturnDeadlinePassed { action_id } => { - return return_deadline_serving_matches_on( - conn, - authority, - resource, - loan, - return_context, - *action_id, - ); - } - ServingReleaseProvenance::CompletedTrainerResult { - action_id, task_id, .. - } - | ServingReleaseProvenance::StoppedTrainerCheckpoint { - action_id, task_id, .. - } - | ServingReleaseProvenance::EndedTrainerLockReleased { - action_id, task_id, .. - } => (*action_id, *task_id), - }; - if resource.registered_background_task != Some(task_id) { - return Ok(false); - } - - let receipt_json: Option = conn - .query_row( - "SELECT receipt_json FROM resource_release_completions WHERE action_id = ?1", - [action_id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional()?; - let Some(receipt_json) = receipt_json else { - return Ok(false); - }; - let receipt: ReleaseCompletionReceipt = - stored_json("release completion receipt", &receipt_json)?; - if receipt.action_id != action_id - || receipt.authority_machine != authority - || receipt.resource_id != resource.id - || receipt.return_context != *return_context - { - return Ok(false); - } - let receipt_provenance = receipt.release_provenance; - - let ReleaseCompletionResult::Assigned { - loan: completed_loan, - request: completed_request, - state_revision, - } = receipt.result - else { - return Ok(false); - }; - if completed_loan.id != loan.id - || completed_loan.resource_id != resource.id - || completed_request.resource_id != resource.id - || !matches!( - completed_request.state, - ResourceRequestState::Assigned { loan_id } if loan_id == loan.id - ) - || receipt.expected_state_revision.next() != Some(state_revision) - { - return Ok(false); - } - - let receipt_has_matching_provenance = matches!( - completed_loan.state, - LoanState::Active { - phase: LoanPhase::Serving { - return_context: completed_return_context, - release_provenance: completed_provenance, - .. - } - } if completed_return_context == *return_context && completed_provenance == *provenance - ); - if !receipt_has_matching_provenance { - return Ok(false); - } - - match provenance { - // an idle opening, an operator attestation, and an expired return window have - // no release completion receipt - ServingReleaseProvenance::IdleBoundary { .. } - | ServingReleaseProvenance::OperatorAttestedGpuFree { .. } - | ServingReleaseProvenance::ReturnDeadlinePassed { .. } => Ok(false), - ServingReleaseProvenance::CompletedTrainerResult { task_id, .. } => Ok(matches!( - return_context, - ReturnContext::AlreadyCompleted { task_id: returned_task, .. } - if returned_task == task_id - )), - ServingReleaseProvenance::StoppedTrainerCheckpoint { .. } => { - stopped_serving_release_matches_saved_proof( - conn, - authority, - resource, - receipt.expected_state_revision, - provenance, - return_context, - ) - } - // only a receipt that saved this exact basis can permit activation - ServingReleaseProvenance::EndedTrainerLockReleased { - task_id, outcome, .. - } => Ok(receipt_provenance == *provenance - && matches!( - return_context, - ReturnContext::EndedWithoutResult { - task_id: returned_task, - outcome: returned_outcome, - } if returned_task == task_id && returned_outcome == outcome - )), - } -} - -fn stopped_serving_release_matches_saved_proof( - conn: &Connection, - authority: MachineId, - resource: &Resource, - expected_state_revision: ResourceRevision, - provenance: &ServingReleaseProvenance, - return_context: &ReturnContext, -) -> Result { - let ServingReleaseProvenance::StoppedTrainerCheckpoint { - action_id, - task_id, - generation_id, - record_sha256, - inventory_sha256, - } = provenance - else { - return Ok(false); - }; - let saved: Option<(String, String)> = conn - .query_row( - "SELECT resource_id, state_json FROM resource_release_checkpoint_states - WHERE action_id = ?1", - [action_id.as_uuid().to_string()], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .optional()?; - let Some((saved_resource_id, state_json)) = saved else { - return Ok(false); - }; - if saved_resource_id != resource.id.as_uuid().to_string() { - return Ok(false); - } - let Ok(state) = serde_json::from_str::(&state_json) else { - return Ok(false); - }; - if state.validate().is_err() - || state.action.resource_id != resource.id - || state.action.action_id != *action_id - || state.action.state_revision != expected_state_revision - { - return Ok(false); - } - let ReleaseCheckpointPhase::CancellationCommitted { - baseline, - decision, - cancellation, - } = state.phase - else { - return Ok(false); - }; - let checkpoint = &decision.selected_checkpoint; - if state.action.observed_background_task != *task_id - || cancellation.task_id != *task_id - || decision.binding != baseline.binding - || baseline.binding.action.state_revision != state.action.state_revision - || baseline.binding.association.resource_id != resource.id - || baseline.binding.association.authority_machine != authority - || baseline.binding.association.task_id != *task_id - || checkpoint.generation_id != *generation_id - || checkpoint.record_sha256 != *record_sha256 - || checkpoint.inventory_sha256 != *inventory_sha256 - || !matches!( - return_context, - ReturnContext::Stopped { - task_id: returned_task, - checkpoint_ref, - recovery_ref, - } if *returned_task == *task_id - && checkpoint_ref == &format!( - "{}#sha256={}", - checkpoint.path.display(), - checkpoint.record_sha256 - ) - && recovery_ref == &checkpoint.generation_id - ) - { - return Ok(false); - } - - let association: Option<(String, String, String)> = conn - .query_row( - "SELECT resource_id, authority_machine, association_json - FROM trainer_attempt_associations WHERE task_id = ?1", - [task_id.to_string()], - |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?)), - ) - .optional()?; - let Some((association_resource, association_authority, association_json)) = association else { - return Ok(false); - }; - let Ok(association) = - serde_json::from_str::(&association_json) - else { - return Ok(false); - }; - if association_resource != resource.id.as_uuid().to_string() - || association_authority != authority.as_uuid().to_string() - || association != baseline.binding.association - { - return Ok(false); - } - - let Some(task) = - crate::store::task_by_id_on(conn, *task_id).map_err(ResourceStoreError::TaskRow)? - else { - return Ok(false); - }; - if !matches!( - task.state, - TaskState::Finished { - reason: crate::domain::ExitReason::Cancelled - } - ) || task.cancel_requested_at != Some(cancellation.cancel_requested_at) - || task.process_group_exit_evidence() != ProcessGroupExitEvidence::ConfirmedExited - { - return Ok(false); - } - - let identity_json: Option = conn - .query_row( - "SELECT identity_json FROM executor_identities WHERE task_id = ?1", - [task_id.to_string()], - |row| row.get(0), - ) - .optional()?; - let Some(identity_json) = identity_json else { - return Ok(false); - }; - let Ok(ExecutorIdentity::Accepted(identity)) = serde_json::from_str(&identity_json) else { - return Ok(false); - }; - let Some(spec) = identity.current_spec() else { - return Ok(false); - }; - Ok(identity.task == *task_id - && identity.execution_machine == authority - && identity.state == ProcessStatus::Cancelled - && identity.has_valid_spec_owners() - && normalized_spec_sha256(spec).map_err(|error| { - ResourceStoreError::TaskPreparation(crate::error::AppError::from(error)) - })? == association.normalized_spec_sha256) -} diff --git a/src/resource/store/queue.rs b/src/resource/store/queue.rs deleted file mode 100644 index 0e0b9c5..0000000 --- a/src/resource/store/queue.rs +++ /dev/null @@ -1,418 +0,0 @@ -//! Authority queue acceptance, serving order, and reads of resource requests - -use rusqlite::{Connection, OptionalExtension, Row, TransactionBehavior, params}; - -use super::assigned_task::resource_task_ownership_risk; -use super::codec::{ - collect_decoded, encode_json, stored_column, stored_count, stored_id, stored_json, stored_uuid, -}; -use super::error::{ConflictReason, ResourceStoreError}; -use super::rows::{ - REQUEST_COLUMNS, check_resource_authority, decode_request, select_request_by_id, -}; -use crate::domain::TaskId; -use crate::machine::MachineId; -use crate::resource::{ - AcceptanceSequence, CommandSpec, ResourceId, ResourceRequest, ResourceRequestState, -}; -use crate::spec::NormalizedSpec; -use crate::submission::{ExecutorIdentity, RequestId}; - -const PREVENTION_COLUMNS: &str = "request_id, task_id, resource_id, origin_machine"; - -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(super) struct RequestIdentity { - pub(super) request_id: RequestId, - pub(super) task_id: TaskId, - pub(super) resource_id: ResourceId, - pub(super) origin_machine: MachineId, -} - -/// Accept one command request on the resource's fixed authority -/// -/// The origin route is stored by the caller before it sends this queue request -pub(crate) fn accept_request_for_authority( - conn: &mut Connection, - authority_machine: MachineId, - request_id: RequestId, - task_id: TaskId, - resource_id: ResourceId, - origin_machine: MachineId, - normalized_spec: NormalizedSpec, -) -> Result { - let identity = RequestIdentity { - request_id, - task_id, - resource_id, - origin_machine, - }; - { - // an exact retry is answered from the saved request before the command - // checks, so a file that changed after acceptance cannot reject it - let tx = conn.transaction_with_behavior(TransactionBehavior::Deferred)?; - if let Some(saved) = replay_request_on(&tx, authority_machine, identity, &normalized_spec)? - { - tx.commit()?; - return Ok(saved); - } - } - // the entry-point check reads the file system outside the IMMEDIATE write transaction - check_queued_command_ownership(&normalized_spec)?; - - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - if let Some(saved) = replay_request_on(&tx, authority_machine, identity, &normalized_spec)? { - tx.commit()?; - return Ok(saved); - } - - if task_id_exists(&tx, task_id)? { - return Err(ResourceStoreError::Conflict( - ConflictReason::TaskIdentityInUse, - )); - } - - if prevention_exists(&tx, identity)? { - return Err(ResourceStoreError::Prevented); - } - if executor_identity_exists(&tx, task_id)? { - return Err(ResourceStoreError::Conflict( - ConflictReason::TaskIdentityInUse, - )); - } - - let command_spec = CommandSpec::try_from(normalized_spec)?; - check_resource_authority(&tx, resource_id, authority_machine)?; - let spec_json = encode_json(command_spec.as_normalized())?; - let state_json = encode_json(&ResourceRequestState::Queued)?; - - tx.execute( - "INSERT INTO resource_requests ( - request_id, task_id, resource_id, origin_machine, spec_json, state_json, queue_rank - ) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7)", - params![ - request_id.0.to_string(), - task_id.to_string(), - resource_id.as_uuid().to_string(), - origin_machine.as_uuid().to_string(), - spec_json, - state_json, - next_queue_rank(&tx, resource_id)?, - ], - )?; - let raw_sequence = tx.last_insert_rowid(); - let acceptance_sequence = - AcceptanceSequence::new(stored_count("request acceptance sequence", raw_sequence)?); - let saved = ResourceRequest::new( - request_id, - task_id, - resource_id, - acceptance_sequence, - origin_machine, - command_spec.as_normalized().clone(), - )?; - tx.commit()?; - Ok(saved) -} - -/// Return the saved request for an exact retry, or a conflict for changed content -fn replay_request_on( - conn: &Connection, - authority_machine: MachineId, - identity: RequestIdentity, - normalized_spec: &NormalizedSpec, -) -> Result, ResourceStoreError> { - let Some(saved) = select_request_by_id(conn, identity.request_id)? else { - return Ok(None); - }; - if saved.task_id != identity.task_id - || saved.resource_id != identity.resource_id - || saved.origin_machine != identity.origin_machine - { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestIdentityMismatch, - )); - } - let command_spec = CommandSpec::try_from(normalized_spec.clone())?; - if saved.spec().as_normalized() != command_spec.as_normalized() { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestSpecMismatch, - )); - } - check_resource_authority(conn, identity.resource_id, authority_machine)?; - - Ok(Some(saved)) -} - -/// Refuse new queued work outside the ownership contract -/// -/// A path-qualified entry point is inspected now. A bare program name resolves -/// from the executor `PATH`, so its task binding inspects it before any spawn. -/// The `cwd` and a container's mount sources must exist on this authority -fn check_queued_command_ownership(spec: &NormalizedSpec) -> Result<(), ResourceStoreError> { - let command_spec = CommandSpec::try_from(spec.clone())?; - if let Some(risk) = resource_task_ownership_risk(&command_spec) { - return Err(ResourceStoreError::UnsupportedCommandOwnership { risk }); - } - crate::spec::check_spec_host(spec).map_err(ResourceStoreError::HostInputRejected)?; - crate::resource::foreground::inspect_path_qualified_entry_point(spec) - .map_err(|risk| ResourceStoreError::UnsupportedCommandOwnership { risk }) -} - -/// Read all requests for one resource in serving order -pub(crate) fn requests_for_resource_for_authority( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, -) -> Result, ResourceStoreError> { - check_resource_authority(conn, resource_id, authority_machine)?; - let sql = format!( - "SELECT {REQUEST_COLUMNS} FROM resource_requests \ - WHERE resource_id = ?1 ORDER BY queue_rank, acceptance_sequence" - ); - let mut statement = conn.prepare(&sql)?; - let rows = statement.query_map([resource_id.as_uuid().to_string()], |row| { - Ok(decode_request(row)) - })?; - collect_decoded(rows) -} - -/// Load assigned requests for one authority so task ownership can be checked durably -pub(crate) fn assigned_resource_requests_for_authority( - conn: &Connection, - authority_machine: MachineId, -) -> Result, ResourceStoreError> { - let mut statement = conn.prepare( - "SELECT rr.acceptance_sequence, rr.request_id, rr.task_id, rr.resource_id, - rr.origin_machine, rr.spec_json, rr.state_json - FROM resource_requests AS rr - JOIN resources AS r ON r.id = rr.resource_id - WHERE r.authority_machine = ?1 - AND json_extract(rr.state_json, '$.type') = 'assigned' - ORDER BY rr.resource_id, rr.queue_rank, rr.acceptance_sequence", - )?; - let rows = statement.query_map([authority_machine.to_string()], |row| { - Ok(decode_request(row)) - })?; - collect_decoded(rows) -} - -/// Read the next queued request for one resource in serving order -pub(crate) fn next_queued_request_for_authority( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, -) -> Result, ResourceStoreError> { - check_resource_authority(conn, resource_id, authority_machine)?; - let sql = format!( - "SELECT {REQUEST_COLUMNS} FROM resource_requests \ - WHERE resource_id = ?1 AND json_extract(state_json, '$.type') = 'queued' \ - ORDER BY queue_rank, acceptance_sequence LIMIT 1" - ); - conn.query_row(&sql, [resource_id.as_uuid().to_string()], |row| { - Ok(decode_request(row)) - }) - .optional()? - .transpose() -} - -/// Whether a queued or assigned request sits before `request_id` in serving order -pub(super) fn earlier_active_request_exists( - conn: &Connection, - resource_id: ResourceId, - request_id: RequestId, -) -> Result { - let exists: bool = conn.query_row( - "SELECT EXISTS( - SELECT 1 FROM resource_requests AS earlier - JOIN resource_requests AS selected - ON selected.request_id = ?2 - AND selected.resource_id = earlier.resource_id - WHERE earlier.resource_id = ?1 - AND (earlier.queue_rank, earlier.acceptance_sequence) - < (selected.queue_rank, selected.acceptance_sequence) - AND json_extract(earlier.state_json, '$.type') IN ('queued', 'assigned') - )", - params![resource_id.as_uuid().to_string(), request_id.0.to_string()], - |row| row.get(0), - )?; - Ok(exists) -} - -/// Rewrite queued ranks as a permutation of their current values -/// -/// The caller supplies a validated serving order from the same write transaction -pub(crate) fn rewrite_queued_request_ranks( - conn: &Connection, - resource_id: ResourceId, - request_order: &[RequestId], -) -> Result<(), ResourceStoreError> { - let mut statement = conn.prepare( - "SELECT request_id, queue_rank FROM resource_requests - WHERE resource_id = ?1 AND json_extract(state_json, '$.type') = 'queued' - ORDER BY queue_rank, acceptance_sequence", - )?; - let rows = statement.query_map([resource_id.as_uuid().to_string()], |row| { - Ok((row.get::<_, String>(0)?, row.get::<_, i64>(1)?)) - })?; - let current = rows.collect::, _>>()?; - // rusqlite forbids another statement while this query statement is live - drop(statement); - - let mut current_ids = current - .iter() - .map(|(request_id, _)| request_id.clone()) - .collect::>(); - let mut requested_ids = request_order - .iter() - .map(|request_id| request_id.0.to_string()) - .collect::>(); - current_ids.sort_unstable(); - requested_ids.sort_unstable(); - if current_ids != requested_ids { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestStateChanged, - )); - } - - let mut ranks = current - .into_iter() - .map(|(_, queue_rank)| queue_rank) - .collect::>(); - ranks.sort_unstable(); - for (request_id, queue_rank) in request_order.iter().zip(ranks) { - let changed = conn.execute( - "UPDATE resource_requests SET queue_rank = ?1 - WHERE request_id = ?2 AND resource_id = ?3 - AND json_extract(state_json, '$.type') = 'queued'", - params![ - queue_rank, - request_id.0.to_string(), - resource_id.as_uuid().to_string(), - ], - )?; - if changed != 1 { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestStateChanged, - )); - } - } - - Ok(()) -} - -fn next_queue_rank(conn: &Connection, resource_id: ResourceId) -> Result { - let maximum: i64 = conn.query_row( - "SELECT COALESCE(MAX(queue_rank), 0) FROM resource_requests WHERE resource_id = ?1", - [resource_id.as_uuid().to_string()], - |row| row.get(0), - )?; - maximum.checked_add(1).ok_or(ResourceStoreError::Conflict( - ConflictReason::QueueRankExhausted, - )) -} - -pub(super) fn prevention_exists( - conn: &Connection, - identity: RequestIdentity, -) -> Result { - let sql = format!( - "SELECT {PREVENTION_COLUMNS} FROM resource_request_preventions \ - WHERE request_id = ?1 OR task_id = ?2" - ); - let mut statement = conn.prepare(&sql)?; - let rows = statement.query_map( - params![ - identity.request_id.0.to_string(), - identity.task_id.to_string() - ], - |row| Ok(decode_prevention(row)), - )?; - let saved = collect_decoded(rows)?; - if saved.iter().any(|saved| *saved != identity) { - return Err(ResourceStoreError::Conflict( - ConflictReason::PreventionIdentityMismatch, - )); - } - - Ok(!saved.is_empty()) -} - -pub(super) fn request_matches_identity( - request: &ResourceRequest, - identity: RequestIdentity, -) -> bool { - request.request_id == identity.request_id - && request.task_id == identity.task_id - && request.resource_id == identity.resource_id - && request.origin_machine == identity.origin_machine -} - -pub(super) fn task_id_exists(conn: &Connection, task_id: TaskId) -> Result { - conn.query_row( - "SELECT EXISTS(SELECT 1 FROM resource_requests WHERE task_id = ?1)", - [task_id.to_string()], - |row| row.get(0), - ) -} - -pub(super) fn executor_identity_exists( - conn: &Connection, - task_id: TaskId, -) -> Result { - conn.query_row( - "SELECT EXISTS(SELECT 1 FROM executor_identities WHERE task_id = ?1)", - [task_id.to_string()], - |row| row.get(0), - ) -} - -pub(super) fn select_executor_identity( - conn: &Connection, - task_id: TaskId, -) -> Result, ResourceStoreError> { - let data: Option = conn - .query_row( - "SELECT identity_json FROM executor_identities WHERE task_id = ?1", - [task_id.to_string()], - |row| row.get(0), - ) - .optional()?; - let identity = data - .as_deref() - .map(|data| stored_json("executor identity", data)) - .transpose()?; - Ok(identity) -} - -pub(super) fn local_task_exists( - conn: &Connection, - task_id: TaskId, -) -> Result { - conn.query_row( - "SELECT EXISTS(SELECT 1 FROM tasks WHERE id = ?1)", - [task_id.to_string()], - |row| row.get(0), - ) -} - -fn decode_prevention(row: &Row<'_>) -> Result { - let text = |index, what| stored_column::(row, index, what); - Ok(RequestIdentity { - request_id: RequestId(stored_uuid( - "prevention request identity", - &text(0, "prevention request identity")?, - )?), - task_id: TaskId(stored_uuid( - "prevention task identity", - &text(1, "prevention task identity")?, - )?), - resource_id: stored_id( - "prevention resource identity", - &text(2, "prevention resource identity")?, - )?, - origin_machine: MachineId::from_uuid(stored_uuid( - "prevention origin machine", - &text(3, "prevention origin machine")?, - )?), - }) -} diff --git a/src/resource/store/release_completion.rs b/src/resource/store/release_completion.rs deleted file mode 100644 index 61cc5d6..0000000 --- a/src/resource/store/release_completion.rs +++ /dev/null @@ -1,729 +0,0 @@ -//! Proof-gated completion of a release action - -use crate::domain::ExitReason; -use rusqlite::{Connection, OptionalExtension, Transaction, TransactionBehavior, params}; -use serde::Serialize; - -use super::checkpoint::{ReleaseCheckpointError, release_checkpoint_state_for_action}; -use super::codec::{sqlite_integer, stored_json}; -use super::error::{ResourceStoreError, TrainerAttemptAssociationStoreError}; -use super::notice::{ - SupervisorNoticeStoreError, insert_supervisor_notice_in_transaction, - select_supervisor_notice_record_by_action, -}; -use super::queue::next_queued_request_for_authority; -use super::revision::swap_resource_revision; -use super::rows::{check_authority, decode_loan, select_resource}; -use crate::domain::{ProcessGroupExitEvidence, TaskId, TaskState}; -use crate::error::AppError; -use crate::machine::MachineId; -use crate::resource::command_shape::DirectSegmentCommandShapeError; -use crate::resource::ownership_lock::OwnershipLockProbeError; -use crate::resource::trainer_publication::WatcherError; -use crate::resource::{ - ActionId, Loan, LoanId, LoanPhase, LoanState, NoticeId, ReleaseCheckpointPhase, Resource, - ResourceId, ResourceRequest, ResourceRequestState, ResourceRevision, ReturnContext, - ServingReleaseProvenance, SupervisorNotice, SupervisorNoticeDelivery, SupervisorNoticePayload, -}; -use crate::store::{IdentityError, VerifiedReleaseProof}; -use crate::submission::RequestId; - -/// Stable result committed when a release action completes -#[derive(Debug, Clone, Serialize, serde::Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub(crate) enum ReleaseCompletionResult { - /// The next queued request now owns the resource through this loan - Assigned { - /// Loan updated from AwaitingRelease to Serving - loan: Loan, - /// Exact request selected at completion time - request: ResourceRequest, - /// Resource revision committed with the assignment - state_revision: ResourceRevision, - }, - /// The queue was empty and the supervisor now owns the return decision - ReturnRequired { - /// Loan updated from AwaitingRelease to AwaitingReturn - loan: Loan, - /// Durable notice for the exact current supervisor assignment - notice: SupervisorNotice, - }, -} - -/// Completion evidence for one saved release action -#[derive(Debug, Clone, Serialize, serde::Deserialize)] -#[serde(deny_unknown_fields)] -pub(super) struct ReleaseCompletionReceipt { - pub(super) action_id: ActionId, - pub(super) authority_machine: MachineId, - pub(super) resource_id: ResourceId, - pub(super) expected_state_revision: ResourceRevision, - pub(super) return_context: ReturnContext, - /// Proof basis accepted by this completion, for either result - pub(super) release_provenance: ServingReleaseProvenance, - pub(super) result: ReleaseCompletionResult, -} - -/// A release action could not be completed for the supplied evidence -#[derive(Debug, thiserror::Error)] -pub(crate) enum CompleteReleaseError { - /// Resource authority or stored resource data failed validation - #[error(transparent)] - Resource(#[from] ResourceStoreError), - /// A supervisor notice could not be read or inserted - #[error(transparent)] - Notice(#[from] SupervisorNoticeStoreError), - /// A trainer association could not be read or does not match the release task - #[error(transparent)] - TrainerAssociation(#[from] TrainerAttemptAssociationStoreError), - /// A persisted task identity could not be read or did not match its authority - #[error(transparent)] - Identity(#[from] IdentityError), - /// The accepted task row could not be read - #[error(transparent)] - TaskStorage(#[from] AppError), - /// The accepted command does not match its saved trainer association - #[error(transparent)] - CommandShape(#[from] DirectSegmentCommandShapeError), - /// The result watcher could not verify the completed publication - #[error(transparent)] - Watcher(#[from] WatcherError), - /// The saved trainer ownership lock could not be verified - #[error(transparent)] - OwnershipLock(#[from] OwnershipLockProbeError), - /// The action's durable checkpoint or cancellation record could not be verified - #[error(transparent)] - Checkpoint(#[from] ReleaseCheckpointError), - /// The action has neither an active release phase nor a completion receipt - #[error("release action {action_id:?} was not found")] - ActionNotFound { - /// Stable release action identity - action_id: ActionId, - }, - /// The action is no longer in the expected AwaitingRelease phase - #[error("release action {action_id:?} is not awaiting release on loan {loan_id:?}")] - NotAwaitingRelease { - /// Loan that should own the release action - loan_id: LoanId, - /// Stable release action identity - action_id: ActionId, - }, - /// The action's durable release notice is missing or inconsistent - #[error("release action {action_id:?} has an invalid durable notice")] - InvalidReleaseNotice { - /// Stable release action identity - action_id: ActionId, - }, - /// The caller's expected revision does not match current resource state - #[error("stale resource revision: expected {expected:?}, found {actual:?}")] - StaleRevision { - /// Resource revision associated with the release notice - expected: ResourceRevision, - /// Current resource revision in SQLite - actual: ResourceRevision, - }, - /// The observed task has no ordinary task row on this authority - #[error("observed background task {task_id} is missing locally")] - BackgroundTaskMissing { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The ordinary task row is not terminal yet - #[error("observed background task {task_id} is not terminal (state {state})")] - BackgroundTaskNotTerminal { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - /// Process state currently saved in the task table - state: String, - }, - /// A lost task row cannot establish that the resource is free - #[error("observed background task {task_id} is lost and needs attention")] - BackgroundTaskLost { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The release action has no durable association for its exact trainer task - #[error("release task {task_id} has no trainer-attempt association")] - TrainerAssociationMissing { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The saved trainer association no longer matches its task or resource - #[error("trainer-attempt association does not match release task {task_id}")] - TrainerAssociationMismatch { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The accepted identity row is missing for the exact trainer task - #[error("accepted trainer identity for task {task_id} is missing")] - TrainerIdentityMissing { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The accepted trainer identity no longer matches its saved proof - #[error("accepted trainer identity for task {task_id} changed")] - TrainerIdentityChanged { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The task was not cancelled by this action's saved stop decision - #[error("trainer task {task_id} has no committed stop decision for this release action")] - StoppedProofUnavailable { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The exact cancel marker no longer matches the release action record - #[error("trainer task {task_id} cancellation marker differs from the saved release action")] - TrainerCancellationMarkerChanged { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The selected checkpoint publication changed during stopped release verification - #[error("trainer task {task_id} selected checkpoint changed during release verification")] - StoppedCheckpointChanged { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The exact task exited successfully but did not confirm its child process group exited - #[error("trainer task {task_id} has no confirmed worker-exit evidence")] - WorkerExitUnconfirmed { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The verified result publication differs from the result held by the proof - #[error("trainer task {task_id} completed result changed during release verification")] - CompletedResultChanged { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The verified result request differs from the request saved at registration - #[error("trainer task {task_id} result request differs from its registered attempt")] - CompletedResultRequestMismatch { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The exact saved ownership lock is still held by a process - #[error("trainer task {task_id} ownership lock is still held")] - OwnershipLockStillHeld { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The persisted task state or worker-exit evidence changed during proof creation - #[error("trainer task {task_id} state changed during release verification")] - TaskStateChanged { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The persisted task command binding changed during release verification - #[error("trainer task {task_id} command binding changed during release verification")] - TaskCommandChanged { - /// Exact task recorded in the AwaitingRelease phase - task_id: TaskId, - }, - /// The resource state revision cannot be incremented - #[error("resource revision {revision:?} cannot be incremented")] - RevisionExhausted { - /// Current resource revision - revision: ResourceRevision, - }, - /// A durable receipt already exists for different input identity or evidence - #[error("release action {action_id:?} was retried with conflicting input")] - ConflictingRetry { - /// Stable release action identity - action_id: ActionId, - }, - /// A concurrent or inconsistent queue change prevented assignment - #[error("queued request {request_id:?} changed during release completion")] - RequestChanged { - /// Request selected at the head of the serving order - request_id: RequestId, - }, - /// A concurrent or inconsistent loan change prevented completion - #[error("loan {loan_id:?} changed during release completion")] - LoanChanged { - /// Loan that owns the release action - loan_id: LoanId, - }, - /// SQLite or serialized stored data failed - #[error("release completion storage error: {0}")] - Storage(#[from] rusqlite::Error), -} - -/// Return an exact committed release result without requiring the original artifacts -pub(crate) fn release_completion_for_retry( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, - action_id: ActionId, - expected_state_revision: ResourceRevision, -) -> Result, CompleteReleaseError> { - let Some(receipt) = select_release_completion_receipt(conn, action_id)? else { - return Ok(None); - }; - if receipt.action_id != action_id - || receipt.authority_machine != authority_machine - || receipt.resource_id != resource_id - || receipt.expected_state_revision != expected_state_revision - { - return Err(CompleteReleaseError::ConflictingRetry { action_id }); - } - - Ok(Some(receipt.result)) -} - -/// Whether a release notice names an assignment its action may still complete under -/// -/// A replacement retargets only undelivered notices. A delivered notice keeps the -/// older assignment whose supervisor launched the watcher, so it stays valid -fn release_notice_assignment_is_valid(notice: &SupervisorNotice, resource: &Resource) -> bool { - let current = notice.destination == resource.supervisor - && notice.assignment_revision == resource.assignment_revision; - let delivered_before_replacement = - matches!(notice.delivery, SupervisorNoticeDelivery::Delivered { .. }) - && notice.assignment_revision.get() < resource.assignment_revision.get(); - current || delivered_before_replacement -} - -/// Complete a saved release action with an authority-built proof and atomically assign work -pub(crate) fn complete_release_for_authority( - conn: &mut Connection, - proof: VerifiedReleaseProof, -) -> Result { - let authority_machine = proof.authority_machine(); - let resource_id = proof.resource_id(); - let action_id = proof.action_id(); - let expected_state_revision = proof.expected_state_revision(); - let observed_task_id = proof.task_id(); - let return_context = proof.return_context(); - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - - if let Some(receipt) = select_release_completion_receipt(&tx, action_id)? { - if receipt.action_id != action_id - || receipt.authority_machine != authority_machine - || receipt.resource_id != resource_id - || receipt.expected_state_revision != expected_state_revision - || receipt.return_context != return_context - { - return Err(CompleteReleaseError::ConflictingRetry { action_id }); - } - - tx.commit()?; - return Ok(receipt.result); - } - - let resource = - select_resource(&tx, resource_id)?.ok_or(ResourceStoreError::ResourceNotFound)?; - check_authority(resource.authority_machine(), authority_machine)?; - - let (notice, _) = select_supervisor_notice_record_by_action(&tx, action_id) - .map_err(ResourceStoreError::from)? - .ok_or(CompleteReleaseError::ActionNotFound { action_id })?; - let loan = tx - .query_row( - "SELECT id, resource_id, state_json FROM loans WHERE id = ?1", - [notice.loan_id.as_uuid().to_string()], - |row| Ok(decode_loan(row)), - ) - .optional()? - .transpose()? - .ok_or(CompleteReleaseError::InvalidReleaseNotice { action_id })?; - - let LoanState::Active { - phase: - LoanPhase::AwaitingRelease { - action_id: saved_action_id, - observed_background_task, - .. - }, - } = &loan.state - else { - return Err(CompleteReleaseError::NotAwaitingRelease { - loan_id: loan.id, - action_id, - }); - }; - if *saved_action_id != action_id { - return Err(CompleteReleaseError::NotAwaitingRelease { - loan_id: loan.id, - action_id, - }); - } - if *observed_background_task != observed_task_id - || resource.registered_background_task != Some(observed_task_id) - { - return Err(CompleteReleaseError::TrainerAssociationMismatch { - task_id: observed_task_id, - }); - } - - if loan.resource_id != resource_id - || notice.loan_id != loan.id - || notice.action_id != action_id - || notice.payload - != (SupervisorNoticePayload::ReleaseRequired { - task_id: observed_task_id, - }) - || !release_notice_assignment_is_valid(¬ice, &resource) - { - return Err(CompleteReleaseError::InvalidReleaseNotice { action_id }); - } - - if notice.state_revision != expected_state_revision { - return Err(CompleteReleaseError::StaleRevision { - expected: expected_state_revision, - actual: notice.state_revision, - }); - } - if resource.state_revision != expected_state_revision { - return Err(CompleteReleaseError::StaleRevision { - expected: expected_state_revision, - actual: resource.state_revision, - }); - } - - require_release_proof_matches_transaction(&tx, &proof)?; - let release_provenance = proof.serving_release_provenance(); - - let next_revision = - expected_state_revision - .next() - .ok_or(CompleteReleaseError::RevisionExhausted { - revision: expected_state_revision, - })?; - - let result = if let Some(mut request) = - next_queued_request_for_authority(&tx, authority_machine, resource_id)? - { - request.state = ResourceRequestState::Assigned { loan_id: loan.id }; - let request_state_json = encode_completion_json(&request.state)?; - let changed = tx.execute( - "UPDATE resource_requests SET state_json = ?1 - WHERE request_id = ?2 AND resource_id = ?3 AND acceptance_sequence = ?4 - AND json_extract(state_json, '$.type') = 'queued'", - params![ - request_state_json, - request.request_id.0.to_string(), - resource_id.as_uuid().to_string(), - sqlite_integer(request.acceptance_sequence.get())?, - ], - )?; - if changed != 1 { - return Err(CompleteReleaseError::RequestChanged { - request_id: request.request_id, - }); - } - - let updated_loan = Loan { - id: loan.id, - resource_id, - state: LoanState::Active { - phase: LoanPhase::Serving { - return_context: return_context.clone(), - current_request_id: request.request_id, - release_provenance, - }, - }, - }; - update_release_loan(&tx, &updated_loan, action_id)?; - update_resource_revision( - &tx, - authority_machine, - resource_id, - expected_state_revision, - next_revision, - )?; - - ReleaseCompletionResult::Assigned { - loan: updated_loan, - request, - state_revision: next_revision, - } - } else { - let return_action_id = ActionId::new(); - let updated_loan = Loan { - id: loan.id, - resource_id, - state: LoanState::Active { - phase: LoanPhase::AwaitingReturn { - action_id: return_action_id, - return_context: return_context.clone(), - }, - }, - }; - update_release_loan(&tx, &updated_loan, action_id)?; - update_resource_revision( - &tx, - authority_machine, - resource_id, - expected_state_revision, - next_revision, - )?; - - let notice = SupervisorNotice { - id: NoticeId::new(), - loan_id: loan.id, - action_id: return_action_id, - state_revision: next_revision, - destination: resource.supervisor, - assignment_revision: resource.assignment_revision, - payload: SupervisorNoticePayload::ReturnRequired { - return_context: return_context.clone(), - }, - delivery: SupervisorNoticeDelivery::Pending { attempts: 0 }, - }; - let notice = insert_supervisor_notice_in_transaction(&tx, ¬ice)?; - - ReleaseCompletionResult::ReturnRequired { - loan: updated_loan, - notice, - } - }; - - let receipt = ReleaseCompletionReceipt { - action_id, - authority_machine, - resource_id, - expected_state_revision, - return_context, - release_provenance: proof.serving_release_provenance(), - result: result.clone(), - }; - let receipt_json = encode_completion_json(&receipt)?; - tx.execute( - "INSERT INTO resource_release_completions (action_id, receipt_json) - VALUES (?1, ?2)", - params![action_id.as_uuid().to_string(), receipt_json], - )?; - - proof.verify_external_evidence()?; - tx.commit()?; - - Ok(result) -} - -fn require_release_proof_matches_transaction( - conn: &Connection, - proof: &VerifiedReleaseProof, -) -> Result<(), CompleteReleaseError> { - let task_id = proof.task_id(); - let current_task = crate::store::task_by_id_on(conn, task_id)? - .ok_or(CompleteReleaseError::BackgroundTaskMissing { task_id })?; - let saved_task = proof.task_row(); - if current_task.id != saved_task.id - || current_task.name != saved_task.name - || current_task.thread != saved_task.thread - || current_task.workload != saved_task.workload - || current_task.cwd != saved_task.cwd - || current_task.timeout != saved_task.timeout - || current_task.env != saved_task.env - || current_task.binary != saved_task.binary - { - return Err(CompleteReleaseError::TaskCommandChanged { task_id }); - } - if current_task.state != saved_task.state { - return Err(CompleteReleaseError::TaskStateChanged { task_id }); - } - match proof.stopped_decision_and_cancellation() { - Some((expected_decision, expected_cancellation)) => { - if !matches!( - current_task.state, - TaskState::Finished { - reason: ExitReason::Cancelled - } - ) { - return Err(CompleteReleaseError::TaskStateChanged { task_id }); - } - if current_task.cancel_requested_at != Some(expected_cancellation.cancel_requested_at) { - return Err(CompleteReleaseError::TrainerCancellationMarkerChanged { task_id }); - } - - let Some((checkpoint_state, _)) = - release_checkpoint_state_for_action(conn, proof.resource_id(), proof.action_id())? - else { - return Err(CompleteReleaseError::StoppedProofUnavailable { task_id }); - }; - let ReleaseCheckpointPhase::CancellationCommitted { - decision, - cancellation, - .. - } = checkpoint_state.phase - else { - return Err(CompleteReleaseError::StoppedProofUnavailable { task_id }); - }; - if checkpoint_state.action.resource_id != proof.resource_id() - || checkpoint_state.action.action_id != proof.action_id() - || checkpoint_state.action.state_revision != proof.expected_state_revision() - || checkpoint_state.action.observed_background_task != task_id - || *decision != *expected_decision - || cancellation != *expected_cancellation - { - return Err(CompleteReleaseError::StoppedProofUnavailable { task_id }); - } - } - None => { - let expected_reason = proof - .ended_outcome() - .cloned() - .unwrap_or(ExitReason::Exit { code: 0 }); - if !matches!( - ¤t_task.state, - TaskState::Finished { reason } if *reason == expected_reason - ) { - return Err(CompleteReleaseError::TaskStateChanged { task_id }); - } - // an ended cancellation is generic only while this action committed no stop - if proof.ended_outcome() == Some(&ExitReason::Cancelled) - && let Some((checkpoint_state, _)) = release_checkpoint_state_for_action( - conn, - proof.resource_id(), - proof.action_id(), - )? - && matches!( - checkpoint_state.phase, - ReleaseCheckpointPhase::CancellationCommitted { .. } - ) - { - return Err(CompleteReleaseError::TaskStateChanged { task_id }); - } - } - } - if current_task.process_group_exit_evidence() != ProcessGroupExitEvidence::ConfirmedExited { - return Err(CompleteReleaseError::WorkerExitUnconfirmed { task_id }); - } - - let association: Option<(String, String, String)> = conn - .query_row( - "SELECT resource_id, authority_machine, association_json - FROM trainer_attempt_associations WHERE task_id = ?1", - [task_id.to_string()], - |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?)), - ) - .optional()?; - let Some((resource_id, authority_machine, association_json)) = association else { - return Err(CompleteReleaseError::TrainerAssociationMissing { task_id }); - }; - if resource_id != proof.resource_id().as_uuid().to_string() - || authority_machine != proof.authority_machine().as_uuid().to_string() - || association_json != proof.association_json() - { - return Err(CompleteReleaseError::TrainerAssociationMismatch { task_id }); - } - - let identity_json: Option = conn - .query_row( - "SELECT identity_json FROM executor_identities WHERE task_id = ?1", - [task_id.to_string()], - |row| row.get(0), - ) - .optional()?; - let Some(identity_json) = identity_json else { - return Err(CompleteReleaseError::TrainerIdentityMissing { task_id }); - }; - if identity_json != proof.identity_json() { - return Err(CompleteReleaseError::TrainerIdentityChanged { task_id }); - } - - Ok(()) -} - -fn update_release_loan( - tx: &Transaction<'_>, - loan: &Loan, - action_id: ActionId, -) -> Result<(), CompleteReleaseError> { - let state_json = encode_completion_json(&loan.state)?; - let changed = tx.execute( - "UPDATE loans SET state_json = ?1 - WHERE id = ?2 AND resource_id = ?3 - AND json_extract(state_json, '$.type') = 'active' - AND json_extract(state_json, '$.phase.type') = 'awaiting_release' - AND json_extract(state_json, '$.phase.action_id') = ?4", - params![ - state_json, - loan.id.as_uuid().to_string(), - loan.resource_id.as_uuid().to_string(), - action_id.as_uuid().to_string(), - ], - )?; - if changed != 1 { - return Err(CompleteReleaseError::LoanChanged { loan_id: loan.id }); - } - - Ok(()) -} - -fn update_resource_revision( - tx: &Transaction<'_>, - authority_machine: MachineId, - resource_id: ResourceId, - expected_revision: ResourceRevision, - next_revision: ResourceRevision, -) -> Result<(), CompleteReleaseError> { - let swapped = swap_resource_revision::( - tx, - authority_machine, - resource_id, - expected_revision, - next_revision, - )?; - if !swapped { - let actual = select_resource(tx, resource_id)? - .ok_or(ResourceStoreError::ResourceNotFound)? - .state_revision; - return Err(CompleteReleaseError::StaleRevision { - expected: expected_revision, - actual, - }); - } - - Ok(()) -} - -/// Read the action and return context that a verified release receipt gave one loan -pub(crate) fn release_completion_for_loan( - conn: &Connection, - resource_id: ResourceId, - loan_id: LoanId, -) -> Result, ResourceStoreError> { - let receipt_json: Option = conn - .query_row( - "SELECT receipt_json FROM resource_release_completions - WHERE json_extract(receipt_json, '$.resource_id') = ?1 - AND json_extract(receipt_json, '$.result.loan.id') = ?2", - params![ - resource_id.as_uuid().to_string(), - loan_id.as_uuid().to_string() - ], - |row| row.get(0), - ) - .optional()?; - let receipt = receipt_json - .map(|json| stored_json::("release completion receipt", &json)) - .transpose()?; - Ok(receipt.map(|receipt| (receipt.action_id, receipt.return_context))) -} - -fn select_release_completion_receipt( - conn: &Connection, - action_id: ActionId, -) -> Result, CompleteReleaseError> { - let receipt_json: Option = conn - .query_row( - "SELECT receipt_json FROM resource_release_completions WHERE action_id = ?1", - [action_id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional()?; - - let receipt = receipt_json - .map(|json| stored_json("release completion receipt", &json)) - .transpose() - .map_err(ResourceStoreError::from)?; - Ok(receipt) -} - -fn encode_completion_json(value: &T) -> Result { - serde_json::to_string(value).map_err(|error| { - CompleteReleaseError::Storage(rusqlite::Error::ToSqlConversionFailure(Box::new(error))) - }) -} diff --git a/src/resource/store/release_loan.rs b/src/resource/store/release_loan.rs deleted file mode 100644 index 61a754b..0000000 --- a/src/resource/store/release_loan.rs +++ /dev/null @@ -1,465 +0,0 @@ -//! Queue reconciliation and release-loan opening - -use rusqlite::{Connection, OptionalExtension, Transaction, TransactionBehavior, params}; - -use super::checkpoint::{ReleaseCheckpointError, insert_release_checkpoint_state}; -use super::codec::encode_json; -use super::error::ResourceStoreError; -use super::notice::{ - SupervisorNoticeStoreError, insert_supervisor_notice_in_transaction, - select_supervisor_notice_record_by_action, -}; -use super::queue::next_queued_request_for_authority; -use super::revision::swap_resource_revision; -use super::rows::{check_authority, select_non_closed_loan, select_resource}; -use crate::domain::{ProcessGroupExitEvidence, ProcessStatus, TaskId}; -use crate::machine::MachineId; -use crate::resource::{ - ActionId, IdleBoundaryDecision, Loan, LoanId, LoanPhase, LoanState, NoticeId, - ReleaseCheckpointAction, ReleaseCheckpointPhase, ReleaseCheckpointState, ResourceId, - ResourceQueueAttentionReason, ResourceQueueReconcileOutcome, ResourceRevision, - SupervisorNotice, SupervisorNoticeDelivery, SupervisorNoticePayload, -}; - -/// Outcome of opening a release loan for one queued resource request -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) enum OpenReleaseLoanResult { - /// A new release loan and its durable supervisor notice were committed - Opened { - /// Loan created for the resource interruption - loan: Loan, - /// Notice saved in the same transaction as the loan - notice: SupervisorNotice, - }, - /// The existing release action and its saved notice were returned unchanged - AlreadyAwaitingRelease { - /// Existing active loan for this resource - loan: Loan, - /// Saved notice for the existing release action - notice: SupervisorNotice, - }, -} - -/// Failure to read or open the authority-owned queue state -#[derive(Debug, thiserror::Error)] -pub(crate) enum ResourceQueueReconcileError { - /// Resource authority or stored resource data failed validation - #[error(transparent)] - Resource(#[from] ResourceStoreError), - /// Opening a release action failed - #[error(transparent)] - Release(#[from] OpenReleaseLoanError), - /// SQLite failed during queue reconciliation - #[error("resource queue reconciliation storage error: {0}")] - Storage(#[from] rusqlite::Error), -} - -/// A release loan could not be opened for the current resource state -#[derive(Debug, thiserror::Error)] -pub(crate) enum OpenReleaseLoanError { - /// Resource authority or stored resource data failed validation - #[error(transparent)] - Resource(#[from] ResourceStoreError), - /// The saved release notice could not be read or inserted - #[error(transparent)] - Notice(#[from] SupervisorNoticeStoreError), - /// The fresh action checkpoint state could not be persisted atomically - #[error(transparent)] - Checkpoint(#[from] ReleaseCheckpointError), - /// The caller's expected revision does not match the resource state - #[error("stale resource revision: expected {expected:?}, found {actual:?}")] - StaleRevision { - /// Resource revision supplied by the caller - expected: ResourceRevision, - /// Current resource revision in SQLite - actual: ResourceRevision, - }, - /// No request remains queued for this resource - #[error("resource has no queued request")] - NoQueuedRequest, - /// The resource does not have a registered background task - #[error("resource has no registered background task")] - BackgroundTaskNotRegistered, - /// The registered background task has no local task row - #[error("registered background task {task_id} is missing locally")] - BackgroundTaskMissing { - /// Exact task registered on the resource - task_id: TaskId, - }, - /// The registered background task is not in the local running state - #[error("registered background task {task_id} is not running (state {state})")] - BackgroundTaskNotRunning { - /// Exact task registered on the resource - task_id: TaskId, - /// Process state observed in the local task table - state: String, - }, - /// Another active or attention-needed loan already owns this resource - #[error("resource already has a non-closed loan {loan:?}")] - ExistingLoan { - /// Existing non-closed loan and its typed state - loan: Box, - }, - /// A saved AwaitingRelease action has no durable notice - #[error("awaiting-release loan {loan_id:?} has no notice for action {action_id:?}")] - MissingReleaseNotice { - /// Existing loan that owns the release action - loan_id: LoanId, - /// Stable release action identity - action_id: ActionId, - }, - /// A saved release notice does not match the existing loan action - #[error("saved release notice does not match loan {loan_id:?} action {action_id:?}")] - InvalidReleaseNotice { - /// Existing loan that owns the release action - loan_id: LoanId, - /// Stable release action identity - action_id: ActionId, - }, - /// The resource state revision cannot be incremented - #[error("resource revision {revision:?} cannot be incremented")] - RevisionExhausted { - /// Current resource revision - revision: ResourceRevision, - }, - /// SQLite failed during the release transaction - #[error("release loan storage error: {0}")] - Storage(#[from] rusqlite::Error), -} - -/// Open one release action and persist its supervisor notice atomically -pub(super) fn open_release_loan_in_transaction( - tx: Transaction<'_>, - authority_machine: MachineId, - resource_id: ResourceId, - expected_state_revision: ResourceRevision, -) -> Result { - let resource = - select_resource(&tx, resource_id)?.ok_or(ResourceStoreError::ResourceNotFound)?; - check_authority(resource.authority_machine(), authority_machine)?; - - // check issued actions before queue readiness because cancellation cannot revoke them - if let Some(loan) = select_non_closed_loan(&tx, resource_id)? { - let LoanState::Active { - phase: - LoanPhase::AwaitingRelease { - action_id, - observed_background_task, - .. - }, - } = &loan.state - else { - return Err(OpenReleaseLoanError::ExistingLoan { - loan: Box::new(loan), - }); - }; - - let Some((notice, _)) = select_supervisor_notice_record_by_action(&tx, *action_id) - .map_err(ResourceStoreError::from)? - else { - return Err(OpenReleaseLoanError::MissingReleaseNotice { - loan_id: loan.id, - action_id: *action_id, - }); - }; - if notice.loan_id != loan.id - || notice.payload - != (SupervisorNoticePayload::ReleaseRequired { - task_id: *observed_background_task, - }) - { - return Err(OpenReleaseLoanError::InvalidReleaseNotice { - loan_id: loan.id, - action_id: *action_id, - }); - } - - // retries retain the original expected revision after the opening increments it - if expected_state_revision.next() != Some(notice.state_revision) - || resource.state_revision != notice.state_revision - { - return Err(OpenReleaseLoanError::StaleRevision { - expected: expected_state_revision, - actual: resource.state_revision, - }); - } - - tx.commit()?; - return Ok(OpenReleaseLoanResult::AlreadyAwaitingRelease { loan, notice }); - } - - if resource.state_revision != expected_state_revision { - return Err(OpenReleaseLoanError::StaleRevision { - expected: expected_state_revision, - actual: resource.state_revision, - }); - } - - if next_queued_request_for_authority(&tx, authority_machine, resource_id)?.is_none() { - return Err(OpenReleaseLoanError::NoQueuedRequest); - } - - let task_id = resource - .registered_background_task - .ok_or(OpenReleaseLoanError::BackgroundTaskNotRegistered)?; - match observe_registered_background_on(&tx, resource_id, task_id)? { - RegisteredBackgroundObservation::Missing => { - return Err(OpenReleaseLoanError::BackgroundTaskMissing { task_id }); - } - RegisteredBackgroundObservation::Running - | RegisteredBackgroundObservation::EndedCandidate => {} - RegisteredBackgroundObservation::NotReleasable { state } => { - return Err(OpenReleaseLoanError::BackgroundTaskNotRunning { task_id, state }); - } - } - - let next_revision = - expected_state_revision - .next() - .ok_or(OpenReleaseLoanError::RevisionExhausted { - revision: expected_state_revision, - })?; - let action_id = ActionId::new(); - let loan = Loan { - id: LoanId::new(), - resource_id, - state: LoanState::Active { - phase: LoanPhase::AwaitingRelease { - action_id, - observed_background_task: task_id, - watcher_intent: None, - }, - }, - }; - let state_json = encode_json(&loan.state)?; - tx.execute( - "INSERT INTO loans (id, resource_id, state_json) VALUES (?1, ?2, ?3)", - params![ - loan.id.as_uuid().to_string(), - resource_id.as_uuid().to_string(), - state_json, - ], - )?; - - let swapped = swap_resource_revision::( - &tx, - authority_machine, - resource_id, - expected_state_revision, - next_revision, - )?; - if !swapped { - let actual = select_resource(&tx, resource_id)? - .map_or(resource.state_revision, |saved| saved.state_revision); - return Err(OpenReleaseLoanError::StaleRevision { - expected: expected_state_revision, - actual, - }); - } - - let notice = SupervisorNotice { - id: NoticeId::new(), - loan_id: loan.id, - action_id, - state_revision: next_revision, - destination: resource.supervisor, - assignment_revision: resource.assignment_revision, - payload: SupervisorNoticePayload::ReleaseRequired { task_id }, - delivery: SupervisorNoticeDelivery::Pending { attempts: 0 }, - }; - insert_supervisor_notice_in_transaction(&tx, ¬ice)?; - insert_release_checkpoint_state( - &tx, - &ReleaseCheckpointState { - action: ReleaseCheckpointAction { - resource_id, - action_id, - state_revision: next_revision, - observed_background_task: task_id, - }, - phase: ReleaseCheckpointPhase::WatcherBindingPending, - }, - )?; - tx.commit()?; - - Ok(OpenReleaseLoanResult::Opened { loan, notice }) -} - -/// Reconcile the next queued request using only authority-owned state -pub(crate) fn reconcile_resource_queue_for_authority( - conn: &mut Connection, - authority_machine: MachineId, - resource_id: ResourceId, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let resource = - select_resource(&tx, resource_id)?.ok_or(ResourceStoreError::ResourceNotFound)?; - check_authority(resource.authority_machine(), authority_machine)?; - - if let Some(loan) = select_non_closed_loan(&tx, resource_id)? { - tx.commit()?; - return Ok(ResourceQueueReconcileOutcome::LoanAlreadyActive { loan }); - } - - // a first background launch with a confirmed start becomes the registered - // task before any queue decision reads the registration - let resource = match crate::store::promote_started_background_launch_on(&tx, &resource)? { - Some(promoted) => promoted, - None => resource, - }; - - let Some(request) = next_queued_request_for_authority(&tx, authority_machine, resource_id)? - else { - tx.commit()?; - return Ok(ResourceQueueReconcileOutcome::NoQueuedRequest); - }; - - if let Some(task_id) = crate::store::pending_background_launch_on(&tx, &resource)? { - tx.commit()?; - return Ok(ResourceQueueReconcileOutcome::AttentionRequired { - request, - reason: ResourceQueueAttentionReason::BackgroundLaunchPending { task_id }, - }); - } - - let Some(task_id) = resource.registered_background_task else { - let outcome = match crate::store::idle_boundary_decision_on(&tx, &resource)? { - IdleBoundaryDecision::Proven(proof) => { - let (loan, request) = crate::store::open_idle_serving_loan_on( - &tx, - authority_machine, - &resource, - request, - proof.clone(), - )?; - ResourceQueueReconcileOutcome::IdleServing { - loan, - request, - proof, - } - } - IdleBoundaryDecision::Unproven(gap) => { - ResourceQueueReconcileOutcome::AttentionRequired { - request, - reason: ResourceQueueAttentionReason::IdleNotProven { gap }, - } - } - }; - tx.commit()?; - return Ok(outcome); - }; - - match observe_registered_background_on(&tx, resource_id, task_id)? { - RegisteredBackgroundObservation::Missing => { - tx.commit()?; - Ok(ResourceQueueReconcileOutcome::AttentionRequired { - request, - reason: ResourceQueueAttentionReason::BackgroundTaskMissing { task_id }, - }) - } - // an ended trainer opens the same release action as a running one, so - // only the authority release proof can serve the queue from it - RegisteredBackgroundObservation::Running - | RegisteredBackgroundObservation::EndedCandidate => { - let outcome = open_release_loan_in_transaction( - tx, - authority_machine, - resource_id, - resource.state_revision, - )?; - match outcome { - OpenReleaseLoanResult::Opened { loan, notice } - | OpenReleaseLoanResult::AlreadyAwaitingRelease { loan, notice } => { - Ok(ResourceQueueReconcileOutcome::ReleaseRequired { loan, notice }) - } - } - } - RegisteredBackgroundObservation::NotReleasable { state } => { - tx.commit()?; - Ok(ResourceQueueReconcileOutcome::AttentionRequired { - request, - reason: ResourceQueueAttentionReason::BackgroundTaskNotRunning { task_id, state }, - }) - } - } -} - -/// Authority view of the registered background task when queued work needs the resource -#[derive(Debug, Clone, PartialEq, Eq)] -enum RegisteredBackgroundObservation { - /// The registered task has no authority task row - Missing, - /// The task is running, so a watcher must stop it at a checkpoint - Running, - /// The task succeeded, failed, or was cancelled, its process-group exit is - /// confirmed, and a trainer attempt association is saved - /// - /// These facts do not release the resource. They only let the release action - /// open, and the authority release proof must still verify the released - /// ownership lock, and any result or stop evidence, before it serves the queue - EndedCandidate, - /// The task ended without the facts a release proof needs - /// - /// A lost or unconfirmed end, or a missing association, can leave trainer - /// work alive, so the queue stays blocked for an owner - NotReleasable { - /// Durable task status observed by the authority - state: String, - }, -} - -fn observe_registered_background_on( - conn: &Connection, - resource_id: ResourceId, - task_id: TaskId, -) -> Result { - let status: Option = conn - .query_row( - "SELECT status FROM tasks WHERE id = ?1", - [task_id.to_string()], - |row| row.get(0), - ) - .optional()?; - let Some(status) = status else { - return Ok(RegisteredBackgroundObservation::Missing); - }; - match ProcessStatus::from_storage(&status).ok() { - Some(ProcessStatus::Running) => return Ok(RegisteredBackgroundObservation::Running), - // a lost task never has a confirmed exit, so it is not a candidate - Some(ProcessStatus::Succeeded | ProcessStatus::Failed | ProcessStatus::Cancelled) - if ended_release_candidate(conn, resource_id, task_id)? => - { - return Ok(RegisteredBackgroundObservation::EndedCandidate); - } - _ => {} - } - - Ok(RegisteredBackgroundObservation::NotReleasable { state: status }) -} - -/// Check the saved facts that an ended-trainer release proof needs before it can run -fn ended_release_candidate( - conn: &Connection, - resource_id: ResourceId, - task_id: TaskId, -) -> Result { - let evidence: Option = conn.query_row( - "SELECT process_group_exit_evidence FROM tasks WHERE id = ?1", - [task_id.to_string()], - |row| row.get(0), - )?; - if ProcessGroupExitEvidence::from_storage(evidence.as_deref()) - != ProcessGroupExitEvidence::ConfirmedExited - { - return Ok(false); - } - - Ok(conn.query_row( - "SELECT EXISTS( - SELECT 1 FROM trainer_attempt_associations - WHERE resource_id = ?1 AND task_id = ?2 - )", - params![resource_id.as_uuid().to_string(), task_id.to_string()], - |row| row.get(0), - )?) -} diff --git a/src/resource/store/release_watcher.rs b/src/resource/store/release_watcher.rs deleted file mode 100644 index e617361..0000000 --- a/src/resource/store/release_watcher.rs +++ /dev/null @@ -1,340 +0,0 @@ -//! Release-watcher identity binding for one release action - -use rusqlite::{Connection, TransactionBehavior, params}; - -use super::checkpoint::ReleaseCheckpointError; -use super::codec::{encode_json, sqlite_integer, stored_column, stored_json}; -use super::error::{ConflictReason, ResourceStoreError}; -use super::notice::select_supervisor_notice_record_by_action; -use super::rows::{check_authority, select_non_closed_loan, select_resource}; -use crate::domain::TaskId; -use crate::machine::MachineId; -use crate::resource::{ - Loan, LoanId, LoanPhase, LoanState, ReleaseWatcherIntent, ResourceId, SupervisorAddress, - SupervisorNoticePayload, -}; -use crate::spec::NormalizedSpec; -use crate::store::IdentityError; -use crate::submission::RequestId; - -/// Result of accepting the fixed identity for one local release watcher -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) enum ReleaseWatcherAcceptance { - /// The watcher row, route, identity, event, and loan binding were inserted - Inserted { task: TaskId }, - /// An exact saved acceptance was returned without adding durable records - Existing { - task: TaskId, - state: crate::domain::TaskState, - }, - /// The supervisor runs elsewhere and must launch through the remote action path - UnsupportedRemoteSupervisor { - /// Machine that owns this resource - authority_machine: MachineId, - /// Supervisor that must receive a future remote launch request - supervisor: SupervisorAddress, - }, -} - -/// Fixed task and owner data required for one release-watcher acceptance -pub(crate) struct ReleaseWatcherAcceptanceInput { - /// Authority machine recorded on the resource - pub(crate) authority_machine: MachineId, - /// Resource whose release action owns this watcher - pub(crate) resource_id: ResourceId, - /// Supervisor address recorded on the resource - pub(crate) supervisor: SupervisorAddress, - /// Fixed action, request, task, and normalized-spec identities - pub(crate) intent: ReleaseWatcherIntent, - /// Queued task row to persist with the accepted command - pub(crate) row: crate::domain::TaskRow, - /// Normalized command used for the accepted task and origin route - pub(crate) spec: NormalizedSpec, - /// Local callback environment and executable - pub(crate) callback: crate::submission::CallbackContext, -} - -/// Failure to accept the exact release-watcher identity -#[derive(Debug, thiserror::Error)] -pub(crate) enum ReleaseWatcherAcceptanceError { - /// Release action or fixed identity does not match the saved authority state - #[error("release watcher acceptance conflict")] - Conflict, - /// Resource authority state rejected the acceptance - #[error(transparent)] - Resource(#[from] ResourceStoreError), - /// Route or executor identity data is invalid or conflicting - #[error(transparent)] - Identity(#[from] IdentityError), - /// The exact release action has no verified checkpoint baseline - #[error(transparent)] - Checkpoint(#[from] ReleaseCheckpointError), - /// Task, event, or SQLite storage failed - #[error(transparent)] - Storage(#[from] crate::error::AppError), - /// SQLite transaction or storage operation failed - #[error("release watcher acceptance storage error: {0}")] - Sqlite(#[from] rusqlite::Error), -} - -/// Bind one preallocated Homebased watcher launch identity to the exact release action -pub(crate) fn bind_release_watcher_for_authority( - conn: &mut Connection, - authority_machine: MachineId, - resource_id: ResourceId, - intent: ReleaseWatcherIntent, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let result = bind_release_watcher_intent_on(&tx, authority_machine, resource_id, intent)?; - tx.commit()?; - Ok(result) -} - -pub(crate) fn validate_release_watcher_for_local_acceptance( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, - supervisor: SupervisorAddress, - intent: &ReleaseWatcherIntent, -) -> Result { - let resource = - select_resource(conn, resource_id)?.ok_or(ResourceStoreError::ResourceNotFound)?; - check_authority(resource.authority_machine(), authority_machine)?; - if resource.supervisor != supervisor { - return Err(ResourceStoreError::Conflict( - ConflictReason::ResourceAssignmentChanged, - )); - } - validate_release_watcher_intent_on(conn, authority_machine, resource_id, intent)?; - Ok(resource.supervisor) -} - -pub(crate) fn bind_release_watcher_intent_on( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, - intent: ReleaseWatcherIntent, -) -> Result { - let mut loan = - validate_release_watcher_intent_on(conn, authority_machine, resource_id, &intent)?; - let LoanState::Active { - phase: LoanPhase::AwaitingRelease { watcher_intent, .. }, - } = &mut loan.state - else { - return Err(ResourceStoreError::Conflict( - ConflictReason::ReleaseActionChanged, - )); - }; - if let Some(saved) = watcher_intent { - return if saved == &intent { - Ok(saved.clone()) - } else { - Err(ResourceStoreError::Conflict( - ConflictReason::WatcherIntentMismatch, - )) - }; - } - - // keep the action revision stable for completion using the release notice - *watcher_intent = Some(intent.clone()); - let loan_state_json = encode_json(&loan.state)?; - let changed = conn.execute( - "UPDATE loans SET state_json = ?1 - WHERE id = ?2 AND resource_id = ?3 - AND json_extract(state_json, '$.type') = 'active' - AND json_extract(state_json, '$.phase.type') = 'awaiting_release' - AND json_extract(state_json, '$.phase.action_id') = ?4 - AND json_extract(state_json, '$.phase.observed_background_task') = ?5 - AND ( - json_type(state_json, '$.phase.watcher_intent') IS NULL - OR json_type(state_json, '$.phase.watcher_intent') = 'null' - ) - AND EXISTS ( - SELECT 1 FROM resources - WHERE id = ?3 AND authority_machine = ?6 - AND state_revision = ?7 AND registered_background_task = ?5 - )", - params![ - loan_state_json, - loan.id.as_uuid().to_string(), - resource_id.as_uuid().to_string(), - intent.action_id.as_uuid().to_string(), - intent.observed_background_task.to_string(), - authority_machine.as_uuid().to_string(), - sqlite_integer(intent.state_revision.get())?, - ], - )?; - if changed != 1 { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - } - Ok(intent) -} - -pub(crate) fn validate_release_watcher_intent_on( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, - intent: &ReleaseWatcherIntent, -) -> Result { - let resource = - select_resource(conn, resource_id)?.ok_or(ResourceStoreError::ResourceNotFound)?; - check_authority(resource.authority_machine(), authority_machine)?; - let watcher_task_id = intent.watcher_task_id.as_task_id(); - if resource.state_revision != intent.state_revision { - return Err(ResourceStoreError::Conflict( - ConflictReason::ResourceRevisionChanged, - )); - } - if resource.registered_background_task != Some(intent.observed_background_task) { - return Err(ResourceStoreError::Conflict( - ConflictReason::ResourceAssignmentChanged, - )); - } - if intent.validate().is_err() { - return Err(ResourceStoreError::Conflict( - ConflictReason::WatcherIdentityInvalid, - )); - } - - let loan = select_non_closed_loan(conn, resource_id)? - .ok_or(ResourceStoreError::Conflict(ConflictReason::LoanChanged))?; - let LoanState::Active { - phase: - LoanPhase::AwaitingRelease { - action_id, - observed_background_task, - watcher_intent, - }, - } = &loan.state - else { - return Err(ResourceStoreError::Conflict( - ConflictReason::ReleaseActionChanged, - )); - }; - if *action_id != intent.action_id - || *observed_background_task != intent.observed_background_task - { - return Err(ResourceStoreError::Conflict( - ConflictReason::ReleaseActionChanged, - )); - } - - let Some((notice, _)) = select_supervisor_notice_record_by_action(conn, intent.action_id)? - else { - return Err(ResourceStoreError::Conflict( - ConflictReason::ReleaseNoticeMismatch, - )); - }; - if notice.loan_id != loan.id - || notice.state_revision != intent.state_revision - || notice.payload - != (SupervisorNoticePayload::ReleaseRequired { - task_id: intent.observed_background_task, - }) - { - return Err(ResourceStoreError::Conflict( - ConflictReason::ReleaseNoticeMismatch, - )); - } - - if release_watcher_identity_is_claimed(conn, loan.id, intent.request_id, watcher_task_id)? { - return Err(ResourceStoreError::Conflict( - ConflictReason::WatcherIdentityClaimed, - )); - } - if let Some(saved) = watcher_intent { - if saved != intent { - return Err(ResourceStoreError::Conflict( - ConflictReason::WatcherIntentMismatch, - )); - } - return Ok(loan); - } - if request_identity_exists(conn, intent.request_id)? - || local_task_identity_exists(conn, watcher_task_id)? - { - return Err(ResourceStoreError::Conflict( - ConflictReason::WatcherIdentityClaimed, - )); - } - Ok(loan) -} - -fn local_task_identity_exists(conn: &Connection, task_id: TaskId) -> Result { - conn.query_row( - "SELECT EXISTS( - SELECT 1 FROM tasks WHERE id = ?1 - UNION ALL SELECT 1 FROM executor_identities WHERE task_id = ?1 - UNION ALL SELECT 1 FROM resource_requests WHERE task_id = ?1 - UNION ALL SELECT 1 FROM resource_request_preventions WHERE task_id = ?1 - UNION ALL SELECT 1 FROM origin_routes WHERE task_id = ?1 - )", - [task_id.to_string()], - |row| row.get(0), - ) -} - -fn request_identity_exists( - conn: &Connection, - request_id: RequestId, -) -> Result { - conn.query_row( - "SELECT EXISTS( - SELECT 1 FROM origin_routes WHERE request_id = ?1 - UNION ALL SELECT 1 FROM resource_requests WHERE request_id = ?1 - UNION ALL SELECT 1 FROM resource_request_preventions WHERE request_id = ?1 - )", - [request_id.0.to_string()], - |row| row.get(0), - ) -} - -fn release_watcher_identity_is_claimed( - conn: &Connection, - loan_id: LoanId, - request_id: RequestId, - task_id: TaskId, -) -> Result { - let mut statement = conn.prepare( - "SELECT state_json FROM loans - WHERE id != ?1 - AND ( - json_type(state_json, '$.phase.watcher_intent') IS NOT NULL - OR json_type(state_json, '$.last_safe_phase.watcher_intent') IS NOT NULL - )", - )?; - - // decode saved bindings so incomplete identities cannot be ignored by uniqueness checks - let states = statement.query_map([loan_id.as_uuid().to_string()], |row| { - Ok(stored_column::(row, 0, "loan state") - .and_then(|json| stored_json::("loan state", &json))) - })?; - - for state in states { - let state = state??; - let intent = match state { - LoanState::Active { - phase: - LoanPhase::AwaitingRelease { - watcher_intent: Some(intent), - .. - }, - } - | LoanState::NeedsAttention { - last_safe_phase: - LoanPhase::AwaitingRelease { - watcher_intent: Some(intent), - .. - }, - .. - } => intent, - _ => continue, - }; - - if intent.request_id == request_id || intent.watcher_task_id.as_task_id() == task_id { - return Ok(true); - } - } - - Ok(false) -} diff --git a/src/resource/store/resources.rs b/src/resource/store/resources.rs deleted file mode 100644 index 862ff2f..0000000 --- a/src/resource/store/resources.rs +++ /dev/null @@ -1,172 +0,0 @@ -//! Resource registration and authority-wide resource reads - -use rusqlite::{Connection, Row, TransactionBehavior, params}; - -use super::codec::{collect_decoded, encode_json, sqlite_integer, stored_column, stored_json}; -use super::error::{ConflictReason, ResourceStoreError}; -use super::rows::{check_authority, decode_loan_at, decode_resource, select_resource}; -use crate::machine::MachineId; -use crate::resource::{Loan, LoanState, Resource, ResourceId, ResourceRegistrationReceipt}; - -/// One authority-owned resource and the loan that must be restored with it -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) struct ResourceSnapshot { - /// Resource whose authority matches the loading daemon - pub(crate) resource: Resource, - /// The resource's current non-closed loan, if one exists - pub(crate) loan: Option, -} - -const RESOURCE_SNAPSHOT_COLUMNS: &str = "r.id, r.display_name, r.authority_machine, - r.supervisor_machine, r.supervisor_thread, r.assignment_revision, r.state_revision, - r.registered_background_task, l.id, l.resource_id, l.state_json"; - -/// Register one resource on its declared authority without changing fixed content -pub(crate) fn register_resource_for_authority( - conn: &mut Connection, - authority_machine: MachineId, - resource: &Resource, -) -> Result { - check_authority(resource.authority_machine(), authority_machine)?; - - let assignment_revision = sqlite_integer(resource.assignment_revision.get())?; - let state_revision = sqlite_integer(resource.state_revision.get())?; - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - - let receipt = ResourceRegistrationReceipt::initial(resource); - if let Some(saved) = select_resource(&tx, resource.id)? { - // an exact retry matches the first registration, even after a supervisor replacement - if select_resource_registration(&tx, resource.id)? != receipt { - return Err(ResourceStoreError::RegistrationConflict { - resource: resource.id, - }); - } - - tx.commit()?; - return Ok(saved); - } - - tx.execute( - "INSERT INTO resources ( - id, display_name, authority_machine, supervisor_machine, supervisor_thread, - assignment_revision, state_revision, registered_background_task - ) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8)", - params![ - resource.id.as_uuid().to_string(), - resource.display_name, - resource.authority_machine().as_uuid().to_string(), - resource.supervisor.machine.as_uuid().to_string(), - resource.supervisor.thread.to_string(), - assignment_revision, - state_revision, - resource - .registered_background_task - .map(|task| task.to_string()), - ], - )?; - insert_resource_registration(&tx, &receipt)?; - tx.commit()?; - Ok(resource.clone()) -} - -fn insert_resource_registration( - conn: &Connection, - receipt: &ResourceRegistrationReceipt, -) -> Result<(), ResourceStoreError> { - let json = encode_json(receipt)?; - conn.execute( - "INSERT INTO resource_registration_receipts (resource_id, receipt_json) - VALUES (?1, ?2)", - params![receipt.resource_id.as_uuid().to_string(), json], - )?; - Ok(()) -} - -/// Read the first registration saved for one existing resource -/// -/// Registration saves the resource row and this receipt in one transaction, so -/// a missing receipt is a storage error -fn select_resource_registration( - conn: &Connection, - resource_id: ResourceId, -) -> Result { - let receipt: ResourceRegistrationReceipt = stored_json( - "resource registration receipt", - &conn.query_row( - "SELECT receipt_json FROM resource_registration_receipts WHERE resource_id = ?1", - [resource_id.as_uuid().to_string()], - |row| row.get::<_, String>(0), - )?, - )?; - if receipt.resource_id != resource_id { - return Err(ResourceStoreError::Conflict( - ConflictReason::RegistrationReceiptMismatch, - )); - } - Ok(receipt) -} - -/// Load each resource owned by one authority with its current non-closed loan -pub(crate) fn resources_for_authority( - conn: &Connection, - authority_machine: MachineId, -) -> Result, ResourceStoreError> { - let sql = format!( - "SELECT {RESOURCE_SNAPSHOT_COLUMNS} - FROM resources AS r - LEFT JOIN loans AS l - ON l.resource_id = r.id - WHERE r.authority_machine = ?1 - ORDER BY r.id, l.id" - ); - let mut statement = conn.prepare(&sql)?; - let rows = statement.query_map([authority_machine.to_string()], |row| { - Ok(decode_resource_snapshot(row)) - })?; - let mut snapshots: Vec = Vec::new(); - - for mut candidate in collect_decoded(rows)? { - match snapshots.last_mut() { - Some(snapshot) if snapshot.resource.id == candidate.resource.id => { - let Some(loan) = candidate.loan.take() else { - continue; - }; - if matches!(&loan.state, LoanState::Closed { .. }) { - continue; - } - if snapshot.loan.is_some() { - return Err(ResourceStoreError::corrupt( - "resource loans", - format!( - "resource {:?} has multiple non-closed loans", - snapshot.resource.id - ), - )); - } - snapshot.loan = Some(loan); - } - _ => { - if candidate - .loan - .as_ref() - .is_some_and(|loan| matches!(&loan.state, LoanState::Closed { .. })) - { - candidate.loan = None; - } - snapshots.push(candidate); - } - } - } - - Ok(snapshots) -} - -fn decode_resource_snapshot(row: &Row<'_>) -> Result { - let resource = decode_resource(row)?; - // the left join leaves every loan column null for a resource with no loans - let loan = stored_column::>(row, 8, "loan identity")? - .map(|_| decode_loan_at(row, 8)) - .transpose()?; - - Ok(ResourceSnapshot { resource, loan }) -} diff --git a/src/resource/store/return_window.rs b/src/resource/store/return_window.rs deleted file mode 100644 index 29bcae4..0000000 --- a/src/resource/store/return_window.rs +++ /dev/null @@ -1,375 +0,0 @@ -//! Saved return decision windows and the queue serving that follows an expired one -//! -//! Every ReturnRequired notice opens the decision window of its action in the -//! same transaction, so an AwaitingReturn loan always has one. When the window -//! closes with a request queued, one IMMEDIATE transaction moves the same loan -//! to Serving, assigns the request, advances the resource revision, and saves -//! the deadline receipt that the new release provenance names. The return -//! context stays on the loan, so the next drained queue opens a new return -//! action for the same obligation - -use chrono::{DateTime, Utc}; -use rusqlite::{Connection, OptionalExtension, TransactionBehavior, params}; -use serde::{Deserialize, Serialize}; - -use super::codec::{StoredReadError, encode_json, stored_id, stored_json}; -use super::error::{ConflictReason, ResourceStoreError}; -use super::queue::next_queued_request_for_authority; -use super::revision::swap_resource_revision; -use super::rows::{check_authority, select_non_closed_loan, select_resource}; -use crate::domain::TaskId; -use crate::machine::MachineId; -use crate::resource::{ - ActionId, Loan, LoanId, LoanPhase, LoanState, NilIdentity, Resource, ResourceId, - ResourceRequest, ResourceRequestState, ResourceRevision, ReturnContext, ReturnDecisionWindow, - ServingReleaseProvenance, -}; -use crate::submission::RequestId; - -/// Evidence that queued work took the resource from an undecided return action -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct ReturnDeadlineReceipt { - /// Expired window, as saved when the queue took the resource - window: ReturnDecisionWindow, - /// Authority that committed the transition - authority_machine: MachineId, - /// Return obligation that the loan keeps while it serves - return_context: ReturnContext, - /// Registration that the loan kept while it waited for the decision - registered_background_task: Option, - /// Request that the loan started to serve - request_id: RequestId, - /// Resource revision of the AwaitingReturn loan - expected_state_revision: ResourceRevision, - /// Resource revision committed with the transition - state_revision: ResourceRevision, - /// Authority time at which the transition committed - served_at: DateTime, -} - -/// Result of checking the decision window of the current return action -#[derive(Debug, Clone)] -pub(crate) enum ReturnDeadlineOutcome { - /// The resource has no loan that awaits a return decision - NotAwaiting, - /// The supervisor still has time to decide - Open { - /// Current window of the pending action - window: ReturnDecisionWindow, - }, - /// The window closed, but no request waits for the resource - NoQueuedRequest, - /// The window closed, and the same loan now serves the next queued request - Served { - // boxed because both records are large and the other variants are small - /// Serving loan that keeps the return context - loan: Box, - /// Request assigned by the transition - request: Box, - }, -} - -/// Open the decision window of one return action in the caller's transaction -/// -/// A retry of the same notice finds the saved window and keeps it -pub(crate) fn open_return_window_on( - conn: &Connection, - action_id: ActionId, - loan_id: LoanId, - opened_at: DateTime, -) -> Result<(), rusqlite::Error> { - let resource_id: String = conn.query_row( - "SELECT resource_id FROM loans WHERE id = ?1", - [loan_id.as_uuid().to_string()], - |row| row.get(0), - )?; - let resource_id: ResourceId = saved_identity("loan resource identity", &resource_id)?; - let window = ReturnDecisionWindow::open(action_id, loan_id, resource_id, opened_at); - let window_json = serde_json::to_string(&window) - .map_err(|error| rusqlite::Error::ToSqlConversionFailure(Box::new(error)))?; - conn.execute( - "INSERT INTO resource_return_windows (action_id, loan_id, resource_id, window_json) - VALUES (?1, ?2, ?3, ?4) - ON CONFLICT (action_id) DO NOTHING", - params![ - action_id.as_uuid().to_string(), - loan_id.as_uuid().to_string(), - resource_id.as_uuid().to_string(), - window_json, - ], - )?; - Ok(()) -} - -/// Open a window at `opened_at` for every AwaitingReturn loan saved without one -/// -/// Loans that entered AwaitingReturn before windows existed get their full -/// grace from the upgrade, not from their unknown opening time -pub(crate) fn open_missing_return_windows_on( - conn: &Connection, - opened_at: DateTime, -) -> Result<(), rusqlite::Error> { - let pending = { - let mut statement = conn.prepare( - "SELECT json_extract(loan.state_json, '$.phase.action_id'), loan.id - FROM loans AS loan - WHERE json_extract(loan.state_json, '$.type') = 'active' - AND json_extract(loan.state_json, '$.phase.type') = 'awaiting_return' - AND NOT EXISTS ( - SELECT 1 FROM resource_return_windows AS window - WHERE window.action_id = json_extract(loan.state_json, '$.phase.action_id') - )", - )?; - let rows = statement.query_map([], |row| { - Ok((row.get::<_, String>(0)?, row.get::<_, String>(1)?)) - })?; - rows.collect::, _>>()? - }; - for (action_id, loan_id) in pending { - let action_id: ActionId = saved_identity("awaiting return action identity", &action_id)?; - let loan_id: LoanId = saved_identity("awaiting return loan identity", &loan_id)?; - open_return_window_on(conn, action_id, loan_id, opened_at)?; - } - Ok(()) -} - -/// Parse one saved identity for a step whose callers handle only SQLite errors -/// -/// Notice insertion and the schema upgrade run on plain rusqlite results, so a -/// nil or malformed identity becomes a conversion failure that aborts them -fn saved_identity(what: &'static str, text: &str) -> Result -where - T: TryFrom, -{ - stored_id(what, text).map_err(|error| match error { - StoredReadError::Storage(error) => error, - StoredReadError::Corrupt { what, reason } => rusqlite::Error::FromSqlConversionFailure( - 0, - rusqlite::types::Type::Text, - format!("{what}: {reason}").into(), - ), - }) -} - -/// Read the saved decision window of one return action -pub(crate) fn return_window_on( - conn: &Connection, - action_id: ActionId, -) -> Result, ResourceStoreError> { - let saved: Option = conn - .query_row( - "SELECT window_json FROM resource_return_windows WHERE action_id = ?1", - [action_id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional()?; - let Some(saved) = saved else { - return Ok(None); - }; - let window: ReturnDecisionWindow = stored_json("return decision window", &saved)?; - if window.action_id() != action_id { - return Err(ResourceStoreError::corrupt( - "return decision window", - "saved window names another action", - )); - } - Ok(Some(window)) -} - -/// Replace the deadline of a saved window in the caller's transaction -/// -/// Only the deadline may change; the identities and opening time are fixed -pub(crate) fn save_held_return_window_on( - conn: &Connection, - saved: &ReturnDecisionWindow, - held: &ReturnDecisionWindow, -) -> Result<(), ResourceStoreError> { - if held.action_id() != saved.action_id() - || held.loan_id() != saved.loan_id() - || held.resource_id() != saved.resource_id() - || held.opened_at() != saved.opened_at() - { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - } - let changed = conn.execute( - "UPDATE resource_return_windows SET window_json = ?1 - WHERE action_id = ?2 AND window_json = ?3", - params![ - encode_json(held)?, - saved.action_id().as_uuid().to_string(), - encode_json(saved)?, - ], - )?; - if changed != 1 { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - } - Ok(()) -} - -/// Serve the next queued request once the current return action's window closed -pub(crate) fn serve_after_return_deadline_for_authority( - conn: &mut Connection, - authority_machine: MachineId, - resource_id: ResourceId, - now: DateTime, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let resource = - select_resource(&tx, resource_id)?.ok_or(ResourceStoreError::ResourceNotFound)?; - check_authority(resource.authority_machine(), authority_machine)?; - let Some(loan) = select_non_closed_loan(&tx, resource_id)? else { - return Ok(ReturnDeadlineOutcome::NotAwaiting); - }; - let LoanState::Active { - phase: - LoanPhase::AwaitingReturn { - action_id, - return_context, - }, - } = &loan.state - else { - return Ok(ReturnDeadlineOutcome::NotAwaiting); - }; - let window = return_window_on(&tx, *action_id)?.ok_or_else(|| { - ResourceStoreError::corrupt( - "return decision window", - "an AwaitingReturn loan has no saved window", - ) - })?; - if window.loan_id() != loan.id || window.resource_id() != resource_id { - return Err(ResourceStoreError::corrupt( - "return decision window", - "saved window names another loan", - )); - } - if !window.expired_at(now) { - return Ok(ReturnDeadlineOutcome::Open { window }); - } - let Some(mut request) = next_queued_request_for_authority(&tx, authority_machine, resource_id)? - else { - return Ok(ReturnDeadlineOutcome::NoQueuedRequest); - }; - - let state_revision = resource - .state_revision - .next() - .ok_or(ResourceStoreError::Conflict( - ConflictReason::RevisionExhausted, - ))?; - request.state = ResourceRequestState::Assigned { loan_id: loan.id }; - let changed = tx.execute( - "UPDATE resource_requests SET state_json = ?1 - WHERE request_id = ?2 AND resource_id = ?3 - AND json_extract(state_json, '$.type') = 'queued'", - params![ - encode_json(&request.state)?, - request.request_id.0.to_string(), - resource_id.as_uuid().to_string(), - ], - )?; - if changed != 1 { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestStateChanged, - )); - } - - let serving = Loan { - id: loan.id, - resource_id, - state: LoanState::Active { - phase: LoanPhase::Serving { - return_context: return_context.clone(), - current_request_id: request.request_id, - release_provenance: ServingReleaseProvenance::ReturnDeadlinePassed { - action_id: *action_id, - }, - }, - }, - }; - let changed = tx.execute( - "UPDATE loans SET state_json = ?1 - WHERE id = ?2 AND resource_id = ?3 - AND json_extract(state_json, '$.type') = 'active' - AND json_extract(state_json, '$.phase.type') = 'awaiting_return' - AND json_extract(state_json, '$.phase.action_id') = ?4", - params![ - encode_json(&serving.state)?, - loan.id.as_uuid().to_string(), - resource_id.as_uuid().to_string(), - action_id.as_uuid().to_string(), - ], - )?; - if changed != 1 { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - } - if !swap_resource_revision::( - &tx, - authority_machine, - resource_id, - resource.state_revision, - state_revision, - )? { - return Err(ResourceStoreError::Conflict( - ConflictReason::ResourceRevisionChanged, - )); - } - - let receipt = ReturnDeadlineReceipt { - window, - authority_machine, - return_context: return_context.clone(), - registered_background_task: resource.registered_background_task, - request_id: request.request_id, - expected_state_revision: resource.state_revision, - state_revision, - served_at: now, - }; - tx.execute( - "INSERT INTO resource_return_deadline_servings - (action_id, loan_id, resource_id, receipt_json) - VALUES (?1, ?2, ?3, ?4)", - params![ - action_id.as_uuid().to_string(), - loan.id.as_uuid().to_string(), - resource_id.as_uuid().to_string(), - encode_json(&receipt)?, - ], - )?; - tx.commit()?; - - Ok(ReturnDeadlineOutcome::Served { - loan: Box::new(serving), - request: Box::new(request), - }) -} - -/// Whether the deadline receipt of `action_id` still proves this Serving loan -pub(super) fn return_deadline_serving_matches_on( - conn: &Connection, - authority_machine: MachineId, - resource: &Resource, - loan: &Loan, - return_context: &ReturnContext, - action_id: ActionId, -) -> Result { - let saved: Option = conn - .query_row( - "SELECT receipt_json FROM resource_return_deadline_servings WHERE action_id = ?1", - [action_id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional()?; - let Some(saved) = saved else { - return Ok(false); - }; - let receipt: ReturnDeadlineReceipt = stored_json("return deadline receipt", &saved)?; - Ok(receipt.window.action_id() == action_id - && receipt.window.loan_id() == loan.id - && receipt.window.resource_id() == resource.id - && loan.resource_id == resource.id - && receipt.authority_machine == authority_machine - && resource.authority_machine() == authority_machine - && receipt.return_context == *return_context - && receipt.registered_background_task == resource.registered_background_task) -} diff --git a/src/resource/store/revision.rs b/src/resource/store/revision.rs deleted file mode 100644 index 1bd2737..0000000 --- a/src/resource/store/revision.rs +++ /dev/null @@ -1,35 +0,0 @@ -//! Compare-and-swap of the resource state revision - -use rusqlite::{Connection, params}; - -use super::codec::sqlite_integer; -use super::error::ResourceStoreError; -use crate::machine::MachineId; -use crate::resource::{ResourceId, ResourceRevision}; - -/// Advance the revision of an authority-owned resource only from `expected` -/// -/// Returns `false` when the resource, its authority, or its revision no longer -/// match, so each caller can report the stale state in its own error type -pub(crate) fn swap_resource_revision( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, - expected: ResourceRevision, - next: ResourceRevision, -) -> Result -where - E: From + From, -{ - let changed = conn.execute( - "UPDATE resources SET state_revision = ?1 - WHERE id = ?2 AND authority_machine = ?3 AND state_revision = ?4", - params![ - sqlite_integer(next.get())?, - resource_id.as_uuid().to_string(), - authority_machine.as_uuid().to_string(), - sqlite_integer(expected.get())?, - ], - )?; - Ok(changed == 1) -} diff --git a/src/resource/store/rows.rs b/src/resource/store/rows.rs deleted file mode 100644 index 09f6c1b..0000000 --- a/src/resource/store/rows.rs +++ /dev/null @@ -1,200 +0,0 @@ -//! Shared row reads and authority checks -//! -//! Decoders return `ResourceStoreError` so a saved value that cannot decode -//! surfaces as `CorruptRecord` for attention instead of a retryable SQLite error - -use rusqlite::{Connection, OptionalExtension, Row}; - -use super::codec::{stored_column, stored_count, stored_id, stored_json, stored_uuid}; -use super::error::ResourceStoreError; -use crate::domain::{TaskId, ThreadId}; -use crate::machine::MachineId; -use crate::resource::{ - AcceptanceSequence, AssignmentRevision, Loan, LoanId, LoanState, Resource, ResourceId, - ResourceRequest, ResourceRequestState, ResourceRevision, SupervisorAddress, -}; -use crate::spec::NormalizedSpec; -use crate::submission::RequestId; - -const RESOURCE_COLUMNS: &str = "id, display_name, authority_machine, supervisor_machine, - supervisor_thread, assignment_revision, state_revision, registered_background_task"; - -pub(super) const REQUEST_COLUMNS: &str = "acceptance_sequence, request_id, task_id, resource_id, - origin_machine, spec_json, state_json"; - -pub(crate) fn select_resource( - conn: &Connection, - resource_id: ResourceId, -) -> Result, ResourceStoreError> { - let sql = format!("SELECT {RESOURCE_COLUMNS} FROM resources WHERE id = ?1"); - conn.query_row(&sql, [resource_id.as_uuid().to_string()], |row| { - Ok(decode_resource(row)) - }) - .optional()? - .transpose() -} - -pub(crate) fn select_non_closed_loan( - conn: &Connection, - resource_id: ResourceId, -) -> Result, ResourceStoreError> { - conn.query_row( - "SELECT id, resource_id, state_json FROM loans - WHERE resource_id = ?1 AND json_extract(state_json, '$.type') != 'closed'", - [resource_id.as_uuid().to_string()], - |row| Ok(decode_loan(row)), - ) - .optional()? - .transpose() -} - -/// Decode a loan from `id, resource_id, state_json` columns starting at `first` -pub(super) fn decode_loan_at(row: &Row<'_>, first: usize) -> Result { - let id: LoanId = stored_id( - "loan identity", - &stored_column::(row, first, "loan identity")?, - )?; - let resource_id: ResourceId = stored_id( - "loan resource identity", - &stored_column::(row, first + 1, "loan resource identity")?, - )?; - let state: LoanState = stored_json( - "loan state", - &stored_column::(row, first + 2, "loan state")?, - )?; - Ok(Loan { - id, - resource_id, - state, - }) -} - -pub(super) fn decode_loan(row: &Row<'_>) -> Result { - decode_loan_at(row, 0) -} - -pub(crate) fn select_request_by_id( - conn: &Connection, - request_id: RequestId, -) -> Result, ResourceStoreError> { - let sql = format!("SELECT {REQUEST_COLUMNS} FROM resource_requests WHERE request_id = ?1"); - conn.query_row(&sql, [request_id.0.to_string()], |row| { - Ok(decode_request(row)) - }) - .optional()? - .transpose() -} - -/// Require an existing resource whose fixed authority is `authority_machine` -pub(super) fn check_resource_authority( - conn: &Connection, - resource_id: ResourceId, - authority_machine: MachineId, -) -> Result<(), ResourceStoreError> { - check_authority(resource_authority(conn, resource_id)?, authority_machine) -} - -fn resource_authority( - conn: &Connection, - resource_id: ResourceId, -) -> Result { - select_resource(conn, resource_id)? - .map(|resource| resource.authority_machine()) - .ok_or(ResourceStoreError::ResourceNotFound) -} - -pub(super) fn check_authority( - expected: MachineId, - found: MachineId, -) -> Result<(), ResourceStoreError> { - if expected == found { - return Ok(()); - } - - Err(ResourceStoreError::WrongAuthority { expected, found }) -} - -pub(super) fn decode_resource(row: &Row<'_>) -> Result { - let text = |index, what| stored_column::(row, index, what); - let id: ResourceId = stored_id("resource identity", &text(0, "resource identity")?)?; - let display_name = stored_column(row, 1, "resource display name")?; - let authority_machine = MachineId::from_uuid(stored_uuid( - "resource authority machine", - &text(2, "resource authority machine")?, - )?); - let supervisor_machine = MachineId::from_uuid(stored_uuid( - "resource supervisor machine", - &text(3, "resource supervisor machine")?, - )?); - let supervisor_thread = ThreadId(stored_uuid( - "resource supervisor thread", - &text(4, "resource supervisor thread")?, - )?); - let assignment_revision = AssignmentRevision::new(stored_count( - "resource assignment revision", - stored_column(row, 5, "resource assignment revision")?, - )?); - let state_revision = ResourceRevision::new(stored_count( - "resource state revision", - stored_column(row, 6, "resource state revision")?, - )?); - let registered_background_task = - stored_column::>(row, 7, "resource registered background task")? - .map(|value| stored_uuid("resource registered background task", &value).map(TaskId)) - .transpose()?; - - Ok(Resource::new( - id, - display_name, - authority_machine, - SupervisorAddress { - machine: supervisor_machine, - thread: supervisor_thread, - }, - assignment_revision, - state_revision, - registered_background_task, - )) -} - -pub(super) fn decode_request(row: &Row<'_>) -> Result { - let text = |index, what| stored_column::(row, index, what); - let sequence = AcceptanceSequence::new(stored_count( - "request acceptance sequence", - stored_column(row, 0, "request acceptance sequence")?, - )?); - let request_id = RequestId(stored_uuid( - "request identity", - &text(1, "request identity")?, - )?); - let task_id = TaskId(stored_uuid( - "request task identity", - &text(2, "request task identity")?, - )?); - let resource_id: ResourceId = stored_id( - "request resource identity", - &text(3, "request resource identity")?, - )?; - let origin_machine = MachineId::from_uuid(stored_uuid( - "request origin machine", - &text(4, "request origin machine")?, - )?); - let normalized_spec: NormalizedSpec = - stored_json("resource request spec", &text(5, "resource request spec")?)?; - let state: ResourceRequestState = stored_json( - "resource request state", - &text(6, "resource request state")?, - )?; - - let mut request = ResourceRequest::new( - request_id, - task_id, - resource_id, - sequence, - origin_machine, - normalized_spec, - ) - .map_err(|error| ResourceStoreError::corrupt("resource request spec", error))?; - request.state = state; - Ok(request) -} diff --git a/src/resource/store/test_support.rs b/src/resource/store/test_support.rs deleted file mode 100644 index cad209f..0000000 --- a/src/resource/store/test_support.rs +++ /dev/null @@ -1,234 +0,0 @@ -//! Test fixtures that drive the authority store without a daemon - -use rusqlite::{Connection, TransactionBehavior, params}; - -use super::codec::encode_json; -use super::error::{ConflictReason, ResourceStoreError}; -use super::notice::{SupervisorNoticeStoreError, retarget_supervisor_notice_in_transaction}; -use super::release_completion::{ReleaseCompletionReceipt, ReleaseCompletionResult}; -use super::release_loan::{ - OpenReleaseLoanError, OpenReleaseLoanResult, open_release_loan_in_transaction, -}; -use super::rows::{select_request_by_id, select_resource}; -use super::schema::RESOURCE_SCHEMA; -use super::{ - QueueCancellationResult, accept_request_for_authority, cancel_request_before_activation_on, - next_queued_request_for_authority, register_resource_for_authority, - requests_for_resource_for_authority, -}; -use crate::domain::TaskId; -use crate::machine::MachineId; -use crate::resource::{ - ActionId, AssignmentRevision, Loan, LoanPhase, LoanState, NoticeId, Resource, ResourceId, - ResourceRequest, ResourceRevision, ServingReleaseProvenance, SupervisorAddress, - SupervisorNotice, -}; -use crate::spec::NormalizedSpec; -use crate::submission::RequestId; - -/// Install the resource schema on an explicitly supplied connection -pub(crate) fn install_schema(conn: &mut Connection) -> Result<(), ResourceStoreError> { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - tx.execute_batch(RESOURCE_SCHEMA)?; - tx.commit()?; - Ok(()) -} - -/// Authority saved for a resource -/// -/// An unknown resource gets a fresh machine, because the store reports -/// `ResourceNotFound` before it compares authorities -fn saved_authority(conn: &Connection, resource_id: ResourceId) -> MachineId { - select_resource(conn, resource_id) - .ok() - .flatten() - .map_or_else(MachineId::new, |resource| resource.authority_machine()) -} - -/// Register a resource on its own declared authority -pub(crate) fn register_resource( - conn: &mut Connection, - resource: &Resource, -) -> Result { - register_resource_for_authority(conn, resource.authority_machine(), resource) -} - -/// Accept a request on the saved authority of its resource -pub(crate) fn accept_request( - conn: &mut Connection, - request_id: RequestId, - task_id: TaskId, - resource_id: ResourceId, - origin_machine: MachineId, - normalized_spec: NormalizedSpec, -) -> Result { - let authority = saved_authority(conn, resource_id); - accept_request_for_authority( - conn, - authority, - request_id, - task_id, - resource_id, - origin_machine, - normalized_spec, - ) -} - -/// Open one release action in its own transaction -pub(crate) fn open_release_loan_for_authority( - conn: &mut Connection, - authority_machine: MachineId, - resource_id: ResourceId, - expected_state_revision: ResourceRevision, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - open_release_loan_in_transaction(tx, authority_machine, resource_id, expected_state_revision) -} - -/// Retarget an undelivered notice in its own transaction -pub(crate) fn retarget_supervisor_notice( - conn: &mut Connection, - notice_id: NoticeId, - expected_assignment_revision: AssignmentRevision, - destination: SupervisorAddress, - new_assignment_revision: AssignmentRevision, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let notice = retarget_supervisor_notice_in_transaction( - &tx, - notice_id, - expected_assignment_revision, - destination, - new_assignment_revision, - )?; - tx.commit()?; - Ok(notice) -} - -/// Cancel a queued or assigned request on its fixed authority in its own transaction -pub(crate) fn cancel_request_before_activation_for_authority( - conn: &mut Connection, - authority_machine: MachineId, - request_id: RequestId, - task_id: TaskId, - resource_id: ResourceId, - origin_machine: MachineId, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let result = cancel_request_before_activation_on( - &tx, - authority_machine, - request_id, - task_id, - resource_id, - origin_machine, - )?; - tx.commit()?; - Ok(result) -} - -/// Cancel a request on the saved authority of its resource -pub(crate) fn cancel_request_before_activation( - conn: &mut Connection, - request_id: RequestId, - task_id: TaskId, - resource_id: ResourceId, - origin_machine: MachineId, -) -> Result { - let authority = saved_authority(conn, resource_id); - cancel_request_before_activation_for_authority( - conn, - authority, - request_id, - task_id, - resource_id, - origin_machine, - ) -} - -/// Read all requests of a resource on its saved authority -pub(crate) fn requests_for_resource( - conn: &Connection, - resource_id: ResourceId, -) -> Result, ResourceStoreError> { - requests_for_resource_for_authority(conn, saved_authority(conn, resource_id), resource_id) -} - -/// Read the next queued request of a resource on its saved authority -pub(crate) fn next_queued_request( - conn: &Connection, - resource_id: ResourceId, -) -> Result, ResourceStoreError> { - next_queued_request_for_authority(conn, saved_authority(conn, resource_id), resource_id) -} - -/// Save a verified completed-trainer release receipt for one Serving loan -pub(crate) fn seed_verified_serving_provenance_for_test( - conn: &mut Connection, - mut loan: Loan, - authority_machine: MachineId, - resource_id: ResourceId, - request_id: RequestId, - expected_state_revision: ResourceRevision, - state_revision: ResourceRevision, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let resource = - select_resource(&tx, resource_id)?.ok_or(ResourceStoreError::ResourceNotFound)?; - let task_id = resource - .registered_background_task - .unwrap_or_else(TaskId::new); - let action_id = ActionId::new(); - let publication_sha256 = crate::digest::Sha256Digest::from_bytes([0xab; 32]); - let LoanState::Active { - phase: - LoanPhase::Serving { - return_context, - release_provenance, - .. - }, - } = &mut loan.state - else { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - }; - *release_provenance = ServingReleaseProvenance::CompletedTrainerResult { - action_id, - task_id, - publication_sha256, - }; - let request = select_request_by_id(&tx, request_id)? - .ok_or(ResourceStoreError::Conflict(ConflictReason::RequestMissing))?; - let receipt = ReleaseCompletionReceipt { - action_id, - authority_machine, - resource_id, - expected_state_revision, - return_context: return_context.clone(), - release_provenance: release_provenance.clone(), - result: ReleaseCompletionResult::Assigned { - loan: loan.clone(), - request, - state_revision, - }, - }; - let state_json = encode_json(&loan.state)?; - let changed = tx.execute( - "UPDATE loans SET state_json = ?1 WHERE id = ?2 AND resource_id = ?3", - params![ - state_json, - loan.id.as_uuid().to_string(), - resource_id.as_uuid().to_string(), - ], - )?; - if changed != 1 { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)); - } - tx.execute( - "INSERT INTO resource_release_completions (action_id, receipt_json) - VALUES (?1, ?2)", - params![action_id.as_uuid().to_string(), encode_json(&receipt)?,], - )?; - tx.commit()?; - - Ok(loan) -} diff --git a/src/resource/store/tests.rs b/src/resource/store/tests.rs deleted file mode 100644 index 52445ad..0000000 --- a/src/resource/store/tests.rs +++ /dev/null @@ -1,2574 +0,0 @@ -//! Authority store behavior against an in-memory schema - -use std::path::Path; - -use rusqlite::{Connection, TransactionBehavior, params}; -use serde_json::json; -use tempfile::tempdir; -use uuid::Uuid; - -use super::assigned_task::{ - AssignedResourceTaskAttention, ResourceTaskReleaseProof, resource_task_release_proof, -}; -use super::cancellation::QueueCancellationResult; -use super::error::{ConflictReason, ResourceStoreError}; -use super::notice::{ - INTERRUPTED_DELIVERY_ERROR, RETARGETED_DELIVERY_ERROR, SupervisorNoticeStoreError, - insert_supervisor_notice_in_transaction, pending_supervisor_notices, - recover_sending_supervisor_notices, reserve_supervisor_notice_attempt, - select_supervisor_notice_record_by_action, settle_supervisor_notice_attempt, supervisor_notice, -}; -use super::queue::{ - accept_request_for_authority, executor_identity_exists, next_queued_request_for_authority, - requests_for_resource_for_authority, -}; -use super::release_loan::{ - OpenReleaseLoanError, OpenReleaseLoanResult, reconcile_resource_queue_for_authority, -}; -use super::resources::{register_resource_for_authority, resources_for_authority}; -use super::rows::{select_non_closed_loan, select_request_by_id, select_resource}; -use super::test_support::{ - accept_request, cancel_request_before_activation, install_schema, next_queued_request, - register_resource, requests_for_resource, -}; -use super::test_support::{ - cancel_request_before_activation_for_authority, open_release_loan_for_authority, - retarget_supervisor_notice, -}; -use crate::domain::{ExitReason, TaskId, ThreadId, WorkExitEvidence}; -use crate::machine::{MachineId, MachineName}; -use crate::resource::{ - AcceptanceSequence, ActionId, AssignmentRevision, CommandSpecError, DeliveryAttemptId, Loan, - LoanClosure, LoanId, LoanPhase, LoanState, NoticeId, Resource, ResourceId, - ResourceQueueAttentionReason, ResourceQueueReconcileOutcome, ResourceRequest, - ResourceRequestState, ResourceRevision, ResourceTaskOwnershipRisk, ReturnContext, - SupervisorAddress, SupervisorNotice, SupervisorNoticeDelivery, SupervisorNoticePayload, -}; -use crate::spec::NormalizedSpec; -use crate::submission::{ExecutionRecord, ExecutorIdentity, RejectionTombstone, RequestId}; - -fn connection() -> Connection { - let mut conn = Connection::open_in_memory().unwrap(); - conn.pragma_update(None, "foreign_keys", "ON").unwrap(); - install_schema(&mut conn).unwrap(); - conn.execute_batch( - "CREATE TABLE tasks ( - id TEXT PRIMARY KEY, - status TEXT NOT NULL DEFAULT 'queued' - ); - CREATE TABLE executor_identities ( - task_id TEXT PRIMARY KEY, - origin_machine TEXT NOT NULL, - identity_json TEXT NOT NULL - ); - CREATE TABLE origin_routes ( - request_id TEXT PRIMARY KEY, - task_id TEXT NOT NULL UNIQUE, - execution_machine TEXT NOT NULL, - spec_json TEXT NOT NULL, - route_json TEXT NOT NULL - );", - ) - .unwrap(); - conn -} - -fn resource() -> Resource { - resource_for_authority(MachineId::new()) -} - -fn resource_for_authority(authority: MachineId) -> Resource { - Resource::new( - ResourceId::new(), - "gpu-0".into(), - authority, - SupervisorAddress { - machine: authority, - thread: ThreadId(Uuid::now_v7()), - }, - AssignmentRevision::new(0), - ResourceRevision::new(0), - None, - ) -} - -fn resource_with_background_task(task_id: TaskId) -> Resource { - let mut resource = resource(); - resource.registered_background_task = Some(task_id); - resource -} - -fn queue_request(conn: &mut Connection, resource_id: ResourceId) -> ResourceRequest { - accept_request( - conn, - RequestId::new(), - TaskId::new(), - resource_id, - MachineId::new(), - command_spec(&["echo", "queued"]), - ) - .unwrap() -} - -fn insert_local_task_status(conn: &Connection, task_id: TaskId, status: &str) { - conn.execute( - "INSERT INTO tasks (id, status) VALUES (?1, ?2)", - params![task_id.to_string(), status], - ) - .unwrap(); -} - -fn command_spec(command: &[&str]) -> NormalizedSpec { - serde_json::from_value(json!({ - "api_version": 1, - "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "gpu command", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": command } - })) - .unwrap() -} - -#[test] -fn no_child_evidence_proves_release_only_for_pre_spawn_outcomes() { - let request = ResourceRequest::new( - RequestId::new(), - TaskId::new(), - ResourceId::new(), - AcceptanceSequence::new(1), - MachineId::new(), - command_spec(&["ssh", "gpu-host"]), - ) - .unwrap(); - let spawn_failed = ExitReason::SpawnFailed { - message: "fake spawn failure".into(), - }; - // no child ran, so the command shape cannot have left work behind - assert_eq!( - resource_task_release_proof(&request, &spawn_failed, WorkExitEvidence::NoWorkStarted), - Ok(ResourceTaskReleaseProof::NoChildSpawnedAfterSpawnFailure) - ); - assert_eq!( - resource_task_release_proof( - &request, - &ExitReason::Cancelled, - WorkExitEvidence::NoWorkStarted - ), - Ok(ResourceTaskReleaseProof::NoChildSpawnedAfterQueuedCancel) - ); - for outcome in [ - ExitReason::Exit { code: 0 }, - ExitReason::Signal { signal: 9 }, - ] { - assert_eq!( - resource_task_release_proof(&request, &outcome, WorkExitEvidence::NoWorkStarted), - Err(AssignedResourceTaskAttention::InvalidNoChildSpawnEvidence) - ); - } -} - -#[test] -fn assigned_task_release_proof_rejects_commands_that_can_outlive_the_group() { - for (command, risk) in [ - ( - &["ssh", "gpu-host"][..], - ResourceTaskOwnershipRisk::RemoteShell, - ), - ( - &["docker", "run"][..], - ResourceTaskOwnershipRisk::ContainerClient, - ), - ( - &["sh", "-c", "bench"][..], - ResourceTaskOwnershipRisk::ShellWrapper, - ), - ( - &["setsid", "bench"][..], - ResourceTaskOwnershipRisk::DetachedLauncher, - ), - // wrappers whose own exit hides a detached container or other launch - ( - &["env", "docker", "run", "-d", "gpu"][..], - ResourceTaskOwnershipRisk::ProgramLauncher, - ), - ( - &["sudo", "/opt/gpu/bench"][..], - ResourceTaskOwnershipRisk::ProgramLauncher, - ), - ( - &["timeout", "1h", "bench"][..], - ResourceTaskOwnershipRisk::ProgramLauncher, - ), - ( - &["/opt/tools/runner", "docker", "run", "-d"][..], - ResourceTaskOwnershipRisk::ContainerClient, - ), - ] { - // a request saved before the contract existed still cannot release - let request = ResourceRequest::new( - RequestId::new(), - TaskId::new(), - ResourceId::new(), - AcceptanceSequence::new(1), - MachineId::new(), - command_spec(command), - ) - .unwrap(); - assert_eq!( - resource_task_release_proof( - &request, - &ExitReason::Exit { code: 0 }, - WorkExitEvidence::ProcessGroupExited, - ), - Err(AssignedResourceTaskAttention::OwnershipUncertain(risk)) - ); - } -} - -fn agent_spec() -> NormalizedSpec { - serde_json::from_value(json!({ - "api_version": 1, - "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "agent request", - "cwd": "/tmp", - "timeout": "4h", - "workload": { - "type": "agent", - "agent": "codex", - "prompt": "run this agent", - "report_trailer": true - } - })) - .unwrap() -} - -fn request_id_at(timestamp_ms: u128) -> RequestId { - RequestId(Uuid::from_u128( - (timestamp_ms << 80) | (0x7 << 76) | (0x2 << 62), - )) -} - -fn insert_loan( - conn: &Connection, - loan_id: LoanId, - resource_id: ResourceId, - state: LoanState, -) -> rusqlite::Result<()> { - conn.execute( - "INSERT INTO loans (id, resource_id, state_json) VALUES (?1, ?2, ?3)", - params![ - loan_id.as_uuid().to_string(), - resource_id.as_uuid().to_string(), - serde_json::to_string(&state).unwrap(), - ], - )?; - Ok(()) -} - -fn assign_request_to_serving_loan( - conn: &Connection, - request: &ResourceRequest, - return_context: ReturnContext, -) -> Loan { - let loan_id = LoanId::new(); - let assigned_state = ResourceRequestState::Assigned { loan_id }; - conn.execute( - "UPDATE resource_requests SET state_json = ?1 WHERE request_id = ?2", - params![ - serde_json::to_string(&assigned_state).unwrap(), - request.request_id.0.to_string(), - ], - ) - .unwrap(); - let loan = Loan { - id: loan_id, - resource_id: request.resource_id, - state: LoanState::Active { - phase: LoanPhase::Serving { - return_context, - current_request_id: request.request_id, - release_provenance: crate::store::unreceipted_release_provenance(), - }, - }, - }; - insert_loan(conn, loan.id, loan.resource_id, loan.state.clone()).unwrap(); - loan -} - -fn insert_accepted_executor_identity( - conn: &Connection, - request: &ResourceRequest, - authority: MachineId, -) { - let identity = ExecutorIdentity::Accepted(ExecutionRecord { - task: request.task_id, - origin_machine: request.origin_machine, - execution_machine: authority, - spec: request.spec().as_normalized().clone().into(), - state: crate::domain::ProcessStatus::Queued, - }); - conn.execute( - "INSERT INTO executor_identities (task_id, origin_machine, identity_json) - VALUES (?1, ?2, ?3)", - params![ - request.task_id.to_string(), - request.origin_machine.as_uuid().to_string(), - serde_json::to_string(&identity).unwrap(), - ], - ) - .unwrap(); -} - -fn connection_at(path: &Path) -> Connection { - let mut conn = Connection::open(path).unwrap(); - conn.pragma_update(None, "foreign_keys", "ON").unwrap(); - install_schema(&mut conn).unwrap(); - conn -} - -fn persist_notice_for_test( - conn: &mut Connection, - notice: &SupervisorNotice, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let saved = insert_supervisor_notice_in_transaction(&tx, notice)?; - tx.commit()?; - Ok(saved) -} - -fn insert_notice_fixture(conn: &mut Connection) -> SupervisorNotice { - let resource = resource(); - register_resource(conn, &resource).unwrap(); - let loan_id = LoanId::new(); - let action_id = ActionId::new(); - let task_id = TaskId::new(); - let state = LoanState::Active { - phase: LoanPhase::AwaitingRelease { - action_id, - observed_background_task: task_id, - watcher_intent: None, - }, - }; - insert_loan(conn, loan_id, resource.id, state).unwrap(); - - SupervisorNotice { - id: NoticeId::new(), - loan_id, - action_id, - state_revision: resource.state_revision, - destination: resource.supervisor, - assignment_revision: resource.assignment_revision, - payload: SupervisorNoticePayload::ReleaseRequired { task_id }, - delivery: SupervisorNoticeDelivery::Pending { attempts: 0 }, - } -} - -#[test] -fn acceptance_sequence_is_not_request_uuid_time() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let origin = MachineId::new(); - let earlier_uuid = request_id_at(1); - let later_uuid = request_id_at(2); - - let accepted_later_uuid = accept_request( - &mut conn, - later_uuid, - TaskId::new(), - resource.id, - origin, - command_spec(&["echo", "later uuid"]), - ) - .unwrap(); - let accepted_earlier_uuid = accept_request( - &mut conn, - earlier_uuid, - TaskId::new(), - resource.id, - origin, - command_spec(&["echo", "earlier uuid"]), - ) - .unwrap(); - - assert!(earlier_uuid.0 < later_uuid.0); - assert!(accepted_later_uuid.acceptance_sequence < accepted_earlier_uuid.acceptance_sequence); - let serving_order = requests_for_resource(&conn, resource.id).unwrap(); - assert_eq!(serving_order[0].request_id, later_uuid); - assert_eq!(serving_order[1].request_id, earlier_uuid); - assert_eq!( - next_queued_request(&conn, resource.id) - .unwrap() - .unwrap() - .request_id, - later_uuid - ); -} - -#[test] -fn cancellation_before_acceptance_prevents_a_delayed_acceptance() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let request_id = RequestId::new(); - let task_id = TaskId::new(); - let origin = MachineId::new(); - - assert!(matches!( - cancel_request_before_activation(&mut conn, request_id, task_id, resource.id, origin,), - Ok(QueueCancellationResult::PreventedBeforeAcceptance) - )); - assert!(matches!( - cancel_request_before_activation(&mut conn, request_id, task_id, resource.id, origin,), - Ok(QueueCancellationResult::PreventedBeforeAcceptance) - )); - assert!(matches!( - accept_request( - &mut conn, - request_id, - task_id, - resource.id, - origin, - command_spec(&["echo", "delayed"]), - ), - Err(ResourceStoreError::Prevented) - )); - - let request_count: i64 = conn - .query_row("SELECT COUNT(*) FROM resource_requests", [], |row| { - row.get(0) - }) - .unwrap(); - let prevention_count: i64 = conn - .query_row( - "SELECT COUNT(*) FROM resource_request_preventions", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(request_count, 0); - assert_eq!(prevention_count, 1); - - let first_accepted = accept_request( - &mut conn, - RequestId::new(), - TaskId::new(), - resource.id, - origin, - command_spec(&["echo", "next"]), - ) - .unwrap(); - assert_eq!( - first_accepted.acceptance_sequence, - AcceptanceSequence::new(1) - ); -} - -#[test] -fn cancellation_after_acceptance_retains_one_cancelled_row_and_its_sequence() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let request_id = RequestId::new(); - let task_id = TaskId::new(); - let origin = MachineId::new(); - let spec = command_spec(&["echo", "queued"]); - let accepted = accept_request( - &mut conn, - request_id, - task_id, - resource.id, - origin, - spec.clone(), - ) - .unwrap(); - - let first_cancel = - cancel_request_before_activation(&mut conn, request_id, task_id, resource.id, origin) - .unwrap(); - let QueueCancellationResult::Request(cancelled) = first_cancel else { - panic!("an accepted request must retain its queue row"); - }; - assert!(matches!( - cancelled.state, - ResourceRequestState::CancelledBeforeLaunch - )); - assert_eq!(cancelled.acceptance_sequence, accepted.acceptance_sequence); - - let retry = - cancel_request_before_activation(&mut conn, request_id, task_id, resource.id, origin) - .unwrap(); - let QueueCancellationResult::Request(retried) = retry else { - panic!("a repeated cancellation must return the saved request"); - }; - assert_eq!(retried.acceptance_sequence, accepted.acceptance_sequence); - assert!(matches!( - retried.state, - ResourceRequestState::CancelledBeforeLaunch - )); - - let accepted_retry = - accept_request(&mut conn, request_id, task_id, resource.id, origin, spec).unwrap(); - assert_eq!( - accepted_retry.acceptance_sequence, - accepted.acceptance_sequence - ); - assert!(matches!( - accepted_retry.state, - ResourceRequestState::CancelledBeforeLaunch - )); - assert_eq!(requests_for_resource(&conn, resource.id).unwrap().len(), 1); - assert!(next_queued_request(&conn, resource.id).unwrap().is_none()); - - let prevention_count: i64 = conn - .query_row( - "SELECT COUNT(*) FROM resource_request_preventions", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(prevention_count, 0); -} - -#[test] -fn cancellation_rejects_reuse_of_either_prevented_identity() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let request_id = RequestId::new(); - let task_id = TaskId::new(); - let origin = MachineId::new(); - cancel_request_before_activation(&mut conn, request_id, task_id, resource.id, origin).unwrap(); - - assert!(matches!( - cancel_request_before_activation(&mut conn, request_id, TaskId::new(), resource.id, origin,), - Err(ResourceStoreError::Conflict( - ConflictReason::PreventionIdentityMismatch - )) - )); - assert!(matches!( - cancel_request_before_activation(&mut conn, RequestId::new(), task_id, resource.id, origin,), - Err(ResourceStoreError::Conflict( - ConflictReason::PreventionIdentityMismatch - )) - )); - assert!(matches!( - cancel_request_before_activation(&mut conn, request_id, task_id, ResourceId::new(), origin,), - Err(ResourceStoreError::Conflict( - ConflictReason::PreventionIdentityMismatch - )) - )); - assert!(matches!( - cancel_request_before_activation( - &mut conn, - request_id, - task_id, - resource.id, - MachineId::new(), - ), - Err(ResourceStoreError::Conflict( - ConflictReason::PreventionIdentityMismatch - )) - )); - assert!(matches!( - accept_request( - &mut conn, - request_id, - TaskId::new(), - resource.id, - origin, - command_spec(&["echo", "conflict"]), - ), - Err(ResourceStoreError::Conflict( - ConflictReason::PreventionIdentityMismatch - )) - )); - assert!(matches!( - accept_request( - &mut conn, - RequestId::new(), - task_id, - resource.id, - origin, - command_spec(&["echo", "conflict"]), - ), - Err(ResourceStoreError::Conflict( - ConflictReason::PreventionIdentityMismatch - )) - )); - assert!(matches!( - accept_request( - &mut conn, - request_id, - task_id, - ResourceId::new(), - origin, - command_spec(&["echo", "conflict"]), - ), - Err(ResourceStoreError::Conflict( - ConflictReason::PreventionIdentityMismatch - )) - )); - assert!(matches!( - accept_request( - &mut conn, - request_id, - task_id, - resource.id, - MachineId::new(), - command_spec(&["echo", "conflict"]), - ), - Err(ResourceStoreError::Conflict( - ConflictReason::PreventionIdentityMismatch - )) - )); - - let request_count: i64 = conn - .query_row("SELECT COUNT(*) FROM resource_requests", [], |row| { - row.get(0) - }) - .unwrap(); - let prevention_count: i64 = conn - .query_row( - "SELECT COUNT(*) FROM resource_request_preventions", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(request_count, 0); - assert_eq!(prevention_count, 1); -} - -#[test] -fn queue_selection_skips_a_cancelled_request() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let origin = MachineId::new(); - let cancelled = accept_request( - &mut conn, - RequestId::new(), - TaskId::new(), - resource.id, - origin, - command_spec(&["echo", "cancelled"]), - ) - .unwrap(); - let ready = accept_request( - &mut conn, - RequestId::new(), - TaskId::new(), - resource.id, - origin, - command_spec(&["echo", "ready"]), - ) - .unwrap(); - - cancel_request_before_activation( - &mut conn, - cancelled.request_id, - cancelled.task_id, - cancelled.resource_id, - cancelled.origin_machine, - ) - .unwrap(); - - let selected = next_queued_request(&conn, resource.id).unwrap().unwrap(); - assert_eq!(selected.request_id, ready.request_id); - assert_eq!(selected.acceptance_sequence, ready.acceptance_sequence); - let saved = requests_for_resource(&conn, resource.id).unwrap(); - assert_eq!(saved.len(), 2); - assert!(matches!( - saved[0].state, - ResourceRequestState::CancelledBeforeLaunch - )); - assert!(matches!(saved[1].state, ResourceRequestState::Queued)); -} - -#[test] -fn cancellation_preserves_assigned_state_when_executor_won_and_terminal_states() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let origin = MachineId::new(); - let assigned = accept_request( - &mut conn, - RequestId::new(), - TaskId::new(), - resource.id, - origin, - command_spec(&["echo", "assigned"]), - ) - .unwrap(); - let loan = assign_request_to_serving_loan( - &conn, - &assigned, - ReturnContext::AlreadyCompleted { - task_id: TaskId::new(), - result_ref: "saved-result".into(), - }, - ); - insert_accepted_executor_identity(&conn, &assigned, resource.authority_machine()); - - let result = cancel_request_before_activation( - &mut conn, - assigned.request_id, - assigned.task_id, - assigned.resource_id, - assigned.origin_machine, - ); - assert!(matches!( - result, - Err(ResourceStoreError::ExecutorAlreadyAccepted { task }) if task == assigned.task_id - )); - let saved_assigned = select_request_by_id(&conn, assigned.request_id) - .unwrap() - .unwrap(); - assert!(matches!( - saved_assigned.state, - ResourceRequestState::Assigned { loan_id: saved } if saved == loan.id - )); - assert_eq!( - select_non_closed_loan(&conn, resource.id).unwrap(), - Some(loan.clone()) - ); - assert_eq!( - select_resource(&conn, resource.id) - .unwrap() - .unwrap() - .state_revision, - resource.state_revision - ); - conn.execute( - "DELETE FROM executor_identities WHERE task_id = ?1", - [assigned.task_id.to_string()], - ) - .unwrap(); - insert_local_task_status(&conn, assigned.task_id, "queued"); - assert!(matches!( - cancel_request_before_activation( - &mut conn, - assigned.request_id, - assigned.task_id, - assigned.resource_id, - assigned.origin_machine, - ), - Err(ResourceStoreError::Conflict( - ConflictReason::TaskIdentityInUse - )) - )); - assert!(matches!( - select_request_by_id(&conn, assigned.request_id) - .unwrap() - .unwrap() - .state, - ResourceRequestState::Assigned { loan_id } if loan_id == loan.id - )); - assert_eq!( - select_non_closed_loan(&conn, resource.id).unwrap(), - Some(loan.clone()) - ); - let notice_count: i64 = conn - .query_row( - "SELECT COUNT(*) FROM resource_supervisor_notices", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(notice_count, 0); - - let finished = accept_request( - &mut conn, - RequestId::new(), - TaskId::new(), - resource.id, - origin, - command_spec(&["echo", "finished"]), - ) - .unwrap(); - let finished_state = ResourceRequestState::Finished { - outcome: ExitReason::Cancelled, - }; - conn.execute( - "UPDATE resource_requests SET state_json = ?1 WHERE request_id = ?2", - params![ - serde_json::to_string(&finished_state).unwrap(), - finished.request_id.0.to_string(), - ], - ) - .unwrap(); - - let result = cancel_request_before_activation( - &mut conn, - finished.request_id, - finished.task_id, - finished.resource_id, - finished.origin_machine, - ) - .unwrap(); - let QueueCancellationResult::Request(saved_finished) = result else { - panic!("a terminal request must remain in the queue store"); - }; - assert!(matches!( - saved_finished.state, - ResourceRequestState::Finished { - outcome: ExitReason::Cancelled - } - )); -} - -#[test] -fn assigned_cancellation_assigns_next_queued_request_to_the_same_loan() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let cancelled = queue_request(&mut conn, resource.id); - let next = queue_request(&mut conn, resource.id); - let later = queue_request(&mut conn, resource.id); - let return_context = ReturnContext::Stopped { - task_id: TaskId::new(), - checkpoint_ref: "checkpoint-9".into(), - recovery_ref: "recovery-9".into(), - }; - let loan = assign_request_to_serving_loan(&conn, &cancelled, return_context.clone()); - let LoanState::Active { - phase: LoanPhase::Serving { - release_provenance, .. - }, - } = &loan.state - else { - panic!("the fixture loan must be Serving"); - }; - let release_provenance = release_provenance.clone(); - - let result = cancel_request_before_activation_for_authority( - &mut conn, - resource.authority_machine(), - cancelled.request_id, - cancelled.task_id, - cancelled.resource_id, - cancelled.origin_machine, - ) - .unwrap(); - let QueueCancellationResult::Request(cancelled) = result else { - panic!("an assigned request must retain its row"); - }; - assert!(matches!( - cancelled.state, - ResourceRequestState::CancelledBeforeLaunch - )); - - let saved = requests_for_resource(&conn, resource.id).unwrap(); - assert!(matches!( - saved[1].state, - ResourceRequestState::Assigned { loan_id } if loan_id == loan.id - )); - assert_eq!(saved[1].request_id, next.request_id); - assert!(matches!(saved[2].state, ResourceRequestState::Queued)); - assert_eq!(saved[2].request_id, later.request_id); - let saved_loan = select_non_closed_loan(&conn, resource.id).unwrap().unwrap(); - assert_eq!( - saved_loan.state, - LoanState::Active { - phase: LoanPhase::Serving { - return_context, - current_request_id: next.request_id, - release_provenance, - }, - } - ); - assert_eq!(saved_loan.id, loan.id); - assert_eq!( - select_resource(&conn, resource.id) - .unwrap() - .unwrap() - .state_revision, - ResourceRevision::new(1) - ); - let identity_json: String = conn - .query_row( - "SELECT identity_json FROM executor_identities WHERE task_id = ?1", - [cancelled.task_id.to_string()], - |row| row.get(0), - ) - .unwrap(); - assert!(matches!( - serde_json::from_str::(&identity_json).unwrap(), - ExecutorIdentity::Rejected(RejectionTombstone { task, .. }) if task == cancelled.task_id - )); -} - -#[test] -fn assigned_cancellation_with_empty_queue_persists_return_notice() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let assigned = queue_request(&mut conn, resource.id); - let return_context = ReturnContext::AlreadyCompleted { - task_id: TaskId::new(), - result_ref: "final-result-51".into(), - }; - let loan = assign_request_to_serving_loan(&conn, &assigned, return_context.clone()); - - let result = cancel_request_before_activation_for_authority( - &mut conn, - resource.authority_machine(), - assigned.request_id, - assigned.task_id, - assigned.resource_id, - assigned.origin_machine, - ) - .unwrap(); - assert!(matches!( - result, - QueueCancellationResult::Request(request) - if matches!(request.state, ResourceRequestState::CancelledBeforeLaunch) - )); - - let saved_loan = select_non_closed_loan(&conn, resource.id).unwrap().unwrap(); - let LoanState::Active { - phase: - LoanPhase::AwaitingReturn { - action_id, - return_context: saved_context, - }, - } = saved_loan.state - else { - panic!("an empty queue must reserve the return decision"); - }; - assert_eq!(saved_loan.id, loan.id); - assert_eq!(saved_context, return_context); - assert_eq!( - select_resource(&conn, resource.id) - .unwrap() - .unwrap() - .state_revision, - ResourceRevision::new(1) - ); - let notice = select_supervisor_notice_record_by_action(&conn, action_id) - .unwrap() - .unwrap() - .0; - assert_eq!(notice.loan_id, loan.id); - assert_eq!(notice.action_id, action_id); - assert_eq!(notice.state_revision, ResourceRevision::new(1)); - assert_eq!(notice.destination, resource.supervisor); - assert_eq!(notice.assignment_revision, resource.assignment_revision); - assert_eq!( - notice.payload, - SupervisorNoticePayload::ReturnRequired { return_context } - ); - assert_eq!( - notice.delivery, - SupervisorNoticeDelivery::Pending { attempts: 0 } - ); -} - -#[test] -fn retrying_assigned_cancellation_does_not_advance_queue_or_add_a_notice() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let assigned = queue_request(&mut conn, resource.id); - let return_context = ReturnContext::AlreadyCompleted { - task_id: TaskId::new(), - result_ref: "final-result-52".into(), - }; - assign_request_to_serving_loan(&conn, &assigned, return_context); - let first = cancel_request_before_activation( - &mut conn, - assigned.request_id, - assigned.task_id, - assigned.resource_id, - assigned.origin_machine, - ) - .unwrap(); - let QueueCancellationResult::Request(first_request) = first else { - panic!("an assigned request must retain its row"); - }; - assert!(matches!( - first_request.state, - ResourceRequestState::CancelledBeforeLaunch - )); - let first_loan = select_non_closed_loan(&conn, resource.id).unwrap().unwrap(); - let first_notice_count: i64 = conn - .query_row( - "SELECT COUNT(*) FROM resource_supervisor_notices", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(first_notice_count, 1); - - let later = queue_request(&mut conn, resource.id); - let retry = cancel_request_before_activation( - &mut conn, - assigned.request_id, - assigned.task_id, - assigned.resource_id, - assigned.origin_machine, - ) - .unwrap(); - let QueueCancellationResult::Request(retried_request) = retry else { - panic!("a repeated cancellation must retain the saved row"); - }; - assert!(matches!( - retried_request.state, - ResourceRequestState::CancelledBeforeLaunch - )); - assert_eq!( - select_non_closed_loan(&conn, resource.id).unwrap().unwrap(), - first_loan - ); - assert!(matches!( - select_request_by_id(&conn, later.request_id) - .unwrap() - .unwrap() - .state, - ResourceRequestState::Queued - )); - let notice_count: i64 = conn - .query_row( - "SELECT COUNT(*) FROM resource_supervisor_notices", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(notice_count, 1); - assert_eq!( - select_resource(&conn, resource.id) - .unwrap() - .unwrap() - .state_revision, - ResourceRevision::new(1) - ); -} - -#[test] -fn assigned_cancellation_refuses_a_loan_needing_attention() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let assigned = queue_request(&mut conn, resource.id); - let return_context = ReturnContext::AlreadyCompleted { - task_id: TaskId::new(), - result_ref: "final-result-53".into(), - }; - let loan = assign_request_to_serving_loan(&conn, &assigned, return_context.clone()); - let attention_state = LoanState::NeedsAttention { - action_id: ActionId::new(), - last_safe_phase: LoanPhase::Serving { - return_context, - current_request_id: assigned.request_id, - release_provenance: crate::store::unreceipted_release_provenance(), - }, - reason: "executor state is uncertain".into(), - }; - conn.execute( - "UPDATE loans SET state_json = ?1 WHERE id = ?2", - params![ - serde_json::to_string(&attention_state).unwrap(), - loan.id.as_uuid().to_string(), - ], - ) - .unwrap(); - - assert!(matches!( - cancel_request_before_activation( - &mut conn, - assigned.request_id, - assigned.task_id, - assigned.resource_id, - assigned.origin_machine, - ), - Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged)) - )); - assert!(matches!( - select_request_by_id(&conn, assigned.request_id) - .unwrap() - .unwrap() - .state, - ResourceRequestState::Assigned { loan_id } if loan_id == loan.id - )); - assert_eq!( - select_non_closed_loan(&conn, resource.id) - .unwrap() - .unwrap() - .state, - attention_state - ); - assert!(!executor_identity_exists(&conn, assigned.task_id).unwrap()); - assert_eq!( - select_resource(&conn, resource.id) - .unwrap() - .unwrap() - .state_revision, - resource.state_revision - ); -} - -#[test] -fn identical_retry_returns_saved_request_after_its_state_changes() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let request_id = RequestId::new(); - let task_id = TaskId::new(); - let origin = MachineId::new(); - let spec = command_spec(&["echo", "hello"]); - let first = accept_request( - &mut conn, - request_id, - task_id, - resource.id, - origin, - spec.clone(), - ) - .unwrap(); - let changed_state = ResourceRequestState::Assigned { - loan_id: LoanId::new(), - }; - conn.execute( - "UPDATE resource_requests SET state_json = ?1 WHERE request_id = ?2", - params![ - serde_json::to_string(&changed_state).unwrap(), - request_id.0.to_string(), - ], - ) - .unwrap(); - - let retry = accept_request(&mut conn, request_id, task_id, resource.id, origin, spec).unwrap(); - assert_eq!(retry.acceptance_sequence, first.acceptance_sequence); - assert_eq!(retry.state, changed_state); - assert_eq!(requests_for_resource(&conn, resource.id).unwrap().len(), 1); - assert!(next_queued_request(&conn, resource.id).unwrap().is_none()); -} - -#[test] -fn conflicting_retry_and_task_uuid_reuse_are_rejected() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let request_id = RequestId::new(); - let task_id = TaskId::new(); - let origin = MachineId::new(); - accept_request( - &mut conn, - request_id, - task_id, - resource.id, - origin, - command_spec(&["echo", "first"]), - ) - .unwrap(); - - assert!(matches!( - accept_request( - &mut conn, - request_id, - task_id, - resource.id, - origin, - command_spec(&["echo", "changed"]), - ), - Err(ResourceStoreError::Conflict( - ConflictReason::RequestSpecMismatch - )) - )); - assert!(matches!( - accept_request( - &mut conn, - RequestId::new(), - task_id, - resource.id, - origin, - command_spec(&["echo", "first"]), - ), - Err(ResourceStoreError::Conflict( - ConflictReason::TaskIdentityInUse - )) - )); - assert_eq!(requests_for_resource(&conn, resource.id).unwrap().len(), 1); -} - -#[test] -fn agent_workloads_are_rejected_before_a_request_is_written() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - - assert!(matches!( - accept_request( - &mut conn, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - agent_spec(), - ), - Err(ResourceStoreError::InvalidCommandSpec( - CommandSpecError::AgentWorkload - )) - )); - assert!( - requests_for_resource(&conn, resource.id) - .unwrap() - .is_empty() - ); -} - -#[test] -fn explicit_machine_is_rejected_before_a_request_is_written() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let mut spec = command_spec(&["echo", "gpu"]); - spec.machine = Some(MachineName::parse("other").unwrap()); - - assert!(matches!( - accept_request( - &mut conn, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - spec, - ), - Err(ResourceStoreError::InvalidCommandSpec( - CommandSpecError::ExplicitMachine - )) - )); - assert!( - requests_for_resource(&conn, resource.id) - .unwrap() - .is_empty() - ); -} - -#[test] -fn request_for_an_unknown_resource_returns_a_typed_error() { - let mut conn = connection(); - - assert!(matches!( - accept_request( - &mut conn, - RequestId::new(), - TaskId::new(), - ResourceId::new(), - MachineId::new(), - command_spec(&["echo", "hello"]), - ), - Err(ResourceStoreError::ResourceNotFound) - )); - let count: i64 = conn - .query_row("SELECT COUNT(*) FROM resource_requests", [], |row| { - row.get(0) - }) - .unwrap(); - assert_eq!(count, 0); -} - -#[test] -fn resource_registration_is_idempotent_and_does_not_replace_content() { - let mut conn = connection(); - let resource = resource(); - - assert_eq!(register_resource(&mut conn, &resource).unwrap(), resource); - assert_eq!(register_resource(&mut conn, &resource).unwrap(), resource); - let mut conflicting = resource.clone(); - conflicting.display_name = "replacement".into(); - assert!(matches!( - register_resource(&mut conn, &conflicting), - Err(ResourceStoreError::RegistrationConflict { .. }) - )); - let conflicting_authority = Resource::new( - resource.id, - resource.display_name.clone(), - MachineId::new(), - resource.supervisor, - resource.assignment_revision, - resource.state_revision, - resource.registered_background_task, - ); - assert!(matches!( - register_resource(&mut conn, &conflicting_authority), - Err(ResourceStoreError::RegistrationConflict { .. }) - )); - assert_eq!(select_resource(&conn, resource.id).unwrap(), Some(resource)); -} - -#[test] -fn partial_index_allows_only_one_non_closed_loan_per_resource() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let closed = LoanState::Closed { - result: LoanClosure::NoResume { - return_context: ReturnContext::Idle, - reason: "completed".into(), - }, - }; - insert_loan(&conn, LoanId::new(), resource.id, closed).unwrap(); - - let active = LoanState::Active { - phase: LoanPhase::AwaitingRelease { - action_id: ActionId::new(), - observed_background_task: TaskId::new(), - watcher_intent: None, - }, - }; - insert_loan(&conn, LoanId::new(), resource.id, active.clone()).unwrap(); - assert!(insert_loan(&conn, LoanId::new(), resource.id, active).is_err()); - let count: i64 = conn - .query_row( - "SELECT COUNT(*) FROM loans WHERE resource_id = ?1", - [resource.id.as_uuid().to_string()], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(count, 2); -} - -#[test] -fn authority_snapshot_restores_only_owned_non_closed_loans_in_stable_order() { - let mut conn = connection(); - let authority = MachineId::new(); - let other_authority = MachineId::new(); - let active_resource = resource_for_authority(authority); - let closed_resource = resource_for_authority(authority); - let empty_resource = resource_for_authority(authority); - let foreign_resource = resource_for_authority(other_authority); - - for resource in [ - &active_resource, - &closed_resource, - &empty_resource, - &foreign_resource, - ] { - register_resource(&mut conn, resource).unwrap(); - } - - let active_loan_id = LoanId::new(); - let active_state = LoanState::Active { - phase: LoanPhase::AwaitingRelease { - action_id: ActionId::new(), - observed_background_task: TaskId::new(), - watcher_intent: None, - }, - }; - let active_loan = Loan { - id: active_loan_id, - resource_id: active_resource.id, - state: active_state.clone(), - }; - insert_loan( - &conn, - active_loan_id, - active_resource.id, - active_loan.state.clone(), - ) - .unwrap(); - insert_loan( - &conn, - LoanId::new(), - closed_resource.id, - LoanState::Closed { - result: LoanClosure::NoResume { - return_context: ReturnContext::Idle, - reason: "completed".into(), - }, - }, - ) - .unwrap(); - insert_loan(&conn, LoanId::new(), foreign_resource.id, active_state).unwrap(); - - let snapshots = resources_for_authority(&conn, authority).unwrap(); - let mut expected_ids = vec![active_resource.id, closed_resource.id, empty_resource.id]; - expected_ids.sort_by_key(ResourceId::as_uuid); - assert_eq!( - snapshots - .iter() - .map(|snapshot| snapshot.resource.id) - .collect::>(), - expected_ids - ); - assert_eq!( - snapshots - .iter() - .find(|snapshot| snapshot.resource.id == active_resource.id) - .unwrap() - .loan, - Some(active_loan) - ); - for resource_id in [closed_resource.id, empty_resource.id] { - assert_eq!( - snapshots - .iter() - .find(|snapshot| snapshot.resource.id == resource_id) - .unwrap() - .loan, - None - ); - } - assert!( - snapshots - .iter() - .all(|snapshot| snapshot.resource.id != foreign_resource.id) - ); - - let malformed_resource = resource_for_authority(authority); - register_resource(&mut conn, &malformed_resource).unwrap(); - conn.execute( - "INSERT INTO loans (id, resource_id, state_json) VALUES (?1, ?2, ?3)", - params![ - LoanId::new().as_uuid().to_string(), - malformed_resource.id.as_uuid().to_string(), - r#"{"type":"closed"}"#, - ], - ) - .unwrap(); - - assert!(matches!( - resources_for_authority(&conn, authority), - Err(ResourceStoreError::CorruptRecord { .. }) - )); -} - -#[test] -fn release_loan_requires_a_queued_request() { - let mut conn = connection(); - let background_task = TaskId::new(); - let resource = resource_with_background_task(background_task); - register_resource(&mut conn, &resource).unwrap(); - insert_local_task_status(&conn, background_task, "running"); - - assert!(matches!( - open_release_loan_for_authority( - &mut conn, - resource.authority_machine(), - resource.id, - resource.state_revision, - ), - Err(OpenReleaseLoanError::NoQueuedRequest) - )); - assert_eq!(select_resource(&conn, resource.id).unwrap(), Some(resource)); - let loan_count: i64 = conn - .query_row("SELECT COUNT(*) FROM loans", [], |row| row.get(0)) - .unwrap(); - assert_eq!(loan_count, 0); -} - -#[test] -fn queue_reconciliation_opens_one_release_action_and_keeps_requests_queued_in_order() { - let mut conn = connection(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let mut resource = resource_for_authority(authority); - resource.registered_background_task = Some(background_task); - register_resource_for_authority(&mut conn, authority, &resource).unwrap(); - insert_local_task_status(&conn, background_task, "running"); - - let first = accept_request_for_authority( - &mut conn, - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - command_spec(&["echo", "first"]), - ) - .unwrap(); - let second = accept_request_for_authority( - &mut conn, - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - command_spec(&["echo", "second"]), - ) - .unwrap(); - - let outcome = - reconcile_resource_queue_for_authority(&mut conn, authority, resource.id).unwrap(); - let ResourceQueueReconcileOutcome::ReleaseRequired { loan, notice } = outcome else { - panic!("running background work must reserve one release action"); - }; - assert_eq!(notice.loan_id, loan.id); - assert_eq!( - loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { - action_id: notice.action_id, - observed_background_task: background_task, - watcher_intent: None, - }, - } - ); - - let repeated = - reconcile_resource_queue_for_authority(&mut conn, authority, resource.id).unwrap(); - assert!(matches!( - repeated, - ResourceQueueReconcileOutcome::LoanAlreadyActive { loan: repeated_loan } - if repeated_loan.id == loan.id && repeated_loan.state == loan.state - )); - assert_eq!( - select_non_closed_loan(&conn, resource.id).unwrap(), - Some(loan.clone()) - ); - assert_eq!( - conn.query_row( - "SELECT COUNT(*) FROM resource_supervisor_notices", - [], - |row| row.get::<_, i64>(0), - ) - .unwrap(), - 1 - ); - let requests = requests_for_resource_for_authority(&conn, authority, resource.id).unwrap(); - assert_eq!(requests[0].request_id, first.request_id); - assert_eq!(requests[1].request_id, second.request_id); - assert!( - requests - .iter() - .all(|request| matches!(request.state, ResourceRequestState::Queued)) - ); -} - -#[test] -fn queue_reconciliation_reports_queued_work_and_does_not_infer_idle_from_missing_background() { - let mut conn = connection(); - let authority = MachineId::new(); - let resource = resource_for_authority(authority); - register_resource_for_authority(&mut conn, authority, &resource).unwrap(); - let first = accept_request_for_authority( - &mut conn, - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - command_spec(&["echo", "first"]), - ) - .unwrap(); - let second = accept_request_for_authority( - &mut conn, - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - command_spec(&["echo", "second"]), - ) - .unwrap(); - - let outcome = - reconcile_resource_queue_for_authority(&mut conn, authority, resource.id).unwrap(); - assert!(matches!( - outcome, - ResourceQueueReconcileOutcome::AttentionRequired { - request, - reason: ResourceQueueAttentionReason::IdleNotProven { - gap: crate::resource::IdleProofGap::NoIdleEvidence, - }, - } if request.request_id == first.request_id - )); - assert!( - select_non_closed_loan(&conn, resource.id) - .unwrap() - .is_none() - ); - let requests = requests_for_resource_for_authority(&conn, authority, resource.id).unwrap(); - assert_eq!(requests[0].request_id, first.request_id); - assert_eq!(requests[1].request_id, second.request_id); - assert!( - requests - .iter() - .all(|request| matches!(request.state, ResourceRequestState::Queued)) - ); -} - -#[test] -fn queue_reconciliation_requires_attention_for_missing_or_uncertain_background_state() { - for (task_status, expected_reason) in [ - ( - None, - ResourceQueueAttentionReason::BackgroundTaskMissing { - task_id: TaskId::new(), - }, - ), - ( - Some("lost"), - ResourceQueueAttentionReason::BackgroundTaskNotRunning { - task_id: TaskId::new(), - state: "lost".into(), - }, - ), - ] { - let mut conn = connection(); - let authority = MachineId::new(); - let background_task = match &expected_reason { - ResourceQueueAttentionReason::BackgroundTaskMissing { task_id } - | ResourceQueueAttentionReason::BackgroundTaskNotRunning { task_id, .. } => *task_id, - ResourceQueueAttentionReason::IdleNotProven { .. } - | ResourceQueueAttentionReason::BackgroundLaunchPending { .. } - | ResourceQueueAttentionReason::AcceptedTaskLaunchUncertain { .. } - | ResourceQueueAttentionReason::UnverifiedServingRelease - | ResourceQueueAttentionReason::ReleaseProofUnavailable { .. } - | ResourceQueueAttentionReason::AssignedTaskLaunchUncertain { .. } - | ResourceQueueAttentionReason::AssignedTaskLost { .. } - | ResourceQueueAttentionReason::AssignedTaskExitUnconfirmed { .. } - | ResourceQueueAttentionReason::AssignedTaskIdentityMismatch { .. } - | ResourceQueueAttentionReason::AssignedTaskNoChildSpawnProofInvalid { .. } - | ResourceQueueAttentionReason::AssignedTaskOwnershipUncertain { .. } - | ResourceQueueAttentionReason::AssignedTaskStaleRevision { .. } - | ResourceQueueAttentionReason::AssignedTaskReconcileFailed { .. } => { - unreachable!() - } - }; - let mut resource = resource_for_authority(authority); - resource.registered_background_task = Some(background_task); - register_resource_for_authority(&mut conn, authority, &resource).unwrap(); - if let Some(status) = task_status { - insert_local_task_status(&conn, background_task, status); - } - let request = accept_request_for_authority( - &mut conn, - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - command_spec(&["echo", "queued"]), - ) - .unwrap(); - - let outcome = - reconcile_resource_queue_for_authority(&mut conn, authority, resource.id).unwrap(); - assert!(matches!( - outcome, - ResourceQueueReconcileOutcome::AttentionRequired { - request: saved_request, - reason, - } if saved_request.request_id == request.request_id - && reason == expected_reason - )); - assert!( - select_non_closed_loan(&conn, resource.id) - .unwrap() - .is_none() - ); - assert!(matches!( - next_queued_request_for_authority(&conn, authority, resource.id) - .unwrap() - .unwrap() - .state, - ResourceRequestState::Queued - )); - } -} - -#[test] -fn release_loan_rejects_the_wrong_authority() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - - assert!(matches!( - open_release_loan_for_authority( - &mut conn, - MachineId::new(), - resource.id, - resource.state_revision, - ), - Err(OpenReleaseLoanError::Resource( - ResourceStoreError::WrongAuthority { expected, found } - )) if expected == resource.authority_machine() && found != expected - )); -} - -#[test] -fn release_loan_rejects_a_stale_resource_revision() { - let mut conn = connection(); - let background_task = TaskId::new(); - let resource = resource_with_background_task(background_task); - register_resource(&mut conn, &resource).unwrap(); - queue_request(&mut conn, resource.id); - insert_local_task_status(&conn, background_task, "running"); - - assert!(matches!( - open_release_loan_for_authority( - &mut conn, - resource.authority_machine(), - resource.id, - ResourceRevision::new(1), - ), - Err(OpenReleaseLoanError::StaleRevision { - expected, - actual, - }) - if expected == ResourceRevision::new(1) - && actual == ResourceRevision::new(0) - )); - assert!( - select_non_closed_loan(&conn, resource.id) - .unwrap() - .is_none() - ); -} - -#[test] -fn release_loan_requires_a_registered_background_task() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - queue_request(&mut conn, resource.id); - - assert!(matches!( - open_release_loan_for_authority( - &mut conn, - resource.authority_machine(), - resource.id, - resource.state_revision, - ), - Err(OpenReleaseLoanError::BackgroundTaskNotRegistered) - )); - assert!( - select_non_closed_loan(&conn, resource.id) - .unwrap() - .is_none() - ); -} - -#[test] -fn release_loan_requires_the_registered_background_task_row() { - let mut conn = connection(); - let background_task = TaskId::new(); - let resource = resource_with_background_task(background_task); - register_resource(&mut conn, &resource).unwrap(); - queue_request(&mut conn, resource.id); - - assert!(matches!( - open_release_loan_for_authority( - &mut conn, - resource.authority_machine(), - resource.id, - resource.state_revision, - ), - Err(OpenReleaseLoanError::BackgroundTaskMissing { task_id }) - if task_id == background_task - )); - assert!( - select_non_closed_loan(&conn, resource.id) - .unwrap() - .is_none() - ); -} - -#[test] -fn release_loan_requires_the_registered_background_task_to_be_running() { - let mut conn = connection(); - let background_task = TaskId::new(); - let resource = resource_with_background_task(background_task); - register_resource(&mut conn, &resource).unwrap(); - queue_request(&mut conn, resource.id); - insert_local_task_status(&conn, background_task, "queued"); - - assert!(matches!( - open_release_loan_for_authority( - &mut conn, - resource.authority_machine(), - resource.id, - resource.state_revision, - ), - Err(OpenReleaseLoanError::BackgroundTaskNotRunning { task_id, state }) - if task_id == background_task && state == "queued" - )); - assert!( - select_non_closed_loan(&conn, resource.id) - .unwrap() - .is_none() - ); -} - -#[test] -fn release_loan_opens_with_one_matching_pending_notice_and_revision() { - let mut conn = connection(); - let background_task = TaskId::new(); - let resource = resource_with_background_task(background_task); - register_resource(&mut conn, &resource).unwrap(); - let request = queue_request(&mut conn, resource.id); - insert_local_task_status(&conn, background_task, "running"); - - let result = open_release_loan_for_authority( - &mut conn, - resource.authority_machine(), - resource.id, - resource.state_revision, - ) - .unwrap(); - let OpenReleaseLoanResult::Opened { loan, notice } = result else { - panic!("first call must open a new release loan"); - }; - let LoanState::Active { - phase: - LoanPhase::AwaitingRelease { - action_id, - observed_background_task, - .. - }, - } = loan.state - else { - panic!("new release loan must await release"); - }; - - assert_eq!(observed_background_task, background_task); - assert_eq!(notice.loan_id, loan.id); - assert_eq!(notice.action_id, action_id); - assert_eq!(notice.state_revision, ResourceRevision::new(1)); - assert_eq!(notice.destination, resource.supervisor); - assert_eq!(notice.assignment_revision, resource.assignment_revision); - assert_eq!( - notice.payload, - SupervisorNoticePayload::ReleaseRequired { - task_id: background_task, - } - ); - assert_eq!( - notice.delivery, - SupervisorNoticeDelivery::Pending { attempts: 0 } - ); - - let saved_resource = select_resource(&conn, resource.id).unwrap().unwrap(); - assert_eq!(saved_resource.state_revision, ResourceRevision::new(1)); - let saved_loan = select_non_closed_loan(&conn, resource.id).unwrap().unwrap(); - assert_eq!(saved_loan, loan); - assert_eq!( - next_queued_request(&conn, resource.id) - .unwrap() - .map(|queued| queued.request_id), - Some(request.request_id) - ); - let loan_count: i64 = conn - .query_row("SELECT COUNT(*) FROM loans", [], |row| row.get(0)) - .unwrap(); - let notice_count: i64 = conn - .query_row( - "SELECT COUNT(*) FROM resource_supervisor_notices", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(loan_count, 1); - assert_eq!(notice_count, 1); -} - -#[test] -fn release_loan_retry_returns_the_saved_action_after_queue_cancellation() { - let mut conn = connection(); - let background_task = TaskId::new(); - let resource = resource_with_background_task(background_task); - register_resource(&mut conn, &resource).unwrap(); - let request = queue_request(&mut conn, resource.id); - insert_local_task_status(&conn, background_task, "running"); - - let first = open_release_loan_for_authority( - &mut conn, - resource.authority_machine(), - resource.id, - resource.state_revision, - ) - .unwrap(); - let OpenReleaseLoanResult::Opened { - loan: first_loan, - notice: first_notice, - } = first - else { - panic!("first call must open a new release loan"); - }; - cancel_request_before_activation( - &mut conn, - request.request_id, - request.task_id, - request.resource_id, - request.origin_machine, - ) - .unwrap(); - conn.execute( - "UPDATE tasks SET status = 'succeeded' WHERE id = ?1", - [background_task.to_string()], - ) - .unwrap(); - - let retry = open_release_loan_for_authority( - &mut conn, - resource.authority_machine(), - resource.id, - resource.state_revision, - ) - .unwrap(); - assert_eq!( - retry, - OpenReleaseLoanResult::AlreadyAwaitingRelease { - loan: first_loan, - notice: first_notice, - } - ); - assert!(next_queued_request(&conn, resource.id).unwrap().is_none()); - let loan_count: i64 = conn - .query_row("SELECT COUNT(*) FROM loans", [], |row| row.get(0)) - .unwrap(); - let notice_count: i64 = conn - .query_row( - "SELECT COUNT(*) FROM resource_supervisor_notices", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(loan_count, 1); - assert_eq!(notice_count, 1); -} - -#[test] -fn release_loan_does_not_replace_another_non_closed_loan() { - let mut conn = connection(); - let background_task = TaskId::new(); - let resource = resource_with_background_task(background_task); - register_resource(&mut conn, &resource).unwrap(); - queue_request(&mut conn, resource.id); - insert_local_task_status(&conn, background_task, "running"); - let existing_loan = Loan { - id: LoanId::new(), - resource_id: resource.id, - state: LoanState::Active { - phase: LoanPhase::Serving { - return_context: ReturnContext::Idle, - current_request_id: RequestId::new(), - release_provenance: crate::store::unreceipted_release_provenance(), - }, - }, - }; - insert_loan( - &conn, - existing_loan.id, - resource.id, - existing_loan.state.clone(), - ) - .unwrap(); - - assert!(matches!( - open_release_loan_for_authority( - &mut conn, - resource.authority_machine(), - resource.id, - resource.state_revision, - ), - Err(OpenReleaseLoanError::ExistingLoan { loan }) if *loan == existing_loan - )); - assert_eq!( - select_resource(&conn, resource.id) - .unwrap() - .unwrap() - .state_revision, - resource.state_revision - ); - let loan_count: i64 = conn - .query_row("SELECT COUNT(*) FROM loans", [], |row| row.get(0)) - .unwrap(); - assert_eq!(loan_count, 1); -} - -#[test] -fn release_loan_notice_failure_rolls_back_loan_and_resource_revision() { - let mut conn = connection(); - let background_task = TaskId::new(); - let resource = resource_with_background_task(background_task); - register_resource(&mut conn, &resource).unwrap(); - let request = queue_request(&mut conn, resource.id); - insert_local_task_status(&conn, background_task, "running"); - conn.execute_batch( - "CREATE TRIGGER reject_release_notice - BEFORE INSERT ON resource_supervisor_notices - BEGIN - SELECT RAISE(ABORT, 'injected notice insert failure'); - END;", - ) - .unwrap(); - - assert!(matches!( - open_release_loan_for_authority( - &mut conn, - resource.authority_machine(), - resource.id, - resource.state_revision, - ), - Err(OpenReleaseLoanError::Notice( - SupervisorNoticeStoreError::Storage(_) - )) - )); - assert_eq!( - select_resource(&conn, resource.id) - .unwrap() - .unwrap() - .state_revision, - resource.state_revision - ); - assert!( - select_non_closed_loan(&conn, resource.id) - .unwrap() - .is_none() - ); - assert_eq!( - next_queued_request(&conn, resource.id) - .unwrap() - .map(|queued| queued.request_id), - Some(request.request_id) - ); - let loan_count: i64 = conn - .query_row("SELECT COUNT(*) FROM loans", [], |row| row.get(0)) - .unwrap(); - let notice_count: i64 = conn - .query_row( - "SELECT COUNT(*) FROM resource_supervisor_notices", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(loan_count, 0); - assert_eq!(notice_count, 0); -} - -#[test] -fn notice_identity_is_unique_per_action_and_retries_detect_conflicts() { - let mut conn = connection(); - let notice = insert_notice_fixture(&mut conn); - - let saved = persist_notice_for_test(&mut conn, ¬ice).unwrap(); - assert_eq!(saved, notice); - assert_eq!(persist_notice_for_test(&mut conn, ¬ice).unwrap(), notice); - - let attempt_id = DeliveryAttemptId::new(); - let reserved = reserve_supervisor_notice_attempt(&mut conn, notice.id, attempt_id).unwrap(); - assert!(matches!( - reserved.delivery, - SupervisorNoticeDelivery::Sending { .. } - )); - assert_eq!( - persist_notice_for_test(&mut conn, ¬ice).unwrap(), - reserved - ); - - let mut action_conflict = notice.clone(); - action_conflict.id = NoticeId::new(); - assert!(matches!( - persist_notice_for_test(&mut conn, &action_conflict), - Err(SupervisorNoticeStoreError::Conflict) - )); - - let mut content_conflict = notice.clone(); - content_conflict.payload = SupervisorNoticePayload::AttentionRequired { - reason: "different action content".into(), - }; - assert!(matches!( - persist_notice_for_test(&mut conn, &content_conflict), - Err(SupervisorNoticeStoreError::Conflict) - )); -} - -#[test] -fn notice_insert_uses_the_callers_loan_transaction() { - let mut conn = connection(); - let resource = resource(); - register_resource(&mut conn, &resource).unwrap(); - let loan_id = LoanId::new(); - let action_id = ActionId::new(); - let task_id = TaskId::new(); - let loan_state = LoanState::Active { - phase: LoanPhase::AwaitingRelease { - action_id, - observed_background_task: task_id, - watcher_intent: None, - }, - }; - let notice = SupervisorNotice { - id: NoticeId::new(), - loan_id, - action_id, - state_revision: resource.state_revision, - destination: resource.supervisor, - assignment_revision: resource.assignment_revision, - payload: SupervisorNoticePayload::ReleaseRequired { task_id }, - delivery: SupervisorNoticeDelivery::Pending { attempts: 0 }, - }; - - let tx = conn - .transaction_with_behavior(TransactionBehavior::Immediate) - .unwrap(); - insert_loan(&tx, loan_id, resource.id, loan_state).unwrap(); - assert_eq!( - insert_supervisor_notice_in_transaction(&tx, ¬ice).unwrap(), - notice - ); - tx.rollback().unwrap(); - - let loan_count: i64 = conn - .query_row( - "SELECT COUNT(*) FROM loans WHERE id = ?1", - [loan_id.as_uuid().to_string()], - |row| row.get(0), - ) - .unwrap(); - let notice_count: i64 = conn - .query_row( - "SELECT COUNT(*) FROM resource_supervisor_notices WHERE id = ?1", - [notice.id.as_uuid().to_string()], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(loan_count, 0); - assert_eq!(notice_count, 0); -} - -#[test] -fn only_the_current_attempt_can_settle_and_delivery_does_not_finish_the_loan() { - let mut conn = connection(); - let notice = insert_notice_fixture(&mut conn); - persist_notice_for_test(&mut conn, ¬ice).unwrap(); - let loan_state_before: String = conn - .query_row( - "SELECT state_json FROM loans WHERE id = ?1", - [notice.loan_id.as_uuid().to_string()], - |row| row.get(0), - ) - .unwrap(); - - let first_attempt = DeliveryAttemptId::new(); - reserve_supervisor_notice_attempt(&mut conn, notice.id, first_attempt).unwrap(); - let failed = settle_supervisor_notice_attempt( - &mut conn, - notice.id, - first_attempt, - Err("temporary transport failure".into()), - ) - .unwrap(); - assert_eq!( - failed.delivery, - SupervisorNoticeDelivery::RetryPending { - attempts: 1, - last_error: "temporary transport failure".into(), - } - ); - - let second_attempt = DeliveryAttemptId::new(); - reserve_supervisor_notice_attempt(&mut conn, notice.id, second_attempt).unwrap(); - assert!(matches!( - settle_supervisor_notice_attempt(&mut conn, notice.id, first_attempt, Ok(())), - Err(SupervisorNoticeStoreError::StaleAttempt) - )); - assert!(matches!( - supervisor_notice(&conn, notice.id).unwrap().unwrap().delivery, - SupervisorNoticeDelivery::Sending { attempt_id, attempt: 2 } - if attempt_id == second_attempt - )); - - let delivered = - settle_supervisor_notice_attempt(&mut conn, notice.id, second_attempt, Ok(())).unwrap(); - assert_eq!( - delivered.delivery, - SupervisorNoticeDelivery::Delivered { attempts: 2 } - ); - let loan_state_after: String = conn - .query_row( - "SELECT state_json FROM loans WHERE id = ?1", - [notice.loan_id.as_uuid().to_string()], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(loan_state_after, loan_state_before); - assert!(matches!( - reserve_supervisor_notice_attempt(&mut conn, notice.id, DeliveryAttemptId::new()), - Err(SupervisorNoticeStoreError::AlreadyDelivered) - )); -} - -#[test] -fn retry_budget_and_last_error_survive_database_reopen() { - let directory = tempdir().unwrap(); - let path = directory.path().join("db"); - let notice = { - let mut conn = connection_at(&path); - let notice = insert_notice_fixture(&mut conn); - persist_notice_for_test(&mut conn, ¬ice).unwrap(); - let attempt = DeliveryAttemptId::new(); - reserve_supervisor_notice_attempt(&mut conn, notice.id, attempt).unwrap(); - settle_supervisor_notice_attempt( - &mut conn, - notice.id, - attempt, - Err("first retry error".into()), - ) - .unwrap(); - notice - }; - - let mut conn = connection_at(&path); - assert_eq!( - supervisor_notice(&conn, notice.id) - .unwrap() - .unwrap() - .delivery, - SupervisorNoticeDelivery::RetryPending { - attempts: 1, - last_error: "first retry error".into(), - } - ); - let second_attempt = DeliveryAttemptId::new(); - reserve_supervisor_notice_attempt(&mut conn, notice.id, second_attempt).unwrap(); - settle_supervisor_notice_attempt( - &mut conn, - notice.id, - second_attempt, - Err("second retry error".into()), - ) - .unwrap(); - drop(conn); - - let mut conn = connection_at(&path); - assert_eq!( - supervisor_notice(&conn, notice.id) - .unwrap() - .unwrap() - .delivery, - SupervisorNoticeDelivery::RetryPending { - attempts: 2, - last_error: "second retry error".into(), - } - ); - let final_attempt = DeliveryAttemptId::new(); - reserve_supervisor_notice_attempt(&mut conn, notice.id, final_attempt).unwrap(); - let failed = settle_supervisor_notice_attempt( - &mut conn, - notice.id, - final_attempt, - Err("final retry error".into()), - ) - .unwrap(); - assert_eq!( - failed.delivery, - SupervisorNoticeDelivery::Failed { - attempts: 3, - last_error: "final retry error".into(), - } - ); - drop(conn); - - let mut reopened = connection_at(&path); - assert_eq!( - supervisor_notice(&reopened, notice.id) - .unwrap() - .unwrap() - .delivery, - SupervisorNoticeDelivery::Failed { - attempts: 3, - last_error: "final retry error".into(), - } - ); - assert!(pending_supervisor_notices(&reopened).unwrap().is_empty()); - assert!(matches!( - reserve_supervisor_notice_attempt(&mut reopened, notice.id, DeliveryAttemptId::new()), - Err(SupervisorNoticeStoreError::AttemptBudgetExhausted) - )); -} - -#[test] -fn startup_recovery_retries_or_fails_sending_notices_without_new_identity() { - let directory = tempdir().unwrap(); - let path = directory.path().join("db"); - let (notice, in_flight_attempt) = { - let mut conn = connection_at(&path); - let notice = insert_notice_fixture(&mut conn); - persist_notice_for_test(&mut conn, ¬ice).unwrap(); - for error in ["first error", "second error"] { - let attempt = DeliveryAttemptId::new(); - reserve_supervisor_notice_attempt(&mut conn, notice.id, attempt).unwrap(); - settle_supervisor_notice_attempt(&mut conn, notice.id, attempt, Err(error.into())) - .unwrap(); - } - let in_flight_attempt = DeliveryAttemptId::new(); - reserve_supervisor_notice_attempt(&mut conn, notice.id, in_flight_attempt).unwrap(); - (notice, in_flight_attempt) - }; - - let mut reopened = connection_at(&path); - assert!(matches!( - supervisor_notice(&reopened, notice.id).unwrap().unwrap().delivery, - SupervisorNoticeDelivery::Sending { attempt_id, attempt: 3 } - if attempt_id == in_flight_attempt - )); - let recovered = recover_sending_supervisor_notices(&mut reopened).unwrap(); - assert_eq!(recovered.len(), 1); - assert_eq!(recovered[0].id, notice.id); - assert_eq!(recovered[0].action_id, notice.action_id); - assert_eq!( - recovered[0].delivery, - SupervisorNoticeDelivery::Failed { - attempts: 3, - last_error: INTERRUPTED_DELIVERY_ERROR.into(), - } - ); - assert!(matches!( - settle_supervisor_notice_attempt(&mut reopened, notice.id, in_flight_attempt, Ok((),)), - Err(SupervisorNoticeStoreError::StaleAttempt) - )); - assert!( - recover_sending_supervisor_notices(&mut reopened) - .unwrap() - .is_empty() - ); -} - -#[test] -fn supervisor_retarget_uses_assignment_cas_and_invalidates_old_attempts() { - let mut conn = connection(); - let notice = insert_notice_fixture(&mut conn); - persist_notice_for_test(&mut conn, ¬ice).unwrap(); - let old_assignment = notice.assignment_revision; - let next_assignment = AssignmentRevision::new(old_assignment.get() + 1); - let next_destination = SupervisorAddress { - machine: MachineId::new(), - thread: ThreadId(Uuid::now_v7()), - }; - assert!(matches!( - retarget_supervisor_notice( - &mut conn, - notice.id, - AssignmentRevision::new(old_assignment.get() + 1), - next_destination, - AssignmentRevision::new(old_assignment.get() + 2), - ), - Err(SupervisorNoticeStoreError::StaleAssignmentRevision) - )); - - let old_attempt = DeliveryAttemptId::new(); - reserve_supervisor_notice_attempt(&mut conn, notice.id, old_attempt).unwrap(); - let retargeted = retarget_supervisor_notice( - &mut conn, - notice.id, - old_assignment, - next_destination, - next_assignment, - ) - .unwrap(); - assert_eq!(retargeted.id, notice.id); - assert_eq!(retargeted.loan_id, notice.loan_id); - assert_eq!(retargeted.action_id, notice.action_id); - assert_eq!(retargeted.destination, next_destination); - assert_eq!(retargeted.assignment_revision, next_assignment); - assert_eq!( - retargeted.delivery, - SupervisorNoticeDelivery::RetryPending { - attempts: 1, - last_error: RETARGETED_DELIVERY_ERROR.into(), - } - ); - assert!(matches!( - settle_supervisor_notice_attempt(&mut conn, notice.id, old_attempt, Ok(())), - Err(SupervisorNoticeStoreError::StaleAttempt) - )); - assert!(matches!( - retarget_supervisor_notice( - &mut conn, - notice.id, - old_assignment, - next_destination, - AssignmentRevision::new(next_assignment.get() + 1), - ), - Err(SupervisorNoticeStoreError::StaleAssignmentRevision) - )); - - let current_attempt = DeliveryAttemptId::new(); - reserve_supervisor_notice_attempt(&mut conn, notice.id, current_attempt).unwrap(); - settle_supervisor_notice_attempt(&mut conn, notice.id, current_attempt, Ok(())).unwrap(); - assert!(matches!( - retarget_supervisor_notice( - &mut conn, - notice.id, - next_assignment, - next_destination, - AssignmentRevision::new(next_assignment.get() + 1), - ), - Err(SupervisorNoticeStoreError::AlreadyDelivered) - )); - assert!(matches!( - supervisor_notice(&conn, notice.id) - .unwrap() - .unwrap() - .delivery, - SupervisorNoticeDelivery::Delivered { attempts: 2 } - )); -} - -#[test] -fn pending_notices_are_ordered_by_stable_notice_id() { - let mut conn = connection(); - // each loan awaits one action, so each notice needs its own loan - let mut first = insert_notice_fixture(&mut conn); - first.id = NoticeId::from_uuid(Uuid::from_u128(1)).unwrap(); - let mut second = insert_notice_fixture(&mut conn); - second.id = NoticeId::from_uuid(Uuid::from_u128(2)).unwrap(); - persist_notice_for_test(&mut conn, &second).unwrap(); - persist_notice_for_test(&mut conn, &first).unwrap(); - - let pending = pending_supervisor_notices(&conn).unwrap(); - assert_eq!(pending.len(), 2); - assert_eq!(pending[0].id, first.id); - assert_eq!(pending[1].id, second.id); - assert_eq!(supervisor_notice(&conn, first.id).unwrap(), Some(first)); -} - -#[test] -fn corrupt_notice_record_is_not_a_retryable_storage_error() { - let mut conn = connection(); - let notice = insert_notice_fixture(&mut conn); - persist_notice_for_test(&mut conn, ¬ice).unwrap(); - conn.execute( - "UPDATE resource_supervisor_notices - SET notice_json = json_remove(notice_json, '$.payload.task_id') WHERE id = ?1", - [notice.id.as_uuid().to_string()], - ) - .unwrap(); - - assert!(matches!( - supervisor_notice(&conn, notice.id), - Err(SupervisorNoticeStoreError::CorruptRecord { .. }) - )); - assert!(matches!( - pending_supervisor_notices(&conn), - Err(SupervisorNoticeStoreError::CorruptRecord { .. }) - )); -} - -fn set_loan_state(conn: &Connection, loan_id: LoanId, state: &LoanState) { - conn.execute( - "UPDATE loans SET state_json = ?1 WHERE id = ?2", - params![ - serde_json::to_string(state).unwrap(), - loan_id.as_uuid().to_string() - ], - ) - .unwrap(); -} - -fn return_notice_awaiting_retry(conn: &mut Connection) -> SupervisorNotice { - let release_notice = insert_notice_fixture(conn); - return_notice_after_release(conn, release_notice) -} - -/// Complete the release of `release_notice` and fail the first return notice attempt -fn return_notice_after_release( - conn: &mut Connection, - release_notice: SupervisorNotice, -) -> SupervisorNotice { - let return_context = ReturnContext::Idle; - let action_id = ActionId::new(); - set_loan_state( - conn, - release_notice.loan_id, - &LoanState::Active { - phase: LoanPhase::AwaitingReturn { - action_id, - return_context: return_context.clone(), - }, - }, - ); - let notice = SupervisorNotice { - id: NoticeId::new(), - action_id, - payload: SupervisorNoticePayload::ReturnRequired { return_context }, - ..release_notice - }; - persist_notice_for_test(conn, ¬ice).unwrap(); - let attempt_id = DeliveryAttemptId::new(); - reserve_supervisor_notice_attempt(conn, notice.id, attempt_id).unwrap(); - settle_supervisor_notice_attempt(conn, notice.id, attempt_id, Err("queue timed out".into())) - .unwrap() -} - -fn assert_not_deliverable(conn: &mut Connection, notice: &SupervisorNotice) { - assert!(pending_supervisor_notices(conn).unwrap().is_empty()); - assert!(matches!( - reserve_supervisor_notice_attempt(conn, notice.id, DeliveryAttemptId::new()), - Err(SupervisorNoticeStoreError::ActionNoLongerAwaited) - )); - assert!(matches!( - supervisor_notice(conn, notice.id) - .unwrap() - .unwrap() - .delivery, - SupervisorNoticeDelivery::RetryPending { attempts: 1, .. } - )); -} - -#[test] -fn return_notice_awaiting_retry_is_deliverable_while_the_return_is_undecided() { - let mut conn = connection(); - let notice = return_notice_awaiting_retry(&mut conn); - - let pending = pending_supervisor_notices(&conn).unwrap(); - assert_eq!(pending, vec![notice.clone()]); - reserve_supervisor_notice_attempt(&mut conn, notice.id, DeliveryAttemptId::new()).unwrap(); -} - -#[test] -fn return_notice_stops_once_the_return_decision_is_accepted() { - let mut conn = connection(); - let notice = return_notice_awaiting_retry(&mut conn); - set_loan_state( - &conn, - notice.loan_id, - &LoanState::Active { - phase: LoanPhase::Restoring { - action_id: notice.action_id, - return_context: ReturnContext::Idle, - resume_task_id: TaskId::new(), - }, - }, - ); - - assert_not_deliverable(&mut conn, ¬ice); -} - -#[test] -fn return_notice_stops_once_the_loan_is_closed() { - let mut conn = connection(); - let notice = return_notice_awaiting_retry(&mut conn); - set_loan_state( - &conn, - notice.loan_id, - &LoanState::Closed { - result: LoanClosure::NoResume { - return_context: ReturnContext::Idle, - reason: "done for today".into(), - }, - }, - ); - - assert_not_deliverable(&mut conn, ¬ice); -} - -#[test] -fn release_notice_stops_once_the_loan_awaits_its_return() { - let mut conn = connection(); - let release_notice = insert_notice_fixture(&mut conn); - persist_notice_for_test(&mut conn, &release_notice).unwrap(); - let return_notice = return_notice_after_release(&mut conn, release_notice.clone()); - - let pending = pending_supervisor_notices(&conn).unwrap(); - assert_eq!(pending, vec![return_notice]); - assert!(matches!( - reserve_supervisor_notice_attempt(&mut conn, release_notice.id, DeliveryAttemptId::new()), - Err(SupervisorNoticeStoreError::ActionNoLongerAwaited) - )); -} diff --git a/src/resource/trainer_publication.rs b/src/resource/trainer_publication.rs deleted file mode 100644 index c1bf64b..0000000 --- a/src/resource/trainer_publication.rs +++ /dev/null @@ -1,2832 +0,0 @@ -//! Read-only detection of newly published trainer recovery generations - -use std::collections::BTreeSet; -use std::fs::{self, File, Metadata}; -use std::io::{self, Read}; -use std::marker::PhantomData; -use std::path::{Path, PathBuf}; - -use serde::ser::SerializeStruct; -use serde::{Deserialize, Deserializer, Serialize, Serializer, de, ser}; -use serde_json::value::RawValue; -use sha2::{Digest, Sha256}; - -use super::ownership_lock::TrainerRequestDigest; -use crate::digest::Sha256Digest; -use crate::domain::{ExitReason, TaskState}; - -const PUBLICATIONS_DIRECTORY: &str = "published"; -const RECOVERY_RECORD: &str = "segment-record.json"; -const RECOVERY_SCHEMA: &str = "trainer-direct-recovery-v1"; -const SNAPSHOT_SCHEMA: &str = "trainer-recovery-snapshot-v1"; -const STAGING_PREFIX: &str = ".publish-"; -const TERMINAL_RESULT_PREFIX: &str = "result-"; -const TERMINAL_FILE: &str = "terminal.json"; -const REQUEST_FILE: &str = "request.json"; -const PROTOCOL_SCHEMA_VERSION: u64 = 1; - -/// Exact trainer identity of one direct-segment attempt, not a Homebased task ID -#[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct AttemptBinding { - /// Campaign identity from the trainer request - #[serde(deserialize_with = "deserialize_identifier")] - pub campaign_id: String, - /// Campaign revision identity from the trainer request - #[serde(deserialize_with = "deserialize_identifier")] - pub campaign_revision_id: String, - /// Trainer task identity from the trainer request - #[serde(deserialize_with = "deserialize_identifier")] - pub task_id: String, - /// Direct-segment attempt identity from the trainer request - #[serde(deserialize_with = "deserialize_identifier")] - pub attempt_id: String, - /// One-based attempt number from the trainer request - #[serde(deserialize_with = "deserialize_positive_integer")] - pub attempt_number: u64, - /// Fencing token from the trainer request - #[serde(deserialize_with = "deserialize_identifier")] - pub ownership_token: String, -} - -impl AttemptBinding { - pub(crate) fn validate(&self) -> Result<(), WatcherError> { - for identifier in [ - &self.campaign_id, - &self.campaign_revision_id, - &self.task_id, - &self.attempt_id, - &self.ownership_token, - ] { - validate_identifier(identifier) - .map_err(|reason| WatcherError::InvalidAttemptBinding { reason })?; - } - - if self.attempt_number == 0 { - return Err(WatcherError::InvalidAttemptBinding { - reason: "attempt_number must be positive", - }); - } - - Ok(()) - } -} - -/// Baseline generation identities captured before a release watch starts -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct RecoverySnapshot { - binding: AttemptBinding, - generation_ids: BTreeSet, -} - -impl RecoverySnapshot { - pub(crate) fn is_for(&self, binding: &AttemptBinding) -> bool { - self.binding == *binding - } - - pub(crate) fn contains_generation(&self, generation_id: &str) -> bool { - self.generation_ids.contains(generation_id) - } -} - -impl Serialize for RecoverySnapshot { - fn serialize(&self, serializer: S) -> Result - where - S: Serializer, - { - self.binding.validate().map_err(ser::Error::custom)?; - for generation_id in &self.generation_ids { - validate_identifier(generation_id).map_err(ser::Error::custom)?; - } - - let mut snapshot = serializer.serialize_struct("RecoverySnapshot", 3)?; - snapshot.serialize_field("schema", SNAPSHOT_SCHEMA)?; - snapshot.serialize_field("binding", &self.binding)?; - snapshot.serialize_field("generation_ids", &self.generation_ids)?; - snapshot.end() - } -} - -impl<'de> Deserialize<'de> for RecoverySnapshot { - fn deserialize(deserializer: D) -> Result - where - D: Deserializer<'de>, - { - #[derive(Deserialize)] - #[serde(deny_unknown_fields)] - struct SnapshotDocument { - schema: String, - binding: AttemptBinding, - generation_ids: Vec, - } - - let document = SnapshotDocument::deserialize(deserializer)?; - if document.schema != SNAPSHOT_SCHEMA { - return Err(de::Error::custom("unsupported recovery snapshot schema")); - } - document.binding.validate().map_err(de::Error::custom)?; - - let mut generation_ids = BTreeSet::new(); - for generation_id in document.generation_ids { - validate_identifier(&generation_id).map_err(de::Error::custom)?; - if !generation_ids.insert(generation_id) { - return Err(de::Error::custom("duplicate recovery generation id")); - } - } - - Ok(Self { - binding: document.binding, - generation_ids, - }) - } -} - -/// One complete recovery publication for an exact trainer attempt -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct PublishedRecoveryGeneration { - /// Path to the immutable published generation directory - pub path: PathBuf, - /// Generation identity committed in the recovery record - pub generation_id: String, - /// Trainer update count committed in the recovery record - pub committed_update_count: u64, -} - -/// Content identity for one complete checkpoint publication -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(crate) struct VerifiedCheckpointPublication { - pub(crate) binding: AttemptBinding, - pub(crate) path: PathBuf, - pub(crate) generation_id: String, - pub(crate) committed_update_count: u64, - pub(crate) record_sha256: Sha256Digest, - pub(crate) inventory_sha256: Sha256Digest, -} - -impl VerifiedCheckpointPublication { - fn observed_generation(&self) -> PublishedRecoveryGeneration { - PublishedRecoveryGeneration { - path: self.path.clone(), - generation_id: self.generation_id.clone(), - committed_update_count: self.committed_update_count, - } - } -} - -/// One fully verified final result publication for an exact trainer attempt -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct PublishedTerminalResult { - /// Immutable result publication directory - pub publication_path: PathBuf, - /// Terminal evidence written under the matching trainer attempt - pub terminal_path: PathBuf, - /// Exact trainer identity committed by the request and terminal - pub binding: AttemptBinding, - /// Output paths bound by the result receipt - pub output_paths: Vec, - /// SHA-256 digest of the exact request bytes validated for this result - pub request_sha256: TrainerRequestDigest, - /// SHA-256 digest of the terminal record, request, and verified output inventory - pub publication_sha256: Sha256Digest, -} - -/// Complete publication evidence observed during one release-watch decision -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct ObservedPublications { - /// New complete checkpoint not present in the persisted baseline - pub new_checkpoint: Option, - /// Complete matching final result, if it has been published - pub completed_result: Option, -} - -/// Read-only conclusion from publication evidence and one exact Homebased task state -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum WatchObservation { - /// The task is queued and has not published attempt evidence yet - WaitingForTaskStart, - /// The task is running and has not published a new checkpoint or final result - WaitingForCheckpoint, - /// A matching checkpoint is available while the task is running - /// - /// This only permits a caller to consider requesting a stop. It does not - /// mean that a stop was requested, that the process exited, or that the GPU - /// was released - StopRequestCandidate { - /// New complete checkpoint that prompted this candidate - checkpoint: PublishedRecoveryGeneration, - }, - /// A final result is published, but Homebased still reports the task as running - /// - /// Wait for the exact task to end. Do not request a stop based on a checkpoint - /// observed at the same time as this result - CompletedResultAwaitingTaskExit { - /// Verified final result for the exact trainer attempt - result: PublishedTerminalResult, - /// New checkpoint observed with the result, if any - new_checkpoint: Option, - }, - /// The exact task ended successfully and a matching final result is published - /// - /// This is only a candidate for the resource workflow's `AlreadyCompleted` - /// path. The caller must still prove that the GPU-owning process or container - /// has exited. This observation never confirms GPU release - AlreadyCompletedCandidate { - /// Verified final result for the exact trainer attempt - result: PublishedTerminalResult, - /// New checkpoint observed with the result, if any - new_checkpoint: Option, - }, - /// The task state or its relation to the publications needs caller attention - Attention(WatcherAttention), -} - -/// Task or publication state that cannot safely advance a release watch -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum WatcherAttention { - /// Homebased lost the exact task, so process state is unknown - LostTask { - /// Publications observed before process state became unknown - publications: ObservedPublications, - }, - /// The exact task ended without a successful zero exit - FailedTask { - /// Recorded reason for the task exit - reason: ExitReason, - /// Publications observed before the task ended - publications: ObservedPublications, - }, - /// Attempt evidence appeared before Homebased reported the task as running - PublicationBeforeTaskStart { - /// Publications inconsistent with the queued task state - publications: ObservedPublications, - }, - /// The task exited successfully but did not publish a matching final result - SuccessfulTaskWithoutFinalResult { - /// New checkpoint observed before the successful task exit, if any - new_checkpoint: Option, - }, -} - -/// Read-only publication detection errors that need caller attention -#[derive(Debug, thiserror::Error)] -pub enum WatcherError { - /// A filesystem operation failed while reading publication state - #[error("cannot inspect {path}: {source}")] - Io { - /// Path involved in the failed filesystem operation - path: PathBuf, - /// Underlying filesystem error - #[source] - source: io::Error, - }, - /// The detector found a symbolic link where a regular path is required - #[error("refusing symlinked trainer publication path {path}")] - Symlink { - /// Symlink path that cannot be trusted as a publication - path: PathBuf, - }, - /// The runtime or publication root has an invalid filesystem type - #[error("invalid trainer publication directory {path}: {reason}")] - InvalidDirectory { - /// Directory path with the invalid filesystem type - path: PathBuf, - /// Why the path cannot be used as a publication directory - reason: &'static str, - }, - /// A candidate publication is incomplete or does not follow the record contract - #[error("malformed trainer publication at {path}: {reason}")] - MalformedPublication { - /// Candidate generation directory - path: PathBuf, - /// Why the candidate cannot be trusted as a complete publication - reason: String, - }, - /// Caller-supplied trainer attempt identity is invalid - #[error("invalid trainer attempt binding: {reason}")] - InvalidAttemptBinding { - /// Invalid trainer identifier or attempt number - reason: &'static str, - }, - /// A baseline cannot be reused for a different trainer attempt - #[error("recovery snapshot belongs to a different trainer attempt")] - SnapshotBindingMismatch, -} - -/// Capture complete recovery generation identities for one exact attempt -/// -/// Missing runtime or publication directories produce an empty baseline -/// Malformed candidate publications and unsafe symlinks return an error so the -/// caller can retain the resource and request attention -pub fn snapshot( - runtime_root: &Path, - expected_binding: &AttemptBinding, -) -> Result { - let generations = read_matching_generations(runtime_root, expected_binding)?; - let generation_ids = generations - .into_iter() - .map(|generation| generation.generation_id) - .collect(); - - Ok(RecoverySnapshot { - binding: expected_binding.clone(), - generation_ids, - }) -} - -/// Find the first deterministic complete publication not present in the baseline -/// -/// Results are ordered by generation identity. A returned publication only -/// identifies a checkpoint; it does not establish process exit or release a GPU -pub fn find_new( - runtime_root: &Path, - expected_binding: &AttemptBinding, - baseline: &RecoverySnapshot, -) -> Result, WatcherError> { - Ok( - find_new_publication(runtime_root, expected_binding, baseline)? - .map(|generation| generation.observed_generation()), - ) -} - -fn find_new_publication( - runtime_root: &Path, - expected_binding: &AttemptBinding, - baseline: &RecoverySnapshot, -) -> Result, WatcherError> { - expected_binding.validate()?; - if baseline.binding != *expected_binding { - return Err(WatcherError::SnapshotBindingMismatch); - } - - Ok(read_matching_generations(runtime_root, expected_binding)? - .into_iter() - .find(|generation| !baseline.generation_ids.contains(&generation.generation_id))) -} - -/// Revalidate the exact complete publication selected for a stop reservation -pub(crate) fn revalidate_checkpoint_publication( - runtime_root: &Path, - expected_binding: &AttemptBinding, - expected: &VerifiedCheckpointPublication, -) -> Result { - if expected.binding != *expected_binding { - return Err(WatcherError::SnapshotBindingMismatch); - } - - Ok(read_matching_generations(runtime_root, expected_binding)? - .into_iter() - .find(|generation| generation.generation_id == expected.generation_id) - .is_some_and(|generation| generation == *expected)) -} - -/// Find a complete successful result and matching terminal evidence for one attempt -/// -/// A result is accepted only when the publication, persisted request, and -/// attempt-local terminal file agree on the exact trainer identity. The -/// publication inventory is checked against the files on disk -pub fn find_completed_result( - runtime_root: &Path, - expected_binding: &AttemptBinding, -) -> Result, WatcherError> { - expected_binding.validate()?; - let publications_root = publication_root(runtime_root)?; - - let result_path = publications_root.map(|root| { - root.join(format!( - "{TERMINAL_RESULT_PREFIX}{}", - expected_binding.attempt_id - )) - }); - let result_metadata = match &result_path { - Some(path) => optional_directory(path)?, - None => None, - }; - - let attempts_path = runtime_root.join("attempts"); - let attempts_metadata = optional_directory(&attempts_path)?; - let attempt_path = attempts_path.join(&expected_binding.attempt_id); - let attempt_metadata = if attempts_metadata.is_some() { - optional_directory(&attempt_path)? - } else { - None - }; - let terminal_path = attempt_path.join(TERMINAL_FILE); - let terminal_metadata = if attempt_metadata.is_some() { - optional_file(&terminal_path)? - } else { - None - }; - - let Some(result_metadata) = result_metadata else { - let Some(terminal_metadata) = terminal_metadata else { - return Ok(None); - }; - let terminal = read_terminal_header(&terminal_path, &terminal_metadata, expected_binding)?; - if terminal.is_completed { - return Err(malformed_publication( - &terminal_path, - "completed terminal evidence has no result publication", - )); - } - - return Ok(None); - }; - - let Some(result_path) = result_path else { - return Err(malformed_publication( - runtime_root, - "result directory exists without a publication root", - )); - }; - reject_symlink(&result_path, &result_metadata)?; - if !result_metadata.is_dir() { - return Err(malformed_publication( - &result_path, - "result publication is not a directory", - )); - } - if attempt_metadata.is_none() { - return Err(malformed_publication( - &result_path, - "result publication has no matching attempt directory", - )); - } - let Some(terminal_metadata) = terminal_metadata else { - return Err(malformed_publication( - &terminal_path, - "result publication has no matching terminal evidence", - )); - }; - - let request_path = attempt_path.join(REQUEST_FILE); - let request_metadata = optional_file(&request_path)?.ok_or_else(|| { - malformed_publication(&request_path, "result publication has no matching request") - })?; - let terminal_bytes = read_regular_file(&terminal_path, &terminal_metadata)?; - let publication_record_path = result_path.join(RECOVERY_RECORD); - let publication_record_metadata = - optional_file(&publication_record_path)?.ok_or_else(|| { - malformed_publication(&result_path, "result publication has no terminal record") - })?; - let publication_bytes = - read_regular_file(&publication_record_path, &publication_record_metadata)?; - if terminal_bytes != publication_bytes { - return Err(malformed_publication( - &publication_record_path, - "published terminal record differs from attempt terminal evidence", - )); - } - - let terminal: CompletedTerminalDocument = serde_json::from_slice(&publication_bytes) - .map_err(|error| malformed_publication(&publication_record_path, error.to_string()))?; - let request_bytes = read_regular_file(&request_path, &request_metadata)?; - let request: WorkerRequestProjection = serde_json::from_slice(&request_bytes) - .map_err(|error| malformed_publication(&request_path, error.to_string()))?; - - validate_completed_terminal( - &terminal, - &request, - &request_bytes, - expected_binding, - &result_path, - )?; - validate_result_files(&result_path, &terminal.terminal.outcome.result.outputs)?; - - let request_sha256 = TrainerRequestDigest::of(&request_bytes); - let publication_sha256 = completed_publication_sha256( - &result_path, - &request_bytes, - &publication_bytes, - &terminal.terminal.outcome.result.outputs, - ); - - Ok(Some(PublishedTerminalResult { - publication_path: result_path, - terminal_path, - binding: expected_binding.clone(), - output_paths: terminal - .terminal - .outcome - .result - .expected_outputs - .into_iter() - .map(|output| output.path) - .collect(), - request_sha256, - publication_sha256, - })) -} - -/// Observe one release watch from its exact attempt, baseline, and Homebased task state -/// -/// The task state must belong to the Homebased task that launched `expected_binding` -/// A checkpoint can only produce a stop-request candidate while that task is running -/// A final result takes precedence over a checkpoint and waits for task exit. Even a -/// successful task exit plus a final result is only an `AlreadyCompletedCandidate`; -/// the caller must independently prove that the GPU-owning process or container has -/// exited before it completes resource release. No observation confirms GPU release -/// -/// Strict publication errors are returned unchanged so the caller can retain the -/// resource and request attention -pub fn observe_release( - runtime_root: &Path, - expected_binding: &AttemptBinding, - baseline: &RecoverySnapshot, - task_state: &TaskState, -) -> Result { - observe_release_with_checkpoint_evidence(runtime_root, expected_binding, baseline, task_state) - .map(|(observation, _)| observation) -} - -pub(crate) fn observe_release_with_checkpoint_evidence( - runtime_root: &Path, - expected_binding: &AttemptBinding, - baseline: &RecoverySnapshot, - task_state: &TaskState, -) -> Result<(WatchObservation, Option), WatcherError> { - let verified_checkpoint = find_new_publication(runtime_root, expected_binding, baseline)?; - let publications = ObservedPublications { - new_checkpoint: verified_checkpoint - .as_ref() - .map(VerifiedCheckpointPublication::observed_generation), - completed_result: find_completed_result(runtime_root, expected_binding)?, - }; - - let observation = match task_state { - TaskState::Queued - if publications.new_checkpoint.is_none() && publications.completed_result.is_none() => - { - WatchObservation::WaitingForTaskStart - } - TaskState::Queued => { - WatchObservation::Attention(WatcherAttention::PublicationBeforeTaskStart { - publications, - }) - } - TaskState::Running { .. } => { - let ObservedPublications { - new_checkpoint, - completed_result, - } = publications; - - if let Some(result) = completed_result { - WatchObservation::CompletedResultAwaitingTaskExit { - result, - new_checkpoint, - } - } else if let Some(checkpoint) = new_checkpoint { - WatchObservation::StopRequestCandidate { checkpoint } - } else { - WatchObservation::WaitingForCheckpoint - } - } - TaskState::Finished { - reason: ExitReason::Exit { code: 0 }, - } => { - let ObservedPublications { - new_checkpoint, - completed_result, - } = publications; - - if let Some(result) = completed_result { - WatchObservation::AlreadyCompletedCandidate { - result, - new_checkpoint, - } - } else { - WatchObservation::Attention(WatcherAttention::SuccessfulTaskWithoutFinalResult { - new_checkpoint, - }) - } - } - TaskState::Finished { reason } => { - WatchObservation::Attention(WatcherAttention::FailedTask { - reason: reason.clone(), - publications, - }) - } - TaskState::Lost => WatchObservation::Attention(WatcherAttention::LostTask { publications }), - }; - let selected_checkpoint = - if matches!(observation, WatchObservation::StopRequestCandidate { .. }) { - verified_checkpoint - } else { - None - }; - - Ok((observation, selected_checkpoint)) -} - -fn read_terminal_header( - terminal_path: &Path, - terminal_metadata: &Metadata, - expected_binding: &AttemptBinding, -) -> Result { - let bytes = read_regular_file(terminal_path, terminal_metadata)?; - let terminal: TerminalHeaderDocument = serde_json::from_slice(&bytes) - .map_err(|error| malformed_publication(terminal_path, error.to_string()))?; - if terminal.terminal.schema_version != PROTOCOL_SCHEMA_VERSION { - return Err(malformed_publication( - terminal_path, - "unsupported terminal protocol schema", - )); - } - terminal - .terminal - .binding - .validate() - .map_err(|error| malformed_publication(terminal_path, error.to_string()))?; - if terminal.terminal.binding != *expected_binding { - return Err(malformed_publication( - terminal_path, - "terminal evidence belongs to a different trainer attempt", - )); - } - if terminal.kind != terminal.terminal.outcome.kind { - return Err(malformed_publication( - terminal_path, - "terminal kind differs from its outcome", - )); - } - let binding = &terminal.terminal.binding; - let outcome = &terminal.terminal.outcome; - match outcome.kind.as_str() { - "completed" - if outcome.result.is_some() - && outcome.recovery.is_none() - && outcome.failure.is_none() => {} - "paused_with_recovery" - if outcome.result.is_none() - && outcome.recovery.is_some() - && outcome.failure.is_none() => {} - "failed" - if outcome.result.is_none() - && outcome.recovery.is_none() - && outcome.failure.is_some() => {} - _ => { - return Err(malformed_publication( - terminal_path, - "terminal outcome has an invalid variant", - )); - } - } - if outcome.kind == "completed" { - let Some(result) = &outcome.result else { - unreachable!("completed outcome variant is checked above"); - }; - let result_binding: AttemptBinding = - serde_json::from_value(result.get("binding").cloned().ok_or_else(|| { - malformed_publication(terminal_path, "completed result has no binding") - })?) - .map_err(|error| malformed_publication(terminal_path, error.to_string()))?; - if result_binding != *binding { - return Err(malformed_publication( - terminal_path, - "terminal result binding differs from terminal binding", - )); - } - } - - Ok(TerminalHeader { - is_completed: outcome.kind == "completed", - }) -} - -fn validate_completed_terminal( - terminal: &CompletedTerminalDocument, - request: &WorkerRequestProjection, - request_bytes: &[u8], - expected_binding: &AttemptBinding, - publication_path: &Path, -) -> Result<(), WatcherError> { - let terminal_binding = &terminal.terminal.binding; - let result = &terminal.terminal.outcome.result; - if terminal.kind != "completed" - || terminal.terminal.schema_version != PROTOCOL_SCHEMA_VERSION - || terminal.terminal.outcome.kind != "completed" - { - return Err(malformed_publication( - publication_path, - "result publication is not a completed terminal", - )); - } - if terminal_binding != expected_binding - || &result.binding != expected_binding - || &result.outputs.binding != expected_binding - || &request.binding != expected_binding - { - return Err(malformed_publication( - publication_path, - "result evidence belongs to a different trainer attempt", - )); - } - if request.schema_version != PROTOCOL_SCHEMA_VERSION - || result.schema_version != PROTOCOL_SCHEMA_VERSION - { - return Err(malformed_publication( - publication_path, - "unsupported request or result protocol schema", - )); - } - validate_request_projection(request) - .map_err(|reason| malformed_publication(publication_path, reason))?; - validate_worker_identity(&result.worker) - .map_err(|reason| malformed_publication(publication_path, reason))?; - if request.worker != result.worker || request.expected_outputs != result.expected_outputs { - return Err(malformed_publication( - publication_path, - "result worker or outputs differ from the persisted request", - )); - } - let request_config_digest = configuration_digest(request_bytes) - .map_err(|reason| malformed_publication(publication_path, reason))?; - if result.config_digest != request_config_digest { - return Err(malformed_publication( - publication_path, - "result configuration digest differs from the persisted request", - )); - } - validate_result_receipt(result) - .map_err(|reason| malformed_publication(publication_path, reason)) -} - -fn validate_result_receipt(result: &ResultReceiptProjection) -> Result<(), String> { - if result.outputs.schema_version != PROTOCOL_SCHEMA_VERSION { - return Err("unsupported result inventory schema".into()); - } - validate_output_declarations(&result.expected_outputs)?; - result - .outputs - .binding - .validate() - .map_err(|error| error.to_string())?; - let mut inventory_paths = BTreeSet::new(); - let mut previous_path: Option<&str> = None; - for entry in &result.outputs.entries { - validate_artifact_path(&entry.path)?; - if entry.kind != "file" { - return Err("direct result inventory contains a non-file artifact".into()); - } - if previous_path.is_some_and(|previous| previous >= entry.path.as_str()) { - return Err("result inventory paths are not strictly sorted".into()); - } - previous_path = Some(&entry.path); - if !inventory_paths.insert(entry.path.as_str()) { - return Err("result inventory contains a duplicate path".into()); - } - } - let declarations: BTreeSet<_> = result - .expected_outputs - .iter() - .map(|output| output.path.as_str()) - .collect(); - if declarations.len() != result.expected_outputs.len() || declarations != inventory_paths { - return Err("result inventory does not match expected outputs".into()); - } - let mut metric_ids = BTreeSet::new(); - for metric in &result.metrics { - validate_identifier(&metric.metric_id)?; - if !metric_ids.insert(metric.metric_id.as_str()) { - return Err("result contains duplicate metric ids".into()); - } - if !matches!( - metric.direction.as_str(), - "higher_is_better" | "lower_is_better" - ) || !metric.value.is_finite() - || metric.denominator == 0 - || !metric.uncertainty.is_finite() - || metric.uncertainty < 0.0 - { - return Err("result contains an invalid metric".into()); - } - validate_metric_unit(&metric.unit)?; - } - validate_result_evidence(&result.evidence)?; - validate_validation_receipt(&result.validation)?; - - Ok(()) -} - -fn validate_request_projection(request: &WorkerRequestProjection) -> Result<(), String> { - for document in [ - &request.adapter, - &request.start_mode, - &request.payload, - &request.resources, - ] { - if !document.is_object() { - return Err("persisted request contains a malformed protocol object".into()); - } - } - if request.inputs.iter().any(|input| !input.is_object()) { - return Err("persisted request contains a malformed input reference".into()); - } - request - .binding - .validate() - .map_err(|error| error.to_string())?; - if request.input_view.schema_version != PROTOCOL_SCHEMA_VERSION - || request.input_view.revision_id != request.binding.campaign_revision_id - || request.input_view.root.trim().is_empty() - || request.input_view.root.chars().any(char::is_control) - { - return Err("persisted input view is not bound to the request revision".into()); - } - validate_worker_identity(&request.worker)?; - validate_output_declarations(&request.expected_outputs) -} - -/// Failure while checking one persisted direct-segment attempt request -#[derive(Debug)] -pub(crate) enum AttemptRequestValidationError { - /// The expected binding supplied by the caller is invalid - InvalidExpectedBinding(&'static str), - /// The request does not match the maintained trainer request projection - Malformed(String), - /// The request is valid but binds a different trainer attempt - BindingMismatch(Box), -} - -/// Validate a request with the same strict projection used for terminal results -/// -/// The strict projection rejects duplicate top-level and input-view fields. This -/// also computes the configuration digest used by completed-result validation -pub(crate) fn validate_attempt_request( - request_bytes: &[u8], - expected_binding: &AttemptBinding, -) -> Result<(), AttemptRequestValidationError> { - expected_binding.validate().map_err(|error| match error { - WatcherError::InvalidAttemptBinding { reason } => { - AttemptRequestValidationError::InvalidExpectedBinding(reason) - } - _ => AttemptRequestValidationError::InvalidExpectedBinding( - "trainer attempt binding is invalid", - ), - })?; - - let request: WorkerRequestProjection = serde_json::from_slice(request_bytes) - .map_err(|error| AttemptRequestValidationError::Malformed(error.to_string()))?; - if request.schema_version != PROTOCOL_SCHEMA_VERSION { - return Err(AttemptRequestValidationError::Malformed( - "unsupported request protocol schema".into(), - )); - } - validate_request_projection(&request).map_err(AttemptRequestValidationError::Malformed)?; - configuration_digest(request_bytes).map_err(AttemptRequestValidationError::Malformed)?; - - if request.binding != *expected_binding { - return Err(AttemptRequestValidationError::BindingMismatch(Box::new( - request.binding, - ))); - } - - Ok(()) -} - -fn validate_metric_unit(unit: &serde_json::Value) -> Result<(), String> { - match unit { - serde_json::Value::String(value) - if matches!(value.as_str(), "ratio" | "seconds" | "count" | "bytes") => - { - Ok(()) - } - serde_json::Value::Object(fields) if fields.len() == 1 => { - let Some(serde_json::Value::Object(custom)) = fields.get("custom") else { - return Err("result contains an invalid metric unit".into()); - }; - if custom.len() != 1 - || custom - .get("name") - .and_then(serde_json::Value::as_str) - .is_none_or(|name| name.trim().is_empty()) - { - return Err("result contains an invalid custom metric unit".into()); - } - Ok(()) - } - _ => Err("result contains an invalid metric unit".into()), - } -} - -fn validate_result_evidence(evidence: &ResultEvidenceProjection) -> Result<(), String> { - if evidence.source_exposures.is_empty() || evidence.update_points.is_empty() { - return Err("result evidence is missing source or update evidence".into()); - } - let mut source_digests = BTreeSet::new(); - for exposure in &evidence.source_exposures { - if exposure.samples == 0 || !source_digests.insert(exposure.source_digest) { - return Err("result evidence has an invalid source exposure".into()); - } - } - if !unique_values(&evidence.update_points) || !unique_values(&evidence.probe_points) { - return Err("result evidence contains duplicate points".into()); - } - if let Some(artifact) = &evidence.evidence_artifact { - if artifact.schema.version != PROTOCOL_SCHEMA_VERSION - || artifact.schema.id.trim().is_empty() - || artifact.schema.id.len() > 128 - || artifact.schema.id.chars().any(char::is_control) - || artifact.size == 0 - { - return Err("result evidence artifact has an invalid schema or size".into()); - } - validate_artifact_path(&artifact.path)?; - } - - Ok(()) -} - -fn validate_validation_receipt(receipt: &ValidationReceiptProjection) -> Result<(), String> { - validate_identifier(&receipt.validator_id)?; - if receipt.validator_version.is_empty() - || receipt.validator_version.len() > 32 - || !receipt.validator_version.bytes().all(|byte| { - byte.is_ascii_lowercase() || byte.is_ascii_digit() || matches!(byte, b'-' | b'.' | b'+') - }) - { - return Err("result validation receipt has an invalid validator version".into()); - } - Ok(()) -} - -fn unique_values(values: &[u64]) -> bool { - values.iter().copied().collect::>().len() == values.len() -} - -fn validate_worker_identity(worker: &WorkerIdentityProjection) -> Result<(), String> { - validate_identifier(&worker.worker_id)?; - Ok(()) -} - -fn validate_output_declarations(outputs: &[OutputDeclarationProjection]) -> Result<(), String> { - if outputs.is_empty() { - return Err("direct result declares no outputs".into()); - } - let mut paths = BTreeSet::new(); - for output in outputs { - validate_artifact_path(&output.path)?; - if output.kind != "file" { - return Err("direct result declaration contains a non-file output".into()); - } - if !paths.insert(output.path.as_str()) { - return Err("request contains a duplicate output path".into()); - } - } - - Ok(()) -} - -fn validate_result_files( - result_path: &Path, - inventory: &ArtifactInventoryProjection, -) -> Result<(), WatcherError> { - let mut expected_files = BTreeSet::from([RECOVERY_RECORD.to_owned()]); - for entry in &inventory.entries { - if entry.path == RECOVERY_RECORD || entry.path.starts_with(&format!("{RECOVERY_RECORD}/")) { - return Err(malformed_publication( - result_path, - "result output collides with its terminal record", - )); - } - verify_output_file(result_path, entry)?; - expected_files.insert(entry.path.clone()); - } - - let mut found_files = BTreeSet::new(); - let mut found_directories = BTreeSet::new(); - collect_publication_members( - result_path, - result_path, - &mut found_files, - &mut found_directories, - )?; - if found_files != expected_files { - return Err(malformed_publication( - result_path, - "result publication files differ from its declared inventory", - )); - } - for directory in found_directories { - if !inventory - .entries - .iter() - .any(|entry| entry.path.starts_with(&format!("{directory}/"))) - { - return Err(malformed_publication( - result_path, - "result publication contains an unlisted directory", - )); - } - } - - Ok(()) -} - -fn completed_publication_sha256( - result_path: &Path, - request_bytes: &[u8], - terminal_bytes: &[u8], - inventory: &ArtifactInventoryProjection, -) -> Sha256Digest { - let mut digest = Sha256::new(); - digest.update(b"trainer-completed-publication-v1\0"); - - let publication_path = result_path.to_string_lossy(); - for value in [publication_path.as_bytes(), request_bytes, terminal_bytes] { - digest.update((value.len() as u64).to_be_bytes()); - digest.update(value); - } - - for entry in &inventory.entries { - // the publication digest covers each entry digest as its hex text - let entry_digest = entry.digest.to_hex(); - for value in [ - entry.path.as_bytes(), - entry.kind.as_bytes(), - entry_digest.as_bytes(), - ] { - digest.update((value.len() as u64).to_be_bytes()); - digest.update(value); - } - digest.update(entry.size.to_be_bytes()); - } - - Sha256Digest::from(digest) -} - -fn verify_output_file( - result_path: &Path, - entry: &InventoryEntryProjection, -) -> Result<(), WatcherError> { - let path = result_path.join(&entry.path); - let mut current = result_path.to_path_buf(); - for component in Path::new(&entry.path).components() { - let std::path::Component::Normal(component) = component else { - return Err(malformed_publication( - result_path, - "invalid result output path", - )); - }; - current.push(component); - let current_metadata = metadata(¤t)?; - reject_symlink(¤t, ¤t_metadata)?; - if current != path && !current_metadata.is_dir() { - return Err(malformed_publication( - ¤t, - "result output parent is not a directory", - )); - } - } - let file_metadata = metadata(&path)?; - if !file_metadata.is_file() || file_metadata.len() != entry.size { - return Err(malformed_publication( - &path, - "result output type or size differs from its inventory", - )); - } - - let mut file = File::open(&path).map_err(|source| WatcherError::Io { - path: path.clone(), - source, - })?; - let mut digest = Sha256::new(); - let mut buffer = [0_u8; 64 * 1024]; - loop { - let count = file.read(&mut buffer).map_err(|source| WatcherError::Io { - path: path.clone(), - source, - })?; - if count == 0 { - break; - } - digest.update(&buffer[..count]); - } - if Sha256Digest::from(digest) != entry.digest { - return Err(malformed_publication( - &path, - "result output digest differs from its inventory", - )); - } - - Ok(()) -} - -fn collect_publication_members( - root: &Path, - directory: &Path, - files: &mut BTreeSet, - directories: &mut BTreeSet, -) -> Result<(), WatcherError> { - let entries = fs::read_dir(directory).map_err(|source| WatcherError::Io { - path: directory.to_path_buf(), - source, - })?; - for entry in entries { - let entry = entry.map_err(|source| WatcherError::Io { - path: directory.to_path_buf(), - source, - })?; - let path = entry.path(); - let entry_metadata = metadata(&path)?; - reject_symlink(&path, &entry_metadata)?; - if entry_metadata.is_dir() { - let relative = path - .strip_prefix(root) - .ok() - .and_then(Path::to_str) - .ok_or_else(|| malformed_publication(&path, "result path is not valid UTF-8"))? - .to_owned(); - directories.insert(relative); - collect_publication_members(root, &path, files, directories)?; - } else if entry_metadata.is_file() { - let relative = path - .strip_prefix(root) - .ok() - .and_then(Path::to_str) - .ok_or_else(|| malformed_publication(&path, "result path is not valid UTF-8"))? - .to_owned(); - files.insert(relative); - } else { - return Err(malformed_publication( - &path, - "result publication contains an unsupported filesystem entry", - )); - } - } - - Ok(()) -} - -fn optional_directory(path: &Path) -> Result, WatcherError> { - let Some(metadata) = optional_metadata(path)? else { - return Ok(None); - }; - reject_symlink(path, &metadata)?; - if !metadata.is_dir() { - return Err(malformed_publication(path, "expected a regular directory")); - } - - Ok(Some(metadata)) -} - -fn optional_file(path: &Path) -> Result, WatcherError> { - let Some(metadata) = optional_metadata(path)? else { - return Ok(None); - }; - reject_symlink(path, &metadata)?; - if !metadata.is_file() { - return Err(malformed_publication(path, "expected a regular file")); - } - - Ok(Some(metadata)) -} - -fn read_regular_file(path: &Path, metadata: &Metadata) -> Result, WatcherError> { - reject_symlink(path, metadata)?; - fs::read(path).map_err(|source| WatcherError::Io { - path: path.to_path_buf(), - source, - }) -} - -fn validate_artifact_path(path: &str) -> Result<(), String> { - if path.is_empty() - || path.len() > 1024 - || path.starts_with('/') - || path.ends_with('/') - || path.contains('\\') - || path.contains(':') - || path.split('/').any(|component| { - component.is_empty() - || matches!(component, "." | "..") - || component.chars().any(|character| { - character.is_control() || matches!(character as u32, 0x7f..=0x9f) - }) - }) - { - return Err("invalid trainer artifact path".into()); - } - - Ok(()) -} - -/// Request members that the trainer's configuration digest covers -/// -/// Each member keeps its exact source bytes, so the digest matches the trainer's -/// hash of the persisted request without re-encoding any value -#[derive(Deserialize)] -struct ConfigurationMembers<'a> { - #[serde(borrow)] - adapter: &'a RawValue, - #[serde(borrow)] - start_mode: &'a RawValue, - #[serde(borrow)] - payload: &'a RawValue, - #[serde(borrow)] - input_view: InputViewConfigurationMembers<'a>, - #[serde(borrow)] - inputs: &'a RawValue, - #[serde(borrow)] - expected_outputs: &'a RawValue, - #[serde(borrow)] - resources: &'a RawValue, -} - -/// Input-view members that the trainer's configuration digest covers -#[derive(Deserialize)] -struct InputViewConfigurationMembers<'a> { - #[serde(borrow)] - schema_version: &'a RawValue, - #[serde(borrow)] - revision_id: &'a RawValue, -} - -/// Hash the configuration members of one persisted request in the trainer's order -/// -/// Serde refuses a duplicate member, so no member can have two candidate values -fn configuration_digest(request: &[u8]) -> Result { - let members: ConfigurationMembers<'_> = - serde_json::from_slice(request).map_err(|error| error.to_string())?; - let input_view = &members.input_view; - let canonical = [ - "[", - members.adapter.get(), - ",", - members.start_mode.get(), - ",", - members.payload.get(), - ",{\"schema_version\":", - input_view.schema_version.get(), - ",\"revision_id\":", - input_view.revision_id.get(), - "},", - members.inputs.get(), - ",", - members.expected_outputs.get(), - ",", - members.resources.get(), - "]", - ] - .concat(); - - Ok(Sha256Digest::of(canonical)) -} - -fn read_matching_generations( - runtime_root: &Path, - expected_binding: &AttemptBinding, -) -> Result, WatcherError> { - expected_binding.validate()?; - let Some(publications_root) = publication_root(runtime_root)? else { - return Ok(Vec::new()); - }; - - let entries = fs::read_dir(&publications_root).map_err(|source| WatcherError::Io { - path: publications_root.clone(), - source, - })?; - let mut generations = Vec::new(); - - for entry in entries { - let entry = entry.map_err(|source| WatcherError::Io { - path: publications_root.clone(), - source, - })?; - let path = entry.path(); - let name = entry - .file_name() - .into_string() - .map_err(|_| malformed_publication(&path, "publication directory name is not UTF-8"))?; - - if name.starts_with(STAGING_PREFIX) || name.starts_with(TERMINAL_RESULT_PREFIX) { - continue; - } - - let directory_metadata = metadata(&path)?; - reject_symlink(&path, &directory_metadata)?; - if !directory_metadata.is_dir() { - return Err(malformed_publication( - &path, - "publication entry is not a generation directory", - )); - } - - let record_path = path.join(RECOVERY_RECORD); - let record_metadata = metadata(&record_path)?; - reject_symlink(&record_path, &record_metadata)?; - if !record_metadata.is_file() { - return Err(malformed_publication( - &path, - "recovery record is not a regular file", - )); - } - - let record = read_regular_file(&record_path, &record_metadata)?; - let bindings: BindingProjection = serde_json::from_slice(&record) - .map_err(|error| malformed_publication(&path, error.to_string()))?; - if bindings.request.binding != *expected_binding - || bindings.recovery.binding != *expected_binding - { - continue; - } - - let attempt_request_path = runtime_root - .join("attempts") - .join(&expected_binding.attempt_id) - .join(REQUEST_FILE); - let attempt_request_metadata = optional_file(&attempt_request_path)?.ok_or_else(|| { - malformed_publication(&path, "checkpoint has no matching attempt request") - })?; - let attempt_request = read_regular_file(&attempt_request_path, &attempt_request_metadata)?; - let publication_request = - request_document(&record).map_err(|reason| malformed_publication(&path, reason))?; - let saved_request: serde_json::Value = serde_json::from_slice(&attempt_request) - .map_err(|error| malformed_publication(&attempt_request_path, error.to_string()))?; - let published_request: serde_json::Value = serde_json::from_slice(publication_request) - .map_err(|error| malformed_publication(&path, error.to_string()))?; - if saved_request != published_request { - return Err(malformed_publication( - &path, - "checkpoint request differs from the exact trainer attempt request", - )); - } - - let publication: RecoveryPublicationProjection = serde_json::from_slice(&record) - .map_err(|error| malformed_publication(&path, error.to_string()))?; - if publication.schema != RECOVERY_SCHEMA { - return Err(malformed_publication( - &path, - "unsupported direct recovery schema", - )); - } - if publication.request.binding != *expected_binding - || publication.recovery.binding != *expected_binding - { - continue; - } - if publication.request.schema_version != PROTOCOL_SCHEMA_VERSION - || publication.recovery.schema_version != PROTOCOL_SCHEMA_VERSION - || publication.recovery.inventory.schema_version != PROTOCOL_SCHEMA_VERSION - || publication.recovery.inventory.binding != *expected_binding - || !matches!( - publication.recovery.cause.as_str(), - "periodic_checkpoint" | "stop_requested" | "non_finite_training" - ) - { - return Err(malformed_publication( - &path, - "recovery publication has an unsupported schema or inventory binding", - )); - } - validate_request_projection(&publication.request) - .map_err(|reason| malformed_publication(&path, reason))?; - validate_worker_identity(&publication.recovery.worker) - .map_err(|reason| malformed_publication(&path, reason))?; - if publication.request.worker != publication.recovery.worker - || publication.recovery.compatibility.schema_id != "speakrs-long-run-train-v1" - || publication.recovery.compatibility.source_digest - != publication.request.worker.source_digest - || publication.recovery.compatibility.config_digest - != configuration_digest( - request_document(&record) - .map_err(|reason| malformed_publication(&path, reason))?, - ) - .map_err(|reason| malformed_publication(&path, reason))? - { - return Err(malformed_publication( - &path, - "recovery compatibility differs from the persisted request", - )); - } - validate_identifier(&publication.recovery.generation_id) - .map_err(|reason| malformed_publication(&path, reason))?; - if name != publication.recovery.generation_id { - return Err(malformed_publication( - &path, - "directory name differs from recovery.generation_id", - )); - } - validate_checkpoint_inventory( - &path, - &publication.recovery.inventory, - expected_binding, - publication.recovery.state_digest, - )?; - - let final_record_metadata = metadata(&record_path)?; - let final_record = read_regular_file(&record_path, &final_record_metadata)?; - if final_record != record { - return Err(malformed_publication( - &record_path, - "recovery record changed during publication verification", - )); - } - - let inventory_bytes = serde_json::to_vec(&publication.recovery.inventory) - .map_err(|error| malformed_publication(&path, error.to_string()))?; - - generations.push(VerifiedCheckpointPublication { - binding: expected_binding.clone(), - path, - generation_id: publication.recovery.generation_id, - committed_update_count: publication.recovery.position.update_count, - record_sha256: Sha256Digest::of(&record), - inventory_sha256: Sha256Digest::of(&inventory_bytes), - }); - } - - generations.sort_by(|left, right| left.generation_id.cmp(&right.generation_id)); - Ok(generations) -} - -/// Borrow the exact source bytes of the request member of one recovery record -fn request_document(record: &[u8]) -> Result<&[u8], String> { - #[derive(Deserialize)] - struct RecordRequest<'a> { - #[serde(borrow)] - request: &'a RawValue, - } - - let record: RecordRequest<'_> = - serde_json::from_slice(record).map_err(|error| error.to_string())?; - Ok(record.request.get().as_bytes()) -} - -fn validate_checkpoint_inventory( - path: &Path, - inventory: &ArtifactInventoryProjection, - expected_binding: &AttemptBinding, - state_digest: Sha256Digest, -) -> Result<(), WatcherError> { - let mut previous_path: Option<&str> = None; - let mut paths = BTreeSet::new(); - for entry in &inventory.entries { - validate_artifact_path(&entry.path) - .map_err(|reason| malformed_publication(path, reason))?; - if entry.kind != "file" - || previous_path.is_some_and(|previous| previous >= entry.path.as_str()) - || !paths.insert(entry.path.as_str()) - { - return Err(malformed_publication( - path, - "recovery inventory must contain sorted unique regular files", - )); - } - previous_path = Some(&entry.path); - } - - let recovery_state = inventory - .entries - .iter() - .find(|entry| entry.path == "recovery.json") - .ok_or_else(|| malformed_publication(path, "recovery inventory omits recovery.json"))?; - if recovery_state.digest != state_digest { - return Err(malformed_publication( - path, - "recovery state digest differs from its inventory", - )); - } - if inventory.binding != *expected_binding { - return Err(malformed_publication( - path, - "recovery inventory belongs to a different trainer attempt", - )); - } - - validate_result_files(path, inventory) -} - -fn publication_root(runtime_root: &Path) -> Result, WatcherError> { - let Some(runtime_metadata) = optional_metadata(runtime_root)? else { - return Ok(None); - }; - reject_symlink(runtime_root, &runtime_metadata)?; - if !runtime_metadata.is_dir() { - return Err(WatcherError::InvalidDirectory { - path: runtime_root.to_path_buf(), - reason: "runtime root is not a directory", - }); - } - - let publications_root = runtime_root.join(PUBLICATIONS_DIRECTORY); - let Some(publications_metadata) = optional_metadata(&publications_root)? else { - return Ok(None); - }; - reject_symlink(&publications_root, &publications_metadata)?; - if !publications_metadata.is_dir() { - return Err(WatcherError::InvalidDirectory { - path: publications_root, - reason: "published root is not a directory", - }); - } - - Ok(Some(publications_root)) -} - -fn optional_metadata(path: &Path) -> Result, WatcherError> { - match fs::symlink_metadata(path) { - Ok(metadata) => Ok(Some(metadata)), - Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(None), - Err(source) => Err(WatcherError::Io { - path: path.to_path_buf(), - source, - }), - } -} - -fn metadata(path: &Path) -> Result { - fs::symlink_metadata(path).map_err(|source| WatcherError::Io { - path: path.to_path_buf(), - source, - }) -} - -fn reject_symlink(path: &Path, metadata: &Metadata) -> Result<(), WatcherError> { - if metadata.file_type().is_symlink() { - return Err(WatcherError::Symlink { - path: path.to_path_buf(), - }); - } - - Ok(()) -} - -fn malformed_publication(path: &Path, reason: impl Into) -> WatcherError { - WatcherError::MalformedPublication { - path: path.to_path_buf(), - reason: reason.into(), - } -} - -fn deserialize_identifier<'de, D>(deserializer: D) -> Result -where - D: Deserializer<'de>, -{ - let identifier = String::deserialize(deserializer)?; - validate_identifier(&identifier).map_err(de::Error::custom)?; - Ok(identifier) -} - -fn deserialize_positive_integer<'de, D>(deserializer: D) -> Result -where - D: Deserializer<'de>, -{ - let value = u64::deserialize(deserializer)?; - if value == 0 { - return Err(de::Error::custom("attempt_number must be positive")); - } - - Ok(value) -} - -fn validate_identifier(identifier: &str) -> Result<(), &'static str> { - if identifier.is_empty() || identifier.len() > 128 { - return Err("trainer identifiers must contain 1 to 128 ASCII bytes"); - } - - let mut characters = identifier.chars(); - let Some(first) = characters.next() else { - return Err("trainer identifiers must not be empty"); - }; - if !first.is_ascii_lowercase() && !first.is_ascii_digit() { - return Err("trainer identifier has an invalid first character"); - } - if characters.any(|character| { - !character.is_ascii() - || (!character.is_ascii_lowercase() - && !character.is_ascii_digit() - && !matches!(character, '-' | '_' | '.')) - }) { - return Err("trainer identifier has an invalid character"); - } - - Ok(()) -} - -#[derive(Debug, Deserialize)] -struct BindingProjection { - request: RequestBindingProjection, - recovery: RecoveryBindingProjection, -} - -#[derive(Debug, Deserialize)] -struct RequestBindingProjection { - binding: AttemptBinding, -} - -#[derive(Debug, Deserialize)] -struct RecoveryBindingProjection { - binding: AttemptBinding, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct RecoveryPublicationProjection { - schema: String, - request: WorkerRequestProjection, - recovery: RecoveryMetadataProjection, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct RecoveryMetadataProjection { - schema_version: u64, - binding: AttemptBinding, - generation_id: String, - compatibility: RecoveryCompatibilityProjection, - worker: WorkerIdentityProjection, - inventory: ArtifactInventoryProjection, - state_digest: Sha256Digest, - position: RecoveryPositionProjection, - cause: String, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct RecoveryCompatibilityProjection { - schema_id: String, - config_digest: Sha256Digest, - source_digest: Sha256Digest, -} - -#[derive(Debug, Deserialize)] -struct RecoveryPositionProjection { - update_count: u64, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct TerminalHeaderDocument { - kind: String, - terminal: TerminalHeaderProposal, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct TerminalHeaderProposal { - schema_version: u64, - binding: AttemptBinding, - outcome: TerminalOutcomeHeader, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct TerminalOutcomeHeader { - kind: String, - #[serde(default)] - result: Option, - #[serde(default)] - recovery: Option, - #[serde(default)] - failure: Option, -} - -#[derive(Debug)] -struct TerminalHeader { - is_completed: bool, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct CompletedTerminalDocument { - kind: String, - terminal: CompletedTerminalProposal, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct CompletedTerminalProposal { - schema_version: u64, - binding: AttemptBinding, - outcome: CompletedTerminalOutcome, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct CompletedTerminalOutcome { - kind: String, - result: ResultReceiptProjection, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct WorkerRequestProjection { - schema_version: u64, - binding: AttemptBinding, - adapter: serde_json::Value, - start_mode: serde_json::Value, - payload: serde_json::Value, - input_view: PreparedInputViewProjection, - inputs: Vec, - expected_outputs: Vec, - resources: serde_json::Value, - worker: WorkerIdentityProjection, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct PreparedInputViewProjection { - schema_version: u64, - revision_id: String, - root: String, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct ResultReceiptProjection { - schema_version: u64, - binding: AttemptBinding, - worker: WorkerIdentityProjection, - config_digest: Sha256Digest, - expected_outputs: Vec, - outputs: ArtifactInventoryProjection, - metrics: Vec, - evidence: ResultEvidenceProjection, - validation: ValidationReceiptProjection, -} - -#[derive(Debug, Deserialize, PartialEq, Eq)] -#[serde(deny_unknown_fields)] -struct WorkerIdentityProjection { - worker_id: String, - executable_digest: Sha256Digest, - source_digest: Sha256Digest, - environment_digest: Sha256Digest, -} - -#[derive(Debug, Deserialize, PartialEq, Eq)] -#[serde(deny_unknown_fields)] -struct OutputDeclarationProjection { - path: String, - kind: String, -} - -#[derive(Debug, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct ArtifactInventoryProjection { - schema_version: u64, - binding: AttemptBinding, - entries: Vec, -} - -#[derive(Debug, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct InventoryEntryProjection { - path: String, - kind: String, - size: u64, - digest: Sha256Digest, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct MetricProjection { - metric_id: String, - unit: serde_json::Value, - direction: String, - value: f64, - denominator: u64, - uncertainty: f64, -} - -/// Decode a field only to check its form, keeping no value -/// -/// Some published fields must be well formed although no rule reads them, so -/// the projection records that the form was checked instead of holding a value -fn decode_form<'de, D, T>(deserializer: D) -> Result, D::Error> -where - D: Deserializer<'de>, - T: Deserialize<'de>, -{ - T::deserialize(deserializer).map(|_| PhantomData) -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct ResultEvidenceProjection { - #[serde(deserialize_with = "decode_form")] - score_contract: PhantomData, - #[serde(deserialize_with = "decode_form")] - membership_digest: PhantomData, - source_exposures: Vec, - update_points: Vec, - probe_points: Vec, - #[serde(default)] - evidence_artifact: Option, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct SourceExposureProjection { - source_digest: Sha256Digest, - samples: u64, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct EvidenceArtifactProjection { - schema: EvidenceSchemaProjection, - path: String, - #[serde(deserialize_with = "decode_form")] - digest: PhantomData, - size: u64, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct EvidenceSchemaProjection { - id: String, - version: u64, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct ValidationReceiptProjection { - validator_id: String, - validator_version: String, - #[serde(deserialize_with = "decode_form")] - evidence_digest: PhantomData, - #[serde(deserialize_with = "decode_form")] - validated_at: PhantomData, -} - -#[cfg(test)] -pub(crate) mod tests { - use std::fs; - use std::path::{Path, PathBuf}; - - use serde_json::{Value, json}; - use tempfile::TempDir; - - use crate::digest::Sha256Digest; - use crate::domain::{ExitReason, TaskState}; - - use super::{ - AttemptBinding, PublishedRecoveryGeneration, WatchObservation, WatcherAttention, - WatcherError, find_completed_result, find_new, observe_release, snapshot, - }; - - fn expected_binding() -> AttemptBinding { - AttemptBinding { - campaign_id: "campaign-a".into(), - campaign_revision_id: "revision-a".into(), - task_id: "trainer-task-a".into(), - attempt_id: "attempt-a".into(), - attempt_number: 1, - ownership_token: "owner-a".into(), - } - } - - fn foreign_binding() -> AttemptBinding { - AttemptBinding { - campaign_id: "campaign-b".into(), - campaign_revision_id: "revision-b".into(), - task_id: "trainer-task-b".into(), - attempt_id: "attempt-b".into(), - attempt_number: 1, - ownership_token: "owner-b".into(), - } - } - - fn runtime_root(temp: &TempDir) -> PathBuf { - temp.path().join("runtime") - } - - fn published_root(runtime_root: &Path) -> PathBuf { - let published_root = runtime_root.join("published"); - fs::create_dir_all(&published_root).unwrap(); - published_root - } - - fn record_value( - request_binding: &AttemptBinding, - recovery_binding: &AttemptBinding, - generation_id: &str, - update_count: Value, - ) -> Value { - const CHECKPOINT_BYTES: &[u8] = b"checkpoint bytes"; - const RECOVERY_BYTES: &[u8] = b"recovery state"; - let request = request_value(request_binding); - let request_bytes = serde_json::to_vec(&request).unwrap(); - let worker = worker_identity(); - - json!({ - "schema": "trainer-direct-recovery-v1", - "request": request, - "recovery": { - "schema_version": 1, - "binding": recovery_binding, - "generation_id": generation_id, - "compatibility": { - "schema_id": "speakrs-long-run-train-v1", - "config_digest": super::configuration_digest(&request_bytes).unwrap(), - "source_digest": "b".repeat(64), - }, - "worker": worker, - "inventory": { - "schema_version": 1, - "binding": recovery_binding, - "entries": [ - { - "path": "checkpoint.bin", - "kind": "file", - "size": CHECKPOINT_BYTES.len(), - "digest": digest(CHECKPOINT_BYTES), - }, - { - "path": "recovery.json", - "kind": "file", - "size": RECOVERY_BYTES.len(), - "digest": digest(RECOVERY_BYTES), - }, - ], - }, - "state_digest": digest(RECOVERY_BYTES), - "position": { - "update_count": update_count, - "exposure_count": 0, - "sample_cursor_digest": "1".repeat(64), - "random_state_digest": "2".repeat(64), - }, - "cause": "periodic_checkpoint", - }, - }) - } - - fn write_generation( - published_root: &Path, - directory_name: &str, - generation_id: &str, - update_count: u64, - request_binding: &AttemptBinding, - recovery_binding: &AttemptBinding, - ) -> PathBuf { - write_record( - published_root, - directory_name, - &record_value( - request_binding, - recovery_binding, - generation_id, - json!(update_count), - ), - ) - } - - fn write_record(published_root: &Path, directory_name: &str, record: &Value) -> PathBuf { - let directory = published_root.join(directory_name); - fs::create_dir(&directory).unwrap(); - let request = record.get("request").unwrap(); - let binding: AttemptBinding = - serde_json::from_value(request.get("binding").unwrap().clone()).unwrap(); - let attempt_path = published_root - .parent() - .unwrap() - .join("attempts") - .join(&binding.attempt_id); - fs::create_dir_all(&attempt_path).unwrap(); - fs::write( - attempt_path.join(super::REQUEST_FILE), - serde_json::to_vec(request).unwrap(), - ) - .unwrap(); - fs::write(directory.join("checkpoint.bin"), b"checkpoint bytes").unwrap(); - fs::write(directory.join("recovery.json"), b"recovery state").unwrap(); - fs::write( - directory.join("segment-record.json"), - serde_json::to_vec(record).unwrap(), - ) - .unwrap(); - directory - } - - fn assert_generation( - result: PublishedRecoveryGeneration, - path: &Path, - generation_id: &str, - update_count: u64, - ) { - assert_eq!(result.path, path); - assert_eq!(result.generation_id, generation_id); - assert_eq!(result.committed_update_count, update_count); - } - - fn digest(bytes: &[u8]) -> String { - Sha256Digest::of(bytes).to_hex() - } - - fn worker_identity() -> Value { - json!({ - "worker_id": "trainer-worker", - "executable_digest": "a".repeat(64), - "source_digest": "b".repeat(64), - "environment_digest": "c".repeat(64), - }) - } - - fn output_declarations() -> Value { - json!([{"path": "checkpoint.bin", "kind": "file"}]) - } - - fn request_value(binding: &AttemptBinding) -> Value { - json!({ - "schema_version": 1, - "binding": binding, - "adapter": {}, - "start_mode": {}, - "payload": {}, - "input_view": { - "schema_version": 1, - "revision_id": binding.campaign_revision_id, - "root": "/trainer/input", - }, - "inputs": [], - "expected_outputs": output_declarations(), - "resources": {}, - "worker": worker_identity(), - }) - } - - fn completed_terminal_value(binding: &AttemptBinding, output: &[u8]) -> Value { - let worker = worker_identity(); - let output_digest = digest(output); - json!({ - "kind": "completed", - "terminal": { - "schema_version": 1, - "binding": binding, - "outcome": { - "kind": "completed", - "result": { - "schema_version": 1, - "binding": binding, - "worker": worker, - "config_digest": super::configuration_digest( - &serde_json::to_vec(&request_value(binding)).unwrap(), - ) - .unwrap(), - "expected_outputs": output_declarations(), - "outputs": { - "schema_version": 1, - "binding": binding, - "entries": [{ - "path": "checkpoint.bin", - "kind": "file", - "size": output.len(), - "digest": output_digest, - }], - }, - "metrics": [], - "evidence": { - "score_contract": "e".repeat(64), - "membership_digest": "f".repeat(64), - "source_exposures": [{ - "source_digest": "1".repeat(64), - "samples": 1, - }], - "update_points": [1], - "probe_points": [], - }, - "validation": { - "validator_id": "trainer-validator", - "validator_version": "1.0", - "evidence_digest": "2".repeat(64), - "validated_at": 1, - }, - }, - }, - }, - }) - } - - pub(crate) fn write_request_for_test(runtime_root: &Path, binding: &AttemptBinding) -> PathBuf { - let attempt_path = runtime_root.join("attempts").join(&binding.attempt_id); - fs::create_dir_all(&attempt_path).unwrap(); - let request_path = attempt_path.join("request.json"); - fs::write( - &request_path, - serde_json::to_vec(&request_value(binding)).unwrap(), - ) - .unwrap(); - request_path - } - - pub(crate) fn write_generation_for_test( - runtime_root: &Path, - binding: &AttemptBinding, - generation_id: &str, - update_count: u64, - ) -> PathBuf { - write_generation( - &published_root(runtime_root), - generation_id, - generation_id, - update_count, - binding, - binding, - ) - } - - pub(crate) fn write_completed_result_for_test( - runtime_root: &Path, - binding: &AttemptBinding, - ) -> PathBuf { - let published_root = published_root(runtime_root); - let publication_path = published_root.join(format!("result-{}", binding.attempt_id)); - fs::create_dir(&publication_path).unwrap(); - let output = b"checkpoint bytes"; - fs::write(publication_path.join("checkpoint.bin"), output).unwrap(); - - let attempt_path = runtime_root.join("attempts").join(&binding.attempt_id); - fs::create_dir_all(&attempt_path).unwrap(); - let request_path = attempt_path.join("request.json"); - if !request_path.exists() { - fs::write( - &request_path, - serde_json::to_vec(&request_value(binding)).unwrap(), - ) - .unwrap(); - } - fs::write( - &request_path, - serde_json::to_vec(&request_value(binding)).unwrap(), - ) - .unwrap(); - let terminal = serde_json::to_vec(&completed_terminal_value(binding, output)).unwrap(); - fs::write(attempt_path.join("terminal.json"), &terminal).unwrap(); - fs::write(publication_path.join("segment-record.json"), terminal).unwrap(); - publication_path - } - - fn write_completed_result(runtime_root: &Path, binding: &AttemptBinding) -> PathBuf { - write_completed_result_for_test(runtime_root, binding) - } - - #[test] - fn a_final_result_wins_when_a_new_checkpoint_appears_at_the_same_time() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let expected = expected_binding(); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - let published_root = published_root(&runtime_root); - write_generation( - &published_root, - "generation-a", - "generation-a", - 41, - &expected, - &expected, - ); - let result_path = write_completed_result(&runtime_root, &expected); - - let observation = observe_release( - &runtime_root, - &expected, - &baseline, - &TaskState::Running { pid: Some(42) }, - ) - .unwrap(); - - assert!(matches!( - observation, - WatchObservation::CompletedResultAwaitingTaskExit { - result, - new_checkpoint: Some(PublishedRecoveryGeneration { - generation_id, - .. - }), - } if result.publication_path == result_path && generation_id == "generation-a" - )); - } - - #[test] - fn a_final_result_before_task_exit_waits_without_requesting_a_stop() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let expected = expected_binding(); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - let result_path = write_completed_result(&runtime_root, &expected); - - let observation = observe_release( - &runtime_root, - &expected, - &baseline, - &TaskState::Running { pid: Some(42) }, - ) - .unwrap(); - - assert!(matches!( - observation, - WatchObservation::CompletedResultAwaitingTaskExit { - result, - new_checkpoint: None, - } if result.publication_path == result_path - )); - } - - #[test] - fn a_lost_task_needs_attention_even_without_publications() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let expected = expected_binding(); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - - let observation = - observe_release(&runtime_root, &expected, &baseline, &TaskState::Lost).unwrap(); - - assert!(matches!( - observation, - WatchObservation::Attention(WatcherAttention::LostTask { publications }) - if publications.new_checkpoint.is_none() - && publications.completed_result.is_none() - )); - } - - #[test] - fn a_failed_task_needs_attention_even_without_publications() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let expected = expected_binding(); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - - let observation = observe_release( - &runtime_root, - &expected, - &baseline, - &TaskState::Finished { - reason: ExitReason::Exit { code: 7 }, - }, - ) - .unwrap(); - - assert!(matches!( - observation, - WatchObservation::Attention(WatcherAttention::FailedTask { - reason: ExitReason::Exit { code: 7 }, - publications, - }) if publications.new_checkpoint.is_none() - && publications.completed_result.is_none() - )); - } - - #[test] - fn foreign_and_baseline_checkpoints_do_not_stop_the_running_task() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let published_root = published_root(&runtime_root); - let expected = expected_binding(); - write_generation( - &published_root, - "generation-baseline", - "generation-baseline", - 40, - &expected, - &expected, - ); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - let foreign = foreign_binding(); - write_generation( - &published_root, - "generation-foreign", - "generation-foreign", - 41, - &foreign, - &foreign, - ); - - let observation = observe_release( - &runtime_root, - &expected, - &baseline, - &TaskState::Running { pid: Some(42) }, - ) - .unwrap(); - - assert_eq!(observation, WatchObservation::WaitingForCheckpoint); - } - - #[test] - fn a_new_checkpoint_while_running_is_only_a_stop_request_candidate() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let expected = expected_binding(); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - let path = write_generation( - &published_root(&runtime_root), - "generation-a", - "generation-a", - 41, - &expected, - &expected, - ); - - let observation = observe_release( - &runtime_root, - &expected, - &baseline, - &TaskState::Running { pid: Some(42) }, - ) - .unwrap(); - - assert!(matches!( - observation, - WatchObservation::StopRequestCandidate { checkpoint } - if checkpoint.path == path - )); - } - - #[test] - fn a_successful_task_and_final_result_are_only_an_already_completed_candidate() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let expected = expected_binding(); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - let result_path = write_completed_result(&runtime_root, &expected); - - let observation = observe_release( - &runtime_root, - &expected, - &baseline, - &TaskState::Finished { - reason: ExitReason::Exit { code: 0 }, - }, - ) - .unwrap(); - - assert!(matches!( - observation, - WatchObservation::AlreadyCompletedCandidate { - result, - new_checkpoint: None, - } if result.publication_path == result_path - )); - } - - #[test] - fn a_checkpoint_before_the_task_starts_needs_attention() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let expected = expected_binding(); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - write_generation( - &published_root(&runtime_root), - "generation-a", - "generation-a", - 41, - &expected, - &expected, - ); - - let observation = - observe_release(&runtime_root, &expected, &baseline, &TaskState::Queued).unwrap(); - - assert!(matches!( - observation, - WatchObservation::Attention(WatcherAttention::PublicationBeforeTaskStart { - publications, - }) if publications.new_checkpoint.is_some() - && publications.completed_result.is_none() - )); - } - - #[test] - fn a_successful_task_and_checkpoint_without_a_result_need_attention() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let expected = expected_binding(); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - write_generation( - &published_root(&runtime_root), - "generation-a", - "generation-a", - 41, - &expected, - &expected, - ); - - let observation = observe_release( - &runtime_root, - &expected, - &baseline, - &TaskState::Finished { - reason: ExitReason::Exit { code: 0 }, - }, - ) - .unwrap(); - - assert!(matches!( - observation, - WatchObservation::Attention(WatcherAttention::SuccessfulTaskWithoutFinalResult { - new_checkpoint: Some(PublishedRecoveryGeneration { .. }), - }) - )); - } - - #[test] - fn snapshot_excludes_generations_that_were_already_published() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let published_root = published_root(&runtime_root); - let expected = expected_binding(); - write_generation( - &published_root, - "generation-a", - "generation-a", - 40, - &expected, - &expected, - ); - - let baseline = snapshot(&runtime_root, &expected).unwrap(); - - assert!( - find_new(&runtime_root, &expected, &baseline) - .unwrap() - .is_none() - ); - } - - #[test] - fn serialized_snapshot_restores_the_exact_restart_baseline() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let published_root = published_root(&runtime_root); - let expected = expected_binding(); - write_generation( - &published_root, - "generation-a", - "generation-a", - 40, - &expected, - &expected, - ); - - let baseline = snapshot(&runtime_root, &expected).unwrap(); - let persisted = serde_json::to_vec(&baseline).unwrap(); - let restored = serde_json::from_slice(&persisted).unwrap(); - assert_eq!(restored, baseline); - let new_path = write_generation( - &published_root, - "generation-b", - "generation-b", - 41, - &expected, - &expected, - ); - - assert_eq!( - find_new(&runtime_root, &expected, &restored) - .unwrap() - .unwrap() - .path, - new_path - ); - } - - #[test] - fn serialized_snapshot_rejects_duplicate_and_untrusted_generation_ids() { - let binding = expected_binding(); - let duplicate = json!({ - "schema": "trainer-recovery-snapshot-v1", - "binding": binding, - "generation_ids": ["generation-a", "generation-a"], - }); - assert!(serde_json::from_value::(duplicate).is_err()); - - let untrusted = json!({ - "schema": "trainer-recovery-snapshot-v1", - "binding": expected_binding(), - "generation_ids": ["../outside"], - }); - assert!(serde_json::from_value::(untrusted).is_err()); - } - - #[test] - fn completed_result_requires_matching_attempt_terminal_evidence_and_outputs() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let expected = expected_binding(); - let publication_path = write_completed_result(&runtime_root, &expected); - - let result = find_completed_result(&runtime_root, &expected) - .unwrap() - .unwrap(); - - assert_eq!(result.publication_path, publication_path); - assert_eq!( - result.terminal_path, - runtime_root.join("attempts/attempt-a/terminal.json") - ); - assert_eq!(result.binding, expected); - assert_eq!(result.output_paths, ["checkpoint.bin"]); - } - - #[test] - fn another_attempts_result_is_not_accepted() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let foreign = AttemptBinding { - attempt_id: "attempt-b".into(), - ..expected_binding() - }; - write_completed_result(&runtime_root, &foreign); - - assert!( - find_completed_result(&runtime_root, &expected_binding()) - .unwrap() - .is_none() - ); - } - - #[test] - fn completed_result_with_wrong_attempt_binding_requires_attention() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let expected = expected_binding(); - let publication_path = write_completed_result(&runtime_root, &expected); - let foreign = foreign_binding(); - let terminal = - serde_json::to_vec(&completed_terminal_value(&foreign, b"checkpoint bytes")).unwrap(); - fs::write(publication_path.join("segment-record.json"), &terminal).unwrap(); - fs::write( - runtime_root.join("attempts/attempt-a/terminal.json"), - terminal, - ) - .unwrap(); - - assert!(matches!( - find_completed_result(&runtime_root, &expected), - Err(WatcherError::MalformedPublication { .. }) - )); - } - - #[test] - fn completed_result_with_malformed_output_requires_attention() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let expected = expected_binding(); - let publication_path = write_completed_result(&runtime_root, &expected); - fs::write(publication_path.join("checkpoint.bin"), b"changed bytes").unwrap(); - - assert!(matches!( - find_completed_result(&runtime_root, &expected), - Err(WatcherError::MalformedPublication { .. }) - )); - } - - #[test] - fn symlinked_terminal_evidence_requires_attention() { - use std::os::unix::fs::symlink; - - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let expected = expected_binding(); - write_completed_result(&runtime_root, &expected); - let terminal_path = runtime_root.join("attempts/attempt-a/terminal.json"); - let external_terminal = temp.path().join("external-terminal.json"); - fs::rename(&terminal_path, &external_terminal).unwrap(); - symlink(&external_terminal, terminal_path).unwrap(); - - assert!(matches!( - find_completed_result(&runtime_root, &expected), - Err(WatcherError::Symlink { .. }) - )); - } - - #[test] - fn find_new_returns_a_complete_matching_publication() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let published_root = published_root(&runtime_root); - let expected = expected_binding(); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - let path = write_generation( - &published_root, - "generation-a", - "generation-a", - 41, - &expected, - &expected, - ); - - let result = find_new(&runtime_root, &expected, &baseline) - .unwrap() - .unwrap(); - - assert_generation(result, &path, "generation-a", 41); - } - - #[test] - fn foreign_attempt_publications_are_not_in_the_baseline_or_new_result() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let published_root = published_root(&runtime_root); - let expected = expected_binding(); - let foreign = foreign_binding(); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - write_generation( - &published_root, - "generation-foreign-request", - "generation-foreign-request", - 42, - &foreign, - &expected, - ); - write_generation( - &published_root, - "generation-foreign-recovery", - "generation-foreign-recovery", - 43, - &expected, - &foreign, - ); - - assert!( - find_new(&runtime_root, &expected, &baseline) - .unwrap() - .is_none() - ); - } - - #[test] - fn trainer_staging_and_terminal_result_publications_are_ignored() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let published_root = published_root(&runtime_root); - fs::create_dir(published_root.join(".publish-staging")).unwrap(); - fs::write( - published_root.join(".publish-staging/segment-record.json"), - b"not a recovery record", - ) - .unwrap(); - fs::create_dir(published_root.join("result-attempt-a")).unwrap(); - fs::write( - published_root.join("result-attempt-a/segment-record.json"), - b"not a recovery record", - ) - .unwrap(); - let expected = expected_binding(); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - - assert!( - find_new(&runtime_root, &expected, &baseline) - .unwrap() - .is_none() - ); - } - - #[test] - fn new_generation_at_the_baseline_update_count_is_still_new() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let published_root = published_root(&runtime_root); - let expected = expected_binding(); - write_generation( - &published_root, - "generation-a", - "generation-a", - 50, - &expected, - &expected, - ); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - let path = write_generation( - &published_root, - "generation-b", - "generation-b", - 50, - &expected, - &expected, - ); - - let result = find_new(&runtime_root, &expected, &baseline) - .unwrap() - .unwrap(); - - assert_generation(result, &path, "generation-b", 50); - } - - #[test] - fn malformed_matching_publication_requires_attention() { - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let published_root = published_root(&runtime_root); - let expected = expected_binding(); - let baseline = snapshot(&runtime_root, &expected).unwrap(); - write_record( - &published_root, - "generation-a", - &record_value( - &expected, - &expected, - "generation-a", - json!("not-an-update-count"), - ), - ); - - assert!(matches!( - find_new(&runtime_root, &expected, &baseline), - Err(WatcherError::MalformedPublication { .. }) - )); - } - - #[test] - fn symlinked_publication_root_requires_attention() { - use std::os::unix::fs::symlink; - - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - fs::create_dir_all(&runtime_root).unwrap(); - let external_root = temp.path().join("external-published"); - fs::create_dir(&external_root).unwrap(); - symlink(&external_root, runtime_root.join("published")).unwrap(); - - assert!(matches!( - snapshot(&runtime_root, &expected_binding()), - Err(WatcherError::Symlink { .. }) - )); - } - - #[test] - fn symlinked_generation_directory_requires_attention() { - use std::os::unix::fs::symlink; - - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let published_root = published_root(&runtime_root); - let external_generation = temp.path().join("external-generation"); - fs::create_dir(&external_generation).unwrap(); - symlink(&external_generation, published_root.join("generation-a")).unwrap(); - - assert!(matches!( - snapshot(&runtime_root, &expected_binding()), - Err(WatcherError::Symlink { .. }) - )); - } - - #[test] - fn symlinked_record_file_requires_attention() { - use std::os::unix::fs::symlink; - - let temp = tempfile::tempdir().unwrap(); - let runtime_root = runtime_root(&temp); - let published_root = published_root(&runtime_root); - let expected = expected_binding(); - let generation = published_root.join("generation-a"); - fs::create_dir(&generation).unwrap(); - let external_record = temp.path().join("external-record.json"); - fs::write( - &external_record, - serde_json::to_vec(&record_value( - &expected, - &expected, - "generation-a", - json!(60), - )) - .unwrap(), - ) - .unwrap(); - symlink(external_record, generation.join("segment-record.json")).unwrap(); - - assert!(matches!( - snapshot(&runtime_root, &expected), - Err(WatcherError::Symlink { .. }) - )); - } - - #[test] - fn configuration_digest_hashes_the_exact_member_bytes() { - let request = br#"{ "schema_version" : 1, "resources":{"accelerator" :"cuda"}, - "adapter" : {"kind":"speakrs"} , "start_mode":{ "kind": "fresh" }, - "payload": {"epochs": 4.0, "rate": 1e-3}, - "input_view": {"root": "/in", "revision_id" : "revision-a", "schema_version": 1}, - "inputs": [ ], "expected_outputs": [{"path":"a.bin","kind":"file"}] }"#; - let expected = concat!( - r#"[{"kind":"speakrs"},{ "kind": "fresh" },{"epochs": 4.0, "rate": 1e-3},"#, - r#"{"schema_version":1,"revision_id":"revision-a"},[ ],"#, - r#"[{"path":"a.bin","kind":"file"}],{"accelerator" :"cuda"}]"#, - ); - - assert_eq!( - super::configuration_digest(request).unwrap(), - Sha256Digest::of(expected) - ); - } - - #[test] - fn configuration_digest_refuses_duplicate_members() { - let request = request_value(&expected_binding()); - let document = serde_json::to_string(&request).unwrap(); - let duplicate_top = document.replacen('{', r#"{"payload":{},"#, 1); - let duplicate_view = document.replacen( - r#""input_view":{"#, - r#""input_view":{"revision_id":"other","#, - 1, - ); - - assert!(super::configuration_digest(document.as_bytes()).is_ok()); - assert!(super::configuration_digest(duplicate_top.as_bytes()).is_err()); - assert!(super::configuration_digest(duplicate_view.as_bytes()).is_err()); - } -} diff --git a/src/runner/container.rs b/src/runner/container.rs index 7a76350..bd89465 100644 --- a/src/runner/container.rs +++ b/src/runner/container.rs @@ -63,15 +63,10 @@ pub(super) async fn run( row, ); } - let resource = store.resource_for_task(id).unwrap_or_else(|error| { - warn!(%id, "read resource of container task: {error}"); - None - }); let args = create_args( workload, &CreateContext { task: id, - resource, cidfile: &paths.container_cid, default_user: ContainerUser::current(), }, diff --git a/src/spec.rs b/src/spec.rs index bbf29a3..a81b253 100644 --- a/src/spec.rs +++ b/src/spec.rs @@ -141,7 +141,7 @@ fn workload_schema(generator: &mut schemars::SchemaGenerator) -> schemars::Schem let agent_kind = subschema::(generator); let resume_thread = subschema::>(generator); let command = subschema::(generator); - let container = container_workload_schema(false); + let container = container_workload_schema(); schemars::json_schema!({ "description": "Workload variant. Exactly one of agent, task, or container.", "oneOf": [ diff --git a/src/store.rs b/src/store.rs index b874595..8abd08f 100644 --- a/src/store.rs +++ b/src/store.rs @@ -21,13 +21,12 @@ use crate::domain::{ }; use crate::error::AppError; use crate::events::EventPayload; -use crate::events::{DeliveryState, EventError, TaskEvent}; +use crate::events::{DeliveryState, TaskEvent}; use crate::machine::MachineId; -use crate::resource::store::{RESOURCE_SCHEMA, open_missing_return_windows_on}; use crate::spec::NormalizedSpec; use crate::submission::{ CallbackContext, CallbackExecutable, ExecutionRecord, ExecutorIdentity, OriginRoute, - PersistedSpec, RequestId, ResourceRoutePhase, SubmissionState, + PersistedSpec, RequestId, SubmissionState, }; mod cancellation; @@ -36,35 +35,10 @@ mod dependency; mod events; mod identity; mod message; -mod resource; pub use container::TaskContainerRecord; pub use dependency::{HeldCancel, UnlaunchedTask}; pub(crate) use events::EventRetentionBatch; -pub use identity::{IdentityError, ResourceActionRouteResult, ResourceBackgroundRouteResult}; -pub(crate) use resource::VerifiedReleaseProof; -#[cfg(test)] -pub(crate) use resource::test_support::{ - mark_request_assigned_for_race, unreceipted_release_provenance, -}; -pub(crate) use resource::{ - AcceptedActionTask, EndedRestoreResolution, PreparedReturnTask, - RemoteReleaseWatcherAcceptanceInput, ResourceActionError, RestoreReconcileOutcome, - ReturnClosure, ReturnDecisionError, ReturnTaskAcceptance, ReturnTaskAcceptanceInput, - ReturnTaskOrigin, -}; -pub(crate) use resource::{ - BackgroundLaunchAcceptance, BackgroundLaunchError, BackgroundLaunchInput, - BackgroundLaunchPhase, BackgroundLaunchView, RemoteBackgroundLaunchInput, - idle_boundary_decision_on, idle_opening_matches_on, open_idle_serving_loan_on, - pending_background_launch_on, promote_started_background_launch_on, -}; -pub(crate) use resource::{ - InitialIdleError, OperatorGpuFreeError, operator_serving_release_matches_on, -}; -pub(crate) use resource::{ - ResourceControlEffect, ResourceControlError, ResourceControlRequest, ResourceControlStart, - ResourceReadModel, SupervisorReplacement, open_action_id, -}; +pub use identity::IdentityError; /// Released version 2 schema /// @@ -119,8 +93,8 @@ ALTER TABLE tasks ADD COLUMN name TEXT; /// Schema version of the v0.4.0 release const RELEASED_V0_4_SCHEMA_VERSION: i64 = 27; -/// Everything added after the released version 2 schema, except the resource -/// tables that `RESOURCE_SCHEMA` owns +/// Everything the v0.4.0 schema added to the released version 2 schema, apart +/// from its GPU loan tables, which a later version drops /// /// Versions 3 through 26 were never released, so a version 2 database moves to /// the current schema in one step @@ -242,45 +216,6 @@ const TASK_SELECT: &str = "SELECT id, thread_id, name, workload_json, cwd, timeo process_group_exit_evidence, container_exit_evidence FROM tasks"; -/// Operator attestation receipts gained the Restoring return outcomes after v0.4.0 -/// -/// SQLite cannot change a CHECK constraint in place, so the table is rebuilt -/// Existing receipts keep their bytes, and `RESOURCE_SCHEMA` recreates the -/// resource index that the dropped table owned -const MIGRATE_27_TO_28: &str = r" -CREATE TABLE resource_operator_attestations_v28 ( - operation_id TEXT PRIMARY KEY NOT NULL, - resource_id TEXT NOT NULL REFERENCES resources(id), - task_id TEXT NOT NULL UNIQUE, - preceding_loan TEXT, - preceding_launch TEXT, - receipt_json TEXT NOT NULL CHECK ( - json_valid(receipt_json) - AND COALESCE(json_type(receipt_json) = 'object', 0) - AND COALESCE(json_extract(receipt_json, '$.attestation.operation_id') = operation_id, 0) - AND COALESCE(json_extract(receipt_json, '$.attestation.resource_id') = resource_id, 0) - AND COALESCE(json_extract(receipt_json, '$.attestation.task_id') = task_id, 0) - AND COALESCE( - json_extract(receipt_json, '$.attestation.confirmation') = 'operator_confirmed_gpu_free', - 0 - ) - AND COALESCE(length(trim(json_extract(receipt_json, '$.attestation.observation'))) > 0, 0) - AND COALESCE(json_type(receipt_json, '$.evidence') = 'object', 0) - AND COALESCE(json_extract(receipt_json, '$.outcome.type') IN ( - 'release_resolved_serving', 'release_resolved_return_required', - 'idle_serving', 'idle_boundary', - 'restore_closed_serving', 'restore_closed_idle_boundary' - ), 0) - ) -); -INSERT INTO resource_operator_attestations_v28 - (operation_id, resource_id, task_id, preceding_loan, preceding_launch, receipt_json) -SELECT operation_id, resource_id, task_id, preceding_loan, preceding_launch, receipt_json -FROM resource_operator_attestations ORDER BY rowid; -DROP TABLE resource_operator_attestations; -ALTER TABLE resource_operator_attestations_v28 RENAME TO resource_operator_attestations; -"; - /// Schema version of the v0.5.0 release const RELEASED_V0_5_SCHEMA_VERSION: i64 = 28; @@ -303,177 +238,30 @@ CREATE TABLE task_containers ( ); "; -/// Resource requests and restore closures gained container work after v0.5.0 -/// -/// SQLite cannot change a CHECK constraint in place, so both tables are -/// rebuilt with the constraints that `RESOURCE_SCHEMA` now declares. Existing -/// rows keep their bytes, and `RESOURCE_SCHEMA` recreates the dropped indexes -const MIGRATE_28_TO_29_RESOURCES: &str = r" -CREATE TABLE resource_requests_v29 ( - acceptance_sequence INTEGER PRIMARY KEY AUTOINCREMENT CHECK (acceptance_sequence > 0), - request_id TEXT NOT NULL UNIQUE, - task_id TEXT NOT NULL UNIQUE, - resource_id TEXT NOT NULL REFERENCES resources(id), - origin_machine TEXT NOT NULL, - spec_json TEXT NOT NULL CHECK ( - json_valid(spec_json) - AND COALESCE(json_type(spec_json) = 'object', 0) - AND COALESCE(json_type(spec_json, '$.api_version') = 'integer', 0) - AND COALESCE(json_type(spec_json, '$.thread') = 'text', 0) - AND COALESCE(json_type(spec_json, '$.name') = 'text', 0) - AND COALESCE(json_type(spec_json, '$.cwd') = 'text', 0) - AND COALESCE(json_type(spec_json, '$.timeout') = 'text', 0) - AND COALESCE(json_type(spec_json, '$.workload') = 'object', 0) - AND COALESCE( - ( - json_extract(spec_json, '$.workload.type') = 'task' - AND json_type(spec_json, '$.workload.command') = 'array' - ) OR ( - json_extract(spec_json, '$.workload.type') = 'container' - AND json_type(spec_json, '$.workload.image') = 'text' - AND json_type(spec_json, '$.workload.gpus') IS NOT NULL - ), - 0 - ) - ), - state_json TEXT NOT NULL CHECK ( - json_valid(state_json) - AND COALESCE(json_type(state_json) = 'object', 0) - AND COALESCE(json_type(state_json, '$.type') = 'text', 0) - AND COALESCE(json_extract(state_json, '$.type') IN ( - 'queued', 'assigned', 'finished', 'cancelled_before_launch', 'rejected' - ), 0) - ) -); -INSERT INTO resource_requests_v29 - (acceptance_sequence, request_id, task_id, resource_id, origin_machine, spec_json, state_json) -SELECT acceptance_sequence, request_id, task_id, resource_id, origin_machine, spec_json, state_json -FROM resource_requests ORDER BY acceptance_sequence; -DROP TABLE resource_requests; -ALTER TABLE resource_requests_v29 RENAME TO resource_requests; - -CREATE TABLE resource_restore_closures_v29 ( - action_id TEXT PRIMARY KEY REFERENCES resource_return_decisions(action_id), - task_id TEXT NOT NULL UNIQUE, - receipt_json TEXT NOT NULL CHECK ( - json_valid(receipt_json) - AND COALESCE(json_type(receipt_json) = 'object', 0) - AND COALESCE(json_extract(receipt_json, '$.action_id') = action_id, 0) - AND COALESCE(json_extract(receipt_json, '$.task_id') = task_id, 0) - AND COALESCE(json_extract(receipt_json, '$.basis.type') IN ( - 'confirmed_running', 'foreground_ended', 'container_ended', 'supervisor_resolved_end' - ), 0) - ) -); -INSERT INTO resource_restore_closures_v29 (action_id, task_id, receipt_json) -SELECT action_id, task_id, receipt_json FROM resource_restore_closures ORDER BY rowid; -DROP TABLE resource_restore_closures; -ALTER TABLE resource_restore_closures_v29 RENAME TO resource_restore_closures; -"; - /// Move a released version 2 database to the current schema -/// -/// The version 2 schema has no resource tables, so `RESOURCE_SCHEMA` creates -/// them with the current constraints fn migrate_2_to_current(conn: &Connection) -> Result<(), rusqlite::Error> { conn.execute_batch(MIGRATE_2_TO_CURRENT)?; - conn.execute_batch(MIGRATE_28_TO_29_TASKS)?; - conn.execute_batch(RESOURCE_SCHEMA)?; - migrate_32_to_current(conn) -} - -/// Move a v0.4.0 database to the current schema -fn migrate_27_to_current(conn: &Connection) -> Result<(), rusqlite::Error> { - conn.execute_batch(MIGRATE_27_TO_28)?; migrate_28_to_current(conn) } -/// Move a v0.5.0 database to the current schema +/// Move a v0.4.0 or v0.5.0 database to the current schema +/// +/// Apart from the container witness, every change before version 32 touched +/// only the GPU loan tables, which the version 35 step drops fn migrate_28_to_current(conn: &Connection) -> Result<(), rusqlite::Error> { conn.execute_batch(MIGRATE_28_TO_29_TASKS)?; - conn.execute_batch(MIGRATE_28_TO_29_RESOURCES)?; - migrate_30_to_current(conn) + migrate_32_to_current(conn) } /// Schema version of the v0.5.1 release const RELEASED_V0_5_1_SCHEMA_VERSION: i64 = 29; -/// Move a v0.5.1 database to the current schema -fn migrate_29_to_current(conn: &Connection) -> Result<(), rusqlite::Error> { - migrate_30_to_current(conn) -} - /// Released v0.7.0 and v0.6.0 databases use schema version 30 const RELEASED_V0_7_SCHEMA_VERSION: i64 = 30; -/// Add serving ranks before `RESOURCE_SCHEMA` creates serving-order indexes -/// -/// Rebuilding the table keeps the upgraded column constraints identical to a -/// fresh database. Existing requests retain their acceptance order. -const MIGRATE_30_TO_CURRENT: &str = r" -CREATE TABLE resource_requests_v31 ( - acceptance_sequence INTEGER PRIMARY KEY AUTOINCREMENT CHECK (acceptance_sequence > 0), - queue_rank INTEGER NOT NULL CHECK (queue_rank > 0), - request_id TEXT NOT NULL UNIQUE, - task_id TEXT NOT NULL UNIQUE, - resource_id TEXT NOT NULL REFERENCES resources(id), - origin_machine TEXT NOT NULL, - spec_json TEXT NOT NULL CHECK ( - json_valid(spec_json) - AND COALESCE(json_type(spec_json) = 'object', 0) - AND COALESCE(json_type(spec_json, '$.api_version') = 'integer', 0) - AND COALESCE(json_type(spec_json, '$.thread') = 'text', 0) - AND COALESCE(json_type(spec_json, '$.name') = 'text', 0) - AND COALESCE(json_type(spec_json, '$.cwd') = 'text', 0) - AND COALESCE(json_type(spec_json, '$.timeout') = 'text', 0) - AND COALESCE(json_type(spec_json, '$.workload') = 'object', 0) - AND COALESCE( - ( - json_extract(spec_json, '$.workload.type') = 'task' - AND json_type(spec_json, '$.workload.command') = 'array' - ) OR ( - json_extract(spec_json, '$.workload.type') = 'container' - AND json_type(spec_json, '$.workload.image') = 'text' - AND json_type(spec_json, '$.workload.gpus') IS NOT NULL - ), - 0 - ) - ), - state_json TEXT NOT NULL CHECK ( - json_valid(state_json) - AND COALESCE(json_type(state_json) = 'object', 0) - AND COALESCE(json_type(state_json, '$.type') = 'text', 0) - AND COALESCE(json_extract(state_json, '$.type') IN ( - 'queued', 'assigned', 'finished', 'cancelled_before_launch', 'rejected' - ), 0) - ) -); -INSERT INTO resource_requests_v31 - (acceptance_sequence, queue_rank, request_id, task_id, resource_id, - origin_machine, spec_json, state_json) -SELECT acceptance_sequence, acceptance_sequence, request_id, task_id, resource_id, - origin_machine, spec_json, state_json -FROM resource_requests ORDER BY acceptance_sequence; -DROP TABLE resource_requests; -ALTER TABLE resource_requests_v31 RENAME TO resource_requests; -"; - -/// Move a released schema version 30 database to the current schema -fn migrate_30_to_current(conn: &Connection) -> Result<(), rusqlite::Error> { - conn.execute_batch(MIGRATE_30_TO_CURRENT)?; - migrate_31_to_current(conn) -} - /// Released v0.8 databases use schema version 31 const RELEASED_V0_8_SCHEMA_VERSION: i64 = 31; -/// Add return decision windows, and open one for each loan that already awaits a return -fn migrate_31_to_current(conn: &Connection) -> Result<(), rusqlite::Error> { - conn.execute_batch(RESOURCE_SCHEMA)?; - open_missing_return_windows_on(conn, Utc::now())?; - migrate_32_to_current(conn) -} - /// Released v0.8.7 databases use schema version 32 const RELEASED_V0_8_7_SCHEMA_VERSION: i64 = 32; @@ -545,7 +333,76 @@ WHERE outcome IS NULL /// Move a v0.11 database to the current schema fn migrate_33_to_current(conn: &Connection) -> Result<(), rusqlite::Error> { conn.execute_batch(MIGRATE_33_TO_34)?; - conn.execute_batch(BACKFILL_33_TO_34) + conn.execute_batch(BACKFILL_33_TO_34)?; + migrate_34_to_current(conn) +} + +/// Released v0.13 and v0.14 databases use schema version 34 +const RELEASED_V0_13_SCHEMA_VERSION: i64 = 34; + +/// Remove the GPU loan subsystem, which never ran live work +/// +/// Origin routes, their callback rows, cancellation intents, and notice +/// messages of the loan flows go first, since no current type can read them. +/// Ordinary tasks, including any a loan launched, keep their rows. The loan +/// tables then go children first, so no foreign key sees a missing parent, and +/// `IF EXISTS` covers older databases that never had some of them +const MIGRATE_34_TO_35: &str = r" +DELETE FROM origin_inbox WHERE task_id IN ( + SELECT task_id FROM origin_routes + WHERE json_extract(route_json, '$.submission.type') + IN ('resource', 'resource_action', 'resource_background') +); +DELETE FROM origin_event_receipts WHERE task_id IN ( + SELECT task_id FROM origin_routes + WHERE json_extract(route_json, '$.submission.type') + IN ('resource', 'resource_action', 'resource_background') +); +DELETE FROM cancellation_requests + WHERE json_extract(request_json, '$.target.type') = 'resource' + OR json_extract(request_json, '$.delivery.state') = 'resource_delivered' + OR task_id IN ( + SELECT task_id FROM origin_routes + WHERE json_extract(route_json, '$.submission.type') + IN ('resource', 'resource_action', 'resource_background') + ); +DELETE FROM origin_routes + WHERE json_extract(route_json, '$.submission.type') + IN ('resource', 'resource_action', 'resource_background'); + +DELETE FROM message_receipts WHERE message_id IN ( + SELECT message_id FROM message_attempts + WHERE json_extract(attempt_json, '$.request.source.kind') = 'resource_notice' +); +DELETE FROM message_attempts + WHERE json_extract(attempt_json, '$.request.source.kind') = 'resource_notice'; + +DROP TABLE IF EXISTS resource_return_deadline_servings; +DROP TABLE IF EXISTS resource_return_windows; +DROP TABLE IF EXISTS resource_restore_closures; +DROP TABLE IF EXISTS resource_return_decisions; +DROP TABLE IF EXISTS resource_supervisor_notices; +DROP TABLE IF EXISTS resource_idle_openings; +DROP TABLE IF EXISTS resource_release_checkpoint_states; +DROP TABLE IF EXISTS resource_release_completions; +DROP TABLE IF EXISTS resource_task_completions; +DROP TABLE IF EXISTS resource_action_task_receipts; +DROP TABLE IF EXISTS resource_background_launches; +DROP TABLE IF EXISTS resource_control_operations; +DROP TABLE IF EXISTS resource_operator_attestations; +DROP TABLE IF EXISTS resource_initial_idle_attestations; +DROP TABLE IF EXISTS resource_registration_receipts; +DROP TABLE IF EXISTS resource_cancellation_receipts; +DROP TABLE IF EXISTS resource_request_preventions; +DROP TABLE IF EXISTS resource_requests; +DROP TABLE IF EXISTS trainer_attempt_associations; +DROP TABLE IF EXISTS loans; +DROP TABLE IF EXISTS resources; +"; + +/// Move a v0.13 database to the current schema +fn migrate_34_to_current(conn: &Connection) -> Result<(), rusqlite::Error> { + conn.execute_batch(MIGRATE_34_TO_35) } /// Why `Store::open` refuses a database version @@ -567,43 +424,6 @@ pub struct Store { tasks_dir: std::path::PathBuf, } -fn resource_task_id_is_reserved(conn: &Connection, task: TaskId) -> Result { - if resource_request_task_id_is_reserved(conn, task)? { - return Ok(true); - } - release_watcher_task_id_is_reserved(conn, task) -} - -fn resource_request_task_id_is_reserved( - conn: &Connection, - task: TaskId, -) -> Result { - conn.query_row( - "SELECT EXISTS( - SELECT 1 FROM resource_requests WHERE task_id = ?1 - UNION ALL - SELECT 1 FROM resource_request_preventions WHERE task_id = ?1 - )", - [task.to_string()], - |row| row.get(0), - ) -} - -fn release_watcher_task_id_is_reserved( - conn: &Connection, - task: TaskId, -) -> Result { - conn.query_row( - "SELECT EXISTS( - SELECT 1 FROM loans - WHERE json_extract(state_json, '$.phase.watcher_intent.watcher_task_id') = ?1 - OR json_extract(state_json, '$.last_safe_phase.watcher_intent.watcher_task_id') = ?1 - )", - [task.to_string()], - |row| row.get(0), - ) -} - fn insert_task_with_project_root_on( conn: &Connection, row: &TaskRow, @@ -676,30 +496,7 @@ fn reject_busy_resume_thread_on(conn: &Connection, row: &TaskRow) -> Result<(), Ok(()) } -pub(crate) fn task_by_id_on(conn: &Connection, id: TaskId) -> Result, AppError> { - let mut statement = conn.prepare(&format!("{TASK_SELECT} WHERE id = ?1"))?; - Ok(statement - .query_row(params![id.to_string()], parse_task_row) - .optional()?) -} - -pub(crate) fn executor_identity_for_resource_task_on( - conn: &Connection, - task: TaskId, -) -> Result, IdentityError> { - identity::executor_identity_on(conn, task) -} - -pub(crate) fn initial_queued_event_matches_on( - conn: &Connection, - task: TaskId, - origin_machine: MachineId, - execution_machine: MachineId, -) -> Result { - events::initial_queued_event_matches_on(conn, task, origin_machine, execution_machine) -} - -pub(super) fn validate_local_task_acceptance( +fn validate_local_task_acceptance( row: &TaskRow, spec: &NormalizedSpec, callback: &CallbackContext, @@ -732,16 +529,8 @@ fn insert_local_task_records_on( machine: MachineId, request: RequestId, callback: &CallbackContext, - allowed_watcher_task: Option, ) -> Result<(), AppError> { validate_local_task_acceptance(row, spec, callback)?; - if resource_request_task_id_is_reserved(conn, row.id)? - || (allowed_watcher_task != Some(row.id) - && release_watcher_task_id_is_reserved(conn, row.id)?) - { - return Err(AppError::ClusterTaskConflict { task: row.id }); - } - let project_root = find_project_root(&row.cwd); insert_task_with_project_root_on(conn, row, project_root.as_deref())?; let route = OriginRoute { @@ -840,11 +629,6 @@ fn release_held_local_task_records_on( return Err(AppError::ClusterTaskConflict { task: row.id }); } validate_local_task_acceptance(row, spec, &route.callback)?; - if resource_request_task_id_is_reserved(conn, row.id)? - || release_watcher_task_id_is_reserved(conn, row.id)? - { - return Err(AppError::ClusterTaskConflict { task: row.id }); - } let project_root = find_project_root(&row.cwd); insert_task_with_project_root_on(conn, row, project_root.as_deref())?; @@ -880,126 +664,6 @@ fn release_held_local_task_records_on( Ok(()) } -/// Fixed owners of one authority task whose callback route lives on another machine -pub(crate) struct RemoteOriginTask { - /// Stable retry identity saved in the remote route - pub(crate) request_id: RequestId, - /// Supervisor machine that owns the callback route - pub(crate) origin_machine: MachineId, - /// Authority machine that executes the task - pub(crate) execution_machine: MachineId, - /// Supervisor thread named by the spec - pub(crate) thread: ThreadId, - /// Whether the task is the release watcher that its own action reserved - pub(crate) reserved_watcher: bool, -} - -/// Insert one action-bound task whose callback route lives on the supervisor machine -/// -/// The task row, accepted remote executor identity, first queued event, and exact -/// action receipt commit in the caller's transaction. No origin route is written -/// here because the supervisor machine saved it before sending the launch -pub(crate) fn insert_remote_action_task_records_on( - conn: &Connection, - row: &TaskRow, - spec: &NormalizedSpec, - receipt: &crate::resource::bound_action::ActionTaskReceipt, -) -> Result<(), AppError> { - if row.id != receipt.task_id { - return Err(AppError::ClusterTaskConflict { task: row.id }); - } - insert_remote_origin_task_records_on( - conn, - row, - spec, - &RemoteOriginTask { - request_id: receipt.request_id, - origin_machine: receipt.origin_machine(), - execution_machine: receipt.execution_machine(), - thread: receipt.authority.supervisor.thread, - // only the watcher bound by this exact action may use its reserved identity - reserved_watcher: receipt.kind - == crate::resource::bound_action::ResourceActionKind::ReleaseWatcher, - }, - ) -} - -/// Insert one authority task whose callback route lives on a remote supervisor machine -/// -/// The task row, accepted remote executor identity, and first queued event commit -/// in the caller's transaction. The caller saves its own receipt in the same one -pub(crate) fn insert_remote_origin_task_records_on( - conn: &Connection, - row: &TaskRow, - spec: &NormalizedSpec, - owners: &RemoteOriginTask, -) -> Result<(), AppError> { - let task = row.id; - let origin = owners.origin_machine; - let execution = owners.execution_machine; - if origin == execution - || spec.machine.is_some() - || spec.thread != owners.thread - || matches!(&spec.workload, crate::spec::NormalizedWorkload::Agent(_)) - || row.status() != ProcessStatus::Queued - || row.name.as_ref() != Some(&spec.name) - || row.thread != spec.thread - || row.workload != crate::invocation::persist_workload(&spec.workload) - || row.cwd != spec.cwd - || row.timeout != spec.timeout - || !row.cwd.is_absolute() - || !row.binary.is_absolute() - { - return Err(AppError::ClusterTaskConflict { task }); - } - let watcher_reserved = release_watcher_task_id_is_reserved(conn, task)?; - if resource_request_task_id_is_reserved(conn, task)? - || (watcher_reserved && !owners.reserved_watcher) - { - return Err(AppError::ClusterTaskConflict { task }); - } - let occupied: bool = conn.query_row( - "SELECT EXISTS( - SELECT 1 FROM tasks WHERE id=?1 - UNION ALL SELECT 1 FROM executor_identities WHERE task_id=?1 - UNION ALL SELECT 1 FROM origin_routes WHERE task_id=?1 OR request_id=?2 - UNION ALL SELECT 1 FROM executor_outbox WHERE task_id=?1 - UNION ALL SELECT 1 FROM executor_event_receipts WHERE task_id=?1 - )", - params![task.to_string(), owners.request_id.0.to_string()], - |entry| entry.get(0), - )?; - if occupied { - return Err(AppError::ClusterTaskConflict { task }); - } - - let project_root = find_project_root(&row.cwd); - insert_task_with_project_root_on(conn, row, project_root.as_deref())?; - let identity = ExecutorIdentity::Accepted(ExecutionRecord { - task, - origin_machine: origin, - execution_machine: execution, - spec: spec.clone().into(), - state: ProcessStatus::Queued, - }); - conn.execute( - "INSERT INTO executor_identities (task_id,origin_machine,identity_json) VALUES (?1,?2,?3)", - params![ - task.to_string(), - origin.to_string(), - serde_json::to_string(&identity)? - ], - )?; - events::append_produced_event_on( - conn, - task, - EventPayload::State { - status: ProcessStatus::Queued, - }, - )?; - Ok(()) -} - /// Both machine owners of one accepted execution #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct TaskOwners { @@ -1062,13 +726,15 @@ impl Store { migrate_2_to_current(&transaction)?; } 2 => migrate_2_to_current(&transaction)?, - RELEASED_V0_4_SCHEMA_VERSION => migrate_27_to_current(&transaction)?, - RELEASED_V0_5_SCHEMA_VERSION => migrate_28_to_current(&transaction)?, - RELEASED_V0_5_1_SCHEMA_VERSION => migrate_29_to_current(&transaction)?, - RELEASED_V0_7_SCHEMA_VERSION => migrate_30_to_current(&transaction)?, - RELEASED_V0_8_SCHEMA_VERSION => migrate_31_to_current(&transaction)?, - RELEASED_V0_8_7_SCHEMA_VERSION => migrate_32_to_current(&transaction)?, + RELEASED_V0_4_SCHEMA_VERSION | RELEASED_V0_5_SCHEMA_VERSION => { + migrate_28_to_current(&transaction)?; + } + RELEASED_V0_5_1_SCHEMA_VERSION + | RELEASED_V0_7_SCHEMA_VERSION + | RELEASED_V0_8_SCHEMA_VERSION + | RELEASED_V0_8_7_SCHEMA_VERSION => migrate_32_to_current(&transaction)?, RELEASED_V0_11_SCHEMA_VERSION => migrate_33_to_current(&transaction)?, + RELEASED_V0_13_SCHEMA_VERSION => migrate_34_to_current(&transaction)?, other => return Err(unsupported_schema_version(other)), } transaction.pragma_update(None, "user_version", SCHEMA_VERSION)?; @@ -1106,7 +772,7 @@ impl Store { codex, }; self.immediate(|| { - insert_local_task_records_on(&self.conn, row, spec, machine, request, &callback, None) + insert_local_task_records_on(&self.conn, row, spec, machine, request, &callback) }) } @@ -1132,7 +798,7 @@ impl Store { }; self.immediate(|| { insert_local_task_records_on( - &self.conn, row, spec, machine, *request, &callback, None, + &self.conn, row, spec, machine, *request, &callback, )?; if let Some(after) = after { self.conn.execute( @@ -1185,18 +851,12 @@ impl Store { ExecutorIdentity::Rejected(record) => record.origin_machine == origin && record.execution_machine == execution, }; - return if !same - || (matches!(&identity, ExecutorIdentity::Accepted(_)) - && resource_task_id_is_reserved(&self.conn, row.id)?) - { - Err(AppError::ClusterTaskConflict { task: row.id }) - } else { + return if same { Ok(identity) + } else { + Err(AppError::ClusterTaskConflict { task: row.id }) }; } - if resource_task_id_is_reserved(&self.conn, row.id)? { - return Err(AppError::ClusterTaskConflict { task: row.id }); - } if self.get_task(row.id)?.is_some() { return Err(AppError::ClusterTaskConflict { task: row.id }); } @@ -1489,16 +1149,7 @@ impl Store { route.validate().map_err(|error| AppError::Internal { message: format!("invalid saved origin route: {error}"), })?; - let accepted_submission = match &route.submission { - SubmissionState::Accepted => true, - SubmissionState::Resource { phase, .. } => matches!( - phase, - ResourceRoutePhase::AcceptanceUnknown - | ResourceRoutePhase::Waiting - | ResourceRoutePhase::Activated - ), - _ => false, - }; + let accepted_submission = matches!(route.submission, SubmissionState::Accepted); record.task == id && route.task == id && route.origin_machine == route.execution_machine @@ -2678,8 +2329,8 @@ mod tests { BASE_SCHEMA, CancelResult, NewTask, RELEASED_V0_4_SCHEMA_VERSION, RELEASED_V0_5_1_SCHEMA_VERSION, RELEASED_V0_5_SCHEMA_VERSION, RELEASED_V0_7_SCHEMA_VERSION, RELEASED_V0_8_7_SCHEMA_VERSION, RELEASED_V0_8_SCHEMA_VERSION, - RELEASED_V0_11_SCHEMA_VERSION, Store, new_queued_task, read_exit_json, - write_exit_json_with_evidence, + RELEASED_V0_11_SCHEMA_VERSION, RELEASED_V0_13_SCHEMA_VERSION, Store, new_queued_task, + read_exit_json, write_exit_json_with_evidence, }; use crate::callback::EventKind; use crate::daemon::api::views::TaskSummary; @@ -2693,9 +2344,6 @@ mod tests { use crate::events::{DeliveryOutcome, DeliveryState, EventPayload, OutboxState}; use crate::invocation::CommandLine; use crate::machine::MachineId; - use crate::resource::{ - AssignmentRevision, Resource, ResourceId, ResourceRevision, SupervisorAddress, - }; use crate::spec::NormalizedSpec; use crate::submission::{ CallbackContext, CallbackExecutable, ExecutorIdentity, OriginRoute, PersistedSpec, @@ -2810,7 +2458,10 @@ mod tests { ProcessGroupExitEvidence::ConfirmedExited ); assert_eq!( - store.process_group_exit_evidence(id).unwrap(), + store + .get_task(id) + .unwrap() + .map(|row| row.process_group_exit_evidence()), Some(ProcessGroupExitEvidence::ConfirmedExited) ); let terminal_events = store @@ -2888,7 +2539,10 @@ mod tests { let store = Store::open(&path).unwrap(); assert_eq!( - store.process_group_exit_evidence(id).unwrap(), + store + .get_task(id) + .unwrap() + .map(|row| row.process_group_exit_evidence()), Some(ProcessGroupExitEvidence::NoChildSpawned) ); } @@ -2940,7 +2594,7 @@ mod tests { .pragma_query_value(None, "user_version", |row| row.get(0)) .unwrap(); assert_eq!(version, SCHEMA_VERSION); - assert_resource_tables_installed(&store); + assert_no_loan_tables(&store); let row = store.require_task(id).unwrap(); assert_eq!(row.status(), ProcessStatus::Cancelled); assert_eq!(row.exit_reason(), Some(&ExitReason::Cancelled)); @@ -2949,7 +2603,10 @@ mod tests { ProcessGroupExitEvidence::Unconfirmed ); assert_eq!( - store.process_group_exit_evidence(id).unwrap(), + store + .get_task(id) + .unwrap() + .map(|row| row.process_group_exit_evidence()), Some(ProcessGroupExitEvidence::Unconfirmed) ); } @@ -3020,75 +2677,6 @@ mod tests { .unwrap(); } - #[test] - fn queued_resource_task_cannot_be_accepted_as_local_or_remote_work() { - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let task = TaskId::new(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let (row, spec) = row_at(task, Path::new("/tmp")); - let resource = crate::resource::Resource::new( - crate::resource::ResourceId::new(), - "gpu-0".into(), - authority, - crate::resource::SupervisorAddress { - machine: authority, - thread: row.thread, - }, - crate::resource::AssignmentRevision::new(0), - crate::resource::ResourceRevision::new(0), - None, - ); - store.register_resource(authority, &resource).unwrap(); - store - .accept_resource_request( - authority, - RequestId::new(), - task, - resource.id, - origin, - spec.clone(), - ) - .unwrap(); - - assert!(matches!( - store.insert_local_task( - &row, - &spec, - MachineId::new(), - crate::submission::RequestId::new(), - CallbackExecutable::available("/bin/true".into()), - ), - Err(AppError::ClusterTaskConflict { task: conflict }) if conflict == task - )); - assert!(matches!( - store.insert_remote_task(&row, &spec, origin, authority), - Err(AppError::ClusterTaskConflict { task: conflict }) if conflict == task - )); - assert!(store.get_task(task).unwrap().is_none()); - assert!(store.executor_identity(task).unwrap().is_none()); - assert_eq!( - store - .next_queued_resource_request(authority, resource.id) - .unwrap() - .unwrap() - .task_id, - task - ); - - let ordinary_task = TaskId::new(); - let (ordinary_row, ordinary_spec) = row_at(ordinary_task, Path::new("/tmp")); - assert!(matches!( - store.insert_remote_task(&ordinary_row, &ordinary_spec, origin, authority), - Ok(ExecutorIdentity::Accepted(_)) - )); - assert!(matches!( - store.insert_remote_task(&ordinary_row, &ordinary_spec, origin, authority), - Ok(ExecutorIdentity::Accepted(_)) - )); - } - fn insert_legacy_terminal(store: &Store, id: TaskId, callback: CallbackStatus) { let mut row = task_row(id); row.state = TaskState::Finished { @@ -4023,8 +3611,7 @@ CREATE TABLE reports ( .pragma_query_value(None, "user_version", |row| row.get(0)) .unwrap(); assert_eq!(version, SCHEMA_VERSION); - assert_resource_tables_installed(&store); - assert_resource_request_serving_schema(&store, "fresh"); + assert_no_loan_tables(&store); assert_foreign_keys_enabled(&store); } @@ -4164,57 +3751,6 @@ CREATE TABLE reports ( assert_eq!(worker_thread, None); } - fn assert_resource_tables_installed(store: &Store) { - for table in [ - "resources", - "trainer_attempt_associations", - "resource_requests", - "resource_request_preventions", - "resource_cancellation_receipts", - "loans", - "resource_supervisor_notices", - "resource_release_completions", - "resource_release_checkpoint_states", - "resource_return_decisions", - "resource_restore_closures", - "resource_action_task_receipts", - "resource_background_launches", - "resource_idle_openings", - "resource_registration_receipts", - ] { - let exists: bool = store - .conn - .query_row( - "SELECT EXISTS(SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ?1)", - [table], - |row| row.get(0), - ) - .unwrap(); - assert!(exists, "resource table {table} was not installed"); - } - - let resource_primary_key: i64 = store - .conn - .query_row( - "SELECT pk FROM pragma_table_info('trainer_attempt_associations') - WHERE name='resource_id'", - [], - |row| row.get(0), - ) - .unwrap(); - let task_primary_key: i64 = store - .conn - .query_row( - "SELECT pk FROM pragma_table_info('trainer_attempt_associations') - WHERE name='task_id'", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(resource_primary_key, 0); - assert_eq!(task_primary_key, 1); - } - fn assert_foreign_keys_enabled(store: &Store) { let enabled: bool = store .conn @@ -4223,188 +3759,274 @@ CREATE TABLE reports ( assert!(enabled); } - /// Schema text of one table or index - fn schema_sql(store: &Store, name: &str) -> String { - store + fn assert_no_loan_tables(store: &Store) { + let loan_tables: Vec = store .conn - .query_row( - "SELECT sql FROM sqlite_master WHERE name = ?1", - [name], - |row| row.get(0), + .prepare( + "SELECT name FROM sqlite_master + WHERE type = 'table' + AND (name IN ('resources', 'loans', 'trainer_attempt_associations') + OR name LIKE 'resource!_%' ESCAPE '!')", ) .unwrap() + .query_map([], |row| row.get(0)) + .unwrap() + .collect::>() + .unwrap(); + assert!( + loan_tables.is_empty(), + "loan tables remain: {loan_tables:?}" + ); } - fn prepare_released_resource_queue( - path: &Path, - version: i64, - ) -> (MachineId, ResourceId, Vec) { - let mut store = Store::open(path).unwrap(); - let authority = MachineId::new(); - let resource_id = ResourceId::new(); - let resource = Resource::new( - resource_id, - "migration GPU".into(), - authority, - SupervisorAddress { - machine: authority, - thread: ThreadId::from_str("01a0ab97-a7aa-7463-a5b0-8d500e40e431").unwrap(), - }, - AssignmentRevision::new(0), - ResourceRevision::new(0), - None, - ); - store.register_resource(authority, &resource).unwrap(); - store + /// GPU loan tables of the last release that had them + const RELEASED_LOAN_SCHEMA: &str = include_str!("store/fixtures/loan_schema_v34.sql"); + + /// Every table, index, and trigger by type and name, with normalized SQL + /// + /// SQLite's own tables are left out: `sqlite_sequence` stays once an + /// `AUTOINCREMENT` loan table has existed, and SQLite cannot drop it + fn schema_snapshot(store: &Store) -> Vec<(String, String, String)> { + let mut statement = store .conn - .execute_batch( - "DROP TABLE resource_requests; - CREATE TABLE resource_requests ( - acceptance_sequence INTEGER PRIMARY KEY AUTOINCREMENT, - request_id TEXT NOT NULL UNIQUE, - task_id TEXT NOT NULL UNIQUE, - resource_id TEXT NOT NULL REFERENCES resources(id), - origin_machine TEXT NOT NULL, - spec_json TEXT NOT NULL, - state_json TEXT NOT NULL - );", + .prepare( + "SELECT type, name, sql FROM sqlite_master + WHERE name NOT LIKE 'sqlite!_%' ESCAPE '!' + ORDER BY type, name", ) .unwrap(); + statement + .query_map([], |row| { + Ok(( + row.get(0)?, + row.get(1)?, + normalized_sql(&row.get::<_, String>(2)?), + )) + }) + .unwrap() + .collect::>() + .unwrap() + } - let request_ids = vec![RequestId::new(), RequestId::new()]; - let spec_json = json!({ - "api_version": 1, - "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "migration request", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": ["/bin/echo", "migration"] } - }) - .to_string(); - for (sequence, request_id) in [7_i64, 13_i64].into_iter().zip(&request_ids) { - store - .conn - .execute( - "INSERT INTO resource_requests - (acceptance_sequence, request_id, task_id, resource_id, - origin_machine, spec_json, state_json) - VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7)", - params![ - sequence, - request_id.0.to_string(), - TaskId::new().to_string(), - resource_id.as_uuid().to_string(), - authority.as_uuid().to_string(), - spec_json, - json!({ "type": "queued" }).to_string(), - ], - ) - .unwrap(); + /// Schema SQL with whitespace collapsed and the parts of a table body sorted + /// + /// `ALTER TABLE ADD COLUMN` appends, so an upgraded version 1 database keeps + /// `name` last while a fresh one declares it third. Every query names its + /// columns, so their order is not part of the schema + fn normalized_sql(sql: &str) -> String { + let sql = sql.split_whitespace().collect::>().join(" "); + if !sql.starts_with("CREATE TABLE") { + return sql; } + let (Some(open), Some(close)) = (sql.find('('), sql.rfind(')')) else { + return sql; + }; - if version == RELEASED_V0_4_SCHEMA_VERSION || version == RELEASED_V0_5_SCHEMA_VERSION { - store - .conn - .execute_batch( - "DROP TABLE task_containers; - ALTER TABLE tasks DROP COLUMN container_exit_evidence;", - ) - .unwrap(); + let body = &sql[open + 1..close]; + let mut parts = Vec::new(); + let (mut depth, mut quoted, mut start) = (0_u32, false, 0); + for (index, character) in body.char_indices() { + match character { + '\'' => quoted = !quoted, + '(' if !quoted => depth += 1, + ')' if !quoted => depth -= 1, + ',' if !quoted && depth == 0 => { + parts.push(body[start..index].trim()); + start = index + 1; + } + _ => {} + } } - store - .conn - .execute_batch( - "ALTER TABLE tasks DROP COLUMN worker_thread; - ALTER TABLE origin_routes DROP COLUMN after_json; - ALTER TABLE origin_routes DROP COLUMN outcome;", - ) + parts.push(body[start..].trim()); + parts.sort_unstable(); + format!("{} ({})", sql[..open].trim_end(), parts.join(", ")) + } + + /// Rows that only the loan flows wrote into shared tables, keyed by `loan_task` + fn seed_loan_rows(store: &Store, ordinary_task: TaskId, loan_task: TaskId) { + let resource = uuid::Uuid::now_v7().to_string(); + let loan = uuid::Uuid::now_v7().to_string(); + let attempt = uuid::Uuid::now_v7().to_string(); + let conn = &store.conn; + conn.execute_batch(RELEASED_LOAN_SCHEMA).unwrap(); + // the loan JSON constraints are irrelevant to dropping the tables + conn.pragma_update(None, "ignore_check_constraints", true) + .unwrap(); + conn.execute( + "INSERT INTO resources (id, display_name, authority_machine, supervisor_machine, + supervisor_thread, assignment_revision, state_revision) + VALUES (?1, 'gpu', 'machine', 'machine', 'thread', 0, 0)", + [&resource], + ) + .unwrap(); + conn.execute( + "INSERT INTO loans (id, resource_id, state_json) VALUES (?1, ?2, '{}')", + [&loan, &resource], + ) + .unwrap(); + conn.execute( + "INSERT INTO resource_supervisor_notices (id, loan_id, action_id, notice_json) + VALUES (?1, ?2, ?1, '{}')", + [&attempt, &loan], + ) + .unwrap(); + conn.execute( + "INSERT INTO trainer_attempt_associations + (task_id, resource_id, authority_machine, association_json) + VALUES (?1, ?2, 'machine', '{}')", + [ordinary_task.to_string(), resource], + ) + .unwrap(); + conn.pragma_update(None, "ignore_check_constraints", false) .unwrap(); + + let loan_task = loan_task.to_string(); + conn.execute( + "INSERT INTO origin_routes (request_id, task_id, execution_machine, spec_json, route_json) + VALUES (?1, ?2, 'machine', '{}', '{\"submission\":{\"type\":\"resource\"}}')", + [uuid::Uuid::now_v7().to_string(), loan_task.clone()], + ) + .unwrap(); + conn.execute( + "INSERT INTO origin_inbox (task_id, seq, origin_machine, execution_machine, + event_json, notification_required, delivery_json) + VALUES (?1, 1, 'machine', 'machine', '{}', 0, '{}')", + [&loan_task], + ) + .unwrap(); + conn.execute( + "INSERT INTO cancellation_requests (task_id, request_json) + VALUES (?1, '{\"target\":{\"type\":\"resource\"}}')", + [&loan_task], + ) + .unwrap(); + conn.execute( + "INSERT INTO message_attempts (message_id, attempt_json) + VALUES (?1, '{\"request\":{\"source\":{\"kind\":\"resource_notice\"}}}')", + [&attempt], + ) + .unwrap(); + conn.execute( + "INSERT INTO message_receipts (message_id, receipt_json) VALUES (?1, '{}')", + [&attempt], + ) + .unwrap(); + } + + /// Restore the shape of released schema `version` on a current database + fn downgrade_to_released(store: &Store, version: i64) { + let mut sql = String::new(); + if version < RELEASED_V0_13_SCHEMA_VERSION { + sql.push_str( + "ALTER TABLE origin_routes DROP COLUMN after_json; + ALTER TABLE origin_routes DROP COLUMN outcome;", + ); + } + if version < RELEASED_V0_11_SCHEMA_VERSION { + sql.push_str("ALTER TABLE tasks DROP COLUMN worker_thread;"); + } + if version < RELEASED_V0_8_7_SCHEMA_VERSION { + sql.push_str( + "DROP TABLE resource_return_deadline_servings; + DROP TABLE resource_return_windows;", + ); + } if version < RELEASED_V0_7_SCHEMA_VERSION { - store - .conn - .execute("DROP TABLE resource_initial_idle_attestations", []) - .unwrap(); + sql.push_str("DROP TABLE resource_initial_idle_attestations;"); + } + if version < RELEASED_V0_5_1_SCHEMA_VERSION { + sql.push_str( + "ALTER TABLE tasks DROP COLUMN container_exit_evidence; + DROP TABLE task_containers;", + ); } - drop_return_window_tables(&store); + sql.push_str(&format!("PRAGMA user_version = {version};")); + store.conn.execute_batch(&sql).unwrap(); + } + + fn row_count(store: &Store, sql: &str, task: TaskId) -> i64 { store .conn - .pragma_update(None, "user_version", version) - .unwrap(); - drop(store); - (authority, resource_id, request_ids) + .query_row(sql, [task.to_string()], |row| row.get(0)) + .unwrap() } #[test] - fn released_resource_schemas_backfill_queue_ranks_and_keep_the_request_order() { + fn every_released_schema_upgrades_to_the_fresh_schema_without_loan_data() { + let fresh_dir = tempdir().unwrap(); + let fresh = schema_snapshot(&Store::open(&fresh_dir.path().join("db")).unwrap()); + + for (version, schema) in [(1, SCHEMA_V1), (2, BASE_SCHEMA)] { + let dir = tempdir().unwrap(); + let path = dir.path().join("db"); + let conn = Connection::open(&path).unwrap(); + conn.execute_batch(schema).unwrap(); + conn.pragma_update(None, "user_version", version).unwrap(); + drop(conn); + + let store = Store::open(&path).unwrap(); + assert_eq!(schema_snapshot(&store), fresh, "released version {version}"); + } + for version in [ RELEASED_V0_4_SCHEMA_VERSION, RELEASED_V0_5_SCHEMA_VERSION, RELEASED_V0_5_1_SCHEMA_VERSION, RELEASED_V0_7_SCHEMA_VERSION, + RELEASED_V0_8_SCHEMA_VERSION, + RELEASED_V0_8_7_SCHEMA_VERSION, + RELEASED_V0_11_SCHEMA_VERSION, + RELEASED_V0_13_SCHEMA_VERSION, ] { let dir = tempdir().unwrap(); let path = dir.path().join("db"); - let (authority, resource_id, request_ids) = - prepare_released_resource_queue(&path, version); - let store = Store::open(&path).unwrap(); - let current_version: i64 = store - .conn - .pragma_query_value(None, "user_version", |row| row.get(0)) - .unwrap(); - assert_eq!( - current_version, SCHEMA_VERSION, - "released version {version}" - ); + let ordinary_task = TaskId::new(); + let loan_task = TaskId::new(); + { + let store = Store::open(&path).unwrap(); + insert_local(&store, ordinary_task); + seed_loan_rows(&store, ordinary_task, loan_task); + downgrade_to_released(&store, version); + } - let requests = store.resource_requests(authority, resource_id).unwrap(); + let store = Store::open(&path).unwrap(); + assert_eq!(schema_snapshot(&store), fresh, "released version {version}"); + assert_no_loan_tables(&store); + assert_eq!(store.require_task(ordinary_task).unwrap().id, ordinary_task); assert_eq!( - requests - .iter() - .map(|request| request.request_id) - .collect::>(), - request_ids, + row_count( + &store, + "SELECT COUNT(*) FROM origin_routes WHERE task_id = ?1", + ordinary_task + ), + 1, "released version {version}" ); - let ranks = store + for table in ["origin_routes", "origin_inbox", "cancellation_requests"] { + assert_eq!( + row_count( + &store, + &format!("SELECT COUNT(*) FROM {table} WHERE task_id = ?1"), + loan_task + ), + 0, + "released version {version}: {table}" + ); + } + let messages: i64 = store .conn - .prepare( - "SELECT queue_rank FROM resource_requests - WHERE resource_id = ?1 ORDER BY acceptance_sequence", + .query_row( + "SELECT (SELECT COUNT(*) FROM message_attempts) + + (SELECT COUNT(*) FROM message_receipts)", + [], + |row| row.get(0), ) - .unwrap() - .query_map([resource_id.as_uuid().to_string()], |row| { - row.get::<_, i64>(0) - }) - .unwrap() - .collect::, _>>() .unwrap(); - assert_eq!(ranks, [7, 13], "released version {version}"); - assert_resource_request_serving_schema(&store, &format!("released version {version}")); + assert_eq!(messages, 0, "released version {version}"); + assert_foreign_keys_enabled(&store); } } - fn assert_resource_request_serving_schema(store: &Store, label: &str) { - let request_schema = schema_sql(store, "resource_requests"); - assert!( - request_schema.contains("queue_rank INTEGER NOT NULL CHECK (queue_rank > 0)"), - "{label}: {request_schema}" - ); - assert!( - !request_schema.contains("queue_rank INTEGER NOT NULL DEFAULT"), - "{label}: {request_schema}" - ); - assert!( - schema_sql(store, "resource_requests_serving_order") - .contains("resource_id, queue_rank, acceptance_sequence"), - "{label}" - ); - assert!( - schema_sql(store, "resource_requests_queued_serving_order") - .contains("resource_id, queue_rank, acceptance_sequence"), - "{label}" - ); - } - #[test] fn released_v0_5_database_gains_the_container_witness_and_keeps_its_rows() { let dir = tempdir().unwrap(); @@ -4413,8 +4035,7 @@ CREATE TABLE reports ( { let store = Store::open(&path).unwrap(); insert_local(&store, id); - // restore the v0.5.0 shape: no container column or table, and a - // request CHECK that accepted only command work + // restore the v0.5.0 shape: no container column or table store .conn .execute_batch(&format!( @@ -4423,18 +4044,6 @@ CREATE TABLE reports ( ALTER TABLE origin_routes DROP COLUMN outcome; ALTER TABLE tasks DROP COLUMN container_exit_evidence; DROP TABLE task_containers; - DROP TABLE resource_requests; - CREATE TABLE resource_requests ( - acceptance_sequence INTEGER PRIMARY KEY AUTOINCREMENT, - request_id TEXT NOT NULL UNIQUE, - task_id TEXT NOT NULL UNIQUE, - resource_id TEXT NOT NULL REFERENCES resources(id), - origin_machine TEXT NOT NULL, - spec_json TEXT NOT NULL CHECK ( - COALESCE(json_extract(spec_json, '$.workload.type') = 'task', 0) - ), - state_json TEXT NOT NULL - ); PRAGMA user_version = {RELEASED_V0_5_SCHEMA_VERSION};" )) .unwrap(); @@ -4452,90 +4061,10 @@ CREATE TABLE reports ( ContainerExitEvidence::Unconfirmed ); assert_eq!(store.task_container(id).unwrap(), None); - assert!(schema_sql(&store, "resource_requests").contains("'container'")); - assert!(!schema_sql(&store, "resource_requests").contains("'task', 0) AND")); - assert!(schema_sql(&store, "resource_restore_closures").contains("'container_ended'")); - schema_sql(&store, "resource_requests_serving_order"); - schema_sql(&store, "resource_requests_queued_serving_order"); - assert_resource_tables_installed(&store); + assert_no_loan_tables(&store); assert_foreign_keys_enabled(&store); } - /// Remove the tables that schema version 32 added, as a released database lacks them - fn drop_return_window_tables(store: &Store) { - store - .conn - .execute_batch( - "DROP TABLE resource_return_deadline_servings; - DROP TABLE resource_return_windows;", - ) - .unwrap(); - } - - #[test] - fn released_v0_8_database_gains_return_windows() { - let dir = tempdir().unwrap(); - let path = dir.path().join("db"); - let id = TaskId::new(); - { - let store = Store::open(&path).unwrap(); - insert_local(&store, id); - drop_return_window_tables(&store); - store - .conn - .execute_batch( - "ALTER TABLE tasks DROP COLUMN worker_thread; - ALTER TABLE origin_routes DROP COLUMN after_json; - ALTER TABLE origin_routes DROP COLUMN outcome;", - ) - .unwrap(); - store - .conn - .pragma_update(None, "user_version", RELEASED_V0_8_SCHEMA_VERSION) - .unwrap(); - } - - let store = Store::open(&path).unwrap(); - let version: i64 = store - .conn - .pragma_query_value(None, "user_version", |row| row.get(0)) - .unwrap(); - assert_eq!(version, SCHEMA_VERSION); - assert!(schema_sql(&store, "resource_return_windows").contains("deadline_at")); - assert!(schema_sql(&store, "resource_return_deadline_servings").contains("receipt_json")); - store.require_task(id).unwrap(); - } - - #[test] - fn released_v0_5_1_database_gains_the_initial_idle_table() { - let dir = tempdir().unwrap(); - let path = dir.path().join("db"); - let id = TaskId::new(); - { - let store = Store::open(&path).unwrap(); - insert_local(&store, id); - store - .conn - .execute_batch(&format!( - "DROP TABLE resource_initial_idle_attestations; - ALTER TABLE tasks DROP COLUMN worker_thread; - ALTER TABLE origin_routes DROP COLUMN after_json; - ALTER TABLE origin_routes DROP COLUMN outcome; - PRAGMA user_version = {RELEASED_V0_5_1_SCHEMA_VERSION};" - )) - .unwrap(); - } - - let store = Store::open(&path).unwrap(); - let version: i64 = store - .conn - .pragma_query_value(None, "user_version", |row| row.get(0)) - .unwrap(); - assert_eq!(version, SCHEMA_VERSION); - assert!(schema_sql(&store, "resource_initial_idle_attestations").contains("resource_id")); - store.require_task(id).unwrap(); - } - #[test] fn container_evidence_is_refused_where_no_container_path_records_it() { let dir = tempdir().unwrap(); @@ -4641,7 +4170,7 @@ CREATE TABLE reports ( assert_eq!(listed[0].name, None); assert_eq!(listed[0].display_name(), "echo hi"); let id = listed[0].id; - assert_resource_tables_installed(&store); + assert_no_loan_tables(&store); assert!(!store.is_event_task(id).unwrap()); store .cas_exit(id, ProcessStatus::Queued, &ExitReason::Cancelled) diff --git a/src/store/cancellation.rs b/src/store/cancellation.rs index e17dd1f..a25198b 100644 --- a/src/store/cancellation.rs +++ b/src/store/cancellation.rs @@ -2,16 +2,14 @@ use rusqlite::{OptionalExtension, TransactionBehavior, params}; -use super::{Store, resource_task_id_is_reserved}; +use super::Store; use crate::cancellation::{ CancellationDelivery, CancellationReceipt, CancellationRequest, CancellationRequestIdentity, - ExecutorCancelState, ResourceCancellationRequestIdentity, + ExecutorCancelState, }; use crate::domain::TaskId; use crate::error::AppError; -use crate::submission::{ - ExecutorIdentity, PreAcceptanceRejection, RejectionTombstone, ResourceCancellationReceipt, -}; +use crate::submission::{ExecutorIdentity, PreAcceptanceRejection, RejectionTombstone}; fn encode(value: &T) -> Result { Ok(serde_json::to_string(value)?) @@ -122,63 +120,13 @@ impl Store { tx.commit()?; } CancellationDelivery::Delivered { result } if result == &receipt.state => {} - CancellationDelivery::Delivered { .. } - | CancellationDelivery::ResourceDelivered { .. } => { + CancellationDelivery::Delivered { .. } => { return Err(conflict(request.task)); } } Ok(request) } - /// Settle one resource intent only with its exact authority receipt. - pub fn acknowledge_resource_cancellation( - &mut self, - receipt: &ResourceCancellationReceipt, - ) -> Result { - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate)?; - let data: String = tx.query_row( - "SELECT request_json FROM cancellation_requests WHERE task_id=?1", - [receipt.task.to_string()], - |row| row.get(0), - )?; - let mut request: CancellationRequest = decode(&data)?; - let Some(identity) = request.resource_identity() else { - return Err(conflict(receipt.task)); - }; - if !resource_receipt_matches_identity(receipt, &identity) - || request.identity() - != (CancellationRequestIdentity { - requester_machine: receipt.requester_machine, - cancellation: receipt.cancellation, - task: receipt.task, - origin_machine: receipt.origin_machine, - execution_machine: receipt.authority_machine, - }) - { - return Err(conflict(receipt.task)); - } - match &request.delivery { - CancellationDelivery::Pending => { - request.delivery = CancellationDelivery::ResourceDelivered { - result: receipt.clone(), - }; - tx.execute( - "UPDATE cancellation_requests SET request_json=?1 WHERE task_id=?2", - params![encode(&request)?, request.task.to_string()], - )?; - tx.commit()?; - } - CancellationDelivery::ResourceDelivered { result } if result == receipt => {} - CancellationDelivery::Delivered { .. } - | CancellationDelivery::ResourceDelivered { .. } => { - return Err(conflict(receipt.task)); - } - } - Ok(request) - } - /// Atomically retain executor receipt and pre-acceptance tombstone pub fn receive_cancellation( &mut self, @@ -187,10 +135,6 @@ impl Store { let tx = self .conn .transaction_with_behavior(TransactionBehavior::Immediate)?; - if resource_task_id_is_reserved(&tx, request.task)? { - return Err(AppError::ResourceCancellationUnavailable { task: request.task }); - } - let previous: Option = tx .query_row( "SELECT receipt_json FROM executor_cancellations WHERE cancellation_id=?1", @@ -320,236 +264,3 @@ impl Store { Ok(receipt) } } - -fn resource_receipt_matches_identity( - receipt: &ResourceCancellationReceipt, - identity: &ResourceCancellationRequestIdentity, -) -> bool { - receipt.cancellation == identity.cancellation - && receipt.requester_machine == identity.requester_machine - && receipt.request == identity.request - && receipt.task == identity.task - && receipt.origin_machine == identity.origin_machine - && receipt.authority_machine == identity.authority_machine - && receipt.resource == identity.resource - && receipt.target_phase == identity.target_phase -} - -#[cfg(test)] -mod tests { - use rusqlite::params; - use tempfile::tempdir; - use uuid::Uuid; - - use super::*; - use crate::domain::ThreadId; - use crate::machine::MachineId; - use crate::resource::{ - AssignmentRevision, Resource, ResourceId, ResourceRevision, SupervisorAddress, - }; - use crate::spec::NormalizedSpec; - use crate::submission::{ - RequestId, ResourceCancellationIneligibleReason, ResourceCancellationOutcome, - ResourceCancellationReceipt, ResourceRoutePhase, - }; - - fn resource(authority: MachineId) -> Resource { - Resource::new( - ResourceId::new(), - "gpu-0".into(), - authority, - SupervisorAddress { - machine: authority, - thread: ThreadId(Uuid::now_v7()), - }, - AssignmentRevision::new(0), - ResourceRevision::new(0), - None, - ) - } - - fn spec() -> NormalizedSpec { - serde_json::from_value(serde_json::json!({ - "api_version": 1, - "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "resource command", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": ["/bin/echo", "hello"] } - })) - .unwrap() - } - - fn cancellation( - task: TaskId, - origin: MachineId, - authority: MachineId, - ) -> CancellationRequestIdentity { - CancellationRequestIdentity { - requester_machine: origin, - cancellation: Uuid::now_v7(), - task, - origin_machine: origin, - execution_machine: authority, - } - } - - fn resource_cancellation_request( - origin: MachineId, - authority: MachineId, - resource: ResourceId, - request: RequestId, - task: TaskId, - cancellation: Uuid, - ) -> CancellationRequest { - CancellationRequest { - requester_machine: origin, - cancellation, - task, - origin_machine: origin, - execution_machine: authority, - target: crate::cancellation::CancellationTarget::Resource( - crate::cancellation::ResourceCancellationTarget { - request_id: request, - task_id: task, - resource_id: resource, - origin_machine: origin, - authority_machine: authority, - phase: ResourceRoutePhase::Waiting, - }, - ), - delivery: CancellationDelivery::Pending, - } - } - - #[test] - fn resource_cancellation_intent_and_receipt_settle_by_full_identity() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let origin = MachineId::new(); - let authority = MachineId::new(); - let resource = ResourceId::new(); - let request_id = RequestId::new(); - let task = TaskId::new(); - let cancellation = Uuid::now_v7(); - let request = resource_cancellation_request( - origin, - authority, - resource, - request_id, - task, - cancellation, - ); - assert_eq!( - store.insert_cancellation_request(request.clone()).unwrap(), - (request.clone(), true) - ); - assert_eq!( - store.insert_cancellation_request(request.clone()).unwrap(), - (request.clone(), false) - ); - - let mut changed = request.clone(); - changed.cancellation = Uuid::now_v7(); - assert!(matches!( - store.insert_cancellation_request(changed), - Err(AppError::ClusterTaskConflict { task: found }) if found == task - )); - - let receipt = ResourceCancellationReceipt { - cancellation, - requester_machine: origin, - request: request_id, - task, - origin_machine: origin, - authority_machine: authority, - resource, - target_phase: ResourceRoutePhase::Waiting, - outcome: ResourceCancellationOutcome::CancelledBeforeLaunch, - }; - let settled = store.acknowledge_resource_cancellation(&receipt).unwrap(); - assert_eq!( - settled.delivery, - CancellationDelivery::ResourceDelivered { - result: receipt.clone(), - } - ); - assert_eq!(store.pending_cancellation_requests().unwrap().len(), 0); - assert_eq!( - store.acknowledge_resource_cancellation(&receipt).unwrap(), - settled - ); - - let mut conflicting_receipt = receipt; - conflicting_receipt.outcome = ResourceCancellationOutcome::NotEligible { - reason: ResourceCancellationIneligibleReason::Terminal, - }; - assert!(matches!( - store.acknowledge_resource_cancellation(&conflicting_receipt), - Err(AppError::ClusterTaskConflict { task: found }) if found == task - )); - } - - #[test] - fn generic_cancellation_does_not_tombstone_a_queued_resource_request() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - let task = TaskId::new(); - store - .accept_resource_request( - authority, - RequestId::new(), - task, - resource.id, - origin, - spec(), - ) - .unwrap(); - - assert!(matches!( - store.receive_cancellation(cancellation(task, origin, authority)), - Err(AppError::ResourceCancellationUnavailable { task: found }) if found == task - )); - assert!(store.executor_identity(task).unwrap().is_none()); - assert_eq!( - store - .resource_requests(authority, resource.id) - .unwrap() - .len(), - 1 - ); - } - - #[test] - fn generic_cancellation_does_not_tombstone_a_prevented_resource_request() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let task = TaskId::new(); - store - .conn - .execute( - "INSERT INTO resource_request_preventions ( - request_id, task_id, resource_id, origin_machine - ) VALUES (?1, ?2, ?3, ?4)", - params![ - RequestId::new().0.to_string(), - task.to_string(), - ResourceId::new().as_uuid().to_string(), - origin.as_uuid().to_string(), - ], - ) - .unwrap(); - - assert!(matches!( - store.receive_cancellation(cancellation(task, origin, authority)), - Err(AppError::ResourceCancellationUnavailable { task: found }) if found == task - )); - assert!(store.executor_identity(task).unwrap().is_none()); - } -} diff --git a/src/store/container.rs b/src/store/container.rs index 4cfc2b3..dcc4c2c 100644 --- a/src/store/container.rs +++ b/src/store/container.rs @@ -6,7 +6,6 @@ use rusqlite::{OptionalExtension, params}; use super::{Store, fmt_time, parse_time}; use crate::domain::{ContainerId, TaskId}; use crate::error::AppError; -use crate::resource::ResourceId; /// Container lifecycle saved for one container task #[derive(Debug, Clone, PartialEq, Eq)] @@ -92,30 +91,6 @@ impl Store { Ok(()) } - /// Resource whose request or return loan owns one task, if any - pub(crate) fn resource_for_task(&self, task: TaskId) -> Result, AppError> { - let resource: Option = self - .conn - .query_row( - "SELECT resource_id FROM resource_requests WHERE task_id = ?1 - UNION ALL - SELECT resource_id FROM loans - WHERE json_extract(state_json, '$.phase.resume_task_id') = ?1 - OR json_extract(state_json, '$.last_safe_phase.resume_task_id') = ?1 - LIMIT 1", - [task.to_string()], - |row| row.get(0), - ) - .optional()?; - resource - .map(|raw| { - raw.parse().map_err(|_| AppError::Internal { - message: format!("stored resource id {raw} is invalid"), - }) - }) - .transpose() - } - /// Count one more adopting worker, unless `limit` workers in a row already /// stopped before they reached the container /// diff --git a/src/store/dependency.rs b/src/store/dependency.rs index 1221281..6e4d64d 100644 --- a/src/store/dependency.rs +++ b/src/store/dependency.rs @@ -20,8 +20,7 @@ use crate::domain::{CallbackStatus, ProcessStatus, TaskId}; use crate::error::AppError; use crate::events::{DeliveryState, EventPayload, TaskEvent}; use crate::submission::{ - DependentRoute, HeldPhase, OriginRoute, PreAcceptanceRejection, ResourceActionRoutePhase, - ResourceRoutePhase, SubmissionState, + DependentRoute, HeldPhase, OriginRoute, PreAcceptanceRejection, SubmissionState, }; fn decode_route(json: &str) -> Result { @@ -172,28 +171,12 @@ pub(super) fn refused_launch_ending(reason: &str) -> UnlaunchedEnding { /// so its closure is its ending fn closed_route_outcome(route: &OriginRoute) -> Option { match &route.submission { - SubmissionState::Rejected { .. } - | SubmissionState::Resource { - phase: ResourceRoutePhase::Rejected { .. }, - .. - } - | SubmissionState::ResourceAction { - phase: ResourceActionRoutePhase::Rejected { .. }, - .. - } => Some(TaskOutcome::Failed), - SubmissionState::Resource { - phase: ResourceRoutePhase::CancelledBeforeLaunch, - .. - } - | SubmissionState::Held { + SubmissionState::Rejected { .. } => Some(TaskOutcome::Failed), + SubmissionState::Held { phase: HeldPhase::Cancelled { .. }, } => Some(TaskOutcome::Cancelled), - // a refused background launch can still be proven accepted by its first event SubmissionState::AcceptanceUnknown | SubmissionState::Accepted - | SubmissionState::Resource { .. } - | SubmissionState::ResourceAction { .. } - | SubmissionState::ResourceBackground { .. } | SubmissionState::Held { .. } => None, } } diff --git a/src/store/events.rs b/src/store/events.rs index baca18e..ed88467 100644 --- a/src/store/events.rs +++ b/src/store/events.rs @@ -17,10 +17,7 @@ use crate::events::{ TaskEvent, WaitingInboxEvent, }; use crate::machine::MachineId; -use crate::submission::{ - ExecutorIdentity, HeldPhase, OriginRoute, ResourceActionRoutePhase, - ResourceBackgroundRoutePhase, ResourceRoutePhase, SubmissionState, -}; +use crate::submission::{ExecutorIdentity, HeldPhase, OriginRoute, SubmissionState}; const EVENT_RETENTION_DAYS: i64 = 30; const EVENT_RETENTION_BATCH_SIZE: i64 = 64; @@ -62,7 +59,7 @@ fn decode_route(value: &str) -> Result { let route: OriginRoute = decode(value)?; route.validate().map_err(|error| { EventError::Storage(AppError::Internal { - message: format!("invalid saved resource origin route: {error}"), + message: format!("invalid saved origin route: {error}"), }) })?; Ok(route) @@ -812,53 +809,6 @@ pub(super) fn append_produced_event_on( Ok(()) } -pub(super) fn initial_queued_event_matches_on( - conn: &Connection, - task: TaskId, - origin_machine: MachineId, - execution_machine: MachineId, -) -> Result { - let expected = TaskEvent { - task, - seq: NonZeroU64::MIN, - origin_machine, - execution_machine, - payload: EventPayload::State { - status: ProcessStatus::Queued, - }, - }; - - let outbox: Option<(String, bool)> = conn - .query_row( - "SELECT event_json, notification_required FROM executor_outbox - WHERE task_id=?1 AND seq=1", - [task.to_string()], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .optional() - .map_err(storage)?; - if let Some((event_json, notification_required)) = outbox { - let event: TaskEvent = decode(&event_json)?; - validate(&event)?; - return Ok(event == expected && !notification_required); - } - - let receipt: Option<(String, String)> = conn - .query_row( - "SELECT event_digest, result_json FROM executor_event_receipts - WHERE task_id=?1 AND seq=1", - [task.to_string()], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .optional() - .map_err(storage)?; - let Some((saved_digest, result_json)) = receipt else { - return Ok(false); - }; - Ok(saved_digest == event_digest(&expected)? - && result_json == encode(&OutboxState::Acknowledged)?) -} - fn append_outbound_event_on( conn: &Connection, task: TaskId, @@ -1073,58 +1023,7 @@ impl Store { seq: event.seq.get(), }); } - let activate_resource = match &route.submission { - // the first queued event proves that the authority accepted the fixed - // identity, even when its launch reply was lost - SubmissionState::ResourceAction { phase, .. } => match phase { - ResourceActionRoutePhase::AcceptanceUnknown => { - if event.payload.process_state() != Some(ProcessStatus::Queued) { - return Err(EventError::Invalid { - message: "first resource action task event must report queued state" - .into(), - }); - } - true - } - ResourceActionRoutePhase::Accepted => false, - ResourceActionRoutePhase::Rejected { .. } => { - return Err(EventError::Invalid { - message: "resource action route was rejected before task acceptance".into(), - }); - } - }, - // the authority's durable first queued event proves acceptance even - // after a refusal was saved, since a delayed send may have won; the - // callback owner must keep the running trainer's results - SubmissionState::ResourceBackground { phase, .. } => match phase { - ResourceBackgroundRoutePhase::AcceptanceUnknown - | ResourceBackgroundRoutePhase::Rejected { .. } => { - if event.payload.process_state() != Some(ProcessStatus::Queued) { - return Err(EventError::Invalid { - message: "first background launch event must report queued state" - .into(), - }); - } - true - } - ResourceBackgroundRoutePhase::Accepted => false, - }, - SubmissionState::Resource { phase, .. } => match phase { - ResourceRoutePhase::AcceptanceUnknown | ResourceRoutePhase::Waiting => { - if event.payload.process_state() != Some(ProcessStatus::Queued) { - return Err(EventError::Invalid { - message: "first resource task event must report queued state".into(), - }); - } - true - } - ResourceRoutePhase::Activated => false, - ResourceRoutePhase::CancelledBeforeLaunch | ResourceRoutePhase::Rejected { .. } => { - return Err(EventError::Invalid { - message: "resource route was closed before task activation".into(), - }); - } - }, + let accept_held_launch = match &route.submission { // the executor's first queued event proves that a released launch // was accepted, even when its reply was lost SubmissionState::Held { phase } => match phase { @@ -1175,24 +1074,12 @@ impl Store { if let Some(state) = event.payload.process_state() { route.last_execution_state = Some(state); } - if activate_resource { - match &mut route.submission { - SubmissionState::Resource { phase, .. } => *phase = ResourceRoutePhase::Activated, - SubmissionState::ResourceAction { phase, .. } => { - *phase = ResourceActionRoutePhase::Accepted; - } - SubmissionState::ResourceBackground { phase, .. } => { - *phase = ResourceBackgroundRoutePhase::Accepted; - } - SubmissionState::Held { .. } => route.submission = SubmissionState::Accepted, - SubmissionState::AcceptanceUnknown - | SubmissionState::Accepted - | SubmissionState::Rejected { .. } => {} - } + if accept_held_launch { + route.submission = SubmissionState::Accepted; } route.validate().map_err(|error| { EventError::Storage(AppError::Internal { - message: format!("invalid resource origin route transition: {error}"), + message: format!("invalid origin route transition: {error}"), }) })?; if matches!(delivery, DeliveryState::NotRequired) @@ -1259,7 +1146,6 @@ impl Store { #[cfg(test)] mod tests { use crate::domain::ProcessStatus; - use crate::resource::{AssignmentRevision, ResourceId, ResourceRevision, SupervisorAddress}; use std::path::Path; use tempfile::tempdir; @@ -1273,11 +1159,9 @@ mod tests { THREAD_WAIT_LIMIT, TaskEvent, }; use crate::machine::MachineId; - use crate::store::{IdentityError, Store}; + use crate::store::Store; use crate::submission::{ - CallbackContext, ExecutionRecord, OriginRoute, RequestId, ResourceActionRoutePhase, - ResourceBackgroundRoutePhase, ResourceQueueOutcome, ResourceQueueReceipt, - ResourceRoutePhase, SubmissionState, + CallbackContext, ExecutionRecord, OriginRoute, RequestId, SubmissionState, }; use chrono::Utc; use rusqlite::params; @@ -1365,112 +1249,6 @@ mod tests { .unwrap(); } - fn resource_route() -> OriginRoute { - let mut route = route(); - route.submission = SubmissionState::Resource { - resource: ResourceId::new(), - phase: ResourceRoutePhase::AcceptanceUnknown, - }; - route - } - - #[test] - fn first_queued_resource_event_activates_in_the_cursor_transaction() { - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let route = resource_route(); - store.insert_origin_route(&route).unwrap(); - - let queued = event(&route, 1, ProcessStatus::Queued); - assert_eq!( - store.accept_inbound_event(&queued).unwrap(), - EventAcceptance::Acknowledged { seq: 1 } - ); - let activated = store.origin_route_by_task(route.task).unwrap().unwrap(); - assert!(matches!( - activated.submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::Activated, - .. - } - )); - assert_eq!(activated.last_accepted_seq, 1); - assert_eq!(activated.last_execution_state, Some(ProcessStatus::Queued)); - - store - .accept_inbound_event(&event(&route, 2, ProcessStatus::Running)) - .unwrap(); - let advanced = store.origin_route_by_task(route.task).unwrap().unwrap(); - assert!(matches!( - advanced.submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::Activated, - .. - } - )); - assert_eq!(advanced.last_accepted_seq, 2); - assert_eq!(advanced.last_execution_state, Some(ProcessStatus::Running)); - } - - #[test] - fn resource_route_rejects_a_nonqueued_first_task_event() { - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let route = resource_route(); - store.insert_origin_route(&route).unwrap(); - - assert!(matches!( - store.accept_inbound_event(&event(&route, 1, ProcessStatus::Running,)), - Err(EventError::Invalid { .. }) - )); - let saved = store.origin_route_by_task(route.task).unwrap().unwrap(); - assert!(matches!( - saved.submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::AcceptanceUnknown, - .. - } - )); - assert_eq!(saved.last_accepted_seq, 0); - assert!(store.inbound_events(route.task).unwrap().is_empty()); - } - - #[test] - fn rejected_resource_route_does_not_accept_a_task_event() { - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let route = resource_route(); - store.insert_origin_route(&route).unwrap(); - let SubmissionState::Resource { resource, .. } = &route.submission else { - unreachable!(); - }; - store - .resolve_resource_route(&ResourceQueueReceipt { - request: route.request, - task: route.task, - origin_machine: route.origin_machine, - authority_machine: route.execution_machine, - resource: *resource, - outcome: ResourceQueueOutcome::Rejected { - reason: "queue rejected".into(), - }, - }) - .unwrap(); - - assert!(matches!( - store.accept_inbound_event(&event(&route, 1, ProcessStatus::Queued,)), - Err(EventError::Invalid { .. }) - )); - assert_eq!( - store - .origin_route_by_task(route.task) - .unwrap() - .unwrap() - .last_accepted_seq, - 0 - ); - } - #[test] fn first_duplicate_conflict_gap_and_unknown_route() { let dir = tempdir().unwrap(); @@ -2409,219 +2187,4 @@ mod tests { ); assert!(reopened.pending_outbound_tasks().unwrap().is_empty()); } - - fn action_route() -> OriginRoute { - let base = route(); - OriginRoute::new_resource_action(crate::submission::NewResourceActionRoute { - request: base.request, - task: base.task, - callback: base.callback.clone(), - spec: base.current_spec().unwrap().clone(), - binding: crate::submission::ResourceActionRouteBinding { - kind: crate::resource::bound_action::ResourceActionKind::ReleaseWatcher, - authority: crate::resource::SupervisorActionAuthority { - authority_machine: base.execution_machine, - resource_id: ResourceId::new(), - loan_id: crate::resource::LoanId::new(), - action_id: crate::resource::ActionId::new(), - expected_state_revision: ResourceRevision::new(1), - supervisor: SupervisorAddress { - machine: base.origin_machine, - thread: base.thread, - }, - assignment_revision: AssignmentRevision::new(0), - }, - }, - launch: crate::resource::bound_action::ResourceActionLaunch::ReleaseWatcher { - observed_background_task: TaskId::new(), - }, - }) - .unwrap() - } - - #[test] - fn first_queued_event_accepts_an_action_route_and_duplicates_settle_once() { - use crate::domain::ProcessStatus; - - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let route = action_route(); - store.insert_origin_route(&route).unwrap(); - - assert!(matches!( - store.accept_inbound_event(&event(&route, 1, ProcessStatus::Running)), - Err(EventError::Invalid { .. }) - )); - assert_eq!( - store - .accept_inbound_event(&event(&route, 1, ProcessStatus::Queued)) - .unwrap(), - EventAcceptance::Acknowledged { seq: 1 } - ); - let saved = store.origin_route_by_task(route.task).unwrap().unwrap(); - assert!(matches!( - saved.submission, - SubmissionState::ResourceAction { - phase: ResourceActionRoutePhase::Accepted, - .. - } - )); - // a duplicate is acknowledged from the saved inbox, a later event is ordered - assert_eq!( - store - .accept_inbound_event(&event(&route, 1, ProcessStatus::Queued)) - .unwrap(), - EventAcceptance::Acknowledged { seq: 1 } - ); - assert_eq!( - store - .accept_inbound_event(&event(&route, 3, ProcessStatus::Running)) - .unwrap(), - EventAcceptance::Expected { seq: 2 } - ); - assert_eq!( - store - .accept_inbound_event(&event(&route, 2, ProcessStatus::Running)) - .unwrap(), - EventAcceptance::Acknowledged { seq: 2 } - ); - assert_eq!(store.inbound_events(route.task).unwrap().len(), 2); - // an event from another execution owner is refused - let mut foreign = event(&route, 3, ProcessStatus::Running); - foreign.execution_machine = MachineId::new(); - assert!(matches!( - store.accept_inbound_event(&foreign), - Err(EventError::OwnerConflict { .. }) - )); - } - - #[test] - fn rejected_action_route_refuses_task_events() { - use crate::domain::ProcessStatus; - - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let route = action_route(); - store.insert_origin_route(&route).unwrap(); - store - .resolve_resource_action_route( - route.task, - &crate::store::ResourceActionRouteResult::Rejected( - crate::resource::bound_action::ResourceActionRejection::ActionNotPending, - ), - ) - .unwrap(); - assert!(matches!( - store.accept_inbound_event(&event(&route, 1, ProcessStatus::Queued)), - Err(EventError::Invalid { .. }) - )); - assert!(store.inbound_events(route.task).unwrap().is_empty()); - } - - fn background_route() -> OriginRoute { - let base = route(); - OriginRoute::new_resource_background(crate::submission::NewResourceBackgroundRoute { - request: base.request, - task: base.task, - callback: base.callback.clone(), - spec: base.current_spec().unwrap().clone(), - binding: crate::resource::background_launch::BackgroundLaunchBinding { - assignment: crate::resource::background_launch::BackgroundSupervisorAssignment { - authority_machine: base.execution_machine, - resource_id: ResourceId::new(), - supervisor: SupervisorAddress { - machine: base.origin_machine, - thread: base.thread, - }, - assignment_revision: AssignmentRevision::new(0), - }, - expected_state_revision: ResourceRevision::new(2), - }, - }) - .unwrap() - } - - #[test] - fn background_route_keeps_a_trainer_that_the_authority_accepted_after_a_saved_refusal() { - use crate::domain::ProcessStatus; - use crate::resource::background_launch::{ - RemoteBackgroundLaunchReceipt, ResourceBackgroundRejection, - }; - use crate::store::ResourceBackgroundRouteResult; - - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let route = background_route(); - store.insert_origin_route(&route).unwrap(); - let SubmissionState::ResourceBackground { binding, .. } = &route.submission else { - unreachable!(); - }; - let receipt = RemoteBackgroundLaunchReceipt { - binding: *binding, - request_id: route.request, - task_id: route.task, - normalized_spec_sha256: crate::submission::normalized_spec_sha256( - route.current_spec().unwrap(), - ) - .unwrap(), - }; - // a receipt for another assignment cannot accept this route - let mut other = receipt; - other.binding.assignment.assignment_revision = AssignmentRevision::new(1); - assert!(matches!( - store.resolve_resource_background_route( - route.task, - &ResourceBackgroundRouteResult::Accepted(other) - ), - Err(IdentityError::Conflict) - )); - - // a delayed send may win after a refusal was saved; its first queued - // event is durable acceptance, so the trainer keeps its callback owner - store - .resolve_resource_background_route( - route.task, - &ResourceBackgroundRouteResult::Rejected( - ResourceBackgroundRejection::LaunchPending { - task_id: TaskId::new(), - }, - ), - ) - .unwrap(); - assert!(matches!( - store.accept_inbound_event(&event(&route, 1, ProcessStatus::Running)), - Err(EventError::Invalid { .. }) - )); - assert_eq!( - store - .accept_inbound_event(&event(&route, 1, ProcessStatus::Queued)) - .unwrap(), - EventAcceptance::Acknowledged { seq: 1 } - ); - let saved = store.origin_route_by_task(route.task).unwrap().unwrap(); - assert!(matches!( - saved.submission, - SubmissionState::ResourceBackground { - phase: ResourceBackgroundRoutePhase::Accepted, - .. - } - )); - - // an accepted route never moves back to a refusal; the exact receipt is idempotent - assert!(matches!( - store.resolve_resource_background_route( - route.task, - &ResourceBackgroundRouteResult::Rejected( - ResourceBackgroundRejection::NotCurrentSupervisor - ) - ), - Err(IdentityError::Conflict) - )); - store - .resolve_resource_background_route( - route.task, - &ResourceBackgroundRouteResult::Accepted(receipt), - ) - .unwrap(); - } } diff --git a/src/resource/store/schema.rs b/src/store/fixtures/loan_schema_v34.sql similarity index 99% rename from src/resource/store/schema.rs rename to src/store/fixtures/loan_schema_v34.sql index 3a6d3ba..c80b3d9 100644 --- a/src/resource/store/schema.rs +++ b/src/store/fixtures/loan_schema_v34.sql @@ -1,7 +1,5 @@ -//! SQLite schema for authority-local resource state - -/// Resource tables installed by the Store migration hook -pub(crate) const RESOURCE_SCHEMA: &str = r" +-- GPU loan tables as the v0.14.0 release (schema version 34) created them +-- Upgrade tests seed these to prove that the version 35 step drops them CREATE TABLE IF NOT EXISTS resources ( id TEXT PRIMARY KEY, display_name TEXT NOT NULL, @@ -362,4 +360,3 @@ CREATE TABLE IF NOT EXISTS resource_registration_receipts ( AND COALESCE(json_type(receipt_json, '$.initial_supervisor') = 'object', 0) ) ); -"; diff --git a/src/store/identity.rs b/src/store/identity.rs index 143934e..3d0d25b 100644 --- a/src/store/identity.rs +++ b/src/store/identity.rs @@ -2,21 +2,14 @@ use rusqlite::{OptionalExtension, TransactionBehavior, params}; -use super::{Store, resource_task_id_is_reserved}; +use super::Store; use crate::dependency::TaskDependencies; use crate::domain::{ProcessStatus, TaskId}; use crate::error::AppError; use crate::machine::MachineId; -use crate::resource::ActionId; -use crate::resource::background_launch::{ - RemoteBackgroundLaunchReceipt, ResourceBackgroundRejection, -}; -use crate::resource::bound_action::ActionTaskReceipt; use crate::submission::{ ExecutionRecord, ExecutorIdentity, HeldPhase, OriginRoute, PreAcceptanceRejection, - RejectionTombstone, RequestId, ResourceActionRoutePhase, ResourceBackgroundRoutePhase, - ResourceCancellationOutcome, ResourceCancellationReceipt, ResourceQueueOutcome, - ResourceQueueReceipt, ResourceRoutePhase, SubmissionState, + RejectionTombstone, RequestId, SubmissionState, }; /// A durable identity operation failed without changing its existing owner @@ -45,7 +38,7 @@ fn decode_route(value: &str) -> Result { let route: OriginRoute = decode(value)?; route.validate().map_err(|error| { IdentityError::Storage(AppError::Internal { - message: format!("invalid saved resource origin route: {error}"), + message: format!("invalid saved origin route: {error}"), }) })?; Ok(route) @@ -61,77 +54,10 @@ fn decode_identity(value: &str) -> Result { Ok(identity) } -pub(super) fn origin_route_by_request_on( - conn: &rusqlite::Connection, - request: RequestId, -) -> Result, IdentityError> { - let data: Option = conn - .query_row( - "SELECT route_json FROM origin_routes WHERE request_id=?1", - [request.0.to_string()], - |row| row.get(0), - ) - .optional() - .map_err(storage)?; - let route = data.as_deref().map(decode_route).transpose()?; - if route.as_ref().is_some_and(|route| route.request != request) { - return Err(IdentityError::Conflict); - } - Ok(route) -} - -pub(super) fn origin_route_by_task_on( - conn: &rusqlite::Connection, - task: TaskId, -) -> Result, IdentityError> { - let data: Option = conn - .query_row( - "SELECT route_json FROM origin_routes WHERE task_id=?1", - [task.to_string()], - |row| row.get(0), - ) - .optional() - .map_err(storage)?; - let route = data.as_deref().map(decode_route).transpose()?; - if route.as_ref().is_some_and(|route| route.task != task) { - return Err(IdentityError::Conflict); - } - Ok(route) -} - pub(super) fn executor_identity_on( conn: &rusqlite::Connection, task: TaskId, ) -> Result, IdentityError> { - Ok(executor_identity_with_json_on(conn, task)?.map(|(identity, _)| identity)) -} - -/// Whether the saved origin column of one executor identity names this origin -/// -/// The column indexes the origin outside the identity JSON, so resource -/// acceptance checks that both copies agree -pub(super) fn identity_origin_column_is_on( - conn: &rusqlite::Connection, - task: TaskId, - origin_machine: MachineId, -) -> Result { - let saved: Option = conn - .query_row( - "SELECT origin_machine FROM executor_identities WHERE task_id=?1", - [task.to_string()], - |row| row.get(0), - ) - .optional()?; - Ok(saved.is_some_and(|saved| saved == origin_machine.to_string())) -} - -/// Read one executor identity with the exact saved JSON it was decoded from -/// -/// A release proof keeps the JSON so its commit can detect any later rewrite -pub(super) fn executor_identity_with_json_on( - conn: &rusqlite::Connection, - task: TaskId, -) -> Result, IdentityError> { let data: Option = conn .query_row( "SELECT identity_json FROM executor_identities WHERE task_id=?1", @@ -140,201 +66,26 @@ pub(super) fn executor_identity_with_json_on( ) .optional() .map_err(storage)?; - data.map(|json| Ok((decode_identity(&json)?, json))) - .transpose() -} - -/// Read acceptance-unknown routes of one resource submission type in request order -/// -/// The SQL filter only selects candidates. Each decoded route must still name its -/// row identities and be unknown by `is_unknown`, or the whole scan is a conflict -fn acceptance_unknown_routes_on( - conn: &rusqlite::Connection, - submission_type: &str, - is_unknown: fn(&SubmissionState) -> bool, -) -> Result, IdentityError> { - let mut statement = conn - .prepare( - "SELECT request_id,task_id,route_json FROM origin_routes - WHERE CASE WHEN json_valid(route_json) THEN - json_extract(route_json, '$.submission.type') = ?1 - AND json_extract(route_json, '$.submission.phase.type') = 'acceptance_unknown' - ELSE 0 END - ORDER BY request_id", - ) - .map_err(storage)?; - let rows = statement - .query_map([submission_type], |row| { - Ok(( - row.get::<_, String>(0)?, - row.get::<_, String>(1)?, - row.get::<_, String>(2)?, - )) - }) - .map_err(storage)?; - let mut routes = Vec::new(); - for row in rows { - let (request, task, json) = row.map_err(storage)?; - let route = decode_route(&json)?; - if route.request.0.to_string() != request - || route.task.to_string() != task - || !is_unknown(&route.submission) - { - return Err(IdentityError::Conflict); - } - routes.push(route); - } - Ok(routes) + data.as_deref().map(decode_identity).transpose() } fn validate_route(route: &OriginRoute) -> Result<(), IdentityError> { route.validate().map_err(|error| { IdentityError::Storage(AppError::Internal { - message: format!("invalid resource origin route: {error}"), + message: format!("invalid origin route: {error}"), }) }) } -fn same_route_identity(left: &OriginRoute, right: &OriginRoute) -> Result { - if left.request != right.request || left.origin_machine != right.origin_machine { - return Ok(false); - } - - // direct retries retain the first route's generated task, destination, and callback context - let same_spec = left.spec == right.spec; - let same_thread = left.thread == right.thread; - match (&left.submission, &right.submission) { - ( - SubmissionState::Resource { - resource: left_resource, - .. - }, - SubmissionState::Resource { - resource: right_resource, - .. - }, - ) => Ok(left.task == right.task - && left.execution_machine == right.execution_machine - && same_thread - && left.callback == right.callback - && *left_resource == *right_resource - && same_spec), - (SubmissionState::Resource { .. }, _) | (_, SubmissionState::Resource { .. }) => Ok(false), - // a retry must repeat the prepared identities, content, action, and callback - ( - SubmissionState::ResourceAction { - binding: left_binding, - launch: left_launch, - .. - }, - SubmissionState::ResourceAction { - binding: right_binding, - launch: right_launch, - .. - }, - ) => Ok(left.task == right.task - && left.execution_machine == right.execution_machine - && same_thread - && left.callback == right.callback - && left_binding == right_binding - && left_launch == right_launch - && same_spec), - (SubmissionState::ResourceAction { .. }, _) - | (_, SubmissionState::ResourceAction { .. }) => Ok(false), - // a retry must repeat the fixed identities, content, assignment, and callback - ( - SubmissionState::ResourceBackground { - binding: left_binding, - .. - }, - SubmissionState::ResourceBackground { - binding: right_binding, - .. - }, - ) => Ok(left.task == right.task - && left.execution_machine == right.execution_machine - && same_thread - && left.callback == right.callback - && left_binding == right_binding - && same_spec), - (SubmissionState::ResourceBackground { .. }, _) - | (_, SubmissionState::ResourceBackground { .. }) => Ok(false), - _ => Ok(same_thread && same_spec), - } -} - -/// Definitive authority result that resolves one remote first background launch route -#[derive(Debug, Clone)] -pub enum ResourceBackgroundRouteResult { - /// The authority saved this exact launch binding with the task - Accepted(RemoteBackgroundLaunchReceipt), - /// The authority refused the launch and wrote no task records - Rejected(ResourceBackgroundRejection), -} - -/// Definitive authority result that resolves one action-bound route -#[derive(Debug, Clone)] -pub enum ResourceActionRouteResult { - /// The authority saved this exact action binding with the task - Accepted(ActionTaskReceipt), - /// The authority refused the launch and wrote no task records - Rejected(crate::resource::bound_action::ResourceActionRejection), -} - -pub(super) fn resource_action_route_by_action_on( - conn: &rusqlite::Connection, - action: ActionId, -) -> Result, IdentityError> { - let mut statement = conn - .prepare( - "SELECT route_json FROM origin_routes - WHERE CASE WHEN json_valid(route_json) THEN - json_extract(route_json, '$.submission.type') = 'resource_action' - AND json_extract(route_json, '$.submission.binding.authority.action_id') = ?1 - ELSE 0 END", - ) - .map_err(storage)?; - let routes = statement - .query_map([action.as_uuid().to_string()], |row| { - row.get::<_, String>(0) - }) - .map_err(storage)? - .collect::, _>>() - .map_err(storage)?; - let mut routes = routes - .iter() - .map(|json| decode_route(json)) - .collect::, _>>()?; - if routes.len() > 1 { - return Err(IdentityError::Conflict); - } - Ok(routes.pop()) -} - -fn check_resource_receipt( - route: &OriginRoute, - request: RequestId, - task: TaskId, - origin: MachineId, - authority: MachineId, - resource: crate::resource::ResourceId, -) -> Result<(), IdentityError> { - let SubmissionState::Resource { - resource: route_resource, - .. - } = &route.submission - else { - return Err(IdentityError::Conflict); - }; - if route.request != request - || route.task != task - || route.origin_machine != origin - || route.execution_machine != authority - || *route_resource != resource - { - return Err(IdentityError::Conflict); - } - Ok(()) +/// Whether a retry repeats the saved route's request, owner, thread, and content +/// +/// Direct retries keep the first route's generated task, destination, and +/// callback context +fn same_route_identity(left: &OriginRoute, right: &OriginRoute) -> bool { + left.request == right.request + && left.origin_machine == right.origin_machine + && left.thread == right.thread + && left.spec == right.spec } fn save_route(tx: &rusqlite::Transaction<'_>, route: &OriginRoute) -> Result<(), IdentityError> { @@ -386,7 +137,7 @@ impl Store { let existing = decode_route(&saved)?; let saved_after: Option = saved_after.as_deref().map(decode).transpose()?; - if !same_route_identity(&existing, route)? || saved_after.as_ref() != after { + if !same_route_identity(&existing, route) || saved_after.as_ref() != after { return Err(IdentityError::Conflict); } return Ok(existing); @@ -396,24 +147,6 @@ impl Store { { return Err(IdentityError::Conflict); } - if let SubmissionState::Resource { phase, .. } = &route.submission - && !matches!(phase, ResourceRoutePhase::AcceptanceUnknown) - { - return Err(IdentityError::Conflict); - } - if let SubmissionState::ResourceBackground { phase, .. } = &route.submission - && !matches!(phase, ResourceBackgroundRoutePhase::AcceptanceUnknown) - { - return Err(IdentityError::Conflict); - } - if let SubmissionState::ResourceAction { binding, phase, .. } = &route.submission { - // one action binds at most one task, so another identity for it conflicts - if !matches!(phase, ResourceActionRoutePhase::AcceptanceUnknown) - || resource_action_route_by_action_on(&tx, binding.authority.action_id)?.is_some() - { - return Err(IdentityError::Conflict); - } - } let occupied: bool = tx .query_row( "SELECT EXISTS(SELECT 1 FROM origin_routes WHERE task_id=?1)", @@ -509,190 +242,6 @@ impl Store { Ok(routes) } - /// Find unresolved resource origin routes for one startup recovery pass - pub fn unknown_resource_origin_routes(&self) -> Result, IdentityError> { - acceptance_unknown_routes_on(&self.conn, "resource", |submission| { - matches!( - submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::AcceptanceUnknown, - .. - } - ) - }) - } - - /// Read the one saved action-bound route for a resource action - pub fn resource_action_route_by_action( - &self, - action: ActionId, - ) -> Result, IdentityError> { - resource_action_route_by_action_on(&self.conn, action) - } - - /// Find action-bound routes whose authority acceptance was unknown at startup - pub fn unknown_resource_action_routes(&self) -> Result, IdentityError> { - acceptance_unknown_routes_on(&self.conn, "resource_action", |submission| { - matches!( - submission, - SubmissionState::ResourceAction { - phase: ResourceActionRoutePhase::AcceptanceUnknown, - .. - } - ) - }) - } - - /// Apply a definitive authority result to one action-bound route - /// - /// An acceptance must carry the exact saved binding and identities. A later - /// phase is never replaced, and an identical retry leaves the route unchanged - pub fn resolve_resource_action_route( - &mut self, - task: TaskId, - result: &ResourceActionRouteResult, - ) -> Result { - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate) - .map_err(storage)?; - let data: Option = tx - .query_row( - "SELECT route_json FROM origin_routes WHERE task_id=?1", - [task.to_string()], - |row| row.get(0), - ) - .optional() - .map_err(storage)?; - let mut route = decode_route(data.as_deref().ok_or(IdentityError::RouteNotFound)?)?; - let SubmissionState::ResourceAction { - binding, - launch, - phase, - } = &route.submission - else { - return Err(IdentityError::Conflict); - }; - let target = match result { - ResourceActionRouteResult::Accepted(receipt) => { - let spec = route.current_spec().ok_or(IdentityError::Conflict)?; - let digest = crate::submission::normalized_spec_sha256(spec) - .map_err(|error| IdentityError::Storage(error.into()))?; - if receipt.kind != binding.kind - || receipt.authority != binding.authority - || receipt.request_id != route.request - || receipt.task_id != route.task - || receipt.normalized_spec_sha256 != digest - { - return Err(IdentityError::Conflict); - } - ResourceActionRoutePhase::Accepted - } - ResourceActionRouteResult::Rejected(reason) => ResourceActionRoutePhase::Rejected { - reason: reason.clone(), - }, - }; - let next = match (phase, &target) { - (ResourceActionRoutePhase::AcceptanceUnknown, _) => Some(target), - (saved, target) if saved == target => None, - _ => return Err(IdentityError::Conflict), - }; - if let Some(phase) = next { - route.submission = SubmissionState::ResourceAction { - binding: *binding, - launch: launch.clone(), - phase, - }; - route.last_updated_at = Some(chrono::Utc::now()); - validate_route(&route)?; - save_route(&tx, &route)?; - } - tx.commit().map_err(storage)?; - Ok(route) - } - - /// Find remote first background launch routes whose acceptance was unknown at startup - pub fn unknown_resource_background_routes(&self) -> Result, IdentityError> { - acceptance_unknown_routes_on(&self.conn, "resource_background", |submission| { - matches!( - submission, - SubmissionState::ResourceBackground { - phase: ResourceBackgroundRoutePhase::AcceptanceUnknown, - .. - } - ) - }) - } - - /// Apply a definitive authority result to one remote first background launch route - /// - /// An acceptance must carry the exact saved binding and identities. An accepted - /// route is never replaced, and an identical retry leaves the route unchanged - /// A refusal only replaces an unresolved phase, since a first queued event may - /// already have accepted the route - pub fn resolve_resource_background_route( - &mut self, - task: TaskId, - result: &ResourceBackgroundRouteResult, - ) -> Result { - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate) - .map_err(storage)?; - let data: Option = tx - .query_row( - "SELECT route_json FROM origin_routes WHERE task_id=?1", - [task.to_string()], - |row| row.get(0), - ) - .optional() - .map_err(storage)?; - let mut route = decode_route(data.as_deref().ok_or(IdentityError::RouteNotFound)?)?; - let SubmissionState::ResourceBackground { binding, phase } = &route.submission else { - return Err(IdentityError::Conflict); - }; - let target = match result { - ResourceBackgroundRouteResult::Accepted(receipt) => { - let spec = route.current_spec().ok_or(IdentityError::Conflict)?; - let digest = crate::submission::normalized_spec_sha256(spec) - .map_err(|error| IdentityError::Storage(error.into()))?; - if receipt.binding != *binding - || receipt.request_id != route.request - || receipt.task_id != route.task - || receipt.normalized_spec_sha256 != digest - { - return Err(IdentityError::Conflict); - } - ResourceBackgroundRoutePhase::Accepted - } - ResourceBackgroundRouteResult::Rejected(reason) => { - ResourceBackgroundRoutePhase::Rejected { - reason: reason.clone(), - } - } - }; - let next = match (phase, &target) { - (ResourceBackgroundRoutePhase::AcceptanceUnknown, _) - | ( - ResourceBackgroundRoutePhase::Rejected { .. }, - ResourceBackgroundRoutePhase::Accepted, - ) => Some(target), - (saved, target) if saved == target => None, - _ => return Err(IdentityError::Conflict), - }; - if let Some(phase) = next { - route.submission = SubmissionState::ResourceBackground { - binding: *binding, - phase, - }; - route.last_updated_at = Some(chrono::Utc::now()); - validate_route(&route)?; - save_route(&tx, &route)?; - } - tx.commit().map_err(storage)?; - Ok(route) - } - /// Set a definitive submission result only while acceptance is unknown /// /// A released held route resolves the same way. Nobody waits on its submit @@ -704,11 +253,7 @@ impl Store { ) -> Result { if matches!( outcome, - SubmissionState::AcceptanceUnknown - | SubmissionState::Resource { .. } - | SubmissionState::ResourceAction { .. } - | SubmissionState::ResourceBackground { .. } - | SubmissionState::Held { .. } + SubmissionState::AcceptanceUnknown | SubmissionState::Held { .. } ) { return Err(IdentityError::Conflict); } @@ -765,133 +310,6 @@ impl Store { } } - /// Apply a definitive resource queue response without changing a later phase - pub fn resolve_resource_route( - &mut self, - receipt: &ResourceQueueReceipt, - ) -> Result { - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate) - .map_err(storage)?; - let data: Option = tx - .query_row( - "SELECT route_json FROM origin_routes WHERE task_id=?1", - [receipt.task.to_string()], - |row| row.get(0), - ) - .optional() - .map_err(storage)?; - let mut route = decode_route(data.as_deref().ok_or(IdentityError::RouteNotFound)?)?; - check_resource_receipt( - &route, - receipt.request, - receipt.task, - receipt.origin_machine, - receipt.authority_machine, - receipt.resource, - )?; - let SubmissionState::Resource { resource, phase } = &route.submission else { - return Err(IdentityError::Conflict); - }; - let next = match (phase, &receipt.outcome) { - (ResourceRoutePhase::AcceptanceUnknown, ResourceQueueOutcome::Waiting) - | (ResourceRoutePhase::Waiting, ResourceQueueOutcome::Waiting) => { - Some(ResourceRoutePhase::Waiting) - } - (ResourceRoutePhase::AcceptanceUnknown, ResourceQueueOutcome::Rejected { reason }) => { - Some(ResourceRoutePhase::Rejected { - reason: reason.clone(), - }) - } - ( - ResourceRoutePhase::Rejected { reason: saved }, - ResourceQueueOutcome::Rejected { reason }, - ) if saved == reason => None, - (ResourceRoutePhase::Activated, ResourceQueueOutcome::Waiting) => None, - (ResourceRoutePhase::CancelledBeforeLaunch, ResourceQueueOutcome::Waiting) => None, - ( - ResourceRoutePhase::CancelledBeforeLaunch, - ResourceQueueOutcome::Rejected { reason }, - ) if reason == "cancelled_before_launch" => None, - _ => return Err(IdentityError::Conflict), - }; - if let Some(phase) = next { - route.submission = SubmissionState::Resource { - resource: *resource, - phase, - }; - route.last_updated_at = Some(chrono::Utc::now()); - validate_route(&route)?; - save_route(&tx, &route)?; - } - tx.commit().map_err(storage)?; - Ok(route) - } - - /// Apply a definitive authority cancellation receipt before task activation - pub fn cancel_resource_route_before_launch( - &mut self, - receipt: &ResourceCancellationReceipt, - ) -> Result { - if receipt.requester_machine != receipt.origin_machine - || !matches!( - &receipt.outcome, - ResourceCancellationOutcome::PreventedBeforeAcceptance - | ResourceCancellationOutcome::CancelledBeforeLaunch - ) - { - return Err(IdentityError::Conflict); - } - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate) - .map_err(storage)?; - let data: Option = tx - .query_row( - "SELECT route_json FROM origin_routes WHERE task_id=?1", - [receipt.task.to_string()], - |row| row.get(0), - ) - .optional() - .map_err(storage)?; - let mut route = decode_route(data.as_deref().ok_or(IdentityError::RouteNotFound)?)?; - check_resource_receipt( - &route, - receipt.request, - receipt.task, - receipt.origin_machine, - receipt.authority_machine, - receipt.resource, - )?; - let SubmissionState::Resource { resource, phase } = &route.submission else { - return Err(IdentityError::Conflict); - }; - let next = match (phase, &receipt.outcome) { - ( - ResourceRoutePhase::AcceptanceUnknown, - ResourceCancellationOutcome::PreventedBeforeAcceptance - | ResourceCancellationOutcome::CancelledBeforeLaunch, - ) - | (ResourceRoutePhase::Waiting, ResourceCancellationOutcome::CancelledBeforeLaunch) => { - Some(ResourceRoutePhase::CancelledBeforeLaunch) - } - (ResourceRoutePhase::CancelledBeforeLaunch, _) => None, - _ => return Err(IdentityError::Conflict), - }; - if let Some(phase) = next { - route.submission = SubmissionState::Resource { - resource: *resource, - phase, - }; - route.last_updated_at = Some(chrono::Utc::now()); - validate_route(&route)?; - save_route(&tx, &route)?; - } - tx.commit().map_err(storage)?; - Ok(route) - } - /// Read accepted identity or rejection tombstone by task UUID pub fn executor_identity( &self, @@ -926,16 +344,8 @@ impl Store { { return Err(IdentityError::Conflict); } - if matches!(&existing, ExecutorIdentity::Accepted(_)) - && resource_task_id_is_reserved(&tx, record.task).map_err(storage)? - { - return Err(IdentityError::Conflict); - } return Ok(existing); } - if resource_task_id_is_reserved(&tx, record.task).map_err(storage)? { - return Err(IdentityError::Conflict); - } let task_exists: bool = tx .query_row( "SELECT EXISTS(SELECT 1 FROM tasks WHERE id=?1)", @@ -1055,26 +465,17 @@ impl Store { #[cfg(test)] mod tests { - use crate::events::{EventPayload, TaskEvent}; - use crate::resource::bound_action::ResourceActionLaunch; use std::path::Path; use std::sync::{Arc, Barrier}; use tempfile::tempdir; - use super::{IdentityError, ResourceActionRouteResult, encode}; + use super::{IdentityError, encode}; use crate::domain::{ProcessStatus, TaskEnv, TaskId}; use crate::machine::MachineId; - use crate::resource::bound_action::ActionTaskReceipt; - use crate::resource::{ - ActionId, AssignmentRevision, Resource, ResourceId, ResourceRevision, SupervisorAddress, - }; use crate::store::Store; use crate::submission::{ - CallbackContext, ExecutionRecord, ExecutorIdentity, OriginRoute, PreAcceptanceRejection, - RequestId, ResourceActionRoutePhase, ResourceCancellationOutcome, - ResourceCancellationReceipt, ResourceQueueOutcome, ResourceQueueReceipt, - ResourceRoutePhase, SubmissionState, + CallbackContext, ExecutionRecord, ExecutorIdentity, OriginRoute, RequestId, SubmissionState, }; use rusqlite::params; @@ -1115,28 +516,6 @@ mod tests { } } - fn resource_route() -> OriginRoute { - let spec = spec(); - OriginRoute::new_resource_waiting(crate::submission::NewResourceRoute { - request: RequestId::new(), - task: TaskId::new(), - origin_machine: MachineId::new(), - authority_machine: MachineId::new(), - thread: spec.thread, - callback: CallbackContext { - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - cwd: Path::new("/tmp").to_path_buf(), - codex: Path::new("/bin/echo").to_path_buf().into(), - }, - spec, - resource: ResourceId::new(), - }) - .unwrap() - } - #[test] fn request_identity_and_origin_transition() { let dir = tempdir().unwrap(); @@ -1212,7 +591,7 @@ mod tests { #[test] fn saved_direct_route_json_is_readable_and_recoverable() { let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); + let store = Store::open(&dir.path().join("db")).unwrap(); let route = route(); let json = serde_json::to_value(&route).unwrap(); store @@ -1238,388 +617,13 @@ mod tests { saved.submission, SubmissionState::AcceptanceUnknown )); - let resource = resource_route(); - store.insert_origin_route(&resource).unwrap(); let unknown = store.unknown_origin_routes().unwrap(); assert_eq!(unknown.len(), 1); assert_eq!(unknown[0].request, saved.request); - let unknown_resource = store.unknown_resource_origin_routes().unwrap(); - assert_eq!(unknown_resource.len(), 1); - assert_eq!(unknown_resource[0].request, resource.request); } #[test] - fn unknown_resource_scan_selects_only_resource_acceptance_unknown_routes() { - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let direct = route(); - store.insert_origin_route(&direct).unwrap(); - - let unresolved = resource_route(); - store.insert_origin_route(&unresolved).unwrap(); - - let waiting = resource_route(); - store.insert_origin_route(&waiting).unwrap(); - let SubmissionState::Resource { resource, .. } = &waiting.submission else { - unreachable!(); - }; - store - .resolve_resource_route(&ResourceQueueReceipt { - request: waiting.request, - task: waiting.task, - origin_machine: waiting.origin_machine, - authority_machine: waiting.execution_machine, - resource: *resource, - outcome: ResourceQueueOutcome::Waiting, - }) - .unwrap(); - - let activated = resource_route(); - store.insert_origin_route(&activated).unwrap(); - store - .accept_inbound_event(&TaskEvent { - task: activated.task, - seq: std::num::NonZeroU64::new(1).unwrap(), - origin_machine: activated.origin_machine, - execution_machine: activated.execution_machine, - payload: EventPayload::State { - status: ProcessStatus::Queued, - }, - }) - .unwrap(); - - let cancelled = resource_route(); - store.insert_origin_route(&cancelled).unwrap(); - let SubmissionState::Resource { resource, .. } = &cancelled.submission else { - unreachable!(); - }; - store - .cancel_resource_route_before_launch(&ResourceCancellationReceipt { - cancellation: uuid::Uuid::now_v7(), - requester_machine: cancelled.origin_machine, - request: cancelled.request, - task: cancelled.task, - origin_machine: cancelled.origin_machine, - authority_machine: cancelled.execution_machine, - resource: *resource, - target_phase: ResourceRoutePhase::AcceptanceUnknown, - outcome: ResourceCancellationOutcome::CancelledBeforeLaunch, - }) - .unwrap(); - - let rejected = resource_route(); - store.insert_origin_route(&rejected).unwrap(); - let SubmissionState::Resource { resource, .. } = &rejected.submission else { - unreachable!(); - }; - store - .resolve_resource_route(&ResourceQueueReceipt { - request: rejected.request, - task: rejected.task, - origin_machine: rejected.origin_machine, - authority_machine: rejected.execution_machine, - resource: *resource, - outcome: ResourceQueueOutcome::Rejected { - reason: "resource unavailable".into(), - }, - }) - .unwrap(); - - let resource_unknown = store.unknown_resource_origin_routes().unwrap(); - assert_eq!(resource_unknown.len(), 1); - assert_eq!(resource_unknown[0].request, unresolved.request); - assert_eq!(resource_unknown[0].task, unresolved.task); - - let direct_unknown = store.unknown_origin_routes().unwrap(); - assert_eq!(direct_unknown.len(), 1); - assert_eq!(direct_unknown[0].request, direct.request); - - store - .conn - .execute( - "INSERT INTO origin_routes (request_id,task_id,execution_machine,spec_json,route_json) - VALUES (?1,?2,?3,?4,?5)", - params![ - RequestId::new().0.to_string(), - TaskId::new().to_string(), - MachineId::new().as_uuid().to_string(), - encode(&spec()).unwrap(), - "{malformed route JSON", - ], - ) - .unwrap(); - assert_eq!(store.unknown_resource_origin_routes().unwrap().len(), 1); - } - - #[test] - fn resource_route_validation_and_idempotent_identity_include_callback() { - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let route = resource_route(); - store.insert_origin_route(&route).unwrap(); - assert_eq!(store.insert_origin_route(&route).unwrap().task, route.task); - - let mut changed_callback = route.clone(); - changed_callback.callback.env.home = "/different-origin".into(); - assert!(matches!( - store.insert_origin_route(&changed_callback), - Err(IdentityError::Conflict) - )); - let mut changed_owner = route.clone(); - changed_owner.execution_machine = MachineId::new(); - assert!(matches!( - store.insert_origin_route(&changed_owner), - Err(IdentityError::Conflict) - )); - let mut changed_task = route.clone(); - changed_task.task = TaskId::new(); - assert!(matches!( - store.insert_origin_route(&changed_task), - Err(IdentityError::Conflict) - )); - let mut changed_resource = route.clone(); - if let SubmissionState::Resource { resource, phase } = &route.submission { - let original_resource = *resource; - let changed_id = ResourceId::new(); - changed_resource.submission = SubmissionState::Resource { - resource: changed_id, - phase: phase.clone(), - }; - assert_ne!(original_resource, changed_id); - } - assert!(matches!( - store.insert_origin_route(&changed_resource), - Err(IdentityError::Conflict) - )); - - let mut invalid = resource_route(); - invalid.thread = crate::domain::ThreadId(uuid::Uuid::now_v7()); - assert!(matches!( - store.insert_origin_route(&invalid), - Err(IdentityError::Storage(_)) - )); - invalid = route.clone(); - invalid.spec.current_mut().unwrap().machine = - Some(serde_json::from_value(serde_json::json!("gpu-authority")).unwrap()); - assert!(matches!( - store.insert_origin_route(&invalid), - Err(IdentityError::Storage(_)) - )); - - store - .conn - .execute( - "UPDATE origin_routes SET route_json=?1 WHERE request_id=?2", - params![ - serde_json::to_string(&invalid).unwrap(), - route.request.0.to_string() - ], - ) - .unwrap(); - assert!(matches!( - store.origin_route_by_request(route.request), - Err(IdentityError::Storage(_)) - )); - } - - #[test] - fn resource_queue_receipt_cas_does_not_overwrite_activation() { - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let route = resource_route(); - store.insert_origin_route(&route).unwrap(); - let SubmissionState::Resource { resource, .. } = route.submission else { - unreachable!(); - }; - let receipt = ResourceQueueReceipt { - request: route.request, - task: route.task, - origin_machine: route.origin_machine, - authority_machine: route.execution_machine, - resource, - outcome: ResourceQueueOutcome::Waiting, - }; - - let waiting = store.resolve_resource_route(&receipt).unwrap(); - assert!(matches!( - waiting.submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::Waiting, - .. - } - )); - store - .accept_inbound_event(&TaskEvent { - task: route.task, - seq: std::num::NonZeroU64::new(1).unwrap(), - origin_machine: route.origin_machine, - execution_machine: route.execution_machine, - payload: EventPayload::State { - status: ProcessStatus::Queued, - }, - }) - .unwrap(); - let activated = store.origin_route_by_task(route.task).unwrap().unwrap(); - let late_receipt = store.resolve_resource_route(&receipt).unwrap(); - assert_eq!(late_receipt.submission, activated.submission); - - let mut late_rejection = receipt.clone(); - late_rejection.outcome = ResourceQueueOutcome::Rejected { - reason: "stale rejection".into(), - }; - assert!(matches!( - store.resolve_resource_route(&late_rejection), - Err(IdentityError::Conflict) - )); - let late_cancellation = ResourceCancellationReceipt { - cancellation: uuid::Uuid::now_v7(), - requester_machine: route.origin_machine, - request: route.request, - task: route.task, - origin_machine: route.origin_machine, - authority_machine: route.execution_machine, - resource, - target_phase: ResourceRoutePhase::AcceptanceUnknown, - outcome: ResourceCancellationOutcome::CancelledBeforeLaunch, - }; - assert!(matches!( - store.cancel_resource_route_before_launch(&late_cancellation), - Err(IdentityError::Conflict) - )); - let saved = store.origin_route_by_task(route.task).unwrap().unwrap(); - assert_eq!(saved.submission, activated.submission); - } - - #[test] - fn definitive_cancellation_ignores_late_queue_receipts_without_regressing() { - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let route = resource_route(); - store.insert_origin_route(&route).unwrap(); - let SubmissionState::Resource { resource, .. } = route.submission else { - unreachable!(); - }; - let identity = ResourceCancellationReceipt { - cancellation: uuid::Uuid::now_v7(), - requester_machine: route.origin_machine, - request: route.request, - task: route.task, - origin_machine: route.origin_machine, - authority_machine: route.execution_machine, - resource, - target_phase: ResourceRoutePhase::AcceptanceUnknown, - outcome: ResourceCancellationOutcome::CancelledBeforeLaunch, - }; - let waiting = ResourceQueueReceipt { - request: route.request, - task: route.task, - origin_machine: route.origin_machine, - authority_machine: route.execution_machine, - resource, - outcome: ResourceQueueOutcome::Waiting, - }; - store.resolve_resource_route(&waiting).unwrap(); - - let cancelled = store - .cancel_resource_route_before_launch(&identity) - .unwrap(); - assert!(matches!( - cancelled.submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::CancelledBeforeLaunch, - .. - } - )); - - let late_waiting = store.resolve_resource_route(&waiting).unwrap(); - assert_eq!(late_waiting.submission, cancelled.submission); - let late_cancelled = ResourceQueueReceipt { - outcome: ResourceQueueOutcome::Rejected { - reason: "cancelled_before_launch".into(), - }, - ..waiting.clone() - }; - let late_cancelled = store.resolve_resource_route(&late_cancelled).unwrap(); - assert_eq!(late_cancelled.submission, cancelled.submission); - - let stale_rejection = ResourceQueueReceipt { - outcome: ResourceQueueOutcome::Rejected { - reason: "unrelated rejection".into(), - }, - ..waiting.clone() - }; - assert!(matches!( - store.resolve_resource_route(&stale_rejection), - Err(IdentityError::Conflict) - )); - - let mismatched = ResourceQueueReceipt { - request: RequestId::new(), - ..waiting - }; - assert!(matches!( - store.resolve_resource_route(&mismatched), - Err(IdentityError::Conflict) - )); - assert!(matches!( - store.accept_inbound_event(&TaskEvent { - task: route.task, - seq: std::num::NonZeroU64::new(1).unwrap(), - origin_machine: route.origin_machine, - execution_machine: route.execution_machine, - payload: EventPayload::State { - status: ProcessStatus::Queued, - }, - }), - Err(crate::events::EventError::Invalid { .. }) - )); - let saved = store.origin_route_by_task(route.task).unwrap().unwrap(); - assert_eq!(saved.last_accepted_seq, 0); - assert_eq!(saved.submission, cancelled.submission); - } - - #[test] - fn resource_rejection_is_idempotent_and_cannot_become_waiting() { - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let route = resource_route(); - store.insert_origin_route(&route).unwrap(); - let SubmissionState::Resource { resource, .. } = &route.submission else { - unreachable!(); - }; - let rejected = ResourceQueueReceipt { - request: route.request, - task: route.task, - origin_machine: route.origin_machine, - authority_machine: route.execution_machine, - resource: *resource, - outcome: ResourceQueueOutcome::Rejected { - reason: "resource unavailable".into(), - }, - }; - let saved = store.resolve_resource_route(&rejected).unwrap(); - assert!(matches!( - &saved.submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::Rejected { .. }, - .. - } - )); - assert_eq!( - store.resolve_resource_route(&rejected).unwrap().submission, - saved.submission - ); - - let mut stale_waiting = rejected; - stale_waiting.outcome = ResourceQueueOutcome::Waiting; - assert!(matches!( - store.resolve_resource_route(&stale_waiting), - Err(IdentityError::Conflict) - )); - } - - #[test] - fn accept_and_abandon_serialize_across_connections() { + fn accept_and_abandon_serialize_across_connections() { let dir = tempdir().unwrap(); let path = dir.path().join("db"); let task = TaskId::new(); @@ -1729,128 +733,6 @@ mod tests { ); } - #[test] - fn direct_acceptance_respects_resource_ownership_and_reuses_cancellation_tombstone() { - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let resource = Resource::new( - ResourceId::new(), - "gpu-0".into(), - authority, - SupervisorAddress { - machine: authority, - thread: spec().thread, - }, - AssignmentRevision::new(0), - ResourceRevision::new(0), - None, - ); - store.register_resource(authority, &resource).unwrap(); - - let queued_task = TaskId::new(); - store - .accept_resource_request( - authority, - RequestId::new(), - queued_task, - resource.id, - origin, - spec(), - ) - .unwrap(); - let queued_execution = ExecutionRecord { - task: queued_task, - origin_machine: origin, - execution_machine: authority, - spec: spec().into(), - state: ProcessStatus::Queued, - }; - assert!(matches!( - store.accept_execution(&queued_execution), - Err(IdentityError::Conflict) - )); - assert!(store.executor_identity(queued_task).unwrap().is_none()); - - let prevented_without_tombstone = TaskId::new(); - store - .conn - .execute( - "INSERT INTO resource_request_preventions ( - request_id, task_id, resource_id, origin_machine - ) VALUES (?1, ?2, ?3, ?4)", - params![ - RequestId::new().0.to_string(), - prevented_without_tombstone.to_string(), - resource.id.as_uuid().to_string(), - origin.as_uuid().to_string(), - ], - ) - .unwrap(); - let prevented_execution = ExecutionRecord { - task: prevented_without_tombstone, - origin_machine: origin, - execution_machine: authority, - spec: spec().into(), - state: ProcessStatus::Queued, - }; - assert!(matches!( - store.accept_execution(&prevented_execution), - Err(IdentityError::Conflict) - )); - assert!( - store - .executor_identity(prevented_without_tombstone) - .unwrap() - .is_none() - ); - - let prevented_task = TaskId::new(); - let request = RequestId::new(); - assert!(matches!( - store.cancel_resource_request_before_activation( - authority, - request, - prevented_task, - resource.id, - origin, - ), - Ok(crate::resource::store::QueueCancellationResult::PreventedBeforeAcceptance) - )); - - let cancelled_execution = ExecutionRecord { - task: prevented_task, - origin_machine: origin, - execution_machine: authority, - spec: spec().into(), - state: ProcessStatus::Queued, - }; - let delayed = store.accept_execution(&cancelled_execution).unwrap(); - let ExecutorIdentity::Rejected(delayed_tombstone) = delayed else { - panic!("resource cancellation must retain a rejection"); - }; - assert_eq!( - delayed_tombstone.reason, - PreAcceptanceRejection::Cancelled.as_str() - ); - - let retried = store.accept_execution(&cancelled_execution).unwrap(); - let ExecutorIdentity::Rejected(retried_tombstone) = retried else { - panic!("a delayed retry must reuse the rejection"); - }; - assert_eq!(retried_tombstone.task, delayed_tombstone.task); - assert_eq!( - retried_tombstone.origin_machine, - delayed_tombstone.origin_machine - ); - assert_eq!( - retried_tombstone.execution_machine, - delayed_tombstone.execution_machine - ); - assert_eq!(retried_tombstone.reason, delayed_tombstone.reason); - } - #[test] fn accepted_identity_retains_state_and_rejects_changed_content() { let dir = tempdir().unwrap(); @@ -1875,186 +757,4 @@ mod tests { Err(IdentityError::Conflict) )); } - - fn action_route(launch: ResourceActionLaunch) -> OriginRoute { - let spec = spec(); - let binding = crate::submission::ResourceActionRouteBinding { - kind: launch.kind(), - authority: crate::resource::SupervisorActionAuthority { - authority_machine: MachineId::new(), - resource_id: ResourceId::new(), - loan_id: crate::resource::LoanId::new(), - action_id: ActionId::new(), - expected_state_revision: ResourceRevision::new(4), - supervisor: SupervisorAddress { - machine: MachineId::new(), - thread: spec.thread, - }, - assignment_revision: AssignmentRevision::new(2), - }, - }; - OriginRoute::new_resource_action(crate::submission::NewResourceActionRoute { - request: RequestId::new(), - task: TaskId::new(), - callback: CallbackContext { - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - cwd: Path::new("/tmp").to_path_buf(), - codex: Path::new("/bin/echo").to_path_buf().into(), - }, - spec, - binding, - launch, - }) - .unwrap() - } - - fn watcher_route() -> OriginRoute { - action_route(ResourceActionLaunch::ReleaseWatcher { - observed_background_task: TaskId::new(), - }) - } - - fn action_receipt(route: &OriginRoute) -> ActionTaskReceipt { - let SubmissionState::ResourceAction { binding, .. } = &route.submission else { - panic!("fixture must be an action route"); - }; - ActionTaskReceipt { - kind: binding.kind, - authority: binding.authority, - request_id: route.request, - task_id: route.task, - normalized_spec_sha256: crate::submission::normalized_spec_sha256( - route.current_spec().unwrap(), - ) - .unwrap(), - } - } - - #[test] - fn action_route_retry_needs_the_same_identity_content_and_one_route_per_action() { - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let route = watcher_route(); - let SubmissionState::ResourceAction { binding, .. } = &route.submission else { - unreachable!(); - }; - let action = binding.authority.action_id; - assert_eq!(store.insert_origin_route(&route).unwrap().task, route.task); - // an exact retry returns the saved route - assert_eq!(store.insert_origin_route(&route).unwrap().task, route.task); - - let mut changed = route.clone(); - changed.callback.cwd = Path::new("/var").to_path_buf(); - assert!(matches!( - store.insert_origin_route(&changed), - Err(IdentityError::Conflict) - )); - let mut other_task = route.clone(); - other_task.task = TaskId::new(); - assert!(matches!( - store.insert_origin_route(&other_task), - Err(IdentityError::Conflict) - )); - // a second identity for the same action cannot be saved - let mut second = route.clone(); - second.request = RequestId::new(); - second.task = TaskId::new(); - assert!(matches!( - store.insert_origin_route(&second), - Err(IdentityError::Conflict) - )); - assert_eq!( - store - .resource_action_route_by_action(action) - .unwrap() - .unwrap() - .task, - route.task - ); - - // recovery retries it by identity; the generic abandon scan never sees it - assert_eq!(store.unknown_resource_action_routes().unwrap().len(), 1); - assert!(store.unknown_origin_routes().unwrap().is_empty()); - assert!(store.unknown_resource_origin_routes().unwrap().is_empty()); - - // the saved route survives reopen with the same launch choice - drop(store); - let store = Store::open(&dir.path().join("db")).unwrap(); - let saved = store.origin_route_by_task(route.task).unwrap().unwrap(); - assert!(matches!( - saved.submission, - SubmissionState::ResourceAction { - phase: ResourceActionRoutePhase::AcceptanceUnknown, - .. - } - )); - assert_eq!(saved.callback, route.callback); - } - - #[test] - fn action_route_resolution_requires_the_exact_receipt_and_never_moves_back() { - let dir = tempdir().unwrap(); - let mut store = Store::open(&dir.path().join("db")).unwrap(); - let route = watcher_route(); - store.insert_origin_route(&route).unwrap(); - let receipt = action_receipt(&route); - - let mut other_supervisor = receipt; - other_supervisor.authority.assignment_revision = AssignmentRevision::new(3); - assert!(matches!( - store.resolve_resource_action_route( - route.task, - &ResourceActionRouteResult::Accepted(other_supervisor) - ), - Err(IdentityError::Conflict) - )); - let mut other_digest = receipt; - other_digest.normalized_spec_sha256 = crate::submission::normalized_spec_sha256(&{ - let mut spec = spec(); - spec.timeout = std::time::Duration::from_secs(1); - spec - }) - .unwrap(); - assert!(matches!( - store.resolve_resource_action_route( - route.task, - &ResourceActionRouteResult::Accepted(other_digest) - ), - Err(IdentityError::Conflict) - )); - - let accepted = store - .resolve_resource_action_route( - route.task, - &ResourceActionRouteResult::Accepted(receipt), - ) - .unwrap(); - assert!(matches!( - accepted.submission, - SubmissionState::ResourceAction { - phase: ResourceActionRoutePhase::Accepted, - .. - } - )); - // a duplicate acceptance is idempotent, and a rejection cannot replace it - store - .resolve_resource_action_route( - route.task, - &ResourceActionRouteResult::Accepted(receipt), - ) - .unwrap(); - assert!(matches!( - store.resolve_resource_action_route( - route.task, - &ResourceActionRouteResult::Rejected( - crate::resource::bound_action::ResourceActionRejection::ActionNotPending - ) - ), - Err(IdentityError::Conflict) - )); - assert!(store.unknown_resource_action_routes().unwrap().is_empty()); - } } diff --git a/src/store/resource.rs b/src/store/resource.rs deleted file mode 100644 index 6c030ae..0000000 --- a/src/store/resource.rs +++ /dev/null @@ -1,325 +0,0 @@ -//! StoreActor-facing operations for authority-local resource state - -mod action_task; -mod assigned_task; -mod background; -mod cancellation; -mod controls; -mod initial_idle; -mod operator_release; -mod release_checkpoint; -mod release_completion; -mod release_proof; -mod release_watcher; -mod restore; -#[cfg(test)] -pub(crate) mod test_support; -#[cfg(test)] -mod tests; -mod trainer_association; -mod trainer_lock; - -pub(crate) use action_task::{ - AcceptedActionTask, RemoteReleaseWatcherAcceptanceInput, ResourceActionError, -}; -pub(crate) use background::{ - BackgroundLaunchAcceptance, BackgroundLaunchError, BackgroundLaunchInput, - BackgroundLaunchPhase, BackgroundLaunchView, RemoteBackgroundLaunchInput, - idle_boundary_decision_on, idle_opening_matches_on, open_idle_serving_loan_on, - pending_background_launch_on, promote_started_background_launch_on, -}; -pub(crate) use controls::{ - ResourceControlEffect, ResourceControlError, ResourceControlRequest, ResourceControlStart, - ResourceReadModel, SupervisorReplacement, open_action_id, -}; -pub(crate) use initial_idle::InitialIdleError; -pub(crate) use operator_release::{OperatorGpuFreeError, operator_serving_release_matches_on}; -pub(crate) use release_proof::VerifiedReleaseProof; -pub(crate) use restore::{ - EndedRestoreResolution, PreparedReturnTask, RestoreReconcileOutcome, ReturnClosure, - ReturnDecisionError, ReturnTaskAcceptance, ReturnTaskAcceptanceInput, ReturnTaskOrigin, -}; - -use rusqlite::{Connection, params}; -use serde::Serialize; -use serde::de::DeserializeOwned; - -use super::Store; -use crate::domain::{TaskId, TaskRow}; -use crate::machine::MachineId; -use crate::resource::store::{ - ResourceQueueReconcileError, ResourceSnapshot, ResourceStoreError, ReturnDeadlineOutcome, - SupervisorNoticeStoreError, - bind_release_watcher_for_authority as persist_release_watcher_binding, - pending_supervisor_notices as load_pending_supervisor_notices, - reconcile_resource_queue_for_authority as persist_resource_queue_reconciliation, - recover_sending_supervisor_notices as recover_in_flight_supervisor_notices, - register_resource_for_authority, requests_for_resource_for_authority, - reserve_supervisor_notice_attempt as reserve_notice_attempt, - resources_for_authority as load_resources_for_authority, select_resource, - serve_after_return_deadline_for_authority, - settle_supervisor_notice_attempt as settle_notice_attempt, - supervisor_notice as load_supervisor_notice, -}; -use crate::resource::{ - ActionId, DeliveryAttemptId, Loan, NoticeId, ReleaseWatcherIntent, Resource, ResourceId, - ResourceQueueReconcileOutcome, ResourceRequest, SupervisorNotice, -}; -use crate::spec::NormalizedSpec; -use crate::submission::RequestId; - -impl Store { - /// Register a resource only on the daemon that matches its fixed authority - pub(crate) fn register_resource( - &mut self, - authority_machine: MachineId, - resource: &Resource, - ) -> Result { - register_resource_for_authority(&mut self.conn, authority_machine, resource) - } - - /// Load authority-owned resources and active loans on the store connection - pub(crate) fn resource_snapshots_for_authority( - &self, - authority_machine: MachineId, - ) -> Result, ResourceStoreError> { - load_resources_for_authority(&self.conn, authority_machine) - } - - /// Read a resource's requests in serving order - pub(crate) fn resource_requests( - &self, - authority_machine: MachineId, - resource_id: ResourceId, - ) -> Result, ResourceStoreError> { - requests_for_resource_for_authority(&self.conn, authority_machine, resource_id) - } - - /// Reconcile queued work from authority-owned resource, loan, and task rows - pub(crate) fn reconcile_resource_queue_for_authority( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - ) -> Result { - persist_resource_queue_reconciliation(&mut self.conn, authority_machine, resource_id) - } - - /// Serve the next queued request once the pending return action's window closed - pub(crate) fn serve_after_return_deadline_for_authority( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - ) -> Result { - serve_after_return_deadline_for_authority( - &mut self.conn, - authority_machine, - resource_id, - chrono::Utc::now(), - ) - } - - /// Bind one preallocated watcher launch identity to its saved release action - pub(crate) fn bind_release_watcher_for_authority( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - intent: ReleaseWatcherIntent, - ) -> Result { - persist_release_watcher_binding(&mut self.conn, authority_machine, resource_id, intent) - } - - /// Read one durable supervisor notice by its identity - pub(crate) fn supervisor_notice( - &self, - notice_id: NoticeId, - ) -> Result, SupervisorNoticeStoreError> { - load_supervisor_notice(&self.conn, notice_id) - } - - /// List notices that can receive another delivery attempt - pub(crate) fn pending_supervisor_notices( - &self, - ) -> Result, SupervisorNoticeStoreError> { - load_pending_supervisor_notices(&self.conn) - } - - /// Reserve one bounded delivery attempt on this store connection - pub(crate) fn reserve_supervisor_notice_attempt( - &mut self, - notice_id: NoticeId, - attempt_id: DeliveryAttemptId, - ) -> Result { - reserve_notice_attempt(&mut self.conn, notice_id, attempt_id) - } - - /// Settle only the exact in-flight delivery attempt - pub(crate) fn settle_supervisor_notice_attempt( - &mut self, - notice_id: NoticeId, - attempt_id: DeliveryAttemptId, - result: Result<(), String>, - ) -> Result { - settle_notice_attempt(&mut self.conn, notice_id, attempt_id, result) - } - - /// Recover in-flight notices after a daemon restart - pub(crate) fn recover_sending_supervisor_notices( - &mut self, - ) -> Result, SupervisorNoticeStoreError> { - recover_in_flight_supervisor_notices(&mut self.conn) - } -} - -/// Read one resource and require that this daemon is its fixed authority -fn select_authority_resource( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, -) -> Result { - let resource = - select_resource(conn, resource_id)?.ok_or(ResourceStoreError::ResourceNotFound)?; - if resource.authority_machine() != authority_machine { - return Err(ResourceStoreError::WrongAuthority { - expected: resource.authority_machine(), - found: authority_machine, - }); - } - - Ok(resource) -} - -/// Active loan phase that carries the release or return action being replaced -#[derive(Debug, Clone, Copy)] -enum LoanActionPhase { - AwaitingRelease, - AwaitingReturn, - Restoring, -} - -impl LoanActionPhase { - /// Serialized `$.phase.type` tag of this phase - const fn tag(self) -> &'static str { - match self { - Self::AwaitingRelease => "awaiting_release", - Self::AwaitingReturn => "awaiting_return", - Self::Restoring => "restoring", - } - } -} - -/// Replace one active loan's state only while it is still in the exact action phase -/// -/// Returns false when the loan left that phase, so each caller names its own refusal -fn replace_loan_in_action_phase_on( - conn: &Connection, - loan: &Loan, - phase: LoanActionPhase, - action_id: ActionId, -) -> Result { - let changed = conn.execute( - "UPDATE loans SET state_json = ?1 - WHERE id = ?2 AND resource_id = ?3 - AND json_extract(state_json, '$.type') = 'active' - AND json_extract(state_json, '$.phase.type') = ?4 - AND json_extract(state_json, '$.phase.action_id') = ?5", - params![ - encode_resource_json(&loan.state)?, - loan.id.as_uuid().to_string(), - loan.resource_id.as_uuid().to_string(), - phase.tag(), - action_id.as_uuid().to_string(), - ], - )?; - - Ok(changed == 1) -} - -/// Decode one saved JSON record; a failure is corrupt data, never a retryable storage error -fn decode_resource_json( - what: &'static str, - value: &str, -) -> Result { - serde_json::from_str(value).map_err(|error| ResourceStoreError::corrupt(what, error)) -} - -fn encode_resource_json(value: &T) -> Result { - serde_json::to_string(value).map_err(|error| { - ResourceStoreError::Storage(rusqlite::Error::ToSqlConversionFailure(Box::new(error))) - }) -} - -fn task_has_any_event(conn: &Connection, task: TaskId) -> Result { - conn.query_row( - "SELECT EXISTS( - SELECT 1 FROM executor_outbox WHERE task_id=?1 - UNION ALL SELECT 1 FROM executor_event_receipts WHERE task_id=?1 - UNION ALL SELECT 1 FROM executor_event_cursors WHERE task_id=?1 - UNION ALL SELECT 1 FROM executor_event_routes WHERE task_id=?1 - )", - [task.to_string()], - |row| row.get(0), - ) -} - -/// Whether a task row runs exactly the command of this accepted spec -fn resource_task_row_matches(row: &TaskRow, task: TaskId, spec: &NormalizedSpec) -> bool { - row.id == task - && row.name.as_ref() == Some(&spec.name) - && row.thread == spec.thread - && row.workload == crate::invocation::persist_workload(&spec.workload) - && row.cwd == spec.cwd - && row.timeout == spec.timeout - && binary_is_resolved_or_unresolvable(&row.binary) -} - -/// Accepted rows name an absolute binary, or an empty path when the executable -/// never resolved and the task failed before launch -fn binary_is_resolved_or_unresolvable(binary: &std::path::Path) -> bool { - binary.is_absolute() || binary.as_os_str().is_empty() -} - -/// Whether two task rows share every immutable launch field -fn same_task_binding(left: &TaskRow, right: &TaskRow) -> bool { - left.id == right.id - && left.name == right.name - && left.thread == right.thread - && left.workload == right.workload - && left.cwd == right.cwd - && left.timeout == right.timeout - && left.env == right.env - && left.binary == right.binary -} - -/// Whether any request, task, route, identity, or event already claims a fresh identity pair -fn task_identity_is_used( - conn: &Connection, - request_id: RequestId, - task_id: TaskId, -) -> Result { - Ok(resource_request_identity_exists(conn, request_id, task_id)? - || super::release_watcher_task_id_is_reserved(conn, task_id)? - || super::task_by_id_on(conn, task_id) - .map_err(ResourceStoreError::TaskRow)? - .is_some() - || super::identity::origin_route_by_request_on(conn, request_id)?.is_some() - || super::identity::origin_route_by_task_on(conn, task_id)?.is_some() - || super::identity::executor_identity_on(conn, task_id)?.is_some() - || task_has_any_event(conn, task_id)?) -} - -/// Whether a resource request or prevention already claims this request or task identity -fn resource_request_identity_exists( - conn: &Connection, - request: RequestId, - task: TaskId, -) -> Result { - conn.query_row( - "SELECT EXISTS( - SELECT 1 FROM resource_requests WHERE request_id=?1 OR task_id=?2 - UNION ALL - SELECT 1 FROM resource_request_preventions WHERE request_id=?1 OR task_id=?2 - )", - params![request.0.to_string(), task.to_string()], - |row| row.get(0), - ) -} diff --git a/src/store/resource/action_task.rs b/src/store/resource/action_task.rs deleted file mode 100644 index c494181..0000000 --- a/src/store/resource/action_task.rs +++ /dev/null @@ -1,424 +0,0 @@ -//! Authority acceptance of action-bound tasks whose supervisor runs on another machine -//! -//! The supervisor machine saves the callback route before it sends a launch. The -//! authority saves the task row, remote executor identity, first queued event, and -//! exact action receipt in one IMMEDIATE transaction. A retry with the same receipt -//! observes the saved task and never inserts or spawns it again - -use crate::store::IdentityError; -use rusqlite::{Connection, OptionalExtension, TransactionBehavior, params}; - -use super::release_checkpoint::validate_release_checkpoint_baseline_on; -use super::resource_task_row_matches; -use crate::domain::{ProcessStatus, TaskId, TaskRow}; -use crate::error::AppError; -use crate::machine::MachineId; -use crate::resource::bound_action::{ - ActionTaskAcceptance, ActionTaskIdentity, ActionTaskReceipt, ResourceActionKind, - ResourceActionRejection, -}; -use crate::resource::release_watcher::ReleaseWatcherCommand; -use crate::resource::store::{ - ReleaseCheckpointError, ResourceStoreError, select_non_closed_loan, select_resource, - validate_release_watcher_intent_on, -}; -use crate::resource::{ - ActionId, LoanPhase, LoanState, ReleaseWatcherIntent, SupervisorActionAuthority, -}; -use crate::spec::NormalizedSpec; -use crate::submission::{ExecutorIdentity, normalized_spec_sha256}; - -/// A remote action request was refused, or storage failed -#[derive(Debug, thiserror::Error)] -pub(crate) enum ResourceActionError { - /// The request does not fit the saved action, and nothing was written - #[error("resource action rejected: {0:?}")] - Rejected(ResourceActionRejection), - /// SQLite, encoding, or stored data failed - #[error(transparent)] - Storage(#[from] AppError), -} - -impl From for ResourceActionError { - fn from(error: rusqlite::Error) -> Self { - Self::Storage(error.into()) - } -} - -impl From for ResourceActionError { - fn from(error: ResourceStoreError) -> Self { - match error { - ResourceStoreError::Storage(error) => Self::Storage(error.into()), - error => Self::Storage(AppError::Internal { - message: error.to_string(), - }), - } - } -} - -impl From for ResourceActionError { - fn from(error: serde_json::Error) -> Self { - Self::Storage(error.into()) - } -} - -impl From for ResourceActionError { - fn from(error: IdentityError) -> Self { - match error { - IdentityError::Storage(error) => Self::Storage(error), - IdentityError::Conflict | IdentityError::RouteNotFound => { - Self::Rejected(ResourceActionRejection::IdentityConflict) - } - } - } -} - -fn rejected(reason: ResourceActionRejection) -> Result { - Err(ResourceActionError::Rejected(reason)) -} - -/// Saved receipt and acceptance result for one remote action-bound launch -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(crate) struct AcceptedActionTask { - /// Exact action binding saved with the task - pub(crate) receipt: ActionTaskReceipt, - /// Whether this call inserted the task - pub(crate) acceptance: ActionTaskAcceptance, -} - -/// Canonical watcher task and fixed identities for one remote-supervisor acceptance -#[derive(Debug, Clone)] -pub(crate) struct RemoteReleaseWatcherAcceptanceInput { - /// Exact action authority named by the supervisor machine - pub(crate) authority: SupervisorActionAuthority, - /// Background task named by the release action - pub(crate) observed_background_task: TaskId, - /// Fixed identities and digest from the prepared watcher - pub(crate) task: ActionTaskIdentity, - /// Queued row with the authority's environment and watcher executable - pub(crate) row: TaskRow, - /// Canonical watcher spec built by the authority - pub(crate) spec: NormalizedSpec, -} - -impl crate::store::Store { - /// Accept one remote-supervisor release watcher once for its exact saved intent - pub(crate) fn accept_remote_release_watcher_for_authority( - &mut self, - input: RemoteReleaseWatcherAcceptanceInput, - ) -> Result { - accept_remote_release_watcher(&mut self.conn, input) - } - - /// Read the task identities of release watchers bound by loans on this authority - pub(crate) fn release_watcher_task_ids_for_authority( - &self, - authority_machine: MachineId, - ) -> Result, AppError> { - let mut statement = self.conn.prepare( - "SELECT COALESCE( - json_extract(l.state_json, '$.phase.watcher_intent.watcher_task_id'), - json_extract(l.state_json, '$.last_safe_phase.watcher_intent.watcher_task_id') - ) - FROM loans l JOIN resources r ON r.id = l.resource_id - WHERE r.authority_machine = ?1 - AND json_extract(l.state_json, '$.type') != 'closed'", - )?; - let ids = statement - .query_map([authority_machine.as_uuid().to_string()], |row| { - row.get::<_, Option>(0) - })? - .collect::, _>>()?; - ids.into_iter() - .flatten() - .map(|id| { - id.parse().map_err(|_| AppError::Internal { - message: format!("loan has an invalid watcher task identity {id}"), - }) - }) - .collect() - } -} - -fn accept_remote_release_watcher( - conn: &mut Connection, - input: RemoteReleaseWatcherAcceptanceInput, -) -> Result { - let RemoteReleaseWatcherAcceptanceInput { - authority, - observed_background_task, - task, - row, - spec, - } = input; - let receipt = ActionTaskReceipt { - kind: ResourceActionKind::ReleaseWatcher, - authority, - request_id: task.request_id, - task_id: task.task_id, - normalized_spec_sha256: task.normalized_spec_sha256, - }; - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - // a saved receipt answers a retry even after the release action moved on - if let Some(acceptance) = replay_action_task(&tx, &receipt)? { - tx.commit()?; - return Ok(AcceptedActionTask { - receipt, - acceptance, - }); - } - - let resource_id = authority.resource_id; - let authority_machine = authority.authority_machine; - let resource = current_remote_supervisor(&tx, &authority)?; - let loan = select_non_closed_loan(&tx, resource_id)? - .filter(|loan| loan.id == authority.loan_id) - .ok_or(ResourceActionError::Rejected( - ResourceActionRejection::ActionNotPending, - ))?; - let LoanState::Active { - phase: - LoanPhase::AwaitingRelease { - action_id, - observed_background_task: saved_task, - watcher_intent, - }, - } = &loan.state - else { - return rejected(ResourceActionRejection::ActionNotPending); - }; - if *action_id != authority.action_id || *saved_task != observed_background_task { - return rejected(ResourceActionRejection::ActionNotPending); - } - let Some(intent) = watcher_intent else { - return rejected(ResourceActionRejection::WatcherUnavailable { - reason: "the release action has no complete watcher identity".into(), - }); - }; - if intent.request_id != task.request_id || intent.watcher_task_id.as_task_id() != task.task_id { - return rejected(ResourceActionRejection::IdentityConflict); - } - if intent.state_revision != authority.expected_state_revision { - return rejected(ResourceActionRejection::StaleRevision { - expected: authority.expected_state_revision, - actual: intent.state_revision, - }); - } - if validate_release_watcher_intent_on(&tx, authority_machine, resource_id, intent).is_err() { - return rejected(ResourceActionRejection::ActionNotPending); - } - // the authority accepts only its own command for these exact release identities - let canonical = ReleaseWatcherCommand::from_intent(resource_id, intent) - .normalized_spec(&row.binary, resource.supervisor.thread)?; - let canonical_digest = normalized_spec_sha256(&canonical)?; - if canonical_digest != intent.normalized_spec_sha256 - || task.normalized_spec_sha256 != canonical_digest - || normalized_spec_sha256(&spec)? != canonical_digest - || !resource_task_row_matches(&row, task.task_id, &canonical) - { - return rejected(ResourceActionRejection::SpecMismatch); - } - if let Err(error) = - validate_release_checkpoint_baseline_on(&tx, authority_machine, resource_id, *action_id) - { - return match error { - ReleaseCheckpointError::Storage(error) => Err(error.into()), - error => rejected(ResourceActionRejection::WatcherUnavailable { - reason: format!("checkpoint baseline is unavailable: {error}"), - }), - }; - } - if action_task_receipt_by_action(&tx, authority.action_id)?.is_some() { - return rejected(ResourceActionRejection::ConflictingRetry); - } - - insert_action_task(&tx, &row, &canonical, &receipt)?; - tx.commit()?; - Ok(AcceptedActionTask { - receipt, - acceptance: ActionTaskAcceptance::Inserted, - }) -} - -/// Check that the request names the current remote supervisor assignment and revision -pub(super) fn current_remote_supervisor( - conn: &Connection, - authority: &SupervisorActionAuthority, -) -> Result { - let resource = select_resource(conn, authority.resource_id)?.ok_or( - ResourceActionError::Rejected(ResourceActionRejection::ActionNotPending), - )?; - if resource.authority_machine() != authority.authority_machine { - return rejected(ResourceActionRejection::ActionNotPending); - } - if resource.supervisor != authority.supervisor - || resource.assignment_revision != authority.assignment_revision - || resource.supervisor.machine == resource.authority_machine() - { - return rejected(ResourceActionRejection::NotCurrentSupervisor); - } - if resource.state_revision != authority.expected_state_revision { - return rejected(ResourceActionRejection::StaleRevision { - expected: authority.expected_state_revision, - actual: resource.state_revision, - }); - } - Ok(resource) -} - -/// Return the saved acceptance for an exact retry, or refuse a different one -pub(super) fn replay_action_task( - conn: &Connection, - receipt: &ActionTaskReceipt, -) -> Result, ResourceActionError> { - let by_task = action_task_receipt_by_task(conn, receipt.task_id)?; - let by_action = action_task_receipt_by_action(conn, receipt.authority.action_id)?; - let saved = match (by_task, by_action) { - (None, None) => return Ok(None), - (Some(saved), _) | (None, Some(saved)) => saved, - }; - if saved != *receipt { - return rejected(ResourceActionRejection::ConflictingRetry); - } - let row = remote_action_task_row_on(conn, &saved)?.ok_or(ResourceActionError::Rejected( - ResourceActionRejection::IdentityConflict, - ))?; - Ok(Some(ActionTaskAcceptance::Existing { - state: row.status(), - })) -} - -/// Insert the task records and the exact receipt for one first acceptance -pub(super) fn insert_action_task( - conn: &Connection, - row: &TaskRow, - spec: &NormalizedSpec, - receipt: &ActionTaskReceipt, -) -> Result<(), ResourceActionError> { - crate::store::insert_remote_action_task_records_on(conn, row, spec, receipt).map_err( - |error| match error { - AppError::ClusterTaskConflict { .. } => { - ResourceActionError::Rejected(ResourceActionRejection::IdentityConflict) - } - error => ResourceActionError::Storage(error), - }, - )?; - conn.execute( - "INSERT INTO resource_action_task_receipts - (task_id, request_id, action_id, resource_id, receipt_json) - VALUES (?1, ?2, ?3, ?4, ?5)", - params![ - receipt.task_id.to_string(), - receipt.request_id.0.to_string(), - receipt.authority.action_id.as_uuid().to_string(), - receipt.authority.resource_id.as_uuid().to_string(), - serde_json::to_string(receipt)?, - ], - )?; - Ok(()) -} - -/// Read the exact receipt that accepted one remote action-bound task -pub(crate) fn action_task_receipt_by_task( - conn: &Connection, - task_id: TaskId, -) -> Result, AppError> { - let json: Option = conn - .query_row( - "SELECT receipt_json FROM resource_action_task_receipts WHERE task_id = ?1", - [task_id.to_string()], - |row| row.get(0), - ) - .optional()?; - Ok(json.map(|json| serde_json::from_str(&json)).transpose()?) -} - -fn action_task_receipt_by_action( - conn: &Connection, - action_id: ActionId, -) -> Result, AppError> { - let json: Option = conn - .query_row( - "SELECT receipt_json FROM resource_action_task_receipts WHERE action_id = ?1", - [action_id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional()?; - Ok(json.map(|json| serde_json::from_str(&json)).transpose()?) -} - -/// Return the task row when its identity and first event match one saved receipt -/// -/// The authority keeps no origin route for these tasks, so the accepted identity -/// and first queued event must name the supervisor machine as callback origin -pub(crate) fn remote_action_task_row_on( - conn: &Connection, - receipt: &ActionTaskReceipt, -) -> Result, AppError> { - let task_id = receipt.task_id; - let Some(row) = crate::store::task_by_id_on(conn, task_id)? else { - return Ok(None); - }; - let identity = - crate::store::executor_identity_for_resource_task_on(conn, task_id).map_err(|error| { - AppError::Internal { - message: format!("action task identity: {error}"), - } - })?; - let Some(ExecutorIdentity::Accepted(record)) = identity else { - return Ok(None); - }; - let Some(spec) = record.current_spec() else { - return Ok(None); - }; - let first_event = crate::store::initial_queued_event_matches_on( - conn, - task_id, - receipt.origin_machine(), - receipt.execution_machine(), - ) - .map_err(|error| AppError::Internal { - message: format!("action task first event: {error}"), - })?; - let matches = record.is_owned_by( - task_id, - receipt.origin_machine(), - receipt.execution_machine(), - ) && record.state == row.status() - && spec.thread == receipt.authority.supervisor.thread - && normalized_spec_sha256(spec)? == receipt.normalized_spec_sha256 - && resource_task_row_matches(&row, task_id, spec) - && first_event; - - Ok(matches.then_some(row)) -} - -/// Decide whether the remote-supervisor watcher accepted under one receipt is running -/// -/// The receipt is the immutable acceptance route. It must name this intent, the -/// authority, and a supervisor on another machine. A later supervisor replacement -/// does not change the owner of a watcher that was already accepted -pub(super) fn remote_release_watcher_is_running_on( - conn: &Connection, - authority_machine: MachineId, - resource_id: crate::resource::ResourceId, - intent: &ReleaseWatcherIntent, - receipt: &ActionTaskReceipt, -) -> Result { - let task_id = intent.watcher_task_id.as_task_id(); - let conflict = || ReleaseCheckpointError::WatcherIdentityConflict { task_id }; - if receipt.kind != ResourceActionKind::ReleaseWatcher - || receipt.task_id != task_id - || receipt.request_id != intent.request_id - || receipt.normalized_spec_sha256 != intent.normalized_spec_sha256 - || receipt.authority.action_id != intent.action_id - || receipt.authority.expected_state_revision != intent.state_revision - || receipt.authority.resource_id != resource_id - || receipt.authority.authority_machine != authority_machine - || receipt.authority.supervisor.machine == authority_machine - { - return Err(conflict()); - } - let row = remote_action_task_row_on(conn, receipt)?.ok_or_else(conflict)?; - Ok(row.status() == ProcessStatus::Running && row.pid().is_some()) -} diff --git a/src/store/resource/assigned_task.rs b/src/store/resource/assigned_task.rs deleted file mode 100644 index 033f524..0000000 --- a/src/store/resource/assigned_task.rs +++ /dev/null @@ -1,400 +0,0 @@ -//! Acceptance of authority-assigned resource requests as executor tasks - -use std::path::PathBuf; - -use rusqlite::{Connection, TransactionBehavior}; - -use super::{encode_resource_json, resource_task_row_matches, task_has_any_event}; -use crate::domain::{ProcessStatus, TaskId, TaskRow}; -use crate::error::AppError; -use crate::events::EventPayload; -use crate::machine::MachineId; -use crate::resource::store::{ - AcceptedResourceTask, AssignedResourceTaskReconcileInput, AssignedResourceTaskReconcileOutcome, - ConflictReason, PreLaunchFailure, ResourceStoreError, ResourceTaskAcceptance, - ResourceTaskAcceptanceInput, accept_request_for_authority, - assigned_resource_request_for_acceptance, assigned_resource_requests_for_authority, - reconcile_assigned_resource_task_for_authority as persist_assigned_task_reconciliation, -}; -use crate::resource::{ResourceId, ResourceRequest, ResourceRequestState}; -use crate::spec::{NormalizedSpec, NormalizedWorkload}; -use crate::store::identity::{ - executor_identity_on, identity_origin_column_is_on, origin_route_by_request_on, - origin_route_by_task_on, -}; -use crate::store::{NewTask, Store}; -use crate::submission::{ - ExecutionRecord, ExecutorIdentity, RequestId, ResourceRoutePhase, SubmissionState, -}; - -fn conflict(reason: ConflictReason) -> ResourceStoreError { - ResourceStoreError::Conflict(reason) -} - -impl Store { - /// Read accepted resource task identities from their authority-owned assignments - pub(crate) fn accepted_resource_tasks_for_authority( - &self, - authority_machine: MachineId, - ) -> Result, ResourceStoreError> { - let requests = assigned_resource_requests_for_authority(&self.conn, authority_machine)?; - let mut accepted = Vec::new(); - - for request in requests { - let ResourceRequestState::Assigned { loan_id } = request.state else { - continue; - }; - let Some(executor) = executor_identity_on(&self.conn, request.task_id)? else { - let task_row = crate::store::task_by_id_on(&self.conn, request.task_id) - .map_err(ResourceStoreError::TaskRow)?; - if task_row.is_some() || task_has_any_event(&self.conn, request.task_id)? { - return Err(conflict(ConflictReason::TaskIdentityInUse)); - } - continue; - }; - let ExecutorIdentity::Accepted(record) = executor else { - return Err(conflict(ConflictReason::ExecutorIdentityMismatch)); - }; - if !loan_is_reserved(&self.conn, loan_id, request.resource_id)? { - return Err(conflict(ConflictReason::LoanChanged)); - } - let task_row = crate::store::task_by_id_on(&self.conn, request.task_id) - .map_err(ResourceStoreError::TaskRow)?; - if !accepted_task_matches_request( - &self.conn, - &record, - &request, - authority_machine, - task_row.as_ref(), - )? { - return Err(conflict(ConflictReason::ExecutorIdentityMismatch)); - } - - accepted.push(AcceptedResourceTask { - request, - state: record.state, - }); - } - - Ok(accepted) - } - - /// Accept a resource request and assign its acceptance identity and serving rank - /// - /// The origin route must be persisted by the caller before it sends this request - pub(crate) fn accept_resource_request( - &mut self, - authority_machine: MachineId, - request_id: RequestId, - task_id: TaskId, - resource_id: ResourceId, - origin_machine: MachineId, - normalized_spec: NormalizedSpec, - ) -> Result { - accept_request_for_authority( - &mut self.conn, - authority_machine, - request_id, - task_id, - resource_id, - origin_machine, - normalized_spec, - ) - } - - /// Accept one selected resource request as a task on this store connection - /// - /// A command that cannot be prepared here is still accepted, as - /// [`ResourceTaskAcceptance::Unlaunchable`], so the caller can fail it before - /// launch and release the queue instead of leaving no record behind - pub(crate) fn accept_assigned_resource_task( - &mut self, - input: ResourceTaskAcceptanceInput, - ) -> Result { - self.accept_assigned_resource_task_with(input, None) - } - - /// Accept one selected resource request that must fail before launch - /// - /// An existing acceptance is returned unchanged, so the caller decides what - /// to do with a task that already has records - pub(crate) fn accept_unlaunchable_resource_task( - &mut self, - input: ResourceTaskAcceptanceInput, - failure: PreLaunchFailure, - ) -> Result { - self.accept_assigned_resource_task_with(input, Some(failure)) - } - - fn accept_assigned_resource_task_with( - &mut self, - input: ResourceTaskAcceptanceInput, - forced_failure: Option, - ) -> Result { - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate)?; - let request = assigned_resource_request_for_acceptance(&tx, &input)?; - let normalized_spec = request.spec().as_normalized(); - let executor = executor_identity_on(&tx, input.task_id)?; - let task_row = crate::store::task_by_id_on(&tx, input.task_id)?; - - validate_resource_task_route( - &tx, - &request, - input.authority_machine, - normalized_spec, - executor.is_some(), - )?; - - if let Some(executor) = executor { - let ExecutorIdentity::Accepted(record) = executor else { - return Err(conflict(ConflictReason::ExecutorIdentityMismatch)); - }; - let same_env = task_row - .as_ref() - .is_some_and(|row| row.env == input.executor_env); - if !same_env - || !accepted_task_matches_request( - &tx, - &record, - &request, - input.authority_machine, - task_row.as_ref(), - )? - { - return Err(conflict(ConflictReason::ExecutorIdentityMismatch)); - } - - tx.commit()?; - return Ok(ResourceTaskAcceptance::Existing { - task: input.task_id, - state: record.state, - }); - } - - let first_event_saved = crate::store::events::initial_queued_event_matches_on( - &tx, - input.task_id, - request.origin_machine, - input.authority_machine, - )?; - if task_row.is_some() - || task_has_any_event(&tx, input.task_id)? - || first_event_saved - || crate::store::release_watcher_task_id_is_reserved(&tx, input.task_id)? - { - return Err(conflict(ConflictReason::TaskIdentityInUse)); - } - - let (row, preparation_failure) = new_resource_task_row(&input, normalized_spec)?; - let failure = forced_failure.or(preparation_failure); - let project_root = crate::store::find_project_root(&row.cwd); - crate::store::insert_task_with_project_root_on(&tx, &row, project_root.as_deref()) - .map_err(ResourceStoreError::TaskPreparation)?; - - let identity = ExecutorIdentity::Accepted(ExecutionRecord { - task: input.task_id, - origin_machine: request.origin_machine, - execution_machine: input.authority_machine, - spec: normalized_spec.clone().into(), - state: ProcessStatus::Queued, - }); - tx.execute( - "INSERT INTO executor_identities (task_id,origin_machine,identity_json) - VALUES (?1,?2,?3)", - rusqlite::params![ - input.task_id.to_string(), - request.origin_machine.to_string(), - encode_resource_json(&identity)?, - ], - )?; - crate::store::events::append_produced_event_on( - &tx, - input.task_id, - EventPayload::State { - status: ProcessStatus::Queued, - }, - )?; - - tx.commit()?; - let task = input.task_id; - Ok(match failure { - None => ResourceTaskAcceptance::Inserted { task }, - Some(failure) => ResourceTaskAcceptance::Unlaunchable { task, failure }, - }) - } - - /// Reconcile one exact Serving task and commit completion only with release evidence - pub(crate) fn reconcile_assigned_resource_task_for_authority( - &mut self, - input: AssignedResourceTaskReconcileInput, - ) -> Result { - persist_assigned_task_reconciliation(&mut self.conn, input) - } -} - -/// Build the queued task row for a first acceptance on this executor -/// -/// A bare program name resolves from the executor PATH only here, so the resolved -/// entry point must pass the foreground contract before the task may launch. A -/// container runs through the resolved `docker` CLI, which Homebased drives -/// itself, so its mount sources are checked instead. A command that fails these -/// checks still gets its row with the failure, so it ends before launch with a -/// reason; a binary that never resolved is saved as an empty path -fn new_resource_task_row( - input: &ResourceTaskAcceptanceInput, - normalized_spec: &NormalizedSpec, -) -> Result<(TaskRow, Option), ResourceStoreError> { - let binary = crate::invocation::resolve_workload_binary( - &normalized_spec.workload, - &input.executor_env.path, - &normalized_spec.cwd, - ); - let failure = launch_preparation_failure(normalized_spec, binary.as_ref()); - let row = crate::store::new_queued_task(NewTask { - id: input.task_id, - name: Some(normalized_spec.name.clone()), - thread: normalized_spec.thread, - workload: crate::invocation::persist_workload(&normalized_spec.workload), - cwd: normalized_spec.cwd.clone(), - timeout: normalized_spec.timeout, - env: input.executor_env.clone(), - binary: binary.unwrap_or_default(), - }); - if !resource_task_row_matches(&row, input.task_id, normalized_spec) { - return Err(conflict(ConflictReason::TaskRowMismatch)); - } - - Ok((row, failure)) -} - -/// First reason this executor cannot launch the saved command, if any -fn launch_preparation_failure( - spec: &NormalizedSpec, - binary: Result<&PathBuf, &AppError>, -) -> Option { - let message = |message: String| Some(PreLaunchFailure::Preparation { message }); - if let Err(error) = crate::spec::check_spec_host(spec) { - return message(error.error.to_string()); - } - let binary = match binary { - Ok(binary) => binary, - Err(error) => return message(error.to_string()), - }; - if matches!(spec.workload, NormalizedWorkload::Container(_)) { - return None; - } - crate::resource::foreground::inspect_foreground_entry_point(binary) - .err() - .and_then(|risk| { - message(ResourceStoreError::UnsupportedCommandOwnership { risk }.to_string()) - }) -} - -/// Check an accepted identity, its origin column, first event, and task row against one request -fn accepted_task_matches_request( - conn: &Connection, - record: &ExecutionRecord, - request: &ResourceRequest, - authority_machine: MachineId, - task_row: Option<&TaskRow>, -) -> Result { - let task_id = request.task_id; - let spec = request.spec().as_normalized(); - Ok( - record.is_owned_by(task_id, request.origin_machine, authority_machine) - && record.current_spec() == Some(spec) - && identity_origin_column_is_on(conn, task_id, request.origin_machine)? - && task_has_any_event(conn, task_id)? - && crate::store::events::initial_queued_event_matches_on( - conn, - task_id, - request.origin_machine, - authority_machine, - )? - && task_row.is_some_and(|row| { - resource_task_row_matches(row, task_id, spec) && row.status() == record.state - }), - ) -} - -fn loan_is_reserved( - conn: &Connection, - loan_id: crate::resource::LoanId, - resource_id: ResourceId, -) -> Result { - conn.query_row( - "SELECT EXISTS( - SELECT 1 FROM loans - WHERE id=?1 AND resource_id=?2 - AND json_extract(state_json, '$.type') != 'closed' - )", - rusqlite::params![ - loan_id.as_uuid().to_string(), - resource_id.as_uuid().to_string() - ], - |row| row.get(0), - ) -} - -/// Require the origin route that a local origin saved before sending its request -/// -/// A remote origin keeps its route on its own machine, so none may exist here -fn validate_resource_task_route( - conn: &Connection, - request: &ResourceRequest, - authority: MachineId, - spec: &NormalizedSpec, - already_accepted: bool, -) -> Result<(), ResourceStoreError> { - let route_conflict = || conflict(ConflictReason::OriginRouteMismatch); - let by_request = origin_route_by_request_on(conn, request.request_id)?; - let by_task = origin_route_by_task_on(conn, request.task_id)?; - - if request.origin_machine != authority { - return if by_request.is_none() && by_task.is_none() { - Ok(()) - } else { - Err(route_conflict()) - }; - } - - let (by_request, by_task) = match (by_request, by_task) { - (Some(by_request), Some(by_task)) => (by_request, by_task), - (None, None) => { - return Err(ResourceStoreError::OriginRouteNotFound { - task: request.task_id, - }); - } - _ => return Err(route_conflict()), - }; - if by_request != by_task { - return Err(route_conflict()); - } - - let SubmissionState::Resource { resource, phase } = &by_request.submission else { - return Err(route_conflict()); - }; - if by_request.request != request.request_id - || by_request.task != request.task_id - || by_request.origin_machine != request.origin_machine - || by_request.execution_machine != authority - || by_request.thread != spec.thread - || *resource != request.resource_id - || by_request.current_spec() != Some(spec) - { - return Err(route_conflict()); - } - - match phase { - ResourceRoutePhase::AcceptanceUnknown | ResourceRoutePhase::Waiting => Ok(()), - ResourceRoutePhase::Activated if already_accepted => Ok(()), - ResourceRoutePhase::CancelledBeforeLaunch if !already_accepted => { - Err(ResourceStoreError::Prevented) - } - ResourceRoutePhase::Activated - | ResourceRoutePhase::CancelledBeforeLaunch - | ResourceRoutePhase::Rejected { .. } => Err(route_conflict()), - } -} diff --git a/src/store/resource/background.rs b/src/store/resource/background.rs deleted file mode 100644 index b4c77f8..0000000 --- a/src/store/resource/background.rs +++ /dev/null @@ -1,1504 +0,0 @@ -//! Resource-owned first background launch and the idle boundary that follows it -//! -//! A first background launch binds its stable request, preallocated task, full -//! normalized spec, callback route, accepted executor identity, queued row, and -//! first event in one IMMEDIATE transaction with an immutable launch receipt -//! The receipt does not register the task. The queue owner registers it only -//! after the task layer records a confirmed start, so a queued row is never -//! mistaken for live background work -//! -//! The latest launch receipt and the latest loan form the resource history that -//! the idle boundary reads. An unregistered resource serves queued work only -//! when that history names an explicit reason why no background work holds the -//! GPU. A missing task row is never such a reason -//! -//! A supervisor on another machine owns the callback route of its launch. It -//! saves that route before it sends the launch, so the authority keeps only the -//! task row, the accepted remote identity, the first event, and a receipt that -//! names the exact supervisor assignment and resource revision - -use rusqlite::{Connection, OptionalExtension, Transaction, TransactionBehavior, params}; -use serde::{Deserialize, Serialize}; - -use crate::domain::{ - ExitReason, ProcessStatus, TaskEnv, TaskId, TaskRow, TaskState, ThreadId, WorkExitEvidence, -}; -use crate::error::AppError; -use crate::machine::MachineId; -use crate::resource::background_launch::{BackgroundLaunchBinding, RemoteBackgroundLaunchReceipt}; -use crate::resource::command_shape::{DirectSegmentCommandShape, DirectSegmentCommandShapeError}; -use crate::resource::store::{ - ConflictReason, ResourceStoreError, next_queued_request_for_authority, select_non_closed_loan, -}; -use crate::resource::{ - BackgroundCommandContract, IdleBoundaryDecision, IdleBoundaryProof, IdleProofGap, Loan, - LoanClosure, LoanId, LoanPhase, LoanState, Resource, ResourceId, ResourceRequest, - ResourceRequestState, ResourceRevision, ReturnContext, ServingReleaseProvenance, - SupervisorAddress, -}; -use crate::spec::NormalizedSpec; -use crate::store::IdentityError; -use crate::store::identity::{executor_identity_on, origin_route_by_request_on}; -use crate::submission::{ - CallbackContext, CallbackExecutable, ExecutorIdentity, NormalizedSpecSha256, RequestId, - normalized_spec_sha256, -}; - -use super::trainer_lock::{ - EndedTaskWitness, HeldTrainerRelease, TrainerLockReleaseError, TrainerLockReleaseGap, - ended_task_witness, hold_released_trainer_lock, -}; -use super::{encode_resource_json, select_authority_resource}; - -/// Fixed identities and executor context for one first background launch -#[derive(Debug, Clone)] -pub(crate) struct BackgroundLaunchInput { - /// Machine that owns the resource and executes the task - pub(crate) authority_machine: MachineId, - /// Resource whose background slot receives the task - pub(crate) resource_id: ResourceId, - /// Stable caller retry identity - pub(crate) request_id: RequestId, - /// Task identity used only when this call inserts the launch - pub(crate) task_id: TaskId, - /// Full normalized command spec - pub(crate) spec: NormalizedSpec, - /// Executor environment captured by the co-located supervisor - pub(crate) env: TaskEnv, - /// Codex executable that delivers callbacks to the supervisor thread - pub(crate) callback_codex: CallbackExecutable, -} - -/// Remote supervisor launch that the supervisor machine saved before sending it -#[derive(Debug, Clone)] -pub(crate) struct RemoteBackgroundLaunchInput { - /// Fixed identities, digest, supervisor assignment, and observed revision - pub(crate) receipt: RemoteBackgroundLaunchReceipt, - /// Full normalized command spec saved in the supervisor's route - pub(crate) spec: NormalizedSpec, - /// Executor environment of the authority daemon - pub(crate) env: TaskEnv, -} - -/// Result of binding one first background launch -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) enum BackgroundLaunchAcceptance { - /// The task records and launch receipt committed in this call, so it alone may spawn - Inserted { - /// Bound task identity - task: TaskId, - /// Resource revision committed with the launch receipt - state_revision: ResourceRevision, - }, - /// An exact earlier launch exists; its task is only observed - Existing { - /// Bound task identity - task: TaskId, - /// State retained by the task layer - state: ProcessStatus, - }, - /// The supervisor thread runs on another machine, so nothing was written - UnsupportedRemoteSupervisor { - /// Machine that owns the resource - authority_machine: MachineId, - /// Supervisor that would need a remote callback route - supervisor: SupervisorAddress, - }, -} - -/// Why a first background launch cannot bind -#[derive(Debug, thiserror::Error)] -pub(crate) enum BackgroundLaunchError { - /// Resource authority or stored resource data failed validation - #[error(transparent)] - Resource(#[from] ResourceStoreError), - /// The remote launch does not come from the current supervisor assignment - #[error("the background launch does not come from the current supervisor assignment")] - NotCurrentSupervisor, - /// The resource changed after the supervisor machine read it - #[error("resource revision changed from {expected:?} to {actual:?}")] - StaleRevision { - /// Revision that the supervisor machine observed - expected: ResourceRevision, - /// Revision saved by the authority - actual: ResourceRevision, - }, - /// The launch callback thread is not the assigned supervisor thread - #[error("background launch thread {found} is not the supervisor thread {expected}")] - NotSupervisorThread { - /// Assigned supervisor thread - expected: ThreadId, - /// Thread named by the spec - found: ThreadId, - }, - /// A non-closed loan owns the resource; background work returns through its action - #[error("loan {loan_id:?} owns the resource")] - ActiveLoan { - /// Non-closed loan - loan_id: LoanId, - }, - /// Queued resource work blocks a new background launch - #[error("queued resource request {request_id:?} blocks the background launch")] - QueuedWorkAhead { - /// Queued request that blocks the launch - request_id: RequestId, - }, - /// The registered background task has not ended - #[error("registered background task {task_id} is {state}")] - BackgroundTaskActive { - /// Registered task - task_id: TaskId, - /// Task-layer state - state: ProcessStatus, - }, - /// The registered background task has no task row on the authority - #[error("registered background task {task_id} has no task record")] - BackgroundTaskMissing { - /// Registered task - task_id: TaskId, - }, - /// An earlier first background launch has not reached a confirmed start or an end - #[error("background launch task {task_id} is still pending")] - LaunchPending { - /// Pending launch task - task_id: TaskId, - }, - /// An earlier background task ended without confirmed exit or no-child evidence - /// - /// It is lost, its process-group exit is unconfirmed, or its records do not - /// match, so it may still hold the GPU and needs an owner decision - #[error("background task {task_id} ended without proof that it released the resource")] - PredecessorReleaseUnproven { - /// Earlier background task that may still hold the resource - task_id: TaskId, - }, - /// An earlier background task's wrapper exited, but its trainer lock is not proven free - /// - /// The direct-segment worker runs in its own session, so only the exact lock - /// named by the trainer-attempt association, held through this transaction, - /// proves that the worker released the GPU - #[error("background task {task_id} has no verified trainer lock release: {gap}")] - PredecessorOwnershipUnproven { - /// Earlier background task that may still hold the resource - task_id: TaskId, - /// Missing or failed part of the lock proof - gap: TrainerLockReleaseGap, - }, - /// The request identity belongs to a different launch or task - #[error("request {request_id:?} was retried with different content")] - ConflictingRetry { - /// Reused request identity - request_id: RequestId, - }, - /// The saved launch records do not match the receipt - #[error("background launch task {task_id} records do not match its receipt")] - IdentityConflict { - /// Bound task identity - task_id: TaskId, - }, - /// The command has no ownership contract that release proof can verify - #[error("background command ownership cannot be verified: {0}")] - UnsupportedCommand(#[from] DirectSegmentCommandShapeError), - /// The resource revision cannot be incremented - #[error("resource revision {revision:?} cannot be incremented")] - RevisionExhausted { - /// Current revision - revision: ResourceRevision, - }, - /// Route or executor identity data is invalid - #[error(transparent)] - Identity(#[from] IdentityError), - /// Task preparation or task record insertion failed - #[error("background launch task records failed: {0}")] - TaskRecords(#[from] AppError), - /// A receipt could not be encoded or decoded - #[error("background launch encoding failed: {0}")] - Encoding(#[from] serde_json::Error), - /// SQLite failed - #[error("background launch storage error: {0}")] - Storage(#[from] rusqlite::Error), -} - -/// Callback owner of one first background launch -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -enum BackgroundLaunchOrigin { - /// The supervisor thread runs on the authority, which saved the callback route - CoLocated, - /// The supervisor machine saved the callback route before it sent the launch - RemoteSupervisor { - /// Resource revision that the supervisor machine observed - expected_state_revision: ResourceRevision, - }, -} - -/// Immutable binding of one first background launch -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct BackgroundLaunchReceipt { - resource_id: ResourceId, - authority_machine: MachineId, - request_id: RequestId, - task_id: TaskId, - normalized_spec_sha256: NormalizedSpecSha256, - supervisor: SupervisorAddress, - assignment_revision: crate::resource::AssignmentRevision, - accepted_state_revision: ResourceRevision, - // the ended registration that this launch replaces once it starts - replaces_task: Option, - // the latest loan when the launch was accepted; a later loan supersedes it - preceding_loan: Option, - contract: BackgroundCommandContract, - origin: BackgroundLaunchOrigin, -} - -impl BackgroundLaunchReceipt { - /// Machine that owns the task's callback route - fn origin_machine(&self) -> MachineId { - match self.origin { - BackgroundLaunchOrigin::CoLocated => self.authority_machine, - BackgroundLaunchOrigin::RemoteSupervisor { .. } => self.supervisor.machine, - } - } - - /// Whether this receipt saved exactly one remote supervisor's launch - fn matches_remote(&self, expected: &RemoteBackgroundLaunchReceipt) -> bool { - let BackgroundLaunchBinding { - assignment, - expected_state_revision, - } = expected.binding; - self.origin - == BackgroundLaunchOrigin::RemoteSupervisor { - expected_state_revision, - } - && self.resource_id == assignment.resource_id - && self.authority_machine == assignment.authority_machine - && self.supervisor == assignment.supervisor - && self.assignment_revision == assignment.assignment_revision - && self.request_id == expected.request_id - && self.task_id == expected.task_id - && self.normalized_spec_sha256 == expected.normalized_spec_sha256 - } -} - -/// Loan-opening receipt that permits serving without a registered task -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct IdleOpeningReceipt { - loan_id: LoanId, - resource_id: ResourceId, - authority_machine: MachineId, - request_id: RequestId, - state_revision: ResourceRevision, - proof: IdleBoundaryProof, -} - -/// Task-layer phase of the latest first background launch -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(crate) enum BackgroundLaunchPhase { - /// A later loan exists, so this launch no longer describes the resource - Superseded, - /// The row is queued; its worker may be starting or may have lost its spawn - Queued, - /// The row is running and waits for the queue owner to register it - StartedUnregistered, - /// The task is the registered background task - Registered, - /// The task ended before registration - EndedBeforeRegistration { - /// Task-layer evidence about the process group that the task started - release: EndedLaunchRelease, - }, - /// The launch records do not match the receipt - IdentityMismatch, -} - -/// Task-layer evidence for a first background launch that ended before registration -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(crate) enum EndedLaunchRelease { - /// The task layer recorded that no child process started - NeverSpawned, - /// The worker confirmed that its owned process group exited - ConfirmedExited, - /// The task is lost, or its process-group exit is not confirmed - Unproven, -} - -/// Latest first background launch and its derived phase -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) struct BackgroundLaunchView { - /// Stable launch request identity - pub(crate) request_id: RequestId, - /// Bound task identity - pub(crate) task_id: TaskId, - /// Derived task-layer phase - pub(crate) phase: BackgroundLaunchPhase, -} - -impl BackgroundLaunchView { - /// Return the task that keeps the resource reserved before registration - pub(crate) fn pending_task(&self) -> Option { - matches!( - self.phase, - BackgroundLaunchPhase::Queued | BackgroundLaunchPhase::StartedUnregistered - ) - .then_some(self.task_id) - } - - /// Whether the launch ended before registration with no automatic release proof - /// - /// The maintained trainer starts its worker in another session, and an - /// unregistered task cannot bind the trainer attempt whose lock would prove - /// that worker's exit. Only a saved operator attestation releases it - pub(crate) fn awaits_operator_release(&self) -> bool { - matches!( - self.phase, - BackgroundLaunchPhase::EndedBeforeRegistration { - release: EndedLaunchRelease::ConfirmedExited | EndedLaunchRelease::Unproven, - } - ) - } -} - -/// Proof that no earlier background task can still hold the GPU -/// -/// Keep it alive until the launch transaction commits, because it holds each -/// released trainer lock that the proof relies on -#[must_use = "keep the released trainer locks held until the launch commits"] -struct BackgroundSlotClearance { - held_locks: Vec<(TaskId, HeldTrainerRelease)>, -} - -impl BackgroundSlotClearance { - /// Recheck every held lock against its saved path just before the commit - fn recheck(&self) -> Result<(), BackgroundLaunchError> { - for (task_id, held) in &self.held_locks { - held.recheck() - .map_err(|gap| BackgroundLaunchError::PredecessorOwnershipUnproven { - task_id: *task_id, - gap, - })?; - } - Ok(()) - } -} - -impl crate::store::Store { - /// Bind one first background launch; only an `Inserted` result may spawn - pub(crate) fn accept_background_launch_for_authority( - &mut self, - input: BackgroundLaunchInput, - ) -> Result { - accept_background_launch_for_authority(&mut self.conn, input) - } - - /// Bind one remote supervisor's first background launch; only `Inserted` may spawn - pub(crate) fn accept_remote_background_launch_for_authority( - &mut self, - input: RemoteBackgroundLaunchInput, - ) -> Result { - accept_remote_background_launch_for_authority(&mut self.conn, input) - } - - /// Read the latest first background launch of one authority-owned resource - pub(crate) fn background_launch_for_authority( - &self, - authority_machine: MachineId, - resource_id: ResourceId, - ) -> Result, ResourceStoreError> { - let resource = select_authority_resource(&self.conn, authority_machine, resource_id)?; - current_launch_on(&self.conn, &resource) - } - - /// Read the tasks of first background launches that are still queued - /// - /// Startup keeps these rows queued for the resource owner because a free - /// runner lock cannot show whether the first spawn happened - pub(crate) fn queued_background_launch_tasks_for_authority( - &self, - authority_machine: MachineId, - ) -> Result, ResourceStoreError> { - let mut statement = self.conn.prepare( - "SELECT l.task_id FROM resource_background_launches l - JOIN resources r ON r.id = l.resource_id - JOIN tasks t ON t.id = l.task_id - WHERE r.authority_machine = ?1 AND t.status = 'queued'", - )?; - let ids = statement - .query_map([authority_machine.as_uuid().to_string()], |row| { - row.get::<_, String>(0) - })? - .collect::, _>>()?; - ids.into_iter() - .map(|id| { - id.parse() - .map_err(|error| ResourceStoreError::corrupt("background launch task", error)) - }) - .collect() - } -} - -fn accept_background_launch_for_authority( - conn: &mut Connection, - input: BackgroundLaunchInput, -) -> Result { - let digest = normalized_spec_sha256(&input.spec)?; - { - // an exact retry is answered before any file-system check, so a path - // that changed after the launch cannot turn it into a rejection - let tx = conn.transaction_with_behavior(TransactionBehavior::Deferred)?; - if let Some(existing) = replay_on(&tx, &input, digest)? { - tx.commit()?; - return Ok(existing); - } - let resource = select_authority_resource(&tx, input.authority_machine, input.resource_id)?; - if let Some(unsupported) = remote_supervisor(&resource) { - return Ok(unsupported); - } - tx.commit()?; - } - - // canonical path checks stay outside the IMMEDIATE write transaction - let (row, contract) = prepare_launch_row(input.task_id, &input.spec, &input.env)?; - - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - if let Some(existing) = replay_on(&tx, &input, digest)? { - tx.commit()?; - return Ok(existing); - } - let resource = select_authority_resource(&tx, input.authority_machine, input.resource_id)?; - if let Some(unsupported) = remote_supervisor(&resource) { - return Ok(unsupported); - } - if input.spec.thread != resource.supervisor.thread { - return Err(BackgroundLaunchError::NotSupervisorThread { - expected: resource.supervisor.thread, - found: input.spec.thread, - }); - } - let clearance = require_launch_slot(&tx, input.authority_machine, &resource)?; - - let task_id = input.task_id; - if super::task_identity_is_used(&tx, input.request_id, task_id)? { - return Err(BackgroundLaunchError::IdentityConflict { task_id }); - } - let callback = CallbackContext { - env: input.env, - cwd: input.spec.cwd.clone(), - codex: input.callback_codex, - }; - crate::store::insert_local_task_records_on( - &tx, - &row, - &input.spec, - input.authority_machine, - input.request_id, - &callback, - None, - ) - .map_err(|error| match error { - AppError::ClusterTaskConflict { .. } => BackgroundLaunchError::IdentityConflict { task_id }, - error => BackgroundLaunchError::TaskRecords(error), - })?; - let state_revision = record_launch_on( - &tx, - &resource, - input.request_id, - task_id, - digest, - contract, - BackgroundLaunchOrigin::CoLocated, - )?; - clearance.recheck()?; - tx.commit()?; - // the new worker may start only after the commit, so the locks stay held until then - drop(clearance); - - Ok(BackgroundLaunchAcceptance::Inserted { - task: task_id, - state_revision, - }) -} - -/// Bind one remote supervisor's launch after its route evidence was checked -/// -/// The receipt names the supervisor assignment and the resource revision that -/// the supervisor machine observed. The authority rechecks both, the empty -/// background slot, and the task identity in one IMMEDIATE transaction, then -/// saves the task row, remote identity, first event, and receipt together -fn accept_remote_background_launch_for_authority( - conn: &mut Connection, - input: RemoteBackgroundLaunchInput, -) -> Result { - let RemoteBackgroundLaunchInput { - receipt: expected, - spec, - env, - } = input; - let request_id = expected.request_id; - if normalized_spec_sha256(&spec)? != expected.normalized_spec_sha256 { - return Err(BackgroundLaunchError::ConflictingRetry { request_id }); - } - let assignment = expected.binding.assignment; - { - // an exact retry is answered before the assignment, revision, or path - // checks, so a later transition cannot turn it into a rejection - let tx = conn.transaction_with_behavior(TransactionBehavior::Deferred)?; - if let Some(existing) = replay_remote_on(&tx, &expected)? { - tx.commit()?; - return Ok(existing); - } - let resource = - select_authority_resource(&tx, assignment.authority_machine, assignment.resource_id)?; - check_remote_assignment(&resource, &expected.binding, &spec)?; - tx.commit()?; - } - - let task_id = expected.task_id; - let (row, contract) = prepare_launch_row(task_id, &spec, &env)?; - - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - if let Some(existing) = replay_remote_on(&tx, &expected)? { - tx.commit()?; - return Ok(existing); - } - let resource = - select_authority_resource(&tx, assignment.authority_machine, assignment.resource_id)?; - check_remote_assignment(&resource, &expected.binding, &spec)?; - let clearance = require_launch_slot(&tx, assignment.authority_machine, &resource)?; - if super::task_identity_is_used(&tx, request_id, task_id)? { - return Err(BackgroundLaunchError::IdentityConflict { task_id }); - } - crate::store::insert_remote_origin_task_records_on( - &tx, - &row, - &spec, - &crate::store::RemoteOriginTask { - request_id, - origin_machine: expected.binding.origin_machine(), - execution_machine: expected.binding.execution_machine(), - thread: assignment.supervisor.thread, - reserved_watcher: false, - }, - ) - .map_err(|error| match error { - AppError::ClusterTaskConflict { .. } => BackgroundLaunchError::IdentityConflict { task_id }, - error => BackgroundLaunchError::TaskRecords(error), - })?; - let state_revision = record_launch_on( - &tx, - &resource, - request_id, - task_id, - expected.normalized_spec_sha256, - contract, - BackgroundLaunchOrigin::RemoteSupervisor { - expected_state_revision: expected.binding.expected_state_revision, - }, - )?; - clearance.recheck()?; - tx.commit()?; - drop(clearance); - - Ok(BackgroundLaunchAcceptance::Inserted { - task: task_id, - state_revision, - }) -} - -/// Check that a remote launch names the current supervisor assignment and revision -fn check_remote_assignment( - resource: &Resource, - binding: &BackgroundLaunchBinding, - spec: &NormalizedSpec, -) -> Result<(), BackgroundLaunchError> { - // a co-located supervisor keeps its callback route on the authority - if resource.supervisor.machine == resource.authority_machine() - || !binding.assignment.is_current(resource) - { - return Err(BackgroundLaunchError::NotCurrentSupervisor); - } - if resource.state_revision != binding.expected_state_revision { - return Err(BackgroundLaunchError::StaleRevision { - expected: binding.expected_state_revision, - actual: resource.state_revision, - }); - } - if spec.thread != resource.supervisor.thread { - return Err(BackgroundLaunchError::NotSupervisorThread { - expected: resource.supervisor.thread, - found: spec.thread, - }); - } - Ok(()) -} - -/// Build the queued row and verify the maintained trainer command before any write -fn prepare_launch_row( - task_id: TaskId, - spec: &NormalizedSpec, - env: &TaskEnv, -) -> Result<(TaskRow, BackgroundCommandContract), BackgroundLaunchError> { - crate::spec::check_spec_host(spec).map_err(AppError::from)?; - let binary = crate::invocation::resolve_workload_binary(&spec.workload, &env.path, &spec.cwd)?; - let row = crate::store::new_queued_task(crate::store::NewTask { - id: task_id, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: crate::invocation::persist_workload(&spec.workload), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env: env.clone(), - binary, - }); - let shape = DirectSegmentCommandShape::validate_launch(spec, &row)?; - let contract = BackgroundCommandContract::DirectSegmentTrainer { - runtime_root: shape.runtime_root().to_path_buf(), - }; - Ok((row, contract)) -} - -/// Refuse a launch while a loan, earlier queued work, or background work holds the resource -fn require_launch_slot( - conn: &Connection, - authority_machine: MachineId, - resource: &Resource, -) -> Result { - if let Some(loan) = select_non_closed_loan(conn, resource.id)? { - return Err(BackgroundLaunchError::ActiveLoan { loan_id: loan.id }); - } - if let Some(request) = next_queued_request_for_authority(conn, authority_machine, resource.id)? - { - return Err(BackgroundLaunchError::QueuedWorkAhead { - request_id: request.request_id, - }); - } - require_background_slot_free(conn, resource) -} - -/// Advance the resource revision and save the immutable launch receipt -fn record_launch_on( - tx: &Transaction<'_>, - resource: &Resource, - request_id: RequestId, - task_id: TaskId, - normalized_spec_sha256: NormalizedSpecSha256, - contract: BackgroundCommandContract, - origin: BackgroundLaunchOrigin, -) -> Result { - // the launch changes what the resource may do next, but it registers nothing - let state_revision = advance_resource_on(tx, resource, resource.registered_background_task) - .map_err(|error| match error { - AdvanceError::Exhausted => BackgroundLaunchError::RevisionExhausted { - revision: resource.state_revision, - }, - AdvanceError::Changed => BackgroundLaunchError::Resource(ResourceStoreError::Conflict( - ConflictReason::ResourceRevisionChanged, - )), - AdvanceError::Storage(error) => BackgroundLaunchError::Storage(error), - })?; - let receipt = BackgroundLaunchReceipt { - resource_id: resource.id, - authority_machine: resource.authority_machine(), - request_id, - task_id, - normalized_spec_sha256, - supervisor: resource.supervisor, - assignment_revision: resource.assignment_revision, - accepted_state_revision: state_revision, - replaces_task: resource.registered_background_task, - preceding_loan: latest_loan_on(tx, resource.id)?.map(|loan| loan.id), - contract, - origin, - }; - tx.execute( - "INSERT INTO resource_background_launches (request_id, task_id, resource_id, receipt_json) - VALUES (?1, ?2, ?3, ?4)", - params![ - request_id.0.to_string(), - task_id.to_string(), - resource.id.as_uuid().to_string(), - serde_json::to_string(&receipt)?, - ], - )?; - Ok(state_revision) -} - -// a local route for a remote supervisor thread would send callbacks to the wrong -// machine, so this path writes nothing for it -fn remote_supervisor(resource: &Resource) -> Option { - (resource.supervisor.machine != resource.authority_machine()).then_some( - BackgroundLaunchAcceptance::UnsupportedRemoteSupervisor { - authority_machine: resource.authority_machine(), - supervisor: resource.supervisor, - }, - ) -} - -/// Answer an exact retry from its receipt without writing or spawning -fn replay_on( - conn: &Connection, - input: &BackgroundLaunchInput, - digest: NormalizedSpecSha256, -) -> Result, BackgroundLaunchError> { - let conflict = BackgroundLaunchError::ConflictingRetry { - request_id: input.request_id, - }; - let Some(receipt) = receipt_by_request_on(conn, input.request_id)? else { - // an ordinary task or another resource path already owns this request identity - if origin_route_by_request_on(conn, input.request_id)?.is_some() - || super::resource_request_identity_exists(conn, input.request_id, input.task_id)? - { - return Err(conflict); - } - return Ok(None); - }; - if receipt.origin != BackgroundLaunchOrigin::CoLocated - || receipt.resource_id != input.resource_id - || receipt.authority_machine != input.authority_machine - || receipt.normalized_spec_sha256 != digest - { - return Err(conflict); - } - let task_id = receipt.task_id; - let row = bound_launch_row_on(conn, &receipt)? - .ok_or(BackgroundLaunchError::IdentityConflict { task_id })?; - - Ok(Some(BackgroundLaunchAcceptance::Existing { - task: task_id, - state: row.status(), - })) -} - -/// Answer an exact remote retry from its receipt without writing or spawning -fn replay_remote_on( - conn: &Connection, - expected: &RemoteBackgroundLaunchReceipt, -) -> Result, BackgroundLaunchError> { - let request_id = expected.request_id; - let conflict = BackgroundLaunchError::ConflictingRetry { request_id }; - let saved = match receipt_by_request_on(conn, request_id)? { - Some(saved) => saved, - // another launch owns the task identity, or another path owns the request - None if receipt_by_task_on(conn, expected.task_id)?.is_some() - || origin_route_by_request_on(conn, request_id)?.is_some() - || super::resource_request_identity_exists(conn, request_id, expected.task_id)? => - { - return Err(conflict); - } - None => return Ok(None), - }; - if !saved.matches_remote(expected) { - return Err(conflict); - } - let task_id = saved.task_id; - let row = bound_launch_row_on(conn, &saved)? - .ok_or(BackgroundLaunchError::IdentityConflict { task_id })?; - - Ok(Some(BackgroundLaunchAcceptance::Existing { - task: task_id, - state: row.status(), - })) -} - -/// Refuse a launch while any predecessor background task may still hold the resource -/// -/// The latest launch and the registered task are the only background tasks that -/// can hold the GPU without a non-closed loan; every other background task was -/// released through a loan transition that required its own proof. Each must be -/// terminal with no-child evidence, or with confirmed exit and the witness its -/// command contract needs, before a new launch commits. A direct-segment -/// trainer's witness is its exact released lock, which the returned clearance -/// holds -fn require_background_slot_free( - conn: &Connection, - resource: &Resource, -) -> Result { - let mut clearance = BackgroundSlotClearance { - held_locks: Vec::new(), - }; - if let Some((receipt, phase)) = current_launch_record_on(conn, resource)? { - let task_id = receipt.task_id; - match phase { - BackgroundLaunchPhase::Queued | BackgroundLaunchPhase::StartedUnregistered => { - return Err(BackgroundLaunchError::LaunchPending { task_id }); - } - BackgroundLaunchPhase::IdentityMismatch - | BackgroundLaunchPhase::EndedBeforeRegistration { - release: EndedLaunchRelease::Unproven, - } => { - return Err(BackgroundLaunchError::PredecessorReleaseUnproven { task_id }); - } - // every first launch is the direct-segment trainer, whose wrapper - // exit does not cover its worker - BackgroundLaunchPhase::EndedBeforeRegistration { - release: EndedLaunchRelease::ConfirmedExited, - } => { - let BackgroundCommandContract::DirectSegmentTrainer { .. } = receipt.contract; - let held = hold_predecessor_lock(conn, resource, task_id)?; - clearance.held_locks.push((task_id, held)); - } - BackgroundLaunchPhase::Superseded - | BackgroundLaunchPhase::Registered - | BackgroundLaunchPhase::EndedBeforeRegistration { - release: EndedLaunchRelease::NeverSpawned, - } => {} - } - } - let Some(task_id) = resource.registered_background_task else { - return Ok(clearance); - }; - let row = crate::store::task_by_id_on(conn, task_id)? - .ok_or(BackgroundLaunchError::BackgroundTaskMissing { task_id })?; - if !row.state.is_terminal() { - return Err(BackgroundLaunchError::BackgroundTaskActive { - task_id, - state: row.status(), - }); - } - match ended_task_release(&row) { - EndedLaunchRelease::Unproven => { - return Err(BackgroundLaunchError::PredecessorReleaseUnproven { task_id }); - } - EndedLaunchRelease::NeverSpawned => {} - // the confirmed evidence must be the kind that the task's contract names, - // and a trainer also needs its released lock - EndedLaunchRelease::ConfirmedExited => { - let witness = ended_task_witness(conn, resource.authority_machine(), task_id) - .map_err(|error| predecessor_lock_error(task_id, error))?; - if !witness.accepts(&row.work_exit_evidence()) { - return Err(BackgroundLaunchError::PredecessorReleaseUnproven { task_id }); - } - if witness == EndedTaskWitness::TrainerLock { - let held = hold_predecessor_lock(conn, resource, task_id)?; - clearance.held_locks.push((task_id, held)); - } - } - } - - Ok(clearance) -} - -/// Take the exact released lock named by one ended predecessor's own association -fn hold_predecessor_lock( - conn: &Connection, - resource: &Resource, - task_id: TaskId, -) -> Result { - hold_released_trainer_lock(conn, resource, task_id, task_id) - .map_err(|error| predecessor_lock_error(task_id, error)) -} - -fn predecessor_lock_error( - task_id: TaskId, - error: TrainerLockReleaseError, -) -> BackgroundLaunchError { - match error { - TrainerLockReleaseError::Unproven(gap) => { - BackgroundLaunchError::PredecessorOwnershipUnproven { task_id, gap } - } - TrainerLockReleaseError::Store(error) => BackgroundLaunchError::Resource(error), - } -} - -/// Promote a started first background launch to the registered background task -/// -/// The task layer must record a running row and a running accepted identity -/// The registration compares the prior registration saved in the receipt, so a -/// stale launch cannot replace a newer background task -pub(crate) fn promote_started_background_launch_on( - tx: &Transaction<'_>, - resource: &Resource, -) -> Result, ResourceStoreError> { - let Some((receipt, phase)) = current_launch_record_on(tx, resource)? else { - return Ok(None); - }; - if phase != BackgroundLaunchPhase::StartedUnregistered - || resource.registered_background_task != receipt.replaces_task - || select_non_closed_loan(tx, resource.id)?.is_some() - { - return Ok(None); - } - - let state_revision = - advance_resource_on(tx, resource, Some(receipt.task_id)).map_err(|error| match error { - AdvanceError::Exhausted => { - ResourceStoreError::Conflict(ConflictReason::RevisionExhausted) - } - AdvanceError::Changed => { - ResourceStoreError::Conflict(ConflictReason::ResourceRevisionChanged) - } - AdvanceError::Storage(error) => ResourceStoreError::Storage(error), - })?; - let mut promoted = resource.clone(); - promoted.state_revision = state_revision; - promoted.registered_background_task = Some(receipt.task_id); - - Ok(Some(promoted)) -} - -/// Return the first background launch task that still keeps the resource reserved -pub(crate) fn pending_background_launch_on( - conn: &Connection, - resource: &Resource, -) -> Result, ResourceStoreError> { - Ok(current_launch_on(conn, resource)? - .as_ref() - .and_then(BackgroundLaunchView::pending_task)) -} - -/// Return the latest first background launch and the registration it replaces -pub(super) fn current_launch_and_predecessor_on( - conn: &Connection, - resource: &Resource, -) -> Result)>, ResourceStoreError> { - Ok( - current_launch_record_on(conn, resource)?.map(|(receipt, phase)| { - let view = BackgroundLaunchView { - request_id: receipt.request_id, - task_id: receipt.task_id, - phase, - }; - (view, receipt.replaces_task) - }), - ) -} - -/// Decide whether saved history proves that an unregistered resource is idle -/// -/// The latest record wins. A launch newer than every loan decides by its task -/// outcome; otherwise the latest loan decides by its closure. A resource with -/// neither record has no evidence, and a missing task row is never evidence -pub(crate) fn idle_boundary_decision_on( - conn: &Connection, - resource: &Resource, -) -> Result { - use IdleBoundaryDecision::{Proven, Unproven}; - - if resource.registered_background_task.is_some() { - return Ok(Unproven(IdleProofGap::InconsistentHistory)); - } - // a current operator attestation is newer than every launch and loan it names - if let Some(boundary) = super::operator_release::current_operator_boundary_on(conn, resource)? { - return Ok(Proven(IdleBoundaryProof::OperatorAttestedGpuFree { - operation_id: boundary.operation_id, - task_id: boundary.task_id, - })); - } - if let Some((receipt, phase)) = current_launch_record_on(conn, resource)? { - return Ok(match phase { - BackgroundLaunchPhase::Superseded => loan_idle_decision(conn, resource)?, - BackgroundLaunchPhase::EndedBeforeRegistration { - release: EndedLaunchRelease::NeverSpawned, - } => Proven(IdleBoundaryProof::BackgroundLaunchNeverSpawned { - request_id: receipt.request_id, - task_id: receipt.task_id, - }), - // the maintained trainer starts its worker in another session, so its - // own process-group exit does not prove that the worker released the GPU - BackgroundLaunchPhase::EndedBeforeRegistration { - release: EndedLaunchRelease::ConfirmedExited | EndedLaunchRelease::Unproven, - } => Unproven(IdleProofGap::BackgroundLaunchReleaseUnproven { - task_id: receipt.task_id, - }), - BackgroundLaunchPhase::Queued - | BackgroundLaunchPhase::StartedUnregistered - | BackgroundLaunchPhase::Registered - | BackgroundLaunchPhase::IdentityMismatch => { - Unproven(IdleProofGap::InconsistentHistory) - } - }); - } - // an initial attestation counts only before any loan or launch exists - if let Some(operation_id) = super::initial_idle::initial_idle_boundary_on(conn, resource)? { - return Ok(Proven(IdleBoundaryProof::OperatorAttestedInitialIdle { - operation_id, - })); - } - - loan_idle_decision(conn, resource) -} - -fn loan_idle_decision( - conn: &Connection, - resource: &Resource, -) -> Result { - use IdleBoundaryDecision::{Proven, Unproven}; - - let Some(loan) = latest_loan_on(conn, resource.id)? else { - return Ok(Unproven(IdleProofGap::NoIdleEvidence)); - }; - let LoanState::Closed { result } = &loan.state else { - return Ok(Unproven(IdleProofGap::InconsistentHistory)); - }; - Ok(match result { - // the no-resume decision required the returned run to have ended after a - // verified or operator-attested release, or an idle context with no registered task - LoanClosure::NoResume { .. } => { - Proven(IdleBoundaryProof::SupervisorNoResume { loan_id: loan.id }) - } - // the resolution required a terminal return task with proven process release - LoanClosure::RestoreEnded { task_id, .. } => { - Proven(IdleBoundaryProof::RestoreEndedWithProvenRelease { - loan_id: loan.id, - task_id: *task_id, - }) - } - // the closure required a successful foreground end with a confirmed - // process-group exit, and it cleared the background registration - LoanClosure::ForegroundReturnEnded { task_id, .. } => { - Proven(IdleBoundaryProof::ForegroundReturnEnded { - loan_id: loan.id, - task_id: *task_id, - }) - } - // only the exact saved attestation receipt that closed this loan is evidence - LoanClosure::OperatorAttestedRestoreEnded { - task_id, - operation_id, - .. - } => { - if super::operator_release::operator_restore_closure_matches_on(conn, resource, &loan)? - { - Proven(IdleBoundaryProof::OperatorAttestedGpuFree { - operation_id: *operation_id, - task_id: *task_id, - }) - } else { - Unproven(IdleProofGap::InconsistentHistory) - } - } - // these closures keep or register a background task - LoanClosure::Resumed { .. } | LoanClosure::NotStopped { .. } => { - Unproven(IdleProofGap::InconsistentHistory) - } - }) -} - -/// Open a Serving loan for the next request in serving order from a proven idle boundary -/// -/// The loan, request assignment, resource revision, and opening receipt commit -/// in the caller's transaction. The receipt is the only provenance that lets the -/// assigned command start without a registered background task -pub(crate) fn open_idle_serving_loan_on( - tx: &Transaction<'_>, - authority_machine: MachineId, - resource: &Resource, - mut request: ResourceRequest, - proof: IdleBoundaryProof, -) -> Result<(Loan, ResourceRequest), ResourceStoreError> { - if resource.registered_background_task.is_some() { - return Err(ResourceStoreError::Conflict( - ConflictReason::ResourceAssignmentChanged, - )); - } - if request.resource_id != resource.id || request.state != ResourceRequestState::Queued { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestStateChanged, - )); - } - let loan = Loan { - id: LoanId::new(), - resource_id: resource.id, - state: LoanState::Active { - phase: LoanPhase::Serving { - return_context: ReturnContext::Idle, - current_request_id: request.request_id, - release_provenance: ServingReleaseProvenance::IdleBoundary { - proof: proof.clone(), - }, - }, - }, - }; - tx.execute( - "INSERT INTO loans (id, resource_id, state_json) VALUES (?1, ?2, ?3)", - params![ - loan.id.as_uuid().to_string(), - resource.id.as_uuid().to_string(), - encode_resource_json(&loan.state)?, - ], - )?; - - request.state = ResourceRequestState::Assigned { loan_id: loan.id }; - let changed = tx.execute( - "UPDATE resource_requests SET state_json = ?1 - WHERE request_id = ?2 AND resource_id = ?3 - AND json_extract(state_json, '$.type') = 'queued'", - params![ - encode_resource_json(&request.state)?, - request.request_id.0.to_string(), - resource.id.as_uuid().to_string(), - ], - )?; - if changed != 1 { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestStateChanged, - )); - } - - let state_revision = advance_resource_on(tx, resource, None).map_err(|error| match error { - AdvanceError::Exhausted => ResourceStoreError::Conflict(ConflictReason::RevisionExhausted), - AdvanceError::Changed => { - ResourceStoreError::Conflict(ConflictReason::ResourceRevisionChanged) - } - AdvanceError::Storage(error) => ResourceStoreError::Storage(error), - })?; - let receipt = IdleOpeningReceipt { - loan_id: loan.id, - resource_id: resource.id, - authority_machine, - request_id: request.request_id, - state_revision, - proof, - }; - tx.execute( - "INSERT INTO resource_idle_openings (loan_id, resource_id, receipt_json) - VALUES (?1, ?2, ?3)", - params![ - loan.id.as_uuid().to_string(), - resource.id.as_uuid().to_string(), - encode_resource_json(&receipt)?, - ], - )?; - - Ok((loan, request)) -} - -/// Check that an idle Serving provenance matches its saved loan-opening receipt -pub(crate) fn idle_opening_matches_on( - conn: &Connection, - authority_machine: MachineId, - resource: &Resource, - loan: &Loan, - return_context: &ReturnContext, - proof: &IdleBoundaryProof, -) -> Result { - let saved: Option = conn - .query_row( - "SELECT receipt_json FROM resource_idle_openings WHERE loan_id = ?1", - [loan.id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional()?; - let Some(saved) = saved else { - return Ok(false); - }; - let receipt: IdleOpeningReceipt = serde_json::from_str(&saved) - .map_err(|error| ResourceStoreError::corrupt("idle opening receipt", error))?; - - Ok(receipt.loan_id == loan.id - && receipt.resource_id == resource.id - && loan.resource_id == resource.id - && receipt.authority_machine == authority_machine - && resource.authority_machine() == authority_machine - && receipt.proof == *proof - && *return_context == ReturnContext::Idle - && resource.registered_background_task.is_none()) -} - -/// Read the latest first background launch of one resource with its derived phase -pub(super) fn current_launch_on( - conn: &Connection, - resource: &Resource, -) -> Result, ResourceStoreError> { - Ok( - current_launch_record_on(conn, resource)?.map(|(receipt, phase)| BackgroundLaunchView { - request_id: receipt.request_id, - task_id: receipt.task_id, - phase, - }), - ) -} - -fn current_launch_record_on( - conn: &Connection, - resource: &Resource, -) -> Result, ResourceStoreError> { - let saved: Option = conn - .query_row( - "SELECT receipt_json FROM resource_background_launches - WHERE resource_id = ?1 ORDER BY rowid DESC LIMIT 1", - [resource.id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional()?; - let Some(saved) = saved else { - return Ok(None); - }; - let receipt: BackgroundLaunchReceipt = serde_json::from_str(&saved) - .map_err(|error| ResourceStoreError::corrupt("background launch receipt", error))?; - if receipt.resource_id != resource.id - || receipt.authority_machine != resource.authority_machine() - { - return Ok(Some((receipt, BackgroundLaunchPhase::IdentityMismatch))); - } - if latest_loan_on(conn, resource.id)?.map(|loan| loan.id) != receipt.preceding_loan { - return Ok(Some((receipt, BackgroundLaunchPhase::Superseded))); - } - // an operator attestation saved after this launch cleared the task it registered - if super::operator_release::current_operator_boundary_on(conn, resource)? - .is_some_and(|boundary| boundary.preceding_launch == Some(receipt.request_id)) - { - return Ok(Some((receipt, BackgroundLaunchPhase::Superseded))); - } - if resource.registered_background_task == Some(receipt.task_id) { - return Ok(Some((receipt, BackgroundLaunchPhase::Registered))); - } - - let Some(row) = bound_launch_row_on(conn, &receipt).map_err(|error| match error { - BackgroundLaunchError::Resource(error) => error, - BackgroundLaunchError::Storage(error) => ResourceStoreError::Storage(error), - error => ResourceStoreError::corrupt("background launch records", error), - })? - else { - return Ok(Some((receipt, BackgroundLaunchPhase::IdentityMismatch))); - }; - let phase = match &row.state { - TaskState::Queued => BackgroundLaunchPhase::Queued, - TaskState::Running { .. } => BackgroundLaunchPhase::StartedUnregistered, - TaskState::Lost | TaskState::Finished { .. } => { - BackgroundLaunchPhase::EndedBeforeRegistration { - release: ended_task_release(&row), - } - } - }; - - Ok(Some((receipt, phase))) -} - -/// Classify the task-layer release evidence of one background task row -/// -/// The task layer records that no work started only when it failed or -/// cancelled the row before any worker child or container started, so that -/// evidence with any other outcome proves nothing. A lost or non-terminal row is -/// never released -fn ended_task_release(row: &TaskRow) -> EndedLaunchRelease { - let TaskState::Finished { reason } = &row.state else { - return EndedLaunchRelease::Unproven; - }; - match row.work_exit_evidence() { - WorkExitEvidence::ProcessGroupExited | WorkExitEvidence::ContainerRemoved { .. } => { - EndedLaunchRelease::ConfirmedExited - } - WorkExitEvidence::NoWorkStarted - if matches!( - reason, - ExitReason::SpawnFailed { .. } | ExitReason::Cancelled - ) => - { - EndedLaunchRelease::NeverSpawned - } - WorkExitEvidence::NoWorkStarted | WorkExitEvidence::Unconfirmed => { - EndedLaunchRelease::Unproven - } - } -} - -/// Return the task row when its callback route and accepted identity match the receipt -/// -/// A co-located launch keeps its route on the authority. A remote supervisor's -/// route lives on its own machine, so the accepted identity and the first queued -/// event must name that machine as the callback origin -fn bound_launch_row_on( - conn: &Connection, - receipt: &BackgroundLaunchReceipt, -) -> Result, BackgroundLaunchError> { - let task_id = receipt.task_id; - let authority = receipt.authority_machine; - let origin = receipt.origin_machine(); - let Some(row) = crate::store::task_by_id_on(conn, task_id)? else { - return Ok(None); - }; - let Some(ExecutorIdentity::Accepted(record)) = executor_identity_on(conn, task_id)? else { - return Ok(None); - }; - let Some(spec) = record.current_spec() else { - return Ok(None); - }; - let identity_matches = record.is_owned_by(task_id, origin, authority) - && record.state == row.status() - && normalized_spec_sha256(spec)? == receipt.normalized_spec_sha256 - && row.thread == spec.thread - && row.thread == receipt.supervisor.thread; - if !identity_matches { - return Ok(None); - } - let local_route = origin_route_by_request_on(conn, receipt.request_id)?; - let route_matches = match receipt.origin { - BackgroundLaunchOrigin::CoLocated => local_route.is_some_and(|route| { - route.task == task_id - && route.origin_machine == authority - && route.execution_machine == authority - && route.thread == row.thread - }), - BackgroundLaunchOrigin::RemoteSupervisor { .. } => { - local_route.is_none() - && origin != authority - && super::resource_task_row_matches(&row, task_id, spec) - && crate::store::initial_queued_event_matches_on(conn, task_id, origin, authority) - .map_err(|error| { - BackgroundLaunchError::TaskRecords(AppError::Internal { - message: format!("background launch first event: {error}"), - }) - })? - } - }; - - Ok(route_matches.then_some(row)) -} - -fn receipt_by_request_on( - conn: &Connection, - request_id: RequestId, -) -> Result, BackgroundLaunchError> { - let saved: Option = conn - .query_row( - "SELECT receipt_json FROM resource_background_launches WHERE request_id = ?1", - [request_id.0.to_string()], - |row| row.get(0), - ) - .optional()?; - Ok(saved.map(|json| serde_json::from_str(&json)).transpose()?) -} - -fn receipt_by_task_on( - conn: &Connection, - task_id: TaskId, -) -> Result, BackgroundLaunchError> { - let saved: Option = conn - .query_row( - "SELECT receipt_json FROM resource_background_launches WHERE task_id = ?1", - [task_id.to_string()], - |row| row.get(0), - ) - .optional()?; - Ok(saved.map(|json| serde_json::from_str(&json)).transpose()?) -} - -/// Return the first background launch that registered one ended trainer -/// -/// The receipt, task row, accepted identity, and callback route must all name -/// this resource and authority. The result is the launch request and the digest -/// of the accepted normalized spec -pub(super) fn first_launch_of_registered_trainer_on( - conn: &Connection, - resource: &Resource, - task_id: TaskId, -) -> Result, BackgroundLaunchError> { - let Some(receipt) = receipt_by_task_on(conn, task_id)? else { - return Ok(None); - }; - if receipt.task_id != task_id - || receipt.resource_id != resource.id - || receipt.authority_machine != resource.authority_machine() - || bound_launch_row_on(conn, &receipt)?.is_none() - { - return Ok(None); - } - - Ok(Some((receipt.request_id, receipt.normalized_spec_sha256))) -} - -/// Return the request identity of the latest first background launch of one resource -pub(super) fn latest_launch_request_on( - conn: &Connection, - resource_id: ResourceId, -) -> Result, ResourceStoreError> { - let saved: Option = conn - .query_row( - "SELECT request_id FROM resource_background_launches - WHERE resource_id = ?1 ORDER BY rowid DESC LIMIT 1", - [resource_id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional()?; - saved - .map(|id| { - uuid::Uuid::parse_str(&id).map(RequestId).map_err(|error| { - ResourceStoreError::corrupt("background launch request identity", error) - }) - }) - .transpose() -} - -pub(super) fn latest_loan_on( - conn: &Connection, - resource_id: ResourceId, -) -> Result, ResourceStoreError> { - let saved: Option<(String, String)> = conn - .query_row( - "SELECT id, state_json FROM loans WHERE resource_id = ?1 ORDER BY rowid DESC LIMIT 1", - [resource_id.as_uuid().to_string()], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .optional()?; - let Some((id, state)) = saved else { - return Ok(None); - }; - let id = id - .parse::() - .map_err(|error| ResourceStoreError::corrupt("loan identity", error))?; - let state: LoanState = serde_json::from_str(&state) - .map_err(|error| ResourceStoreError::corrupt("loan state", error))?; - - Ok(Some(Loan { - id, - resource_id, - state, - })) -} - -pub(super) enum AdvanceError { - Exhausted, - Changed, - Storage(rusqlite::Error), -} - -impl From for AdvanceError { - fn from(error: rusqlite::Error) -> Self { - Self::Storage(error) - } -} - -/// Advance the revision and set the registration, comparing the saved resource -pub(super) fn advance_resource_on( - tx: &Connection, - resource: &Resource, - registered_background_task: Option, -) -> Result { - let next = resource - .state_revision - .get() - .checked_add(1) - .map(ResourceRevision::new) - .ok_or(AdvanceError::Exhausted)?; - let sql_integer = |value: u64| i64::try_from(value).map_err(|_| AdvanceError::Exhausted); - // SQLite cannot hold a larger assignment, so no saved row can match the compare - let Ok(assignment_revision) = i64::try_from(resource.assignment_revision.get()) else { - return Err(AdvanceError::Changed); - }; - let changed = tx.execute( - "UPDATE resources SET state_revision = ?1, registered_background_task = ?2 - WHERE id = ?3 AND authority_machine = ?4 AND state_revision = ?5 - AND supervisor_machine = ?6 AND supervisor_thread = ?7 - AND assignment_revision = ?8 AND registered_background_task IS ?9", - params![ - sql_integer(next.get())?, - registered_background_task.map(|task| task.to_string()), - resource.id.as_uuid().to_string(), - resource.authority_machine().as_uuid().to_string(), - sql_integer(resource.state_revision.get())?, - resource.supervisor.machine.as_uuid().to_string(), - resource.supervisor.thread.to_string(), - assignment_revision, - resource - .registered_background_task - .map(|task| task.to_string()), - ], - )?; - if changed != 1 { - return Err(AdvanceError::Changed); - } - - Ok(next) -} diff --git a/src/store/resource/cancellation.rs b/src/store/resource/cancellation.rs deleted file mode 100644 index b864829..0000000 --- a/src/store/resource/cancellation.rs +++ /dev/null @@ -1,265 +0,0 @@ -//! Durable resource cancellation receipts on the authority - -use rusqlite::{Connection, OptionalExtension, Transaction, TransactionBehavior}; - -use super::{decode_resource_json, encode_resource_json}; -use crate::cancellation::ResourceCancellationRequestIdentity; -use crate::machine::MachineId; -use crate::resource::ResourceRequestState; -use crate::resource::store::{ - ConflictReason, QueueCancellationResult, ResourceStoreError, - cancel_request_before_activation_on, requests_for_resource_for_authority, -}; -use crate::store::Store; -use crate::submission::{ - ResourceCancellationIneligibleReason, ResourceCancellationOutcome, ResourceCancellationReceipt, - ResourceRoutePhase, ResourceRouteProof, normalized_spec_sha256, -}; - -impl Store { - /// Return the exact saved authority receipt for one cancellation identity - pub(crate) fn resource_cancellation_receipt( - &self, - identity: &ResourceCancellationRequestIdentity, - ) -> Result, ResourceStoreError> { - saved_cancellation_receipt_on(&self.conn, identity) - } - - /// Apply one resource cancellation and retain its exact result in one transaction - pub(crate) fn cancel_resource_request_with_receipt( - &mut self, - authority_machine: MachineId, - identity: ResourceCancellationRequestIdentity, - proof: ResourceRouteProof, - ) -> Result { - if !proof_matches_identity(authority_machine, &identity, &proof) { - return Err(receipt_mismatch()); - } - - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate)?; - if let Some(receipt) = saved_cancellation_receipt_on(&tx, &identity)? { - tx.commit()?; - return Ok(receipt); - } - check_retained_request(&tx, authority_machine, &identity, &proof)?; - - let outcome = match proof.phase { - ResourceRoutePhase::Activated => { - not_eligible(ResourceCancellationIneligibleReason::Activated) - } - ResourceRoutePhase::Rejected { reason } => { - not_eligible(ResourceCancellationIneligibleReason::Rejected { reason }) - } - ResourceRoutePhase::AcceptanceUnknown - | ResourceRoutePhase::Waiting - | ResourceRoutePhase::CancelledBeforeLaunch => { - queue_cancellation_outcome(cancel_in_savepoint(&tx, authority_machine, &identity)?)? - } - }; - - let receipt = ResourceCancellationReceipt { - cancellation: identity.cancellation, - requester_machine: identity.requester_machine, - request: identity.request, - task: identity.task, - origin_machine: identity.origin_machine, - authority_machine: identity.authority_machine, - resource: identity.resource, - target_phase: identity.target_phase.clone(), - outcome, - }; - tx.execute( - "INSERT INTO resource_cancellation_receipts - (cancellation_id, request_json, receipt_json) VALUES (?1, ?2, ?3)", - rusqlite::params![ - identity.cancellation.to_string(), - encode_resource_json(&identity)?, - encode_resource_json(&receipt)?, - ], - )?; - tx.commit()?; - Ok(receipt) - } -} - -fn receipt_mismatch() -> ResourceStoreError { - ResourceStoreError::Conflict(ConflictReason::CancellationReceiptMismatch) -} - -fn not_eligible(reason: ResourceCancellationIneligibleReason) -> ResourceCancellationOutcome { - ResourceCancellationOutcome::NotEligible { reason } -} - -/// Require that the origin's route proof names this cancellation and a forward phase -fn proof_matches_identity( - authority_machine: MachineId, - identity: &ResourceCancellationRequestIdentity, - proof: &ResourceRouteProof, -) -> bool { - identity.requester_machine == identity.origin_machine - && identity.authority_machine == authority_machine - && proof.request == identity.request - && proof.task == identity.task - && proof.resource == identity.resource - && proof.origin_machine == identity.origin_machine - && proof.authority_machine == authority_machine - && phase_is_forward(&identity.target_phase, &proof.phase) -} - -fn phase_is_forward(target: &ResourceRoutePhase, current: &ResourceRoutePhase) -> bool { - match target { - ResourceRoutePhase::AcceptanceUnknown => true, - ResourceRoutePhase::Waiting => !matches!(current, ResourceRoutePhase::AcceptanceUnknown), - ResourceRoutePhase::CancelledBeforeLaunch => { - matches!(current, ResourceRoutePhase::CancelledBeforeLaunch) - } - ResourceRoutePhase::Activated | ResourceRoutePhase::Rejected { .. } => target == current, - } -} - -/// Require that any retained request is exactly the one the proof names -/// -/// A waiting or activated route proves that the authority retained its request -fn check_retained_request( - conn: &Connection, - authority_machine: MachineId, - identity: &ResourceCancellationRequestIdentity, - proof: &ResourceRouteProof, -) -> Result<(), ResourceStoreError> { - let requests = - match requests_for_resource_for_authority(conn, authority_machine, identity.resource) { - Ok(requests) => requests, - Err(ResourceStoreError::ResourceNotFound) - if matches!(&proof.phase, ResourceRoutePhase::Rejected { .. }) => - { - Vec::new() - } - Err(error) => return Err(error), - }; - let retained = requests - .iter() - .find(|request| request.request_id == identity.request || request.task_id == identity.task); - let Some(retained) = retained else { - return match &proof.phase { - ResourceRoutePhase::Waiting | ResourceRoutePhase::Activated => { - Err(ResourceStoreError::Conflict(ConflictReason::RequestMissing)) - } - _ => Ok(()), - }; - }; - let saved_digest = normalized_spec_sha256(retained.spec().as_normalized()) - .map_err(|error| ResourceStoreError::corrupt("resource request spec", error))?; - if retained.request_id != identity.request - || retained.task_id != identity.task - || retained.resource_id != identity.resource - || retained.origin_machine != identity.origin_machine - || saved_digest != proof.normalized_spec_sha256 - { - return Err(ResourceStoreError::Conflict( - ConflictReason::RequestIdentityMismatch, - )); - } - - Ok(()) -} - -/// Cancel the request inside a savepoint of the receipt transaction -/// -/// A refusal after partial writes, such as an executor acceptance that won the -/// race after a prevention row was written, rolls back only the cancellation; -/// the receipt of that refusal still commits with the outer transaction -fn cancel_in_savepoint( - tx: &Transaction<'_>, - authority_machine: MachineId, - identity: &ResourceCancellationRequestIdentity, -) -> Result, ResourceStoreError> { - tx.execute_batch("SAVEPOINT resource_cancellation")?; - let result = cancel_request_before_activation_on( - tx, - authority_machine, - identity.request, - identity.task, - identity.resource, - identity.origin_machine, - ); - if result.is_err() { - tx.execute_batch("ROLLBACK TO resource_cancellation")?; - } - tx.execute_batch("RELEASE resource_cancellation")?; - Ok(result) -} - -/// Map the queue cancellation result to the outcome its receipt records -fn queue_cancellation_outcome( - result: Result, -) -> Result { - let request = match result { - Ok(QueueCancellationResult::PreventedBeforeAcceptance) => { - return Ok(ResourceCancellationOutcome::PreventedBeforeAcceptance); - } - Ok(QueueCancellationResult::Request(request)) => request, - Err(ResourceStoreError::ExecutorAlreadyAccepted { .. }) => { - return Ok(not_eligible( - ResourceCancellationIneligibleReason::Activated, - )); - } - Err(error) => return Err(error), - }; - match request.state { - ResourceRequestState::CancelledBeforeLaunch => { - Ok(ResourceCancellationOutcome::CancelledBeforeLaunch) - } - ResourceRequestState::Finished { .. } => { - Ok(not_eligible(ResourceCancellationIneligibleReason::Terminal)) - } - ResourceRequestState::Rejected { reason } => Ok(not_eligible( - ResourceCancellationIneligibleReason::Rejected { reason }, - )), - ResourceRequestState::Queued | ResourceRequestState::Assigned { .. } => Err( - ResourceStoreError::Conflict(ConflictReason::RequestStateChanged), - ), - } -} - -/// Read the saved receipt of one cancellation and require its exact identity -fn saved_cancellation_receipt_on( - conn: &Connection, - identity: &ResourceCancellationRequestIdentity, -) -> Result, ResourceStoreError> { - let saved: Option<(String, String)> = conn - .query_row( - "SELECT request_json, receipt_json FROM resource_cancellation_receipts - WHERE cancellation_id=?1", - [identity.cancellation.to_string()], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .optional()?; - let Some((request_json, receipt_json)) = saved else { - return Ok(None); - }; - let saved_identity: ResourceCancellationRequestIdentity = - decode_resource_json("resource cancellation request", &request_json)?; - let receipt: ResourceCancellationReceipt = - decode_resource_json("resource cancellation receipt", &receipt_json)?; - if saved_identity != *identity || !receipt_matches(&receipt, identity) { - return Err(receipt_mismatch()); - } - - Ok(Some(receipt)) -} - -fn receipt_matches( - receipt: &ResourceCancellationReceipt, - identity: &ResourceCancellationRequestIdentity, -) -> bool { - receipt.cancellation == identity.cancellation - && receipt.requester_machine == identity.requester_machine - && receipt.request == identity.request - && receipt.task == identity.task - && receipt.origin_machine == identity.origin_machine - && receipt.authority_machine == identity.authority_machine - && receipt.resource == identity.resource - && receipt.target_phase == identity.target_phase -} diff --git a/src/store/resource/controls.rs b/src/store/resource/controls.rs deleted file mode 100644 index a8266bd..0000000 --- a/src/store/resource/controls.rs +++ /dev/null @@ -1,721 +0,0 @@ -//! Read models and idempotent operator controls for authority-owned resources -//! -//! Each control commits its operation identity, revision check, and any notice -//! transition in one IMMEDIATE transaction. The daemon performs the external -//! effect only after that commit, and an exact retry returns the saved effect -//! without a second revision check or a second notice attempt - -use rusqlite::{Connection, OptionalExtension, Transaction, TransactionBehavior, params}; -use serde::{Deserialize, Serialize}; -use uuid::Uuid; - -use crate::machine::MachineId; -use crate::resource::api::{BrowserResourceAction, QueuePlacement}; -use crate::resource::store::{ - ResourceStoreError, SupervisorNoticeStoreError, decode_supervisor_notice_record, - requests_for_resource_for_authority, resources_for_authority, - retarget_supervisor_notice_in_transaction, return_window_on, rewrite_queued_request_ranks, - select_non_closed_loan, select_request_by_id, select_supervisor_notice_record, - swap_resource_revision, update_supervisor_notice_cas, -}; -use crate::resource::{ - ActionId, AssignmentRevision, DeliveryAttemptId, Loan, LoanId, LoanPhase, LoanState, NoticeId, - Resource, ResourceId, ResourceRequest, ResourceRequestState, ResourceRevision, - ReturnDecisionWindow, ReturnExecutionMode, SupervisorAddress, SupervisorNotice, - SupervisorNoticeDelivery, -}; -use crate::store::{BackgroundLaunchView, Store}; -use crate::submission::RequestId; - -use super::select_authority_resource; - -#[cfg(test)] -mod test_support; - -/// Durable resource state from which every read view is derived -#[derive(Debug, Clone)] -pub(crate) struct ResourceReadModel { - /// Resource owned by the reading authority - pub(crate) resource: Resource, - /// Non-closed loan, if one reserves the resource - pub(crate) loan: Option, - /// Every request in serving order - pub(crate) requests: Vec, - /// Notices that belong to the non-closed loan - pub(crate) notices: Vec, - /// Latest first background launch with its phase derived from durable task state - /// - /// It keeps an unregistered resource reserved while it is pending or while it - /// ended before registration with no automatic or attested release - pub(crate) background_launch: Option, - /// Accepted execution mode of the exact current Restoring loan, if proven - pub(crate) return_execution_mode: Option, - /// Decision window of the pending return action, if the loan awaits one - pub(crate) return_window: Option, -} - -/// Immutable content bound to one control operation identity -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(crate) struct ResourceControlRequest { - /// Resource that the control targets - pub(crate) resource_id: ResourceId, - /// Resource revision that the caller observed - pub(crate) expected_revision: ResourceRevision, - /// Requested control - pub(crate) action: BrowserResourceAction, -} - -/// External effect that the daemon performs after the operation commits -#[derive(Debug, Clone)] -pub(crate) enum ResourceControlEffect { - /// Ask the request origin to cancel this queued request - CancelQueued { - /// Queued request read in the committing transaction - request: ResourceRequest, - }, - /// Queue ranks changed with no work outside the authority - Reordered, - /// Ask the request origin to cancel this active command task - StopActive { - /// Assigned request whose task holds the resource - request: ResourceRequest, - }, - /// Deliver the one explicit attempt reserved by this operation - Renotify { - /// Notice whose automatic attempts failed - notice_id: NoticeId, - /// Attempt reserved for this operation - attempt_id: DeliveryAttemptId, - }, -} - -/// Committed control operation -#[derive(Debug, Clone)] -pub(crate) struct ResourceControlStart { - /// Effect bound to the operation - pub(crate) effect: ResourceControlEffect, - /// Whether an earlier call already committed this operation - pub(crate) replayed: bool, -} - -/// Supervisor assignment after an explicit replacement -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) struct SupervisorReplacement { - /// Resource with its new supervisor and assignment revision - pub(crate) resource: Resource, - /// Undelivered notices moved to the new supervisor - pub(crate) retargeted: Vec, -} - -/// Why an authority refused or failed one resource control -#[derive(Debug, thiserror::Error)] -pub(crate) enum ResourceControlError { - /// The resource does not exist on this authority - #[error("resource not found")] - NotFound, - /// The resource belongs to a different authority - #[error("resource authority mismatch: expected {expected}, found {found}")] - WrongAuthority { - /// Fixed authority recorded for the resource - expected: MachineId, - /// Machine that received the control - found: MachineId, - }, - /// The caller observed an older resource revision - #[error("resource revision is stale")] - StaleRevision { - /// Current resource revision - current: ResourceRevision, - }, - /// The operation identity already names different content - #[error("operation identity already names different content")] - Conflict, - /// The action does not apply to the current resource phase - #[error("{0}")] - NotAllowed(String), - /// Durable notice state refused the transition - #[error(transparent)] - Notice(#[from] SupervisorNoticeStoreError), - /// Stored operation content cannot be decoded - #[error("resource control operation encoding failed: {0}")] - Json(#[from] serde_json::Error), - /// Other resource storage failure - #[error(transparent)] - Resource(ResourceStoreError), - /// SQLite failed - #[error("resource control storage error: {0}")] - Storage(#[from] rusqlite::Error), -} - -impl From for ResourceControlError { - fn from(error: ResourceStoreError) -> Self { - match error { - ResourceStoreError::ResourceNotFound => Self::NotFound, - ResourceStoreError::WrongAuthority { expected, found } => { - Self::WrongAuthority { expected, found } - } - ResourceStoreError::Storage(error) => Self::Storage(error), - other => Self::Resource(other), - } - } -} - -impl Store { - /// Read durable resource state owned by this authority, optionally for one resource - pub(crate) fn resource_read_models( - &self, - authority_machine: MachineId, - resource_id: Option, - ) -> Result, ResourceStoreError> { - resource_read_models(&self.conn, authority_machine, resource_id) - } - - /// Commit one idempotent control operation and return its external effect - pub(crate) fn begin_resource_control( - &mut self, - authority_machine: MachineId, - operation_id: Uuid, - request: &ResourceControlRequest, - attempt_id: DeliveryAttemptId, - ) -> Result { - begin_resource_control( - &mut self.conn, - authority_machine, - operation_id, - request, - attempt_id, - ) - } - - /// Replace the supervisor and retarget undelivered notices in one transaction - pub(crate) fn replace_resource_supervisor( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - expected_revision: ResourceRevision, - supervisor: SupervisorAddress, - ) -> Result { - replace_resource_supervisor( - &mut self.conn, - authority_machine, - resource_id, - expected_revision, - supervisor, - ) - } -} - -fn resource_read_models( - conn: &Connection, - authority_machine: MachineId, - resource_id: Option, -) -> Result, ResourceStoreError> { - let snapshots = resources_for_authority(conn, authority_machine)?; - snapshots - .into_iter() - .filter(|snapshot| resource_id.is_none_or(|id| snapshot.resource.id == id)) - .map(|snapshot| { - let requests = - requests_for_resource_for_authority(conn, authority_machine, snapshot.resource.id)?; - let notices = match &snapshot.loan { - Some(loan) => notices_for_loan(conn, loan.id)?, - None => Vec::new(), - }; - let background_launch = super::background::current_launch_on(conn, &snapshot.resource)?; - let return_execution_mode = super::restore::current_return_execution_mode_on( - conn, - &snapshot.resource, - snapshot.loan.as_ref(), - ) - .map_err(|error| ResourceStoreError::ReturnDecisionRead(error.to_string()))?; - let return_window = match snapshot.loan.as_ref().map(|loan| &loan.state) { - Some(LoanState::Active { - phase: LoanPhase::AwaitingReturn { action_id, .. }, - }) => return_window_on(conn, *action_id)?, - _ => None, - }; - Ok(ResourceReadModel { - resource: snapshot.resource, - loan: snapshot.loan, - requests, - notices, - background_launch, - return_execution_mode, - return_window, - }) - }) - .collect() -} - -fn notices_for_loan( - conn: &Connection, - loan_id: LoanId, -) -> Result, ResourceStoreError> { - let mut statement = conn.prepare( - "SELECT id, loan_id, action_id, notice_json - FROM resource_supervisor_notices - WHERE loan_id = ?1 - ORDER BY id ASC", - )?; - let rows = statement.query_map([loan_id.as_uuid().to_string()], |row| { - Ok(decode_supervisor_notice_record(row)) - })?; - rows.map(|record| Ok(record??.0)).collect() -} - -/// Action identity that the loan still waits on, if any -pub(crate) fn open_action_id(loan: &Loan) -> Option { - match &loan.state { - LoanState::Active { phase } => phase_action_id(phase), - LoanState::NeedsAttention { action_id, .. } => Some(*action_id), - LoanState::Closed { .. } => None, - } -} - -fn phase_action_id(phase: &LoanPhase) -> Option { - match phase { - LoanPhase::AwaitingRelease { action_id, .. } - | LoanPhase::AwaitingReturn { action_id, .. } - | LoanPhase::Restoring { action_id, .. } => Some(*action_id), - LoanPhase::Serving { .. } => None, - } -} - -fn begin_resource_control( - conn: &mut Connection, - authority_machine: MachineId, - operation_id: Uuid, - request: &ResourceControlRequest, - attempt_id: DeliveryAttemptId, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let resource = select_authority_resource(&tx, authority_machine, request.resource_id)?; - - if let Some(saved) = saved_operation(&tx, operation_id)? { - if saved.request != *request { - return Err(ResourceControlError::Conflict); - } - let effect = replayed_effect(&tx, authority_machine, request, saved.attempt_id)?; - tx.commit()?; - return Ok(ResourceControlStart { - effect, - replayed: true, - }); - } - - if resource.state_revision != request.expected_revision { - return Err(ResourceControlError::StaleRevision { - current: resource.state_revision, - }); - } - let loan = select_non_closed_loan(&tx, request.resource_id)?; - let effect = match &request.action { - BrowserResourceAction::CancelQueued { request_id } => { - let queued = request_for_resource(&tx, request.resource_id, *request_id)?; - if let ResourceRequestState::Assigned { .. } = queued.state { - // the task layer owns an assigned request, started or not, so its - // cancellation goes through the task and releases the queue - return Err(ResourceControlError::NotAllowed(format!( - "request is assigned, not queued; cancel its task with `homebased task \ - cancel {}`, which works whether or not the task started and serves the \ - next request", - queued.task_id - ))); - } - if queued.state != ResourceRequestState::Queued { - return Err(ResourceControlError::NotAllowed(format!( - "request is {}, not queued", - request_state_name(&queued.state) - ))); - } - ResourceControlEffect::CancelQueued { request: queued } - } - BrowserResourceAction::MoveQueued { - request_id, - placement, - } => { - move_queued_request( - &tx, - authority_machine, - request.resource_id, - *request_id, - *placement, - )?; - let next_revision = resource.state_revision.next().ok_or_else(|| { - ResourceControlError::NotAllowed("resource revision cannot advance".into()) - })?; - if !swap_resource_revision::( - &tx, - authority_machine, - request.resource_id, - resource.state_revision, - next_revision, - )? { - return Err(ResourceControlError::StaleRevision { - current: resource.state_revision, - }); - } - ResourceControlEffect::Reordered - } - BrowserResourceAction::StopActive { task_id } => { - let active = active_request(&tx, loan.as_ref())?; - if active.task_id != *task_id { - return Err(ResourceControlError::NotAllowed( - "task is not the active command on this resource".into(), - )); - } - ResourceControlEffect::StopActive { request: active } - } - BrowserResourceAction::Renotify { notice_id } => { - reserve_explicit_attempt(&tx, loan.as_ref(), *notice_id, attempt_id)?; - ResourceControlEffect::Renotify { - notice_id: *notice_id, - attempt_id, - } - } - }; - let saved_attempt = match &effect { - ResourceControlEffect::Renotify { attempt_id, .. } => { - Some(attempt_id.as_uuid().to_string()) - } - ResourceControlEffect::CancelQueued { .. } - | ResourceControlEffect::StopActive { .. } - | ResourceControlEffect::Reordered => None, - }; - tx.execute( - "INSERT INTO resource_control_operations - (operation_id, resource_id, request_json, attempt_id) - VALUES (?1, ?2, ?3, ?4)", - params![ - operation_id.to_string(), - request.resource_id.as_uuid().to_string(), - serde_json::to_string(request)?, - saved_attempt, - ], - )?; - tx.commit()?; - Ok(ResourceControlStart { - effect, - replayed: false, - }) -} - -struct SavedOperation { - request: ResourceControlRequest, - attempt_id: Option, -} - -fn saved_operation( - tx: &Transaction<'_>, - operation_id: Uuid, -) -> Result, ResourceControlError> { - let row = tx - .query_row( - "SELECT request_json, attempt_id FROM resource_control_operations - WHERE operation_id = ?1", - [operation_id.to_string()], - |row| Ok((row.get::<_, String>(0)?, row.get::<_, Option>(1)?)), - ) - .optional()?; - let Some((request_json, attempt_id)) = row else { - return Ok(None); - }; - let attempt_id = attempt_id - .map(|value| { - value.parse::().map_err(|_| { - ResourceControlError::NotAllowed( - "saved control operation has an invalid attempt identity".into(), - ) - }) - }) - .transpose()?; - Ok(Some(SavedOperation { - request: serde_json::from_str(&request_json)?, - attempt_id, - })) -} - -// a replay repeats only effects that are idempotent at their owner; a renotify -// replay names its saved attempt and never reserves another one -fn replayed_effect( - tx: &Transaction<'_>, - authority_machine: MachineId, - request: &ResourceControlRequest, - attempt_id: Option, -) -> Result { - match &request.action { - BrowserResourceAction::CancelQueued { request_id } => { - Ok(ResourceControlEffect::CancelQueued { - request: request_for_resource(tx, request.resource_id, *request_id)?, - }) - } - BrowserResourceAction::MoveQueued { .. } => Ok(ResourceControlEffect::Reordered), - BrowserResourceAction::StopActive { task_id } => { - let request = - requests_for_resource_for_authority(tx, authority_machine, request.resource_id)? - .into_iter() - .find(|saved| saved.task_id == *task_id) - .ok_or_else(|| { - ResourceControlError::NotAllowed( - "saved stop operation no longer names a resource request".into(), - ) - })?; - Ok(ResourceControlEffect::StopActive { request }) - } - BrowserResourceAction::Renotify { notice_id } => { - let attempt_id = attempt_id.ok_or_else(|| { - ResourceControlError::NotAllowed( - "saved renotify operation has no attempt identity".into(), - ) - })?; - Ok(ResourceControlEffect::Renotify { - notice_id: *notice_id, - attempt_id, - }) - } - } -} - -fn move_queued_request( - tx: &Transaction<'_>, - authority_machine: MachineId, - resource_id: ResourceId, - request_id: RequestId, - placement: QueuePlacement, -) -> Result<(), ResourceControlError> { - let request = request_for_resource(tx, resource_id, request_id)?; - if request.state != ResourceRequestState::Queued { - return Err(ResourceControlError::NotAllowed(format!( - "request is {}, not queued", - request_state_name(&request.state) - ))); - } - - if let QueuePlacement::Before { - request_id: anchor_id, - } - | QueuePlacement::After { - request_id: anchor_id, - } = placement - { - if anchor_id == request_id { - return Err(ResourceControlError::NotAllowed( - "a request cannot be its own queue anchor".into(), - )); - } - let anchor = request_for_resource(tx, resource_id, anchor_id)?; - if anchor.state != ResourceRequestState::Queued { - return Err(ResourceControlError::NotAllowed(format!( - "anchor request is {}, not queued", - request_state_name(&anchor.state) - ))); - } - } - - let mut order = requests_for_resource_for_authority(tx, authority_machine, resource_id)? - .into_iter() - .filter(|saved| saved.state == ResourceRequestState::Queued) - .map(|saved| saved.request_id) - .collect::>(); - let old_index = queue_position( - &order, - request_id, - "request is no longer queued on this resource", - )?; - order.remove(old_index); - let new_index = match placement { - QueuePlacement::Front => 0, - QueuePlacement::Back => order.len(), - QueuePlacement::Before { - request_id: anchor_id, - } => queue_position( - &order, - anchor_id, - "anchor request is no longer queued on this resource", - )?, - QueuePlacement::After { - request_id: anchor_id, - } => { - queue_position( - &order, - anchor_id, - "anchor request is no longer queued on this resource", - )? + 1 - } - }; - order.insert(new_index, request_id); - rewrite_queued_request_ranks(tx, resource_id, &order)?; - Ok(()) -} - -fn queue_position( - order: &[RequestId], - request_id: RequestId, - missing: &str, -) -> Result { - order - .iter() - .position(|saved| *saved == request_id) - .ok_or_else(|| ResourceControlError::NotAllowed(missing.into())) -} - -fn request_for_resource( - conn: &Connection, - resource_id: ResourceId, - request_id: RequestId, -) -> Result { - select_request_by_id(conn, request_id)? - .filter(|request| request.resource_id == resource_id) - .ok_or_else(|| { - ResourceControlError::NotAllowed("request does not belong to this resource".into()) - }) -} - -fn active_request( - conn: &Connection, - loan: Option<&Loan>, -) -> Result { - let not_serving = - || ResourceControlError::NotAllowed("no command task is active on this resource".into()); - let loan = loan.ok_or_else(not_serving)?; - let LoanState::Active { - phase: LoanPhase::Serving { - current_request_id, .. - }, - } = &loan.state - else { - return Err(not_serving()); - }; - let request = select_request_by_id(conn, *current_request_id)?.ok_or_else(not_serving)?; - match request.state { - ResourceRequestState::Assigned { loan_id } if loan_id == loan.id => Ok(request), - _ => Err(not_serving()), - } -} - -fn reserve_explicit_attempt( - tx: &Transaction<'_>, - loan: Option<&Loan>, - notice_id: NoticeId, - attempt_id: DeliveryAttemptId, -) -> Result<(), ResourceControlError> { - let (mut notice, old_json) = select_supervisor_notice_record(tx, notice_id) - .map_err(SupervisorNoticeStoreError::from)? - .ok_or_else(|| ResourceControlError::NotAllowed("notice not found".into()))?; - let awaited = - loan.is_some_and(|loan| loan.id == notice.loan_id && loan.state.awaits_notice(¬ice)); - if !awaited { - return Err(ResourceControlError::NotAllowed( - "the notice action is no longer pending".into(), - )); - } - let attempts = match ¬ice.delivery { - SupervisorNoticeDelivery::Failed { attempts, .. } => *attempts, - SupervisorNoticeDelivery::Delivered { .. } => { - return Err(ResourceControlError::NotAllowed( - "the notice was already delivered".into(), - )); - } - SupervisorNoticeDelivery::Pending { .. } - | SupervisorNoticeDelivery::RetryPending { .. } - | SupervisorNoticeDelivery::Sending { .. } => { - return Err(ResourceControlError::NotAllowed( - "automatic delivery has not failed for this notice".into(), - )); - } - }; - // one explicit attempt past the exhausted budget; a failure settles back to failed - notice.delivery = SupervisorNoticeDelivery::Sending { - attempt_id, - attempt: attempts.saturating_add(1), - }; - update_supervisor_notice_cas(tx, ¬ice, &old_json)?; - Ok(()) -} - -fn replace_resource_supervisor( - conn: &mut Connection, - authority_machine: MachineId, - resource_id: ResourceId, - expected_revision: ResourceRevision, - supervisor: SupervisorAddress, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let mut resource = select_authority_resource(&tx, authority_machine, resource_id)?; - // an exact retry after a lost response finds the requested assignment already saved - if resource.supervisor == supervisor { - tx.commit()?; - return Ok(SupervisorReplacement { - resource, - retargeted: Vec::new(), - }); - } - if resource.state_revision != expected_revision { - return Err(ResourceControlError::StaleRevision { - current: resource.state_revision, - }); - } - - let previous = resource.assignment_revision; - let next = AssignmentRevision::new(previous.get() + 1); - let updated = tx.execute( - "UPDATE resources - SET supervisor_machine = ?1, supervisor_thread = ?2, assignment_revision = ?3 - WHERE id = ?4 AND assignment_revision = ?5 AND state_revision = ?6", - params![ - supervisor.machine.as_uuid().to_string(), - supervisor.thread.to_string(), - sqlite_revision(next.get())?, - resource_id.as_uuid().to_string(), - sqlite_revision(previous.get())?, - sqlite_revision(expected_revision.get())?, - ], - )?; - if updated != 1 { - return Err(ResourceControlError::StaleRevision { - current: resource.state_revision, - }); - } - - let mut retargeted = Vec::new(); - if let Some(loan) = select_non_closed_loan(&tx, resource_id)? { - for notice in notices_for_loan(&tx, loan.id)? { - if matches!(notice.delivery, SupervisorNoticeDelivery::Delivered { .. }) { - continue; - } - retargeted.push(retarget_supervisor_notice_in_transaction( - &tx, - notice.id, - notice.assignment_revision, - supervisor, - next, - )?); - } - } - tx.commit()?; - - resource.supervisor = supervisor; - resource.assignment_revision = next; - Ok(SupervisorReplacement { - resource, - retargeted, - }) -} - -fn sqlite_revision(value: u64) -> Result { - i64::try_from(value).map_err(|_| { - ResourceControlError::NotAllowed("resource revision exceeds SQLite integer range".into()) - }) -} - -/// Stable snake-case name of a request state for messages -pub(crate) fn request_state_name(state: &ResourceRequestState) -> &'static str { - match state { - ResourceRequestState::Queued => "queued", - ResourceRequestState::Assigned { .. } => "assigned", - ResourceRequestState::Finished { .. } => "finished", - ResourceRequestState::CancelledBeforeLaunch => "cancelled_before_launch", - ResourceRequestState::Rejected { .. } => "rejected", - } -} diff --git a/src/store/resource/controls/test_support.rs b/src/store/resource/controls/test_support.rs deleted file mode 100644 index 4c9429c..0000000 --- a/src/store/resource/controls/test_support.rs +++ /dev/null @@ -1,56 +0,0 @@ -//! Seeding helpers for tests that drive resource controls through a running daemon - -use std::path::Path; - -use rusqlite::params; - -use crate::resource::{Loan, SupervisorNotice}; -use crate::store::Store; - -impl Store { - /// Insert one non-closed loan and one notice exactly as given, beside a running StoreActor - pub(crate) fn seed_loan_notice_for_test( - database: &Path, - loan: &Loan, - notice: &SupervisorNotice, - ) { - let store = Self::open(database).unwrap(); - store - .conn - .execute( - "INSERT INTO loans (id, resource_id, state_json) VALUES (?1, ?2, ?3)", - params![ - loan.id.as_uuid().to_string(), - loan.resource_id.as_uuid().to_string(), - serde_json::to_string(&loan.state).unwrap(), - ], - ) - .unwrap(); - store - .conn - .execute( - "INSERT INTO resource_supervisor_notices (id, loan_id, action_id, notice_json) - VALUES (?1, ?2, ?3, ?4)", - params![ - notice.id.as_uuid().to_string(), - notice.loan_id.as_uuid().to_string(), - notice.action_id.as_uuid().to_string(), - serde_json::to_string(notice).unwrap(), - ], - ) - .unwrap(); - } - - /// Count committed resource control operations beside a running StoreActor - pub(crate) fn resource_control_operation_count_for_test(database: &Path) -> i64 { - Self::open(database) - .unwrap() - .conn - .query_row( - "SELECT COUNT(*) FROM resource_control_operations", - [], - |row| row.get(0), - ) - .unwrap() - } -} diff --git a/src/store/resource/initial_idle.rs b/src/store/resource/initial_idle.rs deleted file mode 100644 index 86d9aba..0000000 --- a/src/store/resource/initial_idle.rs +++ /dev/null @@ -1,224 +0,0 @@ -//! Authority-owned initial idle attestation for a resource with no history -//! -//! One IMMEDIATE transaction checks the attestation against the current -//! resource revision and every record that could hold or release the GPU: a -//! registered task, a loan, a first background launch, and an operator -//! attestation about an ended task. With none of them, it saves one receipt and -//! advances the revision. Queued work then serves through the normal idle -//! reconciliation, which reads the receipt as the idle boundary only while the -//! resource still has no loan and no first background launch. A refusal writes -//! nothing, and an exact retry returns the saved receipt - -use rusqlite::{Connection, OptionalExtension, TransactionBehavior, params}; - -use super::background::{ - AdvanceError, advance_resource_on, latest_launch_request_on, latest_loan_on, -}; -use crate::machine::MachineId; -use crate::resource::initial_idle::{ - InitialIdleAttestation, InitialIdleReceipt, InitialIdleRefusal, InitialIdleResolution, - ResourceHistory, -}; -use crate::resource::operator_release::OperatorAttestationId; -use crate::resource::store::{ResourceStoreError, select_resource}; -use crate::resource::{Resource, ResourceId}; -use crate::store::Store; - -/// Why an initial idle attestation was refused or could not be evaluated -#[derive(Debug, thiserror::Error)] -pub(crate) enum InitialIdleError { - /// The authority refused the attestation and wrote nothing - #[error(transparent)] - Refused(#[from] InitialIdleRefusal), - /// Resource storage failed or stored data is invalid - #[error(transparent)] - Resource(#[from] ResourceStoreError), - /// A receipt could not be encoded or decoded - #[error("initial idle attestation encoding failed: {0}")] - Encoding(#[from] serde_json::Error), - /// A concurrent change invalidated the revision compare-and-set - #[error("resource state changed during the initial idle attestation")] - Changed, - /// SQLite failed - #[error("initial idle attestation storage error: {0}")] - Storage(#[from] rusqlite::Error), -} - -impl Store { - /// Save one initial idle attestation for a resource with no history - /// - /// `authority_machine` is the local daemon machine. The attestation must - /// name it, and it must be the resource authority - pub(crate) fn attest_initial_idle_for_authority( - &mut self, - authority_machine: MachineId, - attestation: InitialIdleAttestation, - ) -> Result { - attest_initial_idle(&mut self.conn, authority_machine, attestation) - } -} - -fn attest_initial_idle( - conn: &mut Connection, - authority_machine: MachineId, - attestation: InitialIdleAttestation, -) -> Result { - attestation.validate()?; - if attestation.authority_machine != authority_machine { - return Err(InitialIdleRefusal::WrongAuthority { - expected: authority_machine, - found: attestation.authority_machine, - } - .into()); - } - - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - if let Some(receipt) = receipt_by_operation_on(&tx, attestation.operation_id)? { - if receipt.attestation != attestation { - return Err(InitialIdleRefusal::ConflictingRetry { - operation_id: attestation.operation_id, - } - .into()); - } - tx.commit()?; - return Ok(InitialIdleResolution { - receipt, - replayed: true, - }); - } - - let resource = select_resource(&tx, attestation.resource_id)? - .ok_or(InitialIdleRefusal::ResourceNotFound)?; - if resource.authority_machine() != attestation.authority_machine { - return Err(InitialIdleRefusal::WrongAuthority { - expected: resource.authority_machine(), - found: attestation.authority_machine, - } - .into()); - } - if let Some(saved) = receipt_by_resource_on(&tx, resource.id)? { - return Err(InitialIdleRefusal::AlreadyAttested { - operation_id: saved.attestation.operation_id, - } - .into()); - } - if resource.state_revision != attestation.expected_state_revision { - return Err(InitialIdleRefusal::StaleRevision { - expected: attestation.expected_state_revision, - actual: resource.state_revision, - } - .into()); - } - if let Some(history) = resource_history_on(&tx, &resource)? { - return Err(InitialIdleRefusal::HistoryExists { history }.into()); - } - - let state_revision = - advance_resource_on(&tx, &resource, None).map_err(|error| match error { - AdvanceError::Exhausted => { - InitialIdleError::Refused(InitialIdleRefusal::RevisionExhausted { - revision: resource.state_revision, - }) - } - AdvanceError::Changed => InitialIdleError::Changed, - AdvanceError::Storage(error) => InitialIdleError::Storage(error), - })?; - let receipt = InitialIdleReceipt { - attestation, - state_revision, - }; - tx.execute( - "INSERT INTO resource_initial_idle_attestations (operation_id, resource_id, receipt_json) - VALUES (?1, ?2, ?3)", - params![ - receipt.attestation.operation_id.as_uuid().to_string(), - resource.id.as_uuid().to_string(), - serde_json::to_string(&receipt)?, - ], - )?; - tx.commit()?; - - Ok(InitialIdleResolution { - receipt, - replayed: false, - }) -} - -/// First record that already decides whether the resource's GPU is idle -fn resource_history_on( - conn: &Connection, - resource: &Resource, -) -> Result, ResourceStoreError> { - if resource.registered_background_task.is_some() { - return Ok(Some(ResourceHistory::RegisteredTask)); - } - if latest_loan_on(conn, resource.id)?.is_some() { - return Ok(Some(ResourceHistory::Loan)); - } - if latest_launch_request_on(conn, resource.id)?.is_some() { - return Ok(Some(ResourceHistory::BackgroundLaunch)); - } - let attested: bool = conn.query_row( - "SELECT EXISTS(SELECT 1 FROM resource_operator_attestations WHERE resource_id = ?1)", - [resource.id.as_uuid().to_string()], - |row| row.get(0), - )?; - Ok(attested.then_some(ResourceHistory::OperatorAttestation)) -} - -/// Initial idle attestation that is the current idle boundary of a resource -/// -/// The receipt counts only while the resource still has no registered task, -/// loan, or first background launch; every later record decides instead -pub(crate) fn initial_idle_boundary_on( - conn: &Connection, - resource: &Resource, -) -> Result, ResourceStoreError> { - let Some(receipt) = receipt_by_resource_on(conn, resource.id)? else { - return Ok(None); - }; - if receipt.attestation.resource_id != resource.id - || receipt.attestation.authority_machine != resource.authority_machine() - || resource_history_on(conn, resource)?.is_some() - { - return Ok(None); - } - Ok(Some(receipt.attestation.operation_id)) -} - -fn receipt_by_operation_on( - conn: &Connection, - operation_id: OperatorAttestationId, -) -> Result, ResourceStoreError> { - receipt_where(conn, "operation_id", &operation_id.as_uuid().to_string()) -} - -fn receipt_by_resource_on( - conn: &Connection, - resource_id: ResourceId, -) -> Result, ResourceStoreError> { - receipt_where(conn, "resource_id", &resource_id.as_uuid().to_string()) -} - -fn receipt_where( - conn: &Connection, - column: &str, - value: &str, -) -> Result, ResourceStoreError> { - let saved: Option = conn - .query_row( - &format!( - "SELECT receipt_json FROM resource_initial_idle_attestations WHERE {column} = ?1" - ), - [value], - |row| row.get(0), - ) - .optional()?; - saved - .map(|json| { - serde_json::from_str(&json).map_err(|error| { - ResourceStoreError::corrupt("initial idle attestation receipt", error) - }) - }) - .transpose() -} diff --git a/src/store/resource/operator_release.rs b/src/store/resource/operator_release.rs deleted file mode 100644 index fccde94..0000000 --- a/src/store/resource/operator_release.rs +++ /dev/null @@ -1,1010 +0,0 @@ -//! Authority-owned operator attestation for an ended trainer with no release proof -//! -//! One IMMEDIATE transaction checks the exact attestation against the current -//! resource revision, registration, loan or launch reservation, task row, and -//! the resource launch that bound the task. It then saves one receipt with the -//! evidence snapshot and commits the queue or loan transition that follows. A -//! refusal writes nothing. An exact retry returns the saved receipt without -//! these checks, so a later transition cannot turn it into a refusal -//! -//! With an AwaitingRelease loan the release action closes into Serving or -//! AwaitingReturn and keeps the return obligation, as a proven release would -//! With no loan, for the registered trainer or for a first background launch -//! that ended before registration, the registration clears. The next queued -//! request then serves from an idle loan in the same transaction, or the -//! receipt becomes the saved idle boundary that a later reconciliation or first -//! launch reads. With a Restoring loan whose direct-segment return task ended -//! before its confirmed start, or whose native foreground return task ended or -//! was lost, the loan closes with the attested end, as a supervisor resolution -//! of a proven end would. The next queued request then serves from an idle -//! loan, or the closure is the saved idle boundary. The attestation never -//! becomes process-group exit evidence; the task row keeps its saved state - -use rusqlite::{Connection, OptionalExtension, Transaction, TransactionBehavior, params}; - -use super::background::{ - AdvanceError, BackgroundLaunchPhase, advance_resource_on, current_launch_and_predecessor_on, - first_launch_of_registered_trainer_on, latest_launch_request_on, latest_loan_on, - open_idle_serving_loan_on, pending_background_launch_on, -}; -use super::restore::{direct_segment_return_of_registered_trainer_on, restoring_return_of_mode_on}; -use super::{LoanActionPhase, replace_loan_in_action_phase_on}; -use crate::domain::{TaskId, TaskState, Workload}; -use crate::machine::MachineId; -use crate::resource::operator_release::{ - AttestedTrainerAssociation, AttestedTrainerEnd, AttestedTrainerLaunch, OperatorAttestationId, - OperatorGpuFreeAttestation, OperatorGpuFreeEvidence, OperatorGpuFreeOutcome, - OperatorGpuFreeReceipt, OperatorGpuFreeRefusal, OperatorGpuFreeResolution, - OperatorStateBinding, -}; -use crate::resource::ownership_lock::TrainerRequestDigest; -use crate::resource::store::{ - ResourceStoreError, SupervisorNoticeStoreError, insert_supervisor_notice_in_transaction, - next_queued_request_for_authority, select_non_closed_loan, select_resource, - select_supervisor_notice_record_by_action, -}; -use crate::resource::{ - ActionId, IdleBoundaryProof, Loan, LoanClosure, LoanId, LoanPhase, LoanState, NoticeId, - Resource, ResourceRequestState, ResourceRevision, ReturnContext, ReturnExecutionMode, - ServingReleaseProvenance, SupervisorNotice, SupervisorNoticeDelivery, SupervisorNoticePayload, -}; -use crate::store::{BackgroundLaunchError, ReturnDecisionError, Store}; -use crate::submission::{NormalizedSpecSha256, RequestId}; - -/// Why an operator attestation was refused or could not be evaluated -#[derive(Debug, thiserror::Error)] -pub(crate) enum OperatorGpuFreeError { - /// The authority refused the attestation and wrote nothing - #[error(transparent)] - Refused(#[from] OperatorGpuFreeRefusal), - /// Resource storage failed or stored data is invalid - #[error(transparent)] - Resource(#[from] ResourceStoreError), - /// A durable supervisor notice could not be saved - #[error(transparent)] - Notice(#[from] SupervisorNoticeStoreError), - /// The first background launch records could not be read - #[error(transparent)] - Launch(#[from] BackgroundLaunchError), - /// The return decision records could not be read - #[error(transparent)] - Return(#[from] ReturnDecisionError), - /// A task row could not be read - #[error(transparent)] - Task(#[from] crate::error::AppError), - /// A receipt could not be encoded or decoded - #[error("operator attestation encoding failed: {0}")] - Encoding(#[from] serde_json::Error), - /// A concurrent change invalidated a compare-and-set update - #[error("resource state changed during the operator attestation")] - Changed, - /// SQLite failed - #[error("operator attestation storage error: {0}")] - Storage(#[from] rusqlite::Error), -} - -/// Saved attestation that is the current idle boundary of its resource -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) struct OperatorIdleBoundary { - /// Attestation whose receipt holds the observation and evidence - pub(crate) operation_id: OperatorAttestationId, - /// Registered trainer or first launch task that the attestation cleared - pub(crate) task_id: TaskId, - /// Latest first background launch when the attestation committed - pub(crate) preceding_launch: Option, -} - -impl Store { - /// Commit one operator attestation for an ended trainer and its queue or loan transition - /// - /// `authority_machine` is the local daemon machine. The attestation must name - /// it, and it must be the resource authority - pub(crate) fn attest_trainer_gpu_free_for_authority( - &mut self, - authority_machine: MachineId, - attestation: OperatorGpuFreeAttestation, - ) -> Result { - attest_trainer_gpu_free(&mut self.conn, authority_machine, attestation) - } -} - -fn attest_trainer_gpu_free( - conn: &mut Connection, - authority_machine: MachineId, - attestation: OperatorGpuFreeAttestation, -) -> Result { - attestation.validate()?; - if attestation.authority_machine != authority_machine { - return Err(OperatorGpuFreeRefusal::WrongAuthority { - expected: authority_machine, - found: attestation.authority_machine, - } - .into()); - } - - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - if let Some(receipt) = saved_receipt_on(&tx, attestation.operation_id)? { - if receipt.attestation != attestation { - return Err(OperatorGpuFreeRefusal::ConflictingRetry { - operation_id: attestation.operation_id, - } - .into()); - } - tx.commit()?; - return Ok(OperatorGpuFreeResolution { - receipt, - replayed: true, - }); - } - - let resource = current_resource(&tx, &attestation)?; - let preceding_loan = latest_loan_on(&tx, resource.id)?.map(|loan| loan.id); - let preceding_launch = latest_launch_request_on(&tx, resource.id)?; - let (evidence, (state_revision, outcome)) = match attestation.state_binding { - OperatorStateBinding::NoLoan | OperatorStateBinding::AwaitingRelease { .. } => { - let release = bound_release_action(&tx, &resource, &attestation)?; - let evidence = registered_trainer_evidence(&tx, &resource, attestation.task_id)?; - let transition = match release { - Some((loan, action_id)) => resolve_release_action( - &tx, - &resource, - &attestation, - &evidence, - loan, - action_id, - )?, - None => resolve_idle(&tx, &resource, &attestation)?, - }; - (evidence, transition) - } - OperatorStateBinding::FirstBackgroundLaunch { request_id } => { - let evidence = bound_first_launch(&tx, &resource, &attestation, request_id)?; - (evidence, resolve_idle(&tx, &resource, &attestation)?) - } - OperatorStateBinding::RestoringReturn { loan_id, action_id } - | OperatorStateBinding::RestoringForegroundReturn { loan_id, action_id } => { - let (loan, return_context, evidence) = - bound_restoring_return(&tx, &resource, &attestation, loan_id, action_id)?; - let transition = resolve_restoring( - &tx, - &resource, - &attestation, - loan, - action_id, - return_context, - )?; - (evidence, transition) - } - }; - let receipt = OperatorGpuFreeReceipt { - attestation, - evidence, - state_revision, - outcome, - }; - tx.execute( - "INSERT INTO resource_operator_attestations ( - operation_id, resource_id, task_id, preceding_loan, preceding_launch, receipt_json - ) VALUES (?1, ?2, ?3, ?4, ?5, ?6)", - params![ - receipt.attestation.operation_id.as_uuid().to_string(), - resource.id.as_uuid().to_string(), - receipt.attestation.task_id.to_string(), - preceding_loan.map(|loan| loan.as_uuid().to_string()), - preceding_launch.map(|request| request.0.to_string()), - serde_json::to_string(&receipt)?, - ], - )?; - tx.commit()?; - - Ok(OperatorGpuFreeResolution { - receipt, - replayed: false, - }) -} - -/// Read the resource and compare its authority and revision -/// -/// A binding that names the registered trainer also requires that registration -fn current_resource( - conn: &Connection, - attestation: &OperatorGpuFreeAttestation, -) -> Result { - let resource = select_resource(conn, attestation.resource_id)? - .ok_or(OperatorGpuFreeRefusal::ResourceNotFound)?; - if resource.authority_machine() != attestation.authority_machine { - return Err(OperatorGpuFreeRefusal::WrongAuthority { - expected: resource.authority_machine(), - found: attestation.authority_machine, - } - .into()); - } - if resource.state_revision != attestation.expected_state_revision { - return Err(OperatorGpuFreeRefusal::StaleRevision { - expected: attestation.expected_state_revision, - actual: resource.state_revision, - } - .into()); - } - if attestation.state_binding.names_registered_trainer() - && resource.registered_background_task != Some(attestation.task_id) - { - return Err(OperatorGpuFreeRefusal::NotRegisteredTrainer { - task_id: attestation.task_id, - registered: resource.registered_background_task, - } - .into()); - } - - Ok(resource) -} - -/// Compare the named state binding with the current loan -/// -/// Returns the AwaitingRelease loan and its action, or `None` for no loan. Any -/// later phase can already run other GPU work, so it is never overridden -fn bound_release_action( - conn: &Connection, - resource: &Resource, - attestation: &OperatorGpuFreeAttestation, -) -> Result, OperatorGpuFreeError> { - let task_id = attestation.task_id; - let current = select_non_closed_loan(conn, resource.id)?; - let (loan, loan_id, action_id) = match (attestation.state_binding, current) { - (OperatorStateBinding::NoLoan, None) => { - // a pending first launch would replace this registration once it starts - if pending_background_launch_on(conn, resource)?.is_some() { - return Err(OperatorGpuFreeRefusal::InconsistentHistory { task_id }.into()); - } - return Ok(None); - } - (OperatorStateBinding::AwaitingRelease { loan_id, action_id }, Some(loan)) - if loan.id == loan_id => - { - (loan, loan_id, action_id) - } - ( - OperatorStateBinding::NoLoan - | OperatorStateBinding::AwaitingRelease { .. } - | OperatorStateBinding::FirstBackgroundLaunch { .. } - | OperatorStateBinding::RestoringReturn { .. } - | OperatorStateBinding::RestoringForegroundReturn { .. }, - current, - ) => { - return Err(OperatorGpuFreeRefusal::LoanStateChanged { - current_loan: current.map(|loan| loan.id), - } - .into()); - } - }; - - let LoanState::Active { - phase: - LoanPhase::AwaitingRelease { - action_id: saved_action, - observed_background_task, - .. - }, - } = &loan.state - else { - return Err(OperatorGpuFreeRefusal::LoanNotAwaitingRelease { loan_id }.into()); - }; - if *saved_action != action_id || *observed_background_task != task_id { - return Err(OperatorGpuFreeRefusal::LoanStateChanged { - current_loan: Some(loan_id), - } - .into()); - } - - let notice = select_supervisor_notice_record_by_action(conn, action_id) - .map_err(ResourceStoreError::from)?; - let notice_matches = notice.is_some_and(|(notice, _)| { - notice.loan_id == loan_id - && notice.action_id == action_id - && notice.payload == SupervisorNoticePayload::ReleaseRequired { task_id } - }); - if !notice_matches { - return Err(OperatorGpuFreeRefusal::InvalidReleaseNotice { action_id }.into()); - } - - Ok(Some((loan, action_id))) -} - -/// Snapshot the registered trainer's end, registering launch, digest, and association -/// -/// The task must have ended or been lost. Its accepted identity and callback -/// route must still match the resource launch that registered it -fn registered_trainer_evidence( - conn: &Connection, - resource: &Resource, - task_id: TaskId, -) -> Result { - let trainer_end = ended_trainer(conn, task_id)?; - let (trainer_launch, normalized_spec_sha256) = if let Some((request_id, digest)) = - first_launch_of_registered_trainer_on(conn, resource, task_id)? - { - ( - AttestedTrainerLaunch::FirstBackgroundLaunch { request_id }, - digest, - ) - } else if let Some((action_id, request_id, digest)) = - direct_segment_return_of_registered_trainer_on(conn, resource, task_id)? - { - ( - AttestedTrainerLaunch::DirectSegmentReturn { - action_id, - request_id, - }, - digest, - ) - } else { - return Err(OperatorGpuFreeRefusal::TrainerLaunchUnproven { task_id }.into()); - }; - - evidence( - conn, - resource, - task_id, - trainer_end, - trainer_launch, - normalized_spec_sha256, - ) -} - -/// Compare the named first background launch with the latest launch and snapshot its evidence -/// -/// No loan may reserve the resource. The launch must be the latest one, bind -/// the named task, and have ended before registration with no automatic -/// release proof. The registration must still be the one that the launch -/// replaced, so a later registration can never be cleared through it -fn bound_first_launch( - conn: &Connection, - resource: &Resource, - attestation: &OperatorGpuFreeAttestation, - request_id: RequestId, -) -> Result { - let task_id = attestation.task_id; - if let Some(loan) = select_non_closed_loan(conn, resource.id)? { - return Err(OperatorGpuFreeRefusal::LoanStateChanged { - current_loan: Some(loan.id), - } - .into()); - } - let current = current_launch_and_predecessor_on(conn, resource)?; - let current_launch = current.as_ref().map(|(view, _)| view.request_id); - let Some((view, replaces_task)) = current.filter(|(view, _)| view.request_id == request_id) - else { - return Err(OperatorGpuFreeRefusal::LaunchNotAwaitingRelease { - request_id, - current_launch, - } - .into()); - }; - if view.task_id != task_id { - return Err(OperatorGpuFreeRefusal::NotBoundTask { - task_id, - bound: view.task_id, - } - .into()); - } - // a queued or running launch task still owns the GPU - let trainer_end = ended_trainer(conn, task_id)?; - match view.phase { - _ if view.awaits_operator_release() => {} - BackgroundLaunchPhase::IdentityMismatch => { - return Err(OperatorGpuFreeRefusal::TrainerLaunchUnproven { task_id }.into()); - } - BackgroundLaunchPhase::Superseded - | BackgroundLaunchPhase::Queued - | BackgroundLaunchPhase::StartedUnregistered - | BackgroundLaunchPhase::Registered - | BackgroundLaunchPhase::EndedBeforeRegistration { .. } => { - return Err(OperatorGpuFreeRefusal::LaunchNotAwaitingRelease { - request_id, - current_launch, - } - .into()); - } - } - if resource.registered_background_task != replaces_task { - return Err(OperatorGpuFreeRefusal::InconsistentHistory { task_id }.into()); - } - let Some((launch_request, digest)) = - first_launch_of_registered_trainer_on(conn, resource, task_id)? - else { - return Err(OperatorGpuFreeRefusal::TrainerLaunchUnproven { task_id }.into()); - }; - if launch_request != request_id { - return Err(OperatorGpuFreeRefusal::InconsistentHistory { task_id }.into()); - } - - evidence( - conn, - resource, - task_id, - trainer_end, - AttestedTrainerLaunch::FirstBackgroundLaunch { request_id }, - digest, - ) -} - -/// Compare the named Restoring loan with the current loan and snapshot its task evidence -/// -/// The loan must still be Restoring under the named action and bind the named -/// task, and the saved decision must bind that task with an execution mode that -/// the binding names. The foreground binding names both modes that hold the -/// loan while they run: native foreground and container. Such a task is never -/// registered, so a registration that names it is inconsistent history. -/// Returns the loan, its return context, and the evidence snapshot -fn bound_restoring_return( - conn: &Connection, - resource: &Resource, - attestation: &OperatorGpuFreeAttestation, - loan_id: LoanId, - action_id: ActionId, -) -> Result<(Loan, ReturnContext, OperatorGpuFreeEvidence), OperatorGpuFreeError> { - let task_id = attestation.task_id; - let modes: &[ReturnExecutionMode] = match attestation.state_binding { - OperatorStateBinding::RestoringReturn { .. } => { - &[ReturnExecutionMode::DirectSegmentTrainer] - } - OperatorStateBinding::RestoringForegroundReturn { .. } => &[ - ReturnExecutionMode::NativeForeground, - ReturnExecutionMode::Container, - ], - OperatorStateBinding::NoLoan - | OperatorStateBinding::AwaitingRelease { .. } - | OperatorStateBinding::FirstBackgroundLaunch { .. } => { - return Err(OperatorGpuFreeRefusal::LoanNotRestoring { loan_id }.into()); - } - }; - let loan = match select_non_closed_loan(conn, resource.id)? { - Some(loan) if loan.id == loan_id => loan, - current => { - return Err(OperatorGpuFreeRefusal::LoanStateChanged { - current_loan: current.map(|loan| loan.id), - } - .into()); - } - }; - let LoanState::Active { - phase: - LoanPhase::Restoring { - action_id: saved_action, - return_context, - resume_task_id, - }, - } = &loan.state - else { - return Err(OperatorGpuFreeRefusal::LoanNotRestoring { loan_id }.into()); - }; - if *saved_action != action_id { - return Err(OperatorGpuFreeRefusal::LoanStateChanged { - current_loan: Some(loan_id), - } - .into()); - } - if *resume_task_id != task_id { - return Err(OperatorGpuFreeRefusal::NotBoundTask { - task_id, - bound: *resume_task_id, - } - .into()); - } - if modes.iter().all(|mode| mode.holds_loan_while_running()) - && resource.registered_background_task == Some(task_id) - { - return Err(OperatorGpuFreeRefusal::InconsistentHistory { task_id }.into()); - } - let return_context = return_context.clone(); - // a queued or running return task still owns the GPU - let trainer_end = ended_trainer(conn, task_id)?; - let mut bound = None; - for mode in modes { - if let Some(found) = - restoring_return_of_mode_on(conn, resource, &loan, action_id, task_id, *mode)? - { - bound = Some((*mode, found)); - break; - } - } - let Some((mode, (request_id, digest))) = bound else { - return Err(OperatorGpuFreeRefusal::TrainerLaunchUnproven { task_id }.into()); - }; - let trainer_launch = match mode { - ReturnExecutionMode::DirectSegmentTrainer => AttestedTrainerLaunch::DirectSegmentReturn { - action_id, - request_id, - }, - ReturnExecutionMode::NativeForeground => AttestedTrainerLaunch::NativeForegroundReturn { - action_id, - request_id, - }, - ReturnExecutionMode::Container => AttestedTrainerLaunch::ContainerReturn { - action_id, - request_id, - }, - }; - let evidence = evidence(conn, resource, task_id, trainer_end, trainer_launch, digest)?; - - Ok((loan, return_context, evidence)) -} - -/// Read the task end, refusing a queued or running task -fn ended_trainer( - conn: &Connection, - task_id: TaskId, -) -> Result { - let row = crate::store::task_by_id_on(conn, task_id)? - .ok_or(OperatorGpuFreeRefusal::TaskMissing { task_id })?; - match &row.state { - TaskState::Queued | TaskState::Running { .. } => { - Err(OperatorGpuFreeRefusal::TaskNotEnded { - task_id, - state: row.status(), - } - .into()) - } - TaskState::Finished { reason } => Ok(AttestedTrainerEnd::Finished { - outcome: reason.clone(), - process_group_exit: row.process_group_exit_evidence(), - container_exit: matches!(row.workload, Workload::Container(_)) - .then(|| row.container_exit_evidence.clone()), - }), - TaskState::Lost => Ok(AttestedTrainerEnd::Lost), - } -} - -/// Complete the evidence snapshot with the task's trainer-attempt association -fn evidence( - conn: &Connection, - resource: &Resource, - task_id: TaskId, - trainer_end: AttestedTrainerEnd, - trainer_launch: AttestedTrainerLaunch, - normalized_spec_sha256: NormalizedSpecSha256, -) -> Result { - let association: Option<(String, Option)> = conn - .query_row( - "SELECT resource_id, json_extract(association_json, '$.request_sha256') - FROM trainer_attempt_associations WHERE task_id = ?1", - [task_id.to_string()], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .optional()?; - let trainer_association = match association { - None => AttestedTrainerAssociation::Missing, - Some((saved_resource, Some(attempt_request_sha256))) - if saved_resource == resource.id.as_uuid().to_string() => - { - let attempt_request_sha256 = TrainerRequestDigest::from_hex(&attempt_request_sha256) - .ok_or(OperatorGpuFreeRefusal::InconsistentHistory { task_id })?; - AttestedTrainerAssociation::Saved { - attempt_request_sha256, - } - } - Some(_) => { - return Err(OperatorGpuFreeRefusal::InconsistentHistory { task_id }.into()); - } - }; - - Ok(OperatorGpuFreeEvidence { - trainer_end, - trainer_launch, - normalized_spec_sha256, - trainer_association, - }) -} - -/// Return context of an attested trainer -/// -/// The run has no result or checkpoint evidence, so it is never a completed or -/// stopped context and never permits a same-run resume -fn attested_return_context(task_id: TaskId, evidence: &OperatorGpuFreeEvidence) -> ReturnContext { - match &evidence.trainer_end { - AttestedTrainerEnd::Finished { outcome, .. } => ReturnContext::EndedWithoutResult { - task_id, - outcome: outcome.clone(), - }, - AttestedTrainerEnd::Lost => ReturnContext::LostWithoutResult { task_id }, - } -} - -/// Close the release action and select the next request in serving order, or reserve the return decision -/// -/// The trainer stays registered until the supervisor decides the return, as -/// after a proven release, so the return obligation is kept -fn resolve_release_action( - tx: &Transaction<'_>, - resource: &Resource, - attestation: &OperatorGpuFreeAttestation, - evidence: &OperatorGpuFreeEvidence, - loan: Loan, - action_id: ActionId, -) -> Result<(ResourceRevision, OperatorGpuFreeOutcome), OperatorGpuFreeError> { - let return_context = attested_return_context(attestation.task_id, evidence); - let authority = resource.authority_machine(); - let queued = next_queued_request_for_authority(tx, authority, resource.id)?; - let state_revision = advance(tx, resource, resource.registered_background_task)?; - - let Some(mut request) = queued else { - let return_action = ActionId::new(); - let loan = Loan { - id: loan.id, - resource_id: resource.id, - state: LoanState::Active { - phase: LoanPhase::AwaitingReturn { - action_id: return_action, - return_context: return_context.clone(), - }, - }, - }; - update_awaiting_release_loan(tx, &loan, action_id)?; - let notice = insert_supervisor_notice_in_transaction( - tx, - &SupervisorNotice { - id: NoticeId::new(), - loan_id: loan.id, - action_id: return_action, - state_revision, - destination: resource.supervisor, - assignment_revision: resource.assignment_revision, - payload: SupervisorNoticePayload::ReturnRequired { return_context }, - delivery: SupervisorNoticeDelivery::Pending { attempts: 0 }, - }, - )?; - return Ok(( - state_revision, - OperatorGpuFreeOutcome::ReleaseResolvedReturnRequired { loan, notice }, - )); - }; - - assign_request(tx, &mut request, loan.id)?; - let loan = Loan { - id: loan.id, - resource_id: resource.id, - state: LoanState::Active { - phase: LoanPhase::Serving { - return_context, - current_request_id: request.request_id, - release_provenance: ServingReleaseProvenance::OperatorAttestedGpuFree { - operation_id: attestation.operation_id, - action_id, - task_id: attestation.task_id, - }, - }, - }, - }; - update_awaiting_release_loan(tx, &loan, action_id)?; - - Ok(( - state_revision, - OperatorGpuFreeOutcome::ReleaseResolvedServing { loan, request }, - )) -} - -/// Clear the registration and serve the next request in serving order, or keep this receipt as the boundary -/// -/// No committed state shows a free resource between the two steps, because both -/// commit in the caller's transaction -fn resolve_idle( - tx: &Transaction<'_>, - resource: &Resource, - attestation: &OperatorGpuFreeAttestation, -) -> Result<(ResourceRevision, OperatorGpuFreeOutcome), OperatorGpuFreeError> { - let authority = resource.authority_machine(); - let cleared_revision = advance(tx, resource, None)?; - let Some(request) = next_queued_request_for_authority(tx, authority, resource.id)? else { - return Ok((cleared_revision, OperatorGpuFreeOutcome::IdleBoundary)); - }; - - let mut cleared = resource.clone(); - cleared.state_revision = cleared_revision; - cleared.registered_background_task = None; - let proof = IdleBoundaryProof::OperatorAttestedGpuFree { - operation_id: attestation.operation_id, - task_id: attestation.task_id, - }; - let (loan, request) = open_idle_serving_loan_on(tx, authority, &cleared, request, proof)?; - let state_revision = select_resource(tx, resource.id)? - .ok_or(ResourceStoreError::ResourceNotFound)? - .state_revision; - - Ok(( - state_revision, - OperatorGpuFreeOutcome::IdleServing { loan, request }, - )) -} - -/// Close the Restoring loan with the attested end and serve the next request in serving order -/// -/// The closure clears the registration, as a supervisor resolution of a proven -/// end would. The next queued request then serves from an idle loan in the -/// same transaction, or the closed loan is the saved idle boundary. No committed -/// state shows a free resource between the two steps -fn resolve_restoring( - tx: &Transaction<'_>, - resource: &Resource, - attestation: &OperatorGpuFreeAttestation, - loan: Loan, - action_id: ActionId, - return_context: ReturnContext, -) -> Result<(ResourceRevision, OperatorGpuFreeOutcome), OperatorGpuFreeError> { - let closed = Loan { - id: loan.id, - resource_id: resource.id, - state: LoanState::Closed { - result: LoanClosure::OperatorAttestedRestoreEnded { - return_context, - task_id: attestation.task_id, - operation_id: attestation.operation_id, - }, - }, - }; - let closed_revision = advance(tx, resource, None)?; - if !replace_loan_in_action_phase_on(tx, &closed, LoanActionPhase::Restoring, action_id)? { - return Err(OperatorGpuFreeError::Changed); - } - - let authority = resource.authority_machine(); - let Some(request) = next_queued_request_for_authority(tx, authority, resource.id)? else { - return Ok(( - closed_revision, - OperatorGpuFreeOutcome::RestoreClosedIdleBoundary { closed }, - )); - }; - let mut cleared = resource.clone(); - cleared.state_revision = closed_revision; - cleared.registered_background_task = None; - let proof = IdleBoundaryProof::OperatorAttestedGpuFree { - operation_id: attestation.operation_id, - task_id: attestation.task_id, - }; - let (loan, request) = open_idle_serving_loan_on(tx, authority, &cleared, request, proof)?; - let state_revision = select_resource(tx, resource.id)? - .ok_or(ResourceStoreError::ResourceNotFound)? - .state_revision; - - Ok(( - state_revision, - OperatorGpuFreeOutcome::RestoreClosedServing { - closed: Box::new(closed), - loan, - request, - }, - )) -} - -fn advance( - tx: &Transaction<'_>, - resource: &Resource, - registered_background_task: Option, -) -> Result { - advance_resource_on(tx, resource, registered_background_task).map_err(|error| match error { - AdvanceError::Exhausted => OperatorGpuFreeRefusal::RevisionExhausted { - revision: resource.state_revision, - } - .into(), - AdvanceError::Changed => OperatorGpuFreeError::Changed, - AdvanceError::Storage(error) => OperatorGpuFreeError::Storage(error), - }) -} - -fn assign_request( - tx: &Transaction<'_>, - request: &mut crate::resource::ResourceRequest, - loan_id: LoanId, -) -> Result<(), OperatorGpuFreeError> { - request.state = ResourceRequestState::Assigned { loan_id }; - let changed = tx.execute( - "UPDATE resource_requests SET state_json = ?1 - WHERE request_id = ?2 AND resource_id = ?3 - AND json_extract(state_json, '$.type') = 'queued'", - params![ - serde_json::to_string(&request.state)?, - request.request_id.0.to_string(), - request.resource_id.as_uuid().to_string(), - ], - )?; - if changed != 1 { - return Err(OperatorGpuFreeError::Changed); - } - Ok(()) -} - -fn update_awaiting_release_loan( - tx: &Transaction<'_>, - loan: &Loan, - action_id: ActionId, -) -> Result<(), OperatorGpuFreeError> { - if !replace_loan_in_action_phase_on(tx, loan, LoanActionPhase::AwaitingRelease, action_id)? { - return Err(OperatorGpuFreeError::Changed); - } - Ok(()) -} - -pub(super) fn saved_receipt_on( - conn: &Connection, - operation_id: OperatorAttestationId, -) -> Result, OperatorGpuFreeError> { - let saved: Option = conn - .query_row( - "SELECT receipt_json FROM resource_operator_attestations WHERE operation_id = ?1", - [operation_id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional()?; - Ok(saved.map(|json| serde_json::from_str(&json)).transpose()?) -} - -/// Return the saved attestation that is the current idle boundary, if any -/// -/// It is current only while no loan and no first background launch was saved -/// after it and the resource is still unregistered. A later record supersedes it -pub(crate) fn current_operator_boundary_on( - conn: &Connection, - resource: &Resource, -) -> Result, ResourceStoreError> { - let saved: Option<(Option, Option, String)> = conn - .query_row( - "SELECT preceding_loan, preceding_launch, receipt_json - FROM resource_operator_attestations - WHERE resource_id = ?1 - AND json_extract(receipt_json, '$.outcome.type') = 'idle_boundary' - ORDER BY rowid DESC LIMIT 1", - [resource.id.as_uuid().to_string()], - |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?)), - ) - .optional()?; - let Some((preceding_loan, preceding_launch, receipt_json)) = saved else { - return Ok(None); - }; - let latest_loan = latest_loan_on(conn, resource.id)?.map(|loan| loan.id.as_uuid().to_string()); - let latest_launch = latest_launch_request_on(conn, resource.id)?; - if latest_loan != preceding_loan - || latest_launch.map(|request| request.0.to_string()) != preceding_launch - { - return Ok(None); - } - - let receipt: OperatorGpuFreeReceipt = - serde_json::from_str(&receipt_json).map_err(|_| invalid_receipt())?; - if receipt.attestation.resource_id != resource.id - || receipt.attestation.authority_machine != resource.authority_machine() - || !matches!( - receipt.attestation.state_binding, - OperatorStateBinding::NoLoan | OperatorStateBinding::FirstBackgroundLaunch { .. } - ) - || !matches!(receipt.outcome, OperatorGpuFreeOutcome::IdleBoundary) - || resource.registered_background_task.is_some() - { - return Ok(None); - } - - Ok(Some(OperatorIdleBoundary { - operation_id: receipt.attestation.operation_id, - task_id: receipt.attestation.task_id, - preceding_launch: latest_launch, - })) -} - -/// Check that an operator-attested Serving provenance matches its saved receipt -/// -/// The receipt must name this resource, authority, loan, release action, task, -/// return context, and provenance, and the trainer must still be registered -pub(crate) fn operator_serving_release_matches_on( - conn: &Connection, - authority: MachineId, - resource: &Resource, - loan: &Loan, - return_context: &ReturnContext, - provenance: &ServingReleaseProvenance, -) -> Result { - let ServingReleaseProvenance::OperatorAttestedGpuFree { - operation_id, - action_id, - task_id, - } = provenance - else { - return Ok(false); - }; - if resource.registered_background_task != Some(*task_id) { - return Ok(false); - } - let receipt = saved_receipt_on(conn, *operation_id).map_err(|error| match error { - OperatorGpuFreeError::Storage(error) => ResourceStoreError::Storage(error), - _ => invalid_receipt(), - })?; - let Some(receipt) = receipt else { - return Ok(false); - }; - let attestation = &receipt.attestation; - let OperatorGpuFreeOutcome::ReleaseResolvedServing { - loan: committed_loan, - .. - } = &receipt.outcome - else { - return Ok(false); - }; - let committed_matches = matches!( - &committed_loan.state, - LoanState::Active { - phase: LoanPhase::Serving { - return_context: committed_context, - release_provenance: committed_provenance, - .. - } - } if committed_context == return_context && committed_provenance == provenance - ); - - Ok(committed_matches - && committed_loan.id == loan.id - && loan.resource_id == resource.id - && attestation.resource_id == resource.id - && attestation.authority_machine == authority - && resource.authority_machine() == authority - && attestation.task_id == *task_id - && attestation.state_binding - == (OperatorStateBinding::AwaitingRelease { - loan_id: loan.id, - action_id: *action_id, - }) - && attested_return_context(*task_id, &receipt.evidence) == *return_context) -} - -/// Check that a loan closed with an operator-attested restore end matches its saved receipt -/// -/// The receipt must name this resource, authority, loan, return task, and the -/// exact closed loan state. A closure without its receipt proves nothing -pub(crate) fn operator_restore_closure_matches_on( - conn: &Connection, - resource: &Resource, - loan: &Loan, -) -> Result { - let LoanState::Closed { - result: - LoanClosure::OperatorAttestedRestoreEnded { - task_id, - operation_id, - .. - }, - } = &loan.state - else { - return Ok(false); - }; - let receipt = saved_receipt_on(conn, *operation_id).map_err(|error| match error { - OperatorGpuFreeError::Storage(error) => ResourceStoreError::Storage(error), - _ => invalid_receipt(), - })?; - let Some(receipt) = receipt else { - return Ok(false); - }; - let attestation = &receipt.attestation; - let closed = match &receipt.outcome { - OperatorGpuFreeOutcome::RestoreClosedServing { closed, .. } => closed.as_ref(), - OperatorGpuFreeOutcome::RestoreClosedIdleBoundary { closed } => closed, - OperatorGpuFreeOutcome::ReleaseResolvedServing { .. } - | OperatorGpuFreeOutcome::ReleaseResolvedReturnRequired { .. } - | OperatorGpuFreeOutcome::IdleServing { .. } - | OperatorGpuFreeOutcome::IdleBoundary => return Ok(false), - }; - let binding_matches = attestation - .state_binding - .restoring_action() - .is_some_and(|(loan_id, _)| loan_id == loan.id); - - Ok(binding_matches - && closed == loan - && loan.resource_id == resource.id - && attestation.resource_id == resource.id - && attestation.authority_machine == resource.authority_machine() - && attestation.task_id == *task_id - && attestation.operation_id == *operation_id) -} - -fn invalid_receipt() -> ResourceStoreError { - ResourceStoreError::corrupt( - "operator attestation receipt", - "receipt does not decode or names other records", - ) -} diff --git a/src/store/resource/release_checkpoint.rs b/src/store/resource/release_checkpoint.rs deleted file mode 100644 index af26869..0000000 --- a/src/store/resource/release_checkpoint.rs +++ /dev/null @@ -1,508 +0,0 @@ -//! Release checkpoint baseline, stop reservation, and trainer cancellation on the authority - -use std::path::Path; - -use chrono::{SecondsFormat, Utc}; -use rusqlite::{Connection, TransactionBehavior, params}; - -use super::release_watcher::release_watcher_is_running_on; -use super::trainer_association::{ - trainer_association_bind_snapshot, trainer_association_by_resource_and_task, -}; -use crate::domain::ProcessStatus; -use crate::machine::MachineId; -use crate::resource::store::{ - ReleaseCheckpointCancellationOutcome, ReleaseCheckpointCancellationResult, - ReleaseCheckpointError, TrainerAttemptAssociationStoreError, - release_checkpoint_state_for_action, select_non_closed_loan, update_release_checkpoint_state, - validate_release_watcher_intent_on, -}; -use crate::resource::trainer_publication::{ - AttemptBinding, VerifiedCheckpointPublication, WatchObservation, WatcherError, - observe_release_with_checkpoint_evidence, revalidate_checkpoint_publication, snapshot, -}; -use crate::resource::{ - ActionId, LoanPhase, LoanState, ReleaseCheckpointAction, ReleaseCheckpointBaseline, - ReleaseCheckpointBinding, ReleaseCheckpointCancellation, ReleaseCheckpointPhase, - ReleaseCheckpointState, ReleaseCheckpointStopDecision, ReleaseCheckpointStopOutcome, - ReleaseStopReservationId, ResourceId, ResourceRevision, TrainerAttemptAssociationProof, -}; -use crate::store::Store; - -/// Opening a release action saves its checkpoint state in the same transaction -const CHECKPOINT_STATE_MISSING: &str = "current release action has no checkpoint state"; - -/// Saved checkpoint state of the current release action and its rebuilt binding -struct CheckpointForAction { - state: ReleaseCheckpointState, - previous_json: String, - binding: ReleaseCheckpointBinding, -} - -impl Store { - /// Capture the authority-built checkpoint baseline for one bound release watcher - pub(crate) fn capture_release_checkpoint_baseline_for_authority( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - action_id: ActionId, - expected_state_revision: ResourceRevision, - ) -> Result { - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate)?; - let CheckpointForAction { - mut state, - previous_json, - binding, - } = load_checkpoint_for_action( - &tx, - authority_machine, - resource_id, - action_id, - expected_state_revision, - )?; - match &state.phase { - ReleaseCheckpointPhase::BaselineCaptured { baseline } - | ReleaseCheckpointPhase::StopReserved { baseline, .. } - | ReleaseCheckpointPhase::CancellationCommitted { baseline, .. } => { - if baseline.binding != binding { - return Err(ReleaseCheckpointError::Conflict); - } - tx.commit()?; - return Ok(baseline.clone()); - } - ReleaseCheckpointPhase::WatcherBindingPending => {} - } - - let checkpoint_baseline = ReleaseCheckpointBaseline { - snapshot: snapshot( - &binding.association.canonical_runtime_root, - &binding.attempt_binding, - )?, - binding, - }; - state.phase = ReleaseCheckpointPhase::BaselineCaptured { - baseline: checkpoint_baseline.clone(), - }; - update_release_checkpoint_state(&tx, &previous_json, &state)?; - tx.commit()?; - - Ok(checkpoint_baseline) - } - - /// Reserve the exact task stop decision after a matching new checkpoint is verified - pub(crate) fn reserve_release_checkpoint_stop_for_authority( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - action_id: ActionId, - expected_state_revision: ResourceRevision, - ) -> Result { - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate)?; - let CheckpointForAction { - mut state, - previous_json, - binding, - } = load_checkpoint_for_action( - &tx, - authority_machine, - resource_id, - action_id, - expected_state_revision, - )?; - let baseline = match &state.phase { - ReleaseCheckpointPhase::WatcherBindingPending => { - return Err(ReleaseCheckpointError::BaselineMissing { action_id }); - } - ReleaseCheckpointPhase::BaselineCaptured { baseline } => baseline.clone(), - ReleaseCheckpointPhase::StopReserved { baseline, decision } - | ReleaseCheckpointPhase::CancellationCommitted { - baseline, decision, .. - } => { - if baseline.binding != binding || decision.binding != binding { - return Err(ReleaseCheckpointError::Conflict); - } - tx.commit()?; - return Ok(ReleaseCheckpointStopOutcome::AlreadyReserved( - decision.as_ref().clone(), - )); - } - }; - if baseline.binding != binding { - return Err(ReleaseCheckpointError::Conflict); - } - - let task_id = state.action.observed_background_task; - let task = crate::store::task_by_id_on(&tx, task_id)? - .ok_or(ReleaseCheckpointError::TaskMissing { task_id })?; - let (observation, checkpoint) = observe_release_with_checkpoint_evidence( - &binding.association.canonical_runtime_root, - &binding.attempt_binding, - &baseline.snapshot, - &task.state, - )?; - let Some(checkpoint) = checkpoint else { - let outcome = unreserved_stop_outcome(observation)?; - tx.commit()?; - return Ok(outcome); - }; - if !matches!(observation, WatchObservation::StopRequestCandidate { .. }) { - return Err(ReleaseCheckpointError::Conflict); - } - - let decision = ReleaseCheckpointStopDecision { - binding, - reservation_id: ReleaseStopReservationId::new(), - selected_checkpoint: checkpoint, - }; - state.phase = ReleaseCheckpointPhase::StopReserved { - baseline, - decision: Box::new(decision.clone()), - }; - update_release_checkpoint_state(&tx, &previous_json, &state)?; - tx.commit()?; - - Ok(ReleaseCheckpointStopOutcome::Reserved(decision)) - } - - /// Atomically commit the exact saved stop decision and its trainer cancel marker - pub(crate) fn commit_release_checkpoint_cancellation_for_authority( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - action_id: ActionId, - expected_state_revision: ResourceRevision, - expected_decision: &ReleaseCheckpointStopDecision, - ) -> Result { - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate)?; - let CheckpointForAction { - mut state, - previous_json, - binding, - } = load_checkpoint_for_action( - &tx, - authority_machine, - resource_id, - action_id, - expected_state_revision, - )?; - let (baseline, decision) = match state.phase { - ReleaseCheckpointPhase::CancellationCommitted { - baseline, - decision, - cancellation, - } => { - check_saved_decision(&binding, &baseline, &decision, expected_decision, action_id)?; - let task = crate::store::task_by_id_on(&tx, cancellation.task_id)?.ok_or( - ReleaseCheckpointError::TaskMissing { - task_id: cancellation.task_id, - }, - )?; - if task.cancel_requested_at.is_none() { - return Err(ReleaseCheckpointError::TrainerCancellationConflict { - task_id: cancellation.task_id, - }); - } - - tx.commit()?; - return Ok(ReleaseCheckpointCancellationOutcome::AlreadyCommitted( - ReleaseCheckpointCancellationResult { - decision: *decision, - cancellation, - }, - )); - } - ReleaseCheckpointPhase::StopReserved { baseline, decision } => { - check_saved_decision(&binding, &baseline, &decision, expected_decision, action_id)?; - (baseline, decision) - } - ReleaseCheckpointPhase::WatcherBindingPending - | ReleaseCheckpointPhase::BaselineCaptured { .. } => { - return Err(ReleaseCheckpointError::StopDecisionMissing { action_id }); - } - }; - - if !release_watcher_is_running_on( - &tx, - authority_machine, - resource_id, - &binding.watcher_intent, - )? { - tx.commit()?; - return Ok(ReleaseCheckpointCancellationOutcome::WatcherNotReady { - watcher_task_id: binding.watcher_intent.watcher_task_id.as_task_id(), - }); - } - - let task_id = state.action.observed_background_task; - let trainer = - trainer_association_bind_snapshot(&tx, authority_machine, resource_id, task_id)?; - if trainer.normalized_spec_sha256 != binding.association.normalized_spec_sha256 - || !trainer.task_row_matches_spec(task_id) - { - return Err(ReleaseCheckpointError::TrainerCommandBindingChanged { task_id }); - } - if trainer.task_row.status() != ProcessStatus::Running { - return Err(ReleaseCheckpointError::TrainerTaskNotRunning { - task_id, - state: trainer.task_row.status(), - }); - } - if trainer.identity_state != ProcessStatus::Running { - return Err(TrainerAttemptAssociationStoreError::IdentityNotRunning { task_id }.into()); - } - if trainer.task_row.cancel_requested_at.is_some() { - return Err(ReleaseCheckpointError::TrainerCancellationConflict { task_id }); - } - revalidate_selected_checkpoint(&binding, &decision, action_id)?; - - let cancel_requested_at = Utc::now(); - let marker = cancel_requested_at.to_rfc3339_opts(SecondsFormat::Nanos, true); - let changed = tx.execute( - "UPDATE tasks SET cancel_requested_at = ?1, updated_at = ?1 - WHERE id = ?2 AND status = 'running' AND cancel_requested_at IS NULL", - params![marker, task_id.to_string()], - )?; - if changed != 1 { - return Err(ReleaseCheckpointError::TrainerCancellationConflict { task_id }); - } - - let cancellation = ReleaseCheckpointCancellation { - task_id, - cancel_requested_at, - }; - let result = ReleaseCheckpointCancellationResult { - decision: decision.as_ref().clone(), - cancellation: cancellation.clone(), - }; - state.phase = ReleaseCheckpointPhase::CancellationCommitted { - baseline, - decision, - cancellation, - }; - update_release_checkpoint_state(&tx, &previous_json, &state)?; - tx.commit()?; - - Ok(ReleaseCheckpointCancellationOutcome::Committed(result)) - } -} - -/// Load the current action's checkpoint state at the expected revision with its binding -fn load_checkpoint_for_action( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, - action_id: ActionId, - expected_state_revision: ResourceRevision, -) -> Result { - let Some((state, previous_json)) = - release_checkpoint_state_for_action(conn, resource_id, action_id)? - else { - return Err(missing_release_checkpoint_state_error( - conn, - resource_id, - action_id, - )?); - }; - if state.action.resource_id != resource_id - || state.action.action_id != action_id - || state.action.state_revision != expected_state_revision - { - return Err(ReleaseCheckpointError::Conflict); - } - let binding = - release_checkpoint_binding_on(conn, authority_machine, resource_id, &state.action)?; - - Ok(CheckpointForAction { - state, - previous_json, - binding, - }) -} - -/// Compare a saved stop decision with the caller's decision and the current binding -fn check_saved_decision( - binding: &ReleaseCheckpointBinding, - baseline: &ReleaseCheckpointBaseline, - decision: &ReleaseCheckpointStopDecision, - expected_decision: &ReleaseCheckpointStopDecision, - action_id: ActionId, -) -> Result<(), ReleaseCheckpointError> { - if decision != expected_decision { - return Err(ReleaseCheckpointError::StopDecisionMismatch { action_id }); - } - if baseline.binding != *binding || decision.binding != *binding { - return Err(ReleaseCheckpointError::Conflict); - } - - Ok(()) -} - -/// Map an observation with no new checkpoint to the stop outcome it reports -fn unreserved_stop_outcome( - observation: WatchObservation, -) -> Result { - Ok(match observation { - WatchObservation::WaitingForTaskStart => ReleaseCheckpointStopOutcome::WaitingForTaskStart, - WatchObservation::WaitingForCheckpoint => { - ReleaseCheckpointStopOutcome::WaitingForCheckpoint - } - WatchObservation::CompletedResultAwaitingTaskExit { .. } => { - ReleaseCheckpointStopOutcome::CompletedResultAwaitingTaskExit - } - WatchObservation::AlreadyCompletedCandidate { .. } => { - ReleaseCheckpointStopOutcome::AlreadyCompleted - } - WatchObservation::Attention(attention) => { - ReleaseCheckpointStopOutcome::Attention(attention) - } - WatchObservation::StopRequestCandidate { .. } => { - return Err(ReleaseCheckpointError::Conflict); - } - }) -} - -/// Rebuild the binding of one release action from its loan, watcher intent, and association -pub(super) fn release_checkpoint_binding_on( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, - action: &ReleaseCheckpointAction, -) -> Result { - let not_awaiting = || ReleaseCheckpointError::NotAwaitingRelease { - action_id: action.action_id, - }; - let loan = select_non_closed_loan(conn, resource_id)?.ok_or_else(not_awaiting)?; - let LoanState::Active { - phase: - LoanPhase::AwaitingRelease { - action_id, - observed_background_task, - watcher_intent, - }, - } = loan.state - else { - return Err(not_awaiting()); - }; - if action.resource_id != resource_id - || action.action_id != action_id - || action.observed_background_task != observed_background_task - { - return Err(ReleaseCheckpointError::Conflict); - } - let intent = watcher_intent.ok_or(ReleaseCheckpointError::WatcherIntentMissing { - action_id: action.action_id, - })?; - validate_release_watcher_intent_on(conn, authority_machine, resource_id, &intent)?; - - let task_id = action.observed_background_task; - let association = trainer_association_by_resource_and_task(conn, resource_id, task_id)? - .ok_or(ReleaseCheckpointError::TrainerAssociationMissing { task_id })?; - if association.resource_id() != resource_id - || association.authority_machine() != authority_machine - || association.task_id() != task_id - { - return Err(ReleaseCheckpointError::Conflict); - } - - let binding = ReleaseCheckpointBinding { - action: action.clone(), - association: TrainerAttemptAssociationProof::from(&association), - attempt_binding: association.verified_attempt().binding().clone(), - watcher_intent: intent, - }; - binding - .validate_for(action) - .map_err(ReleaseCheckpointError::InvalidStoredEvidence)?; - - Ok(binding) -} - -/// Recheck one selected checkpoint publication against its trainer attempt -/// -/// A missing, malformed, or symlinked publication means that the checkpoint -/// changed; only other watcher failures are errors -pub(super) fn checkpoint_publication_unchanged( - runtime_root: &Path, - binding: &AttemptBinding, - checkpoint: &VerifiedCheckpointPublication, -) -> Result { - match revalidate_checkpoint_publication(runtime_root, binding, checkpoint) { - Err(WatcherError::MalformedPublication { .. } | WatcherError::Symlink { .. }) => Ok(false), - result => result, - } -} - -fn revalidate_selected_checkpoint( - binding: &ReleaseCheckpointBinding, - decision: &ReleaseCheckpointStopDecision, - action_id: ActionId, -) -> Result<(), ReleaseCheckpointError> { - if !checkpoint_publication_unchanged( - &binding.association.canonical_runtime_root, - &binding.attempt_binding, - &decision.selected_checkpoint, - )? { - return Err(ReleaseCheckpointError::SelectedCheckpointChanged { action_id }); - } - - Ok(()) -} - -/// Report a missing checkpoint state as corrupt only for the current release action -pub(super) fn missing_release_checkpoint_state_error( - conn: &Connection, - resource_id: ResourceId, - action_id: ActionId, -) -> Result { - let Some(loan) = select_non_closed_loan(conn, resource_id)? else { - return Ok(ReleaseCheckpointError::Conflict); - }; - match loan.state { - LoanState::Active { - phase: - LoanPhase::AwaitingRelease { - action_id: current_action, - .. - }, - } if current_action == action_id => Ok(ReleaseCheckpointError::InvalidStoredEvidence( - CHECKPOINT_STATE_MISSING, - )), - _ => Ok(ReleaseCheckpointError::Conflict), - } -} - -/// Require a captured baseline, and any saved decision, bound to the current action -pub(super) fn validate_release_checkpoint_baseline_on( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, - action_id: ActionId, -) -> Result<(), ReleaseCheckpointError> { - let Some((state, _)) = release_checkpoint_state_for_action(conn, resource_id, action_id)? - else { - return Err(ReleaseCheckpointError::InvalidStoredEvidence( - CHECKPOINT_STATE_MISSING, - )); - }; - let binding = - release_checkpoint_binding_on(conn, authority_machine, resource_id, &state.action)?; - match state.phase { - ReleaseCheckpointPhase::WatcherBindingPending => { - Err(ReleaseCheckpointError::BaselineMissing { action_id }) - } - ReleaseCheckpointPhase::BaselineCaptured { baseline } if baseline.binding == binding => { - Ok(()) - } - ReleaseCheckpointPhase::StopReserved { baseline, decision } - | ReleaseCheckpointPhase::CancellationCommitted { - baseline, decision, .. - } if baseline.binding == binding && decision.binding == binding => Ok(()), - _ => Err(ReleaseCheckpointError::Conflict), - } -} diff --git a/src/store/resource/release_completion.rs b/src/store/resource/release_completion.rs deleted file mode 100644 index 476fc4e..0000000 --- a/src/store/resource/release_completion.rs +++ /dev/null @@ -1,421 +0,0 @@ -//! Verified release proof for one awaiting-release action on the authority - -use rusqlite::{Connection, TransactionBehavior}; - -use super::release_checkpoint::checkpoint_publication_unchanged; -use super::release_proof::{ReleaseProofEvidence, VerifiedReleaseEvidence, VerifiedReleaseProof}; -use super::trainer_association::{SavedTrainerAssociation, saved_trainer_association_on}; -use crate::domain::{ - ExitReason, ProcessGroupExitEvidence, ProcessStatus, TaskId, TaskRow, TaskState, -}; -use crate::error::AppError; -use crate::machine::MachineId; -use crate::resource::command_shape::DirectSegmentCommandShape; -use crate::resource::ownership_lock::{ - OwnershipLockGuard, OwnershipLockProbe, probe_segment_ownership_lock, -}; -use crate::resource::store::{ - CompleteReleaseError, ReleaseCompletionResult, ResourceStoreError, - complete_release_for_authority as persist_release_completion_for_authority, - release_checkpoint_state_for_action, release_completion_for_retry, select_non_closed_loan, - select_resource, -}; -use crate::resource::trainer_publication::find_completed_result; -use crate::resource::{ - ActionId, LoanPhase, LoanState, ReleaseCheckpointAction, ReleaseCheckpointCancellation, - ReleaseCheckpointPhase, ReleaseCheckpointStopDecision, ResourceId, ResourceRevision, - TrainerAttemptAssociation, TrainerAttemptAssociationProof, -}; -use crate::spec::NormalizedSpec; -use crate::store::Store; -use crate::store::identity::executor_identity_with_json_on; -use crate::submission::{ExecutionRecord, ExecutorIdentity, normalized_spec_sha256}; - -/// Release action that the proof must name -#[derive(Clone, Copy)] -struct ReleaseTarget { - authority_machine: MachineId, - resource_id: ResourceId, - action_id: ActionId, - expected_state_revision: ResourceRevision, -} - -/// Saved records read in one snapshot before any file-system evidence is checked -struct ReleaseProofPreflight { - task_id: TaskId, - association: SavedTrainerAssociation, - identity_json: String, - task_row: TaskRow, - normalized_spec: NormalizedSpec, - terminal_evidence: ReleaseProofTerminalEvidence, -} - -enum ReleaseProofTerminalEvidence { - /// Exit status 0; a verified result publication makes it completed, and - /// its absence makes it an ended run with no usable result - Completed, - Stopped { - decision: Box, - cancellation: ReleaseCheckpointCancellation, - }, - /// Any other terminal outcome, or a cancellation without this action's - /// committed stop; only the released lock can prove its release - Ended { outcome: ExitReason }, -} - -impl ReleaseProofTerminalEvidence { - /// Identity state that the executor must have saved for this terminal evidence - fn expected_identity_state(&self, task_row: &TaskRow) -> ProcessStatus { - match self { - Self::Completed => ProcessStatus::Succeeded, - Self::Stopped { .. } => ProcessStatus::Cancelled, - Self::Ended { .. } => task_row.status(), - } - } -} - -impl Store { - /// Complete a saved release action on this store connection - pub(crate) fn complete_release_for_authority( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - action_id: ActionId, - expected_state_revision: ResourceRevision, - ) -> Result { - if let Some(receipt) = release_completion_for_retry( - &self.conn, - authority_machine, - resource_id, - action_id, - expected_state_revision, - )? { - return Ok(receipt); - } - - let proof = self.build_verified_release_proof( - authority_machine, - resource_id, - action_id, - expected_state_revision, - )?; - persist_release_completion_for_authority(&mut self.conn, proof) - } - - /// Read the release evidence and hold the released lock without committing the release - pub(super) fn build_verified_release_proof( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - action_id: ActionId, - expected_state_revision: ResourceRevision, - ) -> Result { - let target = ReleaseTarget { - authority_machine, - resource_id, - action_id, - expected_state_revision, - }; - let preflight = { - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Deferred)?; - let preflight = release_proof_preflight(&tx, target)?; - tx.commit()?; - preflight - }; - - DirectSegmentCommandShape::validate( - &preflight.normalized_spec, - &preflight.task_row, - &preflight.association.association, - )?; - let release_evidence = verify_terminal_evidence(&preflight)?; - let ownership_guard = hold_released_ownership_lock(&preflight)?; - - Ok(VerifiedReleaseProof::new(ReleaseProofEvidence { - authority_machine: target.authority_machine, - resource_id: target.resource_id, - action_id: target.action_id, - expected_state_revision: target.expected_state_revision, - task_id: preflight.task_id, - association: preflight.association.association, - association_json: preflight.association.json, - identity_json: preflight.identity_json, - task_row: preflight.task_row, - release_evidence, - ownership_guard, - })) - } -} - -/// Read the action's trainer, association, identity, and terminal state in one snapshot -fn release_proof_preflight( - conn: &Connection, - target: ReleaseTarget, -) -> Result { - let task_id = awaiting_release_task(conn, target)?; - let association = saved_trainer_association_on(conn, target.resource_id, task_id)? - .ok_or(CompleteReleaseError::TrainerAssociationMissing { task_id })?; - let saved = &association.association; - if saved.authority_machine() != target.authority_machine - || saved.resource_id() != target.resource_id - || saved.task_id() != task_id - { - return Err(CompleteReleaseError::TrainerAssociationMismatch { task_id }); - } - - let task_row = crate::store::task_by_id_on(conn, task_id)? - .ok_or(CompleteReleaseError::BackgroundTaskMissing { task_id })?; - let (identity, identity_json) = executor_identity_with_json_on(conn, task_id)? - .ok_or(CompleteReleaseError::TrainerIdentityMissing { task_id })?; - let record = accepted_trainer_record(identity, task_id, target.authority_machine)?; - - let terminal_evidence = terminal_evidence(conn, target, task_id, &task_row, saved)?; - if record.state != terminal_evidence.expected_identity_state(&task_row) { - return Err(CompleteReleaseError::TrainerIdentityChanged { task_id }); - } - if task_row.process_group_exit_evidence() != ProcessGroupExitEvidence::ConfirmedExited { - return Err(CompleteReleaseError::WorkerExitUnconfirmed { task_id }); - } - - let normalized_spec = record - .current_spec() - .ok_or(CompleteReleaseError::TrainerAssociationMismatch { task_id })? - .clone(); - if normalized_spec_sha256(&normalized_spec).map_err(AppError::from)? - != saved.normalized_spec_sha256() - { - return Err(CompleteReleaseError::TrainerAssociationMismatch { task_id }); - } - - Ok(ReleaseProofPreflight { - task_id, - association, - identity_json, - task_row, - normalized_spec, - terminal_evidence, - }) -} - -/// Return the registered trainer that the current release action observes -fn awaiting_release_task( - conn: &Connection, - target: ReleaseTarget, -) -> Result { - let ReleaseTarget { - authority_machine, - resource_id, - action_id, - expected_state_revision, - } = target; - // another authority's resource is not visible to this daemon - let resource = select_resource(conn, resource_id)? - .filter(|resource| resource.authority_machine() == authority_machine) - .ok_or(ResourceStoreError::ResourceNotFound)?; - let loan = select_non_closed_loan(conn, resource_id)? - .ok_or(CompleteReleaseError::ActionNotFound { action_id })?; - let not_awaiting = || CompleteReleaseError::NotAwaitingRelease { - loan_id: loan.id, - action_id, - }; - let LoanState::Active { - phase: - LoanPhase::AwaitingRelease { - action_id: saved_action_id, - observed_background_task, - .. - }, - } = loan.state - else { - return Err(not_awaiting()); - }; - if saved_action_id != action_id { - return Err(not_awaiting()); - } - if resource.state_revision != expected_state_revision { - return Err(CompleteReleaseError::StaleRevision { - expected: expected_state_revision, - actual: resource.state_revision, - }); - } - if resource.registered_background_task != Some(observed_background_task) { - return Err(CompleteReleaseError::TrainerAssociationMismatch { - task_id: observed_background_task, - }); - } - - Ok(observed_background_task) -} - -/// Require the trainer's accepted identity on this execution authority -fn accepted_trainer_record( - identity: ExecutorIdentity, - task_id: TaskId, - authority_machine: MachineId, -) -> Result { - let ExecutorIdentity::Accepted(record) = identity else { - return Err(CompleteReleaseError::TrainerIdentityChanged { task_id }); - }; - // the trainer's callback origin may be any supervisor machine - if !record.is_executed_by(task_id, authority_machine) { - return Err(CompleteReleaseError::TrainerIdentityChanged { task_id }); - } - - Ok(record) -} - -/// Classify the trainer's terminal task state as release evidence -fn terminal_evidence( - conn: &Connection, - target: ReleaseTarget, - task_id: TaskId, - task_row: &TaskRow, - association: &TrainerAttemptAssociation, -) -> Result { - match &task_row.state { - TaskState::Finished { - reason: ExitReason::Exit { code: 0 }, - } => Ok(ReleaseProofTerminalEvidence::Completed), - TaskState::Finished { - reason: ExitReason::Cancelled, - } => stopped_evidence(conn, target, task_id, task_row, association), - // a failed run never implies that any checkpoint is resumable - TaskState::Finished { reason } => Ok(ReleaseProofTerminalEvidence::Ended { - outcome: reason.clone(), - }), - TaskState::Lost => Err(CompleteReleaseError::BackgroundTaskLost { task_id }), - TaskState::Queued | TaskState::Running { .. } => { - Err(CompleteReleaseError::BackgroundTaskNotTerminal { - task_id, - state: task_row.status().to_string(), - }) - } - } -} - -/// Use this action's committed stop as evidence for a cancelled trainer -/// -/// A cancellation that this action did not commit is a generic end; it names -/// no checkpoint that the run could resume from -fn stopped_evidence( - conn: &Connection, - target: ReleaseTarget, - task_id: TaskId, - task_row: &TaskRow, - association: &TrainerAttemptAssociation, -) -> Result { - let ended = ReleaseProofTerminalEvidence::Ended { - outcome: ExitReason::Cancelled, - }; - let Some((checkpoint_state, _)) = - release_checkpoint_state_for_action(conn, target.resource_id, target.action_id)? - else { - return Ok(ended); - }; - let expected_action = ReleaseCheckpointAction { - resource_id: target.resource_id, - action_id: target.action_id, - state_revision: target.expected_state_revision, - observed_background_task: task_id, - }; - if checkpoint_state.action != expected_action { - return Err(CompleteReleaseError::StoppedProofUnavailable { task_id }); - } - let ReleaseCheckpointPhase::CancellationCommitted { - baseline, - decision, - cancellation, - } = checkpoint_state.phase - else { - return Ok(ended); - }; - if baseline.binding.association != TrainerAttemptAssociationProof::from(association) { - return Err(CompleteReleaseError::TrainerAssociationMismatch { task_id }); - } - if decision.binding != baseline.binding || cancellation.task_id != task_id { - return Err(CompleteReleaseError::StoppedProofUnavailable { task_id }); - } - if task_row.cancel_requested_at != Some(cancellation.cancel_requested_at) { - return Err(CompleteReleaseError::TrainerCancellationMarkerChanged { task_id }); - } - - Ok(ReleaseProofTerminalEvidence::Stopped { - decision, - cancellation, - }) -} - -/// Check the published result or selected checkpoint that the terminal evidence names -fn verify_terminal_evidence( - preflight: &ReleaseProofPreflight, -) -> Result { - let task_id = preflight.task_id; - let verified_attempt = preflight.association.association.verified_attempt(); - match &preflight.terminal_evidence { - ReleaseProofTerminalEvidence::Completed => { - // a zero exit with no publication is an end with no usable result - let Some(completed_result) = find_completed_result( - verified_attempt.canonical_runtime_root(), - verified_attempt.binding(), - )? - else { - return Ok(VerifiedReleaseEvidence::Ended { - outcome: ExitReason::Exit { code: 0 }, - }); - }; - if completed_result.binding != *verified_attempt.binding() - || completed_result.request_sha256 != verified_attempt.request_digest() - { - return Err(CompleteReleaseError::CompletedResultRequestMismatch { task_id }); - } - - Ok(VerifiedReleaseEvidence::Completed(Box::new( - completed_result, - ))) - } - ReleaseProofTerminalEvidence::Stopped { - decision, - cancellation, - } => { - let checkpoint = &decision.selected_checkpoint; - if checkpoint.binding != *verified_attempt.binding() - || !checkpoint_publication_unchanged( - verified_attempt.canonical_runtime_root(), - verified_attempt.binding(), - checkpoint, - )? - { - return Err(CompleteReleaseError::StoppedCheckpointChanged { task_id }); - } - - Ok(VerifiedReleaseEvidence::Stopped { - decision: decision.clone(), - cancellation: cancellation.clone(), - }) - } - ReleaseProofTerminalEvidence::Ended { outcome } => Ok(VerifiedReleaseEvidence::Ended { - outcome: outcome.clone(), - }), - } -} - -/// Hold the exact saved ownership lock, which must be free for every release basis -/// -/// The guard stays held until the release transaction commits -fn hold_released_ownership_lock( - preflight: &ReleaseProofPreflight, -) -> Result { - let verified_attempt = preflight.association.association.verified_attempt(); - match probe_segment_ownership_lock( - verified_attempt.canonical_runtime_root(), - verified_attempt.ownership_lock_identity(), - ) { - OwnershipLockProbe::OwnershipHeld => Err(CompleteReleaseError::OwnershipLockStillHeld { - task_id: preflight.task_id, - }), - OwnershipLockProbe::ExactOwnershipReleased(guard) => Ok(guard), - OwnershipLockProbe::Attention(source) => Err(CompleteReleaseError::OwnershipLock(source)), - } -} diff --git a/src/store/resource/release_proof.rs b/src/store/resource/release_proof.rs deleted file mode 100644 index 9f2b5cf..0000000 --- a/src/store/resource/release_proof.rs +++ /dev/null @@ -1,218 +0,0 @@ -use super::release_checkpoint::checkpoint_publication_unchanged; -use crate::domain::{ExitReason, TaskId, TaskRow}; -use crate::machine::MachineId; -use crate::resource::ownership_lock::{OwnershipLockGuard, verify_ownership_lock_guard}; -use crate::resource::store::CompleteReleaseError; -use crate::resource::trainer_publication::{PublishedTerminalResult, find_completed_result}; -use crate::resource::{ - ActionId, ReleaseCheckpointCancellation, ReleaseCheckpointStopDecision, ResourceId, - ResourceRevision, ReturnContext, ServingReleaseProvenance, TrainerAttemptAssociation, -}; - -pub(super) enum VerifiedReleaseEvidence { - Completed(Box), - Stopped { - decision: Box, - cancellation: ReleaseCheckpointCancellation, - }, - /// The trainer ended with no usable result and no committed checkpoint stop - /// - /// The released lock is the only evidence, so the release names the - /// outcome and carries no resume reference - Ended { - outcome: ExitReason, - }, -} - -/// Evidence fields supplied only by the authority resource-store module -pub(super) struct ReleaseProofEvidence { - pub(super) authority_machine: MachineId, - pub(super) resource_id: ResourceId, - pub(super) action_id: ActionId, - pub(super) expected_state_revision: ResourceRevision, - pub(super) task_id: TaskId, - pub(super) association: TrainerAttemptAssociation, - pub(super) association_json: String, - pub(super) identity_json: String, - pub(super) task_row: TaskRow, - pub(super) release_evidence: VerifiedReleaseEvidence, - pub(super) ownership_guard: OwnershipLockGuard, -} - -/// Authority-created proof for one completed result, action-committed stopped checkpoint, -/// or ended trainer whose exact lock is released -/// -/// This type has no serialization or public constructor. It keeps the exact lock guard alive -/// while the release transaction rechecks the saved task, identity, association, and result -#[must_use = "keep the ownership guard alive through the release transaction"] -pub(crate) struct VerifiedReleaseProof(ReleaseProofEvidence); - -impl VerifiedReleaseProof { - pub(super) fn new(evidence: ReleaseProofEvidence) -> Self { - Self(evidence) - } - - pub(crate) const fn authority_machine(&self) -> MachineId { - self.0.authority_machine - } - - pub(crate) const fn resource_id(&self) -> ResourceId { - self.0.resource_id - } - - pub(crate) const fn action_id(&self) -> ActionId { - self.0.action_id - } - - pub(crate) const fn expected_state_revision(&self) -> ResourceRevision { - self.0.expected_state_revision - } - - pub(crate) const fn task_id(&self) -> TaskId { - self.0.task_id - } - - pub(crate) fn association_json(&self) -> &str { - &self.0.association_json - } - - pub(crate) fn identity_json(&self) -> &str { - &self.0.identity_json - } - - pub(crate) fn task_row(&self) -> &TaskRow { - &self.0.task_row - } - - pub(crate) fn stopped_decision_and_cancellation( - &self, - ) -> Option<( - &ReleaseCheckpointStopDecision, - &ReleaseCheckpointCancellation, - )> { - match &self.0.release_evidence { - VerifiedReleaseEvidence::Completed(_) | VerifiedReleaseEvidence::Ended { .. } => None, - VerifiedReleaseEvidence::Stopped { - decision, - cancellation, - } => Some((decision, cancellation)), - } - } - - /// Outcome of an ended trainer that has no result or stop evidence - pub(crate) fn ended_outcome(&self) -> Option<&ExitReason> { - match &self.0.release_evidence { - VerifiedReleaseEvidence::Ended { outcome } => Some(outcome), - VerifiedReleaseEvidence::Completed(_) | VerifiedReleaseEvidence::Stopped { .. } => None, - } - } - - pub(crate) fn return_context(&self) -> ReturnContext { - match &self.0.release_evidence { - VerifiedReleaseEvidence::Completed(result) => ReturnContext::AlreadyCompleted { - task_id: self.0.task_id, - result_ref: format!( - "{}#sha256={}", - result.publication_path.display(), - result.publication_sha256 - ), - }, - VerifiedReleaseEvidence::Stopped { decision, .. } => { - let checkpoint = &decision.selected_checkpoint; - ReturnContext::Stopped { - task_id: self.0.task_id, - checkpoint_ref: format!( - "{}#sha256={}", - checkpoint.path.display(), - checkpoint.record_sha256 - ), - recovery_ref: checkpoint.generation_id.clone(), - } - } - VerifiedReleaseEvidence::Ended { outcome } => ReturnContext::EndedWithoutResult { - task_id: self.0.task_id, - outcome: outcome.clone(), - }, - } - } - - pub(crate) fn serving_release_provenance(&self) -> ServingReleaseProvenance { - match &self.0.release_evidence { - VerifiedReleaseEvidence::Completed(result) => { - ServingReleaseProvenance::CompletedTrainerResult { - action_id: self.0.action_id, - task_id: self.0.task_id, - publication_sha256: result.publication_sha256, - } - } - VerifiedReleaseEvidence::Stopped { decision, .. } => { - let checkpoint = &decision.selected_checkpoint; - ServingReleaseProvenance::StoppedTrainerCheckpoint { - action_id: self.0.action_id, - task_id: self.0.task_id, - generation_id: checkpoint.generation_id.clone(), - record_sha256: checkpoint.record_sha256, - inventory_sha256: checkpoint.inventory_sha256, - } - } - VerifiedReleaseEvidence::Ended { outcome } => { - ServingReleaseProvenance::EndedTrainerLockReleased { - action_id: self.0.action_id, - task_id: self.0.task_id, - outcome: outcome.clone(), - attempt_request_sha256: self.0.association.verified_attempt().request_digest(), - } - } - } - } - - pub(crate) fn verify_external_evidence(&self) -> Result<(), CompleteReleaseError> { - match &self.0.release_evidence { - VerifiedReleaseEvidence::Completed(expected) => { - let current_result = find_completed_result( - self.0 - .association - .verified_attempt() - .canonical_runtime_root(), - self.0.association.verified_attempt().binding(), - )?; - if current_result.as_ref() != Some(expected.as_ref()) { - return Err(CompleteReleaseError::CompletedResultChanged { - task_id: self.0.task_id, - }); - } - } - VerifiedReleaseEvidence::Stopped { decision, .. } => { - let checkpoint = &decision.selected_checkpoint; - if !checkpoint_publication_unchanged( - self.0 - .association - .verified_attempt() - .canonical_runtime_root(), - self.0.association.verified_attempt().binding(), - checkpoint, - )? { - return Err(CompleteReleaseError::StoppedCheckpointChanged { - task_id: self.0.task_id, - }); - } - } - // the exact released lock below is the whole external evidence - VerifiedReleaseEvidence::Ended { .. } => {} - } - - verify_ownership_lock_guard( - self.0 - .association - .verified_attempt() - .canonical_runtime_root(), - self.0 - .association - .verified_attempt() - .ownership_lock_identity(), - &self.0.ownership_guard, - )?; - - Ok(()) - } -} diff --git a/src/store/resource/release_watcher.rs b/src/store/resource/release_watcher.rs deleted file mode 100644 index b5aba74..0000000 --- a/src/store/resource/release_watcher.rs +++ /dev/null @@ -1,543 +0,0 @@ -//! Release-watcher acceptance, running checks, and poll handling on the authority - -use rusqlite::{Connection, TransactionBehavior}; - -use super::action_task::{action_task_receipt_by_task, remote_release_watcher_is_running_on}; -use super::release_checkpoint::{ - missing_release_checkpoint_state_error, release_checkpoint_binding_on, - validate_release_checkpoint_baseline_on, -}; -use super::{ - resource_request_identity_exists, resource_task_row_matches, same_task_binding, - select_authority_resource, -}; -use crate::domain::{ProcessStatus, TaskId}; -use crate::error::AppError; -use crate::events::EventError; -use crate::machine::MachineId; -use crate::resource::bound_action::ActionTaskReceipt; -use crate::resource::release_watcher::{ - ReleaseWatcherCommand, ReleaseWatcherPollAttention, ReleaseWatcherPollOutcome, - ReleaseWatcherPollRequest, -}; -use crate::resource::store::{ - CompleteReleaseError, ReleaseCheckpointCancellationOutcome, ReleaseCheckpointError, - ReleaseWatcherAcceptance, ReleaseWatcherAcceptanceError, ReleaseWatcherAcceptanceInput, - ResourceStoreError, TrainerAttemptAssociationStoreError, bind_release_watcher_intent_on, - release_checkpoint_state_for_action, release_completion_for_retry, - validate_release_watcher_for_local_acceptance, -}; -use crate::resource::trainer_publication::WatcherAttention; -use crate::resource::{ - ReleaseCheckpointAction, ReleaseCheckpointStopOutcome, ReleaseWatcherIntent, ResourceId, -}; -use crate::spec::{NormalizedSpec, NormalizedWorkload}; -use crate::store::Store; -use crate::store::identity::{ - executor_identity_on, origin_route_by_request_on, origin_route_by_task_on, -}; -use crate::submission::{ - CallbackContext, ExecutorIdentity, OriginRoute, RequestId, SubmissionState, - normalized_spec_sha256, -}; - -/// Saved acceptance that owns one bound watcher task -/// -/// The supervisor assignment can change while an accepted watcher runs, so the -/// running check follows the records written at acceptance, not the current assignment -enum SavedWatcherAcceptance { - /// No record accepts this watcher identity yet - NotAccepted, - /// The authority accepted the watcher with local origin routes - Local(Box), - /// The authority accepted the watcher for a remote supervisor under this receipt - Remote(ActionTaskReceipt), -} - -/// Origin routes saved by one local watcher acceptance -struct LocalWatcherRoutes { - by_request: OriginRoute, - by_task: OriginRoute, -} - -impl Store { - /// Accept one fixed release-watcher task and all durable ownership records - pub(crate) fn accept_release_watcher_for_authority( - &mut self, - input: ReleaseWatcherAcceptanceInput, - ) -> Result { - let ReleaseWatcherAcceptanceInput { - authority_machine, - resource_id, - supervisor, - intent, - row, - spec, - callback, - } = input; - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate)?; - let task_id = intent.watcher_task_id.as_task_id(); - let digest = normalized_spec_sha256(&spec).map_err(AppError::from)?; - if intent.normalized_spec_sha256 != digest - || task_id != row.id - || row.thread != supervisor.thread - || spec.thread != supervisor.thread - || !row.binary.is_absolute() - || row.workload != crate::invocation::persist_workload(&spec.workload) - || callback.env != row.env - || callback.cwd != row.cwd - || !matches!(&spec.workload, NormalizedWorkload::Task(_)) - { - return Err(ReleaseWatcherAcceptanceError::Conflict); - } - // the authority accepts only its own command for these exact release identities - let canonical_digest = ReleaseWatcherCommand::from_intent(resource_id, &intent) - .normalized_spec_sha256(&row.binary, supervisor.thread) - .map_err(|_| ReleaseWatcherAcceptanceError::Conflict)?; - if canonical_digest != digest { - return Err(ReleaseWatcherAcceptanceError::Conflict); - } - crate::store::validate_local_task_acceptance(&row, &spec, &callback) - .map_err(|_| ReleaseWatcherAcceptanceError::Conflict)?; - - let saved_supervisor = validate_release_watcher_for_local_acceptance( - &tx, - authority_machine, - resource_id, - supervisor, - &intent, - )?; - if saved_supervisor.machine != authority_machine { - return Ok(ReleaseWatcherAcceptance::UnsupportedRemoteSupervisor { - authority_machine, - supervisor: saved_supervisor, - }); - } - - validate_release_checkpoint_baseline_on( - &tx, - authority_machine, - resource_id, - intent.action_id, - )?; - bind_release_watcher_intent_on(&tx, authority_machine, resource_id, intent.clone())?; - - if resource_request_identity_exists(&tx, intent.request_id, task_id)? { - return Err(ReleaseWatcherAcceptanceError::Conflict); - } - let saved_task = crate::store::task_by_id_on(&tx, task_id)?; - let route_by_request = origin_route_by_request_on(&tx, intent.request_id)?; - let route_by_task = origin_route_by_task_on(&tx, task_id)?; - let identity = executor_identity_on(&tx, task_id)?; - let has_event = super::task_has_any_event(&tx, task_id)?; - - if saved_task.is_none() - && route_by_request.is_none() - && route_by_task.is_none() - && identity.is_none() - && !has_event - { - crate::store::insert_local_task_records_on( - &tx, - &row, - &spec, - authority_machine, - intent.request_id, - &callback, - Some(task_id), - )?; - tx.commit()?; - return Ok(ReleaseWatcherAcceptance::Inserted { task: task_id }); - } - - let (Some(saved_task), Some(route_by_request), Some(route_by_task), Some(identity)) = - (saved_task, route_by_request, route_by_task, identity) - else { - return Err(ReleaseWatcherAcceptanceError::Conflict); - }; - let initial_event_matches = crate::store::events::initial_queued_event_matches_on( - &tx, - task_id, - authority_machine, - authority_machine, - ) - .map_err(ResourceStoreError::from)?; - if !same_task_binding(&saved_task, &row) - || !watcher_routes_match( - &route_by_request, - &route_by_task, - (intent.request_id, task_id), - authority_machine, - supervisor.thread, - &callback, - &spec, - ) - || !watcher_identity_matches( - &identity, - task_id, - authority_machine, - &spec, - saved_task.status(), - ) - || !initial_event_matches - { - return Err(ReleaseWatcherAcceptanceError::Conflict); - } - - let state = saved_task.state.clone(); - tx.commit()?; - Ok(ReleaseWatcherAcceptance::Existing { - task: task_id, - state, - }) - } - - /// Serve one co-located release-watcher poll on this store connection - /// - /// The authority, action, revision, trainer, saved watcher intent, baseline, and - /// accepted running watcher identity are validated before any publication is read - /// A stop decision is reserved once and commits with the exact trainer cancel - /// marker; retries reuse that saved decision and never select another checkpoint - /// Only storage failures are errors; every domain refusal is typed attention - pub(crate) fn poll_release_watcher_for_authority( - &mut self, - authority_machine: MachineId, - request: ReleaseWatcherPollRequest, - ) -> Result { - match self.poll_release_watcher(authority_machine, request.watcher) { - Ok(outcome) => Ok(outcome), - Err(error) => release_watcher_poll_attention(error) - .map(|reason| ReleaseWatcherPollOutcome::Attention { reason }), - } - } - - fn poll_release_watcher( - &mut self, - authority_machine: MachineId, - watcher: ReleaseWatcherCommand, - ) -> Result { - let resource_id = watcher.resource_id; - let action_id = watcher.action_id; - let revision = watcher.state_revision; - let attention = |reason| Ok(ReleaseWatcherPollOutcome::Attention { reason }); - - match release_completion_for_retry( - &self.conn, - authority_machine, - resource_id, - action_id, - revision, - ) { - Ok(Some(_)) => return Ok(ReleaseWatcherPollOutcome::ReleaseSettled), - Ok(None) => {} - Err(CompleteReleaseError::Storage(error)) => return Err(error.into()), - Err(_) => return attention(ReleaseWatcherPollAttention::ActionNotCurrent), - } - - match select_authority_resource(&self.conn, authority_machine, resource_id) { - Ok(_) => {} - Err(ResourceStoreError::ResourceNotFound) => { - return attention(ReleaseWatcherPollAttention::ActionNotCurrent); - } - Err(error) => return Err(error.into()), - } - - let Some((state, _)) = - release_checkpoint_state_for_action(&self.conn, resource_id, action_id)? - else { - return Err(missing_release_checkpoint_state_error( - &self.conn, - resource_id, - action_id, - )?); - }; - let expected_action = ReleaseCheckpointAction { - resource_id, - action_id, - state_revision: revision, - observed_background_task: watcher.trainer_task_id, - }; - if state.action != expected_action { - return attention(ReleaseWatcherPollAttention::ActionNotCurrent); - } - let binding = release_checkpoint_binding_on( - &self.conn, - authority_machine, - resource_id, - &state.action, - )?; - if binding.watcher_intent.watcher_task_id != watcher.watcher_task_id { - return attention(ReleaseWatcherPollAttention::WrongWatcher); - } - validate_release_checkpoint_baseline_on( - &self.conn, - authority_machine, - resource_id, - action_id, - )?; - if !release_watcher_is_running_on( - &self.conn, - authority_machine, - resource_id, - &binding.watcher_intent, - )? { - return Ok(ReleaseWatcherPollOutcome::WatcherNotRunning); - } - - let decision = match self.reserve_release_checkpoint_stop_for_authority( - authority_machine, - resource_id, - action_id, - revision, - )? { - ReleaseCheckpointStopOutcome::Reserved(decision) - | ReleaseCheckpointStopOutcome::AlreadyReserved(decision) => decision, - ReleaseCheckpointStopOutcome::WaitingForTaskStart => { - return Ok(ReleaseWatcherPollOutcome::WaitingForTrainerStart); - } - ReleaseCheckpointStopOutcome::WaitingForCheckpoint => { - return Ok(ReleaseWatcherPollOutcome::WaitingForCheckpoint); - } - ReleaseCheckpointStopOutcome::CompletedResultAwaitingTaskExit => { - return Ok(ReleaseWatcherPollOutcome::CompletedResultAwaitingTrainerExit); - } - ReleaseCheckpointStopOutcome::AlreadyCompleted => { - return Ok(ReleaseWatcherPollOutcome::TrainerCompleted); - } - ReleaseCheckpointStopOutcome::Attention(watcher_attention) => { - return attention(watcher_poll_attention(&watcher_attention)); - } - }; - - match self.commit_release_checkpoint_cancellation_for_authority( - authority_machine, - resource_id, - action_id, - revision, - &decision, - )? { - ReleaseCheckpointCancellationOutcome::Committed(result) - | ReleaseCheckpointCancellationOutcome::AlreadyCommitted(result) => { - Ok(ReleaseWatcherPollOutcome::StopCommitted { - generation_id: result.decision.selected_checkpoint.generation_id, - cancel_requested_at: result.cancellation.cancel_requested_at, - }) - } - ReleaseCheckpointCancellationOutcome::WatcherNotReady { .. } => { - Ok(ReleaseWatcherPollOutcome::WatcherNotRunning) - } - } - } -} - -/// Decide whether the accepted watcher bound to this intent is running -pub(super) fn release_watcher_is_running_on( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, - intent: &ReleaseWatcherIntent, -) -> Result { - let task_id = intent.watcher_task_id.as_task_id(); - let conflict = || ReleaseCheckpointError::WatcherIdentityConflict { task_id }; - select_authority_resource(conn, authority_machine, resource_id)?; - let routes = match saved_watcher_acceptance(conn, intent)? { - SavedWatcherAcceptance::NotAccepted => return Ok(false), - // a remote supervisor's watcher has no authority-local route; its receipt, - // remote identity, and first event prove the same acceptance instead - SavedWatcherAcceptance::Remote(receipt) => { - return remote_release_watcher_is_running_on( - conn, - authority_machine, - resource_id, - intent, - &receipt, - ); - } - SavedWatcherAcceptance::Local(routes) => routes, - }; - let task = crate::store::task_by_id_on(conn, task_id)?; - let identity = executor_identity_on(conn, task_id).map_err(ResourceStoreError::from)?; - let (Some(task), Some(identity)) = (task, identity) else { - return Ok(false); - }; - let ExecutorIdentity::Accepted(record) = &identity else { - return Ok(false); - }; - let spec = record.current_spec().ok_or_else(conflict)?; - if !record.is_owned_by(task_id, authority_machine, authority_machine) - || spec.machine.is_some() - || normalized_spec_sha256(spec)? != intent.normalized_spec_sha256 - || !matches!(&spec.workload, NormalizedWorkload::Task(_)) - || !resource_task_row_matches(&task, task_id, spec) - { - return Err(conflict()); - } - - if task.status() != ProcessStatus::Running - || task.pid().is_none() - || record.state != task.status() - { - return Ok(false); - } - // the intent digest binds the spec thread, so the saved routes are checked - // against it rather than the current, possibly replaced, supervisor - if !watcher_routes_match( - &routes.by_request, - &routes.by_task, - (intent.request_id, task_id), - authority_machine, - spec.thread, - &routes.by_request.callback, - spec, - ) || !watcher_identity_matches(&identity, task_id, authority_machine, spec, task.status()) - { - return Err(conflict()); - } - - initial_watcher_event_matches(conn, task_id, authority_machine) -} - -fn saved_watcher_acceptance( - conn: &Connection, - intent: &ReleaseWatcherIntent, -) -> Result { - let task_id = intent.watcher_task_id.as_task_id(); - let receipt = action_task_receipt_by_task(conn, task_id)?; - let by_request = - origin_route_by_request_on(conn, intent.request_id).map_err(ResourceStoreError::from)?; - let by_task = origin_route_by_task_on(conn, task_id).map_err(ResourceStoreError::from)?; - match (receipt, by_request, by_task) { - (Some(receipt), None, None) => Ok(SavedWatcherAcceptance::Remote(receipt)), - (None, Some(by_request), Some(by_task)) => Ok(SavedWatcherAcceptance::Local(Box::new( - LocalWatcherRoutes { - by_request, - by_task, - }, - ))), - // a task row without either acceptance record was accepted for another owner - (None, None, None) if crate::store::task_by_id_on(conn, task_id)?.is_none() => { - Ok(SavedWatcherAcceptance::NotAccepted) - } - _ => Err(ReleaseCheckpointError::WatcherIdentityConflict { task_id }), - } -} - -/// Check that the watcher's first saved event is the accepted queued state -/// -/// Storage failures stay retryable; a different first event means the watcher -/// identity no longer matches its acceptance -fn initial_watcher_event_matches( - conn: &Connection, - task: TaskId, - authority: MachineId, -) -> Result { - crate::store::events::initial_queued_event_matches_on(conn, task, authority, authority).map_err( - |error| match error { - EventError::Storage(error) => ReleaseCheckpointError::Task(error), - _ => ReleaseCheckpointError::WatcherIdentityConflict { task_id: task }, - }, - ) -} - -/// Classify a checkpoint failure as watcher attention, keeping storage failures retryable -fn release_watcher_poll_attention( - error: ReleaseCheckpointError, -) -> Result { - let reason = match error { - ReleaseCheckpointError::Storage(_) - | ReleaseCheckpointError::Serialization(_) - | ReleaseCheckpointError::Task(_) - | ReleaseCheckpointError::Resource(ResourceStoreError::Storage(_)) - | ReleaseCheckpointError::TrainerAssociation( - TrainerAttemptAssociationStoreError::Storage(_), - ) => return Err(error), - ReleaseCheckpointError::Resource(ResourceStoreError::CorruptRecord { .. }) => { - ReleaseWatcherPollAttention::CorruptRecord - } - ReleaseCheckpointError::Resource(ResourceStoreError::WrongAuthority { .. }) => { - ReleaseWatcherPollAttention::WrongAuthority - } - ReleaseCheckpointError::WatcherIntentMissing { .. } => { - ReleaseWatcherPollAttention::WatcherIntentMissing - } - ReleaseCheckpointError::WatcherIdentityConflict { .. } => { - ReleaseWatcherPollAttention::WatcherIdentityConflict - } - ReleaseCheckpointError::TrainerAssociation(_) - | ReleaseCheckpointError::TrainerAssociationMissing { .. } - | ReleaseCheckpointError::TaskMissing { .. } - | ReleaseCheckpointError::TrainerTaskNotRunning { .. } - | ReleaseCheckpointError::TrainerCommandBindingChanged { .. } - | ReleaseCheckpointError::TrainerCancellationConflict { .. } => { - ReleaseWatcherPollAttention::TrainerChanged - } - ReleaseCheckpointError::Watcher(_) - | ReleaseCheckpointError::SelectedCheckpointChanged { .. } => { - ReleaseWatcherPollAttention::PublicationChanged - } - ReleaseCheckpointError::Resource(_) - | ReleaseCheckpointError::NotAwaitingRelease { .. } - | ReleaseCheckpointError::BaselineMissing { .. } - | ReleaseCheckpointError::StopDecisionMissing { .. } - | ReleaseCheckpointError::StopDecisionMismatch { .. } - | ReleaseCheckpointError::Conflict - | ReleaseCheckpointError::InvalidStoredEvidence(_) => { - ReleaseWatcherPollAttention::ActionNotCurrent - } - }; - - Ok(reason) -} - -fn watcher_poll_attention(attention: &WatcherAttention) -> ReleaseWatcherPollAttention { - match attention { - WatcherAttention::LostTask { .. } => ReleaseWatcherPollAttention::TrainerLost, - WatcherAttention::FailedTask { .. } => ReleaseWatcherPollAttention::TrainerFailed, - WatcherAttention::PublicationBeforeTaskStart { .. } => { - ReleaseWatcherPollAttention::PublicationBeforeTrainerStart - } - WatcherAttention::SuccessfulTaskWithoutFinalResult { .. } => { - ReleaseWatcherPollAttention::TrainerCompletedWithoutResult - } - } -} - -fn watcher_routes_match( - by_request: &OriginRoute, - by_task: &OriginRoute, - route_identity: (RequestId, TaskId), - authority: MachineId, - thread: crate::domain::ThreadId, - callback: &CallbackContext, - spec: &NormalizedSpec, -) -> bool { - let (request, task) = route_identity; - let matches = |route: &OriginRoute| { - route.request == request - && route.task == task - && route.origin_machine == authority - && route.execution_machine == authority - && route.thread == thread - && route.callback == *callback - && matches!(&route.submission, SubmissionState::Accepted) - && route.spec.current() == Some(spec) - }; - matches(by_request) && matches(by_task) -} - -fn watcher_identity_matches( - identity: &ExecutorIdentity, - task: TaskId, - authority: MachineId, - spec: &NormalizedSpec, - state: ProcessStatus, -) -> bool { - let ExecutorIdentity::Accepted(record) = identity else { - return false; - }; - record.is_owned_by(task, authority, authority) - && record.state == state - && record.current_spec() == Some(spec) -} diff --git a/src/store/resource/restore.rs b/src/store/resource/restore.rs deleted file mode 100644 index 468763f..0000000 --- a/src/store/resource/restore.rs +++ /dev/null @@ -1,1809 +0,0 @@ -//! Supervisor return decisions and the fixed-identity restore that follows a drained queue -//! -//! An AwaitingReturn loan is the scheduling boundary. Only the exact current -//! supervisor assignment can decide it, and each decision commits its loan -//! transition, task records, resource revision, and replay receipt in one -//! IMMEDIATE transaction. The accepting transaction also saves the task's typed -//! execution mode. A direct-segment trainer closes its Restoring loan on a -//! confirmed start and becomes the registered background task. A native -//! foreground or container task keeps the loan reserved while it runs and closes -//! it only after a successful end with its confirmed exit witness: a -//! process-group exit, or a removed container. Any other end needs an explicit -//! supervisor resolution - -use std::time::Duration; - -use chrono::{DateTime, Utc}; -use rusqlite::{Connection, OptionalExtension, Transaction, TransactionBehavior, params}; -use serde::{Deserialize, Serialize}; - -use super::background::{AdvanceError, advance_resource_on}; -use super::trainer_association::trainer_association_by_task; -use super::trainer_lock::{ - HeldTrainerRelease, TrainerLockReleaseError, TrainerLockReleaseGap, hold_released_trainer_lock, -}; -use super::{LoanActionPhase, replace_loan_in_action_phase_on, select_authority_resource}; -use crate::domain::{ - ContainerId, ExitReason, ProcessStatus, TaskEnv, TaskId, TaskRow, TaskState, WorkExitEvidence, -}; -use crate::error::AppError; -use crate::machine::MachineId; -use crate::resource::bound_action::{ActionTaskReceipt, ResourceActionKind}; -use crate::resource::command_shape::{DirectSegmentCommandShape, same_run_resume_command}; -use crate::resource::foreground::{self, CommandOwnershipContract}; -use crate::resource::store::{ - ResourceStoreError, release_checkpoint_state_for_action, release_completion_for_loan, - return_window_on, save_held_return_window_on, select_non_closed_loan, select_resource, - select_supervisor_notice_record_by_action, -}; -use crate::resource::trainer_publication::revalidate_checkpoint_publication; -use crate::resource::{ - ActionId, Loan, LoanClosure, LoanId, LoanPhase, LoanState, ReleaseCheckpointPhase, Resource, - ResourceId, ResourceRevision, ResourceTaskOwnershipRisk, RestoreAttentionReason, ReturnContext, - ReturnDecision, ReturnDecisionRejection, ReturnDecisionWindow, ReturnExecutionMode, - ReturnHoldRejection, ReturnLaunch, ReturnWork, SameRunResumeGap, SupervisorActionAuthority, - SupervisorAddress, SupervisorNoticePayload, -}; -use crate::spec::{NormalizedSpec, NormalizedTaskWorkload, NormalizedWorkload}; -use crate::store::IdentityError; -use crate::store::identity::{executor_identity_on, origin_route_by_request_on}; -use crate::submission::{ - CallbackContext, CallbackExecutable, ExecutorIdentity, NormalizedSpecSha256, RequestId, - normalized_spec_sha256, -}; - -/// Loan closed by one return decision or restore reconciliation -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) struct ReturnClosure { - /// Closed loan with its retained return context - pub(crate) loan: Loan, - /// Resource revision committed with the closure - pub(crate) state_revision: ResourceRevision, -} - -/// Fixed identities, typed work, and executor context for one return launch -#[derive(Debug, Clone)] -pub(crate) struct ReturnTaskAcceptanceInput { - /// Exact supervisor authority for the pending return action - pub(crate) authority: SupervisorActionAuthority, - /// Fixed request and task identities with the chosen work - pub(crate) launch: ReturnLaunch, - /// Executor environment for a supervisor-supplied command - pub(crate) executor_env: TaskEnv, - /// Where the supervisor callback route lives - pub(crate) origin: ReturnTaskOrigin, -} - -/// Canonical return task derived for one pending decision -#[derive(Debug, Clone)] -pub(crate) struct PreparedReturnTask { - /// Spec the remote supervisor saves in its route - pub(crate) spec: NormalizedSpec, - /// Digest of `spec` - pub(crate) normalized_spec_sha256: NormalizedSpecSha256, -} - -/// Owner of the callback route for one return task -#[derive(Debug, Clone)] -pub(crate) enum ReturnTaskOrigin { - /// The supervisor thread runs on the authority, so the route commits with the task - Local { - /// Codex executable that delivers the supervisor callback - callback_codex: CallbackExecutable, - }, - /// The supervisor machine saved the route and sent the prepared spec digest - Remote { - /// Digest of the spec that the supervisor saved in its route - normalized_spec_sha256: NormalizedSpecSha256, - }, -} - -/// Result of binding one fixed return task to its action -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) enum ReturnTaskAcceptance { - /// The task records, Restoring loan, and receipt committed in this transaction - Inserted { - /// Restoring loan that keeps the resource reserved - loan: Box, - /// Bound task identity - task: TaskId, - /// Resource revision committed with the binding - state_revision: ResourceRevision, - }, - /// An exact earlier binding exists, and its current task state is returned - Existing { - /// Bound task identity - task: TaskId, - /// State retained by the task layer - state: ProcessStatus, - }, - /// The supervisor thread runs on another machine, so no task record was written - UnsupportedRemoteSupervisor { - /// Machine that owns the resource - authority_machine: MachineId, - /// Supervisor that would need a remote callback route - supervisor: SupervisorAddress, - }, -} - -/// Supervisor resolution for a return task that ended before a confirmed start -#[derive(Debug, Clone)] -pub(crate) struct EndedRestoreResolution { - /// Exact supervisor authority for the Restoring loan - pub(crate) authority: SupervisorActionAuthority, - /// Bound task that ended - pub(crate) task_id: TaskId, - /// Supervisor's durable resolution reason - pub(crate) reason: String, -} - -/// Authority observation of one Restoring loan -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) enum RestoreReconcileOutcome { - /// The resource has no Restoring loan - NotRestoring, - /// The bound task row is queued and has not reached its running boundary - Queued { - /// Restoring loan that keeps the resource reserved - loan: Loan, - /// Return action bound to the task - action_id: ActionId, - /// Bound task - task_id: TaskId, - }, - /// The bound direct-segment task has a confirmed start, so the loan closed with it registered - Closed { - /// Closed loan and committed revision - closure: ReturnClosure, - /// Task now registered as the resource background task - task_id: TaskId, - }, - /// The bound native foreground task is running, so the loan keeps the resource reserved - ForegroundRunning { - /// Restoring loan that keeps the resource reserved - loan: Loan, - /// Return action bound to the task - action_id: ActionId, - /// Bound task - task_id: TaskId, - }, - /// The bound native foreground task ended successfully with a confirmed - /// process-group exit, so the loan closed and no task is registered - ForegroundEnded { - /// Closed loan and committed revision - closure: ReturnClosure, - /// Native foreground task that ended - task_id: TaskId, - }, - /// The loan stays reserved because the bound task needs attention - Attention { - /// Restoring loan that keeps the resource reserved - loan: Loan, - /// Return action bound to the task - action_id: ActionId, - /// Bound task - task_id: TaskId, - /// Authority-classified reason - reason: RestoreAttentionReason, - }, -} - -/// A return decision, restore reconciliation, or resolution failed -#[derive(Debug, thiserror::Error)] -pub(crate) enum ReturnDecisionError { - /// Resource authority or stored resource data failed validation - #[error(transparent)] - Resource(#[from] ResourceStoreError), - /// The typed decision does not fit the saved action - #[error(transparent)] - Rejected(#[from] ReturnDecisionRejection), - /// The hold cannot move the deadline of the saved action - #[error(transparent)] - HoldRejected(#[from] ReturnHoldRejection), - /// The decision does not come from the current supervisor assignment - #[error("return decision is not from the current supervisor assignment")] - NotCurrentSupervisor, - /// The loan does not have this pending return action - #[error("return action {action_id:?} is not pending on loan {loan_id:?}")] - ActionNotPending { - /// Loan named by the decision - loan_id: LoanId, - /// Action named by the decision - action_id: ActionId, - }, - /// The saved return notice does not match the loan action - #[error("return action {action_id:?} has an invalid durable notice")] - InvalidReturnNotice { - /// Action named by the decision - action_id: ActionId, - }, - /// The caller's expected revision does not match the saved state - #[error("stale resource revision: expected {expected:?}, found {actual:?}")] - StaleRevision { - /// Revision supplied by the caller - expected: ResourceRevision, - /// Revision saved by the authority - actual: ResourceRevision, - }, - /// The registered background task does not match the return context or is still live - #[error("registered background task does not match the return context")] - BackgroundTaskMismatch, - /// The bound return task has not reached a terminal state - #[error("return task {task_id} has not ended")] - RestoreNotEnded { - /// Bound task - task_id: TaskId, - }, - /// The bound return task ended without proof that its process released the resource - #[error("return task {task_id} ended without proven process release")] - RestoreReleaseUnproven { - /// Bound task - task_id: TaskId, - }, - /// The bound direct-segment return task's wrapper exited, but its trainer lock is not proven free - /// - /// The worker runs in its own session, so only the exact lock named by a - /// trainer-attempt association, held through the closing transaction, proves - /// that the worker released the GPU - #[error("return task {task_id} has no verified trainer lock release: {gap}")] - RestoreOwnershipUnproven { - /// Bound task - task_id: TaskId, - /// Missing or failed part of the lock proof - gap: TrainerLockReleaseGap, - }, - /// A saved receipt holds a different decision for the same action - #[error("return action {action_id:?} was retried with a different decision")] - ConflictingRetry { - /// Action named by the decision - action_id: ActionId, - }, - /// The derived return spec differs from the spec the remote supervisor saved - #[error("return task spec differs from the prepared spec")] - SpecMismatch, - /// The callback origin does not match where the assigned supervisor runs - #[error("return task origin does not match the supervisor machine")] - OriginMismatch, - /// The fixed request or task identity already belongs to other records - #[error("return task identity {task_id} is already used")] - IdentityConflict { - /// Task named by the decision - task_id: TaskId, - }, - /// The resource revision cannot be incremented - #[error("resource revision {revision:?} cannot be incremented")] - RevisionExhausted { - /// Current resource revision - revision: ResourceRevision, - }, - /// Route or executor identity data is invalid - #[error(transparent)] - Identity(#[from] IdentityError), - /// Task preparation or task record insertion failed - #[error("return task records failed: {0}")] - TaskRecords(#[from] AppError), - /// A typed receipt could not be encoded or decoded - #[error("return decision encoding failed: {0}")] - Encoding(#[from] serde_json::Error), - /// SQLite failed - #[error("return decision storage error: {0}")] - Storage(#[from] rusqlite::Error), -} - -/// Saved result of one supervisor return decision -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -enum SavedReturnResult { - /// The no-resume decision closed the loan - Closed { - loan: Loan, - state_revision: ResourceRevision, - }, - /// The launch decision bound one task and entered Restoring - RestoreBound { - loan: Loan, - request_id: RequestId, - task_id: TaskId, - normalized_spec_sha256: NormalizedSpecSha256, - state_revision: ResourceRevision, - /// Mode fixed by the validated ownership contract at acceptance - execution_mode: ReturnExecutionMode, - }, -} - -/// Replay receipt for one supervisor return decision -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct ReturnDecisionReceipt { - authority: SupervisorActionAuthority, - decision: ReturnDecision, - result: SavedReturnResult, -} - -/// Evidence that closed one Restoring loan -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -enum RestoreClosureBasis { - /// The authority observed the bound direct-segment task in its running state - ConfirmedRunning, - /// The bound native foreground task ended successfully, and the task layer - /// confirmed that its owned process group exited - ForegroundEnded { outcome: ExitReason }, - /// The bound container task exited with code 0, and the worker removed the - /// exact container and confirmed that its ID is absent - ContainerEnded { - outcome: ExitReason, - container_id: ContainerId, - }, - /// The supervisor explicitly accepted an end that had proven process release - SupervisorResolvedEnd { - authority: SupervisorActionAuthority, - reason: String, - outcome: ExitReason, - }, -} - -/// Replay receipt for one Restoring loan closure -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct RestoreClosureReceipt { - resource_id: ResourceId, - loan_id: LoanId, - action_id: ActionId, - task_id: TaskId, - basis: RestoreClosureBasis, - loan: Loan, - state_revision: ResourceRevision, -} - -impl ReturnDecisionReceipt { - /// Execution mode of a bound launch, or `None` for a no-resume closure - fn execution_mode(&self) -> Option { - match &self.result { - SavedReturnResult::RestoreBound { execution_mode, .. } => Some(*execution_mode), - SavedReturnResult::Closed { .. } => None, - } - } -} - -/// Read the accepted mode only when a saved decision still proves the current Restoring loan -pub(crate) fn current_return_execution_mode_on( - conn: &Connection, - resource: &Resource, - loan: Option<&Loan>, -) -> Result, ReturnDecisionError> { - let Some(loan) = loan else { - return Ok(None); - }; - let LoanState::Active { - phase: - LoanPhase::Restoring { - action_id, - resume_task_id, - .. - }, - } = &loan.state - else { - return Ok(None); - }; - - let Some(receipt) = saved_decision(conn, *action_id)? else { - return Ok(None); - }; - let ReturnDecision::Launch(launch) = &receipt.decision else { - return Ok(None); - }; - let SavedReturnResult::RestoreBound { - loan: bound_loan, - request_id, - task_id, - normalized_spec_sha256, - execution_mode, - .. - } = &receipt.result - else { - return Ok(None); - }; - if bound_loan != loan - || *task_id != *resume_task_id - || launch.request_id != *request_id - || launch.task_id != *task_id - || receipt.authority.action_id != *action_id - || receipt.authority.loan_id != loan.id - || receipt.authority.resource_id != resource.id - || receipt.authority.authority_machine != resource.authority_machine() - { - return Ok(None); - } - if bound_restore_task( - conn, - &receipt.authority, - *request_id, - *task_id, - *normalized_spec_sha256, - )? - .is_none() - { - return Ok(None); - } - - Ok(Some(*execution_mode)) -} - -/// Bound return task with the evidence its reconciliation needs -struct BoundRestoreTask { - row: TaskRow, - execution_mode: ReturnExecutionMode, - /// Stopped run that a same-run resume continues in the same runtime root - resumed_run: Option, -} - -/// AwaitingReturn state that the exact current supervisor may decide -struct PendingReturn { - resource: Resource, - loan: Loan, - return_context: ReturnContext, -} - -impl crate::store::Store { - /// Close one exact AwaitingReturn loan without starting background work - pub(crate) fn record_no_resume_for_authority( - &mut self, - authority: SupervisorActionAuthority, - reason: String, - ) -> Result { - record_no_resume_for_authority(&mut self.conn, authority, reason) - } - - /// Move the decision deadline of one exact pending return action - pub(crate) fn hold_return_for_authority( - &mut self, - authority: SupervisorActionAuthority, - hold: Duration, - ) -> Result { - hold_return_for_authority(&mut self.conn, authority, hold, Utc::now()) - } - - /// Derive the canonical return task for one pending decision without binding it - pub(crate) fn prepare_return_task_for_authority( - &mut self, - authority: SupervisorActionAuthority, - launch: ReturnLaunch, - executor_env: TaskEnv, - ) -> Result { - prepare_return_task_for_authority(&mut self.conn, authority, launch, executor_env) - } - - /// Bind one fixed return task and enter Restoring on this store connection - pub(crate) fn accept_return_task_for_authority( - &mut self, - input: ReturnTaskAcceptanceInput, - ) -> Result { - accept_return_task_for_authority(&mut self.conn, input) - } - - /// Observe one Restoring loan and close it only on a confirmed start - pub(crate) fn reconcile_restoring_loan_for_authority( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - ) -> Result { - reconcile_restoring_loan_for_authority(&mut self.conn, authority_machine, resource_id) - } - - /// Close a Restoring loan after the supervisor resolves an early end - pub(crate) fn resolve_ended_restore_for_authority( - &mut self, - resolution: EndedRestoreResolution, - ) -> Result { - resolve_ended_restore_for_authority(&mut self.conn, resolution) - } - - /// Read the task identities bound by Restoring loans on this authority - pub(crate) fn restoring_task_ids_for_authority( - &self, - authority_machine: MachineId, - ) -> Result, ReturnDecisionError> { - restoring_task_ids_for_authority(&self.conn, authority_machine) - } -} - -/// Move the decision deadline of one exact AwaitingReturn action -/// -/// A hold is not a decision: it changes neither the loan nor the resource -/// revision, so the supervisor decides later with the same pending action -fn hold_return_for_authority( - conn: &mut Connection, - authority: SupervisorActionAuthority, - hold: Duration, - now: DateTime, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let pending = pending_return(&tx, &authority)?; - let window = return_window_on(&tx, authority.action_id)? - .filter(|window| window.loan_id() == pending.loan.id) - .ok_or(ReturnDecisionError::ActionNotPending { - loan_id: authority.loan_id, - action_id: authority.action_id, - })?; - let held = window.hold(now, hold)?; - save_held_return_window_on(&tx, &window, &held)?; - tx.commit()?; - Ok(held) -} - -/// Close one exact AwaitingReturn loan without starting background work -fn record_no_resume_for_authority( - conn: &mut Connection, - authority: SupervisorActionAuthority, - reason: String, -) -> Result { - let decision = ReturnDecision::NoResume { reason }; - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - if let Some(receipt) = saved_decision(&tx, authority.action_id)? { - let SavedReturnResult::Closed { - loan, - state_revision, - } = replayed_result(receipt, &authority, &decision)? - else { - return Err(ReturnDecisionError::ConflictingRetry { - action_id: authority.action_id, - }); - }; - tx.commit()?; - return Ok(ReturnClosure { - loan, - state_revision, - }); - } - - let pending = pending_return(&tx, &authority)?; - decision.validate_for(&pending.return_context, authority.supervisor.thread)?; - require_prior_background_ended(&tx, &pending)?; - let ReturnDecision::NoResume { reason } = &decision else { - unreachable!("the no-resume decision was built above"); - }; - - let loan = Loan { - id: pending.loan.id, - resource_id: pending.loan.resource_id, - state: LoanState::Closed { - result: LoanClosure::NoResume { - return_context: pending.return_context.clone(), - reason: reason.clone(), - }, - }, - }; - // neither the stopped nor the completed run is live, and no work replaces it - let state_revision = advance_resource(&tx, &pending.resource, None)?; - update_active_loan( - &tx, - &loan, - LoanActionPhase::AwaitingReturn, - authority.action_id, - )?; - insert_decision_receipt( - &tx, - &ReturnDecisionReceipt { - authority, - decision: decision.clone(), - result: SavedReturnResult::Closed { - loan: loan.clone(), - state_revision, - }, - }, - )?; - tx.commit()?; - - Ok(ReturnClosure { - loan, - state_revision, - }) -} - -/// Derive the canonical return task for one pending decision without binding it -/// -/// A remote supervisor saves this exact spec in its callback route before it -/// sends the launch. A decision already bound returns the spec it accepted -fn prepare_return_task_for_authority( - conn: &mut Connection, - authority: SupervisorActionAuthority, - launch: ReturnLaunch, - executor_env: TaskEnv, -) -> Result { - let decision = ReturnDecision::Launch(Box::new(launch.clone())); - let tx = conn.transaction_with_behavior(TransactionBehavior::Deferred)?; - if let Some(receipt) = saved_decision(&tx, authority.action_id)? { - let SavedReturnResult::RestoreBound { task_id, .. } = - replayed_result(receipt, &authority, &decision)? - else { - return Err(ReturnDecisionError::ConflictingRetry { - action_id: authority.action_id, - }); - }; - let Some(ExecutorIdentity::Accepted(record)) = executor_identity_on(&tx, task_id)? else { - return Err(ReturnDecisionError::IdentityConflict { task_id }); - }; - let spec = record - .current_spec() - .cloned() - .ok_or(ReturnDecisionError::IdentityConflict { task_id })?; - let normalized_spec_sha256 = normalized_spec_sha256(&spec)?; - return Ok(PreparedReturnTask { - spec, - normalized_spec_sha256, - }); - } - - let pending = pending_return(&tx, &authority)?; - decision.validate_for(&pending.return_context, authority.supervisor.thread)?; - require_prior_background_ended(&tx, &pending)?; - let (spec, _, _) = return_task_spec(&tx, &pending, &authority, &launch, executor_env)?; - crate::spec::check_spec_host(&spec).map_err(AppError::from)?; - let normalized_spec_sha256 = normalized_spec_sha256(&spec)?; - Ok(PreparedReturnTask { - spec, - normalized_spec_sha256, - }) -} - -/// Derive the spec, executor environment, and binary for one return launch -fn return_task_spec( - conn: &Connection, - pending: &PendingReturn, - authority: &SupervisorActionAuthority, - launch: &ReturnLaunch, - executor_env: TaskEnv, -) -> Result<(NormalizedSpec, TaskEnv, std::path::PathBuf), ReturnDecisionError> { - let Some(spec) = launch.work.supervisor_spec() else { - return same_run_resume(conn, pending, authority.supervisor.thread); - }; - let spec = spec.as_normalized().clone(); - let binary = - crate::invocation::resolve_workload_binary(&spec.workload, &executor_env.path, &spec.cwd)?; - Ok((spec, executor_env, binary)) -} - -/// Check the return task row against the ownership contract its work selected -/// -/// A foreground command needs an inspectable native entry point. A trainer argv -/// must pass the full direct-segment check, because only that shape has an -/// ownership lock that later release proof can verify. A container needs no -/// entry point: Homebased drives the `docker` CLI itself and removes the -/// container. The validated contract fixes the task's execution mode -fn check_return_command_ownership( - launch: &ReturnLaunch, - spec: &NormalizedSpec, - row: &TaskRow, -) -> Result { - // the resume argv comes from the stopped run's verified trainer shape - if matches!(launch.work, ReturnWork::SameRunResume { .. }) { - return Ok(ReturnExecutionMode::DirectSegmentTrainer); - } - let unsupported = |risk| { - ReturnDecisionError::Rejected(ReturnDecisionRejection::UnsupportedCommandOwnership { risk }) - }; - let contract = - CommandOwnershipContract::for_return_work(&spec.workload).map_err(unsupported)?; - match contract { - CommandOwnershipContract::ForegroundExecutable => { - foreground::inspect_foreground_entry_point(&row.binary).map_err(unsupported)?; - } - CommandOwnershipContract::DirectSegmentTrainer => { - DirectSegmentCommandShape::validate_launch(spec, row) - .map_err(|_| unsupported(ResourceTaskOwnershipRisk::Interpreter))?; - } - CommandOwnershipContract::Container => {} - } - - Ok(contract.into()) -} - -/// Bind one fixed return task to an exact AwaitingReturn action and enter Restoring -/// -/// The task row, accepted executor identity, first queued event, Restoring phase, -/// resource revision, and receipt commit together. A local origin also commits its -/// callback route; a remote origin commits the exact action-task receipt instead, -/// because its route was saved on the supervisor machine. The caller may spawn a -/// worker only for an `Inserted` result -fn accept_return_task_for_authority( - conn: &mut Connection, - input: ReturnTaskAcceptanceInput, -) -> Result { - let ReturnTaskAcceptanceInput { - authority, - launch, - executor_env, - origin, - } = input; - let decision = ReturnDecision::Launch(Box::new(launch.clone())); - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - if let Some(receipt) = saved_decision(&tx, authority.action_id)? { - let SavedReturnResult::RestoreBound { - request_id, - task_id, - normalized_spec_sha256, - .. - } = replayed_result(receipt, &authority, &decision)? - else { - return Err(ReturnDecisionError::ConflictingRetry { - action_id: authority.action_id, - }); - }; - if let ReturnTaskOrigin::Remote { - normalized_spec_sha256: requested, - } = &origin - && *requested != normalized_spec_sha256 - { - return Err(ReturnDecisionError::SpecMismatch); - } - let row = bound_restore_task(&tx, &authority, request_id, task_id, normalized_spec_sha256)? - .ok_or(ReturnDecisionError::IdentityConflict { task_id })?; - tx.commit()?; - return Ok(ReturnTaskAcceptance::Existing { - task: task_id, - state: row.status(), - }); - } - - let pending = pending_return(&tx, &authority)?; - decision.validate_for(&pending.return_context, authority.supervisor.thread)?; - let remote_supervisor = authority.supervisor.machine != authority.authority_machine; - match &origin { - // a local route for a remote supervisor thread would send callbacks to the - // wrong machine, so the local path writes nothing for it - ReturnTaskOrigin::Local { .. } if remote_supervisor => { - return Ok(ReturnTaskAcceptance::UnsupportedRemoteSupervisor { - authority_machine: authority.authority_machine, - supervisor: authority.supervisor, - }); - } - ReturnTaskOrigin::Remote { .. } if !remote_supervisor => { - return Err(ReturnDecisionError::OriginMismatch); - } - ReturnTaskOrigin::Local { .. } | ReturnTaskOrigin::Remote { .. } => {} - } - require_prior_background_ended(&tx, &pending)?; - - let task_id = launch.task_id; - let (spec, env, binary) = return_task_spec(&tx, &pending, &authority, &launch, executor_env)?; - if super::task_identity_is_used(&tx, launch.request_id, task_id)? { - return Err(ReturnDecisionError::IdentityConflict { task_id }); - } - - crate::spec::check_spec_host(&spec).map_err(AppError::from)?; - let row = crate::store::new_queued_task(crate::store::NewTask { - id: task_id, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: crate::invocation::persist_workload(&spec.workload), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env: env.clone(), - binary, - }); - let execution_mode = check_return_command_ownership(&launch, &spec, &row)?; - match origin { - ReturnTaskOrigin::Local { callback_codex } => { - let callback = CallbackContext { - env, - cwd: spec.cwd.clone(), - codex: callback_codex, - }; - crate::store::insert_local_task_records_on( - &tx, - &row, - &spec, - authority.authority_machine, - launch.request_id, - &callback, - None, - ) - .map_err(|error| match error { - AppError::ClusterTaskConflict { .. } => { - ReturnDecisionError::IdentityConflict { task_id } - } - error => ReturnDecisionError::TaskRecords(error), - })?; - } - ReturnTaskOrigin::Remote { - normalized_spec_sha256: requested, - } => { - if requested != normalized_spec_sha256(&spec)? { - return Err(ReturnDecisionError::SpecMismatch); - } - let receipt = ActionTaskReceipt { - kind: ResourceActionKind::Return, - authority, - request_id: launch.request_id, - task_id, - normalized_spec_sha256: requested, - }; - super::action_task::insert_action_task(&tx, &row, &spec, &receipt).map_err( - |error| match error { - super::action_task::ResourceActionError::Rejected(_) => { - ReturnDecisionError::IdentityConflict { task_id } - } - super::action_task::ResourceActionError::Storage(error) => { - ReturnDecisionError::TaskRecords(error) - } - }, - )?; - } - } - - let loan = Loan { - id: pending.loan.id, - resource_id: pending.loan.resource_id, - state: LoanState::Active { - phase: LoanPhase::Restoring { - action_id: authority.action_id, - return_context: pending.return_context.clone(), - resume_task_id: task_id, - }, - }, - }; - // the prior task stays registered until a direct-segment task has a confirmed - // start or the loan closes without registering its foreground task - let state_revision = advance_resource( - &tx, - &pending.resource, - pending.resource.registered_background_task, - )?; - update_active_loan( - &tx, - &loan, - LoanActionPhase::AwaitingReturn, - authority.action_id, - )?; - insert_decision_receipt( - &tx, - &ReturnDecisionReceipt { - authority, - decision, - result: SavedReturnResult::RestoreBound { - loan: loan.clone(), - request_id: launch.request_id, - task_id, - normalized_spec_sha256: normalized_spec_sha256(&spec)?, - state_revision, - execution_mode, - }, - }, - )?; - tx.commit()?; - - Ok(ReturnTaskAcceptance::Inserted { - loan: Box::new(loan), - task: task_id, - state_revision, - }) -} - -/// Observe one Restoring loan and close it by the task's saved execution mode -/// -/// A direct-segment task closes the loan on a confirmed start and becomes the -/// registered background task. A native foreground or container task keeps the -/// loan while it runs and closes it only after a successful end with its -/// confirmed exit witness, leaving no registered task. Every other state keeps -/// the loan reserved -fn reconcile_restoring_loan_for_authority( - conn: &mut Connection, - authority_machine: MachineId, - resource_id: ResourceId, -) -> Result { - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - let resource = select_authority_resource(&tx, authority_machine, resource_id)?; - let Some(loan) = select_non_closed_loan(&tx, resource_id)? else { - return Ok(RestoreReconcileOutcome::NotRestoring); - }; - let LoanState::Active { - phase: - LoanPhase::Restoring { - action_id, - return_context, - resume_task_id, - }, - } = &loan.state - else { - return Ok(RestoreReconcileOutcome::NotRestoring); - }; - let (action_id, task_id) = (*action_id, *resume_task_id); - let attention = |reason| { - Ok(RestoreReconcileOutcome::Attention { - loan: loan.clone(), - action_id, - task_id, - reason, - }) - }; - - let Some(bound) = restoring_task_row(&tx, &resource, &loan, action_id, task_id)? else { - return attention(RestoreAttentionReason::IdentityMismatch); - }; - let (closed, registered, basis) = match (&bound.row.state, bound.execution_mode) { - (TaskState::Queued, _) => { - return Ok(RestoreReconcileOutcome::Queued { - loan: loan.clone(), - action_id, - task_id, - }); - } - (TaskState::Lost, _) => return attention(RestoreAttentionReason::Lost), - (TaskState::Running { .. }, ReturnExecutionMode::DirectSegmentTrainer) => ( - LoanClosure::Resumed { - return_context: return_context.clone(), - task_id, - }, - Some(task_id), - RestoreClosureBasis::ConfirmedRunning, - ), - (TaskState::Finished { .. }, ReturnExecutionMode::DirectSegmentTrainer) => { - return attention(RestoreAttentionReason::EndedBeforeConfirmedStart { - state: bound.row.status(), - }); - } - ( - TaskState::Running { .. }, - ReturnExecutionMode::NativeForeground | ReturnExecutionMode::Container, - ) => { - return Ok(RestoreReconcileOutcome::ForegroundRunning { - loan: loan.clone(), - action_id, - task_id, - }); - } - (TaskState::Finished { reason }, mode) => { - let basis = match loan_holding_end(&bound.row, reason, mode) { - Ok(basis) => basis, - Err(reason) => return attention(reason), - }; - ( - LoanClosure::ForegroundReturnEnded { - return_context: return_context.clone(), - task_id, - outcome: reason.clone(), - }, - None, - basis, - ) - } - }; - - let closed = Loan { - id: loan.id, - resource_id, - state: LoanState::Closed { result: closed }, - }; - // a foreground end also clears the ended prior registration, so the queue - // can serve its next request from this closure's idle boundary - let state_revision = advance_resource(&tx, &resource, registered)?; - update_active_loan(&tx, &closed, LoanActionPhase::Restoring, action_id)?; - insert_closure_receipt( - &tx, - &RestoreClosureReceipt { - resource_id, - loan_id: loan.id, - action_id, - task_id, - basis, - loan: closed.clone(), - state_revision, - }, - )?; - tx.commit()?; - - let closure = ReturnClosure { - loan: closed, - state_revision, - }; - Ok(match registered { - Some(task_id) => RestoreReconcileOutcome::Closed { closure, task_id }, - None => RestoreReconcileOutcome::ForegroundEnded { closure, task_id }, - }) -} - -/// Return the closure basis of a return task that held its loan, or why its end cannot close it -/// -/// Only a zero exit with the witness of the task's mode is free: a confirmed -/// owned process-group exit for a native command, or a removed container for a -/// container. A failed end needs the supervisor's explicit resolution, and an -/// unconfirmed witness is not proof that the work released the GPU -fn loan_holding_end( - row: &TaskRow, - reason: &ExitReason, - mode: ReturnExecutionMode, -) -> Result { - let state = row.status(); - let success = *reason == ExitReason::Exit { code: 0 }; - let evidence = row.work_exit_evidence(); - match (mode, evidence) { - (ReturnExecutionMode::NativeForeground, WorkExitEvidence::ProcessGroupExited) - if success => - { - Ok(RestoreClosureBasis::ForegroundEnded { - outcome: reason.clone(), - }) - } - ( - ReturnExecutionMode::Container, - WorkExitEvidence::ContainerRemoved { container_id, .. }, - ) if success => Ok(RestoreClosureBasis::ContainerEnded { - outcome: reason.clone(), - container_id, - }), - ( - ReturnExecutionMode::NativeForeground, - WorkExitEvidence::ProcessGroupExited | WorkExitEvidence::NoWorkStarted, - ) => Err(RestoreAttentionReason::ForegroundEnded { state }), - (ReturnExecutionMode::NativeForeground, _) => { - Err(RestoreAttentionReason::ForegroundExitUnconfirmed { state }) - } - ( - ReturnExecutionMode::Container, - WorkExitEvidence::ContainerRemoved { .. } | WorkExitEvidence::NoWorkStarted, - ) => Err(RestoreAttentionReason::ContainerEnded { state }), - (ReturnExecutionMode::Container, _) => { - Err(RestoreAttentionReason::ContainerExitUnconfirmed { state }) - } - (ReturnExecutionMode::DirectSegmentTrainer, _) => { - Err(RestoreAttentionReason::EndedBeforeConfirmedStart { state }) - } - } -} - -/// Close a Restoring loan whose bound task ended without a mode-specific closure -/// -/// This covers a direct-segment task that ended before a confirmed start, and a -/// native foreground or container task that ended without success. The exact -/// current supervisor must name the action, task, and current revision. The task -/// must be terminal with the proven release its mode needs; a lost task stays -/// reserved -fn resolve_ended_restore_for_authority( - conn: &mut Connection, - resolution: EndedRestoreResolution, -) -> Result { - let EndedRestoreResolution { - authority, - task_id, - reason, - } = resolution; - let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; - if let Some(receipt) = saved_closure(&tx, authority.action_id)? { - let RestoreClosureBasis::SupervisorResolvedEnd { - authority: saved_authority, - reason: saved_reason, - .. - } = &receipt.basis - else { - return Err(ReturnDecisionError::ActionNotPending { - loan_id: authority.loan_id, - action_id: authority.action_id, - }); - }; - if *saved_authority != authority || *saved_reason != reason || receipt.task_id != task_id { - return Err(ReturnDecisionError::ConflictingRetry { - action_id: authority.action_id, - }); - } - tx.commit()?; - return Ok(ReturnClosure { - loan: receipt.loan, - state_revision: receipt.state_revision, - }); - } - - if reason.trim().is_empty() { - return Err(ReturnDecisionRejection::EmptyReason.into()); - } - let resource = current_supervisor_resource(&tx, &authority)?; - let not_pending = || ReturnDecisionError::ActionNotPending { - loan_id: authority.loan_id, - action_id: authority.action_id, - }; - let loan = select_non_closed_loan(&tx, authority.resource_id)? - .filter(|loan| loan.id == authority.loan_id) - .ok_or_else(not_pending)?; - let LoanState::Active { - phase: - LoanPhase::Restoring { - action_id, - return_context, - resume_task_id, - }, - } = &loan.state - else { - return Err(not_pending()); - }; - if *action_id != authority.action_id || *resume_task_id != task_id { - return Err(not_pending()); - } - if resource.state_revision != authority.expected_state_revision { - return Err(ReturnDecisionError::StaleRevision { - expected: authority.expected_state_revision, - actual: resource.state_revision, - }); - } - - let bound = restoring_task_row(&tx, &resource, &loan, *action_id, task_id)? - .ok_or(ReturnDecisionError::IdentityConflict { task_id })?; - let row = &bound.row; - let outcome = match &row.state { - TaskState::Finished { reason } => reason.clone(), - TaskState::Lost => return Err(ReturnDecisionError::RestoreReleaseUnproven { task_id }), - TaskState::Queued | TaskState::Running { .. } => { - return Err(ReturnDecisionError::RestoreNotEnded { task_id }); - } - }; - // the task layer records no started work only before its child spawns or its - // container starts, so the other outcomes need the exit witness of the task's - // mode, and a trainer also needs its released lock - let held_lock = match (row.work_exit_evidence(), bound.execution_mode) { - ( - WorkExitEvidence::ProcessGroupExited, - ReturnExecutionMode::NativeForeground | ReturnExecutionMode::DirectSegmentTrainer, - ) => hold_restore_release(&tx, &resource, &bound)?, - (WorkExitEvidence::ContainerRemoved { .. }, ReturnExecutionMode::Container) => None, - (WorkExitEvidence::NoWorkStarted, _) - if matches!( - outcome, - ExitReason::SpawnFailed { .. } | ExitReason::Cancelled - ) => - { - None - } - _ => return Err(ReturnDecisionError::RestoreReleaseUnproven { task_id }), - }; - - let closed = Loan { - id: loan.id, - resource_id: loan.resource_id, - state: LoanState::Closed { - result: LoanClosure::RestoreEnded { - return_context: return_context.clone(), - task_id, - outcome: outcome.clone(), - reason: reason.clone(), - }, - }, - }; - let state_revision = advance_resource(&tx, &resource, None)?; - update_active_loan( - &tx, - &closed, - LoanActionPhase::Restoring, - authority.action_id, - )?; - insert_closure_receipt( - &tx, - &RestoreClosureReceipt { - resource_id: loan.resource_id, - loan_id: loan.id, - action_id: authority.action_id, - task_id, - basis: RestoreClosureBasis::SupervisorResolvedEnd { - authority, - reason, - outcome, - }, - loan: closed.clone(), - state_revision, - }, - )?; - if let Some(held) = &held_lock { - held.recheck() - .map_err(|gap| ReturnDecisionError::RestoreOwnershipUnproven { task_id, gap })?; - } - tx.commit()?; - // the closure lets new GPU work start, so the lock stays held until it commits - drop(held_lock); - - Ok(ReturnClosure { - loan: closed, - state_revision, - }) -} - -/// Prove that a return task whose wrapper exited released the GPU -/// -/// The saved execution mode selects the witness. A native foreground command -/// needs only its confirmed process-group exit. A direct-segment trainer needs -/// its exact released lock. A same-run resume keeps the stopped run's runtime -/// root, so that run's association names the lock; other trainer work has no -/// association before a confirmed start and fails closed -fn hold_restore_release( - conn: &Connection, - resource: &Resource, - bound: &BoundRestoreTask, -) -> Result, ReturnDecisionError> { - let task_id = bound.row.id; - let mode = bound.execution_mode; - if mode.holds_loan_while_running() { - return Ok(None); - } - let witness_task = bound.resumed_run.unwrap_or(task_id); - - hold_released_trainer_lock(conn, resource, task_id, witness_task) - .map(Some) - .map_err(|error| match error { - TrainerLockReleaseError::Unproven(gap) => { - ReturnDecisionError::RestoreOwnershipUnproven { task_id, gap } - } - TrainerLockReleaseError::Store(error) => ReturnDecisionError::Resource(error), - }) -} - -/// Read the task identities bound by Restoring loans on this authority -fn restoring_task_ids_for_authority( - conn: &Connection, - authority_machine: MachineId, -) -> Result, ReturnDecisionError> { - let mut statement = conn.prepare( - "SELECT json_extract(l.state_json, '$.phase.resume_task_id') - FROM loans l JOIN resources r ON r.id = l.resource_id - WHERE r.authority_machine = ?1 - AND json_extract(l.state_json, '$.type') = 'active' - AND json_extract(l.state_json, '$.phase.type') = 'restoring'", - )?; - let ids = statement - .query_map([authority_machine.as_uuid().to_string()], |row| { - row.get::<_, String>(0) - })? - .collect::, _>>()?; - ids.into_iter() - .map(|id| { - id.parse().map_err(|_| { - ReturnDecisionError::TaskRecords(AppError::Internal { - message: format!("restoring loan has an invalid task identity {id}"), - }) - }) - }) - .collect() -} - -fn current_supervisor_resource( - conn: &Connection, - authority: &SupervisorActionAuthority, -) -> Result { - if authority.supervisor.thread.0.is_nil() { - return Err(ReturnDecisionRejection::InvalidIdentity.into()); - } - let resource = - select_authority_resource(conn, authority.authority_machine, authority.resource_id)?; - if resource.supervisor != authority.supervisor - || resource.assignment_revision != authority.assignment_revision - { - return Err(ReturnDecisionError::NotCurrentSupervisor); - } - - Ok(resource) -} - -fn pending_return( - conn: &Connection, - authority: &SupervisorActionAuthority, -) -> Result { - let resource = current_supervisor_resource(conn, authority)?; - let not_pending = || ReturnDecisionError::ActionNotPending { - loan_id: authority.loan_id, - action_id: authority.action_id, - }; - let loan = select_non_closed_loan(conn, authority.resource_id)? - .filter(|loan| loan.id == authority.loan_id) - .ok_or_else(not_pending)?; - let LoanState::Active { - phase: - LoanPhase::AwaitingReturn { - action_id, - return_context, - }, - } = &loan.state - else { - return Err(not_pending()); - }; - if *action_id != authority.action_id { - return Err(not_pending()); - } - - let (notice, _) = select_supervisor_notice_record_by_action(conn, authority.action_id) - .map_err(ResourceStoreError::from)? - .ok_or(ReturnDecisionError::InvalidReturnNotice { - action_id: authority.action_id, - })?; - if notice.loan_id != loan.id - || notice.payload - != (SupervisorNoticePayload::ReturnRequired { - return_context: return_context.clone(), - }) - { - return Err(ReturnDecisionError::InvalidReturnNotice { - action_id: authority.action_id, - }); - } - // the reservation revision is the decision boundary; queued requests accepted - // after it do not change the revision, and only the expired decision window - // lets one of them supersede this action - for actual in [notice.state_revision, resource.state_revision] { - if actual != authority.expected_state_revision { - return Err(ReturnDecisionError::StaleRevision { - expected: authority.expected_state_revision, - actual, - }); - } - } - - let return_context = return_context.clone(); - Ok(PendingReturn { - resource, - loan, - return_context, - }) -} - -// the return context names the run that released the resource; it must still be -// the registered task and must not be live before anything replaces it -fn require_prior_background_ended( - conn: &Connection, - pending: &PendingReturn, -) -> Result<(), ReturnDecisionError> { - let registered = pending.resource.registered_background_task; - let Some(task_id) = pending.return_context.released_task() else { - return match registered { - None => Ok(()), - Some(_) => Err(ReturnDecisionError::BackgroundTaskMismatch), - }; - }; - if registered != Some(task_id) { - return Err(ReturnDecisionError::BackgroundTaskMismatch); - } - let row = crate::store::task_by_id_on(conn, task_id)? - .ok_or(ReturnDecisionError::BackgroundTaskMismatch)?; - // only an operator attestation releases a lost trainer, and it saved the lost context - let ended = match &pending.return_context { - ReturnContext::LostWithoutResult { .. } => row.state.is_terminal(), - ReturnContext::Stopped { .. } - | ReturnContext::AlreadyCompleted { .. } - | ReturnContext::EndedWithoutResult { .. } - | ReturnContext::Idle => matches!(row.state, TaskState::Finished { .. }), - }; - if !ended { - return Err(ReturnDecisionError::BackgroundTaskMismatch); - } - - Ok(()) -} - -/// Return the direct-segment return launch that registered one ended trainer -/// -/// The closure receipt must record the confirmed start that registered the -/// task, and the bound decision must name this resource, a direct-segment -/// execution mode, and the task's matching accepted identity and callback owner -/// The result is the return action, its request, and the accepted spec digest -pub(super) fn direct_segment_return_of_registered_trainer_on( - conn: &Connection, - resource: &Resource, - task_id: TaskId, -) -> Result, ReturnDecisionError> { - let closure: Option = conn - .query_row( - "SELECT receipt_json FROM resource_restore_closures WHERE task_id = ?1", - [task_id.to_string()], - |row| row.get(0), - ) - .optional()?; - let Some(closure) = closure else { - return Ok(None); - }; - let closure: RestoreClosureReceipt = serde_json::from_str(&closure)?; - if closure.task_id != task_id - || closure.resource_id != resource.id - || !matches!(closure.basis, RestoreClosureBasis::ConfirmedRunning) - { - return Ok(None); - } - let Some(decision) = saved_decision(conn, closure.action_id)? else { - return Ok(None); - }; - let SavedReturnResult::RestoreBound { - request_id, - task_id: bound_task, - normalized_spec_sha256, - .. - } = &decision.result - else { - return Ok(None); - }; - let (request_id, digest) = (*request_id, *normalized_spec_sha256); - if *bound_task != task_id - || decision.authority.resource_id != resource.id - || decision.authority.authority_machine != resource.authority_machine() - || decision.execution_mode() != Some(ReturnExecutionMode::DirectSegmentTrainer) - || bound_restore_task(conn, &decision.authority, request_id, task_id, digest)?.is_none() - { - return Ok(None); - } - - Ok(Some((closure.action_id, request_id, digest))) -} - -/// Return the return task of one execution mode that one Restoring loan still binds -/// -/// The saved decision must bind this exact task to the loan and action with -/// `mode` as its accepted execution mode, and the task's accepted identity and -/// callback owner must still match. The result is the return request and the -/// accepted spec digest. A decision whose mode cannot be proven never matches -pub(super) fn restoring_return_of_mode_on( - conn: &Connection, - resource: &Resource, - loan: &Loan, - action_id: ActionId, - task_id: TaskId, - mode: ReturnExecutionMode, -) -> Result, ReturnDecisionError> { - let Some(decision) = saved_decision(conn, action_id)? else { - return Ok(None); - }; - let SavedReturnResult::RestoreBound { - loan: bound_loan, - request_id, - task_id: bound_task, - normalized_spec_sha256, - .. - } = &decision.result - else { - return Ok(None); - }; - let (request_id, digest) = (*request_id, *normalized_spec_sha256); - if *bound_task != task_id - || bound_loan.id != loan.id - || decision.authority.action_id != action_id - || decision.authority.loan_id != loan.id - || decision.authority.resource_id != resource.id - || decision.authority.authority_machine != resource.authority_machine() - || decision.execution_mode() != Some(mode) - || bound_restore_task(conn, &decision.authority, request_id, task_id, digest)?.is_none() - { - return Ok(None); - } - - Ok(Some((request_id, digest))) -} - -/// Derive the resume command from the stopped run's saved records -/// -/// The verified release receipt, committed checkpoint decision, trainer -/// association, accepted identity, and task row must all name the same run -/// Only the callback thread and `--resume` generation differ from that run -fn same_run_resume( - conn: &Connection, - pending: &PendingReturn, - supervisor_thread: crate::domain::ThreadId, -) -> Result<(NormalizedSpec, TaskEnv, std::path::PathBuf), ReturnDecisionError> { - let ReturnContext::Stopped { - task_id, - checkpoint_ref, - recovery_ref, - } = &pending.return_context - else { - return Err(ReturnDecisionRejection::ResumeRequiresStoppedContext.into()); - }; - let unproven = |gap| ReturnDecisionError::from(ReturnDecisionRejection::ResumeUnproven { gap }); - let resource = &pending.resource; - - let (release_action, released_context) = - release_completion_for_loan(conn, resource.id, pending.loan.id)? - .ok_or_else(|| unproven(SameRunResumeGap::ReleaseReceiptMissing))?; - if released_context != pending.return_context { - return Err(unproven(SameRunResumeGap::ReleaseReceiptMissing)); - } - let (checkpoint_state, _) = - release_checkpoint_state_for_action(conn, resource.id, release_action) - .ok() - .flatten() - .ok_or_else(|| unproven(SameRunResumeGap::CheckpointDecisionMismatch))?; - let ReleaseCheckpointPhase::CancellationCommitted { decision, .. } = checkpoint_state.phase - else { - return Err(unproven(SameRunResumeGap::CheckpointDecisionMismatch)); - }; - let checkpoint = &decision.selected_checkpoint; - if checkpoint_state.action.observed_background_task != *task_id - || checkpoint.generation_id != *recovery_ref - || *checkpoint_ref - != format!( - "{}#sha256={}", - checkpoint.path.display(), - checkpoint.record_sha256 - ) - { - return Err(unproven(SameRunResumeGap::CheckpointDecisionMismatch)); - } - - let association = trainer_association_by_task(conn, *task_id) - .ok() - .flatten() - .filter(|association| { - association.resource_id() == resource.id - && association.authority_machine() == resource.authority_machine() - }) - .ok_or_else(|| unproven(SameRunResumeGap::AssociationMissing))?; - let Some(ExecutorIdentity::Accepted(record)) = executor_identity_on(conn, *task_id)? else { - return Err(unproven(SameRunResumeGap::RunRecordsChanged)); - }; - let original = record - .current_spec() - .ok_or_else(|| unproven(SameRunResumeGap::RunRecordsChanged))?; - let row = crate::store::task_by_id_on(conn, *task_id)? - .ok_or_else(|| unproven(SameRunResumeGap::RunRecordsChanged))?; - if !record.is_executed_by(*task_id, resource.authority_machine()) - || normalized_spec_sha256(original)? != association.normalized_spec_sha256() - || !super::resource_task_row_matches(&row, *task_id, original) - || !matches!( - row.state, - TaskState::Finished { - reason: ExitReason::Cancelled - } - ) - { - return Err(unproven(SameRunResumeGap::RunRecordsChanged)); - } - - let attempt = association.verified_attempt(); - if !matches!( - revalidate_checkpoint_publication( - attempt.canonical_runtime_root(), - attempt.binding(), - checkpoint - ), - Ok(true) - ) { - return Err(unproven(SameRunResumeGap::CheckpointUnavailable)); - } - - let NormalizedWorkload::Task(workload) = &original.workload else { - return Err(unproven(SameRunResumeGap::CommandShapeInvalid)); - }; - let command = same_run_resume_command(&workload.command, recovery_ref) - .map_err(|_| unproven(SameRunResumeGap::CommandShapeInvalid))?; - let spec = NormalizedSpec { - api_version: original.api_version, - thread: supervisor_thread, - name: original.name.clone(), - cwd: row.cwd.clone(), - machine: None, - timeout: original.timeout, - workload: NormalizedWorkload::Task(NormalizedTaskWorkload { command }), - }; - let binary = - crate::invocation::resolve_workload_binary(&spec.workload, &row.env.path, &row.cwd) - .map_err(|_| unproven(SameRunResumeGap::InterpreterChanged))?; - if binary != row.binary { - return Err(unproven(SameRunResumeGap::InterpreterChanged)); - } - - Ok((spec, row.env, binary)) -} - -/// Return the bound task when its receipt, route, and accepted identity match -fn restoring_task_row( - conn: &Connection, - resource: &Resource, - loan: &Loan, - action_id: ActionId, - task_id: TaskId, -) -> Result, ReturnDecisionError> { - let Some(receipt) = saved_decision(conn, action_id)? else { - return Ok(None); - }; - let ReturnDecision::Launch(launch) = &receipt.decision else { - return Ok(None); - }; - let resumed_run = match &launch.work { - ReturnWork::SameRunResume { stopped_task, .. } => Some(*stopped_task), - ReturnWork::EvaluationOrNextEpoch { .. } - | ReturnWork::NewBackgroundWork { .. } - | ReturnWork::AfterEndedRun { .. } => None, - }; - let SavedReturnResult::RestoreBound { - request_id, - task_id: bound_task, - normalized_spec_sha256, - execution_mode, - .. - } = receipt.result - else { - return Ok(None); - }; - if bound_task != task_id - || receipt.authority.loan_id != loan.id - || receipt.authority.resource_id != resource.id - || receipt.authority.authority_machine != resource.authority_machine() - { - return Ok(None); - } - - let row = bound_restore_task( - conn, - &receipt.authority, - request_id, - task_id, - normalized_spec_sha256, - )?; - Ok(row.map(|row| BoundRestoreTask { - row, - execution_mode, - resumed_run, - })) -} - -/// Return the bound task row when its identity and callback owner match the decision -/// -/// The deciding supervisor's machine is the callback origin. A local origin keeps -/// its route on the authority; a remote origin keeps the exact action-task receipt -fn bound_restore_task( - conn: &Connection, - authority: &SupervisorActionAuthority, - request_id: RequestId, - task_id: TaskId, - digest: NormalizedSpecSha256, -) -> Result, ReturnDecisionError> { - let authority_machine = authority.authority_machine; - if authority.supervisor.machine != authority_machine { - let expected = ActionTaskReceipt { - kind: ResourceActionKind::Return, - authority: *authority, - request_id, - task_id, - normalized_spec_sha256: digest, - }; - let saved = super::action_task::action_task_receipt_by_task(conn, task_id)?; - if saved != Some(expected) { - return Ok(None); - } - return Ok(super::action_task::remote_action_task_row_on( - conn, &expected, - )?); - } - let Some(row) = crate::store::task_by_id_on(conn, task_id)? else { - return Ok(None); - }; - let Some(ExecutorIdentity::Accepted(record)) = executor_identity_on(conn, task_id)? else { - return Ok(None); - }; - let Some(route) = origin_route_by_request_on(conn, request_id)? else { - return Ok(None); - }; - let spec_matches = record - .current_spec() - .map(normalized_spec_sha256) - .transpose()? - .is_some_and(|saved| saved == digest); - - Ok( - (record.is_owned_by(task_id, authority_machine, authority_machine) - && record.state == row.status() - && spec_matches - && route.task == task_id - && route.origin_machine == authority_machine - && route.execution_machine == authority_machine - && route.thread == row.thread) - .then_some(row), - ) -} - -fn replayed_result( - receipt: ReturnDecisionReceipt, - authority: &SupervisorActionAuthority, - decision: &ReturnDecision, -) -> Result { - if receipt.authority != *authority || receipt.decision != *decision { - return Err(ReturnDecisionError::ConflictingRetry { - action_id: authority.action_id, - }); - } - - Ok(receipt.result) -} - -/// Advance the resource revision under the shared compare-and-set -pub(super) fn advance_resource( - tx: &Transaction<'_>, - resource: &Resource, - registered_background_task: Option, -) -> Result { - match advance_resource_on(tx, resource, registered_background_task) { - Ok(next) => Ok(next), - Err(AdvanceError::Exhausted) => Err(ReturnDecisionError::RevisionExhausted { - revision: resource.state_revision, - }), - Err(AdvanceError::Changed) => { - let actual = select_resource(tx, resource.id)? - .map_or(resource.state_revision, |saved| saved.state_revision); - Err(ReturnDecisionError::StaleRevision { - expected: resource.state_revision, - actual, - }) - } - Err(AdvanceError::Storage(error)) => Err(error.into()), - } -} - -fn update_active_loan( - tx: &Transaction<'_>, - loan: &Loan, - phase: LoanActionPhase, - action_id: ActionId, -) -> Result<(), ReturnDecisionError> { - if !replace_loan_in_action_phase_on(tx, loan, phase, action_id)? { - return Err(ReturnDecisionError::ActionNotPending { - loan_id: loan.id, - action_id, - }); - } - - Ok(()) -} - -fn saved_decision( - conn: &Connection, - action_id: ActionId, -) -> Result, ReturnDecisionError> { - let receipt: Option = conn - .query_row( - "SELECT receipt_json FROM resource_return_decisions WHERE action_id = ?1", - [action_id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional()?; - Ok(receipt - .map(|json| serde_json::from_str(&json)) - .transpose()?) -} - -fn insert_decision_receipt( - tx: &Transaction<'_>, - receipt: &ReturnDecisionReceipt, -) -> Result<(), ReturnDecisionError> { - tx.execute( - "INSERT INTO resource_return_decisions (action_id, resource_id, loan_id, receipt_json) - VALUES (?1, ?2, ?3, ?4)", - params![ - receipt.authority.action_id.as_uuid().to_string(), - receipt.authority.resource_id.as_uuid().to_string(), - receipt.authority.loan_id.as_uuid().to_string(), - serde_json::to_string(receipt)?, - ], - )?; - Ok(()) -} - -fn saved_closure( - conn: &Connection, - action_id: ActionId, -) -> Result, ReturnDecisionError> { - let receipt: Option = conn - .query_row( - "SELECT receipt_json FROM resource_restore_closures WHERE action_id = ?1", - [action_id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional()?; - Ok(receipt - .map(|json| serde_json::from_str(&json)) - .transpose()?) -} - -fn insert_closure_receipt( - tx: &Transaction<'_>, - receipt: &RestoreClosureReceipt, -) -> Result<(), ReturnDecisionError> { - tx.execute( - "INSERT INTO resource_restore_closures (action_id, task_id, receipt_json) - VALUES (?1, ?2, ?3)", - params![ - receipt.action_id.as_uuid().to_string(), - receipt.task_id.to_string(), - serde_json::to_string(receipt)?, - ], - )?; - Ok(()) -} diff --git a/src/store/resource/test_support.rs b/src/store/resource/test_support.rs deleted file mode 100644 index 7221927..0000000 --- a/src/store/resource/test_support.rs +++ /dev/null @@ -1,293 +0,0 @@ -//! Seeding helpers for tests that need authority-local resource rows in exact states - -use std::path::Path; - -use rusqlite::{OptionalExtension, TransactionBehavior, params}; - -use super::OperatorGpuFreeError; -use super::operator_release::saved_receipt_on; -use super::select_authority_resource; -use crate::domain::ProcessGroupExitEvidence; -use crate::domain::TaskId; -use crate::error::AppError; -use crate::machine::MachineId; -use crate::resource::operator_release::{OperatorAttestationId, OperatorGpuFreeReceipt}; -use crate::resource::store::{ - CompleteReleaseError, ConflictReason, OpenReleaseLoanError, OpenReleaseLoanResult, - QueueCancellationResult, ResourceStoreError, SupervisorNoticeStoreError, - cancel_request_before_activation_on, next_queued_request_for_authority, select_non_closed_loan, -}; -use crate::resource::{ - ActionId, AssignmentRevision, Loan, LoanId, LoanPhase, LoanState, NoticeId, ResourceId, - ResourceRequest, ResourceRequestState, ResourceRevision, ReturnContext, - ServingReleaseProvenance, SupervisorAddress, SupervisorNotice, -}; -use crate::store::Store; -use crate::submission::RequestId; - -/// Serving provenance that no saved completion receipt proves -/// -/// A loan that carries it stays Serving, but activation refuses it -pub(crate) fn unreceipted_release_provenance() -> ServingReleaseProvenance { - ServingReleaseProvenance::CompletedTrainerResult { - action_id: ActionId::new(), - task_id: TaskId::new(), - publication_sha256: "ab".repeat(32).parse().expect("fixture digest"), - } -} - -/// Model activation winning after an API control receipt commits -pub(crate) fn mark_request_assigned_for_race(db_path: &Path, request_id: RequestId) { - let store = Store::open(db_path).expect("open race fixture store"); - let assigned = serde_json::to_string(&ResourceRequestState::Assigned { - loan_id: LoanId::new(), - }) - .expect("encode assigned fixture"); - store - .conn - .execute( - "UPDATE resource_requests SET state_json = ?1 WHERE request_id = ?2", - rusqlite::params![assigned, request_id.0.to_string()], - ) - .expect("advance request in race fixture"); -} - -impl Store { - pub(crate) fn seed_verified_serving_loan_for_test( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - request_id: RequestId, - return_context: ReturnContext, - ) -> Result<(Loan, ResourceRevision), CompleteReleaseError> { - let (mut loan, state_revision) = self.seed_serving_loan_for_test( - authority_machine, - resource_id, - request_id, - return_context, - )?; - let expected_state_revision = - ResourceRevision::new(state_revision.get().checked_sub(1).ok_or( - ResourceStoreError::Conflict(ConflictReason::ResourceRevisionChanged), - )?); - loan = crate::resource::store::test_support::seed_verified_serving_provenance_for_test( - &mut self.conn, - loan, - authority_machine, - resource_id, - request_id, - expected_state_revision, - state_revision, - )?; - Ok((loan, state_revision)) - } - - pub(crate) fn seed_serving_loan_for_test( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - request_id: RequestId, - return_context: ReturnContext, - ) -> Result<(Loan, ResourceRevision), CompleteReleaseError> { - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate)?; - let resource = select_authority_resource(&tx, authority_machine, resource_id)?; - - let acceptance_sequence: Option = tx - .query_row( - "SELECT acceptance_sequence FROM resource_requests - WHERE request_id = ?1 AND resource_id = ?2 - AND json_extract(state_json, '$.type') = 'queued'", - params![request_id.0.to_string(), resource_id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional()?; - let Some(acceptance_sequence) = acceptance_sequence else { - return Err(ResourceStoreError::Conflict(ConflictReason::RequestStateChanged).into()); - }; - - let next_revision_value = resource.state_revision.get().checked_add(1).ok_or( - CompleteReleaseError::RevisionExhausted { - revision: resource.state_revision, - }, - )?; - let next_revision = ResourceRevision::new(next_revision_value); - let current_loan = select_non_closed_loan(&tx, resource_id)?; - let loan_id = current_loan - .as_ref() - .map_or_else(LoanId::new, |loan| loan.id); - if current_loan.as_ref().is_some_and(|loan| { - !matches!( - loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { .. } - } - ) - }) { - return Err(ResourceStoreError::Conflict(ConflictReason::LoanChanged).into()); - } - let state_json = serde_json::to_string(&ResourceRequestState::Assigned { loan_id }) - .map_err(AppError::from)?; - let changed = tx.execute( - "UPDATE resource_requests SET state_json = ?1 - WHERE request_id = ?2 AND resource_id = ?3 AND acceptance_sequence = ?4 - AND json_extract(state_json, '$.type') = 'queued'", - params![ - state_json, - request_id.0.to_string(), - resource_id.as_uuid().to_string(), - acceptance_sequence, - ], - )?; - if changed != 1 { - return Err(CompleteReleaseError::RequestChanged { request_id }); - } - - let loan = Loan { - id: loan_id, - resource_id, - state: LoanState::Active { - phase: LoanPhase::Serving { - return_context, - current_request_id: request_id, - release_provenance: unreceipted_release_provenance(), - }, - }, - }; - let loan_state_json = serde_json::to_string(&loan.state).map_err(AppError::from)?; - if current_loan.is_some() { - let changed = tx.execute( - "UPDATE loans SET state_json = ?1 WHERE id = ?2 AND resource_id = ?3", - params![ - loan_state_json, - loan.id.as_uuid().to_string(), - resource_id.as_uuid().to_string(), - ], - )?; - if changed != 1 { - return Err(CompleteReleaseError::LoanChanged { loan_id: loan.id }); - } - } else { - tx.execute( - "INSERT INTO loans (id, resource_id, state_json) VALUES (?1, ?2, ?3)", - params![ - loan.id.as_uuid().to_string(), - resource_id.as_uuid().to_string(), - loan_state_json, - ], - )?; - } - let changed = tx.execute( - "UPDATE resources SET state_revision = ?1 - WHERE id = ?2 AND authority_machine = ?3 AND state_revision = ?4", - params![ - i64::try_from(next_revision_value).map_err(|_| { - CompleteReleaseError::RevisionExhausted { - revision: resource.state_revision, - } - })?, - resource_id.as_uuid().to_string(), - authority_machine.as_uuid().to_string(), - i64::try_from(resource.state_revision.get()).map_err(|_| { - CompleteReleaseError::RevisionExhausted { - revision: resource.state_revision, - } - })?, - ], - )?; - if changed != 1 { - return Err( - ResourceStoreError::Conflict(ConflictReason::ResourceRevisionChanged).into(), - ); - } - tx.commit()?; - - Ok((loan, next_revision)) - } - - /// Read process-group evidence for one exact task identity - pub(crate) fn process_group_exit_evidence( - &self, - id: TaskId, - ) -> Result, AppError> { - Ok(self - .get_task(id)? - .map(|row| row.process_group_exit_evidence())) - } - - /// Read the next queued request for one resource - pub(crate) fn next_queued_resource_request( - &self, - authority_machine: MachineId, - resource_id: ResourceId, - ) -> Result, ResourceStoreError> { - next_queued_request_for_authority(&self.conn, authority_machine, resource_id) - } - - /// Cancel a queued request or atomically retain prevention before acceptance - pub(crate) fn cancel_resource_request_before_activation( - &mut self, - authority_machine: MachineId, - request_id: RequestId, - task_id: TaskId, - resource_id: ResourceId, - origin_machine: MachineId, - ) -> Result { - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate)?; - let result = cancel_request_before_activation_on( - &tx, - authority_machine, - request_id, - task_id, - resource_id, - origin_machine, - )?; - tx.commit()?; - Ok(result) - } - - /// Open or reuse the authority's release loan - pub(crate) fn open_release_loan_for_authority( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - expected_state_revision: ResourceRevision, - ) -> Result { - crate::resource::store::test_support::open_release_loan_for_authority( - &mut self.conn, - authority_machine, - resource_id, - expected_state_revision, - ) - } - - /// Retarget an undelivered notice with an assignment-revision compare-and-set - pub(crate) fn retarget_supervisor_notice( - &mut self, - notice_id: NoticeId, - expected_assignment_revision: AssignmentRevision, - destination: SupervisorAddress, - new_assignment_revision: AssignmentRevision, - ) -> Result { - crate::resource::store::test_support::retarget_supervisor_notice( - &mut self.conn, - notice_id, - expected_assignment_revision, - destination, - new_assignment_revision, - ) - } - - /// Read the saved receipt of one operator attestation on this authority - pub(crate) fn operator_attestation_receipt_for_authority( - &self, - authority_machine: MachineId, - operation_id: OperatorAttestationId, - ) -> Result, OperatorGpuFreeError> { - Ok(saved_receipt_on(&self.conn, operation_id)? - .filter(|receipt| receipt.attestation.authority_machine == authority_machine)) - } -} diff --git a/src/store/resource/tests.rs b/src/store/resource/tests.rs deleted file mode 100644 index 08c8d90..0000000 --- a/src/store/resource/tests.rs +++ /dev/null @@ -1,27 +0,0 @@ -//! Authority-local resource store tests - -mod fixtures; - -mod assigned_task; -mod background; -mod cancellation; -mod completed_boundary; -mod container; -mod controls; -mod docker; -mod ended_trainer; -mod foreground; -mod initial_idle; -mod operator_release; -mod queue_authority; -mod registration; -mod release_checkpoint; -mod release_completion; -mod release_transaction; -mod release_watcher; -mod release_watcher_acceptance; -mod release_watcher_binding; -mod restore; -mod return_deadline; -mod supervised_launch; -mod trainer_association; diff --git a/src/store/resource/tests/assigned_task.rs b/src/store/resource/tests/assigned_task.rs deleted file mode 100644 index 4c2a2f0..0000000 --- a/src/store/resource/tests/assigned_task.rs +++ /dev/null @@ -1,1386 +0,0 @@ -//! Assigned resource task acceptance, queue drain, and completion tests - -use super::fixtures::{ - ServingFixture, accept_and_finish_next_resource_task, accept_and_finish_resource_task, - acceptance_counts, acceptance_input, completion, queue_waiting_request, - refresh_serving_fixture, resource_cancellation_identity, resource_cancellation_proof, - resource_origin_route, serving_fixture, task_reconcile_input, waiting_receipt, -}; -use crate::domain::{ - ExitReason, ProcessGroupExitEvidence, ProcessStatus, TaskEnv, TaskId, TaskState, -}; -use crate::events::{EventAcceptance, EventPayload}; -use crate::machine::MachineId; -use crate::resource::api::{BrowserResourceAction, QueuePlacement}; -use crate::resource::store::{ - AssignedResourceTaskAttention, AssignedResourceTaskReconcileOutcome, ResourceStoreError, - ResourceTaskAcceptance, ResourceTaskAcceptanceInput, ResourceTaskCompletionResult, -}; -use crate::resource::{ - DeliveryAttemptId, Loan, LoanId, LoanPhase, LoanState, ResourceId, ResourceRequest, - ResourceRequestState, ResourceRevision, -}; -use crate::spec::NormalizedWorkload; -use crate::store::{ExecutorIdentity, ResourceControlEffect, ResourceControlRequest, Store}; -use crate::submission::{PreAcceptanceRejection, RequestId, ResourceRoutePhase, SubmissionState}; -use tempfile::tempdir; -use uuid::Uuid; - -fn queue_request(fixture: &mut ServingFixture) -> ResourceRequest { - queue_waiting_request( - &mut fixture.store, - fixture.authority, - fixture.resource.id, - fixture.origin, - &fixture.spec, - ) -} - -#[test] -fn assigned_resource_task_acceptance_commits_task_identity_and_first_event() { - let mut fixture = serving_fixture(true, true); - let input = acceptance_input(&fixture); - - assert_eq!( - fixture - .store - .accept_assigned_resource_task(input.clone()) - .unwrap(), - ResourceTaskAcceptance::Inserted { - task: fixture.request.task_id, - } - ); - let task = fixture - .store - .get_task(fixture.request.task_id) - .unwrap() - .unwrap(); - assert_eq!(task.state, TaskState::Queued); - assert!(matches!( - fixture.store.executor_identity(task.id).unwrap(), - Some(ExecutorIdentity::Accepted(record)) - if record.task == task.id - && record.origin_machine == fixture.origin - && record.execution_machine == fixture.authority - && record.state == ProcessStatus::Queued - )); - - let events = fixture.store.pending_outbound_events(task.id).unwrap(); - assert_eq!(events.len(), 1); - assert_eq!(events[0].event.seq.get(), 1); - assert_eq!(events[0].event.origin_machine, fixture.origin); - assert_eq!(events[0].event.execution_machine, fixture.authority); - assert_eq!( - events[0].event.payload, - EventPayload::State { - status: ProcessStatus::Queued, - } - ); - assert!(matches!( - fixture.store.resource_requests(fixture.authority, fixture.resource.id).unwrap()[0] - .state, - ResourceRequestState::Assigned { loan_id } if loan_id == fixture.loan.id - )); - assert!(matches!( - fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap()[0] - .loan - .as_ref() - .map(|loan| &loan.state), - Some(LoanState::Active { - phase: LoanPhase::Serving { current_request_id, .. } - }) if *current_request_id == fixture.request.request_id - )); -} - -#[test] -fn assigned_resource_task_event_failure_rolls_back_task_and_identity() { - let mut fixture = serving_fixture(true, true); - let input = acceptance_input(&fixture); - fixture - .store - .conn - .execute_batch( - "CREATE TRIGGER reject_resource_task_event - BEFORE INSERT ON executor_outbox - BEGIN SELECT RAISE(ABORT, 'event storage is unavailable'); END;", - ) - .unwrap(); - - assert!( - fixture - .store - .accept_assigned_resource_task(input.clone()) - .is_err() - ); - assert_eq!( - acceptance_counts(&fixture.store, input.request_id, input.task_id)[..4], - [0, 0, 0, 0] - ); - assert!(matches!( - fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap()[0] - .state, - ResourceRequestState::Assigned { loan_id } if loan_id == fixture.loan.id - )); -} - -#[test] -fn assigned_resource_task_retry_after_reopen_returns_existing_without_records() { - let mut fixture = serving_fixture(true, true); - let input = acceptance_input(&fixture); - fixture - .store - .accept_assigned_resource_task(input.clone()) - .unwrap(); - let before = acceptance_counts(&fixture.store, input.request_id, input.task_id); - let database = fixture.directory.path().join("db"); - let ServingFixture { - directory, store, .. - } = fixture; - drop(store); - - let mut reopened = Store::open(&database).unwrap(); - assert_eq!( - reopened - .accept_assigned_resource_task(input.clone()) - .unwrap(), - ResourceTaskAcceptance::Existing { - task: input.task_id, - state: ProcessStatus::Queued, - } - ); - assert_eq!( - acceptance_counts(&reopened, input.request_id, input.task_id), - before - ); - drop(directory); -} - -#[test] -fn assigned_resource_task_retry_rejects_missing_row_or_changed_environment() { - let mut fixture = serving_fixture(true, true); - let input = acceptance_input(&fixture); - fixture - .store - .accept_assigned_resource_task(input.clone()) - .unwrap(); - - let mut changed_environment = input.clone(); - changed_environment.executor_env.home = "/different-home".into(); - assert!(matches!( - fixture - .store - .accept_assigned_resource_task(changed_environment), - Err(ResourceStoreError::Conflict(_)) - )); - - fixture - .store - .conn - .execute("DELETE FROM tasks WHERE id=?1", [input.task_id.to_string()]) - .unwrap(); - assert!(matches!( - fixture.store.accept_assigned_resource_task(input), - Err(ResourceStoreError::Conflict(_)) - )); -} - -#[test] -fn assigned_resource_task_rejects_changed_spec_and_route() { - let mut fixture = serving_fixture(true, true); - let input = acceptance_input(&fixture); - fixture - .store - .accept_assigned_resource_task(input.clone()) - .unwrap(); - - let mut changed_spec = fixture.spec.clone(); - let NormalizedWorkload::Task(task) = &mut changed_spec.workload else { - panic!("resource acceptance uses a command workload"); - }; - task.command = - crate::invocation::CommandLine::try_from_argv(vec!["/bin/echo".into(), "changed".into()]) - .unwrap(); - let mut changed_input = input.clone(); - changed_input.command_spec = crate::resource::CommandSpec::try_from(changed_spec).unwrap(); - let before_changed_spec = acceptance_counts(&fixture.store, input.request_id, input.task_id); - assert!(matches!( - fixture.store.accept_assigned_resource_task(changed_input), - Err(ResourceStoreError::Conflict(_)) - )); - assert_eq!( - acceptance_counts(&fixture.store, input.request_id, input.task_id), - before_changed_spec - ); - - let mut route = fixture - .store - .origin_route_by_task(input.task_id) - .unwrap() - .unwrap(); - route.submission = SubmissionState::Resource { - resource: ResourceId::new(), - phase: ResourceRoutePhase::Waiting, - }; - fixture - .store - .conn - .execute( - "UPDATE origin_routes SET route_json=?1 WHERE task_id=?2", - rusqlite::params![ - serde_json::to_string(&route).unwrap(), - input.task_id.to_string() - ], - ) - .unwrap(); - let before = acceptance_counts(&fixture.store, input.request_id, input.task_id); - assert!(matches!( - fixture.store.accept_assigned_resource_task(input.clone()), - Err(ResourceStoreError::Conflict(_)) - )); - assert_eq!( - acceptance_counts(&fixture.store, input.request_id, input.task_id), - before - ); -} - -#[test] -fn assigned_resource_task_rejects_wrong_authority_loan_request_and_revision() { - let mut fixture = serving_fixture(true, true); - let base = acceptance_input(&fixture); - - let mut wrong_authority = base.clone(); - wrong_authority.authority_machine = MachineId::new(); - assert!(matches!( - fixture.store.accept_assigned_resource_task(wrong_authority), - Err(ResourceStoreError::WrongAuthority { .. }) - )); - - let mut wrong_loan = base.clone(); - wrong_loan.loan_id = LoanId::new(); - assert!(matches!( - fixture.store.accept_assigned_resource_task(wrong_loan), - Err(ResourceStoreError::Conflict(_)) - )); - - let mut wrong_request = base.clone(); - wrong_request.request_id = RequestId::new(); - assert!(matches!( - fixture.store.accept_assigned_resource_task(wrong_request), - Err(ResourceStoreError::Conflict(_)) - )); - - let mut stale_revision = base.clone(); - stale_revision.expected_state_revision = - ResourceRevision::new(fixture.state_revision.get() + 1); - assert!(matches!( - fixture.store.accept_assigned_resource_task(stale_revision), - Err(ResourceStoreError::Conflict(_)) - )); - assert_eq!( - acceptance_counts(&fixture.store, base.request_id, base.task_id)[..4], - [0, 0, 0, 0] - ); -} - -#[test] -fn assigned_resource_task_rejects_a_request_before_it_in_serving_order() { - let mut fixture = serving_fixture(true, true); - let later_request = fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - fixture.spec.clone(), - ) - .unwrap(); - let LoanState::Active { - phase: LoanPhase::Serving { return_context, .. }, - } = fixture.loan.state.clone() - else { - panic!("fixture must have one serving request"); - }; - - fixture - .store - .conn - .execute( - "UPDATE resource_requests SET state_json=?1 WHERE request_id=?2", - rusqlite::params![ - serde_json::to_string(&ResourceRequestState::Queued).unwrap(), - fixture.request.request_id.0.to_string(), - ], - ) - .unwrap(); - fixture - .store - .conn - .execute( - "UPDATE resource_requests SET state_json=?1 WHERE request_id=?2", - rusqlite::params![ - serde_json::to_string(&ResourceRequestState::Assigned { - loan_id: fixture.loan.id, - }) - .unwrap(), - later_request.request_id.0.to_string(), - ], - ) - .unwrap(); - let later_loan = Loan { - id: fixture.loan.id, - resource_id: fixture.resource.id, - state: LoanState::Active { - phase: LoanPhase::Serving { - return_context, - current_request_id: later_request.request_id, - release_provenance: crate::store::unreceipted_release_provenance(), - }, - }, - }; - fixture - .store - .conn - .execute( - "UPDATE loans SET state_json=?1 WHERE id=?2", - rusqlite::params![ - serde_json::to_string(&later_loan.state).unwrap(), - fixture.loan.id.as_uuid().to_string(), - ], - ) - .unwrap(); - - let input = ResourceTaskAcceptanceInput { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - request_id: later_request.request_id, - task_id: later_request.task_id, - acceptance_sequence: later_request.acceptance_sequence, - loan_id: fixture.loan.id, - expected_state_revision: fixture.state_revision, - command_spec: later_request.spec().clone(), - executor_env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - }; - assert!(matches!( - fixture.store.accept_assigned_resource_task(input), - Err(ResourceStoreError::Conflict(_)) - )); - assert_eq!( - acceptance_counts( - &fixture.store, - later_request.request_id, - later_request.task_id - )[..4], - [0, 0, 0, 0] - ); -} - -#[test] -fn confirmed_success_failure_and_cancel_drain_three_requests_in_serving_order() { - let mut fixture = serving_fixture(true, true); - let second = fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - fixture.spec.clone(), - ) - .unwrap(); - let third = fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - fixture.spec.clone(), - ) - .unwrap(); - - let first_input = accept_and_finish_resource_task( - &mut fixture, - ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let first = fixture - .store - .reconcile_assigned_resource_task_for_authority(first_input) - .unwrap(); - let (loan, next_request) = match completion(first) { - Ok(ResourceTaskCompletionResult::Assigned { - finished_request, - loan, - next_request, - .. - }) => { - assert!(matches!( - &finished_request.state, - ResourceRequestState::Finished { - outcome: ExitReason::Exit { code: 0 } - } - )); - (loan, next_request) - } - other => { - panic!("first confirmed completion did not assign the next request: {other:?}") - } - }; - assert_eq!(next_request.request_id, second.request_id); - assert_eq!(loan.id, fixture.loan.id); - assert!(matches!( - &next_request.state, - ResourceRequestState::Assigned { loan_id } if *loan_id == loan.id - )); - assert!(matches!( - fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap()[2] - .state, - ResourceRequestState::Queued - )); - let committed_revision = fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap()[0] - .resource - .state_revision; - // a duplicate terminal event returns the stored assignment without a new revision - let duplicate = fixture - .store - .reconcile_assigned_resource_task_for_authority(first_input) - .unwrap(); - assert!(matches!( - completion(duplicate), - Ok(ResourceTaskCompletionResult::Assigned { - next_request: duplicate_next, - state_revision, - .. - }) if duplicate_next.request_id == second.request_id - && state_revision == committed_revision - )); - assert_eq!( - fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap()[0] - .resource - .state_revision, - committed_revision - ); - refresh_serving_fixture(&mut fixture, loan, next_request); - - let second_input = accept_and_finish_next_resource_task( - &mut fixture, - ExitReason::Exit { code: 7 }, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let second_result = fixture - .store - .reconcile_assigned_resource_task_for_authority(second_input) - .unwrap(); - let (loan, next_request) = match completion(second_result) { - Ok(ResourceTaskCompletionResult::Assigned { - finished_request, - loan, - next_request, - .. - }) => { - assert!(matches!( - &finished_request.state, - ResourceRequestState::Finished { - outcome: ExitReason::Exit { code: 7 } - } - )); - (loan, next_request) - } - other => { - panic!("second confirmed completion did not assign the last request: {other:?}") - } - }; - assert_eq!(next_request.request_id, third.request_id); - assert_eq!(loan.id, fixture.loan.id); - refresh_serving_fixture(&mut fixture, loan, next_request); - - let third_input = accept_and_finish_next_resource_task( - &mut fixture, - ExitReason::Cancelled, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let third_result = fixture - .store - .reconcile_assigned_resource_task_for_authority(third_input) - .unwrap(); - let (loan, notice) = match completion(third_result) { - Ok(ResourceTaskCompletionResult::ReturnRequired { - finished_request, - loan, - notice, - }) => { - assert!(matches!( - &finished_request.state, - ResourceRequestState::Finished { - outcome: ExitReason::Cancelled - } - )); - (loan, notice) - } - other => panic!("empty queue did not reserve the return: {other:?}"), - }; - assert!(matches!( - loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingReturn { action_id, .. } - } if action_id == notice.action_id - )); - assert_eq!(notice.loan_id, fixture.loan.id); - assert!(matches!( - notice.payload, - crate::resource::SupervisorNoticePayload::ReturnRequired { .. } - )); - let notice_count: i64 = fixture - .store - .conn - .query_row( - "SELECT COUNT(*) FROM resource_supervisor_notices - WHERE loan_id=?1 - AND json_extract(notice_json, '$.payload.type') = 'return_required'", - [fixture.loan.id.as_uuid().to_string()], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(notice_count, 1); -} - -#[test] -fn queue_does_not_advance_until_the_exact_process_group_exit_is_confirmed() { - let mut fixture = serving_fixture(true, true); - let next = fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - fixture.spec.clone(), - ) - .unwrap(); - let input = acceptance_input(&fixture); - fixture.store.accept_assigned_resource_task(input).unwrap(); - fixture - .store - .cas_status( - fixture.request.task_id, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - - assert!(matches!( - fixture - .store - .reconcile_assigned_resource_task_for_authority(task_reconcile_input(&fixture)) - .unwrap(), - AssignedResourceTaskReconcileOutcome::Active( - crate::resource::store::AssignedResourceTaskProgress::Running - ) - )); - fixture - .store - .cas_exit_with_evidence( - fixture.request.task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::Unconfirmed, - ) - .unwrap() - .unwrap(); - assert!(matches!( - fixture - .store - .reconcile_assigned_resource_task_for_authority(task_reconcile_input(&fixture)) - .unwrap(), - AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::ExitWitnessUnconfirmed - ) - )); - let requests = fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap(); - assert!(matches!( - &requests[0].state, - ResourceRequestState::Assigned { loan_id } if *loan_id == fixture.loan.id - )); - assert!(matches!(requests[1].state, ResourceRequestState::Queued)); - assert!(fixture.store.get_task(next.task_id).unwrap().is_none()); - assert_eq!( - fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap()[0] - .resource - .state_revision, - fixture.state_revision - ); -} - -#[test] -fn lost_task_retains_its_assignment_and_reports_attention() { - let mut fixture = serving_fixture(true, true); - let next = fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - fixture.spec.clone(), - ) - .unwrap(); - fixture - .store - .accept_assigned_resource_task(acceptance_input(&fixture)) - .unwrap(); - fixture - .store - .cas_status( - fixture.request.task_id, - ProcessStatus::Queued, - ProcessStatus::Lost, - ) - .unwrap() - .unwrap(); - - assert!(matches!( - fixture - .store - .reconcile_assigned_resource_task_for_authority(task_reconcile_input(&fixture)) - .unwrap(), - AssignedResourceTaskReconcileOutcome::Attention(AssignedResourceTaskAttention::TaskLost) - )); - let requests = fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap(); - assert!(matches!( - &requests[0].state, - ResourceRequestState::Assigned { loan_id } if *loan_id == fixture.loan.id - )); - assert!(matches!(requests[1].state, ResourceRequestState::Queued)); - assert!(fixture.store.get_task(next.task_id).unwrap().is_none()); -} - -#[test] -fn mismatched_task_executor_loan_and_revision_do_not_commit_completion() { - let mut fixture = serving_fixture(true, true); - let input = accept_and_finish_resource_task( - &mut fixture, - ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ); - - let mut wrong_task = input; - wrong_task.task_id = TaskId::new(); - assert!(matches!( - fixture - .store - .reconcile_assigned_resource_task_for_authority(wrong_task) - .unwrap(), - AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::TaskIdentityMismatch - ) - )); - let mut wrong_loan = input; - wrong_loan.loan_id = LoanId::new(); - assert!(matches!( - fixture - .store - .reconcile_assigned_resource_task_for_authority(wrong_loan) - .unwrap(), - AssignedResourceTaskReconcileOutcome::Attention(_) - )); - let mut stale_revision = input; - stale_revision.expected_state_revision = - ResourceRevision::new(input.expected_state_revision.get() + 1); - assert!(matches!( - fixture - .store - .reconcile_assigned_resource_task_for_authority(stale_revision) - .unwrap(), - AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::StaleRevision - ) - )); - - let Some(ExecutorIdentity::Accepted(mut accepted)) = - fixture.store.executor_identity(input.task_id).unwrap() - else { - panic!("accepted resource task must retain an executor identity"); - }; - accepted.origin_machine = MachineId::new(); - fixture - .store - .conn - .execute( - "UPDATE executor_identities SET identity_json=?1 WHERE task_id=?2", - rusqlite::params![ - serde_json::to_string(&ExecutorIdentity::Accepted(accepted)).unwrap(), - input.task_id.to_string(), - ], - ) - .unwrap(); - assert!(matches!( - fixture - .store - .reconcile_assigned_resource_task_for_authority(input) - .unwrap(), - AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::TaskIdentityMismatch - ) - )); - assert!(matches!( - fixture.store.resource_requests(fixture.authority, fixture.resource.id).unwrap()[0] - .state, - ResourceRequestState::Assigned { loan_id } if loan_id == fixture.loan.id - )); - let receipt_count: i64 = fixture - .store - .conn - .query_row( - "SELECT COUNT(*) FROM resource_task_completions", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(receipt_count, 0); -} - -#[test] -fn queued_request_cancelled_during_a_serving_task_is_skipped() { - let mut fixture = serving_fixture(true, true); - let cancelled = fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - fixture.spec.clone(), - ) - .unwrap(); - let next = fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - fixture.spec.clone(), - ) - .unwrap(); - fixture - .store - .accept_assigned_resource_task(acceptance_input(&fixture)) - .unwrap(); - fixture - .store - .cas_status( - fixture.request.task_id, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - fixture - .store - .cancel_resource_request_before_activation( - fixture.authority, - cancelled.request_id, - cancelled.task_id, - fixture.resource.id, - cancelled.origin_machine, - ) - .unwrap(); - fixture - .store - .cas_exit_with_evidence( - fixture.request.task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ) - .unwrap() - .unwrap(); - - let result = fixture - .store - .reconcile_assigned_resource_task_for_authority(task_reconcile_input(&fixture)) - .unwrap(); - assert!(matches!( - completion(result), - Ok( - ResourceTaskCompletionResult::Assigned { - next_request, - .. - } - ) if next_request.request_id == next.request_id - )); - let requests = fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap(); - assert!(matches!( - requests - .iter() - .find(|request| request.request_id == cancelled.request_id) - .unwrap() - .state, - ResourceRequestState::CancelledBeforeLaunch - )); - assert!(matches!( - requests.iter().find(|request| request.request_id == next.request_id).unwrap().state, - ResourceRequestState::Assigned { loan_id } if loan_id == fixture.loan.id - )); -} - -#[test] -fn no_child_spawn_after_initial_spawn_failure_is_an_explicit_release_proof() { - let mut fixture = serving_fixture(true, true); - fixture - .store - .accept_assigned_resource_task(acceptance_input(&fixture)) - .unwrap(); - fixture - .store - .cas_exit_with_evidence( - fixture.request.task_id, - ProcessStatus::Queued, - &ExitReason::SpawnFailed { - message: "fake task runner could not start".into(), - }, - ProcessGroupExitEvidence::NoChildSpawned, - ) - .unwrap() - .unwrap(); - - let result = fixture - .store - .reconcile_assigned_resource_task_for_authority(task_reconcile_input(&fixture)) - .unwrap(); - assert!(matches!( - completion(result), - Ok(ResourceTaskCompletionResult::ReturnRequired { .. }) - )); - let receipt: String = fixture - .store - .conn - .query_row( - "SELECT receipt_json FROM resource_task_completions WHERE task_id=?1", - [fixture.request.task_id.to_string()], - |row| row.get(0), - ) - .unwrap(); - assert_eq!( - serde_json::from_str::(&receipt).unwrap()["release_proof"], - "no_child_spawned_after_spawn_failure" - ); - assert!(matches!( - fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap()[0] - .state, - ResourceRequestState::Finished { - outcome: ExitReason::SpawnFailed { .. } - } - )); -} - -#[test] -fn cancelling_an_accepted_task_before_its_worker_starts_is_an_explicit_release_proof() { - let mut fixture = serving_fixture(true, true); - let next = fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - fixture.spec.clone(), - ) - .unwrap(); - fixture - .store - .accept_assigned_resource_task(acceptance_input(&fixture)) - .unwrap(); - assert!(matches!( - fixture - .store - .request_cancel(fixture.request.task_id) - .unwrap(), - crate::store::CancelResult::CancelledQueued(_) - )); - - let result = fixture - .store - .reconcile_assigned_resource_task_for_authority(task_reconcile_input(&fixture)) - .unwrap(); - assert!(matches!( - completion(result), - Ok(ResourceTaskCompletionResult::Assigned { next_request, .. }) - if next_request.request_id == next.request_id - )); - let receipt: String = fixture - .store - .conn - .query_row( - "SELECT receipt_json FROM resource_task_completions WHERE task_id=?1", - [fixture.request.task_id.to_string()], - |row| row.get(0), - ) - .unwrap(); - assert_eq!( - serde_json::from_str::(&receipt).unwrap()["release_proof"], - "no_child_spawned_after_queued_cancel" - ); -} - -#[test] -fn no_child_evidence_after_a_worker_started_requires_a_spawn_failure() { - let mut fixture = serving_fixture(true, true); - fixture - .store - .accept_assigned_resource_task(acceptance_input(&fixture)) - .unwrap(); - fixture - .store - .cas_status( - fixture.request.task_id, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - - assert!( - fixture - .store - .cas_exit_with_evidence( - fixture.request.task_id, - ProcessStatus::Running, - &ExitReason::Cancelled, - ProcessGroupExitEvidence::NoChildSpawned, - ) - .is_err() - ); - assert_eq!( - fixture - .store - .get_task(fixture.request.task_id) - .unwrap() - .unwrap() - .status(), - ProcessStatus::Running - ); -} - -#[test] -fn exact_task_completion_retry_after_reopen_returns_the_same_return_notice() { - let mut fixture = serving_fixture(true, true); - let input = accept_and_finish_resource_task( - &mut fixture, - ExitReason::Cancelled, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let first = fixture - .store - .reconcile_assigned_resource_task_for_authority(input) - .unwrap(); - let (action_id, notice_id, revision) = match completion(first) { - Ok(ResourceTaskCompletionResult::ReturnRequired { notice, .. }) => { - (notice.action_id, notice.id, notice.state_revision) - } - other => panic!("the empty queue must reserve one return notice: {other:?}"), - }; - let database = fixture.directory.path().join("db"); - let committed_revision = fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap()[0] - .resource - .state_revision; - drop(fixture.store); - - let mut reopened = Store::open(&database).unwrap(); - let retry = reopened - .reconcile_assigned_resource_task_for_authority(input) - .unwrap(); - assert!(matches!( - completion(retry), - Ok( - ResourceTaskCompletionResult::ReturnRequired { - notice, - .. - } - ) if notice.action_id == action_id - && notice.id == notice_id - && notice.state_revision == revision - )); - assert_eq!( - reopened - .resource_snapshots_for_authority(fixture.authority) - .unwrap()[0] - .resource - .state_revision, - committed_revision - ); - let receipt_count: i64 = reopened - .conn - .query_row( - "SELECT COUNT(*) FROM resource_task_completions", - [], - |row| row.get(0), - ) - .unwrap(); - let notice_count: i64 = reopened - .conn - .query_row( - "SELECT COUNT(*) FROM resource_supervisor_notices - WHERE json_extract(notice_json, '$.payload.type') = 'return_required'", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(receipt_count, 1); - assert_eq!(notice_count, 1); -} - -#[test] -fn assigned_resource_task_requires_local_callback_route_before_insert() { - let mut fixture = serving_fixture(true, false); - let input = acceptance_input(&fixture); - - assert!(matches!( - fixture.store.accept_assigned_resource_task(input.clone()), - Err(ResourceStoreError::OriginRouteNotFound { task }) if task == input.task_id - )); - assert_eq!( - acceptance_counts(&fixture.store, input.request_id, input.task_id)[..4], - [0, 0, 0, 0] - ); -} - -#[test] -fn assigned_remote_resource_task_uses_the_saved_origin_event_route() { - let mut fixture = serving_fixture(false, false); - let origin_directory = tempdir().unwrap(); - let mut origin_store = Store::open(&origin_directory.path().join("origin-db")).unwrap(); - origin_store - .insert_origin_route(&resource_origin_route( - fixture.request.request_id, - fixture.request.task_id, - fixture.resource.id, - fixture.origin, - fixture.authority, - &fixture.spec, - )) - .unwrap(); - origin_store - .resolve_resource_route(&waiting_receipt( - fixture.request.request_id, - fixture.request.task_id, - fixture.resource.id, - fixture.origin, - fixture.authority, - )) - .unwrap(); - - let input = acceptance_input(&fixture); - assert_eq!( - fixture - .store - .accept_assigned_resource_task(input.clone()) - .unwrap(), - ResourceTaskAcceptance::Inserted { - task: fixture.request.task_id, - } - ); - assert!( - fixture - .store - .origin_route_by_task(fixture.request.task_id) - .unwrap() - .is_none() - ); - let event = fixture - .store - .pending_outbound_events(input.task_id) - .unwrap()[0] - .event - .clone(); - assert_eq!(event.origin_machine, fixture.origin); - assert_eq!(event.execution_machine, fixture.authority); - assert_eq!( - origin_store.accept_inbound_event(&event).unwrap(), - EventAcceptance::Acknowledged { seq: 1 } - ); - fixture - .store - .mark_outbound_acknowledged(input.task_id, event.seq) - .unwrap(); - let route = origin_store - .origin_route_by_task(input.task_id) - .unwrap() - .unwrap(); - assert!(matches!( - route.submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::Activated, - .. - } - )); - assert_eq!(route.origin_machine, fixture.origin); - assert_eq!(route.execution_machine, fixture.authority); - - fixture - .store - .cas_status(input.task_id, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - let running = fixture - .store - .pending_outbound_events(input.task_id) - .unwrap()[0] - .event - .clone(); - assert_eq!(running.seq.get(), 2); - assert_eq!( - origin_store.accept_inbound_event(&running).unwrap(), - EventAcceptance::Acknowledged { seq: 2 } - ); - fixture - .store - .mark_outbound_acknowledged(input.task_id, running.seq) - .unwrap(); - fixture - .store - .cas_exit_with_evidence( - input.task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 9 }, - ProcessGroupExitEvidence::ConfirmedExited, - ) - .unwrap() - .unwrap(); - assert!(matches!( - fixture - .store - .reconcile_assigned_resource_task_for_authority(task_reconcile_input(&fixture)) - .unwrap(), - AssignedResourceTaskReconcileOutcome::Completed(_) - )); - - let terminal = fixture - .store - .pending_outbound_events(input.task_id) - .unwrap()[0] - .event - .clone(); - assert_eq!(terminal.seq.get(), 3); - assert_eq!( - origin_store.accept_inbound_event(&terminal).unwrap(), - EventAcceptance::Acknowledged { seq: 3 } - ); - fixture - .store - .mark_outbound_acknowledged(input.task_id, terminal.seq) - .unwrap(); - let route = origin_store - .origin_route_by_task(input.task_id) - .unwrap() - .unwrap(); - assert_eq!(route.last_accepted_seq, 3); - assert_eq!(route.last_execution_state, Some(ProcessStatus::Failed)); -} - -#[test] -fn cancellation_and_assigned_resource_task_acceptance_resolve_in_either_order() { - let mut cancelled_first = serving_fixture(true, true); - let input = acceptance_input(&cancelled_first); - cancelled_first - .store - .cancel_resource_request_before_activation( - cancelled_first.authority, - input.request_id, - input.task_id, - input.resource_id, - cancelled_first.origin, - ) - .unwrap(); - assert!(matches!( - cancelled_first - .store - .accept_assigned_resource_task(input.clone()), - Err(ResourceStoreError::Prevented) - )); - let counts = acceptance_counts(&cancelled_first.store, input.request_id, input.task_id); - assert_eq!(counts[0], 0); - assert_eq!(&counts[2..4], [0, 0]); - assert!(matches!( - cancelled_first - .store - .resource_requests(cancelled_first.authority, cancelled_first.resource.id) - .unwrap()[0] - .state, - ResourceRequestState::CancelledBeforeLaunch - )); - assert!(matches!( - cancelled_first.store.executor_identity(input.task_id).unwrap(), - Some(ExecutorIdentity::Rejected(rejection)) - if rejection.reason == PreAcceptanceRejection::Cancelled.as_str() - )); - - let mut accepted_first = serving_fixture(true, true); - let input = acceptance_input(&accepted_first); - accepted_first - .store - .accept_assigned_resource_task(input.clone()) - .unwrap(); - let loan_before = accepted_first - .store - .resource_snapshots_for_authority(accepted_first.authority) - .unwrap()[0] - .loan - .clone() - .unwrap(); - let cancellation = resource_cancellation_identity( - Uuid::now_v7(), - input.request_id, - input.task_id, - input.resource_id, - accepted_first.origin, - accepted_first.authority, - ResourceRoutePhase::Waiting, - ); - let receipt = accepted_first - .store - .cancel_resource_request_with_receipt( - accepted_first.authority, - cancellation.clone(), - resource_cancellation_proof( - &cancellation, - &accepted_first.spec, - ResourceRoutePhase::Waiting, - ), - ) - .unwrap(); - assert_eq!( - receipt.outcome, - crate::submission::ResourceCancellationOutcome::NotEligible { - reason: crate::submission::ResourceCancellationIneligibleReason::Activated, - } - ); - assert_eq!( - accepted_first - .store - .resource_snapshots_for_authority(accepted_first.authority) - .unwrap()[0] - .loan - .clone() - .unwrap(), - loan_before - ); - assert!(matches!( - accepted_first - .store - .resource_requests(accepted_first.authority, accepted_first.resource.id) - .unwrap()[0] - .state, - ResourceRequestState::Assigned { loan_id } if loan_id == accepted_first.loan.id - )); -} - -#[test] -fn moving_a_queued_request_ahead_of_an_assigned_request_selects_and_accepts_it_next() { - let mut fixture = serving_fixture(true, true); - let earlier_queued = queue_request(&mut fixture); - let moved_to_front = queue_request(&mut fixture); - let reordered = fixture - .store - .begin_resource_control( - fixture.authority, - Uuid::now_v7(), - &ResourceControlRequest { - resource_id: fixture.resource.id, - expected_revision: fixture.state_revision, - action: BrowserResourceAction::MoveQueued { - request_id: moved_to_front.request_id, - placement: QueuePlacement::Front, - }, - }, - DeliveryAttemptId::new(), - ) - .unwrap(); - assert!(matches!(reordered.effect, ResourceControlEffect::Reordered)); - fixture.state_revision = ResourceRevision::new(fixture.state_revision.get() + 1); - - let reconcile_input = accept_and_finish_resource_task( - &mut fixture, - ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let result = fixture - .store - .reconcile_assigned_resource_task_for_authority(reconcile_input) - .unwrap(); - let ResourceTaskCompletionResult::Assigned { - loan, next_request, .. - } = completion(result).unwrap() - else { - panic!("the completed request must assign the next queued request"); - }; - assert_eq!(next_request.request_id, moved_to_front.request_id); - assert_eq!( - fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|request| request.state == ResourceRequestState::Queued) - .map(|request| request.request_id), - Some(earlier_queued.request_id) - ); - - refresh_serving_fixture(&mut fixture, loan, next_request); - let acceptance = acceptance_input(&fixture); - assert_eq!( - fixture - .store - .accept_assigned_resource_task(acceptance) - .unwrap(), - ResourceTaskAcceptance::Inserted { - task: moved_to_front.task_id, - } - ); -} diff --git a/src/store/resource/tests/background.rs b/src/store/resource/tests/background.rs deleted file mode 100644 index bf928c9..0000000 --- a/src/store/resource/tests/background.rs +++ /dev/null @@ -1,1082 +0,0 @@ -//! First background launch binding, confirmed-start registration, and the idle boundary - -use super::fixtures::{acquire_test_lock, resource, saved_trainer_association_json, spec}; -use crate::domain::{ - ExitReason, ProcessGroupExitEvidence, ProcessStatus, TaskEnv, TaskId, ThreadId, -}; -use crate::invocation::CommandLine; -use crate::machine::MachineId; -use crate::resource::background_launch::{ - BackgroundLaunchBinding, BackgroundSupervisorAssignment, RemoteBackgroundLaunchReceipt, -}; -use crate::resource::command_shape::test_support::FakeTrainer; -use crate::resource::ownership_lock::test_support::start_fake_lock_process; -use crate::resource::ownership_lock::{ - OwnershipLockIdentity, OwnershipLockIdentityMismatchReason, OwnershipLockProbe, - OwnershipLockProbeError, TrainerRequestDigest, VerifiedTrainerAttempt, - build_trainer_attempt_registration_evidence, probe_segment_ownership_lock, - test_support as trainer_attempt_test_support, -}; -use crate::resource::store::{ - ResourceStoreError, ResourceTaskAcceptance, ResourceTaskAcceptanceInput, - TrainerAttemptAssociationStoreError, -}; -use crate::resource::{ - AssignmentRevision, IdleBoundaryProof, IdleProofGap, Loan, LoanPhase, LoanState, Resource, - ResourceQueueAttentionReason, ResourceQueueReconcileOutcome, ResourceRevision, ReturnContext, - ServingReleaseProvenance, -}; -use crate::spec::{NormalizedSpec, NormalizedTaskWorkload, NormalizedWorkload}; -use crate::store::resource::task_has_any_event; -use crate::store::resource::trainer_lock::TrainerLockReleaseGap; -use crate::store::{ - BackgroundLaunchAcceptance, BackgroundLaunchError, BackgroundLaunchInput, - BackgroundLaunchPhase, ExecutorIdentity, RemoteBackgroundLaunchInput, Store, -}; -use crate::submission::{CallbackExecutable, RequestId, normalized_spec_sha256}; -use rusqlite::params; -use std::fs; -use std::path::{Path, PathBuf}; -use tempfile::tempdir; -use uuid::Uuid; - -pub(super) struct LaunchFixture { - _directory: tempfile::TempDir, - pub(super) store: Store, - pub(super) authority: MachineId, - pub(super) resource: Resource, - pub(super) trainer: FakeTrainer, -} - -impl LaunchFixture { - pub(super) fn new() -> Self { - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let mut store = Store::open(&root.join("db")).unwrap(); - let authority = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - Self { - _directory: directory, - store, - authority, - resource, - trainer: FakeTrainer::new(&root), - } - } - - pub(super) fn input( - &self, - request_id: RequestId, - spec: NormalizedSpec, - ) -> BackgroundLaunchInput { - BackgroundLaunchInput { - authority_machine: self.authority, - resource_id: self.resource.id, - request_id, - task_id: TaskId::new(), - spec, - env: self.trainer.env.clone(), - callback_codex: CallbackExecutable::available(PathBuf::from("/bin/echo")), - } - } - - pub(super) fn trainer_spec(&self) -> NormalizedSpec { - self.trainer - .spec(self.resource.supervisor.thread, "direct segment trainer") - } - - pub(super) fn launch(&mut self, request_id: RequestId) -> TaskId { - let input = self.input(request_id, self.trainer_spec()); - let BackgroundLaunchAcceptance::Inserted { task, .. } = self - .store - .accept_background_launch_for_authority(input) - .unwrap() - else { - panic!("the first exact launch must insert its task"); - }; - task - } - - pub(super) fn saved_resource(&self) -> Resource { - self.store - .resource_snapshots_for_authority(self.authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == self.resource.id) - .unwrap() - .resource - } - - pub(super) fn queue_request(&mut self) -> crate::resource::ResourceRequest { - self.store - .accept_resource_request( - self.authority, - RequestId::new(), - TaskId::new(), - self.resource.id, - MachineId::new(), - spec(), - ) - .unwrap() - } - - pub(super) fn reconcile(&mut self) -> ResourceQueueReconcileOutcome { - self.store - .reconcile_resource_queue_for_authority(self.authority, self.resource.id) - .unwrap() - } - - /// Path of the fixture database - pub(super) fn db_path(&self) -> PathBuf { - self._directory.path().canonicalize().unwrap().join("db") - } - - /// Drop the open connection and read the same database as a restarted daemon - pub(super) fn reopen(&mut self) { - self.store = Store::open(&self.db_path()).unwrap(); - } - - pub(super) fn task_count(&self) -> i64 { - self.store - .conn - .query_row("SELECT COUNT(*) FROM tasks", [], |row| row.get(0)) - .unwrap() - } -} - -#[test] -fn first_launch_binds_one_unregistered_task_and_exact_retry_reuses_it() { - let mut fixture = LaunchFixture::new(); - let request_id = RequestId::new(); - let task = fixture.launch(request_id); - - // the queued row, accepted identity, callback route, and first event commit together - let row = fixture.store.get_task(task).unwrap().unwrap(); - assert_eq!(row.status(), ProcessStatus::Queued); - assert_eq!(row.thread, fixture.resource.supervisor.thread); - let Some(ExecutorIdentity::Accepted(record)) = fixture.store.executor_identity(task).unwrap() - else { - panic!("the launch must save an accepted executor identity"); - }; - assert_eq!(record.origin_machine, fixture.authority); - assert_eq!(record.execution_machine, fixture.authority); - let route = fixture - .store - .origin_route_by_request(request_id) - .unwrap() - .unwrap(); - assert_eq!(route.task, task); - assert!(task_has_any_event(&fixture.store.conn, task).unwrap()); - - // an inserted row is not live work, so it is not registered - let saved = fixture.saved_resource(); - assert_eq!(saved.registered_background_task, None); - assert_eq!( - saved.state_revision.get(), - fixture.resource.state_revision.get() + 1 - ); - - // a retry after a lost reply carries a new preallocated task but keeps the saved one - let retry = fixture.input(request_id, fixture.trainer_spec()); - assert_eq!( - fixture - .store - .accept_background_launch_for_authority(retry) - .unwrap(), - BackgroundLaunchAcceptance::Existing { - task, - state: ProcessStatus::Queued, - } - ); - assert_eq!(fixture.task_count(), 1); - - // different content for the same request identity is a conflict and writes nothing - let changed = fixture.input( - request_id, - fixture - .trainer - .spec(fixture.resource.supervisor.thread, "another trainer"), - ); - assert!(matches!( - fixture - .store - .accept_background_launch_for_authority(changed), - Err(BackgroundLaunchError::ConflictingRetry { .. }) - )); - assert_eq!(fixture.task_count(), 1); - assert_eq!(fixture.saved_resource(), saved); -} - -#[test] -fn launch_refuses_unverifiable_ownership_and_other_threads_before_writing() { - let mut fixture = LaunchFixture::new(); - let before = fixture.saved_resource(); - - // a shell wrapper can start work outside its foreground process group - let mut wrapper = fixture.trainer_spec(); - wrapper.workload = NormalizedWorkload::Task(NormalizedTaskWorkload { - command: CommandLine::try_from_argv(vec![ - "/bin/sh".into(), - "-c".into(), - "python3 -m ops.run_segment run &".into(), - ]) - .unwrap(), - }); - // a script named like the trainer module path is not the maintained invocation - let mut shebang = fixture.trainer_spec(); - shebang.workload = NormalizedWorkload::Task(NormalizedTaskWorkload { - command: CommandLine::try_from_argv(vec!["python3".into(), "ops/run_segment.py".into()]) - .unwrap(), - }); - for spec in [wrapper, shebang] { - assert!(matches!( - fixture - .store - .accept_background_launch_for_authority(fixture.input(RequestId::new(), spec)), - Err(BackgroundLaunchError::UnsupportedCommand(_)) - )); - } - - let other_thread = fixture.trainer.spec(ThreadId(Uuid::now_v7()), "trainer"); - assert!(matches!( - fixture - .store - .accept_background_launch_for_authority(fixture.input(RequestId::new(), other_thread)), - Err(BackgroundLaunchError::NotSupervisorThread { .. }) - )); - assert_eq!(fixture.task_count(), 0); - assert_eq!(fixture.saved_resource(), before); -} - -#[test] -fn remote_supervisor_launch_is_unsupported_and_writes_nothing() { - let mut fixture = LaunchFixture::new(); - let mut moved = fixture.resource.clone(); - moved.supervisor.machine = MachineId::new(); - fixture - .store - .conn - .execute( - "UPDATE resources SET supervisor_machine = ?1 WHERE id = ?2", - params![ - moved.supervisor.machine.as_uuid().to_string(), - moved.id.as_uuid().to_string() - ], - ) - .unwrap(); - let input = fixture.input(RequestId::new(), fixture.trainer_spec()); - assert_eq!( - fixture - .store - .accept_background_launch_for_authority(input) - .unwrap(), - BackgroundLaunchAcceptance::UnsupportedRemoteSupervisor { - authority_machine: fixture.authority, - supervisor: moved.supervisor, - } - ); - assert_eq!(fixture.task_count(), 0); - let count: i64 = fixture - .store - .conn - .query_row( - "SELECT COUNT(*) FROM resource_background_launches", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(count, 0); -} - -#[test] -fn launch_is_refused_while_queued_work_exists() { - let mut fixture = LaunchFixture::new(); - let request = fixture.queue_request(); - let input = fixture.input(RequestId::new(), fixture.trainer_spec()); - assert!(matches!( - fixture.store.accept_background_launch_for_authority(input), - Err(BackgroundLaunchError::QueuedWorkAhead { request_id }) if request_id == request.request_id - )); - assert_eq!(fixture.task_count(), 0); -} - -#[test] -fn queued_launch_reserves_the_resource_until_its_confirmed_start_registers_it() { - let mut fixture = LaunchFixture::new(); - let task = fixture.launch(RequestId::new()); - let later = fixture.queue_request(); - - // the queued launch is not proof of idle or of live work - assert!(matches!( - fixture.reconcile(), - ResourceQueueReconcileOutcome::AttentionRequired { - request, - reason: ResourceQueueAttentionReason::BackgroundLaunchPending { task_id }, - } if request.request_id == later.request_id && task_id == task - )); - assert_eq!(fixture.saved_resource().registered_background_task, None); - // a second launch cannot race the pending one - let second = fixture.input(RequestId::new(), fixture.trainer_spec()); - assert!(matches!( - fixture.store.accept_background_launch_for_authority(second), - Err(BackgroundLaunchError::QueuedWorkAhead { .. }) - )); - - // the confirmed start registers the task, and the queued request asks for its release - fixture - .store - .cas_status(task, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - assert!(matches!( - fixture.reconcile(), - ResourceQueueReconcileOutcome::ReleaseRequired { loan, .. } - if matches!( - loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { observed_background_task, .. } - } if observed_background_task == task - ) - )); - assert_eq!( - fixture.saved_resource().registered_background_task, - Some(task) - ); -} - -#[test] -fn launch_that_never_spawned_proves_idle_and_serves_the_queue() { - let mut fixture = LaunchFixture::new(); - let request_id = RequestId::new(); - let task = fixture.launch(request_id); - // the task layer fails the row before any worker child starts - fixture - .store - .cas_exit_with_evidence( - task, - ProcessStatus::Queued, - &ExitReason::SpawnFailed { - message: "no worker".into(), - }, - ProcessGroupExitEvidence::NoChildSpawned, - ) - .unwrap() - .unwrap(); - let request = fixture.queue_request(); - - let ResourceQueueReconcileOutcome::IdleServing { - loan, - request: assigned, - proof, - } = fixture.reconcile() - else { - panic!("a launch with no child spawn must prove the idle boundary"); - }; - assert_eq!( - proof, - IdleBoundaryProof::BackgroundLaunchNeverSpawned { - request_id, - task_id: task - } - ); - assert_eq!(assigned.request_id, request.request_id); - assert!(matches!( - &loan.state, - LoanState::Active { - phase: LoanPhase::Serving { - return_context: ReturnContext::Idle, - release_provenance: ServingReleaseProvenance::IdleBoundary { .. }, - .. - } - } - )); - - // the saved opening receipt is the provenance that lets the command start - let saved = fixture.saved_resource(); - let input = ResourceTaskAcceptanceInput { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - request_id: request.request_id, - task_id: request.task_id, - acceptance_sequence: request.acceptance_sequence, - loan_id: loan.id, - expected_state_revision: saved.state_revision, - command_spec: request.spec().clone(), - executor_env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - }; - fixture - .store - .conn - .execute("DELETE FROM resource_idle_openings", []) - .unwrap(); - assert!(matches!( - fixture.store.accept_assigned_resource_task(input), - Err(ResourceStoreError::Conflict(_)) - )); -} - -#[test] -fn idle_serving_activates_only_with_its_opening_receipt() { - let mut fixture = LaunchFixture::new(); - let task = fixture.launch(RequestId::new()); - fixture - .store - .cas_exit(task, ProcessStatus::Queued, &ExitReason::Cancelled) - .unwrap() - .unwrap(); - let request = fixture.queue_request(); - let ResourceQueueReconcileOutcome::IdleServing { loan, .. } = fixture.reconcile() else { - panic!("a cancelled queued launch must prove the idle boundary"); - }; - let input = ResourceTaskAcceptanceInput { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - request_id: request.request_id, - task_id: request.task_id, - acceptance_sequence: request.acceptance_sequence, - loan_id: loan.id, - expected_state_revision: fixture.saved_resource().state_revision, - command_spec: request.spec().clone(), - executor_env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - }; - assert_eq!( - fixture.store.accept_assigned_resource_task(input).unwrap(), - ResourceTaskAcceptance::Inserted { - task: request.task_id - } - ); -} - -#[test] -fn launch_that_ran_before_registration_keeps_the_queue_reserved() { - let mut fixture = LaunchFixture::new(); - let task = fixture.launch(RequestId::new()); - // the worker started and exited before the owner observed the running row - fixture - .store - .cas_status(task, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - fixture - .store - .cas_exit_with_evidence( - task, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ) - .unwrap() - .unwrap(); - let request = fixture.queue_request(); - assert!(matches!( - fixture.reconcile(), - ResourceQueueReconcileOutcome::AttentionRequired { - request: blocked, - reason: ResourceQueueAttentionReason::IdleNotProven { - gap: IdleProofGap::BackgroundLaunchReleaseUnproven { task_id }, - }, - } if blocked.request_id == request.request_id && task_id == task - )); - assert!(saved_loan_for(&fixture).is_none()); -} - -#[test] -fn trainer_association_waits_for_registration_and_the_exact_runtime_root() { - let mut fixture = LaunchFixture::new(); - let task = fixture.launch(RequestId::new()); - let evidence = |root: &Path| { - VerifiedTrainerAttempt::from_persisted( - root.to_path_buf(), - trainer_attempt_test_support::attempt_binding("attempt-1"), - TrainerRequestDigest::from_hex(&"00".repeat(32)).unwrap(), - OwnershipLockIdentity::new(1, 2), - ) - .unwrap() - }; - - // a queued launch does not hold the trainer lock, so nothing may assert that it does - assert!(matches!( - fixture.store.bind_trainer_attempt_association( - fixture.authority, - fixture.resource.id, - task, - evidence(&fixture.trainer.runtime_root), - ), - Err(TrainerAttemptAssociationStoreError::TaskNotRegistered { .. }) - )); - - fixture - .store - .cas_status(task, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - assert!(matches!( - fixture.reconcile(), - ResourceQueueReconcileOutcome::NoQueuedRequest - )); - assert_eq!( - fixture.saved_resource().registered_background_task, - Some(task) - ); - - // evidence from another runtime root cannot bind to the accepted trainer command - let other_root = fixture.trainer.cwd.join("other-runtime"); - fs::create_dir_all(&other_root).unwrap(); - assert!(matches!( - fixture.store.bind_trainer_attempt_association( - fixture.authority, - fixture.resource.id, - task, - evidence(&other_root), - ), - Err(TrainerAttemptAssociationStoreError::DirectSegmentCommandShape( - crate::resource::command_shape::DirectSegmentCommandShapeError::RuntimeRootMismatch { .. } - )) - )); - - let association = fixture - .store - .bind_trainer_attempt_association( - fixture.authority, - fixture.resource.id, - task, - evidence(&fixture.trainer.runtime_root), - ) - .unwrap(); - assert_eq!(association.task_id(), task); -} - -#[test] -fn an_ended_launch_frees_the_slot_for_the_next_launch() { - let mut fixture = LaunchFixture::new(); - let first = fixture.launch(RequestId::new()); - fixture - .store - .cas_exit(first, ProcessStatus::Queued, &ExitReason::Cancelled) - .unwrap() - .unwrap(); - // an ended launch no longer occupies the slot, so the next launch binds - let second = fixture.launch(RequestId::new()); - assert_ne!(first, second); - let request = fixture.queue_request(); - assert!(matches!( - fixture.reconcile(), - ResourceQueueReconcileOutcome::AttentionRequired { - request: blocked, - reason: ResourceQueueAttentionReason::BackgroundLaunchPending { task_id }, - } if blocked.request_id == request.request_id && task_id == second - )); -} - -/// Launch a trainer and let the queue owner register its confirmed start -pub(super) fn registered_launch(fixture: &mut LaunchFixture, request_id: RequestId) -> TaskId { - let task = fixture.launch(request_id); - fixture - .store - .cas_status(task, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - // with no queued work the reconciliation only registers the confirmed start - assert!(matches!( - fixture.reconcile(), - ResourceQueueReconcileOutcome::NoQueuedRequest - )); - assert_eq!( - fixture.saved_resource().registered_background_task, - Some(task) - ); - task -} - -#[test] -fn predecessor_without_exit_proof_keeps_the_background_slot() { - type End = fn(&mut LaunchFixture, TaskId); - let lose: End = |fixture, task| { - fixture - .store - .cas_status(task, ProcessStatus::Running, ProcessStatus::Lost) - .unwrap() - .unwrap(); - }; - let unconfirmed: End = |fixture, task| { - fixture - .store - .cas_exit_with_evidence( - task, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::Unconfirmed, - ) - .unwrap() - .unwrap(); - }; - for (name, registered, end) in [ - ("lost wrapper before registration", false, lose), - ("lost registered trainer", true, lose), - ("unconfirmed exit before registration", false, unconfirmed), - ("unconfirmed registered exit", true, unconfirmed), - ] { - let mut fixture = LaunchFixture::new(); - let request_id = RequestId::new(); - let task = if registered { - registered_launch(&mut fixture, request_id) - } else { - let task = fixture.launch(request_id); - fixture - .store - .cas_status(task, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - task - }; - end(&mut fixture, task); - let before = fixture.saved_resource(); - - let next = fixture.input(RequestId::new(), fixture.trainer_spec()); - assert!( - matches!( - fixture.store.accept_background_launch_for_authority(next), - Err(BackgroundLaunchError::PredecessorReleaseUnproven { task_id }) if task_id == task - ), - "{name}" - ); - assert_eq!(fixture.task_count(), 1, "{name}"); - assert_eq!(fixture.saved_resource(), before, "{name}"); - // the accepted launch still answers its exact retry from the receipt - let retry = fixture.input(request_id, fixture.trainer_spec()); - assert!( - matches!( - fixture.store.accept_background_launch_for_authority(retry), - Ok(BackgroundLaunchAcceptance::Existing { task: existing, .. }) if existing == task - ), - "{name}" - ); - } -} - -/// Record exact evidence for a registered trainer while a holder keeps its new lock -/// -/// Returns the lock path. The holder is dropped before return, as if the -/// supervisor bound the association while the worker was running -fn bind_real_trainer_lock(fixture: &mut LaunchFixture, task: TaskId) -> PathBuf { - let runtime_root = fixture.trainer.runtime_root.clone(); - let binding = trainer_attempt_test_support::attempt_binding("attempt-1"); - crate::resource::trainer_publication::tests::write_request_for_test(&runtime_root, &binding); - let lock_path = runtime_root.join(".segment.lock"); - let running_worker = acquire_test_lock(&lock_path, true); - let evidence = build_trainer_attempt_registration_evidence(&runtime_root, &binding).unwrap(); - fixture - .store - .bind_trainer_attempt_association(fixture.authority, fixture.resource.id, task, evidence) - .unwrap(); - drop(running_worker); - lock_path -} - -/// Let the wrapper of a running task exit with a confirmed process-group exit -fn end_wrapper(fixture: &mut LaunchFixture, task: TaskId) { - fixture - .store - .cas_exit_with_evidence( - task, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ) - .unwrap() - .unwrap(); -} - -#[test] -fn confirmed_wrapper_exit_frees_the_slot_only_after_the_exact_trainer_lock_is_free() { - let mut fixture = LaunchFixture::new(); - let first_request = RequestId::new(); - let first = registered_launch(&mut fixture, first_request); - let lock_path = bind_real_trainer_lock(&mut fixture, first); - end_wrapper(&mut fixture, first); - - // the worker runs in its own session and outlives the wrapper that confirmed exit - let mut worker = start_fake_lock_process(&lock_path); - let before = fixture.saved_resource(); - let blocked = fixture.input(RequestId::new(), fixture.trainer_spec()); - let blocked_task = blocked.task_id; - assert!(matches!( - fixture.store.accept_background_launch_for_authority(blocked), - Err(BackgroundLaunchError::PredecessorOwnershipUnproven { - task_id, - gap: TrainerLockReleaseGap::OwnershipHeld { task_id: witness }, - }) if task_id == first && witness == first - )); - assert_eq!(fixture.task_count(), 1); - assert!(fixture.store.get_task(blocked_task).unwrap().is_none()); - assert_eq!(fixture.saved_resource(), before); - - worker.release(); - let second = fixture.launch(RequestId::new()); - assert_ne!(first, second); - // the authority held the lock only through the launch transaction - let identity = OwnershipLockIdentity::from_metadata(&fs::metadata(&lock_path).unwrap()); - assert!(matches!( - probe_segment_ownership_lock(&fixture.trainer.runtime_root, identity), - OwnershipLockProbe::ExactOwnershipReleased(_) - )); - // the unchanged first launch is answered from its receipt and spawns nothing - let retry = fixture.input(first_request, fixture.trainer_spec()); - assert_eq!( - fixture - .store - .accept_background_launch_for_authority(retry) - .unwrap(), - BackgroundLaunchAcceptance::Existing { - task: first, - state: ProcessStatus::Succeeded, - } - ); - assert_eq!(fixture.task_count(), 2); -} - -#[test] -fn wrapper_exit_before_registration_has_no_lock_witness_and_keeps_the_slot() { - let mut fixture = LaunchFixture::new(); - let task = fixture.launch(RequestId::new()); - fixture - .store - .cas_status(task, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - end_wrapper(&mut fixture, task); - - let before = fixture.saved_resource(); - let next = fixture.input(RequestId::new(), fixture.trainer_spec()); - assert!(matches!( - fixture.store.accept_background_launch_for_authority(next), - Err(BackgroundLaunchError::PredecessorOwnershipUnproven { - task_id, - gap: TrainerLockReleaseGap::AssociationMissing { task_id: witness }, - }) if task_id == task && witness == task - )); - assert_eq!(fixture.task_count(), 1); - assert_eq!(fixture.saved_resource(), before); -} - -#[test] -fn changed_or_missing_trainer_lock_evidence_keeps_the_slot() { - type Change = fn(&LaunchFixture, TaskId, &Path); - let replace_lock: Change = |_, _, lock_path| { - fs::rename(lock_path, lock_path.with_extension("lock.saved")).unwrap(); - fs::write(lock_path, b"").unwrap(); - }; - let change_saved_identity: Change = |fixture, task, _| { - let saved = saved_trainer_association_json(&fixture.store, task); - let mut association: serde_json::Value = serde_json::from_str(&saved).unwrap(); - let inode = association["ownership_lock_identity"]["inode"] - .as_u64() - .unwrap(); - association["ownership_lock_identity"]["inode"] = (inode + 1).into(); - fixture - .store - .conn - .execute( - "UPDATE trainer_attempt_associations SET association_json = ?1 WHERE task_id = ?2", - params![association.to_string(), task.to_string()], - ) - .unwrap(); - }; - let remove_association: Change = |fixture, task, _| { - fixture - .store - .conn - .execute( - "DELETE FROM trainer_attempt_associations WHERE task_id = ?1", - [task.to_string()], - ) - .unwrap(); - }; - type Expected = fn(&TrainerLockReleaseGap, TaskId) -> bool; - let different_file: Expected = |gap, _| { - matches!( - gap, - TrainerLockReleaseGap::OwnershipLock(OwnershipLockProbeError::IdentityMismatch { - reason: OwnershipLockIdentityMismatchReason::DifferentFile, - .. - }) - ) - }; - let missing: Expected = |gap, task| matches!(gap, TrainerLockReleaseGap::AssociationMissing { task_id } if *task_id == task); - for (name, change, expected) in [ - ("replaced lock file", replace_lock, different_file), - ( - "changed saved lock identity", - change_saved_identity, - different_file, - ), - ("missing association", remove_association, missing), - ] { - let mut fixture = LaunchFixture::new(); - let first = registered_launch(&mut fixture, RequestId::new()); - let lock_path = bind_real_trainer_lock(&mut fixture, first); - end_wrapper(&mut fixture, first); - change(&fixture, first, &lock_path); - - // no process holds any lock, yet a free file is not the saved lock - let before = fixture.saved_resource(); - let next = fixture.input(RequestId::new(), fixture.trainer_spec()); - match fixture.store.accept_background_launch_for_authority(next) { - Err(BackgroundLaunchError::PredecessorOwnershipUnproven { task_id, gap }) => { - assert_eq!(task_id, first, "{name}"); - assert!(expected(&gap, first), "{name}: {gap:?}"); - } - other => panic!("{name}: expected a fail-closed slot, got {other:?}"), - } - assert_eq!(fixture.task_count(), 1, "{name}"); - assert_eq!(fixture.saved_resource(), before, "{name}"); - } -} - -pub(super) fn saved_loan_for(fixture: &LaunchFixture) -> Option { - fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == fixture.resource.id) - .unwrap() - .loan -} - -/// Remote supervisor machine and its saved launch identities for one fixture resource -struct RemoteLaunch { - receipt: RemoteBackgroundLaunchReceipt, - spec: NormalizedSpec, -} - -impl LaunchFixture { - /// Fixture whose supervisor thread runs on another machine - fn remote() -> Self { - let mut fixture = Self::new(); - let supervisor = MachineId::new(); - fixture - .store - .conn - .execute( - "UPDATE resources SET supervisor_machine = ?1 WHERE id = ?2", - params![ - supervisor.as_uuid().to_string(), - fixture.resource.id.as_uuid().to_string() - ], - ) - .unwrap(); - fixture.resource = fixture.saved_resource(); - fixture - } - - /// Launch that the supervisor machine saved for the current resource state - fn remote_launch(&self, name: &str) -> RemoteLaunch { - let resource = self.saved_resource(); - let spec = self.trainer.spec(resource.supervisor.thread, name); - RemoteLaunch { - receipt: RemoteBackgroundLaunchReceipt { - binding: BackgroundLaunchBinding { - assignment: BackgroundSupervisorAssignment::of(&resource), - expected_state_revision: resource.state_revision, - }, - request_id: RequestId::new(), - task_id: TaskId::new(), - normalized_spec_sha256: normalized_spec_sha256(&spec).unwrap(), - }, - spec, - } - } - - fn accept_remote( - &mut self, - launch: &RemoteLaunch, - ) -> Result { - self.store - .accept_remote_background_launch_for_authority(RemoteBackgroundLaunchInput { - receipt: launch.receipt, - spec: launch.spec.clone(), - env: self.trainer.env.clone(), - }) - } -} - -#[test] -fn remote_launch_keeps_the_callback_owner_remote_and_exact_retry_only_observes() { - let mut fixture = LaunchFixture::remote(); - let launch = fixture.remote_launch("remote trainer"); - let task = launch.receipt.task_id; - assert!(matches!( - fixture.accept_remote(&launch).unwrap(), - BackgroundLaunchAcceptance::Inserted { task: inserted, .. } if inserted == task - )); - - // the authority keeps the execution identity and first event; the route stays remote - let Some(ExecutorIdentity::Accepted(record)) = fixture.store.executor_identity(task).unwrap() - else { - panic!("the launch must save an accepted executor identity"); - }; - assert_eq!(record.origin_machine, fixture.resource.supervisor.machine); - assert_eq!(record.execution_machine, fixture.authority); - assert!( - fixture - .store - .origin_route_by_request(launch.receipt.request_id) - .unwrap() - .is_none() - ); - assert!( - crate::store::initial_queued_event_matches_on( - &fixture.store.conn, - task, - fixture.resource.supervisor.machine, - fixture.authority, - ) - .unwrap() - ); - let launched = fixture - .store - .background_launch_for_authority(fixture.authority, fixture.resource.id) - .unwrap() - .unwrap(); - assert_eq!(launched.phase, BackgroundLaunchPhase::Queued); - let saved = fixture.saved_resource(); - assert_eq!(saved.registered_background_task, None); - - // a lost reply is answered from the receipt, even after the revision moved - assert_eq!( - fixture.accept_remote(&launch).unwrap(), - BackgroundLaunchAcceptance::Existing { - task, - state: ProcessStatus::Queued, - } - ); - - // changed content, another task, or a co-located call cannot reuse the request - let mut changed = fixture.remote_launch("changed trainer"); - changed.receipt.request_id = launch.receipt.request_id; - let mut other_task = fixture.remote_launch("remote trainer"); - other_task.receipt.request_id = launch.receipt.request_id; - for retry in [changed, other_task] { - assert!(matches!( - fixture.accept_remote(&retry), - Err(BackgroundLaunchError::ConflictingRetry { .. }) - )); - } - let co_located = fixture.input(launch.receipt.request_id, launch.spec.clone()); - assert!(matches!( - fixture - .store - .accept_background_launch_for_authority(co_located), - Err(BackgroundLaunchError::ConflictingRetry { .. }) - )); - assert_eq!(fixture.task_count(), 1); - assert_eq!(fixture.saved_resource(), saved); -} - -#[test] -fn remote_launch_refuses_a_stale_assignment_revision_or_queue_before_writing() { - let mut fixture = LaunchFixture::remote(); - - let mut old_assignment = fixture.remote_launch("trainer"); - old_assignment - .receipt - .binding - .assignment - .assignment_revision = AssignmentRevision::new(7); - let mut other_thread = fixture.remote_launch("trainer"); - other_thread.receipt.binding.assignment.supervisor.thread = ThreadId(Uuid::now_v7()); - for launch in [old_assignment, other_thread] { - assert!(matches!( - fixture.accept_remote(&launch), - Err(BackgroundLaunchError::NotCurrentSupervisor) - )); - } - let mut stale = fixture.remote_launch("trainer"); - stale.receipt.binding.expected_state_revision = ResourceRevision::new(9); - assert!(matches!( - fixture.accept_remote(&stale), - Err(BackgroundLaunchError::StaleRevision { .. }) - )); - - // queued work blocks a first launch - let request = fixture.queue_request(); - let launch = fixture.remote_launch("trainer"); - assert!(matches!( - fixture.accept_remote(&launch), - Err(BackgroundLaunchError::QueuedWorkAhead { request_id }) if request_id == request.request_id - )); - assert_eq!(fixture.task_count(), 0); - - // a co-located resource cannot accept a launch that claims a remote callback owner - let mut co_located = LaunchFixture::new(); - let mut claimed = co_located.remote_launch("trainer"); - claimed.receipt.binding.assignment.supervisor.machine = MachineId::new(); - assert!(matches!( - co_located.accept_remote(&claimed), - Err(BackgroundLaunchError::NotCurrentSupervisor) - )); - assert_eq!(co_located.task_count(), 0); -} - -#[test] -fn remote_launch_registers_on_its_start_and_a_replaced_supervisor_cannot_launch_again() { - let mut fixture = LaunchFixture::remote(); - let launch = fixture.remote_launch("remote trainer"); - let task = launch.receipt.task_id; - fixture.accept_remote(&launch).unwrap(); - // a second launch cannot race the pending one - assert!(matches!( - fixture.accept_remote(&fixture.remote_launch("second")), - Err(BackgroundLaunchError::LaunchPending { task_id }) if task_id == task - )); - - fixture - .store - .cas_status(task, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - assert!(matches!( - fixture.reconcile(), - ResourceQueueReconcileOutcome::NoQueuedRequest - )); - assert_eq!( - fixture.saved_resource().registered_background_task, - Some(task) - ); - - // a replacement supervisor cannot start another trainer while this one runs - let replaced = fixture.saved_resource(); - fixture - .store - .conn - .execute( - "UPDATE resources SET supervisor_thread = ?1, assignment_revision = ?2 WHERE id = ?3", - params![ - ThreadId(Uuid::now_v7()).to_string(), - i64::try_from(replaced.assignment_revision.get() + 1).unwrap(), - replaced.id.as_uuid().to_string() - ], - ) - .unwrap(); - assert!(matches!( - fixture.accept_remote(&fixture.remote_launch("replacement")), - Err(BackgroundLaunchError::BackgroundTaskActive { task_id, .. }) if task_id == task - )); - // the old supervisor's retry only observes its running task - assert_eq!( - fixture.accept_remote(&launch).unwrap(), - BackgroundLaunchAcceptance::Existing { - task, - state: ProcessStatus::Running, - } - ); - assert_eq!(fixture.task_count(), 1); -} diff --git a/src/store/resource/tests/cancellation.rs b/src/store/resource/tests/cancellation.rs deleted file mode 100644 index e7df158..0000000 --- a/src/store/resource/tests/cancellation.rs +++ /dev/null @@ -1,704 +0,0 @@ -//! Durable resource cancellation and tombstone tests - -use super::fixtures::{ - assert_cancelled_tombstone, identity_count, open_release_for_test, prevention_count, - remote_task, resource, resource_cancellation_identity, resource_cancellation_proof, spec, -}; -use crate::domain::{ExitReason, TaskId}; -use crate::machine::MachineId; -use crate::resource::store::{QueueCancellationResult, ResourceStoreError}; -use crate::resource::{ResourceRequestState, ReturnContext}; -use crate::store::{ExecutorIdentity, IdentityError, Store}; -use crate::submission::{ - ExecutionRecord, RequestId, ResourceCancellationIneligibleReason, ResourceCancellationOutcome, - ResourceRoutePhase, -}; -use rusqlite::params; -use std::sync::{Arc, Barrier}; -use tempfile::tempdir; -use uuid::Uuid; - -#[test] -fn durable_resource_cancellation_receipt_replays_and_conflicts_by_identity() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - let identity = resource_cancellation_identity( - Uuid::now_v7(), - RequestId::new(), - TaskId::new(), - resource.id, - origin, - authority, - ResourceRoutePhase::AcceptanceUnknown, - ); - - let first = store - .cancel_resource_request_with_receipt( - authority, - identity.clone(), - resource_cancellation_proof(&identity, &spec(), ResourceRoutePhase::AcceptanceUnknown), - ) - .unwrap(); - assert_eq!( - first.outcome, - ResourceCancellationOutcome::PreventedBeforeAcceptance - ); - assert_eq!( - store.resource_cancellation_receipt(&identity).unwrap(), - Some(first.clone()) - ); - assert_eq!( - store - .cancel_resource_request_with_receipt( - authority, - identity.clone(), - resource_cancellation_proof( - &identity, - &spec(), - ResourceRoutePhase::AcceptanceUnknown, - ), - ) - .unwrap(), - first - ); - - let mut changed = identity; - changed.task = TaskId::new(); - assert!(matches!( - store.cancel_resource_request_with_receipt( - authority, - changed.clone(), - resource_cancellation_proof(&changed, &spec(), ResourceRoutePhase::AcceptanceUnknown,), - ), - Err(ResourceStoreError::Conflict(_)) - )); - assert_eq!(prevention_count(&store), 1); -} - -#[test] -fn durable_resource_cancellation_cancels_queued_and_assigned_requests() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - let queued_request = RequestId::new(); - let queued_task = TaskId::new(); - store - .accept_resource_request( - authority, - queued_request, - queued_task, - resource.id, - origin, - spec(), - ) - .unwrap(); - let queued_identity = resource_cancellation_identity( - Uuid::now_v7(), - queued_request, - queued_task, - resource.id, - origin, - authority, - ResourceRoutePhase::Waiting, - ); - let queued_receipt = store - .cancel_resource_request_with_receipt( - authority, - queued_identity.clone(), - resource_cancellation_proof(&queued_identity, &spec(), ResourceRoutePhase::Waiting), - ) - .unwrap(); - assert_eq!( - queued_receipt.outcome, - ResourceCancellationOutcome::CancelledBeforeLaunch - ); - assert!(matches!( - store.resource_requests(authority, resource.id).unwrap()[0].state, - ResourceRequestState::CancelledBeforeLaunch - )); - assert_eq!( - store - .cancel_resource_request_with_receipt( - authority, - queued_identity.clone(), - resource_cancellation_proof( - &queued_identity, - &spec(), - ResourceRoutePhase::Waiting, - ), - ) - .unwrap(), - queued_receipt - ); - - let assigned_directory = tempdir().unwrap(); - let mut assigned_store = Store::open(&assigned_directory.path().join("db")).unwrap(); - let assigned_authority = MachineId::new(); - let background_task = TaskId::new(); - let (assigned_resource, _, _) = - open_release_for_test(&mut assigned_store, assigned_authority, background_task); - let assigned = assigned_store - .resource_requests(assigned_authority, assigned_resource.id) - .unwrap()[0] - .clone(); - let next_request_id = RequestId::new(); - let next_task = TaskId::new(); - assigned_store - .accept_resource_request( - assigned_authority, - next_request_id, - next_task, - assigned_resource.id, - MachineId::new(), - spec(), - ) - .unwrap(); - let (loan, _) = assigned_store - .seed_serving_loan_for_test( - assigned_authority, - assigned_resource.id, - assigned.request_id, - ReturnContext::Stopped { - task_id: background_task, - checkpoint_ref: "checkpoint-1".into(), - recovery_ref: "recovery-1".into(), - }, - ) - .unwrap(); - let assigned_identity = resource_cancellation_identity( - Uuid::now_v7(), - assigned.request_id, - assigned.task_id, - assigned_resource.id, - assigned.origin_machine, - assigned_authority, - ResourceRoutePhase::Waiting, - ); - let assigned_receipt = assigned_store - .cancel_resource_request_with_receipt( - assigned_authority, - assigned_identity.clone(), - resource_cancellation_proof(&assigned_identity, &spec(), ResourceRoutePhase::Waiting), - ) - .unwrap(); - assert_eq!( - assigned_receipt.outcome, - ResourceCancellationOutcome::CancelledBeforeLaunch - ); - let requests = assigned_store - .resource_requests(assigned_authority, assigned_resource.id) - .unwrap(); - assert!(matches!( - requests[0].state, - ResourceRequestState::CancelledBeforeLaunch - )); - assert!(matches!( - requests[1].state, - ResourceRequestState::Assigned { loan_id } if loan_id == loan.id - )); - assert_eq!( - assigned_store - .cancel_resource_request_with_receipt( - assigned_authority, - assigned_identity.clone(), - resource_cancellation_proof( - &assigned_identity, - &spec(), - ResourceRoutePhase::Waiting, - ), - ) - .unwrap(), - assigned_receipt - ); -} - -#[test] -fn resource_cancellation_of_a_terminal_request_returns_typed_attention_receipt() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - let request = RequestId::new(); - let task = TaskId::new(); - store - .accept_resource_request(authority, request, task, resource.id, origin, spec()) - .unwrap(); - let terminal_state = ResourceRequestState::Finished { - outcome: ExitReason::Cancelled, - }; - store - .conn - .execute( - "UPDATE resource_requests SET state_json=?1 WHERE request_id=?2", - params![ - serde_json::to_string(&terminal_state).unwrap(), - request.0.to_string(), - ], - ) - .unwrap(); - - let identity = resource_cancellation_identity( - Uuid::now_v7(), - request, - task, - resource.id, - origin, - authority, - ResourceRoutePhase::Waiting, - ); - let receipt = store - .cancel_resource_request_with_receipt( - authority, - identity.clone(), - resource_cancellation_proof(&identity, &spec(), ResourceRoutePhase::Waiting), - ) - .unwrap(); - assert_eq!( - receipt.outcome, - ResourceCancellationOutcome::NotEligible { - reason: ResourceCancellationIneligibleReason::Terminal, - } - ); - assert!(matches!( - store.resource_requests(authority, resource.id).unwrap()[0].state, - ResourceRequestState::Finished { .. } - )); - assert_eq!( - store.resource_cancellation_receipt(&identity).unwrap(), - Some(receipt) - ); -} - -#[test] -fn activated_route_race_returns_attention_without_cancelling_assigned_work() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, _, _) = open_release_for_test(&mut store, authority, background_task); - let assigned = store.resource_requests(authority, resource.id).unwrap()[0].clone(); - let queued_request = RequestId::new(); - let queued_task = TaskId::new(); - store - .accept_resource_request( - authority, - queued_request, - queued_task, - resource.id, - MachineId::new(), - spec(), - ) - .unwrap(); - let (loan, _) = store - .seed_serving_loan_for_test( - authority, - resource.id, - assigned.request_id, - ReturnContext::Stopped { - task_id: background_task, - checkpoint_ref: "checkpoint-1".into(), - recovery_ref: "recovery-1".into(), - }, - ) - .unwrap(); - - let identity = resource_cancellation_identity( - Uuid::now_v7(), - assigned.request_id, - assigned.task_id, - resource.id, - assigned.origin_machine, - authority, - ResourceRoutePhase::Waiting, - ); - let receipt = store - .cancel_resource_request_with_receipt( - authority, - identity.clone(), - resource_cancellation_proof(&identity, &spec(), ResourceRoutePhase::Activated), - ) - .unwrap(); - assert_eq!( - receipt.outcome, - ResourceCancellationOutcome::NotEligible { - reason: ResourceCancellationIneligibleReason::Activated, - } - ); - let requests = store.resource_requests(authority, resource.id).unwrap(); - assert!(matches!( - requests[0].state, - ResourceRequestState::Assigned { loan_id: assigned_loan } - if assigned_loan == loan.id - )); - assert!(matches!(requests[1].state, ResourceRequestState::Queued)); - assert_eq!( - store.resource_cancellation_receipt(&identity).unwrap(), - Some(receipt) - ); -} - -#[test] -fn cancellation_before_acceptance_fences_delayed_remote_task_acceptance() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - let request = RequestId::new(); - let task = TaskId::new(); - - assert!(matches!( - store - .cancel_resource_request_before_activation( - authority, - request, - task, - resource.id, - origin, - ) - .unwrap(), - QueueCancellationResult::PreventedBeforeAcceptance - )); - assert_eq!(prevention_count(&store), 1); - assert_cancelled_tombstone( - store.executor_identity(task).unwrap().unwrap(), - task, - origin, - authority, - ); - assert!(store.origin_route_by_task(task).unwrap().is_none()); - - let execution = store - .insert_remote_task(&remote_task(task, &spec()), &spec(), origin, authority) - .unwrap(); - assert_cancelled_tombstone(execution, task, origin, authority); - assert!(store.get_task(task).unwrap().is_none()); - assert_eq!(identity_count(&store, task), 1); - - assert!(matches!( - store - .cancel_resource_request_before_activation( - authority, - request, - task, - resource.id, - origin, - ) - .unwrap(), - QueueCancellationResult::PreventedBeforeAcceptance - )); - assert_eq!(prevention_count(&store), 1); - assert_eq!(identity_count(&store, task), 1); -} - -#[test] -fn cancellation_after_queue_acceptance_is_atomic_and_idempotent() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - let request = RequestId::new(); - let task = TaskId::new(); - let accepted = store - .accept_resource_request(authority, request, task, resource.id, origin, spec()) - .unwrap(); - assert!(store.origin_route_by_task(task).unwrap().is_none()); - assert_eq!( - store - .next_queued_resource_request(authority, resource.id) - .unwrap() - .unwrap() - .request_id, - request - ); - - let cancelled = store - .cancel_resource_request_before_activation(authority, request, task, resource.id, origin) - .unwrap(); - let QueueCancellationResult::Request(cancelled) = cancelled else { - panic!("accepted cancellation must return its saved request"); - }; - assert_eq!(cancelled.acceptance_sequence, accepted.acceptance_sequence); - assert!(matches!( - cancelled.state, - ResourceRequestState::CancelledBeforeLaunch - )); - assert!( - store - .next_queued_resource_request(authority, resource.id) - .unwrap() - .is_none() - ); - assert_cancelled_tombstone( - store.executor_identity(task).unwrap().unwrap(), - task, - origin, - authority, - ); - - let retry = store - .cancel_resource_request_before_activation(authority, request, task, resource.id, origin) - .unwrap(); - let QueueCancellationResult::Request(retry) = retry else { - panic!("repeated cancellation must return the saved request"); - }; - assert_eq!(retry.acceptance_sequence, accepted.acceptance_sequence); - assert!(matches!( - retry.state, - ResourceRequestState::CancelledBeforeLaunch - )); - - let delayed_acceptance = store - .insert_remote_task(&remote_task(task, &spec()), &spec(), origin, authority) - .unwrap(); - assert_cancelled_tombstone(delayed_acceptance, task, origin, authority); - assert!(store.get_task(task).unwrap().is_none()); - assert_eq!(identity_count(&store, task), 1); -} - -#[test] -fn concurrent_direct_acceptance_cannot_claim_a_resource_queue_id() { - let directory = tempdir().unwrap(); - let path = directory.path().join("db"); - let authority = MachineId::new(); - let origin = MachineId::new(); - let resource = resource(authority); - let request = RequestId::new(); - let task = TaskId::new(); - let barrier = Arc::new(Barrier::new(2)); - - let mut setup = Store::open(&path).unwrap(); - setup.register_resource(authority, &resource).unwrap(); - setup - .accept_resource_request(authority, request, task, resource.id, origin, spec()) - .unwrap(); - drop(setup); - - let accept_path = path.clone(); - let accept_barrier = barrier.clone(); - let spec_for_acceptance = spec(); - let acceptance = std::thread::spawn(move || { - let mut store = Store::open(&accept_path).unwrap(); - accept_barrier.wait(); - store.accept_execution(&ExecutionRecord { - task, - origin_machine: origin, - execution_machine: authority, - spec: spec_for_acceptance.into(), - state: crate::domain::ProcessStatus::Queued, - }) - }); - - let mut cancel_store = Store::open(&path).unwrap(); - barrier.wait(); - let cancellation = cancel_store.cancel_resource_request_before_activation( - authority, - request, - task, - resource.id, - origin, - ); - let execution_identity = acceptance.join().unwrap(); - - let final_store = Store::open(&path).unwrap(); - let saved = final_store.executor_identity(task).unwrap().unwrap(); - let tombstone = match execution_identity { - Err(IdentityError::Conflict) => match saved { - ExecutorIdentity::Rejected(tombstone) => tombstone, - identity => panic!("resource cancellation must retain a rejection: {identity:?}"), - }, - Ok(ExecutorIdentity::Rejected(tombstone)) => tombstone, - outcome => panic!("ordinary acceptance must not win this race: {outcome:?}"), - }; - assert_eq!(tombstone.task, task); - assert_eq!(tombstone.origin_machine, origin); - assert_eq!(tombstone.execution_machine, authority); - assert!(matches!( - cancellation, - Ok(QueueCancellationResult::Request(cancelled)) - if matches!(cancelled.state, ResourceRequestState::CancelledBeforeLaunch) - )); - assert_eq!(prevention_count(&final_store), 0); -} - -#[test] -fn tombstone_insert_failure_rolls_back_queued_request_cancellation() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - let request = RequestId::new(); - let task = TaskId::new(); - store - .accept_resource_request(authority, request, task, resource.id, origin, spec()) - .unwrap(); - store - .conn - .execute_batch( - "CREATE TRIGGER reject_executor_tombstone - BEFORE INSERT ON executor_identities - BEGIN SELECT RAISE(ABORT, 'forced tombstone failure'); END;", - ) - .unwrap(); - - assert!( - store - .cancel_resource_request_before_activation( - authority, - request, - task, - resource.id, - origin, - ) - .is_err() - ); - let requests = store.resource_requests(authority, resource.id).unwrap(); - assert_eq!(requests.len(), 1); - assert!(matches!(requests[0].state, ResourceRequestState::Queued)); - assert_eq!(identity_count(&store, task), 0); - assert!( - store - .next_queued_resource_request(authority, resource.id) - .unwrap() - .is_some() - ); -} - -#[test] -fn terminal_cancellation_does_not_create_executor_tombstones() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - let terminal = store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource.id, - origin, - spec(), - ) - .unwrap(); - let terminal_json = serde_json::to_string(&ResourceRequestState::Finished { - outcome: ExitReason::Cancelled, - }) - .unwrap(); - store - .conn - .execute( - "UPDATE resource_requests SET state_json=?1 WHERE request_id=?2", - rusqlite::params![terminal_json, terminal.request_id.0.to_string()], - ) - .unwrap(); - - let result = store - .cancel_resource_request_before_activation( - authority, - terminal.request_id, - terminal.task_id, - resource.id, - origin, - ) - .unwrap(); - let QueueCancellationResult::Request(saved) = result else { - panic!("terminal cancellation must return its saved state"); - }; - assert!(matches!( - saved.state, - ResourceRequestState::Finished { - outcome: ExitReason::Cancelled - } - )); - assert_eq!(identity_count(&store, terminal.task_id), 0); -} - -#[test] -fn failed_receipt_write_rolls_back_the_queue_cancellation() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - let request = RequestId::new(); - let task = TaskId::new(); - store - .accept_resource_request(authority, request, task, resource.id, origin, spec()) - .unwrap(); - let queued = resource_cancellation_identity( - Uuid::now_v7(), - request, - task, - resource.id, - origin, - authority, - ResourceRoutePhase::Waiting, - ); - let unknown = resource_cancellation_identity( - Uuid::now_v7(), - RequestId::new(), - TaskId::new(), - resource.id, - origin, - authority, - ResourceRoutePhase::AcceptanceUnknown, - ); - store - .conn - .execute_batch( - "CREATE TRIGGER reject_receipt BEFORE INSERT ON resource_cancellation_receipts - BEGIN SELECT RAISE(ABORT, 'receipt insert failed'); END;", - ) - .unwrap(); - - // a crash before the receipt must leave no cancellation or prevention behind - for (identity, phase) in [ - (&queued, ResourceRoutePhase::Waiting), - (&unknown, ResourceRoutePhase::AcceptanceUnknown), - ] { - let proof = resource_cancellation_proof(identity, &spec(), phase); - assert!( - store - .cancel_resource_request_with_receipt(authority, identity.clone(), proof) - .is_err() - ); - } - assert!(matches!( - store.resource_requests(authority, resource.id).unwrap()[0].state, - ResourceRequestState::Queued - )); - assert_eq!(prevention_count(&store), 0); - - store - .conn - .execute_batch("DROP TRIGGER reject_receipt") - .unwrap(); - let receipt = store - .cancel_resource_request_with_receipt( - authority, - queued.clone(), - resource_cancellation_proof(&queued, &spec(), ResourceRoutePhase::Waiting), - ) - .unwrap(); - assert_eq!( - receipt.outcome, - ResourceCancellationOutcome::CancelledBeforeLaunch - ); -} diff --git a/src/store/resource/tests/completed_boundary.rs b/src/store/resource/tests/completed_boundary.rs deleted file mode 100644 index a54663d..0000000 --- a/src/store/resource/tests/completed_boundary.rs +++ /dev/null @@ -1,376 +0,0 @@ -//! A registered trainer that ended outside a loan and the queued work that follows it - -use super::fixtures::{ - TrainerAssociationFixture, completion, machine_other_than, publish_completed_result, spec, -}; -use crate::domain::{ExitReason, ProcessGroupExitEvidence, ProcessStatus, TaskEnv, TaskId}; -use crate::machine::MachineId; -use crate::resource::store::{ - AssignedResourceTaskReconcileInput, CompleteReleaseError, ReleaseCompletionResult, - ResourceTaskAcceptance, ResourceTaskAcceptanceInput, ResourceTaskCompletionResult, -}; -use crate::resource::{ - DeliveryAttemptId, Loan, LoanPhase, LoanState, ResourceQueueAttentionReason, - ResourceQueueReconcileOutcome, ResourceRequest, ResourceRequestState, ResourceRevision, - ReturnContext, ServingReleaseProvenance, SupervisorAddress, -}; -use crate::submission::RequestId; - -fn queue(fixture: &mut TrainerAssociationFixture) -> ResourceRequest { - fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - spec(), - ) - .unwrap() -} - -fn reconcile(fixture: &mut TrainerAssociationFixture) -> ResourceQueueReconcileOutcome { - fixture - .store - .reconcile_resource_queue_for_authority(fixture.authority, fixture.resource.id) - .unwrap() -} - -fn saved_revision(fixture: &TrainerAssociationFixture) -> ResourceRevision { - fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap()[0] - .resource - .state_revision -} - -fn saved_loan(fixture: &TrainerAssociationFixture) -> Option { - fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap() - .remove(0) - .loan -} - -/// Accept the assigned request, end it with confirmed exit, and record its result -fn run_assigned( - fixture: &mut TrainerAssociationFixture, - loan: &Loan, - request: &ResourceRequest, -) -> ResourceTaskCompletionResult { - let input = ResourceTaskAcceptanceInput { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - request_id: request.request_id, - task_id: request.task_id, - acceptance_sequence: request.acceptance_sequence, - loan_id: loan.id, - expected_state_revision: saved_revision(fixture), - command_spec: request.spec().clone(), - executor_env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - }; - assert_eq!( - fixture.store.accept_assigned_resource_task(input).unwrap(), - ResourceTaskAcceptance::Inserted { - task: request.task_id - } - ); - fixture - .store - .cas_status( - request.task_id, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - fixture - .store - .cas_exit_with_evidence( - request.task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ) - .unwrap() - .unwrap(); - let result = fixture - .store - .reconcile_assigned_resource_task_for_authority(AssignedResourceTaskReconcileInput { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - loan_id: loan.id, - request_id: request.request_id, - task_id: request.task_id, - expected_state_revision: saved_revision(fixture), - }) - .unwrap(); - completion(result).unwrap() -} - -#[test] -fn completed_trainer_serves_queued_work_through_its_release_proof_and_returns() { - let mut fixture = TrainerAssociationFixture::new(); - fixture.insert_accepted_running_task(); - let binding = fixture.register_release_attempt(); - // the epoch ends normally while no loan exists - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let first = queue(&mut fixture); - let second = queue(&mut fixture); - - // the completed trainer opens the ordinary release action, not a free resource - let ResourceQueueReconcileOutcome::ReleaseRequired { loan, notice } = reconcile(&mut fixture) - else { - panic!("a completed trainer must open one release action"); - }; - assert!(matches!( - loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { observed_background_task, .. } - } if observed_background_task == fixture.task_id - )); - - // only the authority-built proof of the final result serves the next request in serving order - let completed = fixture - .store - .complete_release_for_authority( - fixture.authority, - fixture.resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(); - let ReleaseCompletionResult::Assigned { loan, request, .. } = completed.clone() else { - panic!("the verified completed result must assign the next request in serving order"); - }; - assert_eq!(request.request_id, first.request_id); - let LoanState::Active { - phase: - LoanPhase::Serving { - return_context, - release_provenance, - .. - }, - } = &loan.state - else { - panic!("the assignment must serve the request"); - }; - assert!(matches!( - return_context, - ReturnContext::AlreadyCompleted { task_id, result_ref } - if *task_id == fixture.task_id && result_ref.contains("#sha256=") - )); - assert!(matches!( - release_provenance, - ServingReleaseProvenance::CompletedTrainerResult { action_id, task_id, .. } - if *action_id == notice.action_id && *task_id == fixture.task_id - )); - // an exact retry answers from the saved receipt - let retried = fixture - .store - .complete_release_for_authority( - fixture.authority, - fixture.resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(); - assert!(matches!( - retried, - ReleaseCompletionResult::Assigned { loan: same, .. } if same == loan - )); - - // the queue drains under the same loan and return context, then reserves the return - let ResourceTaskCompletionResult::Assigned { - loan, next_request, .. - } = run_assigned(&mut fixture, &loan, &request) - else { - panic!("the second request must keep the loan"); - }; - assert_eq!(next_request.request_id, second.request_id); - let ResourceTaskCompletionResult::ReturnRequired { - loan: returning, .. - } = run_assigned(&mut fixture, &loan, &next_request) - else { - panic!("the drained queue must reserve the return"); - }; - assert!(matches!( - &returning.state, - LoanState::Active { - phase: LoanPhase::AwaitingReturn { return_context: saved, .. } - } if saved == return_context - )); -} - -#[test] -fn unsafe_trainer_ends_keep_queued_work_blocked() { - type Setup = fn(&mut TrainerAssociationFixture); - let cases: [(&str, Setup); 4] = [ - ("failed unconfirmed", |fixture| { - fixture.register_release_attempt(); - fixture - .store - .cas_exit_with_evidence( - fixture.task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 1 }, - ProcessGroupExitEvidence::Unconfirmed, - ) - .unwrap() - .unwrap(); - }), - ("lost", |fixture| { - fixture.register_release_attempt(); - fixture - .store - .cas_status(fixture.task_id, ProcessStatus::Running, ProcessStatus::Lost) - .unwrap() - .unwrap(); - }), - ("unconfirmed", |fixture| { - let binding = fixture.register_release_attempt(); - publish_completed_result(fixture, &binding, ProcessGroupExitEvidence::Unconfirmed); - }), - ("no association", |fixture| { - fixture.finish_registered_task_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - }), - ]; - for (name, setup) in cases { - let mut fixture = TrainerAssociationFixture::new(); - fixture.insert_accepted_running_task(); - setup(&mut fixture); - let request = queue(&mut fixture); - let before = saved_revision(&fixture); - - assert!( - matches!( - reconcile(&mut fixture), - ResourceQueueReconcileOutcome::AttentionRequired { - request: blocked, - reason: ResourceQueueAttentionReason::BackgroundTaskNotRunning { task_id, .. }, - } if blocked.request_id == request.request_id && task_id == fixture.task_id - ), - "{name}" - ); - assert!(saved_loan(&fixture).is_none(), "{name}"); - assert_eq!(saved_revision(&fixture), before, "{name}"); - } -} - -#[test] -fn completed_trainer_without_a_verifiable_release_keeps_its_action_reserved() { - // an unpublished final result with a held lock, or a published result with a - // held lock, is not a release - for published in [false, true] { - let mut fixture = TrainerAssociationFixture::new(); - fixture.insert_accepted_running_task(); - let binding = fixture.register_release_attempt(); - if published { - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - } else { - fixture.finish_registered_task_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - } - let _lock = fixture.hold_saved_lock(); - let request = queue(&mut fixture); - let ResourceQueueReconcileOutcome::ReleaseRequired { loan, notice } = - reconcile(&mut fixture) - else { - panic!("the saved facts must open the release action"); - }; - - let error = fixture - .store - .complete_release_for_authority( - fixture.authority, - fixture.resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap_err(); - assert!( - matches!(&error, CompleteReleaseError::OwnershipLockStillHeld { .. }), - "unexpected release error: {error:?}" - ); - assert_eq!(saved_loan(&fixture), Some(loan)); - assert!(matches!( - fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|saved| saved.request_id == request.request_id) - .map(|saved| saved.state), - Some(ResourceRequestState::Queued) - )); - } -} - -#[test] -fn delivered_release_completes_after_the_supervisor_is_replaced() { - let mut fixture = TrainerAssociationFixture::new(); - fixture.insert_accepted_running_task(); - let binding = fixture.register_release_attempt(); - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let request = queue(&mut fixture); - let ResourceQueueReconcileOutcome::ReleaseRequired { notice, .. } = reconcile(&mut fixture) - else { - panic!("a completed trainer must open one release action"); - }; - let attempt = DeliveryAttemptId::new(); - fixture - .store - .reserve_supervisor_notice_attempt(notice.id, attempt) - .unwrap(); - fixture - .store - .settle_supervisor_notice_attempt(notice.id, attempt, Ok(())) - .unwrap(); - - // the delivered notice keeps the assignment whose supervisor received it - let replaced = fixture - .store - .replace_resource_supervisor( - fixture.authority, - fixture.resource.id, - saved_revision(&fixture), - SupervisorAddress { - machine: machine_other_than(fixture.resource.supervisor.machine), - thread: fixture.resource.supervisor.thread, - }, - ) - .unwrap(); - assert!(replaced.retargeted.is_empty()); - - let completed = fixture - .store - .complete_release_for_authority( - fixture.authority, - fixture.resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(); - assert!(matches!( - completed, - ReleaseCompletionResult::Assigned { request: assigned, .. } - if assigned.request_id == request.request_id - )); -} diff --git a/src/store/resource/tests/container.rs b/src/store/resource/tests/container.rs deleted file mode 100644 index 6489936..0000000 --- a/src/store/resource/tests/container.rs +++ /dev/null @@ -1,603 +0,0 @@ -//! Container workloads as queued requests and return work - -use std::os::unix::fs::PermissionsExt; -use std::path::{Path, PathBuf}; - -use serde_json::json; - -use super::fixtures::{ - ServingFixture, accept_and_finish_resource_task, completion, refresh_serving_fixture, - serving_fixture, task_reconcile_input, -}; -use super::operator_release::attestation_for; -use super::restore::{ - accept_post_return_request, awaiting_return_fixture, completed_task, - read_return_execution_mode, reconcile_restore, request_state, restore_closure_basis, - saved_loan, saved_resource, start_task, -}; -use crate::domain::{ - ContainerExitEvidence, ContainerId, ExitReason, ProcessGroupExitEvidence, ProcessStatus, - TaskEnv, TaskExitEvidence, TaskId, -}; -use crate::machine::MachineId; -use crate::resource::operator_release::{ - AttestedTrainerEnd, AttestedTrainerLaunch, OperatorGpuFreeOutcome, OperatorStateBinding, -}; -use crate::resource::store::{ - AssignedResourceTaskAttention, AssignedResourceTaskReconcileOutcome, ResourceTaskAcceptance, - ResourceTaskAcceptanceInput, ResourceTaskCompletionResult, -}; -use crate::resource::{ - CommandSpec, CommandSpecError, IdleBoundaryProof, LoanClosure, LoanPhase, LoanState, - ResourceQueueReconcileOutcome, ResourceRequestState, RestoreAttentionReason, ReturnContext, - ReturnDecision, ReturnExecutionMode, ReturnLaunch, ReturnWork, SupervisorActionAuthority, -}; -use crate::spec::{NormalizedSpec, NormalizedWorkload}; -use crate::store::{ - EndedRestoreResolution, RestoreReconcileOutcome, ReturnDecisionError, ReturnTaskAcceptance, - ReturnTaskAcceptanceInput, ReturnTaskOrigin, Store, -}; -use crate::submission::{CallbackExecutable, RequestId}; - -fn container_spec(fixture: &ServingFixture, gpus: bool) -> NormalizedSpec { - let mut workload = json!({ - "image": format!("eval@sha256:{}", "0".repeat(64)), - "args": ["--checkpoint", "/data/ckpt"], - "memory": "8g", - "mounts": [{ "source": fixture.directory.path(), "target": "/data", "read_only": true }] - }); - if gpus { - workload["gpus"] = json!("all"); - } - let mut spec = fixture.spec.clone(); - spec.name = crate::domain::TaskName::parse("container evaluation").unwrap(); - spec.workload = NormalizedWorkload::Container(Box::new( - crate::container::ContainerWorkload::from_value(&workload).unwrap(), - )); - spec -} - -/// Executor environment with a `docker` CLI on its PATH -/// -/// Store transitions only resolve the CLI; they never run it -fn docker_env(fixture: &ServingFixture) -> TaskEnv { - let bin = fixture.directory.path().join("bin"); - std::fs::create_dir_all(&bin).unwrap(); - let docker = bin.join("docker"); - std::fs::write(&docker, "#!/bin/sh\nexit 1\n").unwrap(); - std::fs::set_permissions(&docker, std::fs::Permissions::from_mode(0o755)).unwrap(); - TaskEnv { - path: bin.display().to_string(), - home: "/tmp".into(), - } -} - -fn container_id(byte: &str) -> ContainerId { - ContainerId::parse(&byte.repeat(64)).unwrap() -} - -fn removed(exit_code: i32) -> TaskExitEvidence { - TaskExitEvidence { - process_group: ProcessGroupExitEvidence::Unconfirmed, - container: ContainerExitEvidence::Confirmed { - container_id: container_id("c"), - exit_code, - }, - } -} - -fn finish(store: &Store, task_id: TaskId, reason: ExitReason, evidence: TaskExitEvidence) { - store - .cas_exit_with_evidence(task_id, ProcessStatus::Running, &reason, evidence) - .unwrap() - .unwrap(); -} - -#[test] -fn resource_containers_must_name_their_gpus() { - let fixture = serving_fixture(true, true); - assert!(matches!( - CommandSpec::try_from(container_spec(&fixture, false)), - Err(CommandSpecError::ContainerWithoutGpus) - )); - CommandSpec::try_from(container_spec(&fixture, true)).unwrap(); -} - -/// Serve the fixture's first request, then assign a queued container request -fn assigned_container_request() -> ServingFixture { - let mut fixture = serving_fixture(true, true); - let spec = container_spec(&fixture, true); - let container = fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - spec, - ) - .unwrap(); - let input = accept_and_finish_resource_task( - &mut fixture, - ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let Ok(ResourceTaskCompletionResult::Assigned { - loan, next_request, .. - }) = completion( - fixture - .store - .reconcile_assigned_resource_task_for_authority(input) - .unwrap(), - ) - else { - panic!("the queued container request must be assigned next"); - }; - assert_eq!(next_request.request_id, container.request_id); - refresh_serving_fixture(&mut fixture, loan, next_request); - fixture -} - -fn accept_container_task(fixture: &mut ServingFixture) -> TaskId { - let env = docker_env(fixture); - let input = ResourceTaskAcceptanceInput { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - request_id: fixture.request.request_id, - task_id: fixture.request.task_id, - acceptance_sequence: fixture.request.acceptance_sequence, - loan_id: fixture.loan.id, - expected_state_revision: fixture.state_revision, - command_spec: fixture.request.spec().clone(), - executor_env: env, - }; - assert_eq!( - fixture.store.accept_assigned_resource_task(input).unwrap(), - ResourceTaskAcceptance::Inserted { - task: fixture.request.task_id, - } - ); - let task_id = fixture.request.task_id; - let row = fixture.store.require_task(task_id).unwrap(); - assert!( - row.binary.ends_with("bin/docker"), - "{}", - row.binary.display() - ); - start_task(&fixture.store, task_id); - task_id -} - -#[test] -fn queued_container_releases_its_turn_only_after_its_container_is_removed() { - let mut fixture = assigned_container_request(); - let task_id = accept_container_task(&mut fixture); - - finish( - &fixture.store, - task_id, - ExitReason::Exit { code: 4 }, - removed(4), - ); - let Ok(ResourceTaskCompletionResult::ReturnRequired { - finished_request, .. - }) = completion( - fixture - .store - .reconcile_assigned_resource_task_for_authority(task_reconcile_input(&fixture)) - .unwrap(), - ) - else { - panic!("a removed container releases the serving turn"); - }; - assert_eq!( - finished_request.state, - ResourceRequestState::Finished { - outcome: ExitReason::Exit { code: 4 } - } - ); - let receipt: String = fixture - .store - .conn - .query_row( - "SELECT json_extract(receipt_json, '$.release_proof') - FROM resource_task_completions WHERE task_id = ?1", - [task_id.to_string()], - |row| row.get(0), - ) - .unwrap(); - assert!(receipt.contains("confirmed_container_removed"), "{receipt}"); -} - -#[test] -fn queued_container_without_container_evidence_keeps_the_resource() { - for (name, evidence) in [ - ( - "unconfirmed container", - TaskExitEvidence { - process_group: ProcessGroupExitEvidence::Unconfirmed, - container: ContainerExitEvidence::Unconfirmed, - }, - ), - // the worker's own Docker clients exited, which says nothing about the container - ( - "process group only", - ProcessGroupExitEvidence::ConfirmedExited.into(), - ), - ] { - let mut fixture = assigned_container_request(); - let task_id = accept_container_task(&mut fixture); - finish( - &fixture.store, - task_id, - ExitReason::Exit { code: 0 }, - evidence, - ); - let outcome = fixture - .store - .reconcile_assigned_resource_task_for_authority(task_reconcile_input(&fixture)) - .unwrap(); - assert!( - matches!( - outcome, - AssignedResourceTaskReconcileOutcome::Attention( - AssignedResourceTaskAttention::ExitWitnessUnconfirmed - ) - ), - "{name}: {outcome:?}" - ); - assert!(matches!( - request_state(&fixture, fixture.request.request_id), - Some(ResourceRequestState::Assigned { .. }) - )); - } -} - -#[test] -fn a_container_that_never_started_releases_its_turn() { - let mut fixture = assigned_container_request(); - let task_id = accept_container_task(&mut fixture); - finish( - &fixture.store, - task_id, - ExitReason::SpawnFailed { - message: "No such image".into(), - }, - TaskExitEvidence { - process_group: ProcessGroupExitEvidence::Unconfirmed, - container: ContainerExitEvidence::NeverStarted, - }, - ); - assert!( - completion( - fixture - .store - .reconcile_assigned_resource_task_for_authority(task_reconcile_input(&fixture)) - .unwrap(), - ) - .is_ok() - ); -} - -fn container_launch(fixture: &ServingFixture) -> ReturnLaunch { - ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::EvaluationOrNextEpoch { - completed_task: completed_task(fixture), - spec: CommandSpec::try_from(container_spec(fixture, true)).unwrap(), - }, - } -} - -fn container_input( - fixture: &ServingFixture, - authority: SupervisorActionAuthority, - launch: ReturnLaunch, -) -> ReturnTaskAcceptanceInput { - ReturnTaskAcceptanceInput { - authority, - launch, - executor_env: docker_env(fixture), - origin: ReturnTaskOrigin::Local { - callback_codex: CallbackExecutable::available(PathBuf::from("/bin/echo")), - }, - } -} - -/// Accept one container evaluation and a later queued request -fn container_restore_fixture() -> ( - ServingFixture, - SupervisorActionAuthority, - ReturnLaunch, - RequestId, -) { - let (mut fixture, authority) = awaiting_return_fixture(); - let later = accept_post_return_request(&mut fixture); - let launch = container_launch(&fixture); - let decision = ReturnDecision::Launch(Box::new(launch.clone())); - decision - .validate_for( - &ReturnContext::AlreadyCompleted { - task_id: completed_task(&fixture), - result_ref: "test serving fixture".into(), - }, - fixture.spec.thread, - ) - .unwrap(); - let ReturnTaskAcceptance::Inserted { state_revision, .. } = fixture - .store - .accept_return_task_for_authority(container_input(&fixture, authority, launch.clone())) - .unwrap() - else { - panic!("the first exact container launch must insert its task"); - }; - let mut current = authority; - current.expected_state_revision = state_revision; - (fixture, current, launch, later.request_id) -} - -fn assert_container_restore_reserved(fixture: &ServingFixture, task_id: TaskId, later: RequestId) { - assert!(matches!( - saved_loan(&fixture.store, fixture.authority).map(|loan| loan.state), - Some(LoanState::Active { - phase: LoanPhase::Restoring { resume_task_id, .. } - }) if resume_task_id == task_id - )); - assert_ne!( - saved_resource(&fixture.store, fixture.authority).registered_background_task, - Some(task_id) - ); - assert_eq!( - request_state(fixture, later), - Some(ResourceRequestState::Queued) - ); -} - -#[test] -fn container_return_holds_its_loan_and_closes_only_on_a_confirmed_removal() { - let (mut fixture, authority, launch, later) = container_restore_fixture(); - let task_id = launch.task_id; - assert_eq!( - read_return_execution_mode(&fixture), - Some(ReturnExecutionMode::Container) - ); - - start_task(&fixture.store, task_id); - assert!(matches!( - reconcile_restore(&mut fixture), - RestoreReconcileOutcome::ForegroundRunning { task_id: running, .. } if running == task_id - )); - assert_container_restore_reserved(&fixture, task_id, later); - - finish( - &fixture.store, - task_id, - ExitReason::Exit { code: 0 }, - removed(0), - ); - let RestoreReconcileOutcome::ForegroundEnded { closure, .. } = reconcile_restore(&mut fixture) - else { - panic!("exit 0 with a removed container must close the loan"); - }; - assert!(matches!( - closure.loan.state, - LoanState::Closed { - result: LoanClosure::ForegroundReturnEnded { task_id: ended, .. } - } if ended == task_id - )); - assert_eq!( - restore_closure_basis(&fixture.store, authority.action_id).as_deref(), - Some("container_ended") - ); - let ResourceQueueReconcileOutcome::IdleServing { request, proof, .. } = fixture - .store - .reconcile_resource_queue_for_authority(fixture.authority, fixture.resource.id) - .unwrap() - else { - panic!("the removed container must let the queue continue"); - }; - assert_eq!(request.request_id, later); - assert_eq!( - proof, - IdleBoundaryProof::ForegroundReturnEnded { - loan_id: authority.loan_id, - task_id, - } - ); -} - -#[test] -fn container_return_without_success_or_witness_stays_reserved_until_resolved() { - let unconfirmed = TaskExitEvidence { - process_group: ProcessGroupExitEvidence::Unconfirmed, - container: ContainerExitEvidence::Unconfirmed, - }; - let cases: [( - &str, - ExitReason, - TaskExitEvidence, - RestoreAttentionReason, - bool, - ); 4] = [ - ( - "failed with a removed container", - ExitReason::Exit { code: 2 }, - removed(2), - RestoreAttentionReason::ContainerEnded { - state: ProcessStatus::Failed, - }, - true, - ), - ( - "never started", - ExitReason::SpawnFailed { - message: "No such image".into(), - }, - TaskExitEvidence { - process_group: ProcessGroupExitEvidence::Unconfirmed, - container: ContainerExitEvidence::NeverStarted, - }, - RestoreAttentionReason::ContainerEnded { - state: ProcessStatus::Failed, - }, - true, - ), - ( - "succeeded without removal", - ExitReason::Exit { code: 0 }, - unconfirmed, - RestoreAttentionReason::ContainerExitUnconfirmed { - state: ProcessStatus::Succeeded, - }, - false, - ), - ( - "only the Docker clients exited", - ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited.into(), - RestoreAttentionReason::ContainerExitUnconfirmed { - state: ProcessStatus::Succeeded, - }, - false, - ), - ]; - for (name, reason, evidence, expected, resolvable) in cases { - let (mut fixture, authority, launch, later) = container_restore_fixture(); - let task_id = launch.task_id; - start_task(&fixture.store, task_id); - finish(&fixture.store, task_id, reason, evidence); - - match reconcile_restore(&mut fixture) { - RestoreReconcileOutcome::Attention { reason, .. } => { - assert_eq!(reason, expected, "{name}"); - } - other => panic!("{name}: expected attention, got {other:?}"), - } - assert_container_restore_reserved(&fixture, task_id, later); - - let result = fixture - .store - .resolve_ended_restore_for_authority(EndedRestoreResolution { - authority, - task_id, - reason: format!("{name} evaluation"), - }); - if resolvable { - assert!( - matches!( - result, - Ok(ref closure) if matches!( - closure.loan.state, - LoanState::Closed { result: LoanClosure::RestoreEnded { .. } } - ) - ), - "{name}: {result:?}" - ); - } else { - assert!( - matches!( - result, - Err(ReturnDecisionError::RestoreReleaseUnproven { .. }) - ), - "{name}: {result:?}" - ); - assert_container_restore_reserved(&fixture, task_id, later); - } - } -} - -#[test] -fn an_unconfirmed_container_return_is_released_by_the_foreground_operator_binding() { - let (mut fixture, authority, launch, _later) = container_restore_fixture(); - let task_id = launch.task_id; - start_task(&fixture.store, task_id); - finish( - &fixture.store, - task_id, - ExitReason::Exit { code: 0 }, - TaskExitEvidence::default(), - ); - - let direct = attestation_for( - &fixture.store, - fixture.authority, - fixture.resource.id, - task_id, - OperatorStateBinding::RestoringReturn { - loan_id: authority.loan_id, - action_id: authority.action_id, - }, - ); - assert!( - fixture - .store - .attest_trainer_gpu_free_for_authority(fixture.authority, direct) - .is_err(), - "the direct-segment binding names another execution mode" - ); - - let attestation = attestation_for( - &fixture.store, - fixture.authority, - fixture.resource.id, - task_id, - OperatorStateBinding::RestoringForegroundReturn { - loan_id: authority.loan_id, - action_id: authority.action_id, - }, - ); - let receipt = fixture - .store - .attest_trainer_gpu_free_for_authority(fixture.authority, attestation) - .unwrap() - .receipt; - assert_eq!( - receipt.evidence.trainer_end, - AttestedTrainerEnd::Finished { - outcome: ExitReason::Exit { code: 0 }, - process_group_exit: ProcessGroupExitEvidence::Unconfirmed, - container_exit: Some(ContainerExitEvidence::Unconfirmed), - } - ); - assert!(matches!( - receipt.evidence.trainer_launch, - AttestedTrainerLaunch::ContainerReturn { action_id, .. } if action_id == authority.action_id - )); - assert!(matches!( - receipt.outcome, - OperatorGpuFreeOutcome::RestoreClosedServing { .. } - )); -} - -#[test] -fn the_container_mount_sources_must_exist_on_the_authority() { - let (fixture, authority) = awaiting_return_fixture(); - let mut launch = container_launch(&fixture); - let mut spec = container_spec(&fixture, true); - let NormalizedWorkload::Container(container) = &mut spec.workload else { - unreachable!("the spec is a container"); - }; - container.mounts[0].source = Path::new("/nonexistent/homebased-test").to_path_buf(); - launch.work = ReturnWork::EvaluationOrNextEpoch { - completed_task: completed_task(&fixture), - spec: CommandSpec::try_from(spec).unwrap(), - }; - let mut store = fixture.store; - assert!( - store - .accept_return_task_for_authority(ReturnTaskAcceptanceInput { - authority, - launch, - executor_env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - origin: ReturnTaskOrigin::Local { - callback_codex: CallbackExecutable::available(PathBuf::from("/bin/echo")), - }, - }) - .is_err() - ); -} diff --git a/src/store/resource/tests/controls.rs b/src/store/resource/tests/controls.rs deleted file mode 100644 index 4a0110a..0000000 --- a/src/store/resource/tests/controls.rs +++ /dev/null @@ -1,317 +0,0 @@ -//! Operator controls for queued resource requests - -use super::fixtures::{queue_waiting_request, resource, spec}; -use crate::machine::MachineId; -use crate::resource::api::{BrowserResourceAction, QueuePlacement}; -use crate::resource::{DeliveryAttemptId, ResourceId, ResourceRequestState, ResourceRevision}; -use crate::store::{ResourceControlEffect, ResourceControlError, ResourceControlRequest, Store}; -use crate::submission::RequestId; -use rusqlite::params; -use tempfile::tempdir; -use uuid::Uuid; - -fn queue_request(store: &mut Store, authority: MachineId, resource_id: ResourceId) -> RequestId { - queue_waiting_request(store, authority, resource_id, MachineId::new(), &spec()).request_id -} - -fn state_revision( - store: &Store, - authority: MachineId, - resource_id: ResourceId, -) -> ResourceRevision { - store - .resource_snapshots_for_authority(authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == resource_id) - .unwrap() - .resource - .state_revision -} - -fn serving_order(store: &Store, authority: MachineId, resource_id: ResourceId) -> Vec { - store - .resource_read_models(authority, Some(resource_id)) - .unwrap() - .into_iter() - .next() - .unwrap() - .requests - .into_iter() - .filter(|request| request.state == ResourceRequestState::Queued) - .map(|request| request.request_id) - .collect() -} - -fn queue_ranks(store: &Store, resource_id: ResourceId) -> Vec { - store - .conn - .prepare( - "SELECT queue_rank FROM resource_requests - WHERE resource_id = ?1 ORDER BY queue_rank", - ) - .unwrap() - .query_map([resource_id.as_uuid().to_string()], |row| row.get(0)) - .unwrap() - .collect::, _>>() - .unwrap() -} - -fn move_request( - store: &mut Store, - authority: MachineId, - resource_id: ResourceId, - request_id: RequestId, - placement: QueuePlacement, - expected_revision: ResourceRevision, - operation_id: Uuid, -) -> Result { - store.begin_resource_control( - authority, - operation_id, - &ResourceControlRequest { - resource_id, - expected_revision, - action: BrowserResourceAction::MoveQueued { - request_id, - placement, - }, - }, - DeliveryAttemptId::new(), - ) -} - -#[test] -fn queued_moves_change_serving_order_keep_ranks_bounded_and_append_at_the_back() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - let first = queue_request(&mut store, authority, resource.id); - let second = queue_request(&mut store, authority, resource.id); - let third = queue_request(&mut store, authority, resource.id); - let fourth = queue_request(&mut store, authority, resource.id); - let original_ranks = queue_ranks(&store, resource.id); - let cases = [ - ( - third, - QueuePlacement::Front, - vec![third, first, second, fourth], - ), - ( - third, - QueuePlacement::Back, - vec![first, second, fourth, third], - ), - ( - fourth, - QueuePlacement::Before { request_id: second }, - vec![first, fourth, second, third], - ), - ( - fourth, - QueuePlacement::After { request_id: third }, - vec![first, second, third, fourth], - ), - ]; - - for (step, (request_id, placement, expected_order)) in cases.into_iter().enumerate() { - let previous_revision = state_revision(&store, authority, resource.id); - let started = move_request( - &mut store, - authority, - resource.id, - request_id, - placement, - previous_revision, - Uuid::now_v7(), - ) - .unwrap(); - assert!(matches!(started.effect, ResourceControlEffect::Reordered)); - assert!(!started.replayed); - let current_revision = state_revision(&store, authority, resource.id); - assert_eq!( - current_revision.get(), - previous_revision.get() + 1, - "step {step}" - ); - assert_eq!( - serving_order(&store, authority, resource.id), - expected_order - ); - assert_eq!( - store - .next_queued_resource_request(authority, resource.id) - .unwrap() - .unwrap() - .request_id, - expected_order[0], - "step {step}" - ); - assert_eq!(queue_ranks(&store, resource.id), original_ranks); - } - - let appended = queue_request(&mut store, authority, resource.id); - let mut expected = vec![first, second, third, fourth]; - expected.push(appended); - assert_eq!(serving_order(&store, authority, resource.id), expected); - assert_eq!( - queue_ranks(&store, resource.id), - [original_ranks, vec![5]].concat() - ); -} - -#[test] -fn queued_move_revision_replay_noop_and_operation_conflict_are_atomic() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - let first = queue_request(&mut store, authority, resource.id); - let second = queue_request(&mut store, authority, resource.id); - let revision = state_revision(&store, authority, resource.id); - let operation_id = Uuid::now_v7(); - - let first_result = move_request( - &mut store, - authority, - resource.id, - first, - QueuePlacement::Front, - revision, - operation_id, - ) - .unwrap(); - assert!(!first_result.replayed); - assert_eq!( - state_revision(&store, authority, resource.id).get(), - revision.get() + 1 - ); - assert_eq!( - serving_order(&store, authority, resource.id), - [first, second] - ); - - let replay = move_request( - &mut store, - authority, - resource.id, - first, - QueuePlacement::Front, - revision, - operation_id, - ) - .unwrap(); - assert!(replay.replayed); - assert!(matches!(replay.effect, ResourceControlEffect::Reordered)); - assert_eq!( - state_revision(&store, authority, resource.id).get(), - revision.get() + 1 - ); - assert_eq!( - serving_order(&store, authority, resource.id), - [first, second] - ); - - assert!(matches!( - move_request( - &mut store, - authority, - resource.id, - first, - QueuePlacement::Back, - revision, - operation_id, - ), - Err(ResourceControlError::Conflict) - )); - assert_eq!( - state_revision(&store, authority, resource.id).get(), - revision.get() + 1 - ); - assert_eq!( - serving_order(&store, authority, resource.id), - [first, second] - ); -} - -#[test] -fn queued_move_refuses_nonqueued_requests_invalid_anchors_and_stale_revisions() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let main_resource = resource(authority); - store.register_resource(authority, &main_resource).unwrap(); - let moving = queue_request(&mut store, authority, main_resource.id); - let queued_anchor = queue_request(&mut store, authority, main_resource.id); - let nonqueued = queue_request(&mut store, authority, main_resource.id); - let other_resource = resource(authority); - store.register_resource(authority, &other_resource).unwrap(); - let other_anchor = queue_request(&mut store, authority, other_resource.id); - store - .conn - .execute( - "UPDATE resource_requests SET state_json = ?1 WHERE request_id = ?2", - params![ - serde_json::to_string(&ResourceRequestState::CancelledBeforeLaunch).unwrap(), - nonqueued.0.to_string(), - ], - ) - .unwrap(); - let revision = state_revision(&store, authority, main_resource.id); - - let invalid = [ - (nonqueued, QueuePlacement::Front), - (moving, QueuePlacement::Before { request_id: moving }), - ( - moving, - QueuePlacement::Before { - request_id: nonqueued, - }, - ), - ( - moving, - QueuePlacement::After { - request_id: other_anchor, - }, - ), - (other_anchor, QueuePlacement::Front), - ]; - for (request_id, placement) in invalid { - assert!(matches!( - move_request( - &mut store, - authority, - main_resource.id, - request_id, - placement, - revision, - Uuid::now_v7(), - ), - Err(ResourceControlError::NotAllowed(_)) - )); - } - - assert!(matches!( - move_request( - &mut store, - authority, - main_resource.id, - moving, - QueuePlacement::Back, - ResourceRevision::new(revision.get() + 1), - Uuid::now_v7(), - ), - Err(ResourceControlError::StaleRevision { current }) if current == revision - )); - assert_eq!( - state_revision(&store, authority, main_resource.id), - revision - ); - assert_eq!( - serving_order(&store, authority, main_resource.id), - [moving, queued_anchor] - ); -} diff --git a/src/store/resource/tests/docker.rs b/src/store/resource/tests/docker.rs deleted file mode 100644 index fc73f4d..0000000 --- a/src/store/resource/tests/docker.rs +++ /dev/null @@ -1,392 +0,0 @@ -//! Container resource work through real daemon actors, workers, and Docker Engine -//! -//! These tests need a Docker Engine and pull a small public image. The engine -//! may have no GPU, so a `docker` wrapper first on `PATH` drops `--gpus` before -//! it calls the real CLI. Run them with `just test-docker` - -use std::os::unix::fs::PermissionsExt; -use std::path::{Path, PathBuf}; -use std::process::Command; - -use ractor::Actor; -use serde_json::json; -use tempfile::tempdir; - -use super::fixtures::{ - ServingFixture, machine_other_than, serving_fixture_at, spec, stop_test_supervisor, - wait_for_awaiting_return, wait_for_terminal_task, -}; -use crate::daemon::actors::supervisor::SUPERVISOR_TEST_LOCK; -use crate::daemon::actors::{StoreMsg, SupervisorActor, SupervisorArgs, SupervisorMsg, call}; -use crate::domain::{ContainerExitEvidence, ExitReason, ProcessStatus, TaskId, TaskRow}; -use crate::home::Home; -use crate::machine::load_or_create_machine_id; -use crate::resource::{ - CommandSpec, Loan, LoanClosure, LoanPhase, LoanState, ResourceId, ReturnDecision, ReturnLaunch, - ReturnWork, SupervisorActionAuthority, -}; -use crate::spec::{NormalizedSpec, NormalizedWorkload}; -use crate::store::{EndedRestoreResolution, Store}; -use crate::submission::RequestId; - -const IMAGE: &str = "busybox:1.37"; - -/// Real Docker CLI, a pulled image, and a GPU-dropping wrapper on `PATH` -struct DockerEngine { - _wrapper: tempfile::TempDir, - real: PathBuf, - image_id: String, -} - -impl DockerEngine { - fn prepare() -> Self { - let real = which::which("docker").expect("these tests need the docker CLI on PATH"); - let run = |args: &[&str]| { - let output = Command::new(&real).args(args).output().unwrap(); - assert!( - output.status.success(), - "docker {args:?}: {}", - String::from_utf8_lossy(&output.stderr) - ); - String::from_utf8(output.stdout).unwrap() - }; - run(&["info", "--format", "{{.ServerVersion}}"]); - run(&["pull", "--quiet", IMAGE]); - let image_id = run(&["image", "inspect", "--format", "{{.Id}}", IMAGE]) - .trim() - .to_owned(); - - let wrapper = tempdir().unwrap(); - let script = wrapper.path().join("docker"); - std::fs::write( - &script, - format!( - "#!/bin/sh\n\ - # this engine may have no GPU, so drop --gpus and its value\n\ - skip=0\n\ - for arg; do\n\ - shift\n\ - if [ \"$skip\" = 1 ]; then skip=0; continue; fi\n\ - if [ \"$arg\" = --gpus ]; then skip=1; continue; fi\n\ - set -- \"$@\" \"$arg\"\n\ - done\n\ - exec '{}' \"$@\"\n", - real.display() - ), - ) - .unwrap(); - std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o755)).unwrap(); - let path = format!( - "{}:{}", - wrapper.path().display(), - std::env::var("PATH").unwrap_or_default() - ); - // the daemon actors capture PATH as the executor environment; these - // tests run one at a time under the supervisor test lock - unsafe { std::env::set_var("PATH", path) }; - Self { - _wrapper: wrapper, - real, - image_id, - } - } - - fn container_exists(&self, task: TaskId) -> bool { - let name = crate::container::docker::container_name(task); - let output = Command::new(&self.real) - .args([ - "container", - "ls", - "--all", - "--filter", - &format!("name={name}"), - "--format", - "{{.Names}}", - ]) - .output() - .unwrap(); - assert!(output.status.success()); - String::from_utf8_lossy(&output.stdout) - .lines() - .any(|line| line == name) - } - - fn spec(&self, fixture: &ServingFixture, script: &str) -> NormalizedSpec { - let mut spec = fixture.spec.clone(); - spec.name = crate::domain::TaskName::parse("docker container test").unwrap(); - spec.workload = NormalizedWorkload::Container(Box::new( - crate::container::ContainerWorkload::from_value(&json!({ - "image": self.image_id, - "entrypoint": ["sh", "-c", script], - "gpus": "all", - "memory": "64m" - })) - .unwrap(), - )); - spec - } -} - -/// Daemon home with a Serving loan whose first request is a native command -struct DaemonFixture { - home: Home, - fixture: ServingFixture, -} - -fn daemon_fixture() -> DaemonFixture { - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let fixture = serving_fixture_at(directory, &home.db_path(), authority, true, true, spec()); - DaemonFixture { home, fixture } -} - -async fn task_row(store: &ractor::ActorRef, id: TaskId) -> TaskRow { - call(store, |reply| StoreMsg::GetTask { id, reply }) - .await - .unwrap() - .unwrap() -} - -fn output(home: &Home, task: TaskId) -> String { - std::fs::read_to_string(home.task_paths(task).output).unwrap_or_default() -} - -fn assert_removed(row: &TaskRow, exit_code: i32) { - assert!( - matches!( - &row.container_exit_evidence, - ContainerExitEvidence::Confirmed { exit_code: code, .. } if *code == exit_code - ), - "{:?}", - row.container_exit_evidence - ); -} - -/// Authority of the exact pending return action on one AwaitingReturn loan -fn return_authority( - fixture: &ServingFixture, - loan: &Loan, - database: &Path, -) -> SupervisorActionAuthority { - let LoanState::Active { - phase: LoanPhase::AwaitingReturn { action_id, .. }, - } = &loan.state - else { - panic!("the loan must await its return"); - }; - let store = Store::open(database).unwrap(); - let resource = store - .resource_snapshots_for_authority(fixture.authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == fixture.resource.id) - .unwrap() - .resource; - SupervisorActionAuthority { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - loan_id: loan.id, - action_id: *action_id, - expected_state_revision: resource.state_revision, - supervisor: resource.supervisor, - assignment_revision: resource.assignment_revision, - } -} - -fn saved_loan_state(database: &Path, loan: &Loan) -> LoanState { - let store = Store::open(database).unwrap(); - let json: String = store - .conn - .query_row( - "SELECT state_json FROM loans WHERE id = ?1", - [loan.id.as_uuid().to_string()], - |row| row.get(0), - ) - .unwrap(); - serde_json::from_str(&json).unwrap() -} - -async fn reconcile(supervisor: &ractor::ActorRef, resource: ResourceId) { - call(supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource, - reply, - }) - .await - .unwrap(); -} - -#[tokio::test] -#[ignore = "requires Docker Engine; run with just test-docker"] -async fn queued_container_runs_in_docker_and_releases_its_turn_after_removal() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - crate::runner::set_task_run_executable_for_tests(assert_cmd::cargo::cargo_bin("homebased")); - let docker = DockerEngine::prepare(); - let DaemonFixture { home, mut fixture } = daemon_fixture(); - let container = fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - machine_other_than(fixture.authority), - docker.spec(&fixture, "echo queued container output; exit 3"), - ) - .unwrap(); - let resource_id = fixture.resource.id; - drop(fixture.store); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let row = wait_for_terminal_task(&store, container.task_id).await; - assert_eq!(row.exit_reason(), Some(&ExitReason::Exit { code: 3 })); - assert_removed(&row, 3); - assert!(output(&home, container.task_id).contains("queued container output")); - assert!(!docker.container_exists(container.task_id)); - wait_for_awaiting_return(&supervisor, resource_id).await; - - stop_test_supervisor(supervisor, handle).await; -} - -/// Run the fixture's first request to the return decision, then launch one container return -async fn container_return( - script: &str, -) -> ( - DockerEngine, - Home, - ServingFixture, - ractor::ActorRef, - ractor::concurrency::JoinHandle<()>, - SupervisorActionAuthority, - Loan, - TaskId, -) { - let docker = DockerEngine::prepare(); - let DaemonFixture { home, fixture } = daemon_fixture(); - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let loan = wait_for_awaiting_return(&supervisor, fixture.resource.id).await; - let authority = return_authority(&fixture, &loan, &home.db_path()); - let launch = ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::EvaluationOrNextEpoch { - completed_task: fixture.resource.registered_background_task.unwrap(), - spec: CommandSpec::try_from(docker.spec(&fixture, script)).unwrap(), - }, - }; - let task_id = launch.task_id; - call(&supervisor, |reply| SupervisorMsg::DecideReturn { - authority, - decision: Box::new(ReturnDecision::Launch(Box::new(launch))), - reply, - }) - .await - .unwrap() - .unwrap(); - ( - docker, home, fixture, supervisor, handle, authority, loan, task_id, - ) -} - -#[tokio::test] -#[ignore = "requires Docker Engine; run with just test-docker"] -async fn container_return_closes_its_loan_after_its_container_is_removed() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - crate::runner::set_task_run_executable_for_tests(assert_cmd::cargo::cargo_bin("homebased")); - let (docker, home, fixture, supervisor, handle, _authority, loan, task_id) = - container_return("sleep 1; echo evaluation done").await; - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - - let row = wait_for_terminal_task(&store, task_id).await; - assert_eq!(row.status(), ProcessStatus::Succeeded); - assert_removed(&row, 0); - assert!(output(&home, task_id).contains("evaluation done")); - assert!(!docker.container_exists(task_id)); - reconcile(&supervisor, fixture.resource.id).await; - assert!(matches!( - saved_loan_state(&home.db_path(), &loan), - LoanState::Closed { - result: LoanClosure::ForegroundReturnEnded { task_id: ended, .. } - } if ended == task_id - )); - assert_eq!( - task_row(&store, task_id).await.status(), - ProcessStatus::Succeeded - ); - - stop_test_supervisor(supervisor, handle).await; -} - -#[tokio::test] -#[ignore = "requires Docker Engine; run with just test-docker"] -async fn failed_container_return_stays_reserved_until_the_supervisor_resolves_it() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - crate::runner::set_task_run_executable_for_tests(assert_cmd::cargo::cargo_bin("homebased")); - let (docker, home, fixture, supervisor, handle, mut authority, loan, task_id) = - container_return("echo evaluation failed; exit 4").await; - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - - let row = wait_for_terminal_task(&store, task_id).await; - assert_eq!(row.exit_reason(), Some(&ExitReason::Exit { code: 4 })); - assert_removed(&row, 4); - assert!(!docker.container_exists(task_id)); - reconcile(&supervisor, fixture.resource.id).await; - assert!(matches!( - saved_loan_state(&home.db_path(), &loan), - LoanState::Active { - phase: LoanPhase::Restoring { resume_task_id, .. } - } if resume_task_id == task_id - )); - - let resource = Store::open(&home.db_path()) - .unwrap() - .resource_snapshots_for_authority(fixture.authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == fixture.resource.id) - .unwrap() - .resource; - authority.expected_state_revision = resource.state_revision; - call(&supervisor, |reply| SupervisorMsg::ResolveEndedRestore { - resolution: Box::new(EndedRestoreResolution { - authority, - task_id, - reason: "evaluation failed; no background work".into(), - }), - reply, - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - saved_loan_state(&home.db_path(), &loan), - LoanState::Closed { - result: LoanClosure::RestoreEnded { - outcome: ExitReason::Exit { code: 4 }, - .. - } - } - )); - - stop_test_supervisor(supervisor, handle).await; -} diff --git a/src/store/resource/tests/ended_trainer.rs b/src/store/resource/tests/ended_trainer.rs deleted file mode 100644 index 092fed0..0000000 --- a/src/store/resource/tests/ended_trainer.rs +++ /dev/null @@ -1,461 +0,0 @@ -//! A registered trainer that ended without a usable result or a saved checkpoint stop -//! -//! Only its exact trainer-attempt association, a confirmed exit, and the exact -//! saved lock held free through the transition can release the resource. The -//! release names the outcome and never permits a same-run resume - -use super::fixtures::{ - TrainerAssociationFixture, release_completion_fixture, saved_trainer_association_json, spec, -}; -use crate::domain::{ExitReason, ProcessGroupExitEvidence, ProcessStatus, TaskEnv, TaskId}; -use crate::machine::MachineId; -use crate::resource::ownership_lock::TrainerRequestDigest; -use crate::resource::store::{ - CompleteReleaseError, ReleaseCompletionResult, ResourceSnapshot, ResourceTaskAcceptance, - ResourceTaskAcceptanceInput, -}; -use crate::resource::{ - ActionId, CommandSpec, LoanClosure, LoanPhase, LoanState, ResourceQueueReconcileOutcome, - ResourceRequest, ResourceRequestState, ResourceRevision, ReturnContext, - ReturnDecisionRejection, ReturnLaunch, ReturnWork, ServingReleaseProvenance, - SupervisorActionAuthority, SupervisorNoticePayload, -}; -use crate::store::{ReturnDecisionError, Store}; -use crate::submission::{RequestId, normalized_spec_sha256}; - -fn queue(fixture: &mut TrainerAssociationFixture) -> ResourceRequest { - fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - spec(), - ) - .unwrap() -} - -fn fail_trainer( - fixture: &mut TrainerAssociationFixture, - code: i32, - evidence: ProcessGroupExitEvidence, -) { - fixture - .store - .cas_exit_with_evidence( - fixture.task_id, - ProcessStatus::Running, - &ExitReason::Exit { code }, - evidence, - ) - .unwrap() - .unwrap(); - fixture - .store - .update_execution_state(fixture.task_id, ProcessStatus::Failed) - .unwrap(); -} - -fn complete( - fixture: &mut TrainerAssociationFixture, - action_id: ActionId, - revision: ResourceRevision, -) -> Result { - fixture.store.complete_release_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ) -} - -fn saved_snapshot(fixture: &TrainerAssociationFixture) -> ResourceSnapshot { - fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap() - .remove(0) -} - -fn request_state( - fixture: &TrainerAssociationFixture, - request_id: RequestId, -) -> ResourceRequestState { - fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|request| request.request_id == request_id) - .unwrap() - .state -} - -fn saved_request_digest(fixture: &TrainerAssociationFixture) -> TrainerRequestDigest { - let association: serde_json::Value = serde_json::from_str(&saved_trainer_association_json( - &fixture.store, - fixture.task_id, - )) - .unwrap(); - TrainerRequestDigest::from_hex(association["request_sha256"].as_str().unwrap()).unwrap() -} - -#[test] -fn failed_trainer_with_its_released_lock_serves_the_next_request() { - let (mut fixture, binding, first_request, action_id, revision, loan_id) = - release_completion_fixture(); - let second = queue(&mut fixture); - // artifacts on disk do not change the basis of a failed run - crate::resource::trainer_publication::tests::write_generation_for_test( - &fixture.runtime_root, - &binding, - "generation-before-failure", - 12, - ); - fail_trainer(&mut fixture, 3, ProcessGroupExitEvidence::ConfirmedExited); - - let result = complete(&mut fixture, action_id, revision).unwrap(); - let ReleaseCompletionResult::Assigned { - loan, - request, - state_revision, - } = &result - else { - panic!("the queued request must be assigned after the ended release"); - }; - assert_eq!(loan.id, loan_id); - assert_eq!(request.request_id, first_request); - assert_eq!(*state_revision, ResourceRevision::new(revision.get() + 1)); - let LoanState::Active { - phase: - LoanPhase::Serving { - return_context, - release_provenance, - .. - }, - } = &loan.state - else { - panic!("the assignment must serve the request"); - }; - assert_eq!( - *return_context, - ReturnContext::EndedWithoutResult { - task_id: fixture.task_id, - outcome: ExitReason::Exit { code: 3 }, - } - ); - assert_eq!( - *release_provenance, - ServingReleaseProvenance::EndedTrainerLockReleased { - action_id, - task_id: fixture.task_id, - outcome: ExitReason::Exit { code: 3 }, - attempt_request_sha256: saved_request_digest(&fixture), - } - ); - assert!(matches!( - request_state(&fixture, second.request_id), - ResourceRequestState::Queued - )); - - // the saved receipt is the provenance that lets the assigned command start - let input = ResourceTaskAcceptanceInput { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - request_id: request.request_id, - task_id: request.task_id, - acceptance_sequence: request.acceptance_sequence, - loan_id, - expected_state_revision: *state_revision, - command_spec: request.spec().clone(), - executor_env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - }; - assert_eq!( - fixture.store.accept_assigned_resource_task(input).unwrap(), - ResourceTaskAcceptance::Inserted { - task: request.task_id - } - ); -} - -#[test] -fn failed_trainer_with_a_held_lock_stays_reserved_until_the_lock_is_free() { - let (mut fixture, _, request_id, action_id, revision, _) = release_completion_fixture(); - fail_trainer(&mut fixture, 1, ProcessGroupExitEvidence::ConfirmedExited); - let before = saved_snapshot(&fixture); - let lock_holder = fixture.hold_saved_lock(); - - assert!(matches!( - complete(&mut fixture, action_id, revision), - Err(CompleteReleaseError::OwnershipLockStillHeld { task_id }) if task_id == fixture.task_id - )); - let after = saved_snapshot(&fixture); - assert_eq!(after.loan, before.loan); - assert_eq!(after.resource, before.resource); - assert!(matches!( - request_state(&fixture, request_id), - ResourceRequestState::Queued - )); - - // the escaped worker exits, so the same action now has its proof - drop(lock_holder); - let result = complete(&mut fixture, action_id, revision).unwrap(); - assert!(matches!( - result, - ReleaseCompletionResult::Assigned { request, .. } if request.request_id == request_id - )); -} - -#[test] -fn lost_unconfirmed_or_unassociated_ended_trainers_are_never_free() { - let (mut lost, _, _, action_id, revision, _) = release_completion_fixture(); - lost.store - .cas_status(lost.task_id, ProcessStatus::Running, ProcessStatus::Lost) - .unwrap() - .unwrap(); - assert!(matches!( - complete(&mut lost, action_id, revision), - Err(CompleteReleaseError::BackgroundTaskLost { task_id }) if task_id == lost.task_id - )); - - let (mut unconfirmed, _, _, action_id, revision, _) = release_completion_fixture(); - fail_trainer(&mut unconfirmed, 2, ProcessGroupExitEvidence::Unconfirmed); - assert!(matches!( - complete(&mut unconfirmed, action_id, revision), - Err(CompleteReleaseError::WorkerExitUnconfirmed { task_id }) - if task_id == unconfirmed.task_id - )); - - // no saved association names a lock, so no fallback clears the loan - let (mut unassociated, _, request_id, action_id, revision, _) = release_completion_fixture(); - fail_trainer( - &mut unassociated, - 2, - ProcessGroupExitEvidence::ConfirmedExited, - ); - unassociated - .store - .conn - .execute( - "DELETE FROM trainer_attempt_associations WHERE task_id = ?1", - [unassociated.task_id.to_string()], - ) - .unwrap(); - assert!(matches!( - complete(&mut unassociated, action_id, revision), - Err(CompleteReleaseError::TrainerAssociationMissing { task_id }) - if task_id == unassociated.task_id - )); - assert!(matches!( - request_state(&unassociated, request_id), - ResourceRequestState::Queued - )); -} - -#[test] -fn ended_release_with_an_empty_queue_keeps_the_supervisor_decision_boundary() { - let (mut fixture, _, request_id, action_id, revision, loan_id) = release_completion_fixture(); - let queued = fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|request| request.request_id == request_id) - .unwrap(); - fixture - .store - .cancel_resource_request_before_activation( - fixture.authority, - queued.request_id, - queued.task_id, - fixture.resource.id, - queued.origin_machine, - ) - .unwrap(); - assert!(matches!( - fixture.store.request_cancel(fixture.task_id).unwrap(), - crate::store::CancelResult::SignalWorker(_) - )); - fixture - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - - let ReleaseCompletionResult::ReturnRequired { loan, notice } = - complete(&mut fixture, action_id, revision).unwrap() - else { - panic!("the empty queue must reserve the supervisor's return decision"); - }; - let ended = ReturnContext::EndedWithoutResult { - task_id: fixture.task_id, - outcome: ExitReason::Cancelled, - }; - assert!(matches!( - &loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingReturn { action_id: return_action, return_context } - } if *return_action == notice.action_id && *return_context == ended - )); - assert_eq!( - notice.payload, - SupervisorNoticePayload::ReturnRequired { - return_context: ended.clone() - } - ); - - let authority = SupervisorActionAuthority { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - loan_id, - action_id: notice.action_id, - expected_state_revision: notice.state_revision, - supervisor: fixture.resource.supervisor, - assignment_revision: fixture.resource.assignment_revision, - }; - let executor_env = TaskEnv { - path: fixture.bin.to_string_lossy().into_owned(), - home: fixture.home.to_string_lossy().into_owned(), - }; - let launch = |work| ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work, - }; - - // the ended run has no resume evidence, whatever checkpoint exists - assert!(matches!( - fixture.store.prepare_return_task_for_authority( - authority, - launch(ReturnWork::SameRunResume { - stopped_task: fixture.task_id, - recovery_ref: "generation-after-cancel".into(), - }), - executor_env.clone(), - ), - Err(ReturnDecisionError::Rejected( - ReturnDecisionRejection::ResumeRequiresStoppedContext - )) - )); - // new work named for the ended run is a valid supervisor choice - let prepared = fixture - .store - .prepare_return_task_for_authority( - authority, - launch(ReturnWork::AfterEndedRun { - ended_task: fixture.task_id, - spec: CommandSpec::try_from(fixture.spec.clone()).unwrap(), - }), - executor_env, - ) - .unwrap(); - assert_eq!( - prepared.normalized_spec_sha256, - normalized_spec_sha256(&fixture.spec).unwrap() - ); - - let closure = fixture - .store - .record_no_resume_for_authority(authority, "ended run is not retried".into()) - .unwrap(); - assert!(matches!( - closure.loan.state, - LoanState::Closed { - result: LoanClosure::NoResume { return_context, .. } - } if return_context == ended - )); - assert_eq!( - saved_snapshot(&fixture).resource.registered_background_task, - None - ); -} - -#[test] -fn exact_ended_release_retry_survives_restart_and_stale_input_conflicts() { - let (mut fixture, _, _, action_id, revision, _) = release_completion_fixture(); - fail_trainer(&mut fixture, 5, ProcessGroupExitEvidence::ConfirmedExited); - let stale = ResourceRevision::new(revision.get() + 1); - assert!(matches!( - complete(&mut fixture, action_id, stale), - Err(CompleteReleaseError::StaleRevision { expected, actual }) - if expected == stale && actual == revision - )); - let other_action = ActionId::new(); - assert!(matches!( - complete(&mut fixture, other_action, revision), - Err(CompleteReleaseError::NotAwaitingRelease { action_id, .. }) if action_id == other_action - )); - - let first = complete(&mut fixture, action_id, revision).unwrap(); - // a restarted daemon reads the receipt; it needs no lock and relaunches nothing - fixture.store = Store::open(&fixture.database).unwrap(); - let _lock_holder = fixture.hold_saved_lock(); - let retry = complete(&mut fixture, action_id, revision).unwrap(); - assert_eq!( - serde_json::to_value(&retry).unwrap(), - serde_json::to_value(&first).unwrap() - ); - assert!(matches!( - complete(&mut fixture, action_id, stale), - Err(CompleteReleaseError::ConflictingRetry { action_id: conflict }) if conflict == action_id - )); -} - -#[test] -fn trainer_that_failed_before_a_loan_opens_a_release_action_and_serves_queued_work() { - let mut fixture = TrainerAssociationFixture::new(); - fixture.insert_accepted_running_task(); - fixture.register_release_attempt(); - fail_trainer(&mut fixture, 7, ProcessGroupExitEvidence::ConfirmedExited); - let first = queue(&mut fixture); - let second = queue(&mut fixture); - - // the ended trainer opens the ordinary release action, not a free resource - let ResourceQueueReconcileOutcome::ReleaseRequired { loan, notice } = fixture - .store - .reconcile_resource_queue_for_authority(fixture.authority, fixture.resource.id) - .unwrap() - else { - panic!("an ended trainer with its association must open one release action"); - }; - assert!(matches!( - loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { observed_background_task, .. } - } if observed_background_task == fixture.task_id - )); - assert!(matches!( - request_state(&fixture, first.request_id), - ResourceRequestState::Queued - )); - - let result = complete(&mut fixture, notice.action_id, notice.state_revision).unwrap(); - let ReleaseCompletionResult::Assigned { request, .. } = &result else { - panic!("the proved release must assign the next request in serving order"); - }; - assert_eq!(request.request_id, first.request_id); - assert_eq!( - saved_snapshot(&fixture).loan.map(|loan| loan.state), - Some(LoanState::Active { - phase: LoanPhase::Serving { - return_context: ReturnContext::EndedWithoutResult { - task_id: fixture.task_id, - outcome: ExitReason::Exit { code: 7 }, - }, - current_request_id: first.request_id, - release_provenance: ServingReleaseProvenance::EndedTrainerLockReleased { - action_id: notice.action_id, - task_id: fixture.task_id, - outcome: ExitReason::Exit { code: 7 }, - attempt_request_sha256: saved_request_digest(&fixture), - }, - } - }) - ); - assert!(matches!( - request_state(&fixture, second.request_id), - ResourceRequestState::Queued - )); -} diff --git a/src/store/resource/tests/fixtures.rs b/src/store/resource/tests/fixtures.rs deleted file mode 100644 index 0b232e3..0000000 --- a/src/store/resource/tests/fixtures.rs +++ /dev/null @@ -1,1562 +0,0 @@ -//! Shared fixtures for authority-local resource store tests - -use crate::domain::TaskRow; -use crate::submission::ResourceRouteProof; -use nix::fcntl::{Flock, FlockArg}; - -use crate::cancellation::ResourceCancellationRequestIdentity; -use crate::daemon::actors::resource::ResourceMsg; -use crate::daemon::actors::{StoreMsg, SupervisorMsg, call}; -use crate::domain::{ - ExitReason, ProcessGroupExitEvidence, ProcessStatus, TaskEnv, TaskId, TaskWorkload, ThreadId, - Workload, -}; -use crate::home::Home; -use crate::machine::MachineId; -use crate::resource::ownership_lock::{ - OwnershipLockIdentity, TrainerRequestDigest, VerifiedTrainerAttempt, - build_trainer_attempt_registration_evidence, test_support as trainer_attempt_test_support, -}; -use crate::resource::release_watcher::ReleaseWatcherCommand; -use crate::resource::store::{ - AssignedResourceTaskReconcileInput, AssignedResourceTaskReconcileOutcome, - OpenReleaseLoanResult, ReleaseCheckpointCancellationOutcome, ReleaseCompletionResult, - ReleaseWatcherAcceptance, ReleaseWatcherAcceptanceError, ReleaseWatcherAcceptanceInput, - ResourceTaskAcceptance, ResourceTaskAcceptanceInput, ResourceTaskCompletionResult, - TrainerAttemptAssociationStoreError, -}; -use crate::resource::trainer_publication::AttemptBinding; -use crate::resource::{ - ActionId, AssignmentRevision, Loan, LoanId, LoanPhase, LoanState, ReleaseCheckpointBaseline, - ReleaseCheckpointCancellation, ReleaseCheckpointStopDecision, ReleaseCheckpointStopOutcome, - ReleaseWatcherIntent, ReleaseWatcherTaskId, Resource, ResourceId, ResourceRequest, - ResourceRevision, ReturnContext, ServingReleaseProvenance, SupervisorAddress, SupervisorNotice, - TrainerAttemptAssociation, -}; -use crate::spec::{NormalizedSpec, NormalizedWorkload}; -use crate::store::resource::trainer_association::{ - trainer_association_by_resource_and_task, trainer_association_json, -}; -use crate::store::{ExecutorIdentity, NewTask, Store, new_queued_task}; -use crate::submission::{ - CallbackContext, CallbackExecutable, NewResourceRoute, OriginRoute, PreAcceptanceRejection, - RequestId, ResourceQueueOutcome, ResourceQueueReceipt, ResourceRoutePhase, - normalized_spec_sha256, -}; -use rusqlite::params; -use serde_json::json; -use std::fs::{self, File, OpenOptions}; -use std::os::unix::fs::PermissionsExt; -use std::path::{Path, PathBuf}; -use tempfile::tempdir; -use uuid::Uuid; - -pub(super) fn resource(authority: MachineId) -> Resource { - Resource::new( - ResourceId::new(), - "gpu-0".into(), - authority, - SupervisorAddress { - machine: authority, - thread: ThreadId(Uuid::now_v7()), - }, - AssignmentRevision::new(0), - ResourceRevision::new(0), - None, - ) -} - -pub(super) fn machine_other_than(machine: MachineId) -> MachineId { - let first = MachineId::from_uuid(Uuid::from_u128(1)); - if first != machine { - return first; - } - - MachineId::from_uuid(Uuid::from_u128(2)) -} - -/// Hold an exclusive lock on `path` the way a running trainer worker does -/// -/// No child inherits the test descriptor, so the lock may end when the guard drops -pub(super) fn acquire_test_lock(path: &Path, create: bool) -> Flock { - let mut options = OpenOptions::new(); - options.read(true).write(true); - if create { - options.create_new(true); - } - let file = options.open(path).unwrap(); - Flock::lock(file, FlockArg::LockExclusiveNonblock) - .map_err(|(_, errno)| errno) - .unwrap() -} - -pub(super) fn spec() -> NormalizedSpec { - serde_json::from_value(json!({ - "api_version": 1, - "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "resource command", - "cwd": "/tmp", - "timeout": "4h", - "workload": { "type": "task", "command": ["/bin/echo", "hello"] } - })) - .unwrap() -} - -pub(super) fn resource_cancellation_identity( - cancellation: uuid::Uuid, - request: RequestId, - task: TaskId, - resource: ResourceId, - origin: MachineId, - authority: MachineId, - target_phase: ResourceRoutePhase, -) -> ResourceCancellationRequestIdentity { - ResourceCancellationRequestIdentity { - requester_machine: origin, - cancellation, - request, - task, - origin_machine: origin, - authority_machine: authority, - resource, - target_phase, - } -} - -pub(super) fn resource_cancellation_proof( - identity: &ResourceCancellationRequestIdentity, - normalized_spec: &NormalizedSpec, - phase: ResourceRoutePhase, -) -> ResourceRouteProof { - ResourceRouteProof { - request: identity.request, - task: identity.task, - origin_machine: identity.origin_machine, - authority_machine: identity.authority_machine, - resource: identity.resource, - thread: normalized_spec.thread, - normalized_spec_sha256: crate::submission::normalized_spec_sha256(normalized_spec).unwrap(), - phase, - } -} - -pub(super) fn remote_task(task: TaskId, spec: &NormalizedSpec) -> TaskRow { - let NormalizedWorkload::Task(workload) = spec.workload.clone() else { - panic!("resource spec must be a command"); - }; - new_queued_task(NewTask { - id: task, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: Workload::Task(TaskWorkload { - command: workload.command, - }), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - binary: PathBuf::from("/bin/echo"), - }) -} - -pub(super) fn direct_segment_spec( - trainer_root: &Path, - task_file: &Path, - input_root: &Path, - runtime_root: &Path, -) -> NormalizedSpec { - serde_json::from_value(json!({ - "api_version": 1, - "thread": Uuid::now_v7(), - "name": "direct segment trainer", - "cwd": trainer_root, - "timeout": "4h", - "workload": { - "type": "task", - "command": [ - "python3", - "-m", - "ops.run_segment", - "run", - "--task", - task_file, - "--input-root", - input_root, - "--runtime-root", - runtime_root, - "--image-digest", - "test-image-digest" - ] - } - })) - .unwrap() -} - -pub(super) fn trainer_task( - task: TaskId, - spec: &NormalizedSpec, - python: &Path, - bin: &Path, - home: &Path, -) -> TaskRow { - let NormalizedWorkload::Task(workload) = spec.workload.clone() else { - panic!("trainer spec must be a command"); - }; - let requested_program = workload.command.program(); - let binary = if requested_program == "python3" { - python.to_path_buf() - } else { - PathBuf::from(requested_program) - }; - new_queued_task(NewTask { - id: task, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: Workload::Task(TaskWorkload { - command: workload.command, - }), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env: TaskEnv { - path: bin.to_string_lossy().into_owned(), - home: home.to_string_lossy().into_owned(), - }, - binary, - }) -} - -pub(super) struct TrainerAssociationFixture { - pub(super) _directory: tempfile::TempDir, - pub(super) database: PathBuf, - pub(super) store: Store, - pub(super) authority: MachineId, - pub(super) resource: Resource, - pub(super) task_id: TaskId, - pub(super) spec: NormalizedSpec, - pub(super) trainer_root: PathBuf, - pub(super) task_file: PathBuf, - pub(super) input_root: PathBuf, - pub(super) runtime_root: PathBuf, - pub(super) bin: PathBuf, - pub(super) python: PathBuf, - pub(super) home: PathBuf, -} - -impl TrainerAssociationFixture { - pub(super) fn new() -> Self { - let directory = tempdir().unwrap(); - let authority = MachineId::new(); - let database = directory.path().join("db"); - Self::new_with_database(directory, database, authority) - } - - pub(super) fn new_for_home( - directory: tempfile::TempDir, - home: &Home, - authority: MachineId, - ) -> Self { - Self::new_with_database(directory, home.db_path(), authority) - } - - pub(super) fn new_with_database( - directory: tempfile::TempDir, - database: PathBuf, - authority: MachineId, - ) -> Self { - let mut store = Store::open(&database).unwrap(); - let task_id = TaskId::new(); - let home = directory.path().canonicalize().unwrap(); - let trainer_root = home.join("trainer"); - let ops = trainer_root.join("ops"); - fs::create_dir_all(&ops).unwrap(); - fs::write(ops.join("run_segment.py"), b"# maintained trainer\n").unwrap(); - fs::write( - ops.join("segment_artifacts.py"), - b"# maintained artifacts\n", - ) - .unwrap(); - - let task_file = home.join("task.json"); - fs::write(&task_file, b"{}\n").unwrap(); - let input_root = home.join("inputs"); - fs::create_dir(&input_root).unwrap(); - let runtime_root = home.join("runtime"); - fs::create_dir(&runtime_root).unwrap(); - - let bin = home.join("bin"); - fs::create_dir(&bin).unwrap(); - let python = bin.join("python3"); - fs::write(&python, b"#!/bin/sh\nexit 0\n").unwrap(); - fs::set_permissions(&python, fs::Permissions::from_mode(0o755)).unwrap(); - - let spec = direct_segment_spec(&trainer_root, &task_file, &input_root, &runtime_root); - let mut resource = resource(authority); - resource.supervisor.thread = spec.thread; - resource.registered_background_task = Some(task_id); - store.register_resource(authority, &resource).unwrap(); - - Self { - _directory: directory, - database, - store, - authority, - resource, - task_id, - spec, - trainer_root, - task_file, - input_root, - runtime_root, - bin, - python, - home, - } - } - - pub(super) fn insert_accepted_running_task(&mut self) { - let row = self.trainer_task(self.task_id, &self.spec); - self.store - .insert_local_task( - &row, - &self.spec, - self.authority, - crate::submission::RequestId::new(), - PathBuf::from("/bin/echo").into(), - ) - .unwrap(); - self.store - .cas_status(self.task_id, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - } - - pub(super) fn evidence(&self) -> VerifiedTrainerAttempt { - VerifiedTrainerAttempt::from_persisted( - self.runtime_root.clone(), - trainer_attempt_test_support::attempt_binding("attempt-1"), - TrainerRequestDigest::from_hex(&"42".repeat(32)).unwrap(), - OwnershipLockIdentity::new(7, 11), - ) - .unwrap() - } - - pub(super) fn register_release_attempt(&mut self) -> AttemptBinding { - let binding = trainer_attempt_test_support::attempt_binding("attempt-1"); - crate::resource::trainer_publication::tests::write_request_for_test( - &self.runtime_root, - &binding, - ); - let lock_path = self.runtime_root.join(".segment.lock"); - let lock_file = acquire_test_lock(&lock_path, true); - let evidence = - build_trainer_attempt_registration_evidence(&self.runtime_root, &binding).unwrap(); - self.bind(evidence).unwrap(); - drop(lock_file); - binding - } - - pub(super) fn hold_saved_lock(&self) -> Flock { - acquire_test_lock(&self.runtime_root.join(".segment.lock"), false) - } - - pub(super) fn trainer_task(&self, task: TaskId, spec: &NormalizedSpec) -> TaskRow { - trainer_task(task, spec, &self.python, &self.bin, &self.home) - } - - pub(super) fn command(&self) -> Vec { - let NormalizedWorkload::Task(workload) = &self.spec.workload else { - panic!("trainer spec must be a command"); - }; - workload.command.to_vec() - } - - pub(super) fn set_command(&mut self, command: Vec) { - let NormalizedWorkload::Task(workload) = &mut self.spec.workload else { - panic!("trainer spec must be a command"); - }; - workload.command = crate::invocation::CommandLine::try_from_argv(command).unwrap(); - } - - pub(super) fn remove_shape_files(&self) { - fs::remove_file(&self.task_file).unwrap(); - fs::remove_dir(&self.input_root).unwrap(); - fs::remove_dir(&self.runtime_root).unwrap(); - fs::remove_dir_all(&self.trainer_root).unwrap(); - fs::remove_dir_all(&self.bin).unwrap(); - } - - pub(super) fn finish_registered_task(&mut self) { - self.store - .cas_exit( - self.task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ) - .unwrap() - .unwrap(); - self.store - .update_execution_state(self.task_id, ProcessStatus::Succeeded) - .unwrap(); - } - - pub(super) fn finish_registered_task_with_evidence( - &mut self, - evidence: ProcessGroupExitEvidence, - ) { - self.store - .cas_exit_with_evidence( - self.task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - evidence, - ) - .unwrap() - .unwrap(); - self.store - .update_execution_state(self.task_id, ProcessStatus::Succeeded) - .unwrap(); - } - - pub(super) fn finish_registered_task_cancelled_with_evidence( - &mut self, - evidence: ProcessGroupExitEvidence, - ) { - self.store - .cas_exit_with_evidence( - self.task_id, - ProcessStatus::Running, - &ExitReason::Cancelled, - evidence, - ) - .unwrap() - .unwrap(); - self.store - .update_execution_state(self.task_id, ProcessStatus::Cancelled) - .unwrap(); - } - - pub(super) fn bind( - &mut self, - evidence: VerifiedTrainerAttempt, - ) -> Result { - self.store.bind_trainer_attempt_association( - self.authority, - self.resource.id, - self.task_id, - evidence, - ) - } - - pub(super) fn add_loan(&self, state: LoanState) { - self.store - .conn - .execute( - "INSERT INTO loans (id, resource_id, state_json) VALUES (?1, ?2, ?3)", - params![ - LoanId::new().as_uuid().to_string(), - self.resource.id.as_uuid().to_string(), - serde_json::to_string(&state).unwrap(), - ], - ) - .unwrap(); - } -} - -pub(super) fn saved_trainer_association_json(store: &Store, task_id: TaskId) -> String { - store - .conn - .query_row( - "SELECT association_json FROM trainer_attempt_associations WHERE task_id=?1", - [task_id.to_string()], - |row| row.get(0), - ) - .unwrap() -} - -pub(super) fn awaiting_release_state(task_id: TaskId) -> LoanState { - LoanState::Active { - phase: LoanPhase::AwaitingRelease { - action_id: ActionId::new(), - observed_background_task: task_id, - watcher_intent: None, - }, - } -} - -pub(super) fn open_release_for_test( - store: &mut Store, - authority: MachineId, - background_task: TaskId, -) -> (Resource, Loan, SupervisorNotice) { - let mut resource = resource(authority); - resource.supervisor.thread = spec().thread; - resource.registered_background_task = Some(background_task); - store.register_resource(authority, &resource).unwrap(); - let trainer_spec = spec(); - let trainer_row = remote_task(background_task, &trainer_spec); - store - .insert_local_task( - &trainer_row, - &trainer_spec, - authority, - crate::submission::RequestId::new(), - PathBuf::from("/bin/echo").into(), - ) - .unwrap(); - store - .cas_status( - background_task, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - store - .update_execution_state(background_task, ProcessStatus::Running) - .unwrap(); - test_release_association(store, authority, resource.id, background_task); - store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - spec(), - ) - .unwrap(); - - let OpenReleaseLoanResult::Opened { loan, notice } = store - .open_release_loan_for_authority(authority, resource.id, resource.state_revision) - .unwrap() - else { - panic!("fixture must create a new release action"); - }; - - (resource, loan, notice) -} - -pub(super) fn test_release_association( - store: &mut Store, - authority: MachineId, - resource_id: ResourceId, - task_id: TaskId, -) -> PathBuf { - let runtime_root = store.tasks_dir.join(format!("release-proof-{task_id}")); - fs::create_dir_all(&runtime_root).unwrap(); - let attempt_binding = AttemptBinding { - campaign_id: "release-campaign".into(), - campaign_revision_id: "release-revision".into(), - task_id: "release-trainer-task".into(), - attempt_id: format!("attempt-{}", task_id.0.simple()), - attempt_number: 1, - ownership_token: "release-owner".into(), - }; - let evidence = VerifiedTrainerAttempt::from_persisted( - runtime_root.clone(), - attempt_binding, - TrainerRequestDigest::from_hex(&"42".repeat(32)).unwrap(), - OwnershipLockIdentity::new(7, 11), - ) - .unwrap(); - let association = TrainerAttemptAssociation::from_components( - resource_id, - authority, - task_id, - evidence, - normalized_spec_sha256(&spec()).unwrap(), - ) - .unwrap(); - store - .conn - .execute( - "INSERT INTO trainer_attempt_associations - (task_id, resource_id, authority_machine, association_json) - VALUES (?1, ?2, ?3, ?4)", - params![ - task_id.to_string(), - resource_id.as_uuid().to_string(), - authority.as_uuid().to_string(), - trainer_association_json(&association).unwrap(), - ], - ) - .unwrap(); - runtime_root -} - -pub(super) struct ServingFixture { - pub(super) directory: tempfile::TempDir, - pub(super) store: Store, - pub(super) authority: MachineId, - pub(super) origin: MachineId, - pub(super) resource: Resource, - pub(super) request: ResourceRequest, - pub(super) loan: Loan, - pub(super) state_revision: ResourceRevision, - pub(super) spec: NormalizedSpec, -} - -pub(super) fn resource_origin_route( - request: RequestId, - task: TaskId, - resource: ResourceId, - origin: MachineId, - authority: MachineId, - spec: &NormalizedSpec, -) -> OriginRoute { - OriginRoute::new_resource_waiting(NewResourceRoute { - request, - task, - origin_machine: origin, - authority_machine: authority, - thread: spec.thread, - callback: CallbackContext { - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - cwd: PathBuf::from("/tmp"), - codex: CallbackExecutable::available(PathBuf::from("/bin/echo")), - }, - spec: spec.clone(), - resource, - }) - .unwrap() -} - -pub(super) fn queue_waiting_request( - store: &mut Store, - authority: MachineId, - resource_id: ResourceId, - origin: MachineId, - spec: &NormalizedSpec, -) -> ResourceRequest { - let request_id = RequestId::new(); - let task_id = TaskId::new(); - store - .insert_origin_route(&resource_origin_route( - request_id, - task_id, - resource_id, - origin, - authority, - spec, - )) - .unwrap(); - let request = store - .accept_resource_request( - authority, - request_id, - task_id, - resource_id, - origin, - spec.clone(), - ) - .unwrap(); - store - .resolve_resource_route(&waiting_receipt( - request_id, - task_id, - resource_id, - origin, - authority, - )) - .unwrap(); - request -} - -pub(super) fn waiting_receipt( - request: RequestId, - task: TaskId, - resource: ResourceId, - origin: MachineId, - authority: MachineId, -) -> ResourceQueueReceipt { - ResourceQueueReceipt { - request, - task, - origin_machine: origin, - authority_machine: authority, - resource, - outcome: ResourceQueueOutcome::Waiting, - } -} - -pub(super) fn serving_fixture(local_origin: bool, save_local_route: bool) -> ServingFixture { - serving_fixture_with_spec(local_origin, save_local_route, spec()) -} - -pub(super) fn serving_fixture_with_spec( - local_origin: bool, - save_local_route: bool, - spec: NormalizedSpec, -) -> ServingFixture { - let directory = tempdir().unwrap(); - let database = directory.path().join("db"); - serving_fixture_at( - directory, - &database, - MachineId::new(), - local_origin, - save_local_route, - spec, - ) -} - -/// Serving fixture whose store and authority are the ones a daemon home uses -pub(super) fn serving_fixture_at( - directory: tempfile::TempDir, - database: &Path, - authority: MachineId, - local_origin: bool, - save_local_route: bool, - spec: NormalizedSpec, -) -> ServingFixture { - let mut store = Store::open(database).unwrap(); - let origin = if local_origin { - authority - } else { - MachineId::new() - }; - let background_task = TaskId::new(); - let request_id = RequestId::new(); - let task_id = TaskId::new(); - let mut resource = resource(authority); - resource.supervisor.thread = spec.thread; - resource.registered_background_task = Some(background_task); - - store.register_resource(authority, &resource).unwrap(); - store - .insert_task(&remote_task(background_task, &spec)) - .unwrap(); - store - .cas_status( - background_task, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - if local_origin && save_local_route { - store - .insert_origin_route(&resource_origin_route( - request_id, - task_id, - resource.id, - origin, - authority, - &spec, - )) - .unwrap(); - } - let request = store - .accept_resource_request( - authority, - request_id, - task_id, - resource.id, - origin, - spec.clone(), - ) - .unwrap(); - if local_origin && save_local_route { - store - .resolve_resource_route(&waiting_receipt( - request_id, - task_id, - resource.id, - origin, - authority, - )) - .unwrap(); - } - - let OpenReleaseLoanResult::Opened { .. } = store - .open_release_loan_for_authority(authority, resource.id, resource.state_revision) - .unwrap() - else { - panic!("fixture must open one release action"); - }; - store - .cas_exit( - background_task, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ) - .unwrap() - .unwrap(); - let (loan, state_revision) = store - .seed_verified_serving_loan_for_test( - authority, - resource.id, - request.request_id, - ReturnContext::AlreadyCompleted { - task_id: background_task, - result_ref: "test serving fixture".into(), - }, - ) - .unwrap(); - - ServingFixture { - directory, - store, - authority, - origin, - resource, - request, - loan, - state_revision, - spec, - } -} - -pub(super) fn acceptance_input(fixture: &ServingFixture) -> ResourceTaskAcceptanceInput { - ResourceTaskAcceptanceInput { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - request_id: fixture.request.request_id, - task_id: fixture.request.task_id, - acceptance_sequence: fixture.request.acceptance_sequence, - loan_id: fixture.loan.id, - expected_state_revision: fixture.state_revision, - command_spec: crate::resource::CommandSpec::try_from(fixture.spec.clone()).unwrap(), - executor_env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - } -} - -pub(super) fn completion( - outcome: AssignedResourceTaskReconcileOutcome, -) -> Result { - match outcome { - AssignedResourceTaskReconcileOutcome::Completed(result) => Ok(*result), - other => Err(other), - } -} - -pub(super) fn task_reconcile_input(fixture: &ServingFixture) -> AssignedResourceTaskReconcileInput { - AssignedResourceTaskReconcileInput { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - loan_id: fixture.loan.id, - request_id: fixture.request.request_id, - task_id: fixture.request.task_id, - expected_state_revision: fixture.state_revision, - } -} - -pub(super) fn accept_and_finish_resource_task( - fixture: &mut ServingFixture, - outcome: ExitReason, - evidence: ProcessGroupExitEvidence, -) -> AssignedResourceTaskReconcileInput { - let acceptance = acceptance_input(fixture); - assert_eq!( - fixture - .store - .accept_assigned_resource_task(acceptance) - .unwrap(), - ResourceTaskAcceptance::Inserted { - task: fixture.request.task_id, - } - ); - fixture - .store - .cas_status( - fixture.request.task_id, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - fixture - .store - .cas_exit_with_evidence( - fixture.request.task_id, - ProcessStatus::Running, - &outcome, - evidence, - ) - .unwrap() - .unwrap(); - task_reconcile_input(fixture) -} - -pub(super) fn refresh_serving_fixture( - fixture: &mut ServingFixture, - loan: Loan, - request: ResourceRequest, -) { - fixture.loan = loan; - fixture.request = request; - fixture.state_revision = fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == fixture.resource.id) - .unwrap() - .resource - .state_revision; -} - -pub(super) fn accept_and_finish_next_resource_task( - fixture: &mut ServingFixture, - outcome: ExitReason, - evidence: ProcessGroupExitEvidence, -) -> AssignedResourceTaskReconcileInput { - let input = ResourceTaskAcceptanceInput { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - request_id: fixture.request.request_id, - task_id: fixture.request.task_id, - acceptance_sequence: fixture.request.acceptance_sequence, - loan_id: fixture.loan.id, - expected_state_revision: fixture.state_revision, - command_spec: fixture.request.spec().clone(), - executor_env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - }; - assert_eq!( - fixture.store.accept_assigned_resource_task(input).unwrap(), - ResourceTaskAcceptance::Inserted { - task: fixture.request.task_id, - } - ); - fixture - .store - .cas_status( - fixture.request.task_id, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - fixture - .store - .cas_exit_with_evidence( - fixture.request.task_id, - ProcessStatus::Running, - &outcome, - evidence, - ) - .unwrap() - .unwrap(); - task_reconcile_input(fixture) -} - -pub(super) fn acceptance_counts(store: &Store, request: RequestId, task: TaskId) -> [i64; 7] { - [ - store - .conn - .query_row( - "SELECT COUNT(*) FROM tasks WHERE id=?1", - [task.to_string()], - |row| row.get(0), - ) - .unwrap(), - store - .conn - .query_row( - "SELECT COUNT(*) FROM executor_identities WHERE task_id=?1", - [task.to_string()], - |row| row.get(0), - ) - .unwrap(), - store - .conn - .query_row( - "SELECT COUNT(*) FROM executor_outbox WHERE task_id=?1", - [task.to_string()], - |row| row.get(0), - ) - .unwrap(), - store - .conn - .query_row( - "SELECT COUNT(*) FROM executor_event_cursors WHERE task_id=?1", - [task.to_string()], - |row| row.get(0), - ) - .unwrap(), - store - .conn - .query_row( - "SELECT COUNT(*) FROM resource_requests WHERE request_id=?1", - [request.0.to_string()], - |row| row.get(0), - ) - .unwrap(), - store - .conn - .query_row( - "SELECT COUNT(*) FROM origin_routes WHERE request_id=?1", - [request.0.to_string()], - |row| row.get(0), - ) - .unwrap(), - store - .conn - .query_row( - "SELECT COUNT(*) FROM executor_event_receipts WHERE task_id=?1", - [task.to_string()], - |row| row.get(0), - ) - .unwrap(), - ] -} - -pub(super) const TEST_WATCHER_EXECUTABLE: &str = "/bin/echo"; - -pub(super) fn release_watcher_intent( - resource_id: ResourceId, - notice: &SupervisorNotice, - background_task: TaskId, - watcher_task: TaskId, -) -> ReleaseWatcherIntent { - let watcher_task_id = ReleaseWatcherTaskId::new(watcher_task); - let command = ReleaseWatcherCommand { - resource_id, - action_id: notice.action_id, - state_revision: notice.state_revision, - trainer_task_id: background_task, - watcher_task_id, - }; - ReleaseWatcherIntent { - action_id: notice.action_id, - state_revision: notice.state_revision, - observed_background_task: background_task, - watcher_task_id, - request_id: RequestId::new(), - normalized_spec_sha256: command - .normalized_spec_sha256( - Path::new(TEST_WATCHER_EXECUTABLE), - notice.destination.thread, - ) - .unwrap(), - } -} - -pub(super) fn watcher_spec(resource: &Resource, intent: &ReleaseWatcherIntent) -> NormalizedSpec { - ReleaseWatcherCommand::from_intent(resource.id, intent) - .normalized_spec( - Path::new(TEST_WATCHER_EXECUTABLE), - resource.supervisor.thread, - ) - .unwrap() -} - -pub(super) fn prepare_release_checkpoint_baseline( - store: &mut Store, - authority: MachineId, - resource: &Resource, - intent: &ReleaseWatcherIntent, -) -> ReleaseCheckpointBaseline { - store - .bind_release_watcher_for_authority(authority, resource.id, intent.clone()) - .unwrap(); - store - .capture_release_checkpoint_baseline_for_authority( - authority, - resource.id, - intent.action_id, - intent.state_revision, - ) - .unwrap() -} - -pub(super) fn saved_release_association( - store: &Store, - resource_id: ResourceId, - task_id: TaskId, -) -> TrainerAttemptAssociation { - trainer_association_by_resource_and_task(&store.conn, resource_id, task_id) - .unwrap() - .unwrap() -} - -pub(super) fn watcher_task_and_callback( - intent: &ReleaseWatcherIntent, - spec: &NormalizedSpec, -) -> (TaskRow, CallbackContext) { - let row = remote_task(intent.watcher_task_id.as_task_id(), spec); - let callback = CallbackContext { - env: row.env.clone(), - cwd: row.cwd.clone(), - codex: CallbackExecutable::available(PathBuf::from("/bin/echo")), - }; - (row, callback) -} - -pub(super) fn watcher_acceptance_input( - resource: &Resource, - intent: ReleaseWatcherIntent, - row: &TaskRow, - spec: &NormalizedSpec, - callback: &CallbackContext, -) -> ReleaseWatcherAcceptanceInput { - ReleaseWatcherAcceptanceInput { - authority_machine: resource.authority_machine(), - resource_id: resource.id, - supervisor: resource.supervisor, - intent, - row: row.clone(), - spec: spec.clone(), - callback: callback.clone(), - } -} - -pub(super) fn accept_watcher( - store: &mut Store, - resource: &Resource, - intent: ReleaseWatcherIntent, - row: &TaskRow, - spec: &NormalizedSpec, - callback: &CallbackContext, -) -> Result { - store.accept_release_watcher_for_authority(watcher_acceptance_input( - resource, intent, row, spec, callback, - )) -} - -pub(super) fn checkpoint_decision_for_cancellation( - store: &mut Store, - authority: MachineId, - background_task: TaskId, -) -> ( - Resource, - SupervisorNotice, - ReleaseWatcherIntent, - ReleaseCheckpointStopDecision, -) { - let (resource, _, notice) = open_release_for_test(store, authority, background_task); - let association = saved_release_association(store, resource.id, background_task); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - prepare_release_checkpoint_baseline(store, authority, &resource, &intent); - crate::resource::trainer_publication::tests::write_generation_for_test( - association.verified_attempt().canonical_runtime_root(), - association.verified_attempt().binding(), - "generation-cancellation", - 81, - ); - let ReleaseCheckpointStopOutcome::Reserved(decision) = store - .reserve_release_checkpoint_stop_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap() - else { - panic!("the exact checkpoint must reserve a stop decision"); - }; - - (resource, notice, intent, decision) -} - -pub(super) fn start_release_watcher_for_test( - store: &mut Store, - resource: &Resource, - intent: &ReleaseWatcherIntent, -) { - let watcher_spec = watcher_spec(resource, intent); - let (row, callback) = watcher_task_and_callback(intent, &watcher_spec); - accept_watcher( - store, - resource, - intent.clone(), - &row, - &watcher_spec, - &callback, - ) - .unwrap(); - store - .cas_status(row.id, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - store.set_pid(row.id, std::process::id() as i32).unwrap(); - store - .update_execution_state(row.id, ProcessStatus::Running) - .unwrap(); -} - -pub(super) fn watcher_acceptance_counts( - store: &Store, - request: RequestId, - task: TaskId, -) -> [i64; 4] { - [ - store - .conn - .query_row( - "SELECT COUNT(*) FROM tasks WHERE id=?1", - [task.to_string()], - |row| row.get(0), - ) - .unwrap(), - store - .conn - .query_row( - "SELECT COUNT(*) FROM origin_routes WHERE request_id=?1", - [request.0.to_string()], - |row| row.get(0), - ) - .unwrap(), - identity_count(store, task), - store - .conn - .query_row( - "SELECT COUNT(*) FROM executor_outbox WHERE task_id=?1", - [task.to_string()], - |row| row.get(0), - ) - .unwrap(), - ] -} - -pub(super) fn prevention_count(store: &Store) -> i64 { - store - .conn - .query_row( - "SELECT COUNT(*) FROM resource_request_preventions", - [], - |row| row.get(0), - ) - .unwrap() -} - -pub(super) fn identity_count(store: &Store, task: TaskId) -> i64 { - store - .conn - .query_row( - "SELECT COUNT(*) FROM executor_identities WHERE task_id=?1", - [task.to_string()], - |row| row.get(0), - ) - .unwrap() -} - -pub(super) fn assert_cancelled_tombstone( - identity: ExecutorIdentity, - task: TaskId, - origin: MachineId, - authority: MachineId, -) { - let ExecutorIdentity::Rejected(tombstone) = identity else { - panic!("pre-activation cancellation must retain a rejection"); - }; - assert_eq!(tombstone.task, task); - assert_eq!(tombstone.origin_machine, origin); - assert_eq!(tombstone.execution_machine, authority); - assert_eq!(tombstone.reason, PreAcceptanceRejection::Cancelled.as_str()); -} - -pub(super) fn release_completion_fixture() -> ( - TrainerAssociationFixture, - AttemptBinding, - RequestId, - ActionId, - ResourceRevision, - LoanId, -) { - release_completion_fixture_with(TrainerAssociationFixture::new(), spec(), MachineId::new()) -} - -pub(super) fn release_completion_fixture_with( - mut fixture: TrainerAssociationFixture, - request_spec: NormalizedSpec, - request_origin: MachineId, -) -> ( - TrainerAssociationFixture, - AttemptBinding, - RequestId, - ActionId, - ResourceRevision, - LoanId, -) { - fixture.insert_accepted_running_task(); - let binding = fixture.register_release_attempt(); - let request_id = RequestId::new(); - fixture - .store - .accept_resource_request( - fixture.authority, - request_id, - TaskId::new(), - fixture.resource.id, - request_origin, - request_spec, - ) - .unwrap(); - let OpenReleaseLoanResult::Opened { loan, notice } = fixture - .store - .open_release_loan_for_authority( - fixture.authority, - fixture.resource.id, - fixture.resource.state_revision, - ) - .unwrap() - else { - panic!("fixture must open one release action"); - }; - - ( - fixture, - binding, - request_id, - notice.action_id, - notice.state_revision, - loan.id, - ) -} - -pub(super) fn publish_completed_result( - fixture: &mut TrainerAssociationFixture, - binding: &AttemptBinding, - exit_evidence: ProcessGroupExitEvidence, -) { - crate::resource::trainer_publication::tests::write_completed_result_for_test( - &fixture.runtime_root, - binding, - ); - fixture.finish_registered_task_with_evidence(exit_evidence); -} - -pub(super) fn commit_stopped_release_decision( - fixture: &mut TrainerAssociationFixture, - action_id: ActionId, - revision: ResourceRevision, -) -> (ReleaseCheckpointStopDecision, ReleaseCheckpointCancellation) { - let notice_json: String = fixture - .store - .conn - .query_row( - "SELECT notice_json FROM resource_supervisor_notices WHERE action_id = ?1", - [action_id.as_uuid().to_string()], - |row| row.get(0), - ) - .unwrap(); - let notice: SupervisorNotice = serde_json::from_str(¬ice_json).unwrap(); - let intent = - release_watcher_intent(fixture.resource.id, ¬ice, fixture.task_id, TaskId::new()); - prepare_release_checkpoint_baseline( - &mut fixture.store, - fixture.authority, - &fixture.resource, - &intent, - ); - crate::resource::trainer_publication::tests::write_generation_for_test( - &fixture.runtime_root, - &trainer_attempt_test_support::attempt_binding("attempt-1"), - "generation-stopped", - 81, - ); - let ReleaseCheckpointStopOutcome::Reserved(decision) = fixture - .store - .reserve_release_checkpoint_stop_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ) - .unwrap() - else { - panic!("the saved checkpoint must reserve a stopped release decision"); - }; - start_release_watcher_for_test(&mut fixture.store, &fixture.resource, &intent); - let ReleaseCheckpointCancellationOutcome::Committed(cancellation) = fixture - .store - .commit_release_checkpoint_cancellation_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision, - &decision, - ) - .unwrap() - else { - panic!("the running watcher must permit the saved stop decision"); - }; - - (decision, cancellation.cancellation) -} - -pub(super) fn fake_resource_task_spec(root: &Path, marker: &Path) -> NormalizedSpec { - native_resource_command_spec(root, "fake resource command", &[marker]) -} - -// the command stays in its foreground process group until the test creates the -// gate, so a test can inspect the Serving loan while the task is running -pub(super) fn gated_resource_task_spec(root: &Path, marker: &Path, gate: &Path) -> NormalizedSpec { - native_resource_command_spec(root, "gated resource command", &[marker, gate]) -} - -pub(super) fn native_resource_command_spec( - root: &Path, - name: &str, - args: &[&Path], -) -> NormalizedSpec { - let mut command = vec![crate::resource::foreground::test_support::native_fake_command()]; - command.extend_from_slice(args); - - serde_json::from_value(json!({ - "api_version": 1, - "thread": Uuid::now_v7(), - "name": name, - "cwd": root, - "timeout": "4h", - "workload": { "type": "task", "command": command } - })) - .unwrap() -} - -pub(super) async fn wait_for_running_task( - store: &ractor::ActorRef, - task_id: TaskId, -) -> TaskRow { - tokio::time::timeout(std::time::Duration::from_secs(15), async { - loop { - if let Some(row) = call(store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - && row.status() == ProcessStatus::Running - { - return row; - } - - tokio::time::sleep(std::time::Duration::from_millis(20)).await; - } - }) - .await - .expect("resource task must reach the running state") -} - -/// Wait for the task-terminal event path, without a client wake, to reserve the return -pub(super) async fn wait_for_awaiting_return( - supervisor: &ractor::ActorRef, - resource_id: ResourceId, -) -> Loan { - tokio::time::timeout(std::time::Duration::from_secs(15), async { - loop { - let inspection = call(supervisor, |reply| SupervisorMsg::InspectResource { - id: resource_id, - reply, - }) - .await - .unwrap() - .unwrap(); - if let Some(loan) = inspection.loan - && matches!( - loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingReturn { .. } - } - ) - { - return loan; - } - - tokio::time::sleep(std::time::Duration::from_millis(20)).await; - } - }) - .await - .expect("empty queue must reserve the return") -} - -pub(super) fn return_notice_count(home: &Home, loan_id: LoanId) -> i64 { - Store::open(&home.db_path()) - .unwrap() - .conn - .query_row( - "SELECT COUNT(*) FROM resource_supervisor_notices - WHERE loan_id = ?1 - AND json_extract(notice_json, '$.payload.type') = 'return_required'", - [loan_id.as_uuid().to_string()], - |row| row.get(0), - ) - .unwrap() -} - -pub(super) async fn wait_for_terminal_task( - store: &ractor::ActorRef, - task_id: TaskId, -) -> TaskRow { - tokio::time::timeout(std::time::Duration::from_secs(15), async { - loop { - if let Some(row) = call(store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - && row.status().is_terminal() - { - return row; - } - - tokio::task::yield_now().await; - } - }) - .await - .expect("resource task must reach a terminal state") -} - -pub(super) async fn stop_test_supervisor( - supervisor: ractor::ActorRef, - handle: ractor::concurrency::JoinHandle<()>, -) { - supervisor.stop(None); - let _ = handle.await; -} - -pub(super) async fn stop_test_resource_actor( - actor: ractor::ActorRef, - actor_handle: ractor::concurrency::JoinHandle<()>, - store: ractor::ActorRef, - store_handle: ractor::concurrency::JoinHandle<()>, -) { - actor.stop(None); - let _ = actor_handle.await; - store.stop(None); - let _ = store_handle.await; -} - -/// Assert that a release served the queue only as a non-resumable ended run -pub(super) fn assert_ended_release( - result: &ReleaseCompletionResult, - task: TaskId, - action: ActionId, - outcome: &ExitReason, -) { - let ReleaseCompletionResult::Assigned { loan, .. } = result else { - panic!("the queued request must be assigned after an ended release: {result:?}"); - }; - assert!( - matches!( - &loan.state, - LoanState::Active { - phase: LoanPhase::Serving { - return_context: ReturnContext::EndedWithoutResult { - task_id, - outcome: saved_outcome, - }, - release_provenance: ServingReleaseProvenance::EndedTrainerLockReleased { - action_id, - task_id: proven_task, - outcome: proven_outcome, - .. - }, - .. - } - } if *task_id == task - && saved_outcome == outcome - && *action_id == action - && *proven_task == task - && proven_outcome == outcome - ), - "unexpected ended release: {loan:?}" - ); -} diff --git a/src/store/resource/tests/foreground.rs b/src/store/resource/tests/foreground.rs deleted file mode 100644 index 62bd6fa..0000000 --- a/src/store/resource/tests/foreground.rs +++ /dev/null @@ -1,193 +0,0 @@ -//! Foreground ownership contract at request acceptance and task binding - -use std::os::unix::fs::symlink; - -use super::fixtures::{acceptance_input, resource, serving_fixture_with_spec}; -use crate::domain::{ProcessStatus, TaskId}; -use crate::machine::MachineId; -use crate::resource::ResourceTaskOwnershipRisk; -use crate::resource::foreground::test_support::native_fake_command; -use crate::resource::store::{PreLaunchFailure, ResourceStoreError, ResourceTaskAcceptance}; -use crate::spec::NormalizedSpec; -use crate::store::Store; -use crate::submission::RequestId; -use serde_json::json; -use std::fs; -use std::os::unix::fs::PermissionsExt; -use std::path::Path; -use tempfile::tempdir; - -fn command_in(root: &Path, argv: &[&Path]) -> NormalizedSpec { - serde_json::from_value(json!({ - "api_version": 1, - "thread": "01a0ab97-a7aa-7463-a5b0-8d500e40e431", - "name": "gpu command", - "cwd": root, - "timeout": "4h", - "workload": { "type": "task", "command": argv } - })) - .unwrap() -} - -fn write_script(path: &Path) { - fs::write(path, "#!/bin/sh\n/opt/gpu/bench &\n").unwrap(); - fs::set_permissions(path, fs::Permissions::from_mode(0o755)).unwrap(); -} - -#[test] -fn wrapped_detached_and_script_launches_never_enter_the_queue() { - let directory = tempdir().unwrap(); - let root = directory.path().canonicalize().unwrap(); - let mut store = Store::open(&root.join("db")).unwrap(); - let authority = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - let native = native_fake_command(); - let marker = root.join("marker"); - let script = root.join("prepared-command"); - write_script(&script); - // a renamed link to a shell is judged by its target - let renamed = root.join("bench"); - symlink("/bin/sh", &renamed).unwrap(); - - let path = Path::new; - let refused: [(&[&Path], ResourceTaskOwnershipRisk); 6] = [ - ( - &[ - path("/usr/bin/env"), - path("docker"), - path("run"), - path("-d"), - path("gpu"), - ], - ResourceTaskOwnershipRisk::ProgramLauncher, - ), - ( - &[path("sudo"), path("-n"), native], - ResourceTaskOwnershipRisk::ProgramLauncher, - ), - ( - &[path("timeout"), path("1h"), native], - ResourceTaskOwnershipRisk::ProgramLauncher, - ), - // an unrecognized launcher still exposes the nested container client - ( - &[native, path("docker"), path("run"), path("-d"), path("gpu")], - ResourceTaskOwnershipRisk::ContainerClient, - ), - ( - &[path("/bin/sh"), path("-c"), path("/opt/gpu/bench &")], - ResourceTaskOwnershipRisk::ShellWrapper, - ), - (&[&script], ResourceTaskOwnershipRisk::ScriptEntryPoint), - ]; - for (argv, expected) in refused { - let result = store.accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - command_in(&root, argv), - ); - assert!( - matches!( - result, - Err(ResourceStoreError::UnsupportedCommandOwnership { risk }) if risk == expected - ), - "{argv:?}: {result:?}" - ); - } - assert!(matches!( - store.accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - command_in(&root, &[&renamed]), - ), - Err(ResourceStoreError::UnsupportedCommandOwnership { .. }) - )); - assert!( - store - .resource_requests(authority, resource.id) - .unwrap() - .is_empty() - ); - - // a direct native command enters the queue, and its exact retry answers from - // the saved request even after the prepared file changed - let prepared = root.join("prepared-bench"); - symlink(native, &prepared).unwrap(); - let request_id = RequestId::new(); - let task_id = TaskId::new(); - let origin = MachineId::new(); - let spec = command_in(&root, &[&prepared, &marker]); - let accepted = store - .accept_resource_request( - authority, - request_id, - task_id, - resource.id, - origin, - spec.clone(), - ) - .unwrap(); - fs::remove_file(&prepared).unwrap(); - write_script(&prepared); - let retried = store - .accept_resource_request(authority, request_id, task_id, resource.id, origin, spec) - .unwrap(); - assert_eq!(retried.acceptance_sequence, accepted.acceptance_sequence); - assert_eq!( - store - .resource_requests(authority, resource.id) - .unwrap() - .len(), - 1 - ); -} - -#[test] -fn bare_program_that_resolves_to_a_script_is_accepted_only_to_fail_before_launch() { - let accept_on = |directory: &str, native: bool| { - let spec = command_in(Path::new("/tmp"), &[Path::new("prepared-command")]); - let mut fixture = serving_fixture_with_spec(true, true, spec); - let bin = fixture.directory.path().join(directory); - fs::create_dir(&bin).unwrap(); - if native { - symlink(native_fake_command(), bin.join("prepared-command")).unwrap(); - } else { - write_script(&bin.join("prepared-command")); - } - let mut input = acceptance_input(&fixture); - input.executor_env.path = bin.display().to_string(); - let acceptance = fixture.store.accept_assigned_resource_task(input).unwrap(); - (fixture, acceptance) - }; - - // the task gets its records so it can end with a reason, but it must never spawn - let (fixture, acceptance) = accept_on("scripts", false); - let task = fixture.request.task_id; - let ResourceTaskAcceptance::Unlaunchable { - task: accepted, - failure: PreLaunchFailure::Preparation { message }, - } = acceptance - else { - panic!("a script entry point must be accepted as unlaunchable, got {acceptance:?}"); - }; - assert_eq!(accepted, task); - assert!(message.contains("ScriptEntryPoint"), "{message}"); - let row = fixture.store.get_task(task).unwrap().unwrap(); - assert_eq!(row.status(), ProcessStatus::Queued); - - // the same bare name on a PATH with a native executable binds normally - let (fixture, acceptance) = accept_on("natives", true); - assert_eq!( - acceptance, - ResourceTaskAcceptance::Inserted { - task: fixture.request.task_id - } - ); -} diff --git a/src/store/resource/tests/initial_idle.rs b/src/store/resource/tests/initial_idle.rs deleted file mode 100644 index 4f6dfd7..0000000 --- a/src/store/resource/tests/initial_idle.rs +++ /dev/null @@ -1,211 +0,0 @@ -//! Initial idle attestation for a resource with no history - -use tempfile::tempdir; - -use super::fixtures::{resource, serving_fixture, spec}; -use crate::domain::TaskId; -use crate::machine::MachineId; -use crate::resource::initial_idle::{InitialIdleAttestation, InitialIdleRefusal, ResourceHistory}; -use crate::resource::operator_release::{ - OperatorAttestationId, OperatorGpuFreeConfirmation, OperatorObservation, -}; -use crate::resource::{ - IdleBoundaryProof, IdleProofGap, Resource, ResourceQueueAttentionReason, - ResourceQueueReconcileOutcome, ResourceRevision, -}; -use crate::store::{InitialIdleError, Store}; -use crate::submission::RequestId; - -struct FreshResource { - _directory: tempfile::TempDir, - store: Store, - authority: MachineId, - resource: Resource, -} - -fn fresh_resource() -> FreshResource { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - FreshResource { - _directory: directory, - store, - authority, - resource, - } -} - -fn attestation(fixture: &FreshResource, revision: ResourceRevision) -> InitialIdleAttestation { - InitialIdleAttestation { - operation_id: OperatorAttestationId::new(), - resource_id: fixture.resource.id, - authority_machine: fixture.authority, - expected_state_revision: revision, - observation: OperatorObservation::try_from( - "nvidia-smi on the authority shows no compute processes".to_owned(), - ) - .unwrap(), - confirmation: OperatorGpuFreeConfirmation::OperatorConfirmedGpuFree, - } -} - -fn refusal(result: Result) -> InitialIdleRefusal { - match result { - Err(InitialIdleError::Refused(refusal)) => refusal, - other => panic!("expected a typed refusal, got {other:?}"), - } -} - -fn reconcile(fixture: &mut FreshResource) -> ResourceQueueReconcileOutcome { - fixture - .store - .reconcile_resource_queue_for_authority(fixture.authority, fixture.resource.id) - .unwrap() -} - -#[test] -fn a_new_resource_serves_its_first_request_only_after_an_initial_idle_attestation() { - let mut fixture = fresh_resource(); - let request = fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - spec(), - ) - .unwrap(); - assert!(matches!( - reconcile(&mut fixture), - ResourceQueueReconcileOutcome::AttentionRequired { - reason: ResourceQueueAttentionReason::IdleNotProven { - gap: IdleProofGap::NoIdleEvidence - }, - .. - } - )); - - let saved = attestation(&fixture, fixture.resource.state_revision); - let resolution = fixture - .store - .attest_initial_idle_for_authority(fixture.authority, saved.clone()) - .unwrap(); - assert!(!resolution.replayed); - assert_eq!(resolution.receipt.attestation, saved); - assert_eq!( - resolution.receipt.state_revision, - ResourceRevision::new(fixture.resource.state_revision.get() + 1) - ); - - // an exact retry replays; changed content under the same id conflicts - let replay = fixture - .store - .attest_initial_idle_for_authority(fixture.authority, saved.clone()) - .unwrap(); - assert!(replay.replayed); - assert_eq!(replay.receipt, resolution.receipt); - let mut changed = saved.clone(); - changed.observation = OperatorObservation::try_from("changed".to_owned()).unwrap(); - assert!(matches!( - refusal( - fixture - .store - .attest_initial_idle_for_authority(fixture.authority, changed) - ), - InitialIdleRefusal::ConflictingRetry { .. } - )); - - let ResourceQueueReconcileOutcome::IdleServing { - request: served, - proof, - .. - } = reconcile(&mut fixture) - else { - panic!("the attested idle boundary must serve the next request in serving order"); - }; - assert_eq!(served.request_id, request.request_id); - assert_eq!( - proof, - IdleBoundaryProof::OperatorAttestedInitialIdle { - operation_id: saved.operation_id, - } - ); - - // the resource now has a loan, so a second initial attestation is refused - let current = fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap() - .remove(0) - .resource - .state_revision; - assert!(matches!( - refusal( - fixture - .store - .attest_initial_idle_for_authority(fixture.authority, attestation(&fixture, current)) - ), - InitialIdleRefusal::AlreadyAttested { operation_id } if operation_id == saved.operation_id - )); -} - -#[test] -fn an_initial_idle_attestation_is_refused_for_the_wrong_state() { - let mut fixture = fresh_resource(); - let stale = attestation( - &fixture, - ResourceRevision::new(fixture.resource.state_revision.get() + 1), - ); - assert!(matches!( - refusal( - fixture - .store - .attest_initial_idle_for_authority(fixture.authority, stale) - ), - InitialIdleRefusal::StaleRevision { .. } - )); - let other_authority = attestation(&fixture, fixture.resource.state_revision); - assert!(matches!( - refusal( - fixture - .store - .attest_initial_idle_for_authority(MachineId::new(), other_authority) - ), - InitialIdleRefusal::WrongAuthority { .. } - )); - - // a resource with history decides its idle state from that history - let mut serving = serving_fixture(true, true); - let history = InitialIdleAttestation { - operation_id: OperatorAttestationId::new(), - resource_id: serving.resource.id, - authority_machine: serving.authority, - expected_state_revision: serving.state_revision, - observation: OperatorObservation::try_from("checked".to_owned()).unwrap(), - confirmation: OperatorGpuFreeConfirmation::OperatorConfirmedGpuFree, - }; - assert!(matches!( - refusal( - serving - .store - .attest_initial_idle_for_authority(serving.authority, history) - ), - InitialIdleRefusal::HistoryExists { - history: ResourceHistory::RegisteredTask - } - )); - let count: i64 = serving - .store - .conn - .query_row( - "SELECT COUNT(*) FROM resource_initial_idle_attestations", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(count, 0, "a refusal writes nothing"); -} diff --git a/src/store/resource/tests/operator_release.rs b/src/store/resource/tests/operator_release.rs deleted file mode 100644 index ef1d588..0000000 --- a/src/store/resource/tests/operator_release.rs +++ /dev/null @@ -1,2036 +0,0 @@ -//! Operator attestation for a trainer that ended with no release proof -//! -//! The trainer is the registered trainer, a first background launch task, or a -//! Restoring loan's direct-segment return task that ended before registration -//! A Restoring loan's native foreground return task that ended or was lost -//! without release proof has its own binding -//! The automatic proof stays fail-closed for an unbound trainer. Only one exact, -//! explicit attestation releases it, and the attestation commits its receipt and -//! the queue or loan transition together. It never claims a confirmed exit, a -//! released trainer lock, a completed result, or a resumable stop - -use super::background::{LaunchFixture, registered_launch, saved_loan_for}; -use super::fixtures::{ServingFixture, TrainerAssociationFixture, resource, spec}; -use super::restore::{ - awaiting_return_fixture, evaluation, launch_input, resume_input, start_task, - stopped_return_fixture, -}; -use crate::daemon::actors::{StoreActor, StoreMsg, call}; -use crate::domain::{ExitReason, ProcessGroupExitEvidence, ProcessStatus, TaskId, ThreadId}; -use crate::machine::MachineId; -use crate::resource::operator_release::{ - AttestedTrainerAssociation, AttestedTrainerEnd, AttestedTrainerLaunch, OperatorAttestationId, - OperatorGpuFreeAttestation, OperatorGpuFreeConfirmation, OperatorGpuFreeEvidence, - OperatorGpuFreeOutcome, OperatorGpuFreeRefusal, OperatorGpuFreeResolution, OperatorObservation, - OperatorStateBinding, -}; -use crate::resource::store::ResourceTaskAcceptanceInput; -use crate::resource::{ - ActionId, IdleBoundaryDecision, IdleBoundaryProof, IdleProofGap, Loan, LoanClosure, LoanId, - LoanPhase, LoanState, Resource, ResourceId, ResourceQueueAttentionReason, - ResourceQueueReconcileOutcome, ResourceRequest, ResourceRequestState, ResourceRevision, - RestoreAttentionReason, ReturnContext, ServingReleaseProvenance, SupervisorActionAuthority, - SupervisorNoticePayload, -}; -use crate::store::{ - BackgroundLaunchAcceptance, BackgroundLaunchError, BackgroundLaunchPhase, - EndedRestoreResolution, OperatorGpuFreeError, RestoreReconcileOutcome, ReturnDecisionError, - ReturnTaskAcceptance, Store, -}; -use crate::submission::{RequestId, normalized_spec_sha256}; -use ractor::Actor; -use rusqlite::params; -use serde_json::json; -use tempfile::tempdir; -use uuid::Uuid; - -const OBSERVATION: &str = "nvidia-smi on the authority lists no trainer process"; - -pub(super) fn attestation_for( - store: &Store, - authority: MachineId, - resource_id: ResourceId, - task_id: TaskId, - state_binding: OperatorStateBinding, -) -> OperatorGpuFreeAttestation { - let revision = store - .resource_snapshots_for_authority(authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == resource_id) - .unwrap() - .resource - .state_revision; - OperatorGpuFreeAttestation { - operation_id: OperatorAttestationId::new(), - resource_id, - authority_machine: authority, - task_id, - expected_state_revision: revision, - state_binding, - observation: OperatorObservation::try_from(OBSERVATION.to_owned()).unwrap(), - confirmation: OperatorGpuFreeConfirmation::OperatorConfirmedGpuFree, - } -} - -fn attestation( - fixture: &LaunchFixture, - task_id: TaskId, - state_binding: OperatorStateBinding, -) -> OperatorGpuFreeAttestation { - attestation_for( - &fixture.store, - fixture.authority, - fixture.resource.id, - task_id, - state_binding, - ) -} - -fn attest( - fixture: &mut LaunchFixture, - attestation: OperatorGpuFreeAttestation, -) -> Result { - fixture - .store - .attest_trainer_gpu_free_for_authority(fixture.authority, attestation) -} - -/// Build a fresh attestation for the current revision and commit it -fn attest_binding( - fixture: &mut LaunchFixture, - task_id: TaskId, - state_binding: OperatorStateBinding, -) -> Result { - let attestation = attestation(fixture, task_id, state_binding); - attest(fixture, attestation) -} - -fn refusal( - result: Result, -) -> OperatorGpuFreeRefusal { - match result { - Err(OperatorGpuFreeError::Refused(refusal)) => refusal, - other => panic!("expected a typed refusal, got {other:?}"), - } -} - -fn attestation_count(store: &Store) -> i64 { - store - .conn - .query_row( - "SELECT COUNT(*) FROM resource_operator_attestations", - [], - |row| row.get(0), - ) - .unwrap() -} - -fn fail(fixture: &LaunchFixture, task_id: TaskId, code: i32, evidence: ProcessGroupExitEvidence) { - fixture - .store - .cas_exit_with_evidence( - task_id, - ProcessStatus::Running, - &ExitReason::Exit { code }, - evidence, - ) - .unwrap() - .unwrap(); -} - -fn lose(fixture: &LaunchFixture, task_id: TaskId) { - fixture - .store - .cas_status(task_id, ProcessStatus::Running, ProcessStatus::Lost) - .unwrap() - .unwrap(); -} - -/// Check the activation gate that an assigned command task must pass -fn activation_accepts(fixture: &LaunchFixture, loan: &Loan, request: &ResourceRequest) -> bool { - let input = ResourceTaskAcceptanceInput { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - request_id: request.request_id, - task_id: request.task_id, - acceptance_sequence: request.acceptance_sequence, - loan_id: loan.id, - expected_state_revision: fixture.saved_resource().state_revision, - command_spec: request.spec().clone(), - executor_env: fixture.trainer.env.clone(), - }; - crate::resource::store::assigned_resource_request_for_acceptance(&fixture.store.conn, &input) - .is_ok() -} - -/// Open the release action for a running registered trainer and return its identities -fn awaiting_release(fixture: &mut LaunchFixture) -> (Loan, ActionId, ResourceRevision) { - let ResourceQueueReconcileOutcome::ReleaseRequired { loan, notice } = fixture.reconcile() - else { - panic!("a running registered trainer must open a release action for queued work"); - }; - let LoanState::Active { - phase: LoanPhase::AwaitingRelease { action_id, .. }, - } = loan.state - else { - panic!("the new loan must await release"); - }; - (loan, action_id, notice.state_revision) -} - -#[test] -fn unbound_failed_trainer_stays_reserved_until_the_attestation_serves_the_queue() { - let mut fixture = LaunchFixture::new(); - let launch_request = RequestId::new(); - let task = registered_launch(&mut fixture, launch_request); - fail(&fixture, task, 3, ProcessGroupExitEvidence::Unconfirmed); - let request = fixture.queue_request(); - - // no association names a lock, so no release action can open and the queue waits - for _ in 0..2 { - assert!(matches!( - fixture.reconcile(), - ResourceQueueReconcileOutcome::AttentionRequired { - reason: ResourceQueueAttentionReason::BackgroundTaskNotRunning { task_id, .. }, - .. - } if task_id == task - )); - } - assert!(saved_loan_for(&fixture).is_none()); - let before = fixture.saved_resource(); - - let resolution = attest_binding(&mut fixture, task, OperatorStateBinding::NoLoan).unwrap(); - assert!(!resolution.replayed); - let receipt = resolution.receipt; - // the snapshot keeps the unconfirmed wrapper exit; the attestation proves nothing more - assert_eq!( - receipt.evidence, - OperatorGpuFreeEvidence { - trainer_end: AttestedTrainerEnd::Finished { - outcome: ExitReason::Exit { code: 3 }, - process_group_exit: ProcessGroupExitEvidence::Unconfirmed, - container_exit: None, - }, - trainer_launch: AttestedTrainerLaunch::FirstBackgroundLaunch { - request_id: launch_request, - }, - normalized_spec_sha256: normalized_spec_sha256(&fixture.trainer_spec()).unwrap(), - trainer_association: AttestedTrainerAssociation::Missing, - } - ); - - // the registration clears and the next request in serving order serves in the same transaction - let OperatorGpuFreeOutcome::IdleServing { - loan, - request: selected, - } = &receipt.outcome - else { - panic!("queued work must be selected with the attestation"); - }; - assert_eq!(selected.request_id, request.request_id); - assert_eq!( - loan.state, - LoanState::Active { - phase: LoanPhase::Serving { - return_context: ReturnContext::Idle, - current_request_id: request.request_id, - release_provenance: ServingReleaseProvenance::IdleBoundary { - proof: IdleBoundaryProof::OperatorAttestedGpuFree { - operation_id: receipt.attestation.operation_id, - task_id: task, - }, - }, - }, - } - ); - let saved = fixture.saved_resource(); - assert_eq!(saved.registered_background_task, None); - assert_eq!(saved.state_revision.get(), before.state_revision.get() + 2); - assert_eq!(receipt.state_revision, saved.state_revision); - assert_eq!(saved_loan_for(&fixture), Some(loan.clone())); - assert!(activation_accepts(&fixture, loan, selected)); -} - -#[test] -fn running_trainer_wrong_identities_and_stale_state_are_refused_without_records() { - let mut fixture = LaunchFixture::new(); - let task = registered_launch(&mut fixture, RequestId::new()); - - // a running wrapper still owns the GPU - assert_eq!( - refusal(attest_binding( - &mut fixture, - task, - OperatorStateBinding::NoLoan - )), - OperatorGpuFreeRefusal::TaskNotEnded { - task_id: task, - state: ProcessStatus::Running, - } - ); - - lose(&fixture, task); - let before = fixture.saved_resource(); - let other_machine = MachineId::new(); - let other_task = TaskId::new(); - type Change = fn(&mut OperatorGpuFreeAttestation, MachineId, TaskId); - let cases: [(&str, Change, MachineId, OperatorGpuFreeRefusal); 6] = [ - ( - "unregistered task", - |attestation, _, task| attestation.task_id = task, - fixture.authority, - OperatorGpuFreeRefusal::NotRegisteredTrainer { - task_id: other_task, - registered: Some(task), - }, - ), - ( - "stale revision", - |attestation, _, _| { - attestation.expected_state_revision = - ResourceRevision::new(attestation.expected_state_revision.get() - 1); - }, - fixture.authority, - OperatorGpuFreeRefusal::StaleRevision { - expected: ResourceRevision::new(before.state_revision.get() - 1), - actual: before.state_revision, - }, - ), - ( - "attestation names another authority", - |attestation, machine, _| attestation.authority_machine = machine, - fixture.authority, - OperatorGpuFreeRefusal::WrongAuthority { - expected: fixture.authority, - found: other_machine, - }, - ), - ( - "another daemon receives it", - |attestation, machine, _| attestation.authority_machine = machine, - other_machine, - OperatorGpuFreeRefusal::WrongAuthority { - expected: fixture.authority, - found: other_machine, - }, - ), - ( - "binding names a release action that does not exist", - |attestation, _, _| { - attestation.state_binding = OperatorStateBinding::AwaitingRelease { - loan_id: LoanId::new(), - action_id: ActionId::new(), - }; - }, - fixture.authority, - OperatorGpuFreeRefusal::LoanStateChanged { current_loan: None }, - ), - ( - "unknown resource", - |attestation, _, _| attestation.resource_id = ResourceId::new(), - fixture.authority, - OperatorGpuFreeRefusal::ResourceNotFound, - ), - ]; - for (name, change, daemon, expected) in cases { - let mut changed = attestation(&fixture, task, OperatorStateBinding::NoLoan); - change(&mut changed, other_machine, other_task); - let result = fixture - .store - .attest_trainer_gpu_free_for_authority(daemon, changed); - assert_eq!(refusal(result), expected, "{name}"); - assert_eq!(attestation_count(&fixture.store), 0, "{name}"); - assert_eq!(fixture.saved_resource(), before, "{name}"); - } - - // the observation and the explicit confirmation are required at the boundary - assert_eq!( - OperatorObservation::try_from(" ".to_owned()), - Err(OperatorGpuFreeRefusal::EmptyObservation) - ); - let mut encoded = - serde_json::to_value(attestation(&fixture, task, OperatorStateBinding::NoLoan)).unwrap(); - encoded["observation"] = json!(" "); - assert!(serde_json::from_value::(encoded.clone()).is_err()); - encoded["observation"] = json!(OBSERVATION); - encoded.as_object_mut().unwrap().remove("confirmation"); - assert!(serde_json::from_value::(encoded).is_err()); -} - -#[test] -fn missing_or_changed_launch_records_are_refused() { - let change_thread = |fixture: &LaunchFixture, task: TaskId| { - fixture - .store - .conn - .execute( - "UPDATE tasks SET thread_id = ?1 WHERE id = ?2", - params![ThreadId(Uuid::now_v7()).to_string(), task.to_string()], - ) - .unwrap(); - }; - let drop_receipt = |fixture: &LaunchFixture, task: TaskId| { - fixture - .store - .conn - .execute( - "DELETE FROM resource_background_launches WHERE task_id = ?1", - [task.to_string()], - ) - .unwrap(); - }; - for (name, change) in [ - ( - "no launch receipt", - &drop_receipt as &dyn Fn(&LaunchFixture, TaskId), - ), - ("task row no longer matches its identity", &change_thread), - ] { - let mut fixture = LaunchFixture::new(); - let task = registered_launch(&mut fixture, RequestId::new()); - fail(&fixture, task, 1, ProcessGroupExitEvidence::ConfirmedExited); - change(&fixture, task); - let before = fixture.saved_resource(); - - assert_eq!( - refusal(attest_binding( - &mut fixture, - task, - OperatorStateBinding::NoLoan - )), - OperatorGpuFreeRefusal::TrainerLaunchUnproven { task_id: task }, - "{name}" - ); - assert_eq!(attestation_count(&fixture.store), 0, "{name}"); - assert_eq!(fixture.saved_resource(), before, "{name}"); - } -} - -#[test] -fn lost_trainer_release_keeps_the_return_obligation_for_the_supervisor() { - let mut fixture = LaunchFixture::new(); - let task = registered_launch(&mut fixture, RequestId::new()); - let request = fixture.queue_request(); - let (loan, action_id, revision) = awaiting_release(&mut fixture); - fixture - .store - .cancel_resource_request_before_activation( - fixture.authority, - request.request_id, - request.task_id, - fixture.resource.id, - request.origin_machine, - ) - .unwrap(); - lose(&fixture, task); - - // the automatic proof stays fail-closed for a lost, unbound trainer - assert!( - fixture - .store - .complete_release_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision - ) - .is_err() - ); - assert_eq!(saved_loan_for(&fixture), Some(loan.clone())); - - // the binding must name the exact release action - for binding in [ - OperatorStateBinding::NoLoan, - OperatorStateBinding::AwaitingRelease { - loan_id: loan.id, - action_id: ActionId::new(), - }, - ] { - assert_eq!( - refusal(attest_binding(&mut fixture, task, binding)), - OperatorGpuFreeRefusal::LoanStateChanged { - current_loan: Some(loan.id) - } - ); - } - assert_eq!(attestation_count(&fixture.store), 0); - - let binding = OperatorStateBinding::AwaitingRelease { - loan_id: loan.id, - action_id, - }; - let receipt = attest_binding(&mut fixture, task, binding).unwrap().receipt; - assert_eq!(receipt.evidence.trainer_end, AttestedTrainerEnd::Lost); - let lost = ReturnContext::LostWithoutResult { task_id: task }; - let OperatorGpuFreeOutcome::ReleaseResolvedReturnRequired { - loan: returned, - notice, - } = &receipt.outcome - else { - panic!("an empty queue must reserve the supervisor return decision"); - }; - assert_eq!( - returned.state, - LoanState::Active { - phase: LoanPhase::AwaitingReturn { - action_id: notice.action_id, - return_context: lost.clone(), - }, - } - ); - assert_eq!( - notice.payload, - SupervisorNoticePayload::ReturnRequired { - return_context: lost.clone() - } - ); - assert_eq!(notice.state_revision, receipt.state_revision); - // the trainer stays registered until the supervisor decides the return - assert_eq!( - fixture.saved_resource().registered_background_task, - Some(task) - ); - - // the supervisor can close the lost run without resuming it - let authority = SupervisorActionAuthority { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - loan_id: returned.id, - action_id: notice.action_id, - expected_state_revision: notice.state_revision, - supervisor: fixture.resource.supervisor, - assignment_revision: fixture.resource.assignment_revision, - }; - let closure = fixture - .store - .record_no_resume_for_authority(authority, "operator resolved the lost run".into()) - .unwrap(); - assert!(matches!( - closure.loan.state, - LoanState::Closed { - result: LoanClosure::NoResume { return_context, .. } - } if return_context == lost - )); - assert_eq!(fixture.saved_resource().registered_background_task, None); - fixture.launch(RequestId::new()); -} - -#[test] -fn failed_trainer_release_serves_requests_in_order_and_replays_after_restart() { - let mut fixture = LaunchFixture::new(); - let task = registered_launch(&mut fixture, RequestId::new()); - let first = fixture.queue_request(); - let second = fixture.queue_request(); - let (loan, action_id, revision) = awaiting_release(&mut fixture); - // a confirmed wrapper exit does not cover the detached worker without an association - fail(&fixture, task, 3, ProcessGroupExitEvidence::ConfirmedExited); - assert!( - fixture - .store - .complete_release_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision - ) - .is_err() - ); - - let binding = OperatorStateBinding::AwaitingRelease { - loan_id: loan.id, - action_id, - }; - let saved_attestation = attestation(&fixture, task, binding); - let receipt = attest(&mut fixture, saved_attestation.clone()) - .unwrap() - .receipt; - let OperatorGpuFreeOutcome::ReleaseResolvedServing { - loan: serving, - request: selected, - } = &receipt.outcome - else { - panic!("the next queued request must serve after the attestation"); - }; - assert_eq!(selected.request_id, first.request_id); - // an ended run without a result is never a completed or resumable context - assert_eq!( - serving.state, - LoanState::Active { - phase: LoanPhase::Serving { - return_context: ReturnContext::EndedWithoutResult { - task_id: task, - outcome: ExitReason::Exit { code: 3 }, - }, - current_request_id: first.request_id, - release_provenance: ServingReleaseProvenance::OperatorAttestedGpuFree { - operation_id: saved_attestation.operation_id, - action_id, - task_id: task, - }, - }, - } - ); - assert_eq!( - fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|request| request.request_id == second.request_id) - .unwrap() - .state, - ResourceRequestState::Queued - ); - assert_eq!( - fixture.saved_resource().registered_background_task, - Some(task) - ); - assert!(activation_accepts(&fixture, serving, selected)); - - // a Serving loan can already run other GPU work, so it is never overridden - assert_eq!( - refusal(attest_binding(&mut fixture, task, binding)), - OperatorGpuFreeRefusal::LoanNotAwaitingRelease { loan_id: loan.id } - ); - - fixture.reopen(); - let replay = attest(&mut fixture, saved_attestation.clone()).unwrap(); - assert!(replay.replayed); - assert_eq!( - serde_json::to_value(&replay.receipt).unwrap(), - serde_json::to_value(&receipt).unwrap() - ); - let mut changed = saved_attestation.clone(); - changed.observation = OperatorObservation::try_from("a different account".to_owned()).unwrap(); - assert_eq!( - refusal(attest(&mut fixture, changed)), - OperatorGpuFreeRefusal::ConflictingRetry { - operation_id: saved_attestation.operation_id - } - ); - assert_eq!(attestation_count(&fixture.store), 1); - assert!( - fixture - .store - .operator_attestation_receipt_for_authority( - fixture.authority, - saved_attestation.operation_id - ) - .unwrap() - .is_some() - ); -} - -#[test] -fn idle_boundary_survives_restart_and_serves_a_later_request() { - let mut fixture = LaunchFixture::new(); - let task = registered_launch(&mut fixture, RequestId::new()); - lose(&fixture, task); - let before = fixture.saved_resource(); - - let saved_attestation = attestation(&fixture, task, OperatorStateBinding::NoLoan); - let receipt = attest(&mut fixture, saved_attestation.clone()) - .unwrap() - .receipt; - assert!(matches!( - receipt.outcome, - OperatorGpuFreeOutcome::IdleBoundary - )); - let saved = fixture.saved_resource(); - assert_eq!(saved.registered_background_task, None); - assert_eq!(saved.state_revision.get(), before.state_revision.get() + 1); - - fixture.reopen(); - assert!( - attest(&mut fixture, saved_attestation.clone()) - .unwrap() - .replayed - ); - let request = fixture.queue_request(); - assert!(matches!( - fixture.reconcile(), - ResourceQueueReconcileOutcome::IdleServing { - request: served, - proof: IdleBoundaryProof::OperatorAttestedGpuFree { operation_id, task_id }, - .. - } if served.request_id == request.request_id - && operation_id == saved_attestation.operation_id - && task_id == task - )); -} - -#[test] -fn first_background_launch_proceeds_only_through_the_saved_boundary() { - let mut fixture = LaunchFixture::new(); - let task = registered_launch(&mut fixture, RequestId::new()); - fail(&fixture, task, 1, ProcessGroupExitEvidence::Unconfirmed); - - let blocked = fixture.input(RequestId::new(), fixture.trainer_spec()); - assert!(matches!( - fixture.store.accept_background_launch_for_authority(blocked), - Err(BackgroundLaunchError::PredecessorReleaseUnproven { task_id }) if task_id == task - )); - - attest_binding(&mut fixture, task, OperatorStateBinding::NoLoan).unwrap(); - let next = fixture.input(RequestId::new(), fixture.trainer_spec()); - let next_task = next.task_id; - assert!(matches!( - fixture.store.accept_background_launch_for_authority(next), - Ok(BackgroundLaunchAcceptance::Inserted { task, .. }) if task == next_task - )); - // the new launch supersedes the boundary, so queued work waits for its start - let request = fixture.queue_request(); - assert!(matches!( - fixture.reconcile(), - ResourceQueueReconcileOutcome::AttentionRequired { - request: waiting, - reason: ResourceQueueAttentionReason::BackgroundLaunchPending { task_id }, - } if waiting.request_id == request.request_id && task_id == next_task - )); -} - -#[test] -fn direct_segment_return_trainer_that_failed_before_binding_can_be_attested() { - let (mut fixture, authority, decision) = stopped_return_fixture(); - let input = resume_input( - authority, - fixture.task_id, - &decision.selected_checkpoint.generation_id, - ); - let (request_id, task_id) = (input.launch.request_id, input.launch.task_id); - assert!(matches!( - fixture - .store - .accept_return_task_for_authority(input) - .unwrap(), - ReturnTaskAcceptance::Inserted { .. } - )); - start_task(&fixture.store, task_id); - assert!(matches!( - fixture - .store - .reconcile_restoring_loan_for_authority(fixture.authority, fixture.resource.id) - .unwrap(), - RestoreReconcileOutcome::Closed { .. } - )); - fixture - .store - .cas_exit_with_evidence( - task_id, - ProcessStatus::Running, - &ExitReason::Signal { signal: 9 }, - ProcessGroupExitEvidence::Unconfirmed, - ) - .unwrap() - .unwrap(); - - let attestation = attestation_for( - &fixture.store, - fixture.authority, - fixture.resource.id, - task_id, - OperatorStateBinding::NoLoan, - ); - let receipt = fixture - .store - .attest_trainer_gpu_free_for_authority(fixture.authority, attestation) - .unwrap() - .receipt; - assert_eq!( - receipt.evidence.trainer_launch, - AttestedTrainerLaunch::DirectSegmentReturn { - action_id: authority.action_id, - request_id, - } - ); - assert!(matches!( - receipt.outcome, - OperatorGpuFreeOutcome::IdleBoundary - )); -} - -#[test] -fn attestation_table_checks_the_outcome_and_the_explicit_confirmation() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - - let insert = |outcome: &str, confirmation: serde_json::Value| { - let (operation, task) = (Uuid::now_v7(), TaskId::new()); - let receipt = json!({ - "attestation": { - "operation_id": operation, - "resource_id": resource.id, - "task_id": task, - "observation": OBSERVATION, - "confirmation": confirmation, - }, - "evidence": {}, - "outcome": { "type": outcome }, - }); - store.conn.execute( - "INSERT INTO resource_operator_attestations - (operation_id, resource_id, task_id, receipt_json) - VALUES (?1, ?2, ?3, ?4)", - params![ - operation.to_string(), - resource.id.as_uuid().to_string(), - task.to_string(), - receipt.to_string() - ], - ) - }; - let confirmed = || json!("operator_confirmed_gpu_free"); - for outcome in [ - "release_resolved_serving", - "release_resolved_return_required", - "idle_serving", - "idle_boundary", - "restore_closed_serving", - "restore_closed_idle_boundary", - ] { - assert!(insert(outcome, confirmed()).is_ok(), "{outcome}"); - } - assert!(insert("unknown", confirmed()).is_err()); - // a receipt without the explicit confirmation cannot be stored - assert!(insert("idle_boundary", json!(null)).is_err()); - assert!(insert("idle_boundary", json!("confirmed")).is_err()); -} - -#[tokio::test] -async fn store_actor_commits_and_replays_one_attestation() { - let mut fixture = LaunchFixture::new(); - let task = registered_launch(&mut fixture, RequestId::new()); - lose(&fixture, task); - let saved_attestation = attestation(&fixture, task, OperatorStateBinding::NoLoan); - let authority = fixture.authority; - - let (store, handle) = StoreActor::spawn(None, StoreActor, fixture.db_path()) - .await - .unwrap(); - let attest = |attestation: OperatorGpuFreeAttestation| { - call(&store, move |reply| { - StoreMsg::AttestTrainerGpuFreeForAuthority { - authority_machine: authority, - attestation: Box::new(attestation), - reply, - } - }) - }; - let first = attest(saved_attestation.clone()).await.unwrap().unwrap(); - assert!(!first.replayed); - let replay = attest(saved_attestation.clone()).await.unwrap().unwrap(); - assert!(replay.replayed); - let saved = fixture - .store - .operator_attestation_receipt_for_authority(authority, saved_attestation.operation_id) - .unwrap() - .unwrap(); - assert_eq!(saved.attestation, saved_attestation); - store.stop(None); - let _ = handle.await; -} - -/// Start a first background launch and end it before the owner registers its start -fn launch_ended_before_registration( - fixture: &mut LaunchFixture, - request_id: RequestId, - evidence: ProcessGroupExitEvidence, -) -> TaskId { - let task = fixture.launch(request_id); - start_task(&fixture.store, task); - fail(fixture, task, 1, evidence); - task -} - -fn resource_count_state(fixture: &LaunchFixture) -> (Resource, Option, i64) { - ( - fixture.saved_resource(), - saved_loan_for(fixture), - attestation_count(&fixture.store), - ) -} - -#[test] -fn first_launch_that_ended_before_registration_releases_only_through_its_exact_attestation() { - let mut fixture = LaunchFixture::new(); - let launch_request = RequestId::new(); - let task = launch_ended_before_registration( - &mut fixture, - launch_request, - ProcessGroupExitEvidence::Unconfirmed, - ); - // the ended task is not registered, and neither the slot nor the queue is free - assert_eq!(fixture.saved_resource().registered_background_task, None); - let blocked = fixture.input(RequestId::new(), fixture.trainer_spec()); - assert!(matches!( - fixture.store.accept_background_launch_for_authority(blocked), - Err(BackgroundLaunchError::PredecessorReleaseUnproven { task_id }) if task_id == task - )); - let before = resource_count_state(&fixture); - - let binding = OperatorStateBinding::FirstBackgroundLaunch { - request_id: launch_request, - }; - let other_task = TaskId::new(); - let other_request = RequestId::new(); - let cases = [ - ( - "the registered-trainer binding names no registration", - attestation(&fixture, task, OperatorStateBinding::NoLoan), - OperatorGpuFreeRefusal::NotRegisteredTrainer { - task_id: task, - registered: None, - }, - ), - ( - "another launch request", - attestation( - &fixture, - task, - OperatorStateBinding::FirstBackgroundLaunch { - request_id: other_request, - }, - ), - OperatorGpuFreeRefusal::LaunchNotAwaitingRelease { - request_id: other_request, - current_launch: Some(launch_request), - }, - ), - ( - "another task", - attestation(&fixture, other_task, binding), - OperatorGpuFreeRefusal::NotBoundTask { - task_id: other_task, - bound: task, - }, - ), - ( - "a restore binding with no loan", - attestation( - &fixture, - task, - OperatorStateBinding::RestoringReturn { - loan_id: LoanId::new(), - action_id: ActionId::new(), - }, - ), - OperatorGpuFreeRefusal::LoanStateChanged { current_loan: None }, - ), - ]; - for (name, changed, expected) in cases { - assert_eq!(refusal(attest(&mut fixture, changed)), expected, "{name}"); - assert_eq!(resource_count_state(&fixture), before, "{name}"); - } - let other_authority = MachineId::new(); - let saved_attestation = attestation(&fixture, task, binding); - assert_eq!( - refusal( - fixture - .store - .attest_trainer_gpu_free_for_authority(other_authority, saved_attestation.clone()) - ), - OperatorGpuFreeRefusal::WrongAuthority { - expected: other_authority, - found: fixture.authority, - } - ); - assert_eq!(resource_count_state(&fixture), before); - - let receipt = attest(&mut fixture, saved_attestation.clone()) - .unwrap() - .receipt; - assert_eq!( - receipt.evidence, - OperatorGpuFreeEvidence { - trainer_end: AttestedTrainerEnd::Finished { - outcome: ExitReason::Exit { code: 1 }, - process_group_exit: ProcessGroupExitEvidence::Unconfirmed, - container_exit: None, - }, - trainer_launch: AttestedTrainerLaunch::FirstBackgroundLaunch { - request_id: launch_request, - }, - normalized_spec_sha256: normalized_spec_sha256(&fixture.trainer_spec()).unwrap(), - trainer_association: AttestedTrainerAssociation::Missing, - } - ); - assert!(matches!( - receipt.outcome, - OperatorGpuFreeOutcome::IdleBoundary - )); - let saved = fixture.saved_resource(); - assert_eq!(saved.registered_background_task, None); - assert_eq!( - saved.state_revision.get(), - before.0.state_revision.get() + 1 - ); - assert_eq!( - fixture - .store - .background_launch_for_authority(fixture.authority, fixture.resource.id) - .unwrap() - .unwrap() - .phase, - BackgroundLaunchPhase::Superseded - ); - - // the exact retry replays after a restart even though the revision moved - fixture.reopen(); - let replay = attest(&mut fixture, saved_attestation.clone()).unwrap(); - assert!(replay.replayed); - assert_eq!( - serde_json::to_value(&replay.receipt).unwrap(), - serde_json::to_value(&receipt).unwrap() - ); - let mut changed = saved_attestation.clone(); - changed.observation = OperatorObservation::try_from("another account".to_owned()).unwrap(); - assert_eq!( - refusal(attest(&mut fixture, changed)), - OperatorGpuFreeRefusal::ConflictingRetry { - operation_id: saved_attestation.operation_id - } - ); - - // only the saved boundary lets the queue and the next launch proceed - let request = fixture.queue_request(); - assert!(matches!( - fixture.reconcile(), - ResourceQueueReconcileOutcome::IdleServing { - request: served, - proof: IdleBoundaryProof::OperatorAttestedGpuFree { operation_id, task_id }, - .. - } if served.request_id == request.request_id - && operation_id == saved_attestation.operation_id - && task_id == task - )); -} - -#[test] -fn queued_running_or_never_spawned_first_launch_cannot_be_attested() { - let mut fixture = LaunchFixture::new(); - let launch_request = RequestId::new(); - let task = fixture.launch(launch_request); - let binding = OperatorStateBinding::FirstBackgroundLaunch { - request_id: launch_request, - }; - let before = resource_count_state(&fixture); - assert_eq!( - refusal(attest_binding(&mut fixture, task, binding)), - OperatorGpuFreeRefusal::TaskNotEnded { - task_id: task, - state: ProcessStatus::Queued, - } - ); - // a started launch that the owner has not registered yet is still live work - start_task(&fixture.store, task); - assert_eq!( - refusal(attest_binding(&mut fixture, task, binding)), - OperatorGpuFreeRefusal::TaskNotEnded { - task_id: task, - state: ProcessStatus::Running, - } - ); - assert_eq!(resource_count_state(&fixture), before); - - // a launch with automatic no-child proof needs no attestation - let mut fixture = LaunchFixture::new(); - let never = fixture.launch(launch_request); - fixture - .store - .cas_exit(never, ProcessStatus::Queued, &ExitReason::Cancelled) - .unwrap() - .unwrap(); - assert_eq!( - refusal(attest_binding(&mut fixture, never, binding)), - OperatorGpuFreeRefusal::LaunchNotAwaitingRelease { - request_id: launch_request, - current_launch: Some(launch_request), - } - ); - assert_eq!(attestation_count(&fixture.store), 0); -} - -#[test] -fn lost_first_launch_attestation_serves_the_queue_in_order() { - let mut fixture = LaunchFixture::new(); - let launch_request = RequestId::new(); - let task = fixture.launch(launch_request); - start_task(&fixture.store, task); - lose(&fixture, task); - let first = fixture.queue_request(); - let second = fixture.queue_request(); - for _ in 0..2 { - assert!(matches!( - fixture.reconcile(), - ResourceQueueReconcileOutcome::AttentionRequired { - reason: ResourceQueueAttentionReason::IdleNotProven { - gap: IdleProofGap::BackgroundLaunchReleaseUnproven { task_id }, - }, - .. - } if task_id == task - )); - } - assert!(saved_loan_for(&fixture).is_none()); - - let receipt = attest_binding( - &mut fixture, - task, - OperatorStateBinding::FirstBackgroundLaunch { - request_id: launch_request, - }, - ) - .unwrap() - .receipt; - assert_eq!(receipt.evidence.trainer_end, AttestedTrainerEnd::Lost); - let OperatorGpuFreeOutcome::IdleServing { loan, request } = &receipt.outcome else { - panic!("the next request in serving order must serve with the attestation"); - }; - assert_eq!(request.request_id, first.request_id); - assert!(matches!( - &loan.state, - LoanState::Active { - phase: LoanPhase::Serving { - release_provenance: ServingReleaseProvenance::IdleBoundary { - proof: IdleBoundaryProof::OperatorAttestedGpuFree { operation_id, task_id }, - }, - .. - }, - } if *operation_id == receipt.attestation.operation_id && *task_id == task - )); - assert_eq!(saved_loan_for(&fixture).as_ref(), Some(loan)); - assert!(activation_accepts(&fixture, loan, request)); - assert_eq!( - fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|saved| saved.request_id == second.request_id) - .unwrap() - .state, - ResourceRequestState::Queued - ); -} - -/// Bind a same-run resume and end its task before the owner observes its start -fn restore_ended_before_confirmed_start() -> (TrainerAssociationFixture, Loan, ActionId, TaskId) { - let (mut fixture, authority, decision) = stopped_return_fixture(); - let input = resume_input( - authority, - fixture.task_id, - &decision.selected_checkpoint.generation_id, - ); - let task_id = input.launch.task_id; - let ReturnTaskAcceptance::Inserted { loan, .. } = fixture - .store - .accept_return_task_for_authority(input) - .unwrap() - else { - panic!("the first exact return launch must insert its task"); - }; - start_task(&fixture.store, task_id); - fixture - .store - .cas_exit_with_evidence( - task_id, - ProcessStatus::Running, - &ExitReason::Signal { signal: 9 }, - ProcessGroupExitEvidence::Unconfirmed, - ) - .unwrap() - .unwrap(); - assert!(matches!( - fixture - .store - .reconcile_restoring_loan_for_authority(fixture.authority, fixture.resource.id) - .unwrap(), - RestoreReconcileOutcome::Attention { task_id: attention, .. } if attention == task_id - )); - // the unconfirmed wrapper exit gives the supervisor resolution no release proof - assert!(matches!( - fixture - .store - .resolve_ended_restore_for_authority(EndedRestoreResolution { - authority: SupervisorActionAuthority { - loan_id: loan.id, - expected_state_revision: fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap()[0] - .resource - .state_revision, - ..authority - }, - task_id, - reason: "ended before its start".into(), - }), - Err(ReturnDecisionError::RestoreReleaseUnproven { .. }) - )); - (fixture, *loan, authority.action_id, task_id) -} - -fn restore_state(fixture: &TrainerAssociationFixture) -> (Resource, Option, i64) { - let snapshot = fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap() - .remove(0); - ( - snapshot.resource, - snapshot.loan, - attestation_count(&fixture.store), - ) -} - -#[test] -fn restoring_return_that_ended_before_its_start_closes_through_its_exact_attestation() { - let (mut fixture, loan, action_id, task_id) = restore_ended_before_confirmed_start(); - let (authority, resource_id) = (fixture.authority, fixture.resource.id); - let attest_on = |store: &mut Store, attestation| { - store.attest_trainer_gpu_free_for_authority(authority, attestation) - }; - let binding = OperatorStateBinding::RestoringReturn { - loan_id: loan.id, - action_id, - }; - let before = restore_state(&fixture); - let other_task = TaskId::new(); - let cases = [ - ( - "the unregistered return task has no registered-trainer binding", - OperatorStateBinding::NoLoan, - task_id, - OperatorGpuFreeRefusal::NotRegisteredTrainer { - task_id, - registered: before.0.registered_background_task, - }, - ), - ( - "another return action", - OperatorStateBinding::RestoringReturn { - loan_id: loan.id, - action_id: ActionId::new(), - }, - task_id, - OperatorGpuFreeRefusal::LoanStateChanged { - current_loan: Some(loan.id), - }, - ), - ( - "another loan", - OperatorStateBinding::RestoringReturn { - loan_id: LoanId::new(), - action_id, - }, - task_id, - OperatorGpuFreeRefusal::LoanStateChanged { - current_loan: Some(loan.id), - }, - ), - ( - "another task", - binding, - other_task, - OperatorGpuFreeRefusal::NotBoundTask { - task_id: other_task, - bound: task_id, - }, - ), - ( - "a first launch binding while a loan reserves the resource", - OperatorStateBinding::FirstBackgroundLaunch { - request_id: RequestId::new(), - }, - task_id, - OperatorGpuFreeRefusal::LoanStateChanged { - current_loan: Some(loan.id), - }, - ), - ]; - for (name, state_binding, task, expected) in cases { - let changed = attestation_for(&fixture.store, authority, resource_id, task, state_binding); - assert_eq!( - refusal(attest_on(&mut fixture.store, changed)), - expected, - "{name}" - ); - assert_eq!(restore_state(&fixture), before, "{name}"); - } - - // queued work waits behind the reservation and serves with the closure - let first = fixture - .store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource_id, - MachineId::new(), - spec(), - ) - .unwrap(); - let saved_attestation = - attestation_for(&fixture.store, authority, resource_id, task_id, binding); - let receipt = attest_on(&mut fixture.store, saved_attestation.clone()) - .unwrap() - .receipt; - let AttestedTrainerLaunch::DirectSegmentReturn { - action_id: bound_action, - .. - } = receipt.evidence.trainer_launch - else { - panic!("the evidence must name the return decision that bound the task"); - }; - assert_eq!(bound_action, action_id); - assert_eq!( - receipt.evidence.trainer_association, - AttestedTrainerAssociation::Missing - ); - let OperatorGpuFreeOutcome::RestoreClosedServing { - closed, - loan: serving, - request, - } = &receipt.outcome - else { - panic!("the closure must serve the next queued request"); - }; - assert_eq!(closed.id, loan.id); - assert!(matches!( - &closed.state, - LoanState::Closed { - result: LoanClosure::OperatorAttestedRestoreEnded { task_id: ended, operation_id, .. }, - } if *ended == task_id && *operation_id == saved_attestation.operation_id - )); - assert_eq!(request.request_id, first.request_id); - assert!(matches!( - &serving.state, - LoanState::Active { - phase: LoanPhase::Serving { - release_provenance: ServingReleaseProvenance::IdleBoundary { - proof: IdleBoundaryProof::OperatorAttestedGpuFree { task_id: ended, .. }, - }, - .. - }, - } if *ended == task_id - )); - let (resource, current, count) = restore_state(&fixture); - assert_eq!(resource.registered_background_task, None); - assert_eq!(resource.state_revision, receipt.state_revision); - assert_eq!(current.as_ref(), Some(serving)); - assert_eq!(count, 1); - - fixture.store = Store::open(&fixture.database).unwrap(); - let replay = attest_on(&mut fixture.store, saved_attestation.clone()).unwrap(); - assert!(replay.replayed); - assert_eq!( - serde_json::to_value(&replay.receipt).unwrap(), - serde_json::to_value(&receipt).unwrap() - ); - let mut changed = saved_attestation.clone(); - changed.state_binding = OperatorStateBinding::NoLoan; - assert_eq!( - refusal(attest_on(&mut fixture.store, changed)), - OperatorGpuFreeRefusal::ConflictingRetry { - operation_id: saved_attestation.operation_id - } - ); -} - -#[test] -fn restore_closure_is_an_idle_boundary_only_while_its_receipt_is_saved() { - let (mut fixture, loan, action_id, task_id) = restore_ended_before_confirmed_start(); - let (authority, resource_id) = (fixture.authority, fixture.resource.id); - let saved_attestation = attestation_for( - &fixture.store, - authority, - resource_id, - task_id, - OperatorStateBinding::RestoringReturn { - loan_id: loan.id, - action_id, - }, - ); - let receipt = fixture - .store - .attest_trainer_gpu_free_for_authority(authority, saved_attestation.clone()) - .unwrap() - .receipt; - assert!(matches!( - &receipt.outcome, - OperatorGpuFreeOutcome::RestoreClosedIdleBoundary { closed } if closed.id == loan.id - )); - let (resource, current, _) = restore_state(&fixture); - assert_eq!(resource.registered_background_task, None); - assert_eq!(current, None); - - // a restarted authority reads the closure and its receipt as the boundary - fixture.store = Store::open(&fixture.database).unwrap(); - let saved = fixture - .store - .resource_snapshots_for_authority(authority) - .unwrap()[0] - .resource - .clone(); - assert!(matches!( - crate::store::idle_boundary_decision_on(&fixture.store.conn, &saved).unwrap(), - IdleBoundaryDecision::Proven(IdleBoundaryProof::OperatorAttestedGpuFree { - operation_id, - task_id: ended, - }) if operation_id == saved_attestation.operation_id && ended == task_id - )); - - // the closure without its receipt is not evidence - fixture - .store - .conn - .execute( - "DELETE FROM resource_operator_attestations WHERE operation_id = ?1", - [saved_attestation.operation_id.as_uuid().to_string()], - ) - .unwrap(); - assert_eq!( - crate::store::idle_boundary_decision_on(&fixture.store.conn, &saved).unwrap(), - IdleBoundaryDecision::Unproven(IdleProofGap::InconsistentHistory) - ); -} - -/// Bind a native foreground return task and return it still queued -fn queued_foreground_return() -> (ServingFixture, Loan, ActionId, TaskId) { - let (mut fixture, authority) = awaiting_return_fixture(); - let launch = evaluation(&fixture, &["/bin/echo", "evaluate"]); - let task_id = launch.task_id; - let ReturnTaskAcceptance::Inserted { loan, .. } = fixture - .store - .accept_return_task_for_authority(launch_input(authority, launch)) - .unwrap() - else { - panic!("the first exact return launch must insert its task"); - }; - (fixture, *loan, authority.action_id, task_id) -} - -fn foreground_state(fixture: &ServingFixture) -> (Resource, Option, i64) { - let snapshot = fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap() - .remove(0); - ( - snapshot.resource, - snapshot.loan, - attestation_count(&fixture.store), - ) -} - -fn foreground_attestation( - fixture: &ServingFixture, - task_id: TaskId, - state_binding: OperatorStateBinding, -) -> OperatorGpuFreeAttestation { - attestation_for( - &fixture.store, - fixture.authority, - fixture.resource.id, - task_id, - state_binding, - ) -} - -fn attest_foreground( - fixture: &mut ServingFixture, - attestation: OperatorGpuFreeAttestation, -) -> Result { - fixture - .store - .attest_trainer_gpu_free_for_authority(fixture.authority, attestation) -} - -fn restore_closure_count(store: &Store) -> i64 { - store - .conn - .query_row( - "SELECT COUNT(*) FROM resource_restore_closures", - [], - |row| row.get(0), - ) - .unwrap() -} - -#[test] -fn foreground_return_attestation_refuses_live_tasks_and_every_mismatch() { - let (mut fixture, loan, action_id, task_id) = queued_foreground_return(); - let binding = OperatorStateBinding::RestoringForegroundReturn { - loan_id: loan.id, - action_id, - }; - - // a queued or running foreground task still owns the GPU - let queued = foreground_attestation(&fixture, task_id, binding); - assert_eq!( - refusal(attest_foreground(&mut fixture, queued)), - OperatorGpuFreeRefusal::TaskNotEnded { - task_id, - state: ProcessStatus::Queued, - } - ); - start_task(&fixture.store, task_id); - let running = foreground_attestation(&fixture, task_id, binding); - assert_eq!( - refusal(attest_foreground(&mut fixture, running)), - OperatorGpuFreeRefusal::TaskNotEnded { - task_id, - state: ProcessStatus::Running, - } - ); - fixture - .store - .cas_status(task_id, ProcessStatus::Running, ProcessStatus::Lost) - .unwrap() - .unwrap(); - - let before = foreground_state(&fixture); - let other_task = TaskId::new(); - let cases = [ - ( - "the direct-segment binding names another execution mode", - OperatorStateBinding::RestoringReturn { - loan_id: loan.id, - action_id, - }, - task_id, - OperatorGpuFreeRefusal::TrainerLaunchUnproven { task_id }, - ), - ( - "another return action", - OperatorStateBinding::RestoringForegroundReturn { - loan_id: loan.id, - action_id: ActionId::new(), - }, - task_id, - OperatorGpuFreeRefusal::LoanStateChanged { - current_loan: Some(loan.id), - }, - ), - ( - "another loan", - OperatorStateBinding::RestoringForegroundReturn { - loan_id: LoanId::new(), - action_id, - }, - task_id, - OperatorGpuFreeRefusal::LoanStateChanged { - current_loan: Some(loan.id), - }, - ), - ( - "another task", - binding, - other_task, - OperatorGpuFreeRefusal::NotBoundTask { - task_id: other_task, - bound: task_id, - }, - ), - ( - "a registered-trainer binding", - OperatorStateBinding::NoLoan, - task_id, - OperatorGpuFreeRefusal::NotRegisteredTrainer { - task_id, - registered: before.0.registered_background_task, - }, - ), - ]; - for (name, state_binding, task, expected) in cases { - let changed = foreground_attestation(&fixture, task, state_binding); - assert_eq!( - refusal(attest_foreground(&mut fixture, changed)), - expected, - "{name}" - ); - assert_eq!(foreground_state(&fixture), before, "{name}"); - } - - let current = before.0.state_revision; - let stale = OperatorGpuFreeAttestation { - expected_state_revision: ResourceRevision::new(current.get() + 1), - ..foreground_attestation(&fixture, task_id, binding) - }; - assert_eq!( - refusal(attest_foreground(&mut fixture, stale.clone())), - OperatorGpuFreeRefusal::StaleRevision { - expected: stale.expected_state_revision, - actual: current, - } - ); - let other_authority = MachineId::new(); - let named_elsewhere = OperatorGpuFreeAttestation { - authority_machine: other_authority, - ..foreground_attestation(&fixture, task_id, binding) - }; - assert_eq!( - refusal(attest_foreground(&mut fixture, named_elsewhere)), - OperatorGpuFreeRefusal::WrongAuthority { - expected: fixture.authority, - found: other_authority, - } - ); - let exact = foreground_attestation(&fixture, task_id, binding); - assert_eq!( - refusal( - fixture - .store - .attest_trainer_gpu_free_for_authority(other_authority, exact) - ), - OperatorGpuFreeRefusal::WrongAuthority { - expected: other_authority, - found: fixture.authority, - } - ); - - // a registration that names the foreground task is not its Restoring reservation - fixture - .store - .conn - .execute( - "UPDATE resources SET registered_background_task = ?1 WHERE id = ?2", - params![ - task_id.to_string(), - fixture.resource.id.as_uuid().to_string() - ], - ) - .unwrap(); - let registered = foreground_attestation(&fixture, task_id, binding); - assert_eq!( - refusal(attest_foreground(&mut fixture, registered)), - OperatorGpuFreeRefusal::InconsistentHistory { task_id } - ); - assert_eq!(attestation_count(&fixture.store), 0); - assert_eq!(foreground_state(&fixture).1, before.1); -} - -#[test] -fn lost_foreground_return_closes_through_its_exact_attestation_and_serves_queued_work() { - let (mut fixture, loan, action_id, task_id) = queued_foreground_return(); - start_task(&fixture.store, task_id); - fixture - .store - .cas_status(task_id, ProcessStatus::Running, ProcessStatus::Lost) - .unwrap() - .unwrap(); - assert!(matches!( - fixture - .store - .reconcile_restoring_loan_for_authority(fixture.authority, fixture.resource.id) - .unwrap(), - RestoreReconcileOutcome::Attention { - reason: RestoreAttentionReason::Lost, - .. - } - )); - let (authority, resource_id) = (fixture.authority, fixture.resource.id); - let mut queue = Vec::new(); - for _ in 0..2 { - queue.push( - fixture - .store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource_id, - MachineId::new(), - spec(), - ) - .unwrap(), - ); - } - // queued work does not release the reservation of a lost foreground task - assert_eq!(foreground_state(&fixture).1, Some(loan.clone())); - - let saved_attestation = foreground_attestation( - &fixture, - task_id, - OperatorStateBinding::RestoringForegroundReturn { - loan_id: loan.id, - action_id, - }, - ); - let resolution = attest_foreground(&mut fixture, saved_attestation.clone()).unwrap(); - assert!(!resolution.replayed); - let receipt = resolution.receipt; - assert_eq!(receipt.evidence.trainer_end, AttestedTrainerEnd::Lost); - assert!(matches!( - receipt.evidence.trainer_launch, - AttestedTrainerLaunch::NativeForegroundReturn { action_id: bound, .. } if bound == action_id - )); - let OperatorGpuFreeOutcome::RestoreClosedServing { - closed, - loan: serving, - request, - } = &receipt.outcome - else { - panic!("the closure must serve the next queued request"); - }; - assert!(matches!( - &closed.state, - LoanState::Closed { - result: LoanClosure::OperatorAttestedRestoreEnded { task_id: ended, operation_id, .. }, - } if *ended == task_id && *operation_id == saved_attestation.operation_id - )); - assert_eq!(request.request_id, queue[0].request_id); - assert!(matches!( - &serving.state, - LoanState::Active { - phase: LoanPhase::Serving { - release_provenance: ServingReleaseProvenance::IdleBoundary { - proof: IdleBoundaryProof::OperatorAttestedGpuFree { task_id: ended, .. }, - }, - .. - }, - } if *ended == task_id - )); - let (resource, current, count) = foreground_state(&fixture); - assert_eq!(resource.registered_background_task, None); - assert_eq!(resource.state_revision, receipt.state_revision); - assert_eq!(current.as_ref(), Some(serving)); - assert_eq!(count, 1); - assert_eq!( - fixture - .store - .resource_requests(authority, resource_id) - .unwrap() - .into_iter() - .find(|saved| saved.request_id == queue[1].request_id) - .unwrap() - .state, - ResourceRequestState::Queued - ); - // the human attestation is not saved as task or restore-closure exit evidence - assert_eq!( - fixture.store.get_task(task_id).unwrap().unwrap().status(), - ProcessStatus::Lost - ); - assert_eq!(restore_closure_count(&fixture.store), 0); - - // an exact retry replays after restart, although the loan has since closed - fixture.store = Store::open(&fixture.directory.path().join("db")).unwrap(); - let replay = attest_foreground(&mut fixture, saved_attestation.clone()).unwrap(); - assert!(replay.replayed); - assert_eq!( - serde_json::to_value(&replay.receipt).unwrap(), - serde_json::to_value(&receipt).unwrap() - ); - let changed = OperatorGpuFreeAttestation { - observation: OperatorObservation::try_from("a different inspection".to_owned()).unwrap(), - ..saved_attestation.clone() - }; - assert_eq!( - refusal(attest_foreground(&mut fixture, changed)), - OperatorGpuFreeRefusal::ConflictingRetry { - operation_id: saved_attestation.operation_id - } - ); - assert_eq!(attestation_count(&fixture.store), 1); -} - -#[test] -fn unconfirmed_foreground_exit_closes_into_the_saved_idle_boundary() { - let (mut fixture, loan, action_id, task_id) = queued_foreground_return(); - start_task(&fixture.store, task_id); - fixture - .store - .cas_exit_with_evidence( - task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::Unconfirmed, - ) - .unwrap() - .unwrap(); - // neither the owner nor the supervisor treats an unconfirmed exit as release - assert!(matches!( - fixture - .store - .reconcile_restoring_loan_for_authority(fixture.authority, fixture.resource.id) - .unwrap(), - RestoreReconcileOutcome::Attention { - reason: RestoreAttentionReason::ForegroundExitUnconfirmed { .. }, - .. - } - )); - let revision = foreground_state(&fixture).0.state_revision; - assert!(matches!( - fixture - .store - .resolve_ended_restore_for_authority(EndedRestoreResolution { - authority: SupervisorActionAuthority { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - loan_id: loan.id, - action_id, - expected_state_revision: revision, - supervisor: fixture.resource.supervisor, - assignment_revision: fixture.resource.assignment_revision, - }, - task_id, - reason: "exit not confirmed".into(), - }), - Err(ReturnDecisionError::RestoreReleaseUnproven { .. }) - )); - - let saved_attestation = foreground_attestation( - &fixture, - task_id, - OperatorStateBinding::RestoringForegroundReturn { - loan_id: loan.id, - action_id, - }, - ); - let receipt = attest_foreground(&mut fixture, saved_attestation.clone()) - .unwrap() - .receipt; - assert_eq!( - receipt.evidence.trainer_end, - AttestedTrainerEnd::Finished { - outcome: ExitReason::Exit { code: 0 }, - process_group_exit: ProcessGroupExitEvidence::Unconfirmed, - container_exit: None, - } - ); - assert!(matches!( - &receipt.outcome, - OperatorGpuFreeOutcome::RestoreClosedIdleBoundary { closed } if closed.id == loan.id - )); - assert_eq!( - fixture.store.process_group_exit_evidence(task_id).unwrap(), - Some(ProcessGroupExitEvidence::Unconfirmed) - ); - assert_eq!(restore_closure_count(&fixture.store), 0); - - // a restarted authority reads the closure and its receipt as the idle boundary - fixture.store = Store::open(&fixture.directory.path().join("db")).unwrap(); - let (saved, current, _) = foreground_state(&fixture); - assert_eq!(current, None); - assert_eq!(saved.registered_background_task, None); - assert!(matches!( - crate::store::idle_boundary_decision_on(&fixture.store.conn, &saved).unwrap(), - IdleBoundaryDecision::Proven(IdleBoundaryProof::OperatorAttestedGpuFree { - operation_id, - task_id: ended, - }) if operation_id == saved_attestation.operation_id && ended == task_id - )); - let replay = attest_foreground(&mut fixture, saved_attestation).unwrap(); - assert!(replay.replayed); -} - -#[test] -fn saved_restoring_receipts_decode_unchanged_and_bindings_refuse_unknown_fields() { - // a direct-segment receipt as saved before the foreground binding existed - let (operation, task, loan, action, request) = ( - Uuid::now_v7(), - TaskId::new(), - LoanId::new(), - ActionId::new(), - RequestId::new(), - ); - let closed = json!({ - "id": loan, - "resource_id": Uuid::now_v7(), - "state": { - "type": "closed", - "result": { - "type": "operator_attested_restore_ended", - "return_context": { "type": "idle" }, - "task_id": task, - "operation_id": operation, - }, - }, - }); - let legacy = json!({ - "attestation": { - "operation_id": operation, - "resource_id": closed["resource_id"], - "authority_machine": Uuid::now_v7(), - "task_id": task, - "expected_state_revision": 9, - "state_binding": { "type": "restoring_return", "loan_id": loan, "action_id": action }, - "observation": OBSERVATION, - "confirmation": "operator_confirmed_gpu_free", - }, - "evidence": { - "trainer_end": { - "type": "finished", - "outcome": { "kind": "signal", "signal": 9 }, - "process_group_exit": "unconfirmed", - }, - "trainer_launch": { - "type": "direct_segment_return", - "action_id": action, - "request_id": request, - }, - "normalized_spec_sha256": "42".repeat(32), - "trainer_association": { "type": "missing" }, - }, - "state_revision": 10, - "outcome": { "type": "restore_closed_idle_boundary", "closed": closed }, - }); - let saved: crate::resource::operator_release::OperatorGpuFreeReceipt = - serde_json::from_value(legacy.clone()).unwrap(); - assert_eq!( - saved.attestation.state_binding, - OperatorStateBinding::RestoringReturn { - loan_id: loan, - action_id: action, - } - ); - assert_eq!(serde_json::to_value(&saved).unwrap(), legacy); - - // the tag-only no_loan binding keeps its shape and refuses extra fields - let no_loan: OperatorStateBinding = - serde_json::from_value(json!({ "type": "no_loan" })).unwrap(); - assert_eq!(no_loan, OperatorStateBinding::NoLoan); - assert_eq!( - serde_json::to_value(no_loan).unwrap(), - json!({ "type": "no_loan" }) - ); - for binding in [ - json!({ "type": "no_loan", "loan_id": loan }), - json!({ - "type": "restoring_foreground_return", - "loan_id": loan, - "action_id": action, - "task_id": task, - }), - ] { - assert!(serde_json::from_value::(binding).is_err()); - } -} - -#[test] -fn schema_27_receipts_survive_the_restore_outcome_migration() { - let directory = tempdir().unwrap(); - let path = directory.path().join("db"); - let authority = MachineId::new(); - let resource = resource(authority); - let (operation, task, request) = (Uuid::now_v7(), TaskId::new(), RequestId::new()); - // a receipt exactly as schema 27 saved it for a registered trainer - let legacy = json!({ - "attestation": { - "operation_id": operation, - "resource_id": resource.id, - "authority_machine": authority, - "task_id": task, - "expected_state_revision": 4, - "state_binding": { "type": "no_loan" }, - "observation": OBSERVATION, - "confirmation": "operator_confirmed_gpu_free", - }, - "evidence": { - "trainer_end": { "type": "lost" }, - "trainer_launch": { "type": "first_background_launch", "request_id": request }, - "normalized_spec_sha256": "42".repeat(32), - "trainer_association": { "type": "missing" }, - }, - "state_revision": 5, - "outcome": { "type": "idle_boundary" }, - }); - { - let mut store = Store::open(&path).unwrap(); - store.register_resource(authority, &resource).unwrap(); - store - .conn - .execute_batch(&format!( - "DROP TABLE resource_operator_attestations; - CREATE TABLE resource_operator_attestations ( - operation_id TEXT PRIMARY KEY NOT NULL, - resource_id TEXT NOT NULL REFERENCES resources(id), - task_id TEXT NOT NULL UNIQUE, - preceding_loan TEXT, - preceding_launch TEXT, - receipt_json TEXT NOT NULL CHECK ( - json_extract(receipt_json, '$.outcome.type') IN ( - 'release_resolved_serving', 'release_resolved_return_required', - 'idle_serving', 'idle_boundary' - ) - ) - ); - INSERT INTO resource_operator_attestations - (operation_id, resource_id, task_id, preceding_launch, receipt_json) - VALUES ('{operation}', '{}', '{task}', '{}', '{legacy}'); - ALTER TABLE tasks DROP COLUMN worker_thread; - ALTER TABLE origin_routes DROP COLUMN after_json; - ALTER TABLE origin_routes DROP COLUMN outcome; - ALTER TABLE tasks DROP COLUMN container_exit_evidence; - DROP TABLE task_containers; - PRAGMA user_version = 27;", - resource.id.as_uuid(), - request.0 - )) - .unwrap(); - } - - let store = Store::open(&path).unwrap(); - let version: i64 = store - .conn - .pragma_query_value(None, "user_version", |row| row.get(0)) - .unwrap(); - assert_eq!(version, crate::domain::SCHEMA_VERSION); - let saved = store - .operator_attestation_receipt_for_authority( - authority, - OperatorAttestationId::from_uuid(operation).unwrap(), - ) - .unwrap() - .unwrap(); - assert_eq!( - saved.attestation.state_binding, - OperatorStateBinding::NoLoan - ); - assert!(matches!( - saved.outcome, - OperatorGpuFreeOutcome::IdleBoundary - )); - assert_eq!(serde_json::to_value(&saved).unwrap(), legacy); - - // the rebuilt table accepts the new outcomes and keeps its content checks - let insert = |outcome: &str, confirmation: &str| { - let (operation, task) = (Uuid::now_v7(), TaskId::new()); - let receipt = json!({ - "attestation": { - "operation_id": operation, - "resource_id": resource.id, - "task_id": task, - "observation": OBSERVATION, - "confirmation": confirmation, - }, - "evidence": {}, - "outcome": { "type": outcome }, - }); - store.conn.execute( - "INSERT INTO resource_operator_attestations - (operation_id, resource_id, task_id, receipt_json) - VALUES (?1, ?2, ?3, ?4)", - params![ - operation.to_string(), - resource.id.as_uuid().to_string(), - task.to_string(), - receipt.to_string() - ], - ) - }; - assert!( - insert( - "restore_closed_idle_boundary", - "operator_confirmed_gpu_free" - ) - .is_ok() - ); - assert!(insert("restore_closed_serving", "operator_confirmed_gpu_free").is_ok()); - assert!(insert("unknown", "operator_confirmed_gpu_free").is_err()); - assert!(insert("idle_boundary", "confirmed").is_err()); -} diff --git a/src/store/resource/tests/queue_authority.rs b/src/store/resource/tests/queue_authority.rs deleted file mode 100644 index abbe2a1..0000000 --- a/src/store/resource/tests/queue_authority.rs +++ /dev/null @@ -1,119 +0,0 @@ -//! Resource queue authority tests - -use super::fixtures::{identity_count, resource, spec}; -use crate::domain::TaskId; -use crate::machine::MachineId; -use crate::resource::store::ResourceStoreError; -use crate::store::Store; -use crate::submission::{ExecutionRecord, RejectionTombstone, RequestId}; -use tempfile::tempdir; - -#[test] -fn queue_operations_require_the_registered_authority() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let other_machine = MachineId::new(); - let resource = resource(authority); - - assert!(matches!( - store.register_resource(other_machine, &resource), - Err(ResourceStoreError::WrongAuthority { expected, found }) - if expected == authority && found == other_machine - )); - store.register_resource(authority, &resource).unwrap(); - assert_eq!( - store.register_resource(authority, &resource).unwrap(), - resource - ); - - let request = RequestId::new(); - let task = TaskId::new(); - let origin = MachineId::new(); - assert!(matches!( - store.accept_resource_request( - other_machine, - request, - task, - resource.id, - origin, - spec(), - ), - Err(ResourceStoreError::WrongAuthority { expected, found }) - if expected == authority && found == other_machine - )); - assert!( - store - .resource_requests(authority, resource.id) - .unwrap() - .is_empty() - ); - assert!( - store - .next_queued_resource_request(authority, resource.id) - .unwrap() - .is_none() - ); - assert!(matches!( - store.cancel_resource_request_before_activation( - other_machine, - request, - task, - resource.id, - origin, - ), - Err(ResourceStoreError::WrongAuthority { expected, found }) - if expected == authority && found == other_machine - )); - assert_eq!(identity_count(&store, task), 0); -} - -#[test] -fn queue_acceptance_rejects_task_ids_owned_by_the_executor() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let origin = MachineId::new(); - let resource = resource(authority); - store.register_resource(authority, &resource).unwrap(); - - let accepted_task = TaskId::new(); - store - .accept_execution(&ExecutionRecord { - task: accepted_task, - origin_machine: origin, - execution_machine: authority, - spec: spec().into(), - state: crate::domain::ProcessStatus::Queued, - }) - .unwrap(); - let rejected_task = TaskId::new(); - store - .reject_execution(&RejectionTombstone { - task: rejected_task, - origin_machine: origin, - execution_machine: authority, - reason: "prior rejection".into(), - }) - .unwrap(); - - for task in [accepted_task, rejected_task] { - assert!(matches!( - store.accept_resource_request( - authority, - RequestId::new(), - task, - resource.id, - origin, - spec(), - ), - Err(ResourceStoreError::Conflict(_)) - )); - } - assert!( - store - .resource_requests(authority, resource.id) - .unwrap() - .is_empty() - ); -} diff --git a/src/store/resource/tests/registration.rs b/src/store/resource/tests/registration.rs deleted file mode 100644 index e2a2df5..0000000 --- a/src/store/resource/tests/registration.rs +++ /dev/null @@ -1,84 +0,0 @@ -//! Registration retries compare the saved first registration, not the current row - -use super::fixtures::{machine_other_than, resource}; -use crate::domain::ThreadId; -use crate::machine::MachineId; -use crate::resource::store::ResourceStoreError; -use crate::resource::{AssignmentRevision, Resource, SupervisorAddress}; -use crate::store::Store; -use tempfile::tempdir; -use uuid::Uuid; - -fn replace_supervisor(store: &mut Store, authority: MachineId, machine: MachineId) -> Resource { - let current = store.resource_snapshots_for_authority(authority).unwrap()[0] - .resource - .clone(); - store - .replace_resource_supervisor( - authority, - current.id, - current.state_revision, - SupervisorAddress { - machine, - thread: ThreadId(Uuid::now_v7()), - }, - ) - .unwrap() - .resource -} - -#[test] -fn exact_registration_retry_succeeds_after_a_supervisor_replacement() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let first = resource(authority); - assert_eq!(store.register_resource(authority, &first).unwrap(), first); - assert_eq!(store.register_resource(authority, &first).unwrap(), first); - - let replaced = replace_supervisor(&mut store, authority, machine_other_than(authority)); - assert_eq!(replaced.assignment_revision, AssignmentRevision::new(1)); - - // the retry names the first supervisor and answers with the current assignment - assert_eq!( - store.register_resource(authority, &first).unwrap(), - replaced - ); - let models = store - .resource_read_models(authority, Some(first.id)) - .unwrap(); - assert_eq!(models.len(), 1); - assert_eq!(models[0].resource, replaced); - - // a registration that names the current supervisor was never the first one - let mut current_supervisor = first.clone(); - current_supervisor.supervisor = replaced.supervisor; - let mut other_supervisor = first.clone(); - other_supervisor.supervisor.thread = ThreadId(Uuid::now_v7()); - let mut other_name = first.clone(); - other_name.display_name = "another GPU".into(); - for changed in [current_supervisor, other_supervisor, other_name] { - assert!(matches!( - store.register_resource(authority, &changed), - Err(ResourceStoreError::RegistrationConflict { resource }) if resource == first.id - )); - } - let other_authority = machine_other_than(authority); - let moved = Resource::new( - first.id, - first.display_name.clone(), - other_authority, - first.supervisor, - first.assignment_revision, - first.state_revision, - None, - ); - assert!(matches!( - store.register_resource(other_authority, &moved), - Err(ResourceStoreError::RegistrationConflict { .. }) - )); - assert_eq!( - store.resource_snapshots_for_authority(authority).unwrap()[0].resource, - replaced - ); -} diff --git a/src/store/resource/tests/release_checkpoint.rs b/src/store/resource/tests/release_checkpoint.rs deleted file mode 100644 index eba405c..0000000 --- a/src/store/resource/tests/release_checkpoint.rs +++ /dev/null @@ -1,773 +0,0 @@ -//! Release checkpoint baseline, stop decision, and checkpoint cancellation tests - -use super::fixtures::{ - accept_watcher, checkpoint_decision_for_cancellation, open_release_for_test, - release_watcher_intent, saved_release_association, start_release_watcher_for_test, - watcher_spec, watcher_task_and_callback, -}; -use crate::domain::{ExitReason, ProcessStatus, TaskId}; -use crate::machine::MachineId; -use crate::resource::store::{ - ReleaseCheckpointCancellationOutcome, ReleaseCheckpointError, - TrainerAttemptAssociationStoreError, release_checkpoint_state_for_action, -}; -use crate::resource::trainer_publication::AttemptBinding; -use crate::resource::{ - ActionId, ReleaseCheckpointPhase, ReleaseCheckpointStopOutcome, TrainerAttemptAssociationProof, -}; -use crate::store::{CancelResult, Store}; -use std::fs; -use tempfile::tempdir; - -#[test] -fn release_checkpoint_baseline_and_stop_decision_survive_reopen() { - let directory = tempdir().unwrap(); - let database = directory.path().join("db"); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, notice, intent, association, baseline, decision) = { - let mut store = Store::open(&database).unwrap(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - store - .bind_release_watcher_for_authority(authority, resource.id, intent.clone()) - .unwrap(); - let association = saved_release_association(&store, resource.id, background_task); - let baseline = store - .capture_release_checkpoint_baseline_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(); - let runtime_root = association - .verified_attempt() - .canonical_runtime_root() - .to_path_buf(); - let attempt_binding = association.verified_attempt().binding().clone(); - crate::resource::trainer_publication::tests::write_generation_for_test( - &runtime_root, - &attempt_binding, - "generation-a", - 80, - ); - assert_eq!( - store - .capture_release_checkpoint_baseline_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(), - baseline - ); - let ReleaseCheckpointStopOutcome::Reserved(decision) = store - .reserve_release_checkpoint_stop_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap() - else { - panic!("a new exact checkpoint must reserve the stop decision"); - }; - (resource, notice, intent, association, baseline, decision) - }; - - let mut reopened = Store::open(&database).unwrap(); - assert_eq!( - reopened - .capture_release_checkpoint_baseline_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(), - baseline - ); - assert_eq!( - reopened - .reserve_release_checkpoint_stop_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(), - ReleaseCheckpointStopOutcome::AlreadyReserved(decision.clone()) - ); - assert_eq!(decision.binding.watcher_intent, intent); - assert_eq!( - decision.binding.association, - TrainerAttemptAssociationProof::from(&association) - ); - assert_eq!(decision.selected_checkpoint.generation_id, "generation-a"); - assert_eq!( - decision.selected_checkpoint.record_sha256.to_hex().len(), - 64 - ); - assert_eq!( - decision.selected_checkpoint.inventory_sha256.to_hex().len(), - 64 - ); -} - -#[test] -fn pre_baseline_checkpoint_does_not_reserve_stop() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let association = saved_release_association(&store, resource.id, background_task); - let runtime_root = association - .verified_attempt() - .canonical_runtime_root() - .to_path_buf(); - let attempt_binding = association.verified_attempt().binding().clone(); - crate::resource::trainer_publication::tests::write_generation_for_test( - &runtime_root, - &attempt_binding, - "generation-before-baseline", - 80, - ); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - store - .bind_release_watcher_for_authority(authority, resource.id, intent) - .unwrap(); - let baseline = store - .capture_release_checkpoint_baseline_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(); - - assert!( - baseline - .snapshot - .contains_generation("generation-before-baseline") - ); - assert_eq!( - store - .reserve_release_checkpoint_stop_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(), - ReleaseCheckpointStopOutcome::WaitingForCheckpoint - ); -} - -#[test] -fn foreign_attempt_checkpoint_does_not_reserve_stop() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let association = saved_release_association(&store, resource.id, background_task); - let runtime_root = association - .verified_attempt() - .canonical_runtime_root() - .to_path_buf(); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - store - .bind_release_watcher_for_authority(authority, resource.id, intent) - .unwrap(); - store - .capture_release_checkpoint_baseline_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(); - let foreign_attempt = AttemptBinding { - campaign_id: "foreign-campaign".into(), - campaign_revision_id: "foreign-revision".into(), - task_id: "foreign-trainer-task".into(), - attempt_id: "foreign-attempt".into(), - attempt_number: 1, - ownership_token: "foreign-owner".into(), - }; - crate::resource::trainer_publication::tests::write_generation_for_test( - &runtime_root, - &foreign_attempt, - "generation-foreign", - 81, - ); - - assert_eq!( - store - .reserve_release_checkpoint_stop_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(), - ReleaseCheckpointStopOutcome::WaitingForCheckpoint - ); -} - -#[test] -fn new_exact_checkpoint_reserves_one_stop_decision() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let association = saved_release_association(&store, resource.id, background_task); - let runtime_root = association - .verified_attempt() - .canonical_runtime_root() - .to_path_buf(); - let attempt_binding = association.verified_attempt().binding().clone(); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - store - .bind_release_watcher_for_authority(authority, resource.id, intent.clone()) - .unwrap(); - let baseline = store - .capture_release_checkpoint_baseline_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(); - crate::resource::trainer_publication::tests::write_generation_for_test( - &runtime_root, - &attempt_binding, - "generation-new-exact", - 81, - ); - - let ReleaseCheckpointStopOutcome::Reserved(decision) = store - .reserve_release_checkpoint_stop_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap() - else { - panic!("the complete matching checkpoint must reserve a stop decision"); - }; - assert_eq!(decision.binding.watcher_intent, intent); - assert_eq!( - decision.selected_checkpoint.generation_id, - "generation-new-exact" - ); - assert!( - !baseline - .snapshot - .contains_generation(&decision.selected_checkpoint.generation_id) - ); -} - -#[test] -fn duplicate_stop_decision_reuses_checkpoint_and_changed_action_conflicts() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let association = saved_release_association(&store, resource.id, background_task); - let runtime_root = association - .verified_attempt() - .canonical_runtime_root() - .to_path_buf(); - let attempt_binding = association.verified_attempt().binding().clone(); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - store - .bind_release_watcher_for_authority(authority, resource.id, intent) - .unwrap(); - store - .capture_release_checkpoint_baseline_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(); - crate::resource::trainer_publication::tests::write_generation_for_test( - &runtime_root, - &attempt_binding, - "generation-selected", - 81, - ); - let ReleaseCheckpointStopOutcome::Reserved(first) = store - .reserve_release_checkpoint_stop_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap() - else { - panic!("the first checkpoint must reserve the stop decision"); - }; - crate::resource::trainer_publication::tests::write_generation_for_test( - &runtime_root, - &attempt_binding, - "generation-later", - 82, - ); - - assert_eq!( - store - .reserve_release_checkpoint_stop_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(), - ReleaseCheckpointStopOutcome::AlreadyReserved(first.clone()) - ); - assert_eq!( - first.selected_checkpoint.generation_id, - "generation-selected" - ); - assert!(matches!( - store.reserve_release_checkpoint_stop_for_authority( - authority, - resource.id, - ActionId::new(), - notice.state_revision, - ), - Err(ReleaseCheckpointError::Conflict) - )); -} - -#[test] -fn checkpoint_cancellation_commits_decision_and_exact_trainer_marker_atomically() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, notice, intent, decision) = - checkpoint_decision_for_cancellation(&mut store, authority, background_task); - start_release_watcher_for_test(&mut store, &resource, &intent); - - let ReleaseCheckpointCancellationOutcome::Committed(result) = store - .commit_release_checkpoint_cancellation_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - &decision, - ) - .unwrap() - else { - panic!("the accepted watcher must permit the exact stop decision"); - }; - assert_eq!(result.decision, decision); - assert_eq!(result.cancellation.task_id, background_task); - assert_eq!( - store - .require_task(background_task) - .unwrap() - .cancel_requested_at, - Some(result.cancellation.cancel_requested_at) - ); - let (state, _) = - release_checkpoint_state_for_action(&store.conn, resource.id, notice.action_id) - .unwrap() - .unwrap(); - assert!(matches!( - state.phase, - ReleaseCheckpointPhase::CancellationCommitted { - decision: saved, - cancellation, - .. - } if *saved == decision && cancellation == result.cancellation - )); -} - -#[test] -fn checkpoint_cancellation_rolls_back_task_marker_when_phase_update_fails() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, notice, intent, decision) = - checkpoint_decision_for_cancellation(&mut store, authority, background_task); - start_release_watcher_for_test(&mut store, &resource, &intent); - store - .conn - .execute_batch( - "CREATE TRIGGER reject_checkpoint_cancellation - BEFORE UPDATE ON resource_release_checkpoint_states - WHEN json_extract(NEW.state_json, '$.phase.type') = 'cancellation_committed' - BEGIN SELECT RAISE(ABORT, 'forced cancellation phase failure'); END;", - ) - .unwrap(); - - assert!(matches!( - store.commit_release_checkpoint_cancellation_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - &decision, - ), - Err(ReleaseCheckpointError::Storage(_)) - )); - assert_eq!( - store - .require_task(background_task) - .unwrap() - .cancel_requested_at, - None - ); - let (state, _) = - release_checkpoint_state_for_action(&store.conn, resource.id, notice.action_id) - .unwrap() - .unwrap(); - assert!(matches!( - state.phase, - ReleaseCheckpointPhase::StopReserved { decision: saved, .. } - if *saved == decision - )); -} - -#[test] -fn exact_checkpoint_cancellation_retry_reuses_the_committed_reservation() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, notice, intent, decision) = - checkpoint_decision_for_cancellation(&mut store, authority, background_task); - start_release_watcher_for_test(&mut store, &resource, &intent); - let first = store - .commit_release_checkpoint_cancellation_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - &decision, - ) - .unwrap(); - let ReleaseCheckpointCancellationOutcome::Committed(first_result) = first else { - panic!("the first cancellation must commit"); - }; - crate::resource::trainer_publication::tests::write_generation_for_test( - &decision.binding.association.canonical_runtime_root, - &decision.binding.attempt_binding, - "generation-later", - 82, - ); - std::thread::sleep(std::time::Duration::from_millis(2)); - assert!(matches!( - store.request_cancel(background_task).unwrap(), - CancelResult::SignalWorker(_) - )); - - assert_eq!( - store - .commit_release_checkpoint_cancellation_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - &decision, - ) - .unwrap(), - ReleaseCheckpointCancellationOutcome::AlreadyCommitted(first_result.clone()) - ); - assert_eq!( - store - .reserve_release_checkpoint_stop_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ) - .unwrap(), - ReleaseCheckpointStopOutcome::AlreadyReserved(decision) - ); - assert!( - store - .require_task(background_task) - .unwrap() - .cancel_requested_at - .is_some() - ); -} - -#[test] -fn checkpoint_cancellation_rejects_wrong_task_and_action_decisions() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, notice, intent, decision) = - checkpoint_decision_for_cancellation(&mut store, authority, background_task); - start_release_watcher_for_test(&mut store, &resource, &intent); - - let mut wrong_task = decision.clone(); - wrong_task.binding.action.observed_background_task = TaskId::new(); - assert!(matches!( - store.commit_release_checkpoint_cancellation_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - &wrong_task, - ), - Err(ReleaseCheckpointError::StopDecisionMismatch { action_id }) - if action_id == notice.action_id - )); - - let mut wrong_action = decision.clone(); - wrong_action.binding.action.action_id = ActionId::new(); - assert!(matches!( - store.commit_release_checkpoint_cancellation_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - &wrong_action, - ), - Err(ReleaseCheckpointError::StopDecisionMismatch { action_id }) - if action_id == notice.action_id - )); - assert_eq!( - store - .require_task(background_task) - .unwrap() - .cancel_requested_at, - None - ); -} - -#[test] -fn checkpoint_cancellation_waits_for_the_accepted_watcher_to_run() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, notice, intent, decision) = - checkpoint_decision_for_cancellation(&mut store, authority, background_task); - - assert_eq!( - store - .commit_release_checkpoint_cancellation_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - &decision, - ) - .unwrap(), - ReleaseCheckpointCancellationOutcome::WatcherNotReady { - watcher_task_id: intent.watcher_task_id.as_task_id(), - } - ); - let watcher_spec = watcher_spec(&resource, &intent); - let (watcher_row, callback) = watcher_task_and_callback(&intent, &watcher_spec); - accept_watcher( - &mut store, - &resource, - intent.clone(), - &watcher_row, - &watcher_spec, - &callback, - ) - .unwrap(); - assert_eq!( - store - .commit_release_checkpoint_cancellation_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - &decision, - ) - .unwrap(), - ReleaseCheckpointCancellationOutcome::WatcherNotReady { - watcher_task_id: intent.watcher_task_id.as_task_id(), - } - ); - assert_eq!( - store - .require_task(background_task) - .unwrap() - .cancel_requested_at, - None - ); -} - -#[test] -fn checkpoint_cancellation_rejects_changed_checkpoint_and_prior_generic_marker() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let first_task = TaskId::new(); - let (resource, notice, intent, decision) = - checkpoint_decision_for_cancellation(&mut store, authority, first_task); - start_release_watcher_for_test(&mut store, &resource, &intent); - fs::write( - decision.selected_checkpoint.path.join("checkpoint.bin"), - b"changed checkpoint", - ) - .unwrap(); - - assert!(matches!( - store.commit_release_checkpoint_cancellation_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - &decision, - ), - Err(ReleaseCheckpointError::SelectedCheckpointChanged { action_id }) - if action_id == notice.action_id - )); - assert_eq!( - store.require_task(first_task).unwrap().cancel_requested_at, - None - ); - - let second_task = TaskId::new(); - let (resource, notice, intent, decision) = - checkpoint_decision_for_cancellation(&mut store, authority, second_task); - start_release_watcher_for_test(&mut store, &resource, &intent); - assert!(matches!( - store.request_cancel(second_task).unwrap(), - CancelResult::SignalWorker(_) - )); - let generic_marker = store.require_task(second_task).unwrap().cancel_requested_at; - - assert!(matches!( - store.commit_release_checkpoint_cancellation_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - &decision, - ), - Err(ReleaseCheckpointError::TrainerCancellationConflict { task_id }) - if task_id == second_task - )); - assert_eq!( - store.require_task(second_task).unwrap().cancel_requested_at, - generic_marker - ); -} - -#[test] -fn checkpoint_cancellation_rejects_missing_or_terminal_trainers() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let missing_task = TaskId::new(); - let (resource, notice, intent, decision) = - checkpoint_decision_for_cancellation(&mut store, authority, missing_task); - start_release_watcher_for_test(&mut store, &resource, &intent); - store - .conn - .pragma_update(None, "foreign_keys", "OFF") - .unwrap(); - store - .conn - .execute("DELETE FROM tasks WHERE id=?1", [missing_task.to_string()]) - .unwrap(); - assert!(matches!( - store.commit_release_checkpoint_cancellation_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - &decision, - ), - Err(ReleaseCheckpointError::TrainerAssociation( - TrainerAttemptAssociationStoreError::TaskMissing { task_id } - )) if task_id == missing_task - )); - - let terminal_task = TaskId::new(); - let (resource, notice, intent, decision) = - checkpoint_decision_for_cancellation(&mut store, authority, terminal_task); - start_release_watcher_for_test(&mut store, &resource, &intent); - store - .cas_exit( - terminal_task, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ) - .unwrap() - .unwrap(); - store - .update_execution_state(terminal_task, ProcessStatus::Succeeded) - .unwrap(); - - assert!(matches!( - store.commit_release_checkpoint_cancellation_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - &decision, - ), - Err(ReleaseCheckpointError::TrainerTaskNotRunning { - task_id, - state: ProcessStatus::Succeeded, - }) if task_id == terminal_task - )); - assert_eq!( - store - .require_task(terminal_task) - .unwrap() - .cancel_requested_at, - None - ); -} - -#[test] -fn failed_baseline_persistence_keeps_the_action_unbound_to_a_new_snapshot() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - store - .bind_release_watcher_for_authority(authority, resource.id, intent) - .unwrap(); - store - .conn - .execute_batch(&format!( - "CREATE TRIGGER fail_release_checkpoint_update - BEFORE UPDATE ON resource_release_checkpoint_states - WHEN OLD.action_id='{}' - BEGIN SELECT RAISE(ABORT, 'test checkpoint persistence failure'); END;", - notice.action_id.as_uuid() - )) - .unwrap(); - - assert!(matches!( - store.capture_release_checkpoint_baseline_for_authority( - authority, - resource.id, - notice.action_id, - notice.state_revision, - ), - Err(ReleaseCheckpointError::Storage(_)) - )); - let (state, _) = - release_checkpoint_state_for_action(&store.conn, resource.id, notice.action_id) - .unwrap() - .unwrap(); - assert_eq!(state.phase, ReleaseCheckpointPhase::WatcherBindingPending); -} diff --git a/src/store/resource/tests/release_completion.rs b/src/store/resource/tests/release_completion.rs deleted file mode 100644 index 20a69f0..0000000 --- a/src/store/resource/tests/release_completion.rs +++ /dev/null @@ -1,898 +0,0 @@ -//! Completed and stopped release proof tests - -use super::fixtures::{ - TrainerAssociationFixture, assert_ended_release, commit_stopped_release_decision, - fake_resource_task_spec, machine_other_than, publish_completed_result, - release_completion_fixture, release_completion_fixture_with, saved_trainer_association_json, -}; -use crate::domain::{ExitReason, ProcessGroupExitEvidence, ProcessStatus, TaskEnv, TaskId}; -use crate::resource::ownership_lock::{ - OwnershipLockIdentityMismatchReason, OwnershipLockProbeError, -}; -use crate::resource::store::{ - CompleteReleaseError, ReleaseCompletionResult, ResourceTaskAcceptance, - ResourceTaskAcceptanceInput, -}; -use crate::resource::trainer_publication::find_completed_result; -use crate::resource::{ - ActionId, LoanPhase, LoanState, ResourceRequestState, ResourceRevision, ReturnContext, - ServingReleaseProvenance, -}; -use crate::submission::RequestId; -use rusqlite::params; -use serde_json::json; -use std::fs; - -#[test] -fn completed_release_requires_and_uses_the_authority_verified_result() { - let (mut fixture, binding, request_id, action_id, revision, loan_id) = - release_completion_fixture(); - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let publication = find_completed_result(&fixture.runtime_root, &binding) - .unwrap() - .unwrap(); - - let result = fixture - .store - .complete_release_for_authority(fixture.authority, fixture.resource.id, action_id, revision) - .unwrap(); - let ReleaseCompletionResult::Assigned { - loan, - request, - state_revision, - } = result - else { - panic!("queued work must be assigned after a verified completed result"); - }; - - assert_eq!(loan.id, loan_id); - assert_eq!(request.request_id, request_id); - assert_eq!(state_revision, ResourceRevision::new(2)); - assert!(matches!( - loan.state, - LoanState::Active { - phase: LoanPhase::Serving { - return_context: ReturnContext::AlreadyCompleted { - task_id, - result_ref, - }, - .. - } - } if task_id == fixture.task_id - && result_ref == format!( - "{}#sha256={}", - publication.publication_path.display(), - publication.publication_sha256 - ) - )); - assert!(matches!( - fixture.store.resource_requests(fixture.authority, fixture.resource.id).unwrap()[0] - .state, - ResourceRequestState::Assigned { loan_id: assigned } if assigned == loan_id - )); -} - -#[test] -fn exact_release_retry_does_not_need_removed_artifacts() { - let (mut fixture, binding, _, action_id, revision, _) = release_completion_fixture(); - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let first = fixture - .store - .complete_release_for_authority(fixture.authority, fixture.resource.id, action_id, revision) - .unwrap(); - fs::remove_dir_all(&fixture.runtime_root).unwrap(); - - let retry = fixture - .store - .complete_release_for_authority(fixture.authority, fixture.resource.id, action_id, revision) - .unwrap(); - - assert_eq!( - serde_json::to_value(retry).unwrap(), - serde_json::to_value(first).unwrap() - ); -} - -#[test] -fn authority_proves_a_stopped_trainer_from_its_saved_checkpoint_decision() { - let (mut fixture, _, request_id, action_id, revision, loan_id) = release_completion_fixture(); - let (decision, cancellation) = - commit_stopped_release_decision(&mut fixture, action_id, revision); - fixture - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - - let result = fixture - .store - .complete_release_for_authority(fixture.authority, fixture.resource.id, action_id, revision) - .unwrap(); - let ReleaseCompletionResult::Assigned { - loan, - request, - state_revision, - } = result - else { - panic!("the queued request must be assigned after a proved stop"); - }; - let checkpoint = decision.selected_checkpoint; - assert_eq!(request.request_id, request_id); - assert_eq!(loan.id, loan_id); - assert_eq!(state_revision, ResourceRevision::new(2)); - assert!(matches!( - loan.state, - LoanState::Active { - phase: LoanPhase::Serving { - return_context: ReturnContext::Stopped { - task_id, - checkpoint_ref, - recovery_ref, - }, - release_provenance: ServingReleaseProvenance::StoppedTrainerCheckpoint { - action_id: saved_action, - task_id: saved_task, - generation_id, - record_sha256, - inventory_sha256, - }, - .. - } - } if task_id == fixture.task_id - && saved_action == action_id - && saved_task == fixture.task_id - && checkpoint_ref == format!( - "{}#sha256={}", checkpoint.path.display(), checkpoint.record_sha256 - ) - && recovery_ref == checkpoint.generation_id - && generation_id == checkpoint.generation_id - && record_sha256 == checkpoint.record_sha256 - && inventory_sha256 == checkpoint.inventory_sha256 - )); - assert_eq!( - fixture - .store - .require_task(fixture.task_id) - .unwrap() - .cancel_requested_at, - Some(cancellation.cancel_requested_at) - ); -} - -#[test] -fn stopped_release_without_queued_work_creates_a_verified_return_context() { - let (mut fixture, _, request_id, action_id, revision, _) = release_completion_fixture(); - let queued = fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|request| request.request_id == request_id) - .unwrap(); - assert!(matches!( - fixture - .store - .cancel_resource_request_before_activation( - fixture.authority, - queued.request_id, - queued.task_id, - fixture.resource.id, - queued.origin_machine, - ) - .unwrap(), - crate::resource::store::QueueCancellationResult::Request(_) - )); - let (decision, _) = commit_stopped_release_decision(&mut fixture, action_id, revision); - fixture - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - - let result = fixture - .store - .complete_release_for_authority(fixture.authority, fixture.resource.id, action_id, revision) - .unwrap(); - let ReleaseCompletionResult::ReturnRequired { loan, .. } = result else { - panic!("the empty queue must return the verified stopped context"); - }; - let checkpoint = decision.selected_checkpoint; - assert!(matches!( - loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingReturn { - return_context: ReturnContext::Stopped { - task_id, - checkpoint_ref, - recovery_ref, - }, - .. - } - } if task_id == fixture.task_id - && checkpoint_ref == format!( - "{}#sha256={}", checkpoint.path.display(), checkpoint.record_sha256 - ) - && recovery_ref == checkpoint.generation_id - )); -} - -#[test] -fn exact_stopped_release_retry_uses_its_receipt_after_checkpoint_removal() { - let (mut fixture, _, _, action_id, revision, _) = release_completion_fixture(); - let (decision, _) = commit_stopped_release_decision(&mut fixture, action_id, revision); - fixture - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - let first = fixture - .store - .complete_release_for_authority(fixture.authority, fixture.resource.id, action_id, revision) - .unwrap(); - fs::remove_dir_all(&decision.selected_checkpoint.path).unwrap(); - - let retry = fixture - .store - .complete_release_for_authority(fixture.authority, fixture.resource.id, action_id, revision) - .unwrap(); - - assert_eq!( - serde_json::to_value(retry).unwrap(), - serde_json::to_value(first).unwrap() - ); -} - -#[test] -fn stopped_release_rejects_missing_or_changed_selected_checkpoint() { - for changed in ["missing", "contents"] { - let (mut fixture, _, _, action_id, revision, _) = release_completion_fixture(); - let (decision, _) = commit_stopped_release_decision(&mut fixture, action_id, revision); - fixture.finish_registered_task_cancelled_with_evidence( - ProcessGroupExitEvidence::ConfirmedExited, - ); - if changed == "missing" { - fs::remove_dir_all(&decision.selected_checkpoint.path).unwrap(); - } else { - fs::write( - decision.selected_checkpoint.path.join("checkpoint.bin"), - b"changed checkpoint contents", - ) - .unwrap(); - } - - assert!(matches!( - fixture.store.complete_release_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::StoppedCheckpointChanged { task_id }) - if task_id == fixture.task_id - )); - } -} - -#[test] -fn completed_release_rejects_missing_or_mismatched_association() { - let (mut missing, binding, _, action_id, revision, _) = release_completion_fixture(); - publish_completed_result( - &mut missing, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - missing - .store - .conn - .execute( - "DELETE FROM trainer_attempt_associations WHERE task_id = ?1", - [missing.task_id.to_string()], - ) - .unwrap(); - assert!(matches!( - missing.store.complete_release_for_authority( - missing.authority, - missing.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::TrainerAssociationMissing { task_id }) - if task_id == missing.task_id - )); - - let (mut mismatched, binding, _, action_id, revision, _) = release_completion_fixture(); - publish_completed_result( - &mut mismatched, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let mut association: serde_json::Value = serde_json::from_str(&saved_trainer_association_json( - &mismatched.store, - mismatched.task_id, - )) - .unwrap(); - association["normalized_spec_sha256"] = json!("0".repeat(64)); - mismatched - .store - .conn - .execute( - "UPDATE trainer_attempt_associations SET association_json = ?1 - WHERE task_id = ?2", - params![ - serde_json::to_string(&association).unwrap(), - mismatched.task_id.to_string() - ], - ) - .unwrap(); - assert!(matches!( - mismatched.store.complete_release_for_authority( - mismatched.authority, - mismatched.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::TrainerAssociationMismatch { task_id }) - if task_id == mismatched.task_id - )); -} - -#[test] -fn completed_release_rejects_unconfirmed_worker_exit() { - let (mut fixture, binding, _, action_id, revision, _) = release_completion_fixture(); - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::Unconfirmed, - ); - - assert!(matches!( - fixture.store.complete_release_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::WorkerExitUnconfirmed { task_id }) - if task_id == fixture.task_id - )); -} - -#[test] -fn completed_release_rejects_a_still_held_ownership_lock() { - let (mut fixture, binding, _, action_id, revision, _) = release_completion_fixture(); - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let lock_holder = fixture.hold_saved_lock(); - - assert!(matches!( - fixture.store.complete_release_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::OwnershipLockStillHeld { task_id }) - if task_id == fixture.task_id - )); - drop(lock_holder); -} - -#[test] -fn completed_release_rejects_missing_replaced_and_symlinked_locks() { - for (replacement, reason) in [ - ("missing", OwnershipLockIdentityMismatchReason::Missing), - ( - "replaced", - OwnershipLockIdentityMismatchReason::DifferentFile, - ), - ("symlink", OwnershipLockIdentityMismatchReason::SymbolicLink), - ] { - let (mut fixture, binding, _, action_id, revision, _) = release_completion_fixture(); - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let lock_path = fixture.runtime_root.join(".segment.lock"); - let saved_path = fixture.runtime_root.join(".segment.lock.saved"); - if replacement == "missing" { - fs::remove_file(&lock_path).unwrap(); - } else { - fs::rename(&lock_path, &saved_path).unwrap(); - if replacement == "replaced" { - fs::write(&lock_path, b"new lock file").unwrap(); - } else { - std::os::unix::fs::symlink(&saved_path, &lock_path).unwrap(); - } - } - - let result = fixture.store.complete_release_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ); - assert!(matches!( - result, - Err(CompleteReleaseError::OwnershipLock( - OwnershipLockProbeError::IdentityMismatch { - reason: found, - .. - } - )) if found == reason - )); - } -} - -#[test] -fn completed_release_rejects_wrong_action_or_current_task() { - let (mut fixture, binding, _, action_id, revision, loan_id) = release_completion_fixture(); - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let wrong_action = ActionId::new(); - assert!(matches!( - fixture.store.complete_release_for_authority( - fixture.authority, - fixture.resource.id, - wrong_action, - revision, - ), - Err(CompleteReleaseError::NotAwaitingRelease { loan_id: found, action_id }) - if found == loan_id && action_id == wrong_action - )); - - let different_task = TaskId::new(); - fixture - .store - .conn - .execute( - "UPDATE resources SET registered_background_task = ?1 WHERE id = ?2", - params![ - different_task.to_string(), - fixture.resource.id.as_uuid().to_string() - ], - ) - .unwrap(); - assert!(matches!( - fixture.store.complete_release_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::TrainerAssociationMismatch { task_id }) - if task_id == fixture.task_id - )); -} - -#[test] -fn zero_exit_without_result_is_ended_and_a_mismatched_result_fails_closed() { - let (mut missing, _, _, action_id, revision, _) = release_completion_fixture(); - missing.finish_registered_task_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - let result = missing - .store - .complete_release_for_authority(missing.authority, missing.resource.id, action_id, revision) - .unwrap(); - assert_ended_release( - &result, - missing.task_id, - action_id, - &ExitReason::Exit { code: 0 }, - ); - - let (mut mismatched, binding, _, action_id, revision, _) = release_completion_fixture(); - publish_completed_result( - &mut mismatched, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - fs::write( - mismatched - .runtime_root - .join("attempts") - .join(&binding.attempt_id) - .join("terminal.json"), - b"{}", - ) - .unwrap(); - assert!(matches!( - mismatched.store.complete_release_for_authority( - mismatched.authority, - mismatched.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::Watcher(_)) - )); -} - -#[test] -fn nonzero_exit_releases_only_as_an_ended_run_despite_saved_artifacts() { - let (mut fixture, binding, _, action_id, revision, _) = release_completion_fixture(); - crate::resource::trainer_publication::tests::write_completed_result_for_test( - &fixture.runtime_root, - &binding, - ); - crate::resource::trainer_publication::tests::write_generation_for_test( - &fixture.runtime_root, - &binding, - "checkpoint-after-start", - 4, - ); - fixture - .store - .cas_exit_with_evidence( - fixture.task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 3 }, - ProcessGroupExitEvidence::ConfirmedExited, - ) - .unwrap() - .unwrap(); - fixture - .store - .update_execution_state(fixture.task_id, ProcessStatus::Failed) - .unwrap(); - - // neither the result file nor the checkpoint makes a failed run resumable - let result = fixture - .store - .complete_release_for_authority(fixture.authority, fixture.resource.id, action_id, revision) - .unwrap(); - assert_ended_release( - &result, - fixture.task_id, - action_id, - &ExitReason::Exit { code: 3 }, - ); -} - -#[test] -fn stopped_release_requires_the_exact_cancellation_action_task_and_association() { - let (mut wrong_action, _, _, action_id, revision, _) = release_completion_fixture(); - let (_, _) = commit_stopped_release_decision(&mut wrong_action, action_id, revision); - wrong_action - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - let wrong_action_id = ActionId::new(); - assert!(matches!( - wrong_action.store.complete_release_for_authority( - wrong_action.authority, - wrong_action.resource.id, - wrong_action_id, - revision, - ), - Err(CompleteReleaseError::NotAwaitingRelease { action_id, .. }) - if action_id == wrong_action_id - )); - - let (mut wrong_task, _, _, action_id, revision, _) = release_completion_fixture(); - let _ = commit_stopped_release_decision(&mut wrong_task, action_id, revision); - wrong_task - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - let replacement_task = TaskId::new(); - wrong_task - .store - .conn - .execute( - "UPDATE resources SET registered_background_task = ?1 WHERE id = ?2", - params![ - replacement_task.to_string(), - wrong_task.resource.id.as_uuid().to_string() - ], - ) - .unwrap(); - assert!(matches!( - wrong_task.store.complete_release_for_authority( - wrong_task.authority, - wrong_task.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::TrainerAssociationMismatch { task_id }) - if task_id == wrong_task.task_id - )); - - let (mut wrong_association, _, _, action_id, revision, _) = release_completion_fixture(); - let _ = commit_stopped_release_decision(&mut wrong_association, action_id, revision); - wrong_association - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - let mut association: serde_json::Value = serde_json::from_str(&saved_trainer_association_json( - &wrong_association.store, - wrong_association.task_id, - )) - .unwrap(); - association["request_sha256"] = json!("63".repeat(32)); - wrong_association - .store - .conn - .execute( - "UPDATE trainer_attempt_associations SET association_json = ?1 - WHERE task_id = ?2", - params![ - serde_json::to_string(&association).unwrap(), - wrong_association.task_id.to_string(), - ], - ) - .unwrap(); - assert!(matches!( - wrong_association.store.complete_release_for_authority( - wrong_association.authority, - wrong_association.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::TrainerAssociationMismatch { task_id }) - if task_id == wrong_association.task_id - )); -} - -#[test] -fn generic_cancellation_and_checkpoint_release_only_as_an_ended_run() { - let (mut fixture, binding, _, action_id, revision, _) = release_completion_fixture(); - crate::resource::trainer_publication::tests::write_generation_for_test( - &fixture.runtime_root, - &binding, - "generation-without-stop-decision", - 81, - ); - assert!(matches!( - fixture.store.request_cancel(fixture.task_id).unwrap(), - crate::store::CancelResult::SignalWorker(_) - )); - fixture - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - - // a cancellation that no stop decision committed names no resumable checkpoint - let result = fixture - .store - .complete_release_for_authority(fixture.authority, fixture.resource.id, action_id, revision) - .unwrap(); - assert_ended_release(&result, fixture.task_id, action_id, &ExitReason::Cancelled); -} - -#[test] -fn stopped_release_rejects_nonterminal_and_lost_trainers_and_ends_a_failed_stop() { - let (mut running, _, _, action_id, revision, _) = release_completion_fixture(); - assert!(matches!( - running.store.complete_release_for_authority( - running.authority, - running.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::BackgroundTaskNotTerminal { task_id, .. }) - if task_id == running.task_id - )); - - let (mut lost, _, _, action_id, revision, _) = release_completion_fixture(); - lost.store - .conn - .execute( - "UPDATE tasks SET status = 'lost', exit_reason = NULL WHERE id = ?1", - [lost.task_id.to_string()], - ) - .unwrap(); - assert!(matches!( - lost.store.complete_release_for_authority( - lost.authority, - lost.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::BackgroundTaskLost { task_id }) - if task_id == lost.task_id - )); - - let (mut failed, _, _, action_id, revision, _) = release_completion_fixture(); - let _ = commit_stopped_release_decision(&mut failed, action_id, revision); - failed - .store - .cas_exit_with_evidence( - failed.task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 9 }, - ProcessGroupExitEvidence::ConfirmedExited, - ) - .unwrap() - .unwrap(); - failed - .store - .update_execution_state(failed.task_id, ProcessStatus::Failed) - .unwrap(); - // a saved stop decision does not make a failed run resumable - let result = failed - .store - .complete_release_for_authority(failed.authority, failed.resource.id, action_id, revision) - .unwrap(); - assert_ended_release( - &result, - failed.task_id, - action_id, - &ExitReason::Exit { code: 9 }, - ); -} - -#[test] -fn stopped_release_requires_confirmed_worker_exit() { - let (mut fixture, _, _, action_id, revision, _) = release_completion_fixture(); - let _ = commit_stopped_release_decision(&mut fixture, action_id, revision); - fixture.finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::Unconfirmed); - - assert!(matches!( - fixture.store.complete_release_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::WorkerExitUnconfirmed { task_id }) - if task_id == fixture.task_id - )); -} - -#[test] -fn stopped_release_requires_the_saved_lock_to_be_exact_and_exclusively_free() { - for replacement in ["held", "missing", "replaced", "symlink"] { - let (mut fixture, _, _, action_id, revision, _) = release_completion_fixture(); - let _ = commit_stopped_release_decision(&mut fixture, action_id, revision); - fixture.finish_registered_task_cancelled_with_evidence( - ProcessGroupExitEvidence::ConfirmedExited, - ); - let lock_path = fixture.runtime_root.join(".segment.lock"); - let saved_path = fixture.runtime_root.join(".segment.lock.saved"); - if replacement == "held" { - let lock_holder = fixture.hold_saved_lock(); - assert!(matches!( - fixture.store.complete_release_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::OwnershipLockStillHeld { task_id }) - if task_id == fixture.task_id - )); - drop(lock_holder); - continue; - } - if replacement == "missing" { - fs::remove_file(&lock_path).unwrap(); - } else { - fs::rename(&lock_path, &saved_path).unwrap(); - if replacement == "replaced" { - fs::write(&lock_path, b"new lock file").unwrap(); - } else { - std::os::unix::fs::symlink(&saved_path, &lock_path).unwrap(); - } - } - - let expected_reason = match replacement { - "missing" => OwnershipLockIdentityMismatchReason::Missing, - "replaced" => OwnershipLockIdentityMismatchReason::DifferentFile, - "symlink" => OwnershipLockIdentityMismatchReason::SymbolicLink, - _ => unreachable!(), - }; - assert!(matches!( - fixture.store.complete_release_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ), - Err(CompleteReleaseError::OwnershipLock( - OwnershipLockProbeError::IdentityMismatch { - reason, - .. - } - )) if reason == expected_reason - )); - } -} - -#[test] -fn proven_stopped_release_accepts_the_next_request_after_prelaunch_cancel() { - let fixture = TrainerAssociationFixture::new(); - let authority = fixture.authority; - let marker = fixture.home.join("assigned-command-marker"); - let request_spec = fake_resource_task_spec(&fixture.home, &marker); - let first_origin = machine_other_than(authority); - let (mut fixture, _, first_request_id, action_id, revision, loan_id) = - release_completion_fixture_with(fixture, request_spec.clone(), first_origin); - let second_request = fixture - .store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - machine_other_than(authority), - request_spec, - ) - .unwrap(); - let _ = commit_stopped_release_decision(&mut fixture, action_id, revision); - fixture - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - assert!(matches!( - fixture - .store - .complete_release_for_authority( - authority, - fixture.resource.id, - action_id, - revision, - ) - .unwrap(), - ReleaseCompletionResult::Assigned { request, .. } - if request.request_id == first_request_id - )); - let first_request = fixture - .store - .resource_requests(authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|request| request.request_id == first_request_id) - .unwrap(); - fixture - .store - .cancel_resource_request_before_activation( - authority, - first_request.request_id, - first_request.task_id, - fixture.resource.id, - first_request.origin_machine, - ) - .unwrap(); - let snapshot = fixture - .store - .resource_snapshots_for_authority(authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == fixture.resource.id) - .unwrap(); - assert!(matches!( - snapshot.loan.as_ref().map(|loan| &loan.state), - Some(LoanState::Active { - phase: LoanPhase::Serving { - current_request_id, - release_provenance: ServingReleaseProvenance::StoppedTrainerCheckpoint { - action_id: saved_action, - .. - }, - .. - } - }) if *current_request_id == second_request.request_id - && *saved_action == action_id - )); - - let input = ResourceTaskAcceptanceInput { - authority_machine: authority, - resource_id: fixture.resource.id, - request_id: second_request.request_id, - task_id: second_request.task_id, - acceptance_sequence: second_request.acceptance_sequence, - loan_id, - expected_state_revision: snapshot.resource.state_revision, - command_spec: second_request.spec().clone(), - executor_env: TaskEnv::capture(), - }; - assert_eq!( - fixture.store.accept_assigned_resource_task(input).unwrap(), - ResourceTaskAcceptance::Inserted { - task: second_request.task_id, - } - ); - assert_eq!( - fixture - .store - .get_task(second_request.task_id) - .unwrap() - .unwrap() - .status(), - ProcessStatus::Queued - ); -} diff --git a/src/store/resource/tests/release_transaction.rs b/src/store/resource/tests/release_transaction.rs deleted file mode 100644 index de8c965..0000000 --- a/src/store/resource/tests/release_transaction.rs +++ /dev/null @@ -1,392 +0,0 @@ -//! Release transaction recheck tests - -use super::fixtures::{ - commit_stopped_release_decision, publish_completed_result, release_completion_fixture, - saved_trainer_association_json, -}; -use crate::domain::{ExitReason, ProcessGroupExitEvidence, TaskId}; -use crate::resource::ownership_lock::{ - OwnershipLockIdentityMismatchReason, OwnershipLockProbeError, -}; -use crate::resource::store::{ - CompleteReleaseError, - complete_release_for_authority as persist_release_completion_for_authority, -}; -use crate::resource::{ActionId, LoanPhase, LoanState, ResourceRevision}; -use rusqlite::params; -use serde_json::json; -use std::fs; - -#[test] -fn release_transaction_rechecks_task_state_association_and_identity() { - for changed_record in ["task", "task_command", "association", "identity"] { - let (mut fixture, binding, _, action_id, revision, _) = release_completion_fixture(); - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let proof = fixture - .store - .build_verified_release_proof( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ) - .unwrap(); - - let expected_error = match changed_record { - "task" => { - fixture - .store - .conn - .execute( - "UPDATE tasks SET status = 'failed', exit_reason = ?1 - WHERE id = ?2", - params![ - serde_json::to_string(&ExitReason::Exit { code: 1 }).unwrap(), - fixture.task_id.to_string(), - ], - ) - .unwrap(); - "task" - } - "task_command" => { - fixture - .store - .conn - .execute( - "UPDATE tasks SET cwd = ?1 WHERE id = ?2", - params!["/changed/task/command", fixture.task_id.to_string()], - ) - .unwrap(); - "task_command" - } - "association" => { - let mut association: serde_json::Value = serde_json::from_str( - &saved_trainer_association_json(&fixture.store, fixture.task_id), - ) - .unwrap(); - association["normalized_spec_sha256"] = json!("1".repeat(64)); - fixture - .store - .conn - .execute( - "UPDATE trainer_attempt_associations SET association_json = ?1 - WHERE task_id = ?2", - params![ - serde_json::to_string(&association).unwrap(), - fixture.task_id.to_string(), - ], - ) - .unwrap(); - "association" - } - "identity" => { - let mut identity: serde_json::Value = fixture - .store - .conn - .query_row( - "SELECT identity_json FROM executor_identities WHERE task_id = ?1", - [fixture.task_id.to_string()], - |row| row.get::<_, String>(0), - ) - .map(|json| serde_json::from_str(&json).unwrap()) - .unwrap(); - identity["state"] = json!("failed"); - fixture - .store - .conn - .execute( - "UPDATE executor_identities SET identity_json = ?1 WHERE task_id = ?2", - params![ - serde_json::to_string(&identity).unwrap(), - fixture.task_id.to_string() - ], - ) - .unwrap(); - "identity" - } - _ => unreachable!(), - }; - - let result = persist_release_completion_for_authority(&mut fixture.store.conn, proof); - match expected_error { - "task" => assert!(matches!( - result, - Err(CompleteReleaseError::TaskStateChanged { task_id }) - if task_id == fixture.task_id - )), - "task_command" => assert!(matches!( - result, - Err(CompleteReleaseError::TaskCommandChanged { task_id }) - if task_id == fixture.task_id - )), - "association" => assert!(matches!( - result, - Err(CompleteReleaseError::TrainerAssociationMismatch { task_id }) - if task_id == fixture.task_id - )), - "identity" => assert!(matches!( - result, - Err(CompleteReleaseError::TrainerIdentityChanged { task_id }) - if task_id == fixture.task_id - )), - _ => unreachable!(), - } - let snapshot = fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == fixture.resource.id) - .unwrap(); - assert_eq!(snapshot.resource.state_revision, ResourceRevision::new(1)); - assert!(matches!( - snapshot.loan.unwrap().state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { .. } - } - )); - } -} - -#[test] -fn release_transaction_rechecks_action_revision_and_current_task() { - for changed_binding in ["action", "revision", "task"] { - let (mut fixture, binding, _, action_id, revision, _) = release_completion_fixture(); - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let proof = fixture - .store - .build_verified_release_proof( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ) - .unwrap(); - - match changed_binding { - "action" => { - let mut loan_state: serde_json::Value = fixture - .store - .conn - .query_row( - "SELECT state_json FROM loans WHERE resource_id = ?1", - [fixture.resource.id.as_uuid().to_string()], - |row| row.get::<_, String>(0), - ) - .map(|json| serde_json::from_str(&json).unwrap()) - .unwrap(); - loan_state["phase"]["action_id"] = json!(ActionId::new()); - fixture - .store - .conn - .execute( - "UPDATE loans SET state_json = ?1 WHERE resource_id = ?2", - params![ - serde_json::to_string(&loan_state).unwrap(), - fixture.resource.id.as_uuid().to_string(), - ], - ) - .unwrap(); - } - "revision" => { - fixture - .store - .conn - .execute( - "UPDATE resources SET state_revision = 9 WHERE id = ?1", - [fixture.resource.id.as_uuid().to_string()], - ) - .unwrap(); - } - "task" => { - fixture - .store - .conn - .execute( - "UPDATE resources SET registered_background_task = ?1 WHERE id = ?2", - params![ - TaskId::new().to_string(), - fixture.resource.id.as_uuid().to_string(), - ], - ) - .unwrap(); - } - _ => unreachable!(), - } - - let result = persist_release_completion_for_authority(&mut fixture.store.conn, proof); - match changed_binding { - "action" => assert!(matches!( - result, - Err(CompleteReleaseError::NotAwaitingRelease { .. }) - )), - "revision" => assert!(matches!( - result, - Err(CompleteReleaseError::StaleRevision { - expected, - actual, - }) if expected == revision && actual == ResourceRevision::new(9) - )), - "task" => assert!(matches!( - result, - Err(CompleteReleaseError::TrainerAssociationMismatch { task_id }) - if task_id == fixture.task_id - )), - _ => unreachable!(), - } - } -} - -#[test] -fn release_transaction_rechecks_result_and_lock_path_before_commit() { - let (mut changed_result, binding, _, action_id, revision, _) = release_completion_fixture(); - publish_completed_result( - &mut changed_result, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let proof = changed_result - .store - .build_verified_release_proof( - changed_result.authority, - changed_result.resource.id, - action_id, - revision, - ) - .unwrap(); - fs::write( - changed_result - .runtime_root - .join("published") - .join(format!("result-{}", binding.attempt_id)) - .join("checkpoint.bin"), - b"changed result bytes", - ) - .unwrap(); - assert!(matches!( - persist_release_completion_for_authority(&mut changed_result.store.conn, proof), - Err(CompleteReleaseError::Watcher(_)) - )); - - let (mut changed_lock, binding, _, action_id, revision, _) = release_completion_fixture(); - publish_completed_result( - &mut changed_lock, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let proof = changed_lock - .store - .build_verified_release_proof( - changed_lock.authority, - changed_lock.resource.id, - action_id, - revision, - ) - .unwrap(); - let lock_path = changed_lock.runtime_root.join(".segment.lock"); - let saved_path = changed_lock.runtime_root.join(".segment.lock.saved"); - fs::rename(&lock_path, &saved_path).unwrap(); - fs::write(&lock_path, b"replaced lock").unwrap(); - assert!(matches!( - persist_release_completion_for_authority(&mut changed_lock.store.conn, proof), - Err(CompleteReleaseError::OwnershipLock( - OwnershipLockProbeError::IdentityMismatch { - reason: OwnershipLockIdentityMismatchReason::DifferentFile, - .. - } - )) - )); -} - -#[test] -fn stopped_release_transaction_rechecks_saved_marker_checkpoint_and_lock() { - let (mut changed_marker, _, _, action_id, revision, _) = release_completion_fixture(); - let _ = commit_stopped_release_decision(&mut changed_marker, action_id, revision); - changed_marker - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - let proof = changed_marker - .store - .build_verified_release_proof( - changed_marker.authority, - changed_marker.resource.id, - action_id, - revision, - ) - .unwrap(); - changed_marker - .store - .conn - .execute( - "UPDATE tasks SET cancel_requested_at = ?1 WHERE id = ?2", - params![ - "2026-01-01T00:00:00.000000000Z", - changed_marker.task_id.to_string() - ], - ) - .unwrap(); - assert!(matches!( - persist_release_completion_for_authority(&mut changed_marker.store.conn, proof), - Err(CompleteReleaseError::TrainerCancellationMarkerChanged { task_id }) - if task_id == changed_marker.task_id - )); - - let (mut changed_checkpoint, _, _, action_id, revision, _) = release_completion_fixture(); - let (decision, _) = - commit_stopped_release_decision(&mut changed_checkpoint, action_id, revision); - changed_checkpoint - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - let proof = changed_checkpoint - .store - .build_verified_release_proof( - changed_checkpoint.authority, - changed_checkpoint.resource.id, - action_id, - revision, - ) - .unwrap(); - fs::write( - decision.selected_checkpoint.path.join("checkpoint.bin"), - b"changed after proof construction", - ) - .unwrap(); - assert!(matches!( - persist_release_completion_for_authority(&mut changed_checkpoint.store.conn, proof), - Err(CompleteReleaseError::StoppedCheckpointChanged { task_id }) - if task_id == changed_checkpoint.task_id - )); - - let (mut changed_lock, _, _, action_id, revision, _) = release_completion_fixture(); - let _ = commit_stopped_release_decision(&mut changed_lock, action_id, revision); - changed_lock - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - let proof = changed_lock - .store - .build_verified_release_proof( - changed_lock.authority, - changed_lock.resource.id, - action_id, - revision, - ) - .unwrap(); - let lock_path = changed_lock.runtime_root.join(".segment.lock"); - let saved_path = changed_lock.runtime_root.join(".segment.lock.saved"); - fs::rename(&lock_path, &saved_path).unwrap(); - fs::write(&lock_path, b"replaced lock").unwrap(); - assert!(matches!( - persist_release_completion_for_authority(&mut changed_lock.store.conn, proof), - Err(CompleteReleaseError::OwnershipLock( - OwnershipLockProbeError::IdentityMismatch { - reason: OwnershipLockIdentityMismatchReason::DifferentFile, - .. - } - )) - )); -} diff --git a/src/store/resource/tests/release_watcher.rs b/src/store/resource/tests/release_watcher.rs deleted file mode 100644 index f84e9d7..0000000 --- a/src/store/resource/tests/release_watcher.rs +++ /dev/null @@ -1,1780 +0,0 @@ -//! Bound release-watcher poll, launch, restart, and end-to-end release tests - -use crate::domain::TaskRow; -use crate::resource::SupervisorActionAuthority; -use crate::resource::bound_action::{ - ActionTaskAcceptance, ActionTaskIdentity, ResourceActionOperation, ResourceActionOutcome, - ResourceActionRejection, -}; -use crate::store::{AcceptedActionTask, RemoteReleaseWatcherAcceptanceInput, ResourceActionError}; -use std::time::Duration; - -use tokio::net::UnixListener; - -use super::fixtures::{ - TrainerAssociationFixture, accept_watcher, fake_resource_task_spec, gated_resource_task_spec, - machine_other_than, open_release_for_test, prepare_release_checkpoint_baseline, - release_completion_fixture_with, release_watcher_intent, remote_task, resource, - return_notice_count, saved_release_association, spec, start_release_watcher_for_test, - stop_test_supervisor, test_release_association, wait_for_awaiting_return, - watcher_acceptance_counts, watcher_spec, watcher_task_and_callback, -}; -use crate::daemon::AppState; -use crate::daemon::actors::resource::{ReleaseWatcherAttentionReason, ReleaseWatcherStatus}; -use crate::daemon::actors::supervisor::SUPERVISOR_TEST_LOCK; -use crate::daemon::actors::{StoreMsg, SupervisorActor, SupervisorArgs, SupervisorMsg, call}; -use crate::domain::{ExitReason, ProcessGroupExitEvidence, ProcessStatus, TaskEnv, TaskId}; -use crate::files::StreamSlots; -use crate::fleet::FleetState; -use crate::fleet::directory::LocalMachine; -use crate::fleet::protocol::SUPPORTED_PROTOCOLS; -use crate::home::{Home, LockMode, flock_exclusive}; -use crate::machine::{LocalIdentity, MachineId, MachineName, load_or_create_machine_id}; -use crate::resource::release_watcher::{ - RELEASE_WATCHER_POLL_PATH, ReleaseWatcherCommand, ReleaseWatcherPollAttention, - ReleaseWatcherPollOutcome, ReleaseWatcherPollRequest, -}; -use crate::resource::store::{ - ReleaseWatcherAcceptance, ReleaseWatcherAcceptanceError, ReleaseWatcherAcceptanceInput, -}; -use crate::resource::trainer_publication::AttemptBinding; -use crate::resource::{ - ActionId, LoanPhase, LoanState, ReleaseCheckpointStopOutcome, ReleaseWatcherIntent, - ReleaseWatcherTaskId, Resource, ResourceId, ResourceRevision, ServingReleaseProvenance, - SupervisorAddress, SupervisorNotice, -}; -use crate::store::{ExecutorIdentity, NewTask, Store, new_queued_task}; -use crate::submission::{CallbackContext, CallbackExecutable, RequestId, normalized_spec_sha256}; -use ractor::Actor; -use rusqlite::params; -use std::fs; -use std::path::{Path, PathBuf}; -use tempfile::tempdir; - -const WATCHER_TASK_NAME: &str = "resource release watcher"; - -struct PollFixture { - _directory: tempfile::TempDir, - store: Store, - authority: MachineId, - resource: Resource, - notice: SupervisorNotice, - background_task: TaskId, - intent: ReleaseWatcherIntent, - runtime_root: PathBuf, - binding: AttemptBinding, -} - -impl PollFixture { - /// Open one release action with a saved baseline and no accepted watcher - fn new() -> Self { - Self::with_pre_baseline_generation(false) - } - - fn with_pre_baseline_generation(old_generation: bool) -> Self { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let association = saved_release_association(&store, resource.id, background_task); - let runtime_root = association - .verified_attempt() - .canonical_runtime_root() - .to_path_buf(); - let binding = association.verified_attempt().binding().clone(); - if old_generation { - crate::resource::trainer_publication::tests::write_generation_for_test( - &runtime_root, - &binding, - "generation-before-baseline", - 10, - ); - } - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - prepare_release_checkpoint_baseline(&mut store, authority, &resource, &intent); - - Self { - _directory: directory, - store, - authority, - resource, - notice, - background_task, - intent, - runtime_root, - binding, - } - } - - fn command(&self) -> ReleaseWatcherCommand { - ReleaseWatcherCommand::from_intent(self.resource.id, &self.intent) - } - - fn poll(&mut self, command: ReleaseWatcherCommand) -> ReleaseWatcherPollOutcome { - self.store - .poll_release_watcher_for_authority( - self.authority, - ReleaseWatcherPollRequest::new(command), - ) - .unwrap() - } - - fn start_watcher(&mut self) { - start_release_watcher_for_test(&mut self.store, &self.resource, &self.intent); - } - - fn accept_watcher_without_start(&mut self) { - let spec = watcher_spec(&self.resource, &self.intent); - let (row, callback) = watcher_task_and_callback(&self.intent, &spec); - accept_watcher( - &mut self.store, - &self.resource, - self.intent.clone(), - &row, - &spec, - &callback, - ) - .unwrap(); - } - - /// Replace the supervisor machine through the operator control - fn replace_supervisor(&mut self, machine: MachineId) { - let resource = self - .store - .resource_snapshots_for_authority(self.authority) - .unwrap() - .remove(0) - .resource; - let replaced = self - .store - .replace_resource_supervisor( - self.authority, - resource.id, - resource.state_revision, - SupervisorAddress { - machine, - thread: resource.supervisor.thread, - }, - ) - .unwrap(); - assert!(replaced.resource.assignment_revision.get() > resource.assignment_revision.get()); - } - - fn write_generation(&self, generation_id: &str, update_count: u64) { - crate::resource::trainer_publication::tests::write_generation_for_test( - &self.runtime_root, - &self.binding, - generation_id, - update_count, - ); - } - - fn trainer_cancel_marker(&self) -> Option> { - self.store - .get_task(self.background_task) - .unwrap() - .unwrap() - .cancel_requested_at - } - - fn checkpoint_state_json(&self) -> String { - self.store - .conn - .query_row( - "SELECT state_json FROM resource_release_checkpoint_states WHERE action_id=?1", - [self.notice.action_id.as_uuid().to_string()], - |row| row.get(0), - ) - .unwrap() - } -} - -fn attention(reason: ReleaseWatcherPollAttention) -> ReleaseWatcherPollOutcome { - ReleaseWatcherPollOutcome::Attention { reason } -} - -#[test] -fn poll_waits_for_running_watcher_and_ignores_the_baseline_checkpoint() { - let mut fixture = PollFixture::with_pre_baseline_generation(true); - let command = fixture.command(); - - assert_eq!( - fixture.poll(command), - ReleaseWatcherPollOutcome::WatcherNotRunning - ); - fixture.accept_watcher_without_start(); - assert_eq!( - fixture.poll(command), - ReleaseWatcherPollOutcome::WatcherNotRunning - ); - fixture.start_watcher(); - for _ in 0..2 { - assert_eq!( - fixture.poll(command), - ReleaseWatcherPollOutcome::WaitingForCheckpoint - ); - } - assert!(fixture.trainer_cancel_marker().is_none()); -} - -#[test] -fn new_checkpoint_commits_one_marker_and_exact_retries_reuse_the_saved_decision() { - let mut fixture = PollFixture::new(); - fixture.start_watcher(); - let command = fixture.command(); - fixture.write_generation("generation-after-baseline", 20); - - let ReleaseWatcherPollOutcome::StopCommitted { - generation_id, - cancel_requested_at, - } = fixture.poll(command) - else { - panic!("a new complete checkpoint must commit the exact trainer stop"); - }; - assert_eq!(generation_id, "generation-after-baseline"); - assert_eq!(fixture.trainer_cancel_marker(), Some(cancel_requested_at)); - let committed = fixture.checkpoint_state_json(); - - // a later publication cannot replace the saved checkpoint on retry - fixture.write_generation("generation-later", 30); - for _ in 0..2 { - assert_eq!( - fixture.poll(command), - ReleaseWatcherPollOutcome::StopCommitted { - generation_id: "generation-after-baseline".into(), - cancel_requested_at, - } - ); - } - assert_eq!(fixture.checkpoint_state_json(), committed); - assert_eq!(fixture.trainer_cancel_marker(), Some(cancel_requested_at)); -} - -#[test] -fn final_result_before_stop_waits_for_trainer_exit_without_cancelling() { - let mut fixture = PollFixture::new(); - fixture.start_watcher(); - let command = fixture.command(); - crate::resource::trainer_publication::tests::write_request_for_test( - &fixture.runtime_root, - &fixture.binding, - ); - crate::resource::trainer_publication::tests::write_completed_result_for_test( - &fixture.runtime_root, - &fixture.binding, - ); - fixture.write_generation("generation-with-result", 40); - - assert_eq!( - fixture.poll(command), - ReleaseWatcherPollOutcome::CompletedResultAwaitingTrainerExit - ); - assert!(fixture.trainer_cancel_marker().is_none()); - - fixture - .store - .cas_exit_with_evidence( - fixture.background_task, - ProcessStatus::Running, - &ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ) - .unwrap() - .unwrap(); - assert_eq!( - fixture.poll(command), - ReleaseWatcherPollOutcome::TrainerCompleted - ); - assert!(fixture.trainer_cancel_marker().is_none()); -} - -#[test] -fn wrong_watcher_action_trainer_revision_and_authority_are_typed_without_a_marker() { - let mut fixture = PollFixture::new(); - fixture.start_watcher(); - fixture.write_generation("generation-after-baseline", 20); - let command = fixture.command(); - - let cases = [ - ( - ReleaseWatcherCommand { - watcher_task_id: ReleaseWatcherTaskId::new(TaskId::new()), - ..command - }, - ReleaseWatcherPollAttention::WrongWatcher, - ), - ( - ReleaseWatcherCommand { - watcher_task_id: ReleaseWatcherTaskId::new(fixture.background_task), - ..command - }, - ReleaseWatcherPollAttention::WrongWatcher, - ), - ( - ReleaseWatcherCommand { - action_id: ActionId::new(), - ..command - }, - ReleaseWatcherPollAttention::ActionNotCurrent, - ), - ( - ReleaseWatcherCommand { - trainer_task_id: TaskId::new(), - ..command - }, - ReleaseWatcherPollAttention::ActionNotCurrent, - ), - ( - ReleaseWatcherCommand { - state_revision: ResourceRevision::new(command.state_revision.get() + 1), - ..command - }, - ReleaseWatcherPollAttention::ActionNotCurrent, - ), - ( - ReleaseWatcherCommand { - resource_id: ResourceId::new(), - ..command - }, - ReleaseWatcherPollAttention::ActionNotCurrent, - ), - ]; - for (changed, reason) in cases { - assert_eq!(fixture.poll(changed), attention(reason), "{changed:?}"); - } - assert_eq!( - fixture - .store - .poll_release_watcher_for_authority( - MachineId::new(), - ReleaseWatcherPollRequest::new(command), - ) - .unwrap(), - attention(ReleaseWatcherPollAttention::WrongAuthority) - ); - assert!(fixture.trainer_cancel_marker().is_none()); - assert!(matches!( - fixture.poll(command), - ReleaseWatcherPollOutcome::StopCommitted { .. } - )); -} - -#[test] -fn changed_selected_checkpoint_and_lost_trainer_keep_the_loan_without_a_marker() { - let mut changed = PollFixture::new(); - changed.write_generation("generation-selected", 20); - assert!(matches!( - changed - .store - .reserve_release_checkpoint_stop_for_authority( - changed.authority, - changed.resource.id, - changed.notice.action_id, - changed.notice.state_revision, - ) - .unwrap(), - ReleaseCheckpointStopOutcome::Reserved(_) - )); - fs::remove_dir_all(changed.runtime_root.join("published/generation-selected")).unwrap(); - changed.write_generation("generation-replacement", 21); - changed.start_watcher(); - let command = changed.command(); - assert_eq!( - changed.poll(command), - attention(ReleaseWatcherPollAttention::PublicationChanged) - ); - assert!(changed.trainer_cancel_marker().is_none()); - - let mut lost = PollFixture::new(); - lost.start_watcher(); - lost.store - .cas_status( - lost.background_task, - ProcessStatus::Running, - ProcessStatus::Lost, - ) - .unwrap() - .unwrap(); - let command = lost.command(); - assert_eq!( - lost.poll(command), - attention(ReleaseWatcherPollAttention::TrainerLost) - ); - let snapshot = lost - .store - .resource_snapshots_for_authority(lost.authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == lost.resource.id) - .unwrap(); - assert!(matches!( - snapshot.loan.unwrap().state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { .. } - } - )); -} - -#[test] -fn local_watcher_keeps_its_stop_after_the_supervisor_moves_to_another_machine() { - let mut fixture = PollFixture::new(); - fixture.start_watcher(); - let remote = machine_other_than(fixture.authority); - fixture.replace_supervisor(remote); - fixture.write_generation("generation-after-baseline", 20); - let command = fixture.command(); - - // the saved local routes, not the new remote assignment, own this watcher - let ReleaseWatcherPollOutcome::StopCommitted { - cancel_requested_at, - .. - } = fixture.poll(command) - else { - panic!("the accepted local watcher must keep its stop authority"); - }; - assert_eq!(fixture.trainer_cancel_marker(), Some(cancel_requested_at)); -} - -#[test] -fn local_watcher_with_a_changed_route_is_refused_without_a_marker() { - let mut fixture = PollFixture::new(); - fixture.start_watcher(); - fixture.write_generation("generation-after-baseline", 20); - fixture - .store - .conn - .execute( - "UPDATE origin_routes SET route_json = json_set(route_json, '$.thread', ?1) - WHERE request_id = ?2", - params![ - uuid::Uuid::now_v7().to_string(), - fixture.intent.request_id.0.to_string(), - ], - ) - .unwrap(); - let command = fixture.command(); - - assert_eq!( - fixture.poll(command), - attention(ReleaseWatcherPollAttention::WatcherIdentityConflict) - ); - assert!(fixture.trainer_cancel_marker().is_none()); -} - -#[test] -fn watcher_whose_initial_event_changed_is_not_running_and_leaves_no_marker() { - let changed_outbox: fn(&Store, TaskId) = |store, watcher| { - store - .conn - .execute( - "UPDATE executor_outbox SET notification_required = 1 - WHERE task_id = ?1 AND seq = 1", - [watcher.to_string()], - ) - .unwrap(); - }; - let receipt_with_wrong_digest: fn(&Store, TaskId) = |store, watcher| { - store - .conn - .execute( - "DELETE FROM executor_outbox WHERE task_id = ?1 AND seq = 1", - [watcher.to_string()], - ) - .unwrap(); - store - .conn - .execute( - "INSERT INTO executor_event_receipts - (task_id, seq, event_digest, result_json, terminal_callback) - VALUES (?1, 1, 'not-the-queued-event', '\"acknowledged\"', 0)", - [watcher.to_string()], - ) - .unwrap(); - }; - for tamper in [changed_outbox, receipt_with_wrong_digest] { - let mut fixture = PollFixture::new(); - fixture.start_watcher(); - fixture.write_generation("generation-after-baseline", 20); - tamper(&fixture.store, fixture.intent.watcher_task_id.as_task_id()); - let command = fixture.command(); - - assert_eq!( - fixture.poll(command), - ReleaseWatcherPollOutcome::WatcherNotRunning - ); - assert!(fixture.trainer_cancel_marker().is_none()); - } -} - -#[test] -fn corrupt_checkpoint_state_is_attention_instead_of_a_retryable_error() { - let mut fixture = PollFixture::new(); - fixture.start_watcher(); - fixture - .store - .conn - .execute( - "UPDATE resource_release_checkpoint_states - SET state_json = json_set(state_json, '$.action.state_revision', 'corrupt') - WHERE action_id = ?1", - [fixture.notice.action_id.as_uuid().to_string()], - ) - .unwrap(); - let command = fixture.command(); - - assert_eq!( - fixture.poll(command), - attention(ReleaseWatcherPollAttention::CorruptRecord) - ); - assert!(fixture.trainer_cancel_marker().is_none()); -} - -#[test] -fn corrupt_resource_or_loan_row_is_attention_instead_of_a_retryable_error() { - let corrupt_loan_state = - "UPDATE loans SET state_json = json_set(state_json, '$.phase', 'corrupt') - WHERE resource_id = ?1"; - let malformed_thread = "UPDATE resources SET supervisor_thread = 'not-a-uuid' WHERE id = ?1"; - for tamper in [corrupt_loan_state, malformed_thread] { - let mut fixture = PollFixture::new(); - fixture.start_watcher(); - fixture - .store - .conn - .execute(tamper, [fixture.resource.id.as_uuid().to_string()]) - .unwrap(); - let command = fixture.command(); - - assert_eq!( - fixture.poll(command), - attention(ReleaseWatcherPollAttention::CorruptRecord), - "{tamper}" - ); - assert!(fixture.trainer_cancel_marker().is_none()); - } -} - -#[test] -fn caller_command_that_differs_from_the_canonical_watcher_is_rejected() { - let mut fixture = PollFixture::new(); - let spec = watcher_spec(&fixture.resource, &fixture.intent); - let (mut row, callback) = watcher_task_and_callback(&fixture.intent, &spec); - row.binary = PathBuf::from("/bin/sh"); - - assert!(matches!( - accept_watcher( - &mut fixture.store, - &fixture.resource, - fixture.intent.clone(), - &row, - &spec, - &callback, - ), - Err(ReleaseWatcherAcceptanceError::Conflict) - )); - assert_eq!( - watcher_acceptance_counts(&fixture.store, fixture.intent.request_id, row.id), - [0, 0, 0, 0] - ); -} - -fn watcher_executable() -> PathBuf { - assert_cmd::cargo::cargo_bin("homebased") -} - -/// Bind and baseline the release watcher as the authority would before any acceptance -fn bind_saved_watcher( - fixture: &mut TrainerAssociationFixture, - action_id: ActionId, - revision: ResourceRevision, -) -> ReleaseWatcherIntent { - let command = ReleaseWatcherCommand { - resource_id: fixture.resource.id, - action_id, - state_revision: revision, - trainer_task_id: fixture.task_id, - watcher_task_id: ReleaseWatcherTaskId::new(TaskId::new()), - }; - let intent = ReleaseWatcherIntent { - action_id, - state_revision: revision, - observed_background_task: fixture.task_id, - watcher_task_id: command.watcher_task_id, - request_id: RequestId::new(), - normalized_spec_sha256: command - .normalized_spec_sha256(&watcher_executable(), fixture.resource.supervisor.thread) - .unwrap(), - }; - fixture - .store - .bind_release_watcher_for_authority(fixture.authority, fixture.resource.id, intent.clone()) - .unwrap(); - fixture - .store - .capture_release_checkpoint_baseline_for_authority( - fixture.authority, - fixture.resource.id, - action_id, - revision, - ) - .unwrap(); - - intent -} - -fn watcher_task_count(database: &Path) -> i64 { - Store::open(database) - .unwrap() - .conn - .query_row( - "SELECT COUNT(*) FROM tasks WHERE name = ?1", - [WATCHER_TASK_NAME], - |row| row.get(0), - ) - .unwrap() -} - -async fn wait_for_task( - store: &ractor::ActorRef, - task_id: TaskId, - timeout: Duration, - done: impl Fn(&TaskRow) -> bool, -) -> TaskRow { - tokio::time::timeout(timeout, async { - loop { - if let Some(row) = call(store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - && done(&row) - { - return row; - } - tokio::time::sleep(Duration::from_millis(20)).await; - } - }) - .await - .unwrap_or_else(|_| panic!("task {task_id} did not reach the expected state")) -} - -async fn wait_for_watcher_status( - supervisor: &ractor::ActorRef, - resource_id: ResourceId, - done: impl Fn(&ReleaseWatcherStatus) -> bool, -) -> ReleaseWatcherStatus { - tokio::time::timeout(Duration::from_secs(15), async { - loop { - let inspection = call(supervisor, |reply| SupervisorMsg::InspectResource { - id: resource_id, - reply, - }) - .await - .unwrap() - .unwrap(); - if let Some(status) = inspection.release_watcher - && done(&status) - { - return status; - } - call(supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource_id, - reply, - }) - .await - .unwrap(); - tokio::time::sleep(Duration::from_millis(50)).await; - } - }) - .await - .expect("release watcher status did not settle") -} - -async fn cancel_watcher(supervisor: &ractor::ActorRef, task_id: TaskId) { - let _ = call(supervisor, |reply| SupervisorMsg::Cancel { - id: task_id, - reply, - }) - .await; -} - -#[tokio::test] -async fn bound_watcher_stops_the_trainer_at_a_new_checkpoint_and_one_optimization_runs() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - crate::runner::set_task_run_executable_for_tests(watcher_executable()); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let mut fixture = TrainerAssociationFixture::new_for_home(directory, &home, authority); - // the fake trainer runs until its task-run cancels the child group - fs::write(&fixture.python, b"#!/bin/sh\nexec /bin/sleep 600\n").unwrap(); - let marker = fixture.home.join("activation-count"); - let gate = fixture.home.join("release-commands"); - let request_spec = gated_resource_task_spec(&fixture.home, &marker, &gate); - let resource_id = fixture.resource.id; - let trainer_task_id = fixture.task_id; - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - home.prepare_task(trainer_task_id).unwrap(); - let trainer_row = fixture.trainer_task(trainer_task_id, &fixture.spec); - call(&supervisor, |reply| SupervisorMsg::Launch { - row: Box::new(trainer_row), - spec: Box::new(fixture.spec.clone()), - admission: crate::store::LocalAdmission::Submitted { - request: crate::submission::RequestId::new(), - after: None, - }, - reply, - }) - .await - .unwrap(); - wait_for_task(&store, trainer_task_id, Duration::from_secs(10), |row| { - row.status() == ProcessStatus::Running && row.pid().is_some() - }) - .await; - let binding = fixture.register_release_attempt(); - let request = fixture - .store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource_id, - machine_other_than(authority), - request_spec, - ) - .unwrap(); - - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource_id, - reply, - }) - .await - .unwrap(); - let status = wait_for_watcher_status(&supervisor, resource_id, |status| { - matches!(status, ReleaseWatcherStatus::Running { .. }) - }) - .await; - let ReleaseWatcherStatus::Running { - action_id, - watcher_task_id, - } = status - else { - unreachable!(); - }; - - // the watcher starts while the daemon socket is absent and must retry - tokio::time::sleep(Duration::from_millis(300)).await; - let machine = LocalMachine { - identity: LocalIdentity::start(&home).unwrap(), - name: MachineName::fallback(), - protocol: SUPPORTED_PROTOCOLS, - }; - assert_eq!(machine.identity.machine, authority); - let state = AppState { - home: home.clone(), - store: store.clone(), - supervisor: supervisor.clone(), - web: None, - content: None, - stream_slots: StreamSlots::new(), - machine, - fleet: FleetState::Disabled, - message_receiver: crate::daemon::message_receiver::MessageReceiver::default(), - locks: crate::daemon::DaemonLocks::default(), - thread_titles: None, - }; - let listener = UnixListener::bind(home.sock_path()).unwrap(); - let router = crate::daemon::api::socket_router(state.clone()); - let socket_server = tokio::spawn(async move { axum::serve(listener, router).await }); - tokio::time::sleep(Duration::from_millis(300)).await; - assert!( - call(&store, |reply| StoreMsg::GetTask { - id: trainer_task_id, - reply, - }) - .await - .unwrap() - .unwrap() - .cancel_requested_at - .is_none() - ); - crate::resource::trainer_publication::tests::write_generation_for_test( - &fixture.runtime_root, - &binding, - "generation-after-baseline", - 90, - ); - - let trainer = wait_for_task(&store, trainer_task_id, Duration::from_secs(40), |row| { - row.status().is_terminal() - }) - .await; - assert_eq!(trainer.status(), ProcessStatus::Cancelled); - assert_eq!( - trainer.process_group_exit_evidence(), - ProcessGroupExitEvidence::ConfirmedExited - ); - wait_for_task(&store, request.task_id, Duration::from_secs(30), |row| { - row.status() == ProcessStatus::Running - }) - .await; - let inspection = call(&supervisor, |reply| SupervisorMsg::InspectResource { - id: resource_id, - reply, - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - inspection.loan.as_ref().map(|loan| &loan.state), - Some(LoanState::Active { - phase: LoanPhase::Serving { - release_provenance: ServingReleaseProvenance::StoppedTrainerCheckpoint { - action_id: saved_action, - task_id: saved_task, - generation_id, - .. - }, - .. - } - }) if *saved_action == action_id - && *saved_task == trainer_task_id - && generation_id == "generation-after-baseline" - )); - - fs::write(&gate, b"").unwrap(); - let optimization = wait_for_task(&store, request.task_id, Duration::from_secs(30), |row| { - row.status().is_terminal() - }) - .await; - assert_eq!(optimization.status(), ProcessStatus::Succeeded); - let watcher = wait_for_task(&store, watcher_task_id, Duration::from_secs(30), |row| { - row.status().is_terminal() - }) - .await; - assert_eq!(watcher.status(), ProcessStatus::Succeeded); - - for _ in 0..3 { - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource_id, - reply, - }) - .await - .unwrap(); - } - assert_eq!(fs::read(&marker).unwrap(), b"x"); - assert_eq!(watcher_task_count(&fixture.database), 1); - let loan = wait_for_awaiting_return(&supervisor, resource_id).await; - assert_eq!(return_notice_count(&home, loan.id), 1); - - assert_route_is_socket_only(state).await; - socket_server.abort(); - stop_test_supervisor(supervisor, handle).await; -} - -async fn assert_route_is_socket_only(state: AppState) { - use http_body_util::{BodyExt, Full}; - use hyper_util::rt::TokioIo; - - let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); - let bind = listener.local_addr().unwrap(); - let router = crate::daemon::web::router(state, bind); - let server = tokio::spawn(async move { axum::serve(listener, router).await }); - let stream = tokio::net::TcpStream::connect(bind).await.unwrap(); - let (mut sender, connection) = hyper::client::conn::http1::handshake(TokioIo::new(stream)) - .await - .unwrap(); - tokio::spawn(connection); - let request = hyper::Request::builder() - .method("POST") - .uri(RELEASE_WATCHER_POLL_PATH) - .header("host", bind.to_string()) - .header("content-type", "application/json") - .body(Full::new(bytes::Bytes::from_static(b"{}"))) - .unwrap(); - let response = sender.send_request(request).await.unwrap(); - let status = response.status(); - let _ = response.into_body().collect().await; - server.abort(); - - assert!( - status == hyper::StatusCode::NOT_FOUND || status == hyper::StatusCode::METHOD_NOT_ALLOWED, - "dashboard listener served the watcher route with {status}" - ); -} - -#[tokio::test] -async fn restart_before_acceptance_launches_the_saved_watcher_once_and_later_only_observes() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - crate::runner::set_task_run_executable_for_tests(watcher_executable()); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let fixture = TrainerAssociationFixture::new_for_home(directory, &home, authority); - let marker = fixture.home.join("activation-count"); - let request_spec = fake_resource_task_spec(&fixture.home, &marker); - let (mut fixture, _, _, action_id, revision, _) = - release_completion_fixture_with(fixture, request_spec, machine_other_than(authority)); - // the trainer row is live, so hold its runner lock like a live worker - home.prepare_task(fixture.task_id).unwrap(); - let trainer_lock = flock_exclusive( - &home.task_paths(fixture.task_id).runner_lock, - LockMode::NonBlocking, - ) - .unwrap(); - let intent = bind_saved_watcher(&mut fixture, action_id, revision); - let watcher_task_id = intent.watcher_task_id.as_task_id(); - let resource_id = fixture.resource.id; - let database = fixture.database.clone(); - assert_eq!(watcher_task_count(&database), 0); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let status = wait_for_watcher_status(&supervisor, resource_id, |status| { - matches!(status, ReleaseWatcherStatus::Running { .. }) - }) - .await; - assert_eq!( - status, - ReleaseWatcherStatus::Running { - action_id, - watcher_task_id, - } - ); - let running = wait_for_task(&store, watcher_task_id, Duration::from_secs(10), |row| { - row.status() == ProcessStatus::Running && row.pid().is_some() - }) - .await; - assert_eq!(watcher_task_count(&database), 1); - stop_test_supervisor(supervisor, handle).await; - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let status = wait_for_watcher_status(&supervisor, resource_id, |status| { - matches!(status, ReleaseWatcherStatus::Running { .. }) - }) - .await; - assert_eq!( - status, - ReleaseWatcherStatus::Running { - action_id, - watcher_task_id, - } - ); - let observed = call(&store, |reply| StoreMsg::GetTask { - id: watcher_task_id, - reply, - }) - .await - .unwrap() - .unwrap(); - assert_eq!(observed.pid(), running.pid()); - assert_eq!(watcher_task_count(&database), 1); - assert!( - call(&store, |reply| StoreMsg::GetTask { - id: fixture.task_id, - reply, - }) - .await - .unwrap() - .unwrap() - .cancel_requested_at - .is_none() - ); - - cancel_watcher(&supervisor, watcher_task_id).await; - wait_for_task(&store, watcher_task_id, Duration::from_secs(20), |row| { - row.status().is_terminal() - }) - .await; - stop_test_supervisor(supervisor, handle).await; - drop(trainer_lock); -} - -#[tokio::test] -async fn accepted_queued_watcher_after_restart_is_uncertain_and_never_spawned() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - crate::runner::set_task_run_executable_for_tests(watcher_executable()); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let fixture = TrainerAssociationFixture::new_for_home(directory, &home, authority); - let marker = fixture.home.join("activation-count"); - let request_spec = fake_resource_task_spec(&fixture.home, &marker); - let (mut fixture, _, _, action_id, revision, _) = - release_completion_fixture_with(fixture, request_spec, machine_other_than(authority)); - home.prepare_task(fixture.task_id).unwrap(); - let trainer_lock = flock_exclusive( - &home.task_paths(fixture.task_id).runner_lock, - LockMode::NonBlocking, - ) - .unwrap(); - let intent = bind_saved_watcher(&mut fixture, action_id, revision); - let watcher_task_id = intent.watcher_task_id.as_task_id(); - // the previous daemon committed acceptance and stopped before its spawn - let executable = watcher_executable(); - let spec = ReleaseWatcherCommand::from_intent(fixture.resource.id, &intent) - .normalized_spec(&executable, fixture.resource.supervisor.thread) - .unwrap(); - let env = TaskEnv::capture(); - let row = new_queued_task(NewTask { - id: watcher_task_id, - name: Some(spec.name.clone()), - thread: spec.thread, - workload: crate::invocation::persist_workload(&spec.workload), - cwd: spec.cwd.clone(), - timeout: spec.timeout, - env: env.clone(), - binary: executable, - }); - let callback = CallbackContext { - env, - cwd: spec.cwd.clone(), - codex: CallbackExecutable::available(PathBuf::from("/bin/echo")), - }; - assert_eq!( - fixture - .store - .accept_release_watcher_for_authority(ReleaseWatcherAcceptanceInput { - authority_machine: authority, - resource_id: fixture.resource.id, - supervisor: fixture.resource.supervisor, - intent, - row, - spec, - callback, - }) - .unwrap(), - ReleaseWatcherAcceptance::Inserted { - task: watcher_task_id - } - ); - home.prepare_task(watcher_task_id).unwrap(); - let resource_id = fixture.resource.id; - let database = fixture.database.clone(); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let status = wait_for_watcher_status(&supervisor, resource_id, |status| { - matches!(status, ReleaseWatcherStatus::Attention { .. }) - }) - .await; - assert!( - matches!( - status, - ReleaseWatcherStatus::Attention { - action_id: saved_action, - reason: ReleaseWatcherAttentionReason::LaunchUncertain { watcher_task_id: task } - | ReleaseWatcherAttentionReason::WatcherTaskEnded { - watcher_task_id: task, - state: ProcessStatus::Lost, - }, - } if saved_action == action_id && task == watcher_task_id - ), - "{status:?}" - ); - for _ in 0..3 { - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource_id, - reply, - }) - .await - .unwrap(); - } - let watcher = call(&store, |reply| StoreMsg::GetTask { - id: watcher_task_id, - reply, - }) - .await - .unwrap() - .unwrap(); - assert!(watcher.pid().is_none()); - assert!(!matches!(watcher.status(), ProcessStatus::Running)); - assert_eq!(watcher_task_count(&database), 1); - let inspection = call(&supervisor, |reply| SupervisorMsg::InspectResource { - id: resource_id, - reply, - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - inspection.loan.map(|loan| loan.state), - Some(LoanState::Active { - phase: LoanPhase::AwaitingRelease { .. } - }) - )); - - stop_test_supervisor(supervisor, handle).await; - drop(trainer_lock); -} - -#[tokio::test] -async fn remote_supervisor_watcher_is_bound_then_launched_once_for_the_supervisor_machine() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - crate::runner::set_task_run_executable_for_tests(watcher_executable()); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let remote_supervisor = machine_other_than(authority); - let background_task = TaskId::new(); - let resource_id = { - let mut store = Store::open(&home.db_path()).unwrap(); - let trainer_spec = spec(); - let mut resource = resource(authority); - resource.supervisor = SupervisorAddress { - machine: remote_supervisor, - thread: trainer_spec.thread, - }; - resource.registered_background_task = Some(background_task); - store.register_resource(authority, &resource).unwrap(); - store - .insert_local_task( - &remote_task(background_task, &trainer_spec), - &trainer_spec, - authority, - crate::submission::RequestId::new(), - PathBuf::from("/bin/echo").into(), - ) - .unwrap(); - store - .cas_status( - background_task, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - test_release_association(&mut store, authority, resource.id, background_task); - store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - spec(), - ) - .unwrap(); - resource.id - }; - home.prepare_task(background_task).unwrap(); - let trainer_lock = flock_exclusive( - &home.task_paths(background_task).runner_lock, - LockMode::NonBlocking, - ) - .unwrap(); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let status = wait_for_watcher_status(&supervisor, resource_id, |_| true).await; - let ReleaseWatcherStatus::AwaitingRemoteSupervisor { - watcher_task_id, - supervisor_machine, - .. - } = status - else { - panic!("remote supervisor watcher status was {status:?}"); - }; - assert_eq!(supervisor_machine, remote_supervisor); - let saved = Store::open(&home.db_path()) - .unwrap() - .resource_snapshots_for_authority(authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == resource_id) - .unwrap(); - // the authority binds the identity and baseline, but writes no task until the - // supervisor machine has saved its callback route and asks for the launch - assert!(matches!( - saved.loan.clone().map(|loan| loan.state), - Some(LoanState::Active { - phase: LoanPhase::AwaitingRelease { - watcher_intent: Some(intent), - .. - } - }) if intent.watcher_task_id.as_task_id() == watcher_task_id - )); - assert_eq!(watcher_task_count(&home.db_path()), 0); - - // the supervisor machine prepares, saves its route, and launches the same identity - let SupervisorAddress { thread, .. } = saved.resource.supervisor; - let loan = saved.loan.unwrap(); - let LoanState::Active { - phase: LoanPhase::AwaitingRelease { action_id, .. }, - } = loan.state - else { - unreachable!(); - }; - let action = SupervisorActionAuthority { - authority_machine: authority, - resource_id, - loan_id: loan.id, - action_id, - expected_state_revision: saved.resource.state_revision, - supervisor: SupervisorAddress { - machine: remote_supervisor, - thread, - }, - assignment_revision: saved.resource.assignment_revision, - }; - let send = |operation| { - let supervisor = supervisor.clone(); - async move { - call(&supervisor, |reply| SupervisorMsg::ResourceAction { - request: Box::new(crate::resource::bound_action::ResourceActionRequest::new( - 1, action, operation, - )), - reply, - }) - .await - .unwrap() - } - }; - let ResourceActionOutcome::Prepared { task } = - send(ResourceActionOperation::PrepareReleaseWatcher { - observed_background_task: background_task, - }) - .await - else { - panic!("the authority must prepare the bound watcher"); - }; - assert_eq!(task.task_id, watcher_task_id); - assert!(task.digest_matches()); - assert_eq!(task.spec.thread, thread); - let launch = ResourceActionOperation::LaunchReleaseWatcher { - observed_background_task: background_task, - task: ActionTaskIdentity { - request_id: task.request_id, - task_id: task.task_id, - normalized_spec_sha256: task.normalized_spec_sha256, - }, - }; - let ResourceActionOutcome::Accepted { acceptance, .. } = send(launch.clone()).await else { - panic!("the first launch must be accepted"); - }; - assert_eq!(acceptance, ActionTaskAcceptance::Inserted); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - wait_for_task(&store, watcher_task_id, Duration::from_secs(10), |row| { - row.status() == ProcessStatus::Running - }) - .await; - let ResourceActionOutcome::Accepted { acceptance, .. } = send(launch).await else { - panic!("an exact retry must observe the accepted watcher"); - }; - assert_eq!( - acceptance, - ActionTaskAcceptance::Existing { - state: ProcessStatus::Running - } - ); - assert_eq!(watcher_task_count(&home.db_path()), 1); - wait_for_watcher_status(&supervisor, resource_id, |status| { - matches!(status, ReleaseWatcherStatus::Running { watcher_task_id: running, .. } if *running == watcher_task_id) - }) - .await; - - cancel_watcher(&supervisor, watcher_task_id).await; - stop_test_supervisor(supervisor, handle).await; - drop(trainer_lock); -} - -/// Release action whose supervisor thread runs on another machine -struct RemoteWatcherFixture { - poll: PollFixture, - authority: SupervisorActionAuthority, -} - -impl RemoteWatcherFixture { - fn new() -> Self { - let poll = PollFixture::new(); - let remote = machine_other_than(poll.authority); - poll.store - .conn - .execute( - "UPDATE resources SET supervisor_machine = ?1 WHERE id = ?2", - params![ - remote.as_uuid().to_string(), - poll.resource.id.as_uuid().to_string() - ], - ) - .unwrap(); - let authority = SupervisorActionAuthority { - authority_machine: poll.authority, - resource_id: poll.resource.id, - loan_id: poll.notice.loan_id, - action_id: poll.notice.action_id, - expected_state_revision: poll.notice.state_revision, - supervisor: SupervisorAddress { - machine: remote, - thread: poll.resource.supervisor.thread, - }, - assignment_revision: poll.resource.assignment_revision, - }; - Self { poll, authority } - } - - fn identity(&self) -> ActionTaskIdentity { - ActionTaskIdentity { - request_id: self.poll.intent.request_id, - task_id: self.poll.intent.watcher_task_id.as_task_id(), - normalized_spec_sha256: self.poll.intent.normalized_spec_sha256, - } - } - - fn input(&self) -> RemoteReleaseWatcherAcceptanceInput { - let spec = watcher_spec(&self.poll.resource, &self.poll.intent); - let row = remote_task(self.poll.intent.watcher_task_id.as_task_id(), &spec); - RemoteReleaseWatcherAcceptanceInput { - authority: self.authority, - observed_background_task: self.poll.background_task, - task: self.identity(), - row, - spec, - } - } - - fn accept( - &mut self, - input: RemoteReleaseWatcherAcceptanceInput, - ) -> Result { - self.poll - .store - .accept_remote_release_watcher_for_authority(input) - } - - fn start(&mut self) { - let task = self.poll.intent.watcher_task_id.as_task_id(); - self.poll - .store - .cas_status(task, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - self.poll - .store - .set_pid(task, std::process::id() as i32) - .unwrap(); - self.poll - .store - .update_execution_state(task, ProcessStatus::Running) - .unwrap(); - } - - /// Overwrite the saved receipt JSON as a corrupted or forged acceptance would - fn rewrite_receipt( - &self, - change: impl FnOnce(&mut crate::resource::bound_action::ActionTaskReceipt), - ) { - let task = self.poll.intent.watcher_task_id.as_task_id(); - let mut receipt = crate::store::resource::action_task::action_task_receipt_by_task( - &self.poll.store.conn, - task, - ) - .unwrap() - .unwrap(); - change(&mut receipt); - self.poll - .store - .conn - .execute( - "UPDATE resource_action_task_receipts SET receipt_json = ?1 WHERE task_id = ?2", - params![serde_json::to_string(&receipt).unwrap(), task.to_string()], - ) - .unwrap(); - } - - fn receipt_count(&self) -> i64 { - self.poll - .store - .conn - .query_row( - "SELECT COUNT(*) FROM resource_action_task_receipts", - [], - |row| row.get(0), - ) - .unwrap() - } -} - -fn rejected(result: Result) -> ResourceActionRejection { - match result { - Err(ResourceActionError::Rejected(reason)) => reason, - other => panic!("expected a typed rejection, found {other:?}"), - } -} - -#[test] -fn remote_watcher_acceptance_saves_one_receipt_and_remote_identity_without_a_local_route() { - use crate::resource::bound_action::ActionTaskAcceptance; - - let mut fixture = RemoteWatcherFixture::new(); - let task = fixture.poll.intent.watcher_task_id.as_task_id(); - let request = fixture.poll.intent.request_id; - let accepted = fixture.accept(fixture.input()).unwrap(); - assert_eq!(accepted.acceptance, ActionTaskAcceptance::Inserted); - assert_eq!(accepted.receipt.task_id, task); - assert_eq!( - accepted.receipt.origin_machine(), - fixture.authority.supervisor.machine - ); - // row, identity, and first event exist; the callback route lives on the supervisor machine - assert_eq!( - watcher_acceptance_counts(&fixture.poll.store, request, task), - [1, 0, 1, 1] - ); - assert_eq!(fixture.receipt_count(), 1); - let Some(ExecutorIdentity::Accepted(record)) = - fixture.poll.store.executor_identity(task).unwrap() - else { - panic!("the watcher must have an accepted identity"); - }; - assert_eq!(record.origin_machine, fixture.authority.supervisor.machine); - assert_eq!(record.execution_machine, fixture.poll.authority); - let event = fixture - .poll - .store - .first_pending_outbound(task) - .unwrap() - .unwrap() - .event; - assert_eq!(event.origin_machine, fixture.authority.supervisor.machine); - - // an exact retry, even after reopening, only observes the saved acceptance - let retried = fixture.accept(fixture.input()).unwrap(); - assert_eq!( - retried.acceptance, - ActionTaskAcceptance::Existing { - state: ProcessStatus::Queued - } - ); - let path = fixture.poll._directory.path().join("db"); - fixture.poll.store = Store::open(&path).unwrap(); - assert_eq!( - fixture.accept(fixture.input()).unwrap().acceptance, - ActionTaskAcceptance::Existing { - state: ProcessStatus::Queued - } - ); - assert_eq!( - watcher_acceptance_counts(&fixture.poll.store, request, task), - [1, 0, 1, 1] - ); - - // a changed digest for the saved identity is a conflicting retry - let mut changed = fixture.input(); - changed.task.normalized_spec_sha256 = normalized_spec_sha256(&spec()).unwrap(); - assert_eq!( - rejected(fixture.accept(changed)), - ResourceActionRejection::ConflictingRetry - ); - assert_eq!(fixture.receipt_count(), 1); -} - -#[test] -fn remote_watcher_acceptance_rejects_changed_command_owner_and_revision_without_records() { - use crate::resource::bound_action::ResourceActionRejection; - - let mut fixture = RemoteWatcherFixture::new(); - let task = fixture.poll.intent.watcher_task_id.as_task_id(); - let request = fixture.poll.intent.request_id; - - let mut caller_command = fixture.input(); - caller_command.row.binary = PathBuf::from("/bin/sh"); - assert_eq!( - rejected(fixture.accept(caller_command)), - ResourceActionRejection::SpecMismatch - ); - let mut caller_spec = fixture.input(); - caller_spec.spec.timeout = std::time::Duration::from_secs(60); - assert_eq!( - rejected(fixture.accept(caller_spec)), - ResourceActionRejection::SpecMismatch - ); - let mut other_task = fixture.input(); - other_task.task.task_id = TaskId::new(); - assert_eq!( - rejected(fixture.accept(other_task)), - ResourceActionRejection::IdentityConflict - ); - let mut other_thread = fixture.input(); - other_thread.authority.supervisor.thread = crate::domain::ThreadId(uuid::Uuid::now_v7()); - assert_eq!( - rejected(fixture.accept(other_thread)), - ResourceActionRejection::NotCurrentSupervisor - ); - let mut old_assignment = fixture.input(); - old_assignment.authority.assignment_revision = - crate::resource::AssignmentRevision::new(fixture.authority.assignment_revision.get() + 1); - assert_eq!( - rejected(fixture.accept(old_assignment)), - ResourceActionRejection::NotCurrentSupervisor - ); - let mut stale = fixture.input(); - stale.authority.expected_state_revision = - ResourceRevision::new(fixture.authority.expected_state_revision.get() + 1); - assert!(matches!( - rejected(fixture.accept(stale)), - ResourceActionRejection::StaleRevision { .. } - )); - let mut other_action = fixture.input(); - other_action.authority.action_id = ActionId::new(); - assert_eq!( - rejected(fixture.accept(other_action)), - ResourceActionRejection::ActionNotPending - ); - let mut other_trainer = fixture.input(); - other_trainer.observed_background_task = TaskId::new(); - assert_eq!( - rejected(fixture.accept(other_trainer)), - ResourceActionRejection::ActionNotPending - ); - assert_eq!( - watcher_acceptance_counts(&fixture.poll.store, request, task), - [0, 0, 0, 0] - ); - assert_eq!(fixture.receipt_count(), 0); - - // the remote path never serves a supervisor that runs on the authority - let mut co_located = RemoteWatcherFixture::new(); - co_located - .poll - .store - .conn - .execute( - "UPDATE resources SET supervisor_machine = ?1 WHERE id = ?2", - params![ - co_located.poll.authority.as_uuid().to_string(), - co_located.poll.resource.id.as_uuid().to_string() - ], - ) - .unwrap(); - let mut local_input = co_located.input(); - local_input.authority.supervisor.machine = co_located.poll.authority; - assert_eq!( - rejected(co_located.accept(local_input)), - ResourceActionRejection::NotCurrentSupervisor - ); - assert_eq!(co_located.receipt_count(), 0); -} - -#[test] -fn remote_watcher_uses_the_same_baseline_stop_marker_and_leaves_release_to_process_exit() { - let mut fixture = RemoteWatcherFixture::new(); - let command = fixture.poll.command(); - fixture.accept(fixture.input()).unwrap(); - // a queued acceptance is not a running watcher and cannot stop the trainer - assert_eq!( - fixture.poll.poll(command), - ReleaseWatcherPollOutcome::WatcherNotRunning - ); - fixture.start(); - assert_eq!( - fixture.poll.poll(command), - ReleaseWatcherPollOutcome::WaitingForCheckpoint - ); - assert!(fixture.poll.trainer_cancel_marker().is_none()); - - fixture - .poll - .write_generation("generation-after-baseline", 30); - let ReleaseWatcherPollOutcome::StopCommitted { - generation_id, - cancel_requested_at, - } = fixture.poll.poll(command) - else { - panic!("a new checkpoint must commit the exact stop"); - }; - assert_eq!(generation_id, "generation-after-baseline"); - assert_eq!( - fixture.poll.trainer_cancel_marker(), - Some(cancel_requested_at) - ); - // an exact retry reuses the saved decision and marker - assert_eq!( - fixture.poll.poll(command), - ReleaseWatcherPollOutcome::StopCommitted { - generation_id, - cancel_requested_at, - } - ); - - // the stop marker alone does not release the loan while the trainer runs - assert!( - fixture - .poll - .store - .complete_release_for_authority( - fixture.poll.authority, - fixture.poll.resource.id, - fixture.poll.notice.action_id, - fixture.poll.notice.state_revision, - ) - .is_err() - ); - let snapshot = fixture - .poll - .store - .resource_snapshots_for_authority(fixture.poll.authority) - .unwrap() - .remove(0); - assert!(matches!( - snapshot.loan.unwrap().state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { .. } - } - )); -} - -#[test] -fn accepted_remote_watcher_keeps_polling_after_its_supervisor_is_replaced() { - for local_replacement in [false, true] { - let mut fixture = RemoteWatcherFixture::new(); - let command = fixture.poll.command(); - fixture.accept(fixture.input()).unwrap(); - fixture.start(); - let replacement = if local_replacement { - fixture.poll.authority - } else { - machine_other_than(fixture.authority.supervisor.machine) - }; - fixture.poll.replace_supervisor(replacement); - fixture - .poll - .write_generation("generation-after-baseline", 30); - - // the saved receipt, not the current assignment, owns the accepted watcher - let ReleaseWatcherPollOutcome::StopCommitted { - generation_id, - cancel_requested_at, - } = fixture.poll.poll(command) - else { - panic!("the accepted watcher must keep its stop authority ({local_replacement})"); - }; - assert_eq!( - fixture.poll.trainer_cancel_marker(), - Some(cancel_requested_at) - ); - assert_eq!( - fixture.poll.poll(command), - ReleaseWatcherPollOutcome::StopCommitted { - generation_id, - cancel_requested_at, - } - ); - // an exact acceptance retry observes the saved watcher and inserts nothing - assert!(matches!( - fixture.accept(fixture.input()).unwrap().acceptance, - ActionTaskAcceptance::Existing { .. } - )); - assert_eq!(fixture.receipt_count(), 1); - } -} - -#[test] -fn remote_watcher_acceptance_after_replacement_requires_the_current_assignment() { - use crate::resource::bound_action::ResourceActionRejection; - - let mut fixture = RemoteWatcherFixture::new(); - fixture - .poll - .replace_supervisor(machine_other_than(fixture.authority.supervisor.machine)); - - // an unaccepted watcher named by the replaced assignment is not a new owner - assert_eq!( - rejected(fixture.accept(fixture.input())), - ResourceActionRejection::NotCurrentSupervisor - ); - assert_eq!(fixture.receipt_count(), 0); -} - -#[test] -fn remote_watcher_with_a_missing_or_changed_receipt_is_refused_without_a_marker() { - type Tamper = fn(&RemoteWatcherFixture); - let cases: [(&str, Tamper); 4] = [ - ("missing receipt", |fixture| { - fixture - .poll - .store - .conn - .execute("DELETE FROM resource_action_task_receipts", []) - .unwrap(); - }), - ("receipt names the authority as supervisor", |fixture| { - fixture.rewrite_receipt(|receipt| { - receipt.authority.supervisor.machine = fixture.poll.authority; - }); - }), - ("receipt names another command", |fixture| { - fixture.rewrite_receipt(|receipt| { - receipt.normalized_spec_sha256 = normalized_spec_sha256(&spec()).unwrap(); - }); - }), - ("receipt names another revision", |fixture| { - fixture.rewrite_receipt(|receipt| { - receipt.authority.expected_state_revision = - ResourceRevision::new(receipt.authority.expected_state_revision.get() + 1); - }); - }), - ]; - for (name, tamper) in cases { - let mut fixture = RemoteWatcherFixture::new(); - let command = fixture.poll.command(); - fixture.accept(fixture.input()).unwrap(); - fixture.start(); - fixture - .poll - .write_generation("generation-after-baseline", 30); - tamper(&fixture); - - assert_eq!( - fixture.poll.poll(command), - attention(ReleaseWatcherPollAttention::WatcherIdentityConflict), - "{name}" - ); - assert!(fixture.poll.trainer_cancel_marker().is_none(), "{name}"); - } -} diff --git a/src/store/resource/tests/release_watcher_acceptance.rs b/src/store/resource/tests/release_watcher_acceptance.rs deleted file mode 100644 index 2f2f14d..0000000 --- a/src/store/resource/tests/release_watcher_acceptance.rs +++ /dev/null @@ -1,425 +0,0 @@ -//! Release watcher task acceptance tests - -use super::fixtures::{ - accept_watcher, open_release_for_test, prepare_release_checkpoint_baseline, - release_watcher_intent, remote_task, resource, spec, watcher_acceptance_counts, - watcher_acceptance_input, watcher_spec, watcher_task_and_callback, -}; -use crate::domain::{ProcessStatus, TaskId, TaskState, ThreadId}; -use crate::events::{EventPayload, TaskEvent}; -use crate::machine::MachineId; -use crate::resource::store::{OpenReleaseLoanResult, ReleaseWatcherAcceptance}; -use crate::resource::{ - ActionId, LoanPhase, LoanState, ReleaseWatcherIntent, ReleaseWatcherTaskId, SupervisorAddress, -}; -use crate::spec::NormalizedSpec; -use crate::store::{ExecutorIdentity, Store}; -use crate::submission::{CallbackContext, CallbackExecutable, RequestId, SubmissionState}; -use serde_json::json; -use std::path::PathBuf; -use tempfile::tempdir; -use uuid::Uuid; - -#[test] -fn release_watcher_acceptance_inserts_fixed_task_route_identity_and_event() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - let spec = watcher_spec(&resource, &intent); - prepare_release_checkpoint_baseline(&mut store, authority, &resource, &intent); - let (row, callback) = watcher_task_and_callback(&intent, &spec); - - let accepted = accept_watcher( - &mut store, - &resource, - intent.clone(), - &row, - &spec, - &callback, - ) - .unwrap(); - assert_eq!( - accepted, - ReleaseWatcherAcceptance::Inserted { task: row.id } - ); - assert_eq!( - store.get_task(row.id).unwrap().unwrap().status(), - ProcessStatus::Queued - ); - - let route = store - .origin_route_by_request(intent.request_id) - .unwrap() - .unwrap(); - assert_eq!(route.task, row.id); - assert_eq!(route.origin_machine, authority); - assert_eq!(route.execution_machine, authority); - assert_eq!(route.callback, callback); - assert_eq!( - serde_json::to_value(route.current_spec().unwrap()).unwrap(), - serde_json::to_value(&spec).unwrap() - ); - assert!(matches!(route.submission, SubmissionState::Accepted)); - let ExecutorIdentity::Accepted(identity) = store.executor_identity(row.id).unwrap().unwrap() - else { - panic!("watcher task must have an accepted executor identity"); - }; - assert_eq!(identity.origin_machine, authority); - assert_eq!(identity.execution_machine, authority); - assert_eq!( - serde_json::to_value(identity.current_spec().unwrap()).unwrap(), - serde_json::to_value(&spec).unwrap() - ); - assert_eq!(identity.state, ProcessStatus::Queued); - - let event_json: String = store - .conn - .query_row( - "SELECT event_json FROM executor_outbox WHERE task_id=?1 AND seq=1", - [row.id.to_string()], - |entry| entry.get(0), - ) - .unwrap(); - let event: TaskEvent = serde_json::from_str(&event_json).unwrap(); - assert_eq!(event.task, row.id); - assert_eq!(event.origin_machine, authority); - assert_eq!(event.execution_machine, authority); - assert_eq!( - event.payload, - EventPayload::State { - status: ProcessStatus::Queued - } - ); - assert_eq!( - watcher_acceptance_counts(&store, intent.request_id, row.id), - [1, 1, 1, 1] - ); - - let snapshot = store - .resource_snapshots_for_authority(authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == resource.id) - .unwrap(); - assert!(matches!( - snapshot.loan.unwrap().state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { - watcher_intent: Some(saved), - .. - } - } if saved == intent - )); -} - -#[test] -fn release_watcher_acceptance_exact_retry_survives_reopen_and_old_bind_retry() { - let directory = tempdir().unwrap(); - let database = directory.path().join("db"); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, intent, row, callback, spec) = { - let mut store = Store::open(&database).unwrap(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - let spec = watcher_spec(&resource, &intent); - prepare_release_checkpoint_baseline(&mut store, authority, &resource, &intent); - let (row, callback) = watcher_task_and_callback(&intent, &spec); - assert!(matches!( - accept_watcher( - &mut store, - &resource, - intent.clone(), - &row, - &spec, - &callback - ) - .unwrap(), - ReleaseWatcherAcceptance::Inserted { .. } - )); - (resource, intent, row, callback, spec) - }; - - let mut reopened = Store::open(&database).unwrap(); - let before = watcher_acceptance_counts(&reopened, intent.request_id, row.id); - let retry = accept_watcher( - &mut reopened, - &resource, - intent.clone(), - &row, - &spec, - &callback, - ) - .unwrap(); - assert_eq!( - retry, - ReleaseWatcherAcceptance::Existing { - task: row.id, - state: TaskState::Queued - } - ); - assert_eq!( - watcher_acceptance_counts(&reopened, intent.request_id, row.id), - before - ); - assert_eq!(before, [1, 1, 1, 1]); - assert_eq!( - reopened - .bind_release_watcher_for_authority(authority, resource.id, intent.clone()) - .unwrap(), - intent - ); -} - -#[test] -fn release_watcher_acceptance_rejects_changed_content_and_owners() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - let spec = watcher_spec(&resource, &intent); - prepare_release_checkpoint_baseline(&mut store, authority, &resource, &intent); - let (row, callback) = watcher_task_and_callback(&intent, &spec); - accept_watcher( - &mut store, - &resource, - intent.clone(), - &row, - &spec, - &callback, - ) - .unwrap(); - - let different_action = ReleaseWatcherIntent { - action_id: ActionId::new(), - ..intent.clone() - }; - let different_request = ReleaseWatcherIntent { - request_id: RequestId::new(), - ..intent.clone() - }; - let different_task = ReleaseWatcherIntent { - watcher_task_id: ReleaseWatcherTaskId::new(TaskId::new()), - ..intent.clone() - }; - for changed in [different_action, different_request, different_task] { - assert!(accept_watcher(&mut store, &resource, changed, &row, &spec, &callback).is_err()); - } - - let mut changed_spec_value = serde_json::to_value(&spec).unwrap(); - changed_spec_value["workload"]["command"][1] = json!("changed"); - let changed_spec: NormalizedSpec = serde_json::from_value(changed_spec_value).unwrap(); - assert!( - accept_watcher( - &mut store, - &resource, - intent.clone(), - &row, - &changed_spec, - &callback, - ) - .is_err() - ); - let changed_digest_intent = ReleaseWatcherIntent { - normalized_spec_sha256: crate::submission::normalized_spec_sha256(&changed_spec).unwrap(), - ..intent.clone() - }; - let changed_row = remote_task(row.id, &changed_spec); - let changed_callback = CallbackContext { - env: changed_row.env.clone(), - cwd: changed_row.cwd.clone(), - codex: CallbackExecutable::available(PathBuf::from("/bin/echo")), - }; - assert!( - accept_watcher( - &mut store, - &resource, - changed_digest_intent, - &changed_row, - &changed_spec, - &changed_callback, - ) - .is_err() - ); - - let mut changed_callback = callback.clone(); - changed_callback.codex = CallbackExecutable::available(PathBuf::from("/bin/sh")); - assert!( - accept_watcher( - &mut store, - &resource, - intent.clone(), - &row, - &spec, - &changed_callback, - ) - .is_err() - ); - let mut wrong_owner_input = - watcher_acceptance_input(&resource, intent.clone(), &row, &spec, &callback); - wrong_owner_input.authority_machine = MachineId::new(); - assert!( - store - .accept_release_watcher_for_authority(wrong_owner_input) - .is_err() - ); - let changed_supervisor = SupervisorAddress { - machine: authority, - thread: ThreadId(Uuid::now_v7()), - }; - let mut changed_supervisor_resource = resource.clone(); - changed_supervisor_resource.supervisor = changed_supervisor; - assert!( - accept_watcher( - &mut store, - &changed_supervisor_resource, - intent, - &row, - &spec, - &callback, - ) - .is_err() - ); -} - -#[test] -fn remote_release_watcher_supervisor_is_typed_and_writes_no_task_records() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let remote_supervisor = MachineId::new(); - let background_task = TaskId::new(); - let spec = spec(); - let mut resource = resource(authority); - resource.supervisor = SupervisorAddress { - machine: remote_supervisor, - thread: spec.thread, - }; - resource.registered_background_task = Some(background_task); - store.register_resource(authority, &resource).unwrap(); - store - .insert_task(&remote_task(background_task, &spec)) - .unwrap(); - store - .cas_status( - background_task, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - spec.clone(), - ) - .unwrap(); - let OpenReleaseLoanResult::Opened { notice, .. } = store - .open_release_loan_for_authority(authority, resource.id, resource.state_revision) - .unwrap() - else { - panic!("fixture must create a new release action"); - }; - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - let spec = watcher_spec(&resource, &intent); - let (row, callback) = watcher_task_and_callback(&intent, &spec); - - let result = accept_watcher( - &mut store, - &resource, - intent.clone(), - &row, - &spec, - &callback, - ) - .unwrap(); - assert_eq!( - result, - ReleaseWatcherAcceptance::UnsupportedRemoteSupervisor { - authority_machine: authority, - supervisor: resource.supervisor, - } - ); - assert_eq!( - watcher_acceptance_counts(&store, intent.request_id, row.id), - [0, 0, 0, 0] - ); - let snapshot = store - .resource_snapshots_for_authority(authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == resource.id) - .unwrap(); - assert!(matches!( - snapshot.loan.unwrap().state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { - watcher_intent: None, - .. - } - } - )); -} - -#[test] -fn release_watcher_acceptance_rolls_back_binding_and_task_records_on_event_failure() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - let spec = watcher_spec(&resource, &intent); - prepare_release_checkpoint_baseline(&mut store, authority, &resource, &intent); - let (row, callback) = watcher_task_and_callback(&intent, &spec); - store - .conn - .execute_batch(&format!( - "CREATE TRIGGER fail_release_watcher_event - BEFORE INSERT ON executor_outbox - WHEN NEW.task_id='{}' - BEGIN SELECT RAISE(ABORT, 'test event failure'); END;", - row.id - )) - .unwrap(); - - assert!( - accept_watcher( - &mut store, - &resource, - intent.clone(), - &row, - &spec, - &callback - ) - .is_err() - ); - assert_eq!( - watcher_acceptance_counts(&store, intent.request_id, row.id), - [0, 0, 0, 0] - ); - let snapshot = store - .resource_snapshots_for_authority(authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == resource.id) - .unwrap(); - assert!(matches!( - snapshot.loan.unwrap().state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { - watcher_intent: Some(saved), - .. - } - } if saved == intent - )); -} diff --git a/src/store/resource/tests/release_watcher_binding.rs b/src/store/resource/tests/release_watcher_binding.rs deleted file mode 100644 index b07d004..0000000 --- a/src/store/resource/tests/release_watcher_binding.rs +++ /dev/null @@ -1,427 +0,0 @@ -//! Release notice facade and release watcher binding tests - -use super::fixtures::{open_release_for_test, release_watcher_intent, remote_task, resource, spec}; -use crate::domain::{ProcessStatus, TaskId, ThreadId}; -use crate::machine::MachineId; -use crate::resource::store::{OpenReleaseLoanResult, ResourceStoreError}; -use crate::resource::{ - ActionId, AssignmentRevision, DeliveryAttemptId, LoanPhase, LoanState, ReleaseWatcherIntent, - ReleaseWatcherTaskId, ResourceRevision, SupervisorAddress, SupervisorNoticeDelivery, -}; -use crate::store::Store; -use crate::submission::RequestId; -use serde_json::json; -use std::path::PathBuf; -use tempfile::tempdir; -use uuid::Uuid; - -#[test] -fn release_notice_facade_operations_share_the_store_connection() { - let directory = tempdir().unwrap(); - let database = directory.path().join("db"); - let mut store = Store::open(&database).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let mut resource = resource(authority); - resource.registered_background_task = Some(background_task); - - store.register_resource(authority, &resource).unwrap(); - store - .insert_task(&remote_task(background_task, &spec())) - .unwrap(); - assert!( - store - .cas_status( - background_task, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .is_some() - ); - store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - resource.id, - MachineId::new(), - spec(), - ) - .unwrap(); - - let OpenReleaseLoanResult::Opened { notice, .. } = store - .open_release_loan_for_authority(authority, resource.id, resource.state_revision) - .unwrap() - else { - panic!("first facade call must open a release loan"); - }; - assert_eq!( - store.supervisor_notice(notice.id).unwrap(), - Some(notice.clone()) - ); - assert_eq!( - store.pending_supervisor_notices().unwrap(), - vec![notice.clone()] - ); - - let destination = SupervisorAddress { - machine: MachineId::new(), - thread: ThreadId(Uuid::now_v7()), - }; - let retargeted = store - .retarget_supervisor_notice( - notice.id, - notice.assignment_revision, - destination, - AssignmentRevision::new(1), - ) - .unwrap(); - assert_eq!(retargeted.destination, destination); - - let first_attempt = DeliveryAttemptId::new(); - assert_eq!( - store - .reserve_supervisor_notice_attempt(notice.id, first_attempt) - .unwrap() - .id, - notice.id - ); - assert_eq!( - store - .recover_sending_supervisor_notices() - .unwrap() - .iter() - .map(|notice| notice.id) - .collect::>(), - vec![notice.id] - ); - - let second_attempt = DeliveryAttemptId::new(); - store - .reserve_supervisor_notice_attempt(notice.id, second_attempt) - .unwrap(); - let settled = store - .settle_supervisor_notice_attempt(notice.id, second_attempt, Ok(())) - .unwrap(); - assert_eq!(settled.id, notice.id); - assert!(matches!( - settled.delivery, - SupervisorNoticeDelivery::Delivered { .. } - )); - assert_eq!(store.supervisor_notice(notice.id).unwrap(), Some(settled)); -} - -#[test] -fn release_watcher_binding_returns_the_saved_intent_on_exact_retry() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, loan, notice) = open_release_for_test(&mut store, authority, background_task); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - - let first = store - .bind_release_watcher_for_authority(authority, resource.id, intent.clone()) - .unwrap(); - let reserved_task = remote_task(intent.watcher_task_id.as_task_id(), &spec()); - assert!(matches!( - store.insert_local_task( - &reserved_task, - &spec(), - authority, - crate::submission::RequestId::new(), - PathBuf::from("/bin/echo").into(), - ), - Err(crate::error::AppError::ClusterTaskConflict { task }) - if task == reserved_task.id - )); - let retry = store - .bind_release_watcher_for_authority(authority, resource.id, intent.clone()) - .unwrap(); - - assert_eq!(first, intent); - assert_eq!(retry, first); - let snapshot = store - .resource_snapshots_for_authority(authority) - .unwrap() - .into_iter() - .next() - .unwrap(); - assert_eq!(snapshot.resource.state_revision, notice.state_revision); - assert_eq!( - snapshot.resource.registered_background_task, - Some(background_task) - ); - let saved_loan = snapshot.loan.unwrap(); - assert_eq!(saved_loan.id, loan.id); - assert!(matches!( - saved_loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { - action_id, - observed_background_task, - watcher_intent: Some(saved), - } - } if action_id == notice.action_id - && observed_background_task == background_task - && saved == intent - )); -} - -#[test] -fn release_watcher_binding_rejects_conflicting_identity_and_stale_action_data() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - store - .bind_release_watcher_for_authority(authority, resource.id, intent.clone()) - .unwrap(); - - let different_watcher = ReleaseWatcherIntent { - watcher_task_id: ReleaseWatcherTaskId::new(TaskId::new()), - ..intent.clone() - }; - let different_action = ReleaseWatcherIntent { - action_id: ActionId::new(), - ..intent.clone() - }; - let different_background_task = ReleaseWatcherIntent { - observed_background_task: TaskId::new(), - ..intent.clone() - }; - let stale_revision = ReleaseWatcherIntent { - state_revision: ResourceRevision::new(notice.state_revision.get() - 1), - ..intent.clone() - }; - let different_request = ReleaseWatcherIntent { - request_id: RequestId::new(), - ..intent.clone() - }; - let mut changed_spec = serde_json::to_value(spec()).unwrap(); - changed_spec["workload"]["command"][1] = json!("different"); - let different_digest = ReleaseWatcherIntent { - normalized_spec_sha256: crate::submission::normalized_spec_sha256( - &serde_json::from_value(changed_spec).unwrap(), - ) - .unwrap(), - ..intent.clone() - }; - - for conflicting in [ - different_watcher, - different_action, - different_background_task, - stale_revision, - different_request, - different_digest, - ] { - assert!(matches!( - store.bind_release_watcher_for_authority(authority, resource.id, conflicting,), - Err(ResourceStoreError::Conflict(_)) - )); - } - - let snapshot = store - .resource_snapshots_for_authority(authority) - .unwrap() - .into_iter() - .next() - .unwrap(); - let saved_loan = snapshot.loan.unwrap(); - assert!(matches!( - saved_loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { - watcher_intent: Some(saved), - .. - } - } if saved == intent - )); -} - -#[test] -fn release_watcher_binding_rejects_wrong_authority_and_reused_identities() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let first_background_task = TaskId::new(); - let (first_resource, _, first_notice) = - open_release_for_test(&mut store, authority, first_background_task); - let first_intent = release_watcher_intent( - first_resource.id, - &first_notice, - first_background_task, - TaskId::new(), - ); - - assert!(matches!( - store.bind_release_watcher_for_authority( - MachineId::new(), - first_resource.id, - first_intent.clone(), - ), - Err(ResourceStoreError::WrongAuthority { .. }) - )); - store - .bind_release_watcher_for_authority(authority, first_resource.id, first_intent.clone()) - .unwrap(); - - let second_background_task = TaskId::new(); - let (second_resource, _, second_notice) = - open_release_for_test(&mut store, authority, second_background_task); - let same_watcher_task = ReleaseWatcherIntent { - watcher_task_id: first_intent.watcher_task_id, - ..release_watcher_intent( - second_resource.id, - &second_notice, - second_background_task, - TaskId::new(), - ) - }; - let same_request = ReleaseWatcherIntent { - request_id: first_intent.request_id, - ..release_watcher_intent( - second_resource.id, - &second_notice, - second_background_task, - TaskId::new(), - ) - }; - - assert!(matches!( - store.bind_release_watcher_for_authority(authority, second_resource.id, same_watcher_task,), - Err(ResourceStoreError::Conflict(_)) - )); - assert!(matches!( - store.bind_release_watcher_for_authority(authority, second_resource.id, same_request), - Err(ResourceStoreError::Conflict(_)) - )); - - let second_watcher_task = TaskId::new(); - let same_request_as_task = ReleaseWatcherIntent { - watcher_task_id: ReleaseWatcherTaskId::new(second_watcher_task), - request_id: RequestId(second_watcher_task.0), - ..release_watcher_intent( - second_resource.id, - &second_notice, - second_background_task, - TaskId::new(), - ) - }; - assert!(matches!( - store.bind_release_watcher_for_authority( - authority, - second_resource.id, - same_request_as_task, - ), - Err(ResourceStoreError::Conflict(_)) - )); -} - -#[test] -fn release_watcher_binding_rejects_request_id_used_by_a_task_route() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - store - .conn - .execute( - "INSERT INTO origin_routes (request_id, task_id, execution_machine, spec_json, route_json) - VALUES (?1, ?2, ?3, ?4, ?5)", - rusqlite::params![ - intent.request_id.0.to_string(), - TaskId::new().to_string(), - authority.as_uuid().to_string(), - "{}", - "{}", - ], - ) - .unwrap(); - - assert!(matches!( - store.bind_release_watcher_for_authority(authority, resource.id, intent.clone()), - Err(ResourceStoreError::Conflict(_)) - )); - - store - .conn - .execute( - "DELETE FROM origin_routes WHERE request_id = ?1", - [intent.request_id.0.to_string()], - ) - .unwrap(); - store - .conn - .execute( - "INSERT INTO origin_routes (request_id, task_id, execution_machine, spec_json, route_json) - VALUES (?1, ?2, ?3, ?4, ?5)", - rusqlite::params![ - RequestId::new().0.to_string(), - intent.watcher_task_id.as_task_id().to_string(), - authority.as_uuid().to_string(), - "{}", - "{}", - ], - ) - .unwrap(); - assert!(matches!( - store.bind_release_watcher_for_authority(authority, resource.id, intent), - Err(ResourceStoreError::Conflict(_)) - )); -} - -#[test] -fn release_watcher_binding_survives_store_reopen_without_changing_release_state() { - let directory = tempdir().unwrap(); - let database = directory.path().join("db"); - let authority = MachineId::new(); - let background_task = TaskId::new(); - let (resource, notice, intent) = { - let mut store = Store::open(&database).unwrap(); - let (resource, _, notice) = open_release_for_test(&mut store, authority, background_task); - let intent = release_watcher_intent(resource.id, ¬ice, background_task, TaskId::new()); - store - .bind_release_watcher_for_authority(authority, resource.id, intent.clone()) - .unwrap(); - (resource, notice, intent) - }; - - let mut reopened = Store::open(&database).unwrap(); - let snapshot = reopened - .resource_snapshots_for_authority(authority) - .unwrap() - .into_iter() - .next() - .unwrap(); - assert_eq!(snapshot.resource.state_revision, notice.state_revision); - assert_eq!( - snapshot.resource.registered_background_task, - Some(background_task) - ); - let saved_loan = snapshot.loan.unwrap(); - assert!(matches!( - saved_loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { - action_id, - observed_background_task, - watcher_intent: Some(saved), - } - } if action_id == notice.action_id - && observed_background_task == background_task - && saved == intent - )); - assert_eq!( - reopened - .bind_release_watcher_for_authority(authority, resource.id, intent.clone()) - .unwrap(), - intent - ); -} diff --git a/src/store/resource/tests/restore.rs b/src/store/resource/tests/restore.rs deleted file mode 100644 index 60e1a78..0000000 --- a/src/store/resource/tests/restore.rs +++ /dev/null @@ -1,1909 +0,0 @@ -//! Supervisor return decisions, fixed-identity restore binding, and restore closure - -use super::fixtures::{ - ServingFixture, TrainerAssociationFixture, accept_and_finish_resource_task, - commit_stopped_release_decision, completion, machine_other_than, release_completion_fixture, - resource, serving_fixture, -}; -use crate::domain::{ - ExitReason, ProcessGroupExitEvidence, ProcessStatus, TaskEnv, TaskId, ThreadId, Workload, -}; -use crate::machine::MachineId; -use crate::resource::ownership_lock::test_support::start_fake_lock_process; -use crate::resource::ownership_lock::{ - OwnershipLockIdentity, OwnershipLockIdentityMismatchReason, OwnershipLockProbe, - OwnershipLockProbeError, probe_segment_ownership_lock, -}; -use crate::resource::store::{ - ReleaseCompletionResult, ResourceStoreError, ResourceTaskCompletionResult, -}; -use crate::resource::{ - ActionId, AssignmentRevision, CommandSpec, IdleBoundaryProof, Loan, LoanClosure, LoanId, - LoanPhase, LoanState, ReleaseCheckpointStopDecision, Resource, ResourceQueueReconcileOutcome, - ResourceRequest, ResourceRequestState, ResourceRevision, ResourceTaskOwnershipRisk, - RestoreAttentionReason, ReturnContext, ReturnDecisionRejection, ReturnExecutionMode, - ReturnLaunch, ReturnWork, SameRunResumeGap, SupervisorActionAuthority, -}; -use crate::spec::NormalizedWorkload; -use crate::store::resource::trainer_lock::TrainerLockReleaseGap; -use crate::store::{ - EndedRestoreResolution, ExecutorIdentity, RestoreReconcileOutcome, ReturnClosure, - ReturnDecisionError, ReturnTaskAcceptance, ReturnTaskAcceptanceInput, ReturnTaskOrigin, Store, -}; -use crate::submission::{ - CallbackExecutable, NormalizedSpecSha256, RequestId, SubmissionState, normalized_spec_sha256, -}; -use rusqlite::{OptionalExtension, params}; -use std::fs; -use std::os::unix::fs::PermissionsExt; -use std::path::PathBuf; -use tempfile::tempdir; -use uuid::Uuid; - -/// Drain the serving fixture's only request so the loan awaits its return decision -pub(super) fn awaiting_return_fixture() -> (ServingFixture, SupervisorActionAuthority) { - let mut fixture = serving_fixture(true, true); - let input = accept_and_finish_resource_task( - &mut fixture, - ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let result = fixture - .store - .reconcile_assigned_resource_task_for_authority(input) - .unwrap(); - let Ok(ResourceTaskCompletionResult::ReturnRequired { loan, notice, .. }) = completion(result) - else { - panic!("the drained queue must reserve the return"); - }; - let authority = SupervisorActionAuthority { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - loan_id: loan.id, - action_id: notice.action_id, - expected_state_revision: notice.state_revision, - supervisor: fixture.resource.supervisor, - assignment_revision: fixture.resource.assignment_revision, - }; - fixture.loan = loan; - fixture.state_revision = notice.state_revision; - (fixture, authority) -} - -/// Invalid authority, its no-resume reason, and the exact expected error -type Rejection = ( - SupervisorActionAuthority, - &'static str, - fn(&ReturnDecisionError) -> bool, -); - -pub(super) fn completed_task(fixture: &ServingFixture) -> TaskId { - fixture.resource.registered_background_task.unwrap() -} - -fn command(fixture: &ServingFixture, argv: &[&str]) -> CommandSpec { - let mut spec = fixture.spec.clone(); - spec.workload = NormalizedWorkload::Task(crate::spec::NormalizedTaskWorkload { - command: crate::invocation::CommandLine::try_from_argv( - argv.iter().map(|argument| (*argument).to_owned()).collect(), - ) - .unwrap(), - }); - CommandSpec::try_from(spec).unwrap() -} - -pub(super) fn evaluation(fixture: &ServingFixture, argv: &[&str]) -> ReturnLaunch { - ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::EvaluationOrNextEpoch { - completed_task: completed_task(fixture), - spec: command(fixture, argv), - }, - } -} - -pub(super) fn launch_input( - authority: SupervisorActionAuthority, - launch: ReturnLaunch, -) -> ReturnTaskAcceptanceInput { - ReturnTaskAcceptanceInput { - authority, - launch, - executor_env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - origin: ReturnTaskOrigin::Local { - callback_codex: CallbackExecutable::available(PathBuf::from("/bin/echo")), - }, - } -} - -pub(super) fn saved_loan(store: &Store, authority: MachineId) -> Option { - store.resource_snapshots_for_authority(authority).unwrap()[0] - .loan - .clone() -} - -pub(super) fn saved_resource(store: &Store, authority: MachineId) -> Resource { - store.resource_snapshots_for_authority(authority).unwrap()[0] - .resource - .clone() -} - -fn assert_still_awaiting_return(fixture: &ServingFixture, authority: &SupervisorActionAuthority) { - assert!(matches!( - saved_loan(&fixture.store, fixture.authority).map(|loan| loan.state), - Some(LoanState::Active { - phase: LoanPhase::AwaitingReturn { action_id, .. } - }) if action_id == authority.action_id - )); - assert_eq!( - saved_resource(&fixture.store, fixture.authority).state_revision, - authority.expected_state_revision - ); -} - -pub(super) fn accept_post_return_request(fixture: &mut ServingFixture) -> ResourceRequest { - fixture - .store - .accept_resource_request( - fixture.authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - MachineId::new(), - fixture.spec.clone(), - ) - .unwrap() -} - -#[test] -fn no_resume_closes_only_the_exact_current_return_action_and_replays_its_receipt() { - let (mut fixture, authority) = awaiting_return_fixture(); - let completed = completed_task(&fixture); - // a request accepted after the reservation waits and cannot supersede the action - let later = accept_post_return_request(&mut fixture); - assert!(matches!( - fixture - .store - .reconcile_resource_queue_for_authority(fixture.authority, fixture.resource.id) - .unwrap(), - ResourceQueueReconcileOutcome::LoanAlreadyActive { .. } - )); - - let mut other_thread = authority; - other_thread.supervisor.thread = ThreadId(Uuid::now_v7()); - let mut stale_assignment = authority; - stale_assignment.assignment_revision = AssignmentRevision::new(7); - let mut stale_revision = authority; - stale_revision.expected_state_revision = ResourceRevision::new(1); - let mut other_action = authority; - other_action.action_id = ActionId::new(); - let mut other_authority = authority; - other_authority.authority_machine = machine_other_than(fixture.authority); - let rejections: [Rejection; 6] = [ - (other_thread, "done", |error| { - matches!(error, ReturnDecisionError::NotCurrentSupervisor) - }), - (stale_assignment, "done", |error| { - matches!(error, ReturnDecisionError::NotCurrentSupervisor) - }), - (stale_revision, "done", |error| { - matches!(error, ReturnDecisionError::StaleRevision { .. }) - }), - (other_action, "done", |error| { - matches!(error, ReturnDecisionError::ActionNotPending { .. }) - }), - (other_authority, "done", |error| { - matches!( - error, - ReturnDecisionError::Resource(ResourceStoreError::WrongAuthority { .. }) - ) - }), - (authority, " ", |error| { - matches!( - error, - ReturnDecisionError::Rejected(ReturnDecisionRejection::EmptyReason) - ) - }), - ]; - for (candidate, reason, expected) in rejections { - let error = fixture - .store - .record_no_resume_for_authority(candidate, reason.into()) - .unwrap_err(); - assert!(expected(&error), "unexpected rejection: {error:?}"); - assert_still_awaiting_return(&fixture, &authority); - } - - let closure = fixture - .store - .record_no_resume_for_authority(authority, "evaluation is not needed".into()) - .unwrap(); - let next_revision = ResourceRevision::new(authority.expected_state_revision.get() + 1); - assert_eq!(closure.state_revision, next_revision); - assert!(matches!( - &closure.loan.state, - LoanState::Closed { - result: LoanClosure::NoResume { - return_context: ReturnContext::AlreadyCompleted { task_id, .. }, - reason, - } - } if *task_id == completed && reason == "evaluation is not needed" - )); - let resource = saved_resource(&fixture.store, fixture.authority); - assert_eq!(resource.state_revision, next_revision); - assert_eq!(resource.registered_background_task, None); - assert!(saved_loan(&fixture.store, fixture.authority).is_none()); - - let replay = fixture - .store - .record_no_resume_for_authority(authority, "evaluation is not needed".into()) - .unwrap(); - assert_eq!(replay, closure); - assert!(matches!( - fixture - .store - .record_no_resume_for_authority(authority, "another reason".into()), - Err(ReturnDecisionError::ConflictingRetry { .. }) - )); - assert!(matches!( - fixture.store.accept_return_task_for_authority(launch_input( - authority, - evaluation(&fixture, &["/bin/echo", "evaluate"]) - )), - Err(ReturnDecisionError::ConflictingRetry { .. }) - )); - assert_eq!( - saved_resource(&fixture.store, fixture.authority).state_revision, - next_revision - ); - - // after closure the later request uses the normal next-loan rules; the - // explicit no-resume closure is the saved proof that nothing holds the GPU - assert!(matches!( - fixture - .store - .reconcile_resource_queue_for_authority(fixture.authority, fixture.resource.id) - .unwrap(), - ResourceQueueReconcileOutcome::IdleServing { - loan, - request, - proof: IdleBoundaryProof::SupervisorNoResume { loan_id }, - } if request.request_id == later.request_id - && loan_id == closure.loan.id - && matches!( - &loan.state, - LoanState::Active { - phase: LoanPhase::Serving { - return_context: ReturnContext::Idle, - .. - } - } - ) - )); -} - -#[test] -fn return_launch_binds_one_fixed_task_with_its_callback_route_and_replays_it() { - let (fixture, authority) = awaiting_return_fixture(); - let mut fixture = fixture; - let completed = completed_task(&fixture); - - // choices that do not fit the completed context write nothing - let mut other_thread_spec = fixture.spec.clone(); - other_thread_spec.thread = ThreadId(Uuid::now_v7()); - // a script entry point hides its interpreter and code from the contract - let script = fixture.directory.path().join("evaluate"); - fs::write(&script, "#!/bin/sh\nnohup trainer &\n").unwrap(); - fs::set_permissions(&script, fs::Permissions::from_mode(0o755)).unwrap(); - for (launch, expected) in [ - ( - ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::NewBackgroundWork { - spec: command(&fixture, &["/bin/echo", "new"]), - }, - }, - ReturnDecisionRejection::NewWorkRequiresIdleContext, - ), - ( - ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::SameRunResume { - stopped_task: completed, - recovery_ref: "generation".into(), - }, - }, - ReturnDecisionRejection::ResumeRequiresStoppedContext, - ), - ( - ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::EvaluationOrNextEpoch { - completed_task: TaskId::new(), - spec: command(&fixture, &["/bin/echo", "evaluate"]), - }, - }, - ReturnDecisionRejection::EvaluationRequiresCompletedContext, - ), - ( - ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::EvaluationOrNextEpoch { - completed_task: completed, - spec: CommandSpec::try_from(other_thread_spec.clone()).unwrap(), - }, - }, - ReturnDecisionRejection::ThreadMismatch, - ), - ( - evaluation(&fixture, &["/bin/sh", "-c", "nohup trainer &"]), - ReturnDecisionRejection::UnsupportedCommandOwnership { - risk: ResourceTaskOwnershipRisk::ShellWrapper, - }, - ), - ( - evaluation( - &fixture, - &["/usr/bin/env", "docker", "run", "-d", "trainer"], - ), - ReturnDecisionRejection::UnsupportedCommandOwnership { - risk: ResourceTaskOwnershipRisk::ProgramLauncher, - }, - ), - ( - evaluation(&fixture, &["python3", "-m", "ops.evaluate_checkpoint"]), - ReturnDecisionRejection::UnsupportedCommandOwnership { - risk: ResourceTaskOwnershipRisk::Interpreter, - }, - ), - ( - evaluation(&fixture, &[script.to_str().unwrap()]), - ReturnDecisionRejection::UnsupportedCommandOwnership { - risk: ResourceTaskOwnershipRisk::ScriptEntryPoint, - }, - ), - ] { - let task_id = launch.task_id; - let error = fixture - .store - .accept_return_task_for_authority(launch_input(authority, launch)) - .unwrap_err(); - assert!( - matches!(&error, ReturnDecisionError::Rejected(rejection) if *rejection == expected), - "unexpected rejection: {error:?}" - ); - assert!(fixture.store.get_task(task_id).unwrap().is_none()); - assert_still_awaiting_return(&fixture, &authority); - } - - let launch = evaluation(&fixture, &["/bin/echo", "evaluate"]); - let (request_id, task_id) = (launch.request_id, launch.task_id); - let ReturnTaskAcceptance::Inserted { - loan, - task, - state_revision, - } = fixture - .store - .accept_return_task_for_authority(launch_input(authority, launch.clone())) - .unwrap() - else { - panic!("the first exact launch must insert its task"); - }; - assert_eq!(task, task_id); - assert_eq!( - state_revision, - ResourceRevision::new(authority.expected_state_revision.get() + 1) - ); - assert!(matches!( - &loan.state, - LoanState::Active { - phase: LoanPhase::Restoring { - action_id, - return_context: ReturnContext::AlreadyCompleted { task_id: context_task, .. }, - resume_task_id, - } - } if *action_id == authority.action_id - && *context_task == completed - && *resume_task_id == task_id - )); - assert_eq!(saved_loan(&fixture.store, fixture.authority), Some(*loan)); - // the prior run stays registered until the return task has a confirmed start - assert_eq!( - saved_resource(&fixture.store, fixture.authority).registered_background_task, - Some(completed) - ); - - let row = fixture.store.get_task(task_id).unwrap().unwrap(); - assert_eq!(row.status(), ProcessStatus::Queued); - assert_eq!(row.thread, authority.supervisor.thread); - let route = crate::store::identity::origin_route_by_request_on(&fixture.store.conn, request_id) - .unwrap() - .unwrap(); - assert_eq!(route.task, task_id); - assert_eq!(route.origin_machine, fixture.authority); - assert_eq!(route.execution_machine, fixture.authority); - assert_eq!(route.thread, authority.supervisor.thread); - assert!(matches!(route.submission, SubmissionState::Accepted)); - assert!(matches!( - crate::store::identity::executor_identity_on(&fixture.store.conn, task_id).unwrap(), - Some(ExecutorIdentity::Accepted(record)) - if record.origin_machine == fixture.authority - && record.execution_machine == fixture.authority - && record.state == ProcessStatus::Queued - )); - assert!( - crate::store::initial_queued_event_matches_on( - &fixture.store.conn, - task_id, - fixture.authority, - fixture.authority, - ) - .unwrap() - ); - - // an exact retry observes the saved binding without new records - assert_eq!( - fixture - .store - .accept_return_task_for_authority(launch_input(authority, launch.clone())) - .unwrap(), - ReturnTaskAcceptance::Existing { - task: task_id, - state: ProcessStatus::Queued, - } - ); - let mut changed_command = launch.clone(); - changed_command.work = ReturnWork::EvaluationOrNextEpoch { - completed_task: completed, - spec: command(&fixture, &["/bin/echo", "other"]), - }; - let mut changed_task = launch; - changed_task.task_id = TaskId::new(); - for changed in [changed_command, changed_task] { - assert!(matches!( - fixture - .store - .accept_return_task_for_authority(launch_input(authority, changed)), - Err(ReturnDecisionError::ConflictingRetry { .. }) - )); - } - assert!(matches!( - fixture - .store - .record_no_resume_for_authority(authority, "changed mind".into()), - Err(ReturnDecisionError::ConflictingRetry { .. }) - )); - let task_rows: i64 = fixture - .store - .conn - .query_row( - "SELECT COUNT(*) FROM tasks WHERE thread_id = ?1", - [authority.supervisor.thread.to_string()], - |row| row.get(0), - ) - .unwrap(); - // the completed background task, the drained request, and one return task - assert_eq!(task_rows, 3); -} - -#[test] -fn remote_supervisor_return_launch_is_unsupported_before_any_task_record() { - let (mut fixture, mut authority) = awaiting_return_fixture(); - let remote = machine_other_than(fixture.authority); - fixture - .store - .conn - .execute( - "UPDATE resources SET supervisor_machine = ?1 WHERE id = ?2", - params![ - remote.to_string(), - fixture.resource.id.as_uuid().to_string() - ], - ) - .unwrap(); - authority.supervisor.machine = remote; - - let launch = evaluation(&fixture, &["/bin/echo", "evaluate"]); - let task_id = launch.task_id; - assert_eq!( - fixture - .store - .accept_return_task_for_authority(launch_input(authority, launch)) - .unwrap(), - ReturnTaskAcceptance::UnsupportedRemoteSupervisor { - authority_machine: fixture.authority, - supervisor: authority.supervisor, - } - ); - assert!(fixture.store.get_task(task_id).unwrap().is_none()); - assert_still_awaiting_return(&fixture, &authority); - let receipts: i64 = fixture - .store - .conn - .query_row( - "SELECT COUNT(*) FROM resource_return_decisions", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(receipts, 0); -} - -pub(super) fn reconcile_restore(fixture: &mut ServingFixture) -> RestoreReconcileOutcome { - fixture - .store - .reconcile_restoring_loan_for_authority(fixture.authority, fixture.resource.id) - .unwrap() -} - -pub(super) fn start_task(store: &Store, task_id: TaskId) { - store - .cas_status(task_id, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); -} - -fn finish_running_task( - store: &Store, - task_id: TaskId, - reason: ExitReason, - evidence: ProcessGroupExitEvidence, -) { - store - .cas_exit_with_evidence(task_id, ProcessStatus::Running, &reason, evidence) - .unwrap() - .unwrap(); -} - -pub(super) fn restore_closure_basis(store: &Store, action_id: ActionId) -> Option { - store - .conn - .query_row( - "SELECT json_extract(receipt_json, '$.basis.type') - FROM resource_restore_closures WHERE action_id = ?1", - [action_id.as_uuid().to_string()], - |row| row.get(0), - ) - .optional() - .unwrap() -} - -pub(super) fn request_state( - fixture: &ServingFixture, - request_id: RequestId, -) -> Option { - fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|request| request.request_id == request_id) - .map(|request| request.state) -} - -/// Accept one native foreground evaluation and a later queued request -fn native_restore_fixture() -> ( - ServingFixture, - SupervisorActionAuthority, - ReturnLaunch, - ResourceRequest, -) { - let (mut fixture, authority) = awaiting_return_fixture(); - let later = accept_post_return_request(&mut fixture); - let launch = evaluation(&fixture, &["/bin/echo", "evaluate"]); - let ReturnTaskAcceptance::Inserted { state_revision, .. } = fixture - .store - .accept_return_task_for_authority(launch_input(authority, launch.clone())) - .unwrap() - else { - panic!("the first exact launch must insert its task"); - }; - let mut current = authority; - current.expected_state_revision = state_revision; - (fixture, current, launch, later) -} - -fn assert_native_restore_reserved(fixture: &mut ServingFixture, task_id: TaskId, later: RequestId) { - assert!(matches!( - saved_loan(&fixture.store, fixture.authority).map(|loan| loan.state), - Some(LoanState::Active { - phase: LoanPhase::Restoring { resume_task_id, .. } - }) if resume_task_id == task_id - )); - // the foreground task never becomes the registered trainer - assert_ne!( - saved_resource(&fixture.store, fixture.authority).registered_background_task, - Some(task_id) - ); - assert!(matches!( - fixture - .store - .reconcile_resource_queue_for_authority(fixture.authority, fixture.resource.id) - .unwrap(), - ResourceQueueReconcileOutcome::LoanAlreadyActive { .. } - )); - assert_eq!( - request_state(fixture, later), - Some(ResourceRequestState::Queued) - ); -} - -#[test] -fn native_foreground_return_stays_reserved_while_running_and_its_confirmed_end_serves_the_queue() { - let (mut fixture, authority, launch, later) = native_restore_fixture(); - let task_id = launch.task_id; - let prior = completed_task(&fixture); - - // a queued row is not a confirmed start, and the queue stays behind the loan - assert!(matches!( - reconcile_restore(&mut fixture), - RestoreReconcileOutcome::Queued { task_id: queued, .. } if queued == task_id - )); - assert_native_restore_reserved(&mut fixture, task_id, later.request_id); - - // a running foreground task keeps the loan; no release action is opened for it - start_task(&fixture.store, task_id); - assert!(matches!( - reconcile_restore(&mut fixture), - RestoreReconcileOutcome::ForegroundRunning { task_id: running, action_id, .. } - if running == task_id && action_id == authority.action_id - )); - assert_native_restore_reserved(&mut fixture, task_id, later.request_id); - assert_eq!( - saved_resource(&fixture.store, fixture.authority).registered_background_task, - Some(prior) - ); - assert_eq!( - saved_resource(&fixture.store, fixture.authority).state_revision, - authority.expected_state_revision - ); - assert_eq!( - restore_closure_basis(&fixture.store, authority.action_id), - None - ); - - // a restart reads the saved mode instead of classifying the command again - fixture.store = Store::open(&fixture.directory.path().join("db")).unwrap(); - assert!(matches!( - reconcile_restore(&mut fixture), - RestoreReconcileOutcome::ForegroundRunning { task_id: running, .. } if running == task_id - )); - let mut original = authority; - original.expected_state_revision = fixture.state_revision; - assert_eq!( - fixture - .store - .accept_return_task_for_authority(launch_input(original, launch.clone())) - .unwrap(), - ReturnTaskAcceptance::Existing { - task: task_id, - state: ProcessStatus::Running, - } - ); - assert!(matches!( - fixture - .store - .resolve_ended_restore_for_authority(EndedRestoreResolution { - authority, - task_id, - reason: "still running".into(), - }), - Err(ReturnDecisionError::RestoreNotEnded { .. }) - )); - - finish_running_task( - &fixture.store, - task_id, - ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let RestoreReconcileOutcome::ForegroundEnded { - closure, - task_id: ended, - } = reconcile_restore(&mut fixture) - else { - panic!("a successful confirmed foreground end must close the loan"); - }; - assert_eq!(ended, task_id); - assert_eq!( - closure.state_revision, - ResourceRevision::new(authority.expected_state_revision.get() + 1) - ); - assert!(matches!( - closure.loan.state, - LoanState::Closed { - result: LoanClosure::ForegroundReturnEnded { - task_id: ended, - outcome: ExitReason::Exit { code: 0 }, - .. - } - } if ended == task_id - )); - assert_eq!( - restore_closure_basis(&fixture.store, authority.action_id).as_deref(), - Some("foreground_ended") - ); - assert_eq!( - saved_resource(&fixture.store, fixture.authority).registered_background_task, - None - ); - assert_eq!( - reconcile_restore(&mut fixture), - RestoreReconcileOutcome::NotRestoring - ); - assert_eq!( - fixture - .store - .accept_return_task_for_authority(launch_input(original, launch)) - .unwrap(), - ReturnTaskAcceptance::Existing { - task: task_id, - state: ProcessStatus::Succeeded, - } - ); - - // the post-return request is served from the foreground-end idle boundary - let ResourceQueueReconcileOutcome::IdleServing { - loan, - request, - proof, - } = fixture - .store - .reconcile_resource_queue_for_authority(fixture.authority, fixture.resource.id) - .unwrap() - else { - panic!("the confirmed foreground end must let the queue continue"); - }; - assert_ne!(loan.id, authority.loan_id); - assert_eq!(request.request_id, later.request_id); - assert_eq!( - proof, - IdleBoundaryProof::ForegroundReturnEnded { - loan_id: authority.loan_id, - task_id, - } - ); -} - -#[test] -fn native_foreground_return_without_a_successful_confirmed_end_stays_reserved() { - type Finish = fn(&Store, TaskId); - let failed: Finish = |store, task_id| { - finish_running_task( - store, - task_id, - ExitReason::Exit { code: 1 }, - ProcessGroupExitEvidence::ConfirmedExited, - ); - }; - let unconfirmed: Finish = |store, task_id| { - finish_running_task( - store, - task_id, - ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::Unconfirmed, - ); - }; - let lost: Finish = |store, task_id| { - store - .cas_status(task_id, ProcessStatus::Running, ProcessStatus::Lost) - .unwrap() - .unwrap(); - }; - let changed_identity: Finish = |store, task_id| { - finish_running_task( - store, - task_id, - ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ); - store - .conn - .execute( - "UPDATE tasks SET thread_id = ?1 WHERE id = ?2", - params![Uuid::now_v7().to_string(), task_id.to_string()], - ) - .unwrap(); - }; - type Expected = fn(&RestoreAttentionReason) -> bool; - type Resolved = fn(&Result) -> bool; - let cases: [(&str, Finish, Expected, Resolved); 4] = [ - ( - "failed", - failed, - |reason| { - *reason - == RestoreAttentionReason::ForegroundEnded { - state: ProcessStatus::Failed, - } - }, - |result| { - matches!( - result, - Ok(ReturnClosure { - loan: Loan { - state: LoanState::Closed { - result: LoanClosure::RestoreEnded { - outcome: ExitReason::Exit { code: 1 }, - .. - } - }, - .. - }, - .. - }) - ) - }, - ), - ( - "unconfirmed exit", - unconfirmed, - |reason| { - *reason - == RestoreAttentionReason::ForegroundExitUnconfirmed { - state: ProcessStatus::Succeeded, - } - }, - |result| { - matches!( - result, - Err(ReturnDecisionError::RestoreReleaseUnproven { .. }) - ) - }, - ), - ( - "lost", - lost, - |reason| *reason == RestoreAttentionReason::Lost, - |result| { - matches!( - result, - Err(ReturnDecisionError::RestoreReleaseUnproven { .. }) - ) - }, - ), - ( - "changed identity", - changed_identity, - |reason| *reason == RestoreAttentionReason::IdentityMismatch, - |result| matches!(result, Err(ReturnDecisionError::IdentityConflict { .. })), - ), - ]; - for (name, finish, expected, resolved) in cases { - let (mut fixture, authority, launch, later) = native_restore_fixture(); - let task_id = launch.task_id; - start_task(&fixture.store, task_id); - finish(&fixture.store, task_id); - - match reconcile_restore(&mut fixture) { - RestoreReconcileOutcome::Attention { - task_id: attention, - reason, - .. - } => { - assert_eq!(attention, task_id, "{name}"); - assert!(expected(&reason), "{name}: {reason:?}"); - } - other => panic!("{name}: expected attention, got {other:?}"), - } - assert_native_restore_reserved(&mut fixture, task_id, later.request_id); - assert_eq!( - restore_closure_basis(&fixture.store, authority.action_id), - None, - "{name}" - ); - - // only an exact supervisor resolution with proven release can close it - let result = fixture - .store - .resolve_ended_restore_for_authority(EndedRestoreResolution { - authority, - task_id, - reason: format!("{name} evaluation"), - }); - assert!(resolved(&result), "{name}: {result:?}"); - if result.is_err() { - assert_native_restore_reserved(&mut fixture, task_id, later.request_id); - continue; - } - assert_eq!( - saved_resource(&fixture.store, fixture.authority).registered_background_task, - None, - "{name}" - ); - assert!( - matches!( - fixture - .store - .reconcile_resource_queue_for_authority(fixture.authority, fixture.resource.id) - .unwrap(), - ResourceQueueReconcileOutcome::IdleServing { request, .. } - if request.request_id == later.request_id - ), - "{name}" - ); - } -} - -pub(super) fn read_return_execution_mode(fixture: &ServingFixture) -> Option { - read_return_execution_mode_for(&fixture.store, fixture.authority, fixture.resource.id) -} - -fn read_return_execution_mode_for( - store: &Store, - authority: MachineId, - resource: crate::resource::ResourceId, -) -> Option { - store - .resource_read_models(authority, Some(resource)) - .unwrap() - .pop() - .unwrap() - .return_execution_mode -} - -#[test] -fn resource_read_model_uses_saved_return_modes_only_while_restoring() { - let (mut awaiting, authority) = awaiting_return_fixture(); - assert_eq!(read_return_execution_mode(&awaiting), None); - awaiting - .store - .record_no_resume_for_authority(authority, "no return task".into()) - .unwrap(); - assert_eq!(read_return_execution_mode(&awaiting), None); - - let (native, _, _, _) = native_restore_fixture(); - assert_eq!( - read_return_execution_mode(&native), - Some(ReturnExecutionMode::NativeForeground) - ); - - let (mut direct, direct_authority, decision) = stopped_return_fixture(); - let input = resume_input( - direct_authority, - direct.task_id, - &decision.selected_checkpoint.generation_id, - ); - let ReturnTaskAcceptance::Inserted { .. } = direct - .store - .accept_return_task_for_authority(input) - .unwrap() - else { - panic!("the direct-segment return task must be accepted"); - }; - assert_eq!( - read_return_execution_mode_for(&direct.store, direct.authority, direct.resource.id), - Some(ReturnExecutionMode::DirectSegmentTrainer) - ); -} - -#[test] -fn resource_read_model_hides_mismatched_return_modes() { - let (stale_action, _, _, _) = native_restore_fixture(); - stale_action - .store - .conn - .execute( - "UPDATE loans - SET state_json = json_set(state_json, '$.phase.action_id', ?1) - WHERE id = ?2", - rusqlite::params![ - ActionId::new().as_uuid().to_string(), - stale_action.loan.id.as_uuid().to_string(), - ], - ) - .unwrap(); - assert_eq!(read_return_execution_mode(&stale_action), None); - - for identity in ["loan", "task"] { - let (fixture, current_authority, launch, _) = native_restore_fixture(); - let (column, replacement) = match identity { - "loan" => ("$.result.loan.id", LoanId::new().as_uuid().to_string()), - "task" => ("$.result.task_id", TaskId::new().to_string()), - _ => unreachable!(), - }; - fixture - .store - .conn - .execute( - &format!( - "UPDATE resource_return_decisions SET receipt_json = json_set(receipt_json, '{column}', ?1) WHERE action_id = ?2" - ), - rusqlite::params![replacement, current_authority.action_id.as_uuid().to_string()], - ) - .unwrap(); - assert_eq!( - read_return_execution_mode(&fixture), - None, - "mismatched {identity} identity must not expose a mode for task {}", - launch.task_id - ); - } -} - -#[test] -fn lost_restore_stays_reserved_and_cannot_be_resolved() { - let (mut fixture, authority) = awaiting_return_fixture(); - let launch = evaluation(&fixture, &["/bin/echo", "evaluate"]); - let task_id = launch.task_id; - let ReturnTaskAcceptance::Inserted { state_revision, .. } = fixture - .store - .accept_return_task_for_authority(launch_input(authority, launch)) - .unwrap() - else { - panic!("the first exact launch must insert its task"); - }; - fixture - .store - .cas_status(task_id, ProcessStatus::Queued, ProcessStatus::Lost) - .unwrap() - .unwrap(); - - assert!(matches!( - fixture - .store - .reconcile_restoring_loan_for_authority(fixture.authority, fixture.resource.id) - .unwrap(), - RestoreReconcileOutcome::Attention { - reason: RestoreAttentionReason::Lost, - .. - } - )); - let mut current = authority; - current.expected_state_revision = state_revision; - assert!(matches!( - fixture - .store - .resolve_ended_restore_for_authority(EndedRestoreResolution { - authority: current, - task_id, - reason: "lost worker".into(), - }), - Err(ReturnDecisionError::RestoreReleaseUnproven { task_id: lost }) if lost == task_id - )); - assert!(matches!( - saved_loan(&fixture.store, fixture.authority).map(|loan| loan.state), - Some(LoanState::Active { - phase: LoanPhase::Restoring { resume_task_id, .. } - }) if resume_task_id == task_id - )); -} - -#[test] -fn restore_that_ended_early_closes_only_by_explicit_supervisor_resolution() { - let (mut fixture, authority) = awaiting_return_fixture(); - let launch = evaluation(&fixture, &["/bin/echo", "evaluate"]); - let task_id = launch.task_id; - let ReturnTaskAcceptance::Inserted { state_revision, .. } = fixture - .store - .accept_return_task_for_authority(launch_input(authority, launch)) - .unwrap() - else { - panic!("the first exact launch must insert its task"); - }; - let mut current = authority; - current.expected_state_revision = state_revision; - let resolution = |authority, reason: &str| EndedRestoreResolution { - authority, - task_id, - reason: reason.into(), - }; - assert!(matches!( - fixture - .store - .resolve_ended_restore_for_authority(resolution(current, "not started")), - Err(ReturnDecisionError::RestoreNotEnded { .. }) - )); - - // the task starts and fails between two observations - fixture - .store - .cas_status(task_id, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - fixture - .store - .cas_exit_with_evidence( - task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 1 }, - ProcessGroupExitEvidence::ConfirmedExited, - ) - .unwrap() - .unwrap(); - assert!(matches!( - fixture - .store - .reconcile_restoring_loan_for_authority(fixture.authority, fixture.resource.id) - .unwrap(), - RestoreReconcileOutcome::Attention { - reason: RestoreAttentionReason::ForegroundEnded { - state: ProcessStatus::Failed - }, - .. - } - )); - - let mut other_thread = current; - other_thread.supervisor.thread = ThreadId(Uuid::now_v7()); - assert!(matches!( - fixture - .store - .resolve_ended_restore_for_authority(resolution(other_thread, "failed")), - Err(ReturnDecisionError::NotCurrentSupervisor) - )); - assert!(matches!( - fixture - .store - .resolve_ended_restore_for_authority(resolution(authority, "failed")), - Err(ReturnDecisionError::StaleRevision { .. }) - )); - - let closure = fixture - .store - .resolve_ended_restore_for_authority(resolution(current, "evaluation failed")) - .unwrap(); - assert!(matches!( - &closure.loan.state, - LoanState::Closed { - result: LoanClosure::RestoreEnded { - task_id: ended, - outcome: ExitReason::Exit { code: 1 }, - reason, - .. - } - } if *ended == task_id && reason == "evaluation failed" - )); - assert_eq!( - saved_resource(&fixture.store, fixture.authority).registered_background_task, - None - ); - assert_eq!( - fixture - .store - .resolve_ended_restore_for_authority(resolution(current, "evaluation failed")) - .unwrap(), - closure - ); - assert!(matches!( - fixture - .store - .resolve_ended_restore_for_authority(resolution(current, "other reason")), - Err(ReturnDecisionError::ConflictingRetry { .. }) - )); -} - -/// Release a stopped trainer with no queued work so its checkpoint context awaits return -pub(super) fn stopped_return_fixture() -> ( - TrainerAssociationFixture, - SupervisorActionAuthority, - ReleaseCheckpointStopDecision, -) { - let (mut fixture, _, request_id, action_id, revision, _) = release_completion_fixture(); - let queued = fixture - .store - .resource_requests(fixture.authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|request| request.request_id == request_id) - .unwrap(); - fixture - .store - .cancel_resource_request_before_activation( - fixture.authority, - queued.request_id, - queued.task_id, - fixture.resource.id, - queued.origin_machine, - ) - .unwrap(); - let (decision, _) = commit_stopped_release_decision(&mut fixture, action_id, revision); - fixture - .finish_registered_task_cancelled_with_evidence(ProcessGroupExitEvidence::ConfirmedExited); - let ReleaseCompletionResult::ReturnRequired { loan, notice } = fixture - .store - .complete_release_for_authority(fixture.authority, fixture.resource.id, action_id, revision) - .unwrap() - else { - panic!("the empty queue must return the verified stopped context"); - }; - let authority = SupervisorActionAuthority { - authority_machine: fixture.authority, - resource_id: fixture.resource.id, - loan_id: loan.id, - action_id: notice.action_id, - expected_state_revision: notice.state_revision, - supervisor: fixture.resource.supervisor, - assignment_revision: fixture.resource.assignment_revision, - }; - (fixture, authority, decision) -} - -pub(super) fn resume_input( - authority: SupervisorActionAuthority, - stopped_task: TaskId, - recovery_ref: &str, -) -> ReturnTaskAcceptanceInput { - ReturnTaskAcceptanceInput { - authority, - launch: ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work: ReturnWork::SameRunResume { - stopped_task, - recovery_ref: recovery_ref.into(), - }, - }, - // a resume keeps the run's saved environment, not the daemon's current one - executor_env: TaskEnv { - path: "/usr/bin".into(), - home: "/nonexistent".into(), - }, - origin: ReturnTaskOrigin::Local { - callback_codex: CallbackExecutable::available(PathBuf::from("/bin/echo")), - }, - } -} - -#[test] -fn same_run_resume_derives_the_saved_run_command_with_its_selected_checkpoint() { - let (mut fixture, authority, decision) = stopped_return_fixture(); - let generation = decision.selected_checkpoint.generation_id.clone(); - - assert!(matches!( - fixture.store.accept_return_task_for_authority(resume_input( - authority, - fixture.task_id, - "other" - )), - Err(ReturnDecisionError::Rejected( - ReturnDecisionRejection::ResumeRequiresStoppedContext - )) - )); - - let input = resume_input(authority, fixture.task_id, &generation); - let task_id = input.launch.task_id; - assert!(matches!( - fixture.store.accept_return_task_for_authority(input).unwrap(), - ReturnTaskAcceptance::Inserted { task, .. } if task == task_id - )); - let original = fixture.store.get_task(fixture.task_id).unwrap().unwrap(); - let resumed = fixture.store.get_task(task_id).unwrap().unwrap(); - let mut expected = fixture.command(); - expected.extend(["--resume".to_owned(), generation]); - let Workload::Task(workload) = &resumed.workload else { - panic!("a resume must be a command task"); - }; - assert_eq!(workload.command.to_vec(), expected); - assert_eq!(resumed.binary, original.binary); - assert_eq!(resumed.env, original.env); - assert_eq!(resumed.cwd, original.cwd); - assert_eq!(resumed.thread, authority.supervisor.thread); -} - -#[test] -fn same_run_resume_fails_closed_before_any_record_when_the_checkpoint_is_gone() { - let (mut fixture, authority, decision) = stopped_return_fixture(); - fs::remove_dir_all(&decision.selected_checkpoint.path).unwrap(); - - let input = resume_input( - authority, - fixture.task_id, - &decision.selected_checkpoint.generation_id, - ); - let task_id = input.launch.task_id; - assert!(matches!( - fixture.store.accept_return_task_for_authority(input), - Err(ReturnDecisionError::Rejected( - ReturnDecisionRejection::ResumeUnproven { - gap: SameRunResumeGap::CheckpointUnavailable - } - )) - )); - assert!(fixture.store.get_task(task_id).unwrap().is_none()); - assert!(matches!( - fixture - .store - .resource_snapshots_for_authority(fixture.authority) - .unwrap()[0] - .loan - .as_ref() - .map(|loan| &loan.state), - Some(LoanState::Active { - phase: LoanPhase::AwaitingReturn { .. } - }) - )); -} - -/// Bind a same-run resume, then let it start and fail with a confirmed wrapper exit -/// -/// Returns the resolution authority at the Restoring revision and the resume task -fn resumed_wrapper_exit_fixture() -> (TrainerAssociationFixture, SupervisorActionAuthority, TaskId) -{ - let (mut fixture, authority, decision) = stopped_return_fixture(); - let input = resume_input( - authority, - fixture.task_id, - &decision.selected_checkpoint.generation_id, - ); - let task_id = input.launch.task_id; - let ReturnTaskAcceptance::Inserted { state_revision, .. } = fixture - .store - .accept_return_task_for_authority(input) - .unwrap() - else { - panic!("the first exact resume must insert its task"); - }; - fixture - .store - .cas_status(task_id, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - fixture - .store - .cas_exit_with_evidence( - task_id, - ProcessStatus::Running, - &ExitReason::Exit { code: 1 }, - ProcessGroupExitEvidence::ConfirmedExited, - ) - .unwrap() - .unwrap(); - let mut current = authority; - current.expected_state_revision = state_revision; - (fixture, current, task_id) -} - -fn resolve( - fixture: &mut TrainerAssociationFixture, - authority: SupervisorActionAuthority, - task_id: TaskId, -) -> Result { - fixture - .store - .resolve_ended_restore_for_authority(EndedRestoreResolution { - authority, - task_id, - reason: "resume failed".into(), - }) -} - -fn assert_still_restoring(fixture: &TrainerAssociationFixture, task_id: TaskId) { - assert!(matches!( - saved_loan(&fixture.store, fixture.authority).map(|loan| loan.state), - Some(LoanState::Active { - phase: LoanPhase::Restoring { resume_task_id, .. } - }) if resume_task_id == task_id - )); -} - -#[test] -fn resumed_trainer_wrapper_exit_closes_restore_only_after_the_exact_lock_is_free() { - let (mut fixture, authority, task_id) = resumed_wrapper_exit_fixture(); - let stopped = fixture.task_id; - let lock_path = fixture.runtime_root.join(".segment.lock"); - - // the resumed worker runs in its own session and still holds the run's lock - let mut worker = start_fake_lock_process(&lock_path); - assert!(matches!( - resolve(&mut fixture, authority, task_id), - Err(ReturnDecisionError::RestoreOwnershipUnproven { - task_id: ended, - gap: TrainerLockReleaseGap::OwnershipHeld { task_id: witness }, - }) if ended == task_id && witness == stopped - )); - assert_still_restoring(&fixture, task_id); - assert_eq!( - saved_resource(&fixture.store, fixture.authority).state_revision, - authority.expected_state_revision - ); - - worker.release(); - let closure = resolve(&mut fixture, authority, task_id).unwrap(); - assert!(matches!( - &closure.loan.state, - LoanState::Closed { - result: LoanClosure::RestoreEnded { - task_id: ended, - outcome: ExitReason::Exit { code: 1 }, - .. - } - } if *ended == task_id - )); - assert_eq!( - saved_resource(&fixture.store, fixture.authority).registered_background_task, - None - ); - // the authority held the lock only through the closing transaction - let identity = OwnershipLockIdentity::from_metadata(&fs::metadata(&lock_path).unwrap()); - assert!(matches!( - probe_segment_ownership_lock(&fixture.runtime_root, identity), - OwnershipLockProbe::ExactOwnershipReleased(_) - )); -} - -#[test] -fn resumed_trainer_wrapper_exit_with_changed_or_missing_lock_evidence_stays_reserved() { - type Change = fn(&TrainerAssociationFixture); - let replace_lock: Change = |fixture| { - let lock_path = fixture.runtime_root.join(".segment.lock"); - fs::rename(&lock_path, fixture.runtime_root.join(".segment.lock.saved")).unwrap(); - fs::write(&lock_path, b"").unwrap(); - }; - let remove_association: Change = |fixture| { - fixture - .store - .conn - .execute( - "DELETE FROM trainer_attempt_associations WHERE task_id = ?1", - [fixture.task_id.to_string()], - ) - .unwrap(); - }; - type Expected = fn(&TrainerLockReleaseGap, TaskId) -> bool; - let different_file: Expected = |gap, _| { - matches!( - gap, - TrainerLockReleaseGap::OwnershipLock(OwnershipLockProbeError::IdentityMismatch { - reason: OwnershipLockIdentityMismatchReason::DifferentFile, - .. - }) - ) - }; - let missing: Expected = |gap, stopped| matches!(gap, TrainerLockReleaseGap::AssociationMissing { task_id } if *task_id == stopped); - for (name, change, expected) in [ - ("replaced lock file", replace_lock, different_file), - ("missing association", remove_association, missing), - ] { - let (mut fixture, authority, task_id) = resumed_wrapper_exit_fixture(); - change(&fixture); - - match resolve(&mut fixture, authority, task_id) { - Err(ReturnDecisionError::RestoreOwnershipUnproven { - task_id: ended, - gap, - }) => { - assert_eq!(ended, task_id, "{name}"); - assert!(expected(&gap, fixture.task_id), "{name}: {gap:?}"); - } - other => panic!("{name}: expected a fail-closed restore, got {other:?}"), - } - assert_still_restoring(&fixture, task_id); - } -} - -#[test] -fn resume_that_never_spawned_closes_without_a_lock_witness() { - let (mut fixture, authority, decision) = stopped_return_fixture(); - let input = resume_input( - authority, - fixture.task_id, - &decision.selected_checkpoint.generation_id, - ); - let task_id = input.launch.task_id; - let ReturnTaskAcceptance::Inserted { state_revision, .. } = fixture - .store - .accept_return_task_for_authority(input) - .unwrap() - else { - panic!("the first exact resume must insert its task"); - }; - // cancelling a queued row records that no worker child started - fixture - .store - .cas_exit(task_id, ProcessStatus::Queued, &ExitReason::Cancelled) - .unwrap() - .unwrap(); - assert_eq!( - fixture - .store - .get_task(task_id) - .unwrap() - .unwrap() - .process_group_exit_evidence(), - ProcessGroupExitEvidence::NoChildSpawned - ); - // no association can name a lock, and none is needed - fixture - .store - .conn - .execute( - "DELETE FROM trainer_attempt_associations WHERE task_id = ?1", - [fixture.task_id.to_string()], - ) - .unwrap(); - - let mut current = authority; - current.expected_state_revision = state_revision; - let closure = resolve(&mut fixture, current, task_id).unwrap(); - assert!(matches!( - &closure.loan.state, - LoanState::Closed { - result: LoanClosure::RestoreEnded { - task_id: ended, - outcome: ExitReason::Cancelled, - .. - } - } if *ended == task_id - )); -} - -#[test] -fn same_run_resume_closes_on_its_confirmed_start_and_registers_the_trainer() { - let (mut fixture, authority, decision) = stopped_return_fixture(); - let input = resume_input( - authority, - fixture.task_id, - &decision.selected_checkpoint.generation_id, - ); - let task_id = input.launch.task_id; - let ReturnTaskAcceptance::Inserted { state_revision, .. } = fixture - .store - .accept_return_task_for_authority(input) - .unwrap() - else { - panic!("the first exact resume must insert its task"); - }; - start_task(&fixture.store, task_id); - - let RestoreReconcileOutcome::Closed { - closure, - task_id: registered, - } = fixture - .store - .reconcile_restoring_loan_for_authority(fixture.authority, fixture.resource.id) - .unwrap() - else { - panic!("a confirmed trainer start must close the loan"); - }; - assert_eq!(registered, task_id); - assert_eq!( - closure.state_revision, - ResourceRevision::new(state_revision.get() + 1) - ); - assert!(matches!( - closure.loan.state, - LoanState::Closed { - result: LoanClosure::Resumed { task_id: resumed, .. } - } if resumed == task_id - )); - assert_eq!( - saved_resource(&fixture.store, fixture.authority).registered_background_task, - Some(task_id) - ); - assert_eq!( - restore_closure_basis(&fixture.store, authority.action_id).as_deref(), - Some("confirmed_running") - ); -} - -/// Move the drained fixture's supervisor to another machine -fn remote_awaiting_return_fixture() -> (ServingFixture, SupervisorActionAuthority) { - let (fixture, mut authority) = awaiting_return_fixture(); - let remote = machine_other_than(fixture.authority); - fixture - .store - .conn - .execute( - "UPDATE resources SET supervisor_machine = ?1 WHERE id = ?2", - params![ - remote.to_string(), - fixture.resource.id.as_uuid().to_string() - ], - ) - .unwrap(); - authority.supervisor.machine = remote; - (fixture, authority) -} - -fn remote_input( - authority: SupervisorActionAuthority, - launch: ReturnLaunch, - normalized_spec_sha256: NormalizedSpecSha256, -) -> ReturnTaskAcceptanceInput { - ReturnTaskAcceptanceInput { - authority, - launch, - executor_env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - origin: ReturnTaskOrigin::Remote { - normalized_spec_sha256, - }, - } -} - -fn prepared_digest( - fixture: &mut ServingFixture, - authority: SupervisorActionAuthority, - launch: &ReturnLaunch, -) -> NormalizedSpecSha256 { - fixture - .store - .prepare_return_task_for_authority( - authority, - launch.clone(), - TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - ) - .unwrap() - .normalized_spec_sha256 -} - -fn action_receipt_count(store: &Store) -> i64 { - store - .conn - .query_row( - "SELECT COUNT(*) FROM resource_action_task_receipts", - [], - |row| row.get(0), - ) - .unwrap() -} - -#[test] -fn remote_native_return_binds_once_and_closes_only_after_its_confirmed_end() { - let (mut fixture, authority) = remote_awaiting_return_fixture(); - let launch = evaluation(&fixture, &["/bin/echo", "evaluate"]); - let task_id = launch.task_id; - let digest = prepared_digest(&mut fixture, authority, &launch); - // preparing derives the spec without writing any task or decision record - assert!(fixture.store.get_task(task_id).unwrap().is_none()); - assert_still_awaiting_return(&fixture, &authority); - - let ReturnTaskAcceptance::Inserted { .. } = fixture - .store - .accept_return_task_for_authority(remote_input(authority, launch.clone(), digest)) - .unwrap() - else { - panic!("the first remote launch must insert its task"); - }; - assert_eq!(action_receipt_count(&fixture.store), 1); - let Some(ExecutorIdentity::Accepted(record)) = - fixture.store.executor_identity(task_id).unwrap() - else { - panic!("the return task must have an accepted identity"); - }; - assert_eq!(record.origin_machine, authority.supervisor.machine); - assert_eq!(record.execution_machine, fixture.authority); - // no authority-local callback route is substituted for the remote supervisor - assert!( - fixture - .store - .origin_route_by_task(task_id) - .unwrap() - .is_none() - ); - - // exact retries observe the binding; the local path observes it without a route - for input in [ - remote_input(authority, launch.clone(), digest), - launch_input(authority, launch.clone()), - ] { - assert_eq!( - fixture - .store - .accept_return_task_for_authority(input) - .unwrap(), - ReturnTaskAcceptance::Existing { - task: task_id, - state: ProcessStatus::Queued, - } - ); - } - let other_digest = normalized_spec_sha256(&fixture.spec).unwrap(); - assert!(matches!( - fixture.store.accept_return_task_for_authority(remote_input( - authority, - launch.clone(), - other_digest - )), - Err(ReturnDecisionError::SpecMismatch) - )); - - assert!(matches!( - fixture - .store - .reconcile_restoring_loan_for_authority(fixture.authority, fixture.resource.id) - .unwrap(), - RestoreReconcileOutcome::Queued { task_id: queued, .. } if queued == task_id - )); - start_task(&fixture.store, task_id); - assert!(matches!( - reconcile_restore(&mut fixture), - RestoreReconcileOutcome::ForegroundRunning { task_id: running, .. } if running == task_id - )); - assert_eq!( - fixture - .store - .accept_return_task_for_authority(remote_input(authority, launch.clone(), digest)) - .unwrap(), - ReturnTaskAcceptance::Existing { - task: task_id, - state: ProcessStatus::Running, - } - ); - finish_running_task( - &fixture.store, - task_id, - ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let RestoreReconcileOutcome::ForegroundEnded { - closure, - task_id: ended, - } = reconcile_restore(&mut fixture) - else { - panic!("a successful confirmed end must close the remote restore"); - }; - assert_eq!(ended, task_id); - assert!(matches!( - closure.loan.state, - LoanState::Closed { - result: LoanClosure::ForegroundReturnEnded { task_id: ended, .. } - } if ended == task_id - )); - // the remote callback owner keeps its receipt, and no local route replaces it - assert_eq!(action_receipt_count(&fixture.store), 1); - assert!( - fixture - .store - .origin_route_by_task(task_id) - .unwrap() - .is_none() - ); - assert_eq!( - fixture - .store - .accept_return_task_for_authority(remote_input(authority, launch, digest)) - .unwrap(), - ReturnTaskAcceptance::Existing { - task: task_id, - state: ProcessStatus::Succeeded, - } - ); -} - -#[test] -fn remote_return_rejects_changed_spec_and_wrong_origin_without_records() { - let (mut fixture, authority) = remote_awaiting_return_fixture(); - let launch = evaluation(&fixture, &["/bin/echo", "evaluate"]); - let task_id = launch.task_id; - let wrong = normalized_spec_sha256(&fixture.spec).unwrap(); - assert!(matches!( - fixture.store.accept_return_task_for_authority(remote_input( - authority, - launch.clone(), - wrong - )), - Err(ReturnDecisionError::SpecMismatch) - )); - assert!(fixture.store.get_task(task_id).unwrap().is_none()); - assert_eq!(action_receipt_count(&fixture.store), 0); - assert_still_awaiting_return(&fixture, &authority); - - // a co-located supervisor cannot use the remote origin - let (mut local, local_authority) = awaiting_return_fixture(); - let launch = evaluation(&local, &["/bin/echo", "evaluate"]); - let digest = prepared_digest(&mut local, local_authority, &launch); - assert!(matches!( - local - .store - .accept_return_task_for_authority(remote_input(local_authority, launch, digest)), - Err(ReturnDecisionError::OriginMismatch) - )); - assert_eq!(action_receipt_count(&local.store), 0); - assert_still_awaiting_return(&local, &local_authority); -} - -#[test] -fn remote_return_early_end_and_no_resume_need_the_exact_remote_supervisor() { - let (mut fixture, authority) = remote_awaiting_return_fixture(); - let launch = evaluation(&fixture, &["/bin/echo", "evaluate"]); - let task_id = launch.task_id; - let digest = prepared_digest(&mut fixture, authority, &launch); - let ReturnTaskAcceptance::Inserted { state_revision, .. } = fixture - .store - .accept_return_task_for_authority(remote_input(authority, launch, digest)) - .unwrap() - else { - panic!("the first remote launch must insert its task"); - }; - fixture - .store - .cas_exit_with_evidence( - task_id, - ProcessStatus::Queued, - &ExitReason::Cancelled, - ProcessGroupExitEvidence::NoChildSpawned, - ) - .unwrap() - .unwrap(); - assert!(matches!( - fixture - .store - .reconcile_restoring_loan_for_authority(fixture.authority, fixture.resource.id) - .unwrap(), - RestoreReconcileOutcome::Attention { - reason: RestoreAttentionReason::ForegroundEnded { - state: ProcessStatus::Cancelled - }, - .. - } - )); - let mut current = authority; - current.expected_state_revision = state_revision; - let mut replaced = current; - replaced.supervisor.machine = fixture.authority; - assert!(matches!( - fixture - .store - .resolve_ended_restore_for_authority(EndedRestoreResolution { - authority: replaced, - task_id, - reason: "never started".into(), - }), - Err(ReturnDecisionError::NotCurrentSupervisor) - )); - let closure = fixture - .store - .resolve_ended_restore_for_authority(EndedRestoreResolution { - authority: current, - task_id, - reason: "never started".into(), - }) - .unwrap(); - assert!(matches!( - closure.loan.state, - LoanState::Closed { - result: LoanClosure::RestoreEnded { task_id: ended, .. } - } if ended == task_id - )); - - let (mut no_resume, no_resume_authority) = remote_awaiting_return_fixture(); - let closure = no_resume - .store - .record_no_resume_for_authority(no_resume_authority, "training finished".into()) - .unwrap(); - assert!(matches!( - closure.loan.state, - LoanState::Closed { - result: LoanClosure::NoResume { .. } - } - )); -} - -#[test] -fn unstorable_assignment_revision_is_a_failed_compare_not_a_supervisor_change() { - let directory = tempdir().unwrap(); - let mut store = Store::open(&directory.path().join("db")).unwrap(); - let authority = MachineId::new(); - let saved = resource(authority); - store.register_resource(authority, &saved).unwrap(); - // no stored row can hold this revision, so the compare can never match - let mut snapshot = saved.clone(); - snapshot.assignment_revision = AssignmentRevision::new(u64::MAX); - - let tx = store.conn.transaction().unwrap(); - let result = crate::store::resource::restore::advance_resource(&tx, &snapshot, None); - - assert!( - matches!( - result, - Err(ReturnDecisionError::StaleRevision { expected, actual }) - if expected == saved.state_revision && actual == saved.state_revision - ), - "{result:?}" - ); -} diff --git a/src/store/resource/tests/return_deadline.rs b/src/store/resource/tests/return_deadline.rs deleted file mode 100644 index 2057082..0000000 --- a/src/store/resource/tests/return_deadline.rs +++ /dev/null @@ -1,235 +0,0 @@ -//! Return decision windows: queued work after the deadline, holds, and upgrades - -use std::time::Duration; - -use chrono::{DateTime, Utc}; - -use super::fixtures::{ - ServingFixture, accept_and_finish_resource_task, completion, refresh_serving_fixture, -}; -use super::restore::{ - accept_post_return_request, awaiting_return_fixture, saved_loan, saved_resource, -}; -use crate::domain::{ExitReason, ProcessGroupExitEvidence}; -use crate::resource::store::{ - ResourceTaskCompletionResult, ReturnDeadlineOutcome, open_missing_return_windows_on, - return_window_on, serve_after_return_deadline_for_authority, -}; -use crate::resource::{ - ActionId, LoanPhase, LoanState, RETURN_DECISION_GRACE, RETURN_DECISION_LIMIT, - ResourceRequestState, ReturnContext, ReturnDecisionWindow, ServingReleaseProvenance, - SupervisorActionAuthority, -}; -use crate::store::ReturnDecisionError; - -fn window(fixture: &ServingFixture, action_id: ActionId) -> ReturnDecisionWindow { - return_window_on(&fixture.store.conn, action_id) - .unwrap() - .expect("every return action has a decision window") -} - -fn serve_at(fixture: &mut ServingFixture, now: DateTime) -> ReturnDeadlineOutcome { - serve_after_return_deadline_for_authority( - &mut fixture.store.conn, - fixture.authority, - fixture.resource.id, - now, - ) - .unwrap() -} - -fn awaited_return(fixture: &ServingFixture) -> (ActionId, ReturnContext) { - match saved_loan(&fixture.store, fixture.authority).map(|loan| loan.state) { - Some(LoanState::Active { - phase: - LoanPhase::AwaitingReturn { - action_id, - return_context, - }, - }) => (action_id, return_context), - other => panic!("the loan must await a return decision, found {other:?}"), - } -} - -#[test] -fn a_return_action_opens_its_window_with_the_default_grace() { - let (fixture, authority) = awaiting_return_fixture(); - - let window = window(&fixture, authority.action_id); - - assert_eq!(window.loan_id(), authority.loan_id); - assert_eq!(window.resource_id(), authority.resource_id); - assert_eq!( - window.deadline_at(), - window.opened_at() + RETURN_DECISION_GRACE - ); -} - -#[test] -fn queued_work_takes_the_resource_only_after_the_window_closes() { - let (mut fixture, authority) = awaiting_return_fixture(); - let (_, return_context) = awaited_return(&fixture); - let later = accept_post_return_request(&mut fixture); - let window = window(&fixture, authority.action_id); - - let before = window.deadline_at() - chrono::Duration::seconds(1); - assert!(matches!( - serve_at(&mut fixture, before), - ReturnDeadlineOutcome::Open { window: open } if open == window - )); - assert_eq!(awaited_return(&fixture).0, authority.action_id); - - let ReturnDeadlineOutcome::Served { loan, request } = - serve_at(&mut fixture, window.deadline_at()) - else { - panic!("an expired window with queued work must serve it"); - }; - assert_eq!(loan.id, authority.loan_id); - assert_eq!(request.request_id, later.request_id); - assert_eq!( - request.state, - ResourceRequestState::Assigned { loan_id: loan.id } - ); - assert_eq!( - loan.state, - LoanState::Active { - phase: LoanPhase::Serving { - return_context: return_context.clone(), - current_request_id: later.request_id, - release_provenance: ServingReleaseProvenance::ReturnDeadlinePassed { - action_id: authority.action_id, - }, - }, - } - ); - let resource = saved_resource(&fixture.store, fixture.authority); - assert_eq!( - resource.state_revision, - authority.expected_state_revision.next().unwrap() - ); - - // the supervisor's late decision names an action that no longer waits - let late = fixture - .store - .record_no_resume_for_authority(authority, "too late".into()) - .unwrap_err(); - assert!(matches!(late, ReturnDecisionError::ActionNotPending { .. })); - - // the deadline receipt proves the release, so the served task starts and the - // next drained queue asks for the same return again - refresh_serving_fixture(&mut fixture, *loan, *request); - let input = accept_and_finish_resource_task( - &mut fixture, - ExitReason::Exit { code: 0 }, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let result = fixture - .store - .reconcile_assigned_resource_task_for_authority(input) - .unwrap(); - let Ok(ResourceTaskCompletionResult::ReturnRequired { loan, notice, .. }) = completion(result) - else { - panic!("the drained queue must ask for the return again"); - }; - assert_eq!(loan.id, authority.loan_id); - assert_ne!(notice.action_id, authority.action_id); - assert_eq!(awaited_return(&fixture), (notice.action_id, return_context)); - assert!( - return_window_on(&fixture.store.conn, notice.action_id) - .unwrap() - .is_some() - ); -} - -#[test] -fn an_expired_window_without_queued_work_keeps_the_return_pending() { - let (mut fixture, authority) = awaiting_return_fixture(); - let window = window(&fixture, authority.action_id); - - assert!(matches!( - serve_at(&mut fixture, window.limit_at()), - ReturnDeadlineOutcome::NoQueuedRequest - )); - - assert_eq!(awaited_return(&fixture).0, authority.action_id); - fixture - .store - .record_no_resume_for_authority(authority, "not needed".into()) - .unwrap(); - assert!(matches!( - serve_at(&mut fixture, window.limit_at()), - ReturnDeadlineOutcome::NotAwaiting - )); -} - -#[test] -fn a_hold_keeps_queued_work_waiting_without_changing_the_pending_action() { - let (mut fixture, authority) = awaiting_return_fixture(); - accept_post_return_request(&mut fixture); - let opened = window(&fixture, authority.action_id); - - let held = fixture - .store - .hold_return_for_authority(authority, Duration::from_secs(5 * 60)) - .unwrap(); - - assert!(held.deadline_at() > opened.deadline_at()); - assert!(held.deadline_at() <= opened.opened_at() + RETURN_DECISION_LIMIT); - assert_eq!(window(&fixture, authority.action_id), held); - assert_eq!( - saved_resource(&fixture.store, fixture.authority).state_revision, - authority.expected_state_revision - ); - assert!(matches!( - serve_at(&mut fixture, opened.deadline_at()), - ReturnDeadlineOutcome::Open { .. } - )); - - // the pending action is still decided with the same saved authority - fixture - .store - .record_no_resume_for_authority(authority, "not needed".into()) - .unwrap(); -} - -#[test] -fn a_hold_needs_the_exact_pending_action() { - let (mut fixture, authority) = awaiting_return_fixture(); - let other = SupervisorActionAuthority { - action_id: ActionId::new(), - ..authority - }; - - let error = fixture - .store - .hold_return_for_authority(other, Duration::from_secs(60)) - .unwrap_err(); - - assert!(matches!( - error, - ReturnDecisionError::ActionNotPending { .. } - )); - let error = fixture - .store - .hold_return_for_authority(authority, Duration::ZERO) - .unwrap_err(); - assert!(matches!(error, ReturnDecisionError::HoldRejected(_))); -} - -#[test] -fn an_upgrade_opens_a_full_window_for_a_return_saved_without_one() { - let (fixture, authority) = awaiting_return_fixture(); - fixture - .store - .conn - .execute("DELETE FROM resource_return_windows", []) - .unwrap(); - let upgraded_at = Utc::now(); - - open_missing_return_windows_on(&fixture.store.conn, upgraded_at).unwrap(); - - let window = window(&fixture, authority.action_id); - assert_eq!(window.opened_at(), upgraded_at); - assert_eq!(window.loan_id(), authority.loan_id); - assert_eq!(window.deadline_at(), upgraded_at + RETURN_DECISION_GRACE); -} diff --git a/src/store/resource/tests/supervised_launch.rs b/src/store/resource/tests/supervised_launch.rs deleted file mode 100644 index be2f9a9..0000000 --- a/src/store/resource/tests/supervised_launch.rs +++ /dev/null @@ -1,636 +0,0 @@ -//! Daemon-driven startup, trainer reconcile, and assigned task launch tests - -use super::fixtures::{ - TrainerAssociationFixture, fake_resource_task_spec, gated_resource_task_spec, - machine_other_than, publish_completed_result, release_completion_fixture_with, - return_notice_count, stop_test_resource_actor, stop_test_supervisor, wait_for_awaiting_return, - wait_for_running_task, wait_for_terminal_task, -}; -use crate::daemon::actors::resource::{ResourceActor, ResourceMsg}; -use crate::daemon::actors::supervisor::SUPERVISOR_TEST_LOCK; -use crate::daemon::actors::{ - StoreActor, StoreMsg, SupervisorActor, SupervisorArgs, SupervisorMsg, call, -}; -use crate::domain::{ExitReason, ProcessGroupExitEvidence, ProcessStatus, TaskId}; -use crate::home::{Home, LockMode, flock_exclusive}; -use crate::machine::{MachineId, load_or_create_machine_id}; -use crate::resource::store::ReleaseCompletionResult; -use crate::resource::trainer_publication::find_completed_result; -use crate::resource::{ - LoanPhase, LoanState, ReleaseProofAttentionReason, ResourceQueueReconcileOutcome, - ResourceRequest, ResourceRequestState, ServingReleaseProvenance, -}; -use crate::submission::RequestId; -use ractor::Actor; -use std::fs; -use tempfile::tempdir; - -#[tokio::test] -async fn startup_retries_completed_trainer_proof_and_launches_assigned_task_once() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - crate::runner::set_task_run_executable_for_tests(assert_cmd::cargo::cargo_bin("homebased")); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let fixture = TrainerAssociationFixture::new_for_home(directory, &home, authority); - let marker = fixture.home.join("activation-count"); - let gate = fixture.home.join("release-commands"); - let request_spec = gated_resource_task_spec(&fixture.home, &marker, &gate); - let (mut fixture, binding, request_id, action_id, _, loan_id) = release_completion_fixture_with( - fixture, - request_spec.clone(), - machine_other_than(authority), - ); - let second = fixture - .store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - machine_other_than(authority), - request_spec, - ) - .unwrap(); - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let publication = find_completed_result(&fixture.runtime_root, &binding) - .unwrap() - .unwrap(); - let request = fixture - .store - .resource_requests(authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|request| request.request_id == request_id) - .unwrap(); - assert!(matches!(request.state, ResourceRequestState::Queued)); - let resource_id = fixture.resource.id; - let trainer_task_id = fixture.task_id; - drop(fixture.store); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - wait_for_running_task(&store, request.task_id).await; - for _ in 0..3 { - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource_id, - reply, - }) - .await - .unwrap(); - } - let inspection = call(&supervisor, |reply| SupervisorMsg::InspectResource { - id: resource_id, - reply, - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - inspection.loan.as_ref().map(|loan| &loan.state), - Some(LoanState::Active { - phase: LoanPhase::Serving { - current_request_id, - release_provenance: - ServingReleaseProvenance::CompletedTrainerResult { - action_id: saved_action, - task_id: saved_task, - publication_sha256, - }, - .. - } - }) if *current_request_id == request_id - && *saved_action == action_id - && *saved_task == trainer_task_id - && *publication_sha256 == publication.publication_sha256 - )); - // the next queued request waits for the running task's confirmed exit - assert!( - call(&store, |reply| StoreMsg::GetTask { - id: second.task_id, - reply, - }) - .await - .unwrap() - .is_none() - ); - - fs::write(&gate, b"").unwrap(); - let row = wait_for_terminal_task(&store, request.task_id).await; - assert_eq!(row.status(), ProcessStatus::Succeeded); - let second_row = wait_for_terminal_task(&store, second.task_id).await; - assert_eq!(second_row.status(), ProcessStatus::Succeeded); - let loan = wait_for_awaiting_return(&supervisor, resource_id).await; - assert_eq!(loan.id, loan_id); - for _ in 0..3 { - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource_id, - reply, - }) - .await - .unwrap(); - } - assert_eq!(fs::read(&marker).unwrap(), b"xx"); - assert_eq!(return_notice_count(&home, loan_id), 1); - let requests = call(&store, |reply| StoreMsg::ResourceRequests { - authority_machine: authority, - resource_id, - reply, - }) - .await - .unwrap(); - assert!(requests.iter().all(|request| matches!( - request.state, - ResourceRequestState::Finished { - outcome: ExitReason::Exit { code: 0 } - } - ))); - - stop_test_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn exact_trainer_terminal_event_reconciles_without_a_client_wake() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - crate::runner::set_task_run_executable_for_tests(assert_cmd::cargo::cargo_bin("homebased")); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let fixture = TrainerAssociationFixture::new_for_home(directory, &home, authority); - let marker = fixture.home.join("activation-count"); - let gate = fixture.home.join("release-commands"); - let request_spec = gated_resource_task_spec(&fixture.home, &marker, &gate); - let (fixture, binding, request_id, action_id, _, loan_id) = - release_completion_fixture_with(fixture, request_spec, machine_other_than(authority)); - let request = fixture - .store - .resource_requests(authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|request| request.request_id == request_id) - .unwrap(); - let resource_id = fixture.resource.id; - let trainer_task_id = fixture.task_id; - let runtime_root = fixture.runtime_root.clone(); - home.prepare_task(trainer_task_id).unwrap(); - let trainer_lock = flock_exclusive( - &home.task_paths(trainer_task_id).runner_lock, - LockMode::NonBlocking, - ) - .unwrap(); - drop(fixture.store); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let before = call(&supervisor, |reply| SupervisorMsg::InspectResource { - id: resource_id, - reply, - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - before.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::ReleaseProofUnavailable { - reason: ReleaseProofAttentionReason::TrainerNotCompleted, - .. - }) - )); - - crate::resource::trainer_publication::tests::write_completed_result_for_test( - &runtime_root, - &binding, - ); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let terminal = call(&store, |reply| StoreMsg::CasExit { - id: trainer_task_id, - from: ProcessStatus::Running, - reason: ExitReason::Exit { code: 0 }, - evidence: crate::domain::ProcessGroupExitEvidence::ConfirmedExited.into(), - reply, - }) - .await - .unwrap() - .unwrap(); - assert_eq!(terminal.status(), ProcessStatus::Succeeded); - drop(trainer_lock); - - wait_for_running_task(&store, request.task_id).await; - let after = call(&supervisor, |reply| SupervisorMsg::InspectResource { - id: resource_id, - reply, - }) - .await - .unwrap() - .unwrap(); - assert!(matches!( - after.loan.as_ref().map(|loan| &loan.state), - Some(LoanState::Active { - phase: LoanPhase::Serving { - release_provenance: - ServingReleaseProvenance::CompletedTrainerResult { - action_id: saved_action, - task_id: saved_task, - .. - }, - .. - } - }) if *saved_action == action_id && *saved_task == trainer_task_id - )); - assert!(matches!( - after.loan.map(|loan| loan.id), - Some(saved_loan) if saved_loan == loan_id - )); - - fs::write(&gate, b"").unwrap(); - let row = wait_for_terminal_task(&store, request.task_id).await; - assert_eq!(row.status(), ProcessStatus::Succeeded); - assert_eq!(fs::read(&marker).unwrap(), b"x"); - let loan = wait_for_awaiting_return(&supervisor, resource_id).await; - assert_eq!(loan.id, loan_id); - assert_eq!(return_notice_count(&home, loan_id), 1); - - stop_test_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn pre_activation_cancellation_keeps_release_proof_for_the_next_request() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - crate::runner::set_task_run_executable_for_tests(assert_cmd::cargo::cargo_bin("homebased")); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let fixture = TrainerAssociationFixture::new_for_home(directory, &home, authority); - let marker = fixture.home.join("activation-count"); - let request_spec = fake_resource_task_spec(&fixture.home, &marker); - let (mut fixture, binding, first_request_id, action_id, revision, loan_id) = - release_completion_fixture_with( - fixture, - request_spec.clone(), - machine_other_than(authority), - ); - let second_origin = MachineId::new(); - let second_request = fixture - .store - .accept_resource_request( - authority, - RequestId::new(), - TaskId::new(), - fixture.resource.id, - second_origin, - request_spec, - ) - .unwrap(); - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - assert!(matches!( - fixture - .store - .complete_release_for_authority( - authority, - fixture.resource.id, - action_id, - revision, - ) - .unwrap(), - ReleaseCompletionResult::Assigned { request, .. } - if request.request_id == first_request_id - )); - let first_request = fixture - .store - .resource_requests(authority, fixture.resource.id) - .unwrap() - .into_iter() - .find(|request| request.request_id == first_request_id) - .unwrap(); - assert!(matches!( - fixture - .store - .cancel_resource_request_before_activation( - authority, - first_request.request_id, - first_request.task_id, - fixture.resource.id, - first_request.origin_machine, - ) - .unwrap(), - crate::resource::store::QueueCancellationResult::Request(saved) - if saved.request_id == first_request_id - && matches!(saved.state, ResourceRequestState::CancelledBeforeLaunch) - )); - let snapshot = fixture - .store - .resource_snapshots_for_authority(authority) - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == fixture.resource.id) - .unwrap(); - assert!(matches!( - snapshot.loan.as_ref().map(|loan| &loan.state), - Some(LoanState::Active { - phase: LoanPhase::Serving { - current_request_id, - release_provenance: - ServingReleaseProvenance::CompletedTrainerResult { - action_id: saved_action, - .. - }, - .. - } - }) if *current_request_id == second_request.request_id - && *saved_action == action_id - )); - let resource_id = fixture.resource.id; - let second_task_id = second_request.task_id; - drop(fixture.store); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let row = wait_for_terminal_task(&store, second_task_id).await; - assert_eq!(row.status(), ProcessStatus::Succeeded); - let loan = wait_for_awaiting_return(&supervisor, resource_id).await; - assert_eq!(loan.id, loan_id); - assert_eq!(fs::read(&marker).unwrap(), b"x"); - let requests = call(&store, |reply| StoreMsg::ResourceRequests { - authority_machine: authority, - resource_id, - reply, - }) - .await - .unwrap(); - assert!(matches!( - requests - .iter() - .find(|request| request.request_id == first_request_id), - Some(ResourceRequest { - state: ResourceRequestState::CancelledBeforeLaunch, - .. - }) - )); - assert!(matches!( - requests - .iter() - .find(|request| request.request_id == second_request.request_id), - Some(ResourceRequest { - state: ResourceRequestState::Finished { - outcome: ExitReason::Exit { code: 0 } - }, - .. - }) - )); - - stop_test_supervisor(supervisor, handle).await; -} - -#[tokio::test] -async fn ongoing_unproven_or_unconfirmed_trainer_keeps_request_queued_and_does_not_launch() { - // an ended run with no result is released by its lock, so a held lock keeps it reserved - for (finish_task, publish_result, exit_evidence, expected_reason) in [ - ( - false, - false, - ProcessGroupExitEvidence::ConfirmedExited, - ReleaseProofAttentionReason::TrainerNotCompleted, - ), - ( - true, - false, - ProcessGroupExitEvidence::ConfirmedExited, - ReleaseProofAttentionReason::OwnershipLockUnverified, - ), - ( - true, - true, - ProcessGroupExitEvidence::Unconfirmed, - ReleaseProofAttentionReason::WorkerExitUnconfirmed, - ), - ] { - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let fixture = TrainerAssociationFixture::new_for_home(directory, &home, authority); - let marker = fixture.home.join("activation-count"); - let request_spec = fake_resource_task_spec(&fixture.home, &marker); - let (mut fixture, binding, request_id, _, _, _) = - release_completion_fixture_with(fixture, request_spec, machine_other_than(authority)); - let mut lock_holder = None; - if publish_result { - publish_completed_result(&mut fixture, &binding, exit_evidence); - } else if finish_task { - fixture.finish_registered_task_with_evidence(exit_evidence); - lock_holder = Some(fixture.hold_saved_lock()); - } - let resource_id = fixture.resource.id; - let task_id = fixture - .store - .resource_requests(authority, resource_id) - .unwrap() - .into_iter() - .find(|request| request.request_id == request_id) - .unwrap() - .task_id; - drop(fixture.store); - - let (store, store_handle) = StoreActor::spawn(None, StoreActor, home.db_path()) - .await - .unwrap(); - let snapshot = call(&store, |reply| StoreMsg::ResourceSnapshotsForAuthority { - authority_machine: authority, - reply, - }) - .await - .unwrap() - .into_iter() - .find(|snapshot| snapshot.resource.id == resource_id) - .unwrap(); - let (actor, actor_handle) = ResourceActor::spawn( - None, - ResourceActor, - ( - store.clone(), - crate::daemon::actors::resource::test_support::spawn_stub_supervisor() - .await - .0, - authority, - snapshot.resource, - snapshot.loan, - ), - ) - .await - .unwrap(); - let inspection = call(&actor, |reply| ResourceMsg::Inspect { reply }) - .await - .unwrap(); - assert!(matches!( - inspection.reconcile_outcome, - Some(ResourceQueueReconcileOutcome::ReleaseProofUnavailable { - reason, - .. - }) if reason == expected_reason - )); - let requests = call(&store, |reply| StoreMsg::ResourceRequests { - authority_machine: authority, - resource_id, - reply, - }) - .await - .unwrap(); - assert!(matches!( - requests - .iter() - .find(|request| request.request_id == request_id), - Some(ResourceRequest { - state: ResourceRequestState::Queued, - .. - }) - )); - assert!( - call(&store, |reply| StoreMsg::GetTask { id: task_id, reply }) - .await - .unwrap() - .is_none() - ); - assert!(!marker.exists()); - - stop_test_resource_actor(actor, actor_handle, store, store_handle).await; - drop(lock_holder); - } -} - -#[tokio::test] -async fn an_undecided_return_serves_late_queued_work_after_its_deadline() { - let _guard = SUPERVISOR_TEST_LOCK.lock().await; - crate::runner::set_task_run_executable_for_tests(assert_cmd::cargo::cargo_bin("homebased")); - let directory = tempdir().unwrap(); - let home = Home::resolve(Some(directory.path().to_path_buf())).unwrap(); - home.ensure().unwrap(); - let authority = load_or_create_machine_id(&home).unwrap(); - let fixture = TrainerAssociationFixture::new_for_home(directory, &home, authority); - let marker = fixture.home.join("activation-count"); - let request_spec = fake_resource_task_spec(&fixture.home, &marker); - let (mut fixture, binding, _, _, _, loan_id) = release_completion_fixture_with( - fixture, - request_spec.clone(), - machine_other_than(authority), - ); - publish_completed_result( - &mut fixture, - &binding, - ProcessGroupExitEvidence::ConfirmedExited, - ); - let resource_id = fixture.resource.id; - drop(fixture.store); - - let (supervisor, handle) = SupervisorActor::spawn( - None, - SupervisorActor, - SupervisorArgs::new(home.clone(), None), - ) - .await - .unwrap(); - let store = call(&supervisor, |reply| SupervisorMsg::GetStore { reply }) - .await - .unwrap(); - let loan = wait_for_awaiting_return(&supervisor, resource_id).await; - let LoanState::Active { - phase: LoanPhase::AwaitingReturn { - action_id: expired, .. - }, - } = loan.state - else { - unreachable!("the helper returns only an AwaitingReturn loan"); - }; - - let late = call(&store, |reply| StoreMsg::AcceptResourceRequest { - authority_machine: authority, - request_id: RequestId::new(), - task_id: TaskId::new(), - resource_id, - origin_machine: machine_other_than(authority), - normalized_spec: Box::new(request_spec), - reply, - }) - .await - .unwrap(); - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource_id, - reply, - }) - .await - .unwrap(); - // the open window keeps the late request queued - assert!( - call(&store, |reply| StoreMsg::GetTask { - id: late.task_id, - reply, - }) - .await - .unwrap() - .is_none() - ); - - // close the window now instead of waiting for the default grace - let now = chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Nanos, true); - rusqlite::Connection::open(home.db_path()) - .unwrap() - .execute( - "UPDATE resource_return_windows - SET window_json = json_set(window_json, '$.deadline_at', ?1) - WHERE action_id = ?2", - rusqlite::params![now, expired.as_uuid().to_string()], - ) - .unwrap(); - call(&supervisor, |reply| SupervisorMsg::ReconcileResource { - id: resource_id, - reply, - }) - .await - .unwrap(); - - let row = wait_for_terminal_task(&store, late.task_id).await; - assert_eq!(row.status(), ProcessStatus::Succeeded); - let loan = wait_for_awaiting_return(&supervisor, resource_id).await; - assert_eq!(loan.id, loan_id); - assert!(matches!( - loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingReturn { action_id, .. } - } if action_id != expired - )); - - stop_test_supervisor(supervisor, handle).await; -} diff --git a/src/store/resource/tests/trainer_association.rs b/src/store/resource/tests/trainer_association.rs deleted file mode 100644 index ad79231..0000000 --- a/src/store/resource/tests/trainer_association.rs +++ /dev/null @@ -1,589 +0,0 @@ -//! Trainer attempt association binding, validation, and migration tests - -use super::fixtures::{ - TrainerAssociationFixture, awaiting_release_state, machine_other_than, resource, - saved_trainer_association_json, -}; -use crate::domain::{ProcessStatus, TaskId}; -use crate::machine::MachineId; -use crate::resource::command_shape::DirectSegmentCommandShapeError; -use crate::resource::ownership_lock::{ - OwnershipLockIdentity, TrainerRequestDigest, VerifiedTrainerAttempt, - test_support as trainer_attempt_test_support, -}; -use crate::resource::store::{ResourceStoreError, TrainerAttemptAssociationStoreError}; -use crate::resource::{ActionId, LoanPhase, LoanState, ResourceId, ReturnContext}; -use crate::store::Store; -use crate::submission::{RequestId, normalized_spec_sha256}; -use rusqlite::params; -use std::path::PathBuf; - -#[test] -fn trainer_attempt_association_binds_valid_direct_segment_shape_and_reads_it() { - let mut fixture = TrainerAssociationFixture::new(); - fixture.insert_accepted_running_task(); - let evidence = fixture.evidence(); - - let association = fixture.bind(evidence.clone()).unwrap(); - - assert_eq!(association.resource_id(), fixture.resource.id); - assert_eq!(association.authority_machine(), fixture.authority); - assert_eq!(association.task_id(), fixture.task_id); - assert_eq!(association.verified_attempt(), &evidence); - assert_eq!( - association.normalized_spec_sha256(), - normalized_spec_sha256(&fixture.spec).unwrap() - ); - assert_eq!( - fixture - .store - .trainer_attempt_association_for_task_for_authority(fixture.authority, fixture.task_id) - .unwrap(), - Some(association) - ); - let loan_count: i64 = fixture - .store - .conn - .query_row("SELECT COUNT(*) FROM loans", [], |row| row.get(0)) - .unwrap(); - assert_eq!(loan_count, 0); -} - -#[test] -fn trainer_attempt_association_rejects_shell_and_wrong_module_without_a_row() { - let mut shell = TrainerAssociationFixture::new(); - let evidence = shell.evidence(); - shell.set_command(vec![ - "/bin/sh".into(), - "-c".into(), - "echo not a trainer".into(), - ]); - shell.insert_accepted_running_task(); - assert!(matches!( - shell.bind(evidence), - Err( - TrainerAttemptAssociationStoreError::DirectSegmentCommandShape( - DirectSegmentCommandShapeError::NotPythonExecutable { .. } - ) - ) - )); - let shell_count: i64 = shell - .store - .conn - .query_row( - "SELECT COUNT(*) FROM trainer_attempt_associations", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(shell_count, 0); - - let mut wrong_module = TrainerAssociationFixture::new(); - let evidence = wrong_module.evidence(); - let mut command = wrong_module.command(); - command[2] = "ops.not_run_segment".into(); - wrong_module.set_command(command); - wrong_module.insert_accepted_running_task(); - assert!(matches!( - wrong_module.bind(evidence), - Err( - TrainerAttemptAssociationStoreError::DirectSegmentCommandShape( - DirectSegmentCommandShapeError::InvalidModuleInvocation { .. } - ) - ) - )); - let wrong_module_count: i64 = wrong_module - .store - .conn - .query_row( - "SELECT COUNT(*) FROM trainer_attempt_associations", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(wrong_module_count, 0); -} - -#[test] -fn trainer_attempt_association_rejects_runtime_root_mismatch_without_a_row() { - let mut fixture = TrainerAssociationFixture::new(); - let evidence = fixture.evidence(); - let mut command = fixture.command(); - let runtime_root_index = command - .iter() - .position(|argument| argument == "--runtime-root") - .unwrap(); - command[runtime_root_index + 1] = fixture.home.join("other-runtime").display().to_string(); - fixture.set_command(command); - fixture.insert_accepted_running_task(); - - assert!(matches!( - fixture.bind(evidence), - Err( - TrainerAttemptAssociationStoreError::DirectSegmentCommandShape( - DirectSegmentCommandShapeError::RuntimeRootMismatch { .. } - ) - ) - )); - let association_count: i64 = fixture - .store - .conn - .query_row( - "SELECT COUNT(*) FROM trainer_attempt_associations", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(association_count, 0); -} - -#[test] -fn trainer_attempt_association_exact_retry_after_terminal_and_file_removal() { - let mut fixture = TrainerAssociationFixture::new(); - fixture.insert_accepted_running_task(); - let evidence = fixture.evidence(); - let first = fixture.bind(evidence.clone()).unwrap(); - let before = saved_trainer_association_json(&fixture.store, fixture.task_id); - fixture.finish_registered_task(); - fixture.remove_shape_files(); - let TrainerAssociationFixture { - _directory, - database, - store, - authority, - resource, - task_id, - .. - } = fixture; - drop(store); - - let mut reopened = Store::open(&database).unwrap(); - let retry = reopened - .bind_trainer_attempt_association(authority, resource.id, task_id, evidence) - .unwrap(); - - assert_eq!(retry, first); - assert_eq!(saved_trainer_association_json(&reopened, task_id), before); - let association_count: i64 = reopened - .conn - .query_row( - "SELECT COUNT(*) FROM trainer_attempt_associations", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(association_count, 1); -} - -#[test] -fn trainer_attempt_association_rejects_changed_attempt_lock_and_request_digest() { - let mut fixture = TrainerAssociationFixture::new(); - fixture.insert_accepted_running_task(); - let evidence = fixture.evidence(); - fixture.bind(evidence.clone()).unwrap(); - let saved = saved_trainer_association_json(&fixture.store, fixture.task_id); - fixture.finish_registered_task(); - fixture.remove_shape_files(); - let changed_attempt = trainer_attempt_test_support::verified_attempt( - trainer_attempt_test_support::attempt_binding("attempt-2"), - [0x42; 32], - OwnershipLockIdentity::new(7, 11), - ); - let changed_lock = trainer_attempt_test_support::verified_attempt( - trainer_attempt_test_support::attempt_binding("attempt-1"), - [0x42; 32], - OwnershipLockIdentity::new(7, 12), - ); - let changed_request = trainer_attempt_test_support::verified_attempt( - trainer_attempt_test_support::attempt_binding("attempt-1"), - [0x43; 32], - OwnershipLockIdentity::new(7, 11), - ); - - for changed in [changed_attempt, changed_lock, changed_request] { - assert!(matches!( - fixture.bind(changed), - Err(TrainerAttemptAssociationStoreError::Conflict { resource_id }) - if resource_id == fixture.resource.id - )); - } - assert_eq!( - saved_trainer_association_json(&fixture.store, fixture.task_id), - saved - ); -} - -#[test] -fn trainer_attempt_associations_keep_history_and_read_only_the_current_registration() { - let mut fixture = TrainerAssociationFixture::new(); - fixture.insert_accepted_running_task(); - let first = fixture.bind(fixture.evidence()).unwrap(); - let second_task = TaskId::new(); - let second_evidence = VerifiedTrainerAttempt::from_persisted( - fixture.runtime_root.clone(), - trainer_attempt_test_support::attempt_binding("attempt-2"), - TrainerRequestDigest::from_hex(&"43".repeat(32)).unwrap(), - OwnershipLockIdentity::new(7, 12), - ) - .unwrap(); - - fixture - .store - .conn - .execute( - "UPDATE resources SET registered_background_task=?1 WHERE id=?2", - params![ - second_task.to_string(), - fixture.resource.id.as_uuid().to_string() - ], - ) - .unwrap(); - let second_row = fixture.trainer_task(second_task, &fixture.spec); - fixture - .store - .insert_local_task( - &second_row, - &fixture.spec, - fixture.authority, - crate::submission::RequestId::new(), - PathBuf::from("/bin/echo").into(), - ) - .unwrap(); - fixture - .store - .cas_status(second_task, ProcessStatus::Queued, ProcessStatus::Running) - .unwrap() - .unwrap(); - - let second = fixture - .store - .bind_trainer_attempt_association( - fixture.authority, - fixture.resource.id, - second_task, - second_evidence, - ) - .unwrap(); - - assert_ne!(first.task_id(), second.task_id()); - assert_eq!( - fixture - .store - .trainer_attempt_association_for_task_for_authority(fixture.authority, fixture.task_id) - .unwrap(), - Some(first.clone()) - ); - assert_eq!( - fixture - .store - .trainer_attempt_association_for_task_for_authority(fixture.authority, second_task) - .unwrap(), - Some(second) - ); - let association_count: i64 = fixture - .store - .conn - .query_row( - "SELECT COUNT(*) FROM trainer_attempt_associations WHERE resource_id=?1", - [fixture.resource.id.as_uuid().to_string()], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(association_count, 2); - - fixture - .store - .conn - .execute( - "UPDATE resources SET registered_background_task=NULL WHERE id=?1", - [fixture.resource.id.as_uuid().to_string()], - ) - .unwrap(); - assert_eq!( - fixture - .store - .trainer_attempt_association_for_task_for_authority(fixture.authority, fixture.task_id) - .unwrap(), - Some(first) - ); -} - -#[test] -fn trainer_attempt_association_rejects_wrong_authority_resource_and_task() { - let mut fixture = TrainerAssociationFixture::new(); - fixture.insert_accepted_running_task(); - let evidence = fixture.evidence(); - let wrong_authority = MachineId::new(); - assert!(matches!( - fixture.store.bind_trainer_attempt_association( - wrong_authority, - fixture.resource.id, - fixture.task_id, - evidence.clone(), - ), - Err(TrainerAttemptAssociationStoreError::Resource( - ResourceStoreError::WrongAuthority { .. } - )) - )); - assert!(matches!( - fixture.store.bind_trainer_attempt_association( - fixture.authority, - ResourceId::new(), - fixture.task_id, - evidence.clone(), - ), - Err(TrainerAttemptAssociationStoreError::Resource( - ResourceStoreError::ResourceNotFound - )) - )); - - let wrong_task = TaskId::new(); - assert!(matches!( - fixture.store.bind_trainer_attempt_association( - fixture.authority, - fixture.resource.id, - wrong_task, - evidence.clone(), - ), - Err(TrainerAttemptAssociationStoreError::TaskNotRegistered { task_id }) - if task_id == wrong_task - )); - - let mut other_resource = resource(fixture.authority); - other_resource.registered_background_task = Some(TaskId::new()); - fixture - .store - .register_resource(fixture.authority, &other_resource) - .unwrap(); - assert!(matches!( - fixture.store.bind_trainer_attempt_association( - fixture.authority, - other_resource.id, - fixture.task_id, - evidence, - ), - Err(TrainerAttemptAssociationStoreError::TaskNotRegistered { task_id }) - if task_id == fixture.task_id - )); -} - -#[test] -fn trainer_attempt_association_requires_running_task_identity_and_matching_spec() { - let mut missing_task = TrainerAssociationFixture::new(); - let evidence = missing_task.evidence(); - assert!(matches!( - missing_task.bind(evidence), - Err(TrainerAttemptAssociationStoreError::TaskMissing { task_id }) - if task_id == missing_task.task_id - )); - - let mut queued_task = TrainerAssociationFixture::new(); - let row = queued_task.trainer_task(queued_task.task_id, &queued_task.spec); - queued_task - .store - .insert_local_task( - &row, - &queued_task.spec, - queued_task.authority, - crate::submission::RequestId::new(), - PathBuf::from("/bin/echo").into(), - ) - .unwrap(); - let evidence = queued_task.evidence(); - assert!(matches!( - queued_task.bind(evidence), - Err(TrainerAttemptAssociationStoreError::TaskNotRunning { state, .. }) - if state == "queued" - )); - - let mut missing_identity = TrainerAssociationFixture::new(); - let row = missing_identity.trainer_task(missing_identity.task_id, &missing_identity.spec); - missing_identity.store.insert_task(&row).unwrap(); - missing_identity - .store - .cas_status( - missing_identity.task_id, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - let evidence = missing_identity.evidence(); - assert!(matches!( - missing_identity.bind(evidence), - Err(TrainerAttemptAssociationStoreError::IdentityMissing { task_id }) - if task_id == missing_identity.task_id - )); - - let mut stopped_identity = TrainerAssociationFixture::new(); - stopped_identity.insert_accepted_running_task(); - stopped_identity - .store - .update_execution_state(stopped_identity.task_id, ProcessStatus::Queued) - .unwrap(); - let evidence = stopped_identity.evidence(); - assert!(matches!( - stopped_identity.bind(evidence), - Err(TrainerAttemptAssociationStoreError::IdentityNotRunning { task_id }) - if task_id == stopped_identity.task_id - )); - - let mut changed_row = TrainerAssociationFixture::new(); - changed_row.insert_accepted_running_task(); - changed_row - .store - .conn - .execute( - "UPDATE tasks SET cwd='/different' WHERE id=?1", - [changed_row.task_id.to_string()], - ) - .unwrap(); - let evidence = changed_row.evidence(); - assert!(matches!( - changed_row.bind(evidence), - Err(TrainerAttemptAssociationStoreError::NormalizedSpecMismatch { task_id }) - if task_id == changed_row.task_id - )); -} - -#[test] -fn trainer_attempt_association_checks_the_executor_authority_not_request_origin() { - let mut fixture = TrainerAssociationFixture::new(); - let row = fixture.trainer_task(fixture.task_id, &fixture.spec); - let origin = machine_other_than(fixture.authority); - fixture - .store - .insert_remote_task(&row, &fixture.spec, origin, fixture.authority) - .unwrap(); - fixture - .store - .cas_status( - fixture.task_id, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - let evidence = fixture.evidence(); - assert!(fixture.bind(evidence).is_ok()); - - let mut wrong_executor = TrainerAssociationFixture::new(); - let row = wrong_executor.trainer_task(wrong_executor.task_id, &wrong_executor.spec); - let wrong_executor_machine = machine_other_than(wrong_executor.authority); - wrong_executor - .store - .insert_remote_task( - &row, - &wrong_executor.spec, - wrong_executor.authority, - wrong_executor_machine, - ) - .unwrap(); - wrong_executor - .store - .cas_status( - wrong_executor.task_id, - ProcessStatus::Queued, - ProcessStatus::Running, - ) - .unwrap() - .unwrap(); - let evidence = wrong_executor.evidence(); - assert!(matches!( - wrong_executor.bind(evidence), - Err(TrainerAttemptAssociationStoreError::IdentityMismatch { task_id }) - if task_id == wrong_executor.task_id - )); -} - -#[test] -fn trainer_attempt_association_rejects_duplicate_task_across_resources() { - let mut fixture = TrainerAssociationFixture::new(); - fixture.insert_accepted_running_task(); - let evidence = fixture.evidence(); - fixture.bind(evidence.clone()).unwrap(); - - let mut second_resource = resource(fixture.authority); - second_resource.registered_background_task = Some(fixture.task_id); - fixture - .store - .register_resource(fixture.authority, &second_resource) - .unwrap(); - assert!(matches!( - fixture.store.bind_trainer_attempt_association( - fixture.authority, - second_resource.id, - fixture.task_id, - evidence, - ), - Err(TrainerAttemptAssociationStoreError::TaskAlreadyAssociated { task_id }) - if task_id == fixture.task_id - )); -} - -#[test] -fn trainer_attempt_association_accepts_only_matching_awaiting_release_loan() { - let mut awaiting = TrainerAssociationFixture::new(); - awaiting.insert_accepted_running_task(); - awaiting.add_loan(awaiting_release_state(awaiting.task_id)); - let evidence = awaiting.evidence(); - assert!(awaiting.bind(evidence).is_ok()); - - let mut mismatched = TrainerAssociationFixture::new(); - mismatched.insert_accepted_running_task(); - mismatched.add_loan(awaiting_release_state(TaskId::new())); - let evidence = mismatched.evidence(); - assert!(matches!( - mismatched.bind(evidence), - Err(TrainerAttemptAssociationStoreError::ActiveLoanConflict { resource_id }) - if resource_id == mismatched.resource.id - )); - - let mut serving = TrainerAssociationFixture::new(); - serving.insert_accepted_running_task(); - serving.add_loan(LoanState::Active { - phase: LoanPhase::Serving { - return_context: ReturnContext::Stopped { - task_id: serving.task_id, - checkpoint_ref: "checkpoint".into(), - recovery_ref: "recovery".into(), - }, - current_request_id: RequestId::new(), - release_provenance: crate::store::unreceipted_release_provenance(), - }, - }); - let evidence = serving.evidence(); - assert!(matches!( - serving.bind(evidence), - Err(TrainerAttemptAssociationStoreError::ActiveLoanConflict { .. }) - )); - - let mut returning = TrainerAssociationFixture::new(); - returning.insert_accepted_running_task(); - returning.add_loan(LoanState::Active { - phase: LoanPhase::AwaitingReturn { - action_id: ActionId::new(), - return_context: ReturnContext::Idle, - }, - }); - let evidence = returning.evidence(); - assert!(matches!( - returning.bind(evidence), - Err(TrainerAttemptAssociationStoreError::ActiveLoanConflict { .. }) - )); - - let mut restoring = TrainerAssociationFixture::new(); - restoring.insert_accepted_running_task(); - restoring.add_loan(LoanState::Active { - phase: LoanPhase::Restoring { - action_id: ActionId::new(), - return_context: ReturnContext::Idle, - resume_task_id: TaskId::new(), - }, - }); - let evidence = restoring.evidence(); - assert!(matches!( - restoring.bind(evidence), - Err(TrainerAttemptAssociationStoreError::ActiveLoanConflict { .. }) - )); -} diff --git a/src/store/resource/trainer_association.rs b/src/store/resource/trainer_association.rs deleted file mode 100644 index bb9c8d8..0000000 --- a/src/store/resource/trainer_association.rs +++ /dev/null @@ -1,344 +0,0 @@ -//! Durable associations between a registered background task and its exact trainer attempt - -use rusqlite::{Connection, OptionalExtension, TransactionBehavior, params}; - -use super::{resource_task_row_matches, same_task_binding, select_authority_resource}; -use crate::domain::{ProcessStatus, TaskId, TaskRow}; -use crate::error::AppError; -use crate::machine::MachineId; -use crate::resource::command_shape::DirectSegmentCommandShape; -use crate::resource::ownership_lock::VerifiedTrainerAttempt; -use crate::resource::store::{TrainerAttemptAssociationStoreError, select_non_closed_loan}; -use crate::resource::{ - LoanPhase, LoanState, ResourceId, TrainerAttemptAssociation, TrainerAttemptAssociationProof, -}; -use crate::spec::NormalizedSpec; -use crate::store::Store; -use crate::submission::{ExecutorIdentity, NormalizedSpecSha256, normalized_spec_sha256}; - -/// Registered trainer task and accepted identity read for one bind or stop -#[derive(Debug)] -pub(super) struct TrainerAssociationBindSnapshot { - pub(super) task_row: TaskRow, - pub(super) normalized_spec: NormalizedSpec, - pub(super) normalized_spec_sha256: NormalizedSpecSha256, - pub(super) identity_state: ProcessStatus, -} - -impl TrainerAssociationBindSnapshot { - /// Whether the task row still runs the command of the accepted spec - pub(super) fn task_row_matches_spec(&self, task_id: TaskId) -> bool { - resource_task_row_matches(&self.task_row, task_id, &self.normalized_spec) - } -} - -/// One saved association with the exact JSON it was decoded from -pub(super) struct SavedTrainerAssociation { - pub(super) association: TrainerAttemptAssociation, - pub(super) json: String, -} - -impl Store { - /// Bind exact trainer evidence and direct-segment command shape to a registered task - /// - /// The accepted identity and running task row are read before file-system validation - /// A later IMMEDIATE transaction rechecks their immutable binding before insertion - /// File paths may change between the shape check and that transaction - /// Exact retries return the saved association without checking the current file-system layout - /// This shape does not prove live lock use or permit release completion or `Serving` - pub(crate) fn bind_trainer_attempt_association( - &mut self, - authority_machine: MachineId, - resource_id: ResourceId, - task_id: TaskId, - verified_attempt: VerifiedTrainerAttempt, - ) -> Result { - let (preflight, association) = { - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Deferred)?; - let snapshot = - trainer_association_bind_snapshot(&tx, authority_machine, resource_id, task_id)?; - let association = TrainerAttemptAssociation::from_components( - resource_id, - authority_machine, - task_id, - verified_attempt, - snapshot.normalized_spec_sha256, - ) - .map_err(|reason| { - TrainerAttemptAssociationStoreError::InvalidStoredAssociation(reason.into()) - })?; - - if let Some(saved) = trainer_association_by_task(&tx, task_id)? { - let binding_unchanged = snapshot.task_row_matches_spec(task_id); - let saved = saved_association_retry(saved, &association, binding_unchanged)?; - tx.commit()?; - return Ok(saved); - } - if !snapshot.task_row_matches_spec(task_id) { - return Err( - TrainerAttemptAssociationStoreError::NormalizedSpecMismatch { task_id }, - ); - } - ensure_bindable(&tx, resource_id, task_id, &snapshot)?; - - tx.commit()?; - (snapshot, association) - }; - - // keep canonical path checks outside the IMMEDIATE write transaction - DirectSegmentCommandShape::validate_binding( - &preflight.normalized_spec, - &preflight.task_row, - task_id, - preflight.normalized_spec_sha256, - association.verified_attempt(), - )?; - - let tx = self - .conn - .transaction_with_behavior(TransactionBehavior::Immediate)?; - let current = - trainer_association_bind_snapshot(&tx, authority_machine, resource_id, task_id)?; - let binding_unchanged = current.normalized_spec_sha256 == preflight.normalized_spec_sha256 - && same_task_binding(¤t.task_row, &preflight.task_row) - && current.task_row_matches_spec(task_id); - if let Some(saved) = trainer_association_by_task(&tx, task_id)? { - let saved = saved_association_retry(saved, &association, binding_unchanged)?; - tx.commit()?; - return Ok(saved); - } - if !binding_unchanged { - return Err(TrainerAttemptAssociationStoreError::BindingChanged { task_id }); - } - ensure_bindable(&tx, resource_id, task_id, ¤t)?; - - tx.execute( - "INSERT INTO trainer_attempt_associations ( - resource_id, authority_machine, task_id, association_json - ) VALUES (?1, ?2, ?3, ?4)", - params![ - resource_id.as_uuid().to_string(), - authority_machine.as_uuid().to_string(), - task_id.to_string(), - trainer_association_json(&association)?, - ], - )?; - tx.commit()?; - Ok(association) - } - - /// Read one historical association by exact task without authorizing release or execution - pub(crate) fn trainer_attempt_association_for_task_for_authority( - &self, - authority_machine: MachineId, - task_id: TaskId, - ) -> Result, TrainerAttemptAssociationStoreError> { - let Some(saved) = trainer_association_by_task(&self.conn, task_id)? else { - return Ok(None); - }; - select_authority_resource(&self.conn, authority_machine, saved.resource_id())?; - if saved.authority_machine() != authority_machine { - return Err(association_authority_mismatch()); - } - - Ok(Some(saved)) - } -} - -fn association_authority_mismatch() -> TrainerAttemptAssociationStoreError { - TrainerAttemptAssociationStoreError::InvalidStoredAssociation( - "association authority does not match its resource".into(), - ) -} - -/// Classify an association already saved for the task being bound -/// -/// An association of another resource owns the task. The same resource must hold -/// exactly the expected association over an unchanged task binding, or the retry conflicts -fn saved_association_retry( - saved: TrainerAttemptAssociation, - expected: &TrainerAttemptAssociation, - binding_unchanged: bool, -) -> Result { - if saved.resource_id() != expected.resource_id() { - return Err(TrainerAttemptAssociationStoreError::TaskAlreadyAssociated { - task_id: expected.task_id(), - }); - } - if saved != *expected || !binding_unchanged { - return Err(TrainerAttemptAssociationStoreError::Conflict { - resource_id: expected.resource_id(), - }); - } - - Ok(saved) -} - -/// Require a running registered task and a loan that is free or awaits this task's release -fn ensure_bindable( - conn: &Connection, - resource_id: ResourceId, - task_id: TaskId, - snapshot: &TrainerAssociationBindSnapshot, -) -> Result<(), TrainerAttemptAssociationStoreError> { - let loan_awaits_other_work = select_non_closed_loan(conn, resource_id)?.is_some_and(|loan| { - !matches!( - loan.state, - LoanState::Active { - phase: LoanPhase::AwaitingRelease { - observed_background_task, - .. - } - } if observed_background_task == task_id - ) - }); - if loan_awaits_other_work { - return Err(TrainerAttemptAssociationStoreError::ActiveLoanConflict { resource_id }); - } - if snapshot.task_row.status() != ProcessStatus::Running { - return Err(TrainerAttemptAssociationStoreError::TaskNotRunning { - task_id, - state: snapshot.task_row.status().to_string(), - }); - } - if snapshot.identity_state != ProcessStatus::Running { - return Err(TrainerAttemptAssociationStoreError::IdentityNotRunning { task_id }); - } - - Ok(()) -} - -/// Read the registered task row and its accepted identity on this authority -pub(super) fn trainer_association_bind_snapshot( - conn: &Connection, - authority_machine: MachineId, - resource_id: ResourceId, - task_id: TaskId, -) -> Result { - let resource = select_authority_resource(conn, authority_machine, resource_id)?; - if resource.registered_background_task != Some(task_id) { - return Err(TrainerAttemptAssociationStoreError::TaskNotRegistered { task_id }); - } - - let task_row = crate::store::task_by_id_on(conn, task_id)? - .ok_or(TrainerAttemptAssociationStoreError::TaskMissing { task_id })?; - let identity = crate::store::identity::executor_identity_on(conn, task_id)? - .ok_or(TrainerAttemptAssociationStoreError::IdentityMissing { task_id })?; - let ExecutorIdentity::Accepted(record) = identity else { - return Err(TrainerAttemptAssociationStoreError::IdentityNotAccepted { task_id }); - }; - // the trainer's callback origin may be any supervisor machine - if !record.is_executed_by(task_id, authority_machine) { - return Err(TrainerAttemptAssociationStoreError::IdentityMismatch { task_id }); - } - - let normalized_spec = record - .current_spec() - .ok_or(TrainerAttemptAssociationStoreError::NormalizedSpecMissing { task_id })? - .clone(); - let normalized_spec_sha256 = - normalized_spec_sha256(&normalized_spec).map_err(AppError::from)?; - - Ok(TrainerAssociationBindSnapshot { - task_row, - normalized_spec, - normalized_spec_sha256, - identity_state: record.state, - }) -} - -/// Read the association of one resource and task with its saved JSON -pub(super) fn saved_trainer_association_on( - conn: &Connection, - resource_id: ResourceId, - task_id: TaskId, -) -> Result, TrainerAttemptAssociationStoreError> { - select_association( - conn, - "SELECT resource_id, authority_machine, task_id, association_json - FROM trainer_attempt_associations WHERE resource_id=?1 AND task_id=?2", - params![resource_id.as_uuid().to_string(), task_id.to_string()], - ) -} - -pub(super) fn trainer_association_by_resource_and_task( - conn: &Connection, - resource_id: ResourceId, - task_id: TaskId, -) -> Result, TrainerAttemptAssociationStoreError> { - Ok(saved_trainer_association_on(conn, resource_id, task_id)?.map(|saved| saved.association)) -} - -pub(super) fn trainer_association_by_task( - conn: &Connection, - task_id: TaskId, -) -> Result, TrainerAttemptAssociationStoreError> { - let saved = select_association( - conn, - "SELECT resource_id, authority_machine, task_id, association_json - FROM trainer_attempt_associations WHERE task_id=?1", - params![task_id.to_string()], - )?; - Ok(saved.map(|saved| saved.association)) -} - -fn select_association( - conn: &Connection, - sql: &str, - params: impl rusqlite::Params, -) -> Result, TrainerAttemptAssociationStoreError> { - let row = conn - .query_row(sql, params, |row| { - Ok(( - row.get::<_, String>(0)?, - row.get::<_, String>(1)?, - row.get::<_, String>(2)?, - row.get::<_, String>(3)?, - )) - }) - .optional()?; - let Some((resource_id, authority_machine, task_id, json)) = row else { - return Ok(None); - }; - let association = - decode_trainer_association(&resource_id, &authority_machine, &task_id, &json)?; - Ok(Some(SavedTrainerAssociation { association, json })) -} - -pub(super) fn trainer_association_json( - association: &TrainerAttemptAssociation, -) -> Result { - serde_json::to_string(&TrainerAttemptAssociationProof::from(association)) - .map_err(AppError::from) - .map_err(Into::into) -} - -fn decode_trainer_association( - resource_id: &str, - authority_machine: &str, - task_id: &str, - association_json: &str, -) -> Result { - let invalid = - |reason: String| TrainerAttemptAssociationStoreError::InvalidStoredAssociation(reason); - let uuid = - |value: &str| uuid::Uuid::parse_str(value).map_err(|error| invalid(error.to_string())); - let resource_id = - ResourceId::from_uuid(uuid(resource_id)?).map_err(|error| invalid(error.to_string()))?; - let authority_machine = MachineId::from_uuid(uuid(authority_machine)?); - let task_id = TaskId(uuid(task_id)?); - let association: TrainerAttemptAssociationProof = - serde_json::from_str(association_json).map_err(|error| invalid(error.to_string()))?; - if association.resource_id != resource_id - || association.authority_machine != authority_machine - || association.task_id != task_id - { - return Err(invalid( - "JSON identities do not match the association row".into(), - )); - } - - TrainerAttemptAssociation::try_from(association).map_err(|reason| invalid(reason.into())) -} diff --git a/src/store/resource/trainer_lock.rs b/src/store/resource/trainer_lock.rs deleted file mode 100644 index f333133..0000000 --- a/src/store/resource/trainer_lock.rs +++ /dev/null @@ -1,251 +0,0 @@ -//! Exact trainer ownership-lock release for background slot and restore transitions -//! -//! The maintained direct-segment wrapper starts its GPU worker in a new session, -//! so a confirmed exit of the wrapper's process group does not show that the -//! worker exited. The worker holds the runtime `.segment.lock` until it exits -//! Before a transition lets a new GPU worker start, or closes a Restoring loan, -//! after such a task ended, the authority takes that exact lock. The immutable -//! trainer-attempt association names the lock, and the authority holds it until -//! the transaction that records the transition commits - -use std::path::PathBuf; - -use rusqlite::Connection; - -use super::trainer_association::trainer_association_by_resource_and_task; -use crate::domain::{TaskId, WorkExitEvidence}; -use crate::machine::MachineId; -use crate::resource::command_shape::direct_segment_runtime_root; -use crate::resource::foreground::{self, CommandOwnershipContract}; -use crate::resource::ownership_lock::{ - OwnershipLockGuard, OwnershipLockIdentity, OwnershipLockProbe, OwnershipLockProbeError, - probe_segment_ownership_lock, verify_ownership_lock_guard, -}; -use crate::resource::store::{ResourceStoreError, TrainerAttemptAssociationStoreError}; -use crate::resource::{Resource, ResourceTaskOwnershipRisk}; -use crate::spec::NormalizedSpec; -use crate::store::identity::executor_identity_on; -use crate::submission::{ExecutorIdentity, normalized_spec_sha256}; - -/// Why an ended task has no verified release of its trainer ownership lock -#[derive(Debug, thiserror::Error)] -pub(crate) enum TrainerLockReleaseGap { - /// The task has no accepted identity with a spec on this authority - #[error("task {task_id} has no accepted run records on this authority")] - RunRecordsMissing { - /// Task whose records are missing - task_id: TaskId, - }, - /// The task command has no ownership contract that a release can verify - #[error("task {task_id} command ownership cannot be verified: {risk:?}")] - UnsupportedCommand { - /// Ended task - task_id: TaskId, - /// Why the command can hide or detach its work - risk: ResourceTaskOwnershipRisk, - }, - /// No trainer-attempt association names the lock that the task used - #[error("task {task_id} has no trainer-attempt association")] - AssociationMissing { - /// Task whose association names the lock - task_id: TaskId, - }, - /// The association does not name the same run or runtime root as the ended task - #[error("trainer-attempt association of task {task_id} does not match the ended run")] - AssociationMismatch { - /// Task whose association names the lock - task_id: TaskId, - }, - /// A process still holds the exact saved lock - #[error("trainer ownership lock of task {task_id} is still held")] - OwnershipHeld { - /// Task whose association names the lock - task_id: TaskId, - }, - /// The exact saved lock could not be verified or held - #[error(transparent)] - OwnershipLock(#[from] OwnershipLockProbeError), -} - -/// Failure to prove or hold one trainer ownership-lock release -#[derive(Debug)] -pub(super) enum TrainerLockReleaseError { - /// The saved records or the lock do not prove the release - Unproven(TrainerLockReleaseGap), - /// SQLite or stored data failed, so the result is unknown - Store(ResourceStoreError), -} - -impl From for TrainerLockReleaseError { - fn from(gap: TrainerLockReleaseGap) -> Self { - Self::Unproven(gap) - } -} - -impl From for TrainerLockReleaseError { - fn from(error: ResourceStoreError) -> Self { - Self::Store(error) - } -} - -/// Witness that the ownership contract of an ended task's work requires -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(super) enum EndedTaskWitness { - /// Native foreground command; the confirmed process-group exit is the witness - ProcessGroupExit, - /// Maintained direct-segment trainer; only its released lock is the witness - TrainerLock, - /// Container workload; the exited container, removed and confirmed absent, is the witness - ContainerRemoved, -} - -impl EndedTaskWitness { - /// Whether the task layer's exit evidence is the kind this witness needs - /// - /// The trainer lock also needs the wrapper's confirmed process-group exit; - /// the caller then holds the lock itself - pub(super) fn accepts(self, evidence: &WorkExitEvidence) -> bool { - match self { - Self::ProcessGroupExit | Self::TrainerLock => { - *evidence == WorkExitEvidence::ProcessGroupExited - } - Self::ContainerRemoved => { - matches!(evidence, WorkExitEvidence::ContainerRemoved { .. }) - } - } - } -} - -/// Exclusive hold on one exact released trainer lock -/// -/// Keep this value alive until the transaction that relies on it commits, and -/// call [`HeldTrainerRelease::recheck`] just before that commit -#[must_use = "keep the lock held until the transaction that relies on it commits"] -#[derive(Debug)] -pub(super) struct HeldTrainerRelease { - runtime_root: PathBuf, - identity: OwnershipLockIdentity, - guard: OwnershipLockGuard, -} - -impl HeldTrainerRelease { - /// Check that the guard still holds the file at the exact saved lock path - pub(super) fn recheck(&self) -> Result<(), TrainerLockReleaseGap> { - verify_ownership_lock_guard(&self.runtime_root, self.identity, &self.guard) - .map_err(TrainerLockReleaseGap::from) - } -} - -/// Classify the witness that an ended task needs from its accepted work -pub(super) fn ended_task_witness( - conn: &Connection, - authority_machine: MachineId, - task_id: TaskId, -) -> Result { - let spec = accepted_spec(conn, authority_machine, task_id)?; - match CommandOwnershipContract::for_return_work(&spec.workload) { - Ok(CommandOwnershipContract::ForegroundExecutable) => { - Ok(EndedTaskWitness::ProcessGroupExit) - } - Ok(CommandOwnershipContract::DirectSegmentTrainer) => Ok(EndedTaskWitness::TrainerLock), - Ok(CommandOwnershipContract::Container) => Ok(EndedTaskWitness::ContainerRemoved), - Err(risk) => Err(TrainerLockReleaseGap::UnsupportedCommand { task_id, risk }.into()), - } -} - -/// Take the exact trainer lock that an ended direct-segment task used -/// -/// `witness_task` owns the association that saved the lock identity. It is the -/// ended task itself, or the stopped run that a same-run resume continued in the -/// same runtime root. Both accepted commands must name that `--runtime-root`, -/// and the witness spec must still match its association. A held, missing, -/// replaced, or unverifiable lock fails closed -pub(super) fn hold_released_trainer_lock( - conn: &Connection, - resource: &Resource, - ended_task: TaskId, - witness_task: TaskId, -) -> Result { - let authority_machine = resource.authority_machine(); - let mismatch = TrainerLockReleaseGap::AssociationMismatch { - task_id: witness_task, - }; - let association = trainer_association_by_resource_and_task(conn, resource.id, witness_task) - .map_err(|error| match error { - TrainerAttemptAssociationStoreError::Storage(error) => { - TrainerLockReleaseError::Store(error.into()) - } - TrainerAttemptAssociationStoreError::Resource(error) => { - TrainerLockReleaseError::Store(error) - } - // a stored association that no longer decodes cannot name a lock - _ => TrainerLockReleaseGap::AssociationMismatch { - task_id: witness_task, - } - .into(), - })? - .ok_or(TrainerLockReleaseGap::AssociationMissing { - task_id: witness_task, - })?; - if association.authority_machine() != authority_machine - || association.resource_id() != resource.id - || association.task_id() != witness_task - { - return Err(mismatch.into()); - } - - let witness_spec = accepted_spec(conn, authority_machine, witness_task)?; - let witness_digest = normalized_spec_sha256(&witness_spec) - .map_err(|error| ResourceStoreError::TaskRow(error.into()))?; - if witness_digest != association.normalized_spec_sha256() { - return Err(mismatch.into()); - } - let ended_spec = if ended_task == witness_task { - witness_spec.clone() - } else { - accepted_spec(conn, authority_machine, ended_task)? - }; - let runtime_root = |spec: &NormalizedSpec| { - foreground::task_command(spec).and_then(direct_segment_runtime_root) - }; - let witness_root = runtime_root(&witness_spec); - if witness_root.is_none() || witness_root != runtime_root(&ended_spec) { - return Err(mismatch.into()); - } - - let attempt = association.verified_attempt(); - let runtime_root = attempt.canonical_runtime_root().to_path_buf(); - let identity = attempt.ownership_lock_identity(); - match probe_segment_ownership_lock(&runtime_root, identity) { - OwnershipLockProbe::ExactOwnershipReleased(guard) => Ok(HeldTrainerRelease { - runtime_root, - identity, - guard, - }), - OwnershipLockProbe::OwnershipHeld => Err(TrainerLockReleaseGap::OwnershipHeld { - task_id: witness_task, - } - .into()), - OwnershipLockProbe::Attention(error) => Err(TrainerLockReleaseGap::from(error).into()), - } -} - -/// Read the spec of one task's accepted identity executed by this authority -fn accepted_spec( - conn: &Connection, - authority_machine: MachineId, - task_id: TaskId, -) -> Result { - let identity = executor_identity_on(conn, task_id).map_err(ResourceStoreError::from)?; - let Some(ExecutorIdentity::Accepted(record)) = identity else { - return Err(TrainerLockReleaseGap::RunRecordsMissing { task_id }.into()); - }; - if !record.is_executed_by(task_id, authority_machine) { - return Err(TrainerLockReleaseGap::RunRecordsMissing { task_id }.into()); - } - - record - .current_spec() - .cloned() - .ok_or_else(|| TrainerLockReleaseGap::RunRecordsMissing { task_id }.into()) -} diff --git a/src/submission.rs b/src/submission.rs index 26f5b42..1969e10 100644 --- a/src/submission.rs +++ b/src/submission.rs @@ -7,14 +7,8 @@ use serde::{Deserialize, Serialize}; use uuid::Uuid; use crate::dependency::{HeldCancellation, TaskDependencies}; -use crate::digest::Sha256Digest; use crate::domain::{ProcessStatus, TaskEnv, TaskId, TaskStatus, ThreadId}; use crate::machine::MachineId; -use crate::resource::background_launch::{BackgroundLaunchBinding, ResourceBackgroundRejection}; -use crate::resource::bound_action::{ - ResourceActionKind, ResourceActionLaunch, ResourceActionRejection, -}; -use crate::resource::{ResourceId, SupervisorActionAuthority}; use crate::spec::NormalizedSpec; /// The executable saved for callbacks owned by the origin @@ -194,55 +188,6 @@ pub struct CallbackContext { pub codex: CallbackExecutable, } -/// Durable phase of one resource-routed task request -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceRoutePhase { - /// The queue request may have reached the resource authority - AcceptanceUnknown, - /// The authority accepted the request, but it has not started a task - Waiting, - /// The first queued executor event established task activation - Activated, - /// The authority confirmed cancellation before task activation - CancelledBeforeLaunch, - /// The resource authority rejected the queue request - Rejected { - /// Durable authority rejection reason - reason: String, - }, -} - -/// Durable phase of one origin route bound to a resource action -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceActionRoutePhase { - /// The launch may have reached the resource authority - AcceptanceUnknown, - /// The authority accepted the exact action-bound task - Accepted, - /// The authority refused the launch and wrote no task records - Rejected { - /// Definitive authority reason - reason: ResourceActionRejection, - }, -} - -/// Durable phase of one origin route for a remote first background launch -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceBackgroundRoutePhase { - /// The launch may have reached the resource authority - AcceptanceUnknown, - /// The authority accepted the exact launch, by its receipt or its first queued event - Accepted, - /// The authority refused the launch and wrote no task records - Rejected { - /// Definitive authority reason - reason: ResourceBackgroundRejection, - }, -} - /// Durable phase of one origin route held until its dependencies succeed /// /// A held route has no task row and no executor identity. Its dependencies @@ -277,138 +222,15 @@ impl HeldPhase { } } -/// Resource action that owns one action-bound origin route -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceActionRouteBinding { - /// Kind of task that the action binds - pub kind: ResourceActionKind, - /// Exact authority, resource, loan, action, revision, and supervisor assignment - pub authority: SupervisorActionAuthority, -} - -/// Outcome in a definitive response to a resource queue request -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceQueueOutcome { - /// The resource request is durably waiting in the authority queue - Waiting, - /// The resource authority rejected the request before queue acceptance - Rejected { - /// Durable authority rejection reason - reason: String, - }, -} - -/// Definitive response bound to the exact resource route identity -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceQueueReceipt { - /// Caller retry UUID - pub request: RequestId, - /// Preallocated global task UUID - pub task: TaskId, - /// Machine that owns the callback route - pub origin_machine: MachineId, - /// Fixed resource authority and execution owner - pub authority_machine: MachineId, - /// Resource that owns the waiting request - pub resource: ResourceId, - /// Definitive queue outcome - pub outcome: ResourceQueueOutcome, -} - -/// SHA-256 digest of the compact JSON encoding of a normalized spec -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] -#[serde(transparent)] -pub struct NormalizedSpecSha256(Sha256Digest); - -/// Compute the SHA-256 digest used to bind resource-route proofs to normalized content -/// -/// The digest covers `serde_json::to_vec(spec)` exactly. Keep this helper for both -/// origin proof generation and authority-side queue request verification -pub fn normalized_spec_sha256( - spec: &NormalizedSpec, -) -> Result { - let encoded = serde_json::to_vec(spec)?; - Ok(NormalizedSpecSha256(Sha256Digest::of(encoded))) -} - -/// Why a resource request is no longer eligible for pre-activation cancellation -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceCancellationIneligibleReason { - /// The authority accepted the execution task before cancellation won - Activated, - /// The authority rejected the resource request before activation - Rejected { - /// Durable authority rejection reason - reason: String, - }, - /// The resource request reached a terminal state before cancellation won - Terminal, -} - -/// A definitive authority result for one resource cancellation attempt -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] -pub enum ResourceCancellationOutcome { - /// The authority retained a prevention record before queue acceptance - PreventedBeforeAcceptance, - /// The authority cancelled the queued request before task activation - CancelledBeforeLaunch, - /// The request no longer supports a claim that cancellation prevented launch - NotEligible { - /// Exact authority-side reason that prevented pre-activation cancellation - reason: ResourceCancellationIneligibleReason, - }, -} - -/// Definitive cancellation response bound to the exact resource route identity -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceCancellationReceipt { - /// Origin-owned cancellation identity - pub cancellation: Uuid, - /// Machine that persisted the cancellation intent - pub requester_machine: MachineId, - /// Caller retry UUID - pub request: RequestId, - /// Preallocated global task UUID - pub task: TaskId, - /// Machine that owns the callback route - pub origin_machine: MachineId, - /// Fixed resource authority and execution owner - pub authority_machine: MachineId, - /// Resource that owns the waiting request - pub resource: ResourceId, - /// Resource route phase captured when the origin persisted the intent - pub target_phase: ResourceRoutePhase, - /// Definitive pre-activation result - pub outcome: ResourceCancellationOutcome, -} - -/// Why an origin route cannot represent a resource-waiting request +/// Why a saved origin route breaks its invariants #[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] -pub enum ResourceRouteError { - /// Resource work must use the resource authority, not an explicit spec machine - #[error("resource route spec must not name an execution machine")] - ExplicitMachine, - /// Resource routes accept only bounded command or container workloads - #[error("resource route requires a command or container workload")] - NonCommandWorkload, +pub enum RouteError { /// The route and normalized spec must retain one exact thread identity - #[error("resource route thread does not match its normalized spec")] + #[error("route thread does not match its normalized spec")] ThreadMismatch, - /// Resource callbacks need absolute origin-local callback paths - #[error("resource callback paths must be absolute")] - InvalidCallbackContext, - /// Only an executor event can establish resource task activation - #[error("resource route phase does not agree with its event cursor")] + /// A held route cannot have accepted executor events + #[error("route phase does not agree with its event cursor")] InvalidEventCursor, - /// An action-bound route must be owned by the assigned supervisor and executed by the authority - #[error("resource action route owners do not match its action authority")] - ActionOwnerMismatch, /// A migrated local identity names different origin and execution machines #[error("migrated local identity must have the same origin and execution machine")] InvalidMigratedIdentity, @@ -427,41 +249,11 @@ pub enum SubmissionState { Accepted, /// The executor durably rejected this identity Rejected { reason: String }, - /// This task identity waits for one resource authority to activate it - Resource { - /// Resource identity, retained after activation - resource: ResourceId, - /// Resource-specific acceptance and activation phase - phase: ResourceRoutePhase, - }, - /// This task identity is bound to one resource action on a remote authority - /// - /// Recovery retries the same identities; the generic abandon path never - /// resolves it, because abandoning could strand the action - ResourceAction { - /// Exact action that owns the task - binding: ResourceActionRouteBinding, - /// Supervisor choice that the route launches - launch: Box, - /// Action-specific acceptance phase - phase: ResourceActionRoutePhase, - }, /// This task identity waits on its origin for its dependencies to succeed Held { /// Hold, launch, or cancellation phase phase: HeldPhase, }, - /// This task identity is the first background launch of a remote authority - /// - /// It has no loan or action. Recovery retries the same identities, and the - /// generic abandon path never resolves it, so a lost reply cannot start a - /// second trainer - ResourceBackground { - /// Supervisor assignment and resource revision that chose the launch - binding: BackgroundLaunchBinding, - /// Launch-specific acceptance phase - phase: ResourceBackgroundRoutePhase, - }, } /// Origin-owned request mapping and callback route @@ -569,59 +361,6 @@ pub struct NewHeldRoute { pub spec: NormalizedSpec, } -/// Exact identity and origin-owned context used to create a resource route -#[derive(Debug, Clone)] -pub struct NewResourceRoute { - /// Caller retry UUID allocated before the first authority request - pub request: RequestId, - /// Global task UUID allocated before the first authority request - pub task: TaskId, - /// Machine that owns the requesting thread and callback context - pub origin_machine: MachineId, - /// Fixed resource authority and execution owner - pub authority_machine: MachineId, - /// Exact original Codex thread - pub thread: ThreadId, - /// Exact origin-only callback context - pub callback: CallbackContext, - /// Normalized command spec with no explicit execution machine - pub spec: NormalizedSpec, - /// Resource identity that remains on the route after activation - pub resource: ResourceId, -} - -/// Exact identity and origin-owned context used to create an action-bound route -#[derive(Debug, Clone)] -pub struct NewResourceActionRoute { - /// Stable retry identity prepared by the authority or chosen by the supervisor - pub request: RequestId, - /// Preallocated global task identity - pub task: TaskId, - /// Supervisor-machine callback context - pub callback: CallbackContext, - /// Canonical normalized spec prepared by the authority - pub spec: NormalizedSpec, - /// Action that owns the task - pub binding: ResourceActionRouteBinding, - /// Supervisor choice that the route launches - pub launch: ResourceActionLaunch, -} - -/// Exact identity and supervisor-machine context used to create a background launch route -#[derive(Debug, Clone)] -pub struct NewResourceBackgroundRoute { - /// Caller retry identity - pub request: RequestId, - /// Global task identity allocated before the first authority request - pub task: TaskId, - /// Supervisor-machine callback context - pub callback: CallbackContext, - /// Full normalized trainer spec - pub spec: NormalizedSpec, - /// Supervisor assignment and resource revision that chose the launch - pub binding: BackgroundLaunchBinding, -} - impl OriginRoute { /// Borrow normalized request content when this route came from a submit request #[must_use] @@ -661,228 +400,33 @@ impl OriginRoute { } } - /// Create an initial origin route for one resource-waiting command - /// - /// The request and task IDs are supplied by the caller so this exact route - /// can be stored before the first authority request is sent - pub fn new_resource_waiting(input: NewResourceRoute) -> Result { - let NewResourceRoute { - request, - task, - origin_machine, - authority_machine, - thread, - callback, - spec, - resource, - } = input; - let route = Self { - request, - task, - origin_machine, - execution_machine: authority_machine, - thread, - callback, - spec: spec.into(), - submission: SubmissionState::Resource { - resource, - phase: ResourceRoutePhase::AcceptanceUnknown, - }, - last_execution_state: None, - last_updated_at: Some(Utc::now()), - last_accepted_seq: 0, - last_settled_seq: 0, - }; - route.validate()?; - Ok(route) - } - - /// Create the initial route for one action-bound task before its launch is sent - /// - /// The supervisor machine owns callbacks and the authority executes the task - pub fn new_resource_action(input: NewResourceActionRoute) -> Result { - let NewResourceActionRoute { - request, - task, - callback, - spec, - binding, - launch, - } = input; - let route = Self { - request, - task, - origin_machine: binding.authority.supervisor.machine, - execution_machine: binding.authority.authority_machine, - thread: binding.authority.supervisor.thread, - callback, - spec: spec.into(), - submission: SubmissionState::ResourceAction { - binding, - launch: Box::new(launch), - phase: ResourceActionRoutePhase::AcceptanceUnknown, - }, - last_execution_state: None, - last_updated_at: Some(Utc::now()), - last_accepted_seq: 0, - last_settled_seq: 0, - }; - route.validate()?; - Ok(route) - } - - /// Create the initial route for one remote first background launch before it is sent - /// - /// The supervisor machine owns callbacks and the authority executes the task - pub fn new_resource_background( - input: NewResourceBackgroundRoute, - ) -> Result { - let NewResourceBackgroundRoute { - request, - task, - callback, - spec, - binding, - } = input; - let route = Self { - request, - task, - origin_machine: binding.origin_machine(), - execution_machine: binding.execution_machine(), - thread: binding.assignment.supervisor.thread, - callback, - spec: spec.into(), - submission: SubmissionState::ResourceBackground { - binding, - phase: ResourceBackgroundRoutePhase::AcceptanceUnknown, - }, - last_execution_state: None, - last_updated_at: Some(Utc::now()), - last_accepted_seq: 0, - last_settled_seq: 0, - }; - route.validate()?; - Ok(route) - } - - /// Validate resource-route invariants while leaving direct routes unchanged - pub fn validate(&self) -> Result<(), ResourceRouteError> { + /// Validate migrated and held route invariants while leaving direct routes unchanged + pub fn validate(&self) -> Result<(), RouteError> { if !self .spec .valid_for_owners(self.origin_machine, self.execution_machine) { - return Err(ResourceRouteError::InvalidMigratedIdentity); + return Err(RouteError::InvalidMigratedIdentity); } if matches!(&self.spec, PersistedSpec::MigratedLocal) && !matches!(&self.submission, SubmissionState::Accepted) { - return Err(ResourceRouteError::InvalidMigratedRouteState); + return Err(RouteError::InvalidMigratedRouteState); } - let phase = match &self.submission { - SubmissionState::Resource { phase, .. } => phase, - SubmissionState::ResourceAction { - binding, - launch, - phase, - } => { - return self.validate_action_route(binding, launch, phase); - } - SubmissionState::ResourceBackground { binding, phase } => { - return self.validate_background_route(binding, phase); - } - SubmissionState::Held { phase } => return self.validate_held_route(phase), + match &self.submission { + SubmissionState::Held { phase } => self.validate_held_route(phase), SubmissionState::AcceptanceUnknown | SubmissionState::Accepted - | SubmissionState::Rejected { .. } => return Ok(()), - }; - self.validate_command_route()?; - match phase { - ResourceRoutePhase::Activated - if self.last_accepted_seq == 0 || self.last_execution_state.is_none() => - { - Err(ResourceRouteError::InvalidEventCursor) - } - ResourceRoutePhase::AcceptanceUnknown - | ResourceRoutePhase::Waiting - | ResourceRoutePhase::CancelledBeforeLaunch - | ResourceRoutePhase::Rejected { .. } - if self.last_accepted_seq != 0 || self.last_execution_state.is_some() => - { - Err(ResourceRouteError::InvalidEventCursor) - } - _ => Ok(()), + | SubmissionState::Rejected { .. } => Ok(()), } } - fn validate_action_route( - &self, - binding: &ResourceActionRouteBinding, - launch: &ResourceActionLaunch, - phase: &ResourceActionRoutePhase, - ) -> Result<(), ResourceRouteError> { - let authority = &binding.authority; - if self.origin_machine != authority.supervisor.machine - || self.execution_machine != authority.authority_machine - || self.origin_machine == self.execution_machine - || self.thread != authority.supervisor.thread - { - return Err(ResourceRouteError::ActionOwnerMismatch); - } - self.validate_command_route()?; - // a supervisor-supplied return command is the route content itself - let chosen = match launch { - ResourceActionLaunch::Return { work } => work - .supervisor_spec() - .map(crate::resource::CommandSpec::as_normalized), - ResourceActionLaunch::ReleaseWatcher { .. } => None, - }; - let same_content = chosen.is_none_or(|chosen| self.spec.current() == Some(chosen)); - if launch.kind() != binding.kind || !same_content { - return Err(ResourceRouteError::ActionOwnerMismatch); - } - match phase { - // only the first queued executor event or an authority receipt accepts the route - ResourceActionRoutePhase::AcceptanceUnknown - | ResourceActionRoutePhase::Rejected { .. } - if self.last_accepted_seq != 0 || self.last_execution_state.is_some() => - { - Err(ResourceRouteError::InvalidEventCursor) - } - _ => Ok(()), - } - } - - fn validate_background_route( - &self, - binding: &BackgroundLaunchBinding, - phase: &ResourceBackgroundRoutePhase, - ) -> Result<(), ResourceRouteError> { - if self.origin_machine != binding.origin_machine() - || self.execution_machine != binding.execution_machine() - || self.origin_machine == self.execution_machine - || self.thread != binding.assignment.supervisor.thread - { - return Err(ResourceRouteError::ActionOwnerMismatch); - } - self.validate_command_route()?; - match phase { - // only the first queued executor event or an authority receipt accepts the route - ResourceBackgroundRoutePhase::AcceptanceUnknown - | ResourceBackgroundRoutePhase::Rejected { .. } - if self.last_accepted_seq != 0 || self.last_execution_state.is_some() => - { - Err(ResourceRouteError::InvalidEventCursor) - } - _ => Ok(()), - } - } - - fn validate_held_route(&self, phase: &HeldPhase) -> Result<(), ResourceRouteError> { + fn validate_held_route(&self, phase: &HeldPhase) -> Result<(), RouteError> { let Some(spec) = self.spec.current() else { - return Err(ResourceRouteError::InvalidMigratedRouteState); + return Err(RouteError::InvalidMigratedRouteState); }; if self.thread != spec.thread { - return Err(ResourceRouteError::ThreadMismatch); + return Err(RouteError::ThreadMismatch); } // no executor event can reach a route that never launched; a // cancelled one carries only the origin's own terminal event @@ -894,162 +438,12 @@ impl OriginRoute { || self.last_settled_seq > self.last_accepted_seq || self.last_execution_state.is_some() { - return Err(ResourceRouteError::InvalidEventCursor); - } - Ok(()) - } - - /// Shared checks for routes that carry one bounded command or container without a spec machine - fn validate_command_route(&self) -> Result<(), ResourceRouteError> { - let Some(spec) = self.spec.current() else { - return Err(ResourceRouteError::NonCommandWorkload); - }; - if spec.machine.is_some() { - return Err(ResourceRouteError::ExplicitMachine); - } - if matches!(&spec.workload, crate::spec::NormalizedWorkload::Agent(_)) { - return Err(ResourceRouteError::NonCommandWorkload); - } - if self.thread != spec.thread { - return Err(ResourceRouteError::ThreadMismatch); - } - if !self.callback.cwd.is_absolute() - || self - .callback - .codex - .path() - .is_some_and(|path| !path.is_absolute()) - { - return Err(ResourceRouteError::InvalidCallbackContext); - } - if self.last_settled_seq > self.last_accepted_seq { - return Err(ResourceRouteError::InvalidEventCursor); + return Err(RouteError::InvalidEventCursor); } Ok(()) } } -/// Safe proof of the resource route saved by its origin -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceRouteProof { - /// Caller retry UUID - pub request: RequestId, - /// Preallocated global task UUID - pub task: TaskId, - /// Machine that owns the callback route - pub origin_machine: MachineId, - /// Fixed resource authority and execution owner - pub authority_machine: MachineId, - /// Resource that owns the waiting request - pub resource: ResourceId, - /// Original Codex thread - pub thread: ThreadId, - /// SHA-256 digest of the normalized spec JSON encoding - pub normalized_spec_sha256: NormalizedSpecSha256, - /// Durable resource route phase - pub phase: ResourceRoutePhase, -} - -impl ResourceRouteProof { - /// Derive a safe proof from a valid saved resource route - #[must_use] - pub fn from_route(route: &OriginRoute) -> Option { - let SubmissionState::Resource { resource, phase } = &route.submission else { - return None; - }; - route.validate().ok()?; - let spec = route.current_spec()?; - - Some(Self { - request: route.request, - task: route.task, - origin_machine: route.origin_machine, - authority_machine: route.execution_machine, - resource: *resource, - thread: route.thread, - normalized_spec_sha256: normalized_spec_sha256(spec).ok()?, - phase: phase.clone(), - }) - } -} - -/// Safe proof of one action-bound route saved by the supervisor machine -/// -/// The authority reads it before accepting a launch, so a request cannot claim -/// a callback route that the supervisor machine never saved -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceActionRouteProof { - /// Stable retry identity - pub request: RequestId, - /// Preallocated global task identity - pub task: TaskId, - /// Action that owns the route, with its owners and thread - pub binding: ResourceActionRouteBinding, - /// SHA-256 digest of the normalized spec saved in the route - pub normalized_spec_sha256: NormalizedSpecSha256, - /// Durable action route phase - pub phase: ResourceActionRoutePhase, -} - -impl ResourceActionRouteProof { - /// Derive a safe proof from a valid saved action-bound route - #[must_use] - pub fn from_route(route: &OriginRoute) -> Option { - let SubmissionState::ResourceAction { binding, phase, .. } = &route.submission else { - return None; - }; - route.validate().ok()?; - - Some(Self { - request: route.request, - task: route.task, - binding: *binding, - normalized_spec_sha256: normalized_spec_sha256(route.current_spec()?).ok()?, - phase: phase.clone(), - }) - } -} - -/// Safe proof of one remote first background launch route saved by the supervisor machine -/// -/// The authority reads it before accepting a launch, so a request cannot claim a -/// callback route that the supervisor machine never saved -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceBackgroundRouteProof { - /// Stable retry identity - pub request: RequestId, - /// Preallocated global task identity - pub task: TaskId, - /// Supervisor assignment and resource revision saved with the route - pub binding: BackgroundLaunchBinding, - /// SHA-256 digest of the normalized spec saved in the route - pub normalized_spec_sha256: NormalizedSpecSha256, - /// Durable background route phase - pub phase: ResourceBackgroundRoutePhase, -} - -impl ResourceBackgroundRouteProof { - /// Derive a safe proof from a valid saved background launch route - #[must_use] - pub fn from_route(route: &OriginRoute) -> Option { - let SubmissionState::ResourceBackground { binding, phase } = &route.submission else { - return None; - }; - route.validate().ok()?; - - Some(Self { - request: route.request, - task: route.task, - binding: *binding, - normalized_spec_sha256: normalized_spec_sha256(route.current_spec()?).ok()?, - phase: phase.clone(), - }) - } -} - /// Executor-owned accepted task identity, retained after detail cleanup #[derive(Debug, Clone, Serialize)] #[serde(deny_unknown_fields)] @@ -1113,25 +507,6 @@ impl ExecutionRecord { self.spec .valid_for_owners(self.origin_machine, self.execution_machine) } - - /// Whether this identity accepts exactly this task for these fixed owners - #[must_use] - pub fn is_owned_by( - &self, - task: TaskId, - origin_machine: MachineId, - execution_machine: MachineId, - ) -> bool { - self.origin_machine == origin_machine && self.is_executed_by(task, execution_machine) - } - - /// Whether this identity accepts exactly this task on this executor for any origin - #[must_use] - pub fn is_executed_by(&self, task: TaskId, execution_machine: MachineId) -> bool { - self.task == task - && self.execution_machine == execution_machine - && self.has_valid_spec_owners() - } } /// A definitive rejection that prevents delayed submission @@ -1184,16 +559,11 @@ mod tests { use std::path::PathBuf; use super::{ - CallbackContext, CallbackExecutable, ExecutionRecord, NewResourceActionRoute, - NewResourceRoute, NormalizedSpecSha256, OriginRoute, PersistedSpec, RequestId, - ResourceActionRouteBinding, ResourceActionRoutePhase, ResourceActionRouteProof, - ResourceRouteError, ResourceRoutePhase, ResourceRouteProof, SubmissionState, - normalized_spec_sha256, + CallbackContext, CallbackExecutable, ExecutionRecord, OriginRoute, PersistedSpec, + RequestId, RouteError, SubmissionState, }; - use crate::digest::Sha256Digest; use crate::domain::{ProcessStatus, TaskId}; use crate::machine::MachineId; - use crate::resource::ResourceId; use crate::spec::NormalizedSpec; fn spec() -> NormalizedSpec { @@ -1208,85 +578,6 @@ mod tests { .unwrap() } - fn resource_route() -> OriginRoute { - let spec = spec(); - OriginRoute::new_resource_waiting(NewResourceRoute { - request: RequestId::new(), - task: TaskId::new(), - origin_machine: MachineId::new(), - authority_machine: MachineId::new(), - thread: spec.thread, - callback: CallbackContext { - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - cwd: PathBuf::from("/tmp"), - codex: CallbackExecutable::available(PathBuf::from("/bin/codex")), - }, - spec, - resource: ResourceId::new(), - }) - .unwrap() - } - - #[test] - fn resource_route_proof_binds_identity_phase_and_normalized_spec_digest() { - let route = resource_route(); - let proof = ResourceRouteProof::from_route(&route).unwrap(); - let expected_digest = NormalizedSpecSha256(Sha256Digest::of( - serde_json::to_vec(route.current_spec().unwrap()).unwrap(), - )); - - assert_eq!(proof.request, route.request); - assert_eq!(proof.task, route.task); - assert_eq!(proof.origin_machine, route.origin_machine); - assert_eq!(proof.authority_machine, route.execution_machine); - let SubmissionState::Resource { resource, .. } = &route.submission else { - panic!("resource route constructor must create a resource route"); - }; - assert_eq!(proof.resource, *resource); - assert_eq!(proof.thread, route.thread); - assert_eq!(proof.phase, ResourceRoutePhase::AcceptanceUnknown); - assert_eq!(proof.normalized_spec_sha256, expected_digest); - - let wire = serde_json::to_value(&proof).unwrap(); - assert_eq!(wire["normalized_spec_sha256"].as_str().unwrap().len(), 64); - assert!(wire.get("callback").is_none()); - assert!(wire.get("spec").is_none()); - assert!(wire.get("cwd").is_none()); - assert!(wire.get("env").is_none()); - assert_eq!( - serde_json::from_value::(wire.clone()).unwrap(), - proof - ); - - let mut unknown_field = wire; - unknown_field["unexpected"] = serde_json::json!(true); - assert!(serde_json::from_value::(unknown_field).is_err()); - } - - #[test] - fn resource_route_proof_is_absent_for_missing_direct_or_invalid_resource_routes() { - let absent_route: Option<&OriginRoute> = None; - assert!( - absent_route - .and_then(ResourceRouteProof::from_route) - .is_none() - ); - - let mut direct_route = resource_route(); - direct_route.submission = SubmissionState::Accepted; - assert!(ResourceRouteProof::from_route(&direct_route).is_none()); - - let mut invalid_resource_route = resource_route(); - invalid_resource_route.submission = SubmissionState::Resource { - resource: ResourceId::new(), - phase: ResourceRoutePhase::Activated, - }; - assert!(ResourceRouteProof::from_route(&invalid_resource_route).is_none()); - } - #[test] fn persisted_spec_keeps_current_wire_shape_and_tags_migrated_rows() { let normalized = spec(); @@ -1392,101 +683,13 @@ mod tests { route.submission = SubmissionState::AcceptanceUnknown; assert!(matches!( route.validate(), - Err(ResourceRouteError::InvalidMigratedRouteState) + Err(RouteError::InvalidMigratedRouteState) )); route.submission = SubmissionState::Accepted; route.execution_machine = MachineId::new(); assert!(matches!( route.validate(), - Err(ResourceRouteError::InvalidMigratedIdentity) - )); - } - - fn action_input( - launch: crate::resource::bound_action::ResourceActionLaunch, - ) -> NewResourceActionRoute { - let spec = spec(); - NewResourceActionRoute { - request: RequestId::new(), - task: TaskId::new(), - callback: CallbackContext { - env: TaskEnv { - path: "/bin".into(), - home: "/tmp".into(), - }, - cwd: PathBuf::from("/tmp"), - codex: CallbackExecutable::available(PathBuf::from("/bin/codex")), - }, - binding: ResourceActionRouteBinding { - kind: launch.kind(), - authority: crate::resource::SupervisorActionAuthority { - authority_machine: MachineId::new(), - resource_id: ResourceId::new(), - loan_id: crate::resource::LoanId::new(), - action_id: crate::resource::ActionId::new(), - expected_state_revision: crate::resource::ResourceRevision::new(2), - supervisor: crate::resource::SupervisorAddress { - machine: MachineId::new(), - thread: spec.thread, - }, - assignment_revision: crate::resource::AssignmentRevision::new(0), - }, - }, - spec, - launch, - } - } - - #[test] - fn action_route_is_owned_by_the_supervisor_and_proves_its_exact_content() { - use crate::resource::bound_action::ResourceActionLaunch; - use crate::resource::{CommandSpec, ReturnWork}; - - let input = action_input(ResourceActionLaunch::Return { - work: Box::new(ReturnWork::NewBackgroundWork { - spec: CommandSpec::try_from(spec()).unwrap(), - }), - }); - let route = OriginRoute::new_resource_action(input.clone()).unwrap(); - assert_eq!( - route.origin_machine, - input.binding.authority.supervisor.machine - ); - assert_eq!( - route.execution_machine, - input.binding.authority.authority_machine - ); - let proof = ResourceActionRouteProof::from_route(&route).unwrap(); - assert_eq!(proof.binding, input.binding); - assert_eq!( - proof.normalized_spec_sha256, - normalized_spec_sha256(&spec()).unwrap() - ); - assert_eq!(proof.phase, ResourceActionRoutePhase::AcceptanceUnknown); - let wire = serde_json::to_value(&proof).unwrap(); - assert!(wire.get("callback").is_none()); - - // the authority never owns a callback route for its own supervisor thread - let mut co_located = input.clone(); - co_located.binding.authority.supervisor.machine = - co_located.binding.authority.authority_machine; - assert!(matches!( - OriginRoute::new_resource_action(co_located), - Err(ResourceRouteError::ActionOwnerMismatch) - )); - // a supervisor-supplied return command must be the saved route content - let mut changed = input.clone(); - changed.spec.timeout = std::time::Duration::from_secs(5); - assert!(matches!( - OriginRoute::new_resource_action(changed), - Err(ResourceRouteError::ActionOwnerMismatch) - )); - let mut wrong_kind = input; - wrong_kind.binding.kind = crate::resource::bound_action::ResourceActionKind::ReleaseWatcher; - assert!(matches!( - OriginRoute::new_resource_action(wrong_kind), - Err(ResourceRouteError::ActionOwnerMismatch) + Err(RouteError::InvalidMigratedIdentity) )); - assert!(ResourceActionRouteProof::from_route(&resource_route()).is_none()); } } diff --git a/tests/fleet.rs b/tests/fleet.rs index bf5a46c..28a27a5 100644 --- a/tests/fleet.rs +++ b/tests/fleet.rs @@ -22,12 +22,10 @@ use homebased::fleet::protocol::SUPPORTED_PROTOCOLS; use homebased::fleet::runtime::{FleetHandle, FleetRuntime, FleetStart, RuntimeTimings}; use homebased::machine::{BootId, LocalIdentity, MachineId, MachineName}; use homebased::message::MessageId; -use homebased::resource::{CommandSpec, ResourceId, ResourceQueueRequest}; use homebased::spec::NormalizedSpec; use homebased::store::{NewTask, Store, new_queued_task}; use homebased::submission::{ - CallbackContext, ExecutionRecord, NewResourceRoute, OriginRoute, RequestId, - ResourceQueueReceipt, ResourceRoutePhase, SubmissionState, + CallbackContext, ExecutionRecord, OriginRoute, RequestId, SubmissionState, }; use serde_json::Value; use tempfile::TempDir; @@ -362,112 +360,6 @@ fn seed_origin_route( request } -fn seed_resource_route( - origin: &Daemon, - authority: &Daemon, - resource: ResourceId, - request: RequestId, - task: TaskId, - spec: &NormalizedSpec, -) -> OriginRoute { - let route = OriginRoute::new_resource_waiting(NewResourceRoute { - request, - task, - origin_machine: origin.machine_id(), - authority_machine: authority.machine_id(), - thread: spec.thread, - callback: CallbackContext { - env: TaskEnv { - path: "/bin:/usr/bin".into(), - home: origin.user_home.to_string_lossy().into_owned(), - }, - cwd: origin.user_home.clone(), - codex: PathBuf::from("/bin/true").into(), - }, - spec: spec.clone(), - resource, - }) - .unwrap(); - Store::open(&origin.home.join("homebased.sqlite")) - .unwrap() - .insert_origin_route(&route) - .unwrap(); - route -} - -fn seed_authority_resource(authority: &Daemon, resource: ResourceId) { - let supervisor_thread = ThreadId(uuid::Uuid::now_v7()); - let connection = rusqlite::Connection::open(authority.home.join("homebased.sqlite")).unwrap(); - connection - .execute( - "INSERT INTO resources ( - id, display_name, authority_machine, supervisor_machine, supervisor_thread, - assignment_revision, state_revision, registered_background_task - ) VALUES (?1, 'test-gpu', ?2, ?2, ?3, 0, 0, NULL)", - rusqlite::params![ - resource.as_uuid().to_string(), - authority.machine_id().as_uuid().to_string(), - supervisor_thread.to_string(), - ], - ) - .unwrap(); -} - -async fn post_resource_queue( - authority: &Daemon, - origin: &Daemon, - route: &OriginRoute, -) -> (u16, Value) { - let spec = CommandSpec::try_from(route.current_spec().unwrap().clone()).unwrap(); - let request = ResourceQueueRequest::new( - homebased::fleet::protocol::CLUSTER_PROTOCOL_VERSION.0, - authority.machine_id(), - origin.machine_id(), - route.request, - route.task, - match &route.submission { - SubmissionState::Resource { resource, .. } => *resource, - _ => panic!("resource fixture must retain resource identity"), - }, - spec, - ); - let response = ClusterClient::default() - .post_json( - &authority.address(), - "/v1/cluster/resource-requests", - &request, - ) - .await - .unwrap(); - ( - response.status.as_u16(), - serde_json::from_slice(&response.body).unwrap(), - ) -} - -async fn post_resource_cancel_wire( - authority: &Daemon, - identity: &homebased::cancellation::ResourceCancellationRequestIdentity, -) -> (u16, Value) { - let response = ClusterClient::default() - .post_json( - &authority.address(), - "/v1/cluster/resource-requests/cancel", - &serde_json::json!({ - "api_version": 1, - "protocol_version": homebased::fleet::protocol::CLUSTER_PROTOCOL_VERSION.0, - "destination_machine": authority.machine_id(), - "request": identity, - }), - ) - .await - .unwrap(); - ( - response.status.as_u16(), - serde_json::from_slice(&response.body).unwrap(), - ) -} - /// Give `thread` a Codex session file under `user_home` so the submitting CLI /// accepts it as a callback thread fn register_thread(user_home: &Path, thread: &str) { @@ -2108,369 +2000,6 @@ async fn await_cancel_delivery(daemon: &Daemon, task: TaskId) -> Value { } } -async fn await_resource_cancel_delivery(daemon: &Daemon, task: TaskId) -> Value { - let start = Instant::now(); - loop { - let body = cancel_socket(daemon, task).await; - if body["delivery"]["state"] == "resource_delivered" { - return body; - } - assert!(start.elapsed() < Duration::from_secs(12), "{body}"); - tokio::time::sleep(Duration::from_millis(100)).await; - } -} - -fn resource_cancel_identity( - route: &OriginRoute, - cancellation: uuid::Uuid, - target_phase: ResourceRoutePhase, -) -> homebased::cancellation::ResourceCancellationRequestIdentity { - let SubmissionState::Resource { resource, .. } = &route.submission else { - panic!("resource fixture must retain resource identity"); - }; - homebased::cancellation::ResourceCancellationRequestIdentity { - requester_machine: route.origin_machine, - cancellation, - request: route.request, - task: route.task, - origin_machine: route.origin_machine, - authority_machine: route.execution_machine, - resource: *resource, - target_phase, - } -} - -#[tokio::test(flavor = "multi_thread")] -async fn resource_cancellation_before_acceptance_fences_delayed_queue_acceptance() { - let origin = Daemon::start("resource-cancel-origin", true); - let authority = Daemon::start("resource-cancel-authority", true); - add_peer(&origin, &authority); - add_peer(&authority, &origin); - let resource = ResourceId::new(); - seed_authority_resource(&authority, resource); - let request = RequestId::new(); - let task = TaskId::new(); - let spec = remote_spec(&authority, vec!["/bin/echo", "must-not-start"]); - let route = seed_resource_route(&origin, &authority, resource, request, task, &spec); - - let delivered = await_resource_cancel_delivery(&origin, task).await; - assert_eq!( - delivered["delivery"]["resource"]["outcome"]["type"], - "prevented_before_acceptance" - ); - let saved = Store::open(&origin.home.join("homebased.sqlite")) - .unwrap() - .origin_route_by_task(task) - .unwrap() - .unwrap(); - assert!(matches!( - saved.submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::CancelledBeforeLaunch, - .. - } - )); - - let (status, delayed) = post_resource_queue(&authority, &origin, &route).await; - assert_eq!(status, 200, "{delayed}"); - assert_eq!(delayed["receipt"]["outcome"]["type"], "rejected"); - assert_eq!( - delayed["receipt"]["outcome"]["reason"], - "cancelled_before_launch" - ); - let connection = rusqlite::Connection::open(authority.home.join("homebased.sqlite")).unwrap(); - let prevented: i64 = connection - .query_row( - "SELECT COUNT(*) FROM resource_request_preventions", - [], - |row| row.get(0), - ) - .unwrap(); - let queued: i64 = connection - .query_row( - "SELECT COUNT(*) FROM resource_requests WHERE task_id=?1", - [task.to_string()], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(prevented, 1); - assert_eq!(queued, 0); - assert_eq!(cancel_socket(&origin, task).await, delivered); -} - -#[tokio::test(flavor = "multi_thread")] -async fn resource_cancellation_of_queued_request_uses_the_authority_route() { - let origin = Daemon::start("resource-queued-cancel-origin", true); - let authority = Daemon::start("resource-queued-cancel-authority", true); - add_peer(&origin, &authority); - add_peer(&authority, &origin); - let resource = ResourceId::new(); - seed_authority_resource(&authority, resource); - let request = RequestId::new(); - let task = TaskId::new(); - let spec = remote_spec(&authority, vec!["/bin/echo", "queued-not-started"]); - let route = seed_resource_route(&origin, &authority, resource, request, task, &spec); - let (status, accepted) = post_resource_queue(&authority, &origin, &route).await; - assert_eq!(status, 200, "{accepted}"); - let receipt: ResourceQueueReceipt = - serde_json::from_value(accepted["receipt"].clone()).unwrap(); - assert_eq!( - Store::open(&origin.home.join("homebased.sqlite")) - .unwrap() - .resolve_resource_route(&receipt) - .unwrap() - .submission, - SubmissionState::Resource { - resource, - phase: ResourceRoutePhase::Waiting, - } - ); - - let delivered = await_resource_cancel_delivery(&origin, task).await; - assert_eq!( - delivered["delivery"]["resource"]["outcome"]["type"], - "cancelled_before_launch" - ); - let route = Store::open(&origin.home.join("homebased.sqlite")) - .unwrap() - .origin_route_by_task(task) - .unwrap() - .unwrap(); - assert!(matches!( - route.submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::CancelledBeforeLaunch, - .. - } - )); - let connection = rusqlite::Connection::open(authority.home.join("homebased.sqlite")).unwrap(); - let state: String = connection - .query_row( - "SELECT json_extract(state_json, '$.type') FROM resource_requests WHERE request_id=?1", - [request.0.to_string()], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(state, "cancelled_before_launch"); - let receipts: i64 = connection - .query_row( - "SELECT COUNT(*) FROM resource_cancellation_receipts", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(receipts, 1); -} - -#[tokio::test(flavor = "multi_thread")] -async fn remote_resource_cancellation_persists_intent_at_the_origin() { - let viewer = Daemon::start("resource-remote-cancel-viewer", true); - let origin = Daemon::start("resource-remote-cancel-origin", true); - let authority = Daemon::start("resource-remote-cancel-authority", true); - add_peer(&viewer, &origin); - add_peer(&viewer, &authority); - add_peer(&origin, &authority); - add_peer(&authority, &origin); - - let resource = ResourceId::new(); - seed_authority_resource(&authority, resource); - let request = RequestId::new(); - let task = TaskId::new(); - let spec = remote_spec(&authority, vec!["/bin/echo", "remote-resource-cancel"]); - seed_resource_route(&origin, &authority, resource, request, task, &spec); - - let delivered = await_resource_cancel_delivery(&viewer, task).await; - assert_eq!( - delivered["delivery"]["resource"]["outcome"]["type"], - "prevented_before_acceptance" - ); - - let origin_store = Store::open(&origin.home.join("homebased.sqlite")).unwrap(); - let saved = origin_store - .cancellation_request(task) - .unwrap() - .expect("origin must own the durable cancellation intent"); - assert_eq!(saved.requester_machine, origin.machine_id()); - assert!(matches!( - saved.target, - homebased::cancellation::CancellationTarget::Resource(_) - )); - assert!( - Store::open(&viewer.home.join("homebased.sqlite")) - .unwrap() - .cancellation_request(task) - .unwrap() - .is_none(), - "viewer must not persist a second cancellation identity" - ); - assert_eq!(cancel_socket(&viewer, task).await, delivered); -} - -#[tokio::test(flavor = "multi_thread")] -async fn resource_cancellation_receipt_replays_after_restart_and_conflicts_on_reuse() { - let origin = Daemon::start("resource-receipt-origin", true); - let mut authority = Daemon::start("resource-receipt-authority", true); - add_peer(&origin, &authority); - add_peer(&authority, &origin); - let resource = ResourceId::new(); - seed_authority_resource(&authority, resource); - let request = RequestId::new(); - let task = TaskId::new(); - let spec = remote_spec(&authority, vec!["/bin/echo", "receipt-replay"]); - let route = seed_resource_route(&origin, &authority, resource, request, task, &spec); - let (status, accepted) = post_resource_queue(&authority, &origin, &route).await; - assert_eq!(status, 200, "{accepted}"); - let queue_receipt: ResourceQueueReceipt = - serde_json::from_value(accepted["receipt"].clone()).unwrap(); - Store::open(&origin.home.join("homebased.sqlite")) - .unwrap() - .resolve_resource_route(&queue_receipt) - .unwrap(); - let identity = - resource_cancel_identity(&route, uuid::Uuid::now_v7(), ResourceRoutePhase::Waiting); - - let (status, first) = post_resource_cancel_wire(&authority, &identity).await; - assert_eq!(status, 200, "{first}"); - assert_eq!( - first["receipt"]["outcome"]["type"], - "cancelled_before_launch" - ); - authority.restart(); - let (status, replay) = post_resource_cancel_wire(&authority, &identity).await; - assert_eq!(status, 200, "{replay}"); - assert_eq!(first, replay); - - let mut changed = identity.clone(); - changed.resource = ResourceId::new(); - let (status, conflict) = post_resource_cancel_wire(&authority, &changed).await; - assert_eq!(status, 409, "{conflict}"); -} - -#[tokio::test(flavor = "multi_thread")] -async fn resource_cancellation_origin_restart_replays_a_lost_authority_reply() { - let mut origin = Daemon::start("resource-origin-restart-cancel-origin", true); - let authority = Daemon::start("resource-origin-restart-cancel-authority", true); - add_peer(&authority, &origin); - let resource = ResourceId::new(); - seed_authority_resource(&authority, resource); - let request = RequestId::new(); - let task = TaskId::new(); - let spec = remote_spec(&authority, vec!["/bin/echo", "restart-resource-cancel"]); - let route = seed_resource_route(&origin, &authority, resource, request, task, &spec); - let (status, queued) = post_resource_queue(&authority, &origin, &route).await; - assert_eq!(status, 200, "{queued}"); - assert_eq!(queued["receipt"]["outcome"]["type"], "waiting"); - - let pending = cancel_socket(&origin, task).await; - assert_eq!(pending["delivery"]["state"], "pending"); - let saved = Store::open(&origin.home.join("homebased.sqlite")) - .unwrap() - .cancellation_request(task) - .unwrap() - .expect("origin must save cancellation before delivery"); - let identity = saved - .resource_identity() - .expect("resource route must retain its typed identity"); - - let (status, authority_reply) = post_resource_cancel_wire(&authority, &identity).await; - assert_eq!(status, 200, "{authority_reply}"); - assert_eq!( - authority_reply["receipt"]["outcome"]["type"], - "cancelled_before_launch" - ); - - origin.restart(); - add_peer(&origin, &authority); - let delivered = await_resource_cancel_delivery(&origin, task).await; - assert_eq!(delivered["cancellation"], pending["cancellation"]); - assert_eq!( - delivered["delivery"]["resource"], - authority_reply["receipt"] - ); - let route = Store::open(&origin.home.join("homebased.sqlite")) - .unwrap() - .origin_route_by_task(task) - .unwrap() - .unwrap(); - assert!(matches!( - route.submission, - SubmissionState::Resource { - phase: ResourceRoutePhase::CancelledBeforeLaunch, - .. - } - )); - let connection = rusqlite::Connection::open(authority.home.join("homebased.sqlite")).unwrap(); - let receipts: i64 = connection - .query_row( - "SELECT COUNT(*) FROM resource_cancellation_receipts", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(receipts, 1); -} - -#[tokio::test(flavor = "multi_thread")] -async fn resource_cancellation_rejects_wrong_owner_and_origin_proof() { - let origin = Daemon::start("resource-proof-origin", true); - let authority = Daemon::start("resource-proof-authority", true); - add_peer(&origin, &authority); - add_peer(&authority, &origin); - let resource = ResourceId::new(); - seed_authority_resource(&authority, resource); - let request = RequestId::new(); - let task = TaskId::new(); - let spec = remote_spec(&authority, vec!["/bin/echo", "proof-check"]); - let route = seed_resource_route(&origin, &authority, resource, request, task, &spec); - let identity = resource_cancel_identity( - &route, - uuid::Uuid::now_v7(), - ResourceRoutePhase::AcceptanceUnknown, - ); - - let wrong_destination = ClusterClient::default() - .post_json( - &authority.address(), - "/v1/cluster/resource-requests/cancel", - &serde_json::json!({ - "api_version": 1, - "protocol_version": homebased::fleet::protocol::CLUSTER_PROTOCOL_VERSION.0, - "destination_machine": origin.machine_id(), - "request": identity, - }), - ) - .await - .unwrap(); - assert_eq!(wrong_destination.status.as_u16(), 409); - - let mut wrong_owner = identity.clone(); - wrong_owner.authority_machine = origin.machine_id(); - let (status, owner_conflict) = post_resource_cancel_wire(&authority, &wrong_owner).await; - assert_eq!(status, 409, "{owner_conflict}"); - - let mut wrong_proof = identity.clone(); - wrong_proof.request = RequestId::new(); - let (status, proof_conflict) = post_resource_cancel_wire(&authority, &wrong_proof).await; - assert_eq!(status, 409, "{proof_conflict}"); - let connection = rusqlite::Connection::open(authority.home.join("homebased.sqlite")).unwrap(); - let prevented: i64 = connection - .query_row( - "SELECT COUNT(*) FROM resource_request_preventions", - [], - |row| row.get(0), - ) - .unwrap(); - let receipts: i64 = connection - .query_row( - "SELECT COUNT(*) FROM resource_cancellation_receipts", - [], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(prevented, 0); - assert_eq!(receipts, 0); -} - #[tokio::test(flavor = "multi_thread")] async fn cancellation_before_acceptance_prevents_delayed_submit() { let origin = Daemon::start("cancel-origin", true); @@ -3308,1525 +2837,24 @@ async fn local_duplicate_identity_recovers_after_the_clone_goes_offline() { } } -/// Fixed identities for one AwaitingReturn action whose supervisor runs on another machine -struct RemoteReturnAction { - action: homebased::resource::SupervisorActionAuthority, -} - -/// Seed one resource, AwaitingReturn loan, and delivered return notice on a stopped authority -fn seed_remote_return_action(authority: &Daemon, supervisor: &Daemon) -> RemoteReturnAction { - use homebased::resource::{ - ActionId, AssignmentRevision, LoanId, LoanPhase, LoanState, NoticeId, ResourceRevision, - ReturnContext, SupervisorActionAuthority, SupervisorAddress, SupervisorNotice, - SupervisorNoticeDelivery, SupervisorNoticePayload, - }; - - let action = SupervisorActionAuthority { - authority_machine: authority.machine_id(), - resource_id: ResourceId::new(), - loan_id: LoanId::new(), - action_id: ActionId::new(), - expected_state_revision: ResourceRevision::new(1), - supervisor: SupervisorAddress { - machine: supervisor.machine_id(), - thread: ThreadId(uuid::Uuid::now_v7()), - }, - assignment_revision: AssignmentRevision::new(0), - }; - let loan_state = LoanState::Active { - phase: LoanPhase::AwaitingReturn { - action_id: action.action_id, - return_context: ReturnContext::Idle, - }, - }; - let notice = SupervisorNotice { - id: NoticeId::new(), - loan_id: action.loan_id, - action_id: action.action_id, - state_revision: action.expected_state_revision, - destination: action.supervisor, - assignment_revision: action.assignment_revision, - payload: SupervisorNoticePayload::ReturnRequired { - return_context: ReturnContext::Idle, - }, - delivery: SupervisorNoticeDelivery::Delivered { attempts: 1 }, - }; - let connection = rusqlite::Connection::open(authority.home.join("homebased.sqlite")).unwrap(); - connection - .execute( - "INSERT INTO resources ( - id, display_name, authority_machine, supervisor_machine, supervisor_thread, - assignment_revision, state_revision, registered_background_task - ) VALUES (?1, 'remote-gpu', ?2, ?3, ?4, 0, 1, NULL)", - rusqlite::params![ - action.resource_id.as_uuid().to_string(), - action.authority_machine.as_uuid().to_string(), - action.supervisor.machine.as_uuid().to_string(), - action.supervisor.thread.to_string(), - ], - ) - .unwrap(); - connection - .execute( - "INSERT INTO loans (id, resource_id, state_json) VALUES (?1, ?2, ?3)", - rusqlite::params![ - action.loan_id.as_uuid().to_string(), - action.resource_id.as_uuid().to_string(), - serde_json::to_string(&loan_state).unwrap(), - ], - ) - .unwrap(); - connection - .execute( - "INSERT INTO resource_supervisor_notices (id, loan_id, action_id, notice_json) - VALUES (?1, ?2, ?3, ?4)", - rusqlite::params![ - notice.id.as_uuid().to_string(), - action.loan_id.as_uuid().to_string(), - action.action_id.as_uuid().to_string(), - serde_json::to_string(¬ice).unwrap(), - ], - ) - .unwrap(); - RemoteReturnAction { action } -} - -/// Foreground command on the authority that runs until its gate file exists -fn gated_background_work( - authority: &Daemon, - thread: ThreadId, - name: &str, -) -> (homebased::resource::ReturnWork, PathBuf, PathBuf) { - let marker = authority.user_home.join(format!("{name}.marker")); - let gate = authority.user_home.join(format!("{name}.gate")); - let spec = homebased::spec::parse_normalized_value(&serde_json::json!({ - "api_version": 1, - "thread": thread, - "name": name, - "cwd": authority.user_home, - "timeout": "30m", - "workload": { - "type": "task", - "command": [native_gated_command(), &marker, &gate] - } - })) - .unwrap(); - ( - homebased::resource::ReturnWork::NewBackgroundWork { - spec: CommandSpec::try_from(spec).unwrap(), - }, - marker, - gate, - ) -} - -/// Native foreground command that appends `x` to its marker, then waits for its gate -/// -/// Resource work must name an inspectable native executable, so the tests build -/// one once per process instead of using a shell script -fn native_gated_command() -> &'static Path { - static COMMAND: std::sync::OnceLock = std::sync::OnceLock::new(); - COMMAND.get_or_init(|| { - let directory = - std::env::temp_dir().join(format!("homebased-fleet-gated-{}", std::process::id())); - fs::create_dir_all(&directory).unwrap(); - let source = directory.join("gated.c"); - fs::write( - &source, - "#include \n#include \n\ - int main(int argc, char **argv) {\n\ - if (argc != 3) return 2;\n\ - FILE *marker = fopen(argv[1], \"a\");\n\ - if (marker == NULL || fputs(\"x\", marker) < 0 || fclose(marker) != 0) return 1;\n\ - while (access(argv[2], F_OK) != 0) usleep(50000);\n\ - return 0;\n}\n", - ) - .unwrap(); - let binary = directory.join("gated"); - let status = std::process::Command::new("cc") - .arg(&source) - .arg("-o") - .arg(&binary) - .status() - .unwrap(); - assert!(status.success(), "cc must build the gated command"); - binary - }) -} - -async fn submit_resource_action( - supervisor: &Daemon, - action: homebased::resource::SupervisorActionAuthority, - choice: homebased::resource::bound_action::ResourceActionChoice, -) -> Result { - homebased::client::Client::new(supervisor.home.join("homebased.sock")) - .post( - homebased::resource::bound_action::RESOURCE_ACTION_SUBMIT_PATH, - &serde_json::to_value( - homebased::resource::bound_action::ResourceActionSubmitRequest { - api_version: 1, - authority: action, - choice, - }, - ) - .unwrap(), - ) - .await -} +#[test] +fn remote_submit_with_a_missing_executor_cwd_is_a_typed_rejection() { + let origin = Daemon::start("missing-cwd-origin", true); + let executor = Daemon::start("remote-executor", true); + add_peer(&origin, &executor); + let mut spec = cli_remote_spec(&executor, vec!["/bin/echo", "hello"]); + spec.cwd = PathBuf::from("~/not-present"); + let request = RequestId::new(); -fn return_choice( - launch: &homebased::resource::ReturnLaunch, -) -> homebased::resource::bound_action::ResourceActionChoice { - homebased::resource::bound_action::ResourceActionChoice::Return { - decision: homebased::resource::ReturnDecision::Launch(Box::new(launch.clone())), - } -} - -async fn post_resource_action(authority: &Daemon, request: &Value) -> (u16, Value) { - let response = ClusterClient::default() - .post_json( - &authority.address(), - homebased::resource::bound_action::RESOURCE_ACTION_PATH, - request, - ) - .await - .unwrap(); - ( - response.status.as_u16(), - serde_json::from_slice(&response.body).unwrap_or(Value::Null), - ) -} - -fn task_count(daemon: &Daemon, task: TaskId) -> i64 { - rusqlite::Connection::open(daemon.home.join("homebased.sqlite")) - .unwrap() - .query_row( - "SELECT COUNT(*) FROM tasks WHERE id = ?1", - [task.to_string()], - |row| row.get(0), - ) - .unwrap() -} - -fn action_receipt_count(daemon: &Daemon) -> i64 { - rusqlite::Connection::open(daemon.home.join("homebased.sqlite")) - .unwrap() - .query_row( - "SELECT COUNT(*) FROM resource_action_task_receipts", - [], - |row| row.get(0), - ) - .unwrap() -} - -fn loan_state(daemon: &Daemon, loan: homebased::resource::LoanId) -> Value { - let json: String = rusqlite::Connection::open(daemon.home.join("homebased.sqlite")) - .unwrap() - .query_row( - "SELECT state_json FROM loans WHERE id = ?1", - [loan.as_uuid().to_string()], - |row| row.get(0), - ) - .unwrap(); - serde_json::from_str(&json).unwrap() -} - -async fn wait_for(budget: Duration, mut done: impl FnMut() -> bool) -> bool { - let start = Instant::now(); - while start.elapsed() < budget { - if done() { - return true; - } - tokio::time::sleep(Duration::from_millis(50)).await; - } - done() -} - -/// Two daemons whose peers know each other, with a remote return action on the authority -async fn remote_return_daemons(names: (&str, &str)) -> (Daemon, Daemon, RemoteReturnAction) { - let mut authority = Daemon::start(names.0, true); - let supervisor = Daemon::start(names.1, true); - add_peer(&authority, &supervisor); - add_peer(&supervisor, &authority); - authority.stop(); - let seeded = seed_remote_return_action(&authority, &supervisor); - // the authority restores its resource actor from the seeded records at startup - authority.spawn(); - wait_for_probe(&authority.address()).await; - (authority, supervisor, seeded) -} - -#[tokio::test(flavor = "multi_thread")] -async fn remote_supervisor_return_runs_once_on_the_authority_and_owns_its_callbacks() { - let (authority, supervisor, seeded) = - remote_return_daemons(("remote-return-authority", "remote-return-supervisor")).await; - let action = seeded.action; - let (work, marker, gate) = - gated_background_work(&authority, action.supervisor.thread, "return"); - let launch = homebased::resource::ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work, - }; - let task = launch.task_id; - - let accepted = submit_resource_action(&supervisor, action, return_choice(&launch)) - .await - .unwrap(); - assert_eq!(accepted["outcome"]["type"], "accepted", "{accepted}"); - assert_eq!( - accepted["outcome"]["receipt"]["task_id"], - serde_json::json!(task) - ); - assert_eq!(task_count(&authority, task), 1); - assert_eq!(action_receipt_count(&authority), 1); - // the authority keeps only the execution identity; the route lives on the supervisor - let authority_store = Store::open(&authority.home.join("homebased.sqlite")).unwrap(); - assert!( - authority_store - .origin_route_by_task(task) - .unwrap() - .is_none() - ); - let Some(homebased::submission::ExecutorIdentity::Accepted(record)) = - authority_store.executor_identity(task).unwrap() - else { - panic!("authority must accept the return task"); - }; - assert_eq!(record.origin_machine, supervisor.machine_id()); - assert_eq!(record.execution_machine, authority.machine_id()); - - // the authority's task events reach the supervisor machine's saved route - let supervisor_db = supervisor.home.join("homebased.sqlite"); - assert!( - wait_for(Duration::from_secs(20), || { - let route = Store::open(&supervisor_db) - .unwrap() - .origin_route_by_task(task) - .unwrap() - .unwrap(); - route.last_execution_state == Some(ProcessStatus::Running) - }) - .await, - "running event did not reach the supervisor route" - ); - let route = Store::open(&supervisor_db) - .unwrap() - .origin_route_by_task(task) - .unwrap() - .unwrap(); - assert_eq!(route.thread, action.supervisor.thread); - assert!(matches!( - route.submission, - SubmissionState::ResourceAction { - phase: homebased::submission::ResourceActionRoutePhase::Accepted, - .. - } - )); - - // the running native foreground task keeps its Restoring loan and is never registered - let restoring = loan_state(&authority, action.loan_id); - assert_eq!(restoring["type"], "active", "{restoring}"); - assert_eq!(restoring["phase"]["type"], "restoring", "{restoring}"); - - // an exact retry is answered from the saved route without a second task - let retried = submit_resource_action(&supervisor, action, return_choice(&launch)) - .await - .unwrap(); - assert_eq!(retried["outcome"]["type"], "accepted"); - let mut changed = launch.clone(); - let (other_work, _, _) = gated_background_work(&authority, action.supervisor.thread, "other"); - changed.work = other_work; - assert!(matches!( - submit_resource_action(&supervisor, action, return_choice(&changed)).await, - Err(homebased::error::AppError::SubmissionConflict { .. }) - )); - assert_eq!(task_count(&authority, task), 1); - assert_eq!(action_receipt_count(&authority), 1); - - fs::write(&gate, b"").unwrap(); - assert!( - wait_for(Duration::from_secs(20), || { - Store::open(&authority.home.join("homebased.sqlite")) - .unwrap() - .get_task(task) - .unwrap() - .is_some_and(|row| row.status().is_terminal()) - }) - .await - ); - assert_eq!(fs::read(&marker).unwrap(), b"x"); - - // the successful confirmed end closes the loan without registering the task - assert!( - wait_for(Duration::from_secs(20), || { - loan_state(&authority, action.loan_id)["type"] == "closed" - }) - .await, - "confirmed foreground end did not close the loan: {}", - loan_state(&authority, action.loan_id) - ); - assert_eq!( - loan_state(&authority, action.loan_id)["result"]["type"], - "foreground_return_ended" - ); -} - -#[tokio::test(flavor = "multi_thread")] -async fn saved_remote_return_route_retries_the_same_identity_after_restart_and_lost_reply() { - let (mut authority, mut supervisor, seeded) = - remote_return_daemons(("remote-retry-authority", "remote-retry-supervisor")).await; - let action = seeded.action; - let (work, marker, gate) = gated_background_work(&authority, action.supervisor.thread, "retry"); - let launch = homebased::resource::ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work, - }; - let task = launch.task_id; - let homebased::resource::ReturnWork::NewBackgroundWork { spec } = &launch.work else { - unreachable!(); - }; - // the supervisor saved its fixed-ID route, then the send never reached the authority - supervisor.stop(); - authority.stop(); - let route = OriginRoute::new_resource_action(homebased::submission::NewResourceActionRoute { - request: launch.request_id, - task, - callback: CallbackContext { - env: TaskEnv { - path: "/bin:/usr/bin".into(), - home: supervisor.user_home.to_string_lossy().into_owned(), - }, - cwd: supervisor.user_home.clone(), - codex: PathBuf::from("/bin/true").into(), - }, - spec: spec.as_normalized().clone(), - binding: homebased::submission::ResourceActionRouteBinding { - kind: homebased::resource::bound_action::ResourceActionKind::Return, - authority: action, - }, - launch: homebased::resource::bound_action::ResourceActionLaunch::Return { - work: Box::new(launch.work.clone()), - }, - }) - .unwrap(); - Store::open(&supervisor.home.join("homebased.sqlite")) - .unwrap() - .insert_origin_route(&route) - .unwrap(); - - // startup recovery with the authority offline keeps the same unresolved route - supervisor.spawn(); - tokio::time::sleep(Duration::from_millis(500)).await; - assert_eq!(task_count(&authority, task), 0); - let supervisor_db = supervisor.home.join("homebased.sqlite"); - let unresolved = || { - Store::open(&supervisor_db) - .unwrap() - .unknown_resource_action_routes() - .unwrap() - .len() - }; - assert_eq!(unresolved(), 1); - - authority.spawn(); - wait_for_probe(&authority.address()).await; - supervisor.restart(); - assert!( - wait_for(Duration::from_secs(20), || unresolved() == 0).await, - "startup recovery did not resolve the saved route" - ); - assert_eq!(task_count(&authority, task), 1); - - // a lost reply is answered by the same identity without a second task - let digest = homebased::submission::normalized_spec_sha256(spec.as_normalized()).unwrap(); - let request = homebased::resource::bound_action::ResourceActionRequest::new( - homebased::fleet::protocol::CLUSTER_PROTOCOL_VERSION.0, - action, - homebased::resource::bound_action::ResourceActionOperation::LaunchReturn { - launch: launch.clone(), - normalized_spec_sha256: digest, - }, - ); - let (status, body) = - post_resource_action(&authority, &serde_json::to_value(&request).unwrap()).await; - assert_eq!(status, 200, "{body}"); - assert_eq!(body["outcome"]["type"], "accepted", "{body}"); - assert_eq!(body["outcome"]["acceptance"]["type"], "existing"); - assert_eq!(task_count(&authority, task), 1); - assert_eq!(action_receipt_count(&authority), 1); - - fs::write(&gate, b"").unwrap(); - assert!( - wait_for(Duration::from_secs(20), || { - Store::open(&authority.home.join("homebased.sqlite")) - .unwrap() - .get_task(task) - .unwrap() - .is_some_and(|row| row.status().is_terminal()) - }) - .await - ); - assert_eq!(fs::read(&marker).unwrap(), b"x"); -} - -#[tokio::test(flavor = "multi_thread")] -async fn remote_action_refuses_wrong_route_owner_revision_and_evidence_without_records() { - use homebased::resource::bound_action::{ResourceActionOperation, ResourceActionRequest}; - - let (authority, supervisor, seeded) = - remote_return_daemons(("remote-wrong-authority", "remote-wrong-supervisor")).await; - let action = seeded.action; - let (work, _, _) = gated_background_work(&authority, action.supervisor.thread, "wrong"); - let launch = homebased::resource::ReturnLaunch { - request_id: RequestId::new(), - task_id: TaskId::new(), - work, - }; - let homebased::resource::ReturnWork::NewBackgroundWork { spec } = &launch.work else { - unreachable!(); - }; - let digest = homebased::submission::normalized_spec_sha256(spec.as_normalized()).unwrap(); - let protocol = homebased::fleet::protocol::CLUSTER_PROTOCOL_VERSION.0; - let wire = |action, operation| { - serde_json::to_value(ResourceActionRequest::new(protocol, action, operation)).unwrap() - }; - let launch_operation = ResourceActionOperation::LaunchReturn { - launch: launch.clone(), - normalized_spec_sha256: digest, - }; - - let mut wrong_destination = wire(action, launch_operation.clone()); - wrong_destination["destination_machine"] = serde_json::json!(supervisor.machine_id()); - assert_ne!( - post_resource_action(&authority, &wrong_destination).await.0, - 200 - ); - let mut wrong_source = wire(action, launch_operation.clone()); - wrong_source["source_machine"] = serde_json::json!(MachineId::new()); - assert_ne!(post_resource_action(&authority, &wrong_source).await.0, 200); - let mut unknown_field = wire(action, launch_operation.clone()); - unknown_field["operation"]["command"] = serde_json::json!(["/bin/sh", "-c", "anything"]); - assert_ne!( - post_resource_action(&authority, &unknown_field).await.0, - 200 - ); - - // the supervisor machine has no saved route for this task yet - let (status, body) = post_resource_action(&authority, &wire(action, launch_operation)).await; - assert_eq!(status, 200, "{body}"); - assert_eq!(body["outcome"]["reason"]["type"], "route_evidence_missing"); - - // a replaced thread, stale revision, or watcher request for a return action is refused - let mut other_thread = action; - other_thread.supervisor.thread = ThreadId(uuid::Uuid::now_v7()); - let (_, body) = post_resource_action( - &authority, - &wire( - other_thread, - ResourceActionOperation::PrepareReturn { - launch: launch.clone(), - }, - ), - ) - .await; - assert_eq!(body["outcome"]["reason"]["type"], "not_current_supervisor"); - let mut stale = action; - stale.expected_state_revision = homebased::resource::ResourceRevision::new(7); - let (_, body) = post_resource_action( - &authority, - &wire( - stale, - ResourceActionOperation::PrepareReturn { - launch: launch.clone(), - }, - ), - ) - .await; - assert_eq!(body["outcome"]["reason"]["type"], "stale_revision"); - let (_, body) = post_resource_action( - &authority, - &wire( - action, - ResourceActionOperation::PrepareReleaseWatcher { - observed_background_task: TaskId::new(), - }, - ), - ) - .await; - assert_eq!(body["outcome"]["reason"]["type"], "action_not_pending"); - - // a saved route whose content differs from the launch is wrong-route evidence - let mut other = launch.clone(); - let (other_work, _, _) = gated_background_work(&authority, action.supervisor.thread, "saved"); - other.work = other_work; - let homebased::resource::ReturnWork::NewBackgroundWork { spec: saved_spec } = &other.work - else { - unreachable!(); - }; - Store::open(&supervisor.home.join("homebased.sqlite")) - .unwrap() - .insert_origin_route( - &OriginRoute::new_resource_action(homebased::submission::NewResourceActionRoute { - request: launch.request_id, - task: launch.task_id, - callback: CallbackContext { - env: TaskEnv { - path: "/bin:/usr/bin".into(), - home: supervisor.user_home.to_string_lossy().into_owned(), - }, - cwd: supervisor.user_home.clone(), - codex: PathBuf::from("/bin/true").into(), - }, - spec: saved_spec.as_normalized().clone(), - binding: homebased::submission::ResourceActionRouteBinding { - kind: homebased::resource::bound_action::ResourceActionKind::Return, - authority: action, - }, - launch: homebased::resource::bound_action::ResourceActionLaunch::Return { - work: Box::new(other.work.clone()), - }, - }) - .unwrap(), - ) - .unwrap(); - let (_, body) = post_resource_action( - &authority, - &wire( - action, - ResourceActionOperation::LaunchReturn { - launch: launch.clone(), - normalized_spec_sha256: digest, - }, - ), - ) - .await; - assert_eq!(body["outcome"]["reason"]["type"], "route_evidence_mismatch"); - - assert_eq!(task_count(&authority, launch.task_id), 0); - assert_eq!(action_receipt_count(&authority), 0); - assert_eq!( - loan_state(&authority, action.loan_id)["phase"]["type"], - "awaiting_return" - ); -} - -#[tokio::test(flavor = "multi_thread")] -async fn remote_supervisor_no_resume_closes_the_loan_without_a_task() { - let (authority, supervisor, seeded) = - remote_return_daemons(("remote-no-resume-authority", "remote-no-resume-supervisor")).await; - let action = seeded.action; - let closed = submit_resource_action( - &supervisor, - action, - homebased::resource::bound_action::ResourceActionChoice::Return { - decision: homebased::resource::ReturnDecision::NoResume { - reason: "evaluation is next week".into(), - }, - }, - ) - .await - .unwrap(); - assert_eq!(closed["outcome"]["type"], "closed", "{closed}"); - let state = loan_state(&authority, action.loan_id); - assert_eq!(state["type"], "closed"); - assert_eq!(state["result"]["type"], "no_resume"); - assert_eq!(action_receipt_count(&authority), 0); - - // a repeated identical decision replays the saved closure - let replayed = submit_resource_action( - &supervisor, - action, - homebased::resource::bound_action::ResourceActionChoice::Return { - decision: homebased::resource::ReturnDecision::NoResume { - reason: "evaluation is next week".into(), - }, - }, - ) - .await - .unwrap(); - assert_eq!(replayed["outcome"]["type"], "closed"); -} - -#[tokio::test(flavor = "multi_thread")] -async fn remote_watcher_choice_is_refused_before_any_route_when_the_authority_cannot_bind() { - use homebased::resource::{LoanPhase, LoanState, SupervisorNoticePayload}; - - let mut authority = Daemon::start("remote-watcher-authority", true); - let supervisor = Daemon::start("remote-watcher-supervisor", true); - add_peer(&authority, &supervisor); - add_peer(&supervisor, &authority); - authority.stop(); - let seeded = seed_remote_return_action(&authority, &supervisor); - let action = seeded.action; - // turn the seeded action into a release action for a trainer that is not running - let trainer = TaskId::new(); - let release = LoanState::Active { - phase: LoanPhase::AwaitingRelease { - action_id: action.action_id, - observed_background_task: trainer, - watcher_intent: None, - }, - }; - let connection = rusqlite::Connection::open(authority.home.join("homebased.sqlite")).unwrap(); - connection - .execute( - "UPDATE resources SET registered_background_task = ?1 WHERE id = ?2", - rusqlite::params![ - trainer.to_string(), - action.resource_id.as_uuid().to_string() - ], - ) - .unwrap(); - connection - .execute( - "UPDATE loans SET state_json = ?1 WHERE id = ?2", - rusqlite::params![ - serde_json::to_string(&release).unwrap(), - action.loan_id.as_uuid().to_string() - ], - ) - .unwrap(); - connection - .execute( - "UPDATE resource_supervisor_notices SET notice_json = json_set(notice_json, '$.payload', json(?1)) - WHERE action_id = ?2", - rusqlite::params![ - serde_json::to_string(&SupervisorNoticePayload::ReleaseRequired { task_id: trainer }) - .unwrap(), - action.action_id.as_uuid().to_string() - ], - ) - .unwrap(); - drop(connection); - authority.spawn(); - wait_for_probe(&authority.address()).await; - - let refused = submit_resource_action( - &supervisor, - action, - homebased::resource::bound_action::ResourceActionChoice::ReleaseWatcher { - observed_background_task: trainer, - }, - ) - .await - .unwrap(); - assert_eq!(refused["outcome"]["type"], "rejected", "{refused}"); - assert_eq!( - refused["outcome"]["reason"]["type"], "watcher_unavailable", - "{refused}" - ); - assert!( - Store::open(&supervisor.home.join("homebased.sqlite")) - .unwrap() - .resource_action_route_by_action(action.action_id) - .unwrap() - .is_none() - ); - assert_eq!(action_receipt_count(&authority), 0); - assert_eq!( - loan_state(&authority, action.loan_id)["phase"]["type"], - "awaiting_release" - ); -} - -// ---- remote-supervisor first background launch ---- - -/// Fake maintained trainer layout on the authority whose interpreter is a gated script -/// -/// Every start appends one byte to `marker`; the script exits after `gate` exists -/// or its test directory is removed -struct FleetTrainer { - cwd: PathBuf, - runtime_root: PathBuf, - task_file: PathBuf, - input_root: PathBuf, - python: PathBuf, - marker: PathBuf, - gate: PathBuf, -} - -impl Drop for FleetTrainer { - fn drop(&mut self) { - // a failed test must not leave the gated trainer running - let _ = fs::write(&self.gate, b""); - } -} - -impl FleetTrainer { - fn new(authority: &Daemon) -> Self { - use std::os::unix::fs::PermissionsExt; - - let root = authority - .user_home - .canonicalize() - .unwrap() - .join("trainer-root"); - let cwd = root.join("trainer"); - fs::create_dir_all(cwd.join("ops")).unwrap(); - fs::write(cwd.join("ops/run_segment.py"), b"# maintained runner\n").unwrap(); - fs::write(cwd.join("ops/segment_artifacts.py"), b"# artifacts\n").unwrap(); - let runtime_root = root.join("runtime"); - fs::create_dir_all(&runtime_root).unwrap(); - let task_file = root.join("task.json"); - fs::write(&task_file, b"{}\n").unwrap(); - let input_root = root.join("inputs"); - fs::create_dir_all(&input_root).unwrap(); - let (marker, gate) = (root.join("trainer-starts"), root.join("trainer-gate")); - let bin = root.join("bin"); - fs::create_dir_all(&bin).unwrap(); - let python = bin.join("python3"); - fs::write( - &python, - format!( - "#!/bin/sh\nprintf x >> '{}'\nwhile [ ! -e '{}' ] && [ -d '{}' ]; do sleep 0.05; done\n", - marker.display(), - gate.display(), - root.display() - ), - ) - .unwrap(); - fs::set_permissions(&python, fs::Permissions::from_mode(0o755)).unwrap(); - Self { - cwd, - runtime_root, - task_file, - input_root, - python, - marker, - gate, - } - } - - fn spec(&self, thread: ThreadId, name: &str) -> NormalizedSpec { - homebased::spec::parse_normalized_value(&serde_json::json!({ - "api_version": 1, - "thread": thread, - "name": name, - "cwd": self.cwd, - "timeout": "4h", - "workload": { - "type": "task", - "command": [ - self.python, "-m", "ops.run_segment", "run", - "--task", self.task_file, - "--input-root", self.input_root, - "--runtime-root", self.runtime_root, - "--image-digest", "not-validated-by-shape" - ] - } - })) - .unwrap() - } - - fn starts(&self) -> usize { - fs::read(&self.marker).map_or(0, |bytes| bytes.len()) - } - - /// Write the running attempt's request and hold its ownership lock - fn hold_attempt( - &self, - binding: &homebased::resource::trainer_publication::AttemptBinding, - ) -> fs::File { - let attempt = self.runtime_root.join("attempts").join(&binding.attempt_id); - fs::create_dir_all(&attempt).unwrap(); - let request = serde_json::json!({ - "schema_version": 1, - "binding": binding, - "adapter": {"kind": "speakrs"}, - "start_mode": {"kind": "fresh"}, - "payload": {"training": {"epochs": 1}}, - "input_view": { - "schema_version": 1, - "revision_id": binding.campaign_revision_id, - "root": "/trainer/input", - }, - "inputs": [], - "expected_outputs": [{"path": "checkpoint.bin", "kind": "file"}], - "resources": {"accelerator": "cuda"}, - "worker": { - "worker_id": "trainer-worker", - "executable_digest": "a".repeat(64), - "source_digest": "b".repeat(64), - "environment_digest": "c".repeat(64), - }, - }); - fs::write( - attempt.join("request.json"), - serde_json::to_vec(&request).unwrap(), - ) - .unwrap(); - let lock = fs::OpenOptions::new() - .create(true) - .truncate(false) - .read(true) - .write(true) - .open(self.runtime_root.join(".segment.lock")) - .unwrap(); - lock.try_lock().unwrap(); - lock - } -} - -fn attempt_binding(attempt: &str) -> homebased::resource::trainer_publication::AttemptBinding { - homebased::resource::trainer_publication::AttemptBinding { - campaign_id: "campaign-a".into(), - campaign_revision_id: "revision-a".into(), - task_id: "trainer-task-a".into(), - attempt_id: attempt.into(), - attempt_number: 1, - ownership_token: "owner-a".into(), - } -} - -/// Fake Codex on the supervisor machine that records every callback delivery -struct SupervisorCallbacks { - bin: PathBuf, - log: PathBuf, - cwd: PathBuf, -} - -impl SupervisorCallbacks { - fn new(supervisor: &Daemon) -> Self { - use std::os::unix::fs::PermissionsExt; - - let cwd = supervisor.user_home.canonicalize().unwrap(); - let bin = cwd.join("bin"); - fs::create_dir_all(&bin).unwrap(); - let log = cwd.join("codex-calls"); - let codex = bin.join("codex"); - fs::write( - &codex, - format!("#!/bin/sh\necho \"$@\" >> '{}'\n", log.display()), - ) - .unwrap(); - fs::set_permissions(&codex, fs::Permissions::from_mode(0o755)).unwrap(); - Self { bin, log, cwd } - } - - fn env(&self) -> Value { - serde_json::json!({ - "path": format!("{}:/bin:/usr/bin", self.bin.display()), - "home": self.cwd, - }) - } - - fn calls(&self) -> String { - fs::read_to_string(&self.log).unwrap_or_default() - } -} - -/// Authority and supervisor daemons that know each other, with one resource -/// whose supervisor thread runs on the supervisor machine -struct RemoteBackground { - // the trainer drops first so its gate opens before the daemon directories go - trainer: FleetTrainer, - callbacks: SupervisorCallbacks, - authority: Daemon, - supervisor: Daemon, - resource: ResourceId, - thread: ThreadId, -} - -impl RemoteBackground { - async fn start(names: (&str, &str)) -> Self { - let mut authority = Daemon::start(names.0, true); - let mut supervisor = Daemon::start(names.1, true); - // no daemon may reach a real Codex through an inherited pin or PATH - let callbacks = SupervisorCallbacks::new(&supervisor); - for daemon in [&mut authority, &mut supervisor] { - daemon.codex_override = Some(callbacks.bin.join("codex")); - daemon.restart(); - } - add_peer(&authority, &supervisor); - add_peer(&supervisor, &authority); - let resource = ResourceId::new(); - let thread = ThreadId(uuid::Uuid::now_v7()); - socket(&authority) - .post( - homebased::resource::api::RESOURCE_REGISTER_PATH, - &serde_json::json!({ - "api_version": 1, - "spec": { - "id": resource, - "display_name": "remote gpu", - "supervisor": { "machine": supervisor.machine_id(), "thread": thread }, - }, - }), - ) - .await - .unwrap(); - let trainer = FleetTrainer::new(&authority); - Self { - trainer, - callbacks, - authority, - supervisor, - resource, - thread, - } - } - - fn background_path(&self) -> String { - format!("/v1/resources/{}/background", self.resource.as_uuid()) - } - - fn body(&self, request: RequestId, spec: &NormalizedSpec) -> Value { - serde_json::json!({ - "api_version": 1, - "request_id": request, - "spec": spec, - "env": self.callbacks.env(), - "callback_cwd": self.callbacks.cwd, - }) - } - - async fn launch( - &self, - request: RequestId, - spec: &NormalizedSpec, - ) -> Result { - socket(&self.supervisor) - .post(&self.background_path(), &self.body(request, spec)) - .await - } - - async fn bind_attempt( - &self, - task: TaskId, - attempt: &homebased::resource::trainer_publication::AttemptBinding, - ) -> Result { - socket(&self.supervisor) - .post( - &format!( - "/v1/resources/{}/background/{task}/trainer-attempt", - self.resource.as_uuid() - ), - &serde_json::json!({ "api_version": 1, "attempt_binding": attempt }), - ) - .await - } - - async fn queue_request(&self) -> Value { - let spec = homebased::spec::parse_normalized_value(&serde_json::json!({ - "api_version": 1, - "thread": ThreadId(uuid::Uuid::now_v7()), - "name": "optimization", - "cwd": self.authority.user_home, - "timeout": "30m", - "workload": { "type": "task", "command": ["/bin/echo", "optimize"] } - })) - .unwrap(); - socket(&self.supervisor) - .post( - &format!("/v1/resources/{}/requests", self.resource.as_uuid()), - &self.body(RequestId::new(), &spec), - ) - .await - .unwrap() - } - - async fn detail(&self) -> Value { - socket(&self.authority) - .get(&format!("/v1/resources/{}", self.resource.as_uuid())) - .await - .unwrap() - } - - fn supervisor_route(&self, request: RequestId) -> Option { - Store::open(&self.supervisor.home.join("homebased.sqlite")) - .unwrap() - .origin_route_by_request(request) - .unwrap() - } - - /// Wire launch that a lost reply would repeat, built from a saved supervisor route - fn wire_launch(&self, route: &OriginRoute) -> Value { - use homebased::resource::background_launch::{ - ResourceBackgroundOperation, ResourceBackgroundRequest, - }; - - let SubmissionState::ResourceBackground { binding, .. } = &route.submission else { - panic!("the supervisor must save a background launch route"); - }; - let spec = route.current_spec().unwrap().clone(); - serde_json::to_value(ResourceBackgroundRequest::new( - homebased::fleet::protocol::CLUSTER_PROTOCOL_VERSION.0, - binding.assignment, - ResourceBackgroundOperation::Launch { - expected_state_revision: binding.expected_state_revision, - request_id: route.request, - task_id: route.task, - normalized_spec_sha256: homebased::submission::normalized_spec_sha256(&spec) - .unwrap(), - spec, - }, - )) - .unwrap() - } -} - -/// Whether a socket call failed with the daemon error text for one resource refusal -/// -/// The socket client keeps resource error codes only in the daemon's message -fn refused(result: Result, text: &str) -> bool { - matches!(result, Err(error) if error.to_string().contains(text)) -} - -fn socket(daemon: &Daemon) -> homebased::client::Client { - homebased::client::Client::new(daemon.home.join("homebased.sock")) -} - -async fn post_background(authority: &Daemon, request: &Value) -> (u16, Value) { - let response = ClusterClient::default() - .post_json( - &authority.address(), - homebased::resource::background_launch::RESOURCE_BACKGROUND_PATH, - request, - ) - .await - .unwrap(); - ( - response.status.as_u16(), - serde_json::from_slice(&response.body).unwrap_or(Value::Null), - ) -} - -fn background_receipt_count(daemon: &Daemon) -> i64 { - rusqlite::Connection::open(daemon.home.join("homebased.sqlite")) - .unwrap() - .query_row( - "SELECT COUNT(*) FROM resource_background_launches", - [], - |row| row.get(0), - ) - .unwrap() -} - -#[tokio::test(flavor = "multi_thread")] -async fn remote_supervisor_first_launch_runs_once_and_binds_its_trainer_attempt() { - let fixture = RemoteBackground::start(("bg-authority", "bg-supervisor")).await; - let (authority, supervisor) = (&fixture.authority, &fixture.supervisor); - let request = RequestId::new(); - let spec = fixture.trainer.spec(fixture.thread, "remote trainer"); - - let launched = fixture.launch(request, &spec).await.unwrap(); - assert_eq!(launched["outcome"]["type"], "inserted", "{launched}"); - let task: TaskId = serde_json::from_value(launched["task_id"].clone()).unwrap(); - // the launch binds a task but registers nothing before its confirmed start - assert_eq!( - launched["resource"]["registered_background_task"], - Value::Null - ); - - // the authority keeps the execution identity; the supervisor machine owns the route - assert_eq!(task_count(authority, task), 1); - assert_eq!(background_receipt_count(authority), 1); - let authority_store = Store::open(&authority.home.join("homebased.sqlite")).unwrap(); - assert!( - authority_store - .origin_route_by_task(task) - .unwrap() - .is_none() - ); - let Some(homebased::submission::ExecutorIdentity::Accepted(record)) = - authority_store.executor_identity(task).unwrap() - else { - panic!("the authority must accept the launch"); - }; - assert_eq!(record.origin_machine, supervisor.machine_id()); - assert_eq!(record.execution_machine, authority.machine_id()); - let route = fixture.supervisor_route(request).unwrap(); - assert_eq!(route.task, task); - assert_eq!(route.thread, fixture.thread); - assert_eq!(route.callback.cwd, fixture.callbacks.cwd); - - // the confirmed start reaches the supervisor route and registers the task - let supervisor_db = supervisor.home.join("homebased.sqlite"); - assert!( - wait_for(Duration::from_secs(20), || { - Store::open(&supervisor_db) - .unwrap() - .origin_route_by_task(task) - .unwrap() - .is_some_and(|route| route.last_execution_state == Some(ProcessStatus::Running)) - }) - .await, - "the running event did not reach the supervisor route" - ); - assert!( - wait_for(Duration::from_secs(20), || { - Store::open(&authority.home.join("homebased.sqlite")) - .unwrap() - .origin_route_by_task(task) - .unwrap() - .is_none() - && fixture.trainer.starts() == 1 - }) - .await - ); - let mut registered = false; - for _ in 0..100 { - if fixture.detail().await["resource"]["registered_background_task"] - == serde_json::json!(task) - { - registered = true; - break; - } - tokio::time::sleep(Duration::from_millis(100)).await; - } - assert!(registered, "the confirmed start did not register the task"); - - // an exact retry and a repeated lost reply only observe the running task - let retried = fixture.launch(request, &spec).await.unwrap(); - assert_eq!(retried["outcome"]["type"], "existing", "{retried}"); - assert_eq!(retried["outcome"]["state"], "running"); - assert_eq!(retried["task_id"], serde_json::json!(task)); - let (status, body) = post_background(authority, &fixture.wire_launch(&route)).await; - assert_eq!(status, 200, "{body}"); - assert_eq!(body["outcome"]["type"], "accepted", "{body}"); - assert_eq!(body["outcome"]["acceptance"]["type"], "existing"); - let changed = fixture.trainer.spec(fixture.thread, "changed trainer"); - assert!(refused( - fixture.launch(request, &changed).await, - "resource operation conflict" - )); - assert_eq!(task_count(authority, task), 1); - assert_eq!(background_receipt_count(authority), 1); - - // without a held lock and attempt request the authority refuses the association - let attempt = attempt_binding("attempt-a"); - assert!(refused( - fixture.bind_attempt(task, &attempt).await, - "resource action not allowed" - )); - - // queued work opens a release action whose proof stays blocked without the association - fixture.queue_request().await; - let blocked_reason = |detail: &Value| { - detail["attention"]["code"] == "release_proof_unavailable" - && detail["attention"]["task_id"] == serde_json::json!(task) - }; - let mut blocked = Value::Null; - for _ in 0..100 { - blocked = fixture.detail().await; - if blocked_reason(&blocked) { - break; - } - tokio::time::sleep(Duration::from_millis(100)).await; - } - assert!(blocked_reason(&blocked), "{blocked}"); - // the association is the first release evidence that the authority requires - assert!( - blocked["attention"]["message"] - .as_str() - .unwrap() - .contains("TrainerAssociationMissing"), - "{blocked}" - ); - let phase = &blocked["loan"]["state"]["phase"]; - assert_eq!(phase["type"], "awaiting_release"); - assert_eq!(phase["observed_background_task"], serde_json::json!(task)); - assert_eq!(phase["watcher_intent"], Value::Null); - - // the authority verifies the running attempt from its own runtime evidence - let lock = fixture.trainer.hold_attempt(&attempt); - let bound = fixture.bind_attempt(task, &attempt).await.unwrap(); - assert_eq!(bound["task_id"], serde_json::json!(task)); - assert_eq!( - bound["runtime_root"], - serde_json::json!(fixture.trainer.runtime_root) - ); - assert_eq!(fixture.bind_attempt(task, &attempt).await.unwrap(), bound); - assert!(refused( - fixture - .bind_attempt(task, &attempt_binding("attempt-b")) - .await, - "resource operation conflict" - )); - - // with the association saved, release waits only for the trainer to stop - let mut associated = Value::Null; - for _ in 0..100 { - associated = fixture.detail().await; - if associated["attention"]["message"] - .as_str() - .is_some_and(|message| message.contains("TrainerNotCompleted")) - { - break; - } - tokio::time::sleep(Duration::from_millis(100)).await; - } - assert!( - associated["attention"]["message"] - .as_str() - .is_some_and(|message| message.contains("TrainerNotCompleted")), - "{associated}" - ); - assert_eq!( - associated["loan"]["state"]["phase"]["type"], - "awaiting_release" - ); - - // the trainer's result is delivered to the supervisor thread on its own machine - drop(lock); - fs::write(&fixture.trainer.gate, b"").unwrap(); - assert!( - wait_for(Duration::from_secs(30), || fixture - .callbacks - .calls() - .contains(&fixture.thread.to_string())) - .await, - "no callback reached the supervisor machine: {} route={:?} task={:?}", - fixture.callbacks.calls(), - fixture.supervisor_route(request), - Store::open(&authority.home.join("homebased.sqlite")) - .unwrap() - .get_task(task) - .unwrap() - .map(|row| (row.status(), row.callback_status)), - ); - assert_eq!(fixture.trainer.starts(), 1); -} - -#[tokio::test(flavor = "multi_thread")] -async fn remote_background_launch_refuses_wrong_owners_revision_and_queued_work_without_records() { - use homebased::resource::background_launch::BackgroundLaunchBinding; - - let fixture = RemoteBackground::start(("bg-refuse-authority", "bg-refuse-supervisor")).await; - let (authority, supervisor) = (&fixture.authority, &fixture.supervisor); - let spec = fixture.trainer.spec(fixture.thread, "trainer"); - - // the authority cannot own the remote thread's callbacks, so its socket writes nothing - assert!(refused( - socket(authority) - .post( - &fixture.background_path(), - &fixture.body(RequestId::new(), &spec) - ) - .await, - "resource operation unavailable" - )); - // a spec for another thread is refused before the supervisor saves a route - let other_thread_request = RequestId::new(); - let other_thread = fixture - .trainer - .spec(ThreadId(uuid::Uuid::now_v7()), "trainer"); - assert!(refused( - fixture.launch(other_thread_request, &other_thread).await, - "resource action not allowed" - )); - assert!(fixture.supervisor_route(other_thread_request).is_none()); - - // wire requests with the wrong owners, unknown fields, or no saved route are refused - let saved_route = |binding: BackgroundLaunchBinding, spec: &NormalizedSpec| { - OriginRoute::new_resource_background(homebased::submission::NewResourceBackgroundRoute { - request: RequestId::new(), - task: TaskId::new(), - callback: CallbackContext { - env: TaskEnv { - path: "/bin:/usr/bin".into(), - home: supervisor.user_home.to_string_lossy().into_owned(), - }, - cwd: supervisor.user_home.clone(), - codex: PathBuf::from("/bin/true").into(), - }, - spec: spec.clone(), - binding, - }) - .unwrap() - }; - let detail = fixture.detail().await; - let resource: homebased::resource::Resource = - serde_json::from_value(detail["resource"].clone()).unwrap(); - let current = BackgroundLaunchBinding { - assignment: homebased::resource::background_launch::BackgroundSupervisorAssignment::of( - &resource, - ), - expected_state_revision: resource.state_revision, - }; - let unsaved = saved_route(current, &spec); - let wire = fixture.wire_launch(&unsaved); - let mut wrong_destination = wire.clone(); - wrong_destination["destination_machine"] = serde_json::json!(supervisor.machine_id()); - let mut wrong_source = wire.clone(); - wrong_source["source_machine"] = serde_json::json!(MachineId::new()); - let mut unknown_field = wire.clone(); - unknown_field["operation"]["callback"] = serde_json::json!({"cwd": "/tmp"}); - for request in [wrong_destination, wrong_source, unknown_field] { - assert_ne!(post_background(authority, &request).await.0, 200); - } - let (status, body) = post_background(authority, &wire).await; - assert_eq!(status, 200, "{body}"); - assert_eq!(body["outcome"]["reason"]["type"], "route_evidence_missing"); - - // saved routes whose assignment or revision is stale are refused by the authority - let mut stale_assignment = current; - stale_assignment.assignment.assignment_revision = - serde_json::from_value(serde_json::json!(5)).unwrap(); - let mut stale_revision = current; - stale_revision.expected_state_revision = serde_json::from_value(serde_json::json!(9)).unwrap(); - let supervisor_db = supervisor.home.join("homebased.sqlite"); - for (binding, reason) in [ - (stale_assignment, "not_current_supervisor"), - (stale_revision, "stale_revision"), - ] { - let route = saved_route(binding, &spec); - Store::open(&supervisor_db) - .unwrap() - .insert_origin_route(&route) - .unwrap(); - let (status, body) = post_background(authority, &fixture.wire_launch(&route)).await; - assert_eq!(status, 200, "{body}"); - assert_eq!(body["outcome"]["reason"]["type"], reason, "{body}"); - } - // a saved route with other content is not evidence for this launch - let other = saved_route(current, &fixture.trainer.spec(fixture.thread, "other")); - Store::open(&supervisor_db) - .unwrap() - .insert_origin_route(&other) - .unwrap(); - let mut mismatched = fixture.wire_launch(&unsaved); - mismatched["operation"]["request_id"] = serde_json::json!(other.request); - mismatched["operation"]["task_id"] = serde_json::json!(other.task); - let (_, body) = post_background(authority, &mismatched).await; - assert_eq!(body["outcome"]["reason"]["type"], "route_evidence_mismatch"); - - // queued work keeps its serving place, and a fresh resource does not start it - let queued = fixture.queue_request().await; - let queued_task: TaskId = serde_json::from_value(queued["task_id"].clone()).unwrap(); - let request = RequestId::new(); - assert!(refused( - fixture.launch(request, &spec).await, - "resource action not allowed" - )); - assert!(matches!( - fixture.supervisor_route(request).unwrap().submission, - SubmissionState::ResourceBackground { - phase: homebased::submission::ResourceBackgroundRoutePhase::Rejected { .. }, - .. - } - )); - // the saved refusal answers a retry of the same request - assert!(refused( - fixture.launch(request, &spec).await, - "resource action not allowed" - )); - tokio::time::sleep(Duration::from_millis(500)).await; - assert_eq!(task_count(authority, queued_task), 0); - assert_eq!(background_receipt_count(authority), 0); - assert_eq!(fixture.trainer.starts(), 0); - let detail = fixture.detail().await; - assert_eq!(detail["loan"], Value::Null); - assert_eq!(detail["attention"]["code"], "queue_blocked", "{detail}"); -} - -#[tokio::test(flavor = "multi_thread")] -async fn saved_remote_background_route_retries_after_restart_and_lost_reply_without_a_second_trainer() - { - let mut fixture = RemoteBackground::start(("bg-retry-authority", "bg-retry-supervisor")).await; - let spec = fixture.trainer.spec(fixture.thread, "retried trainer"); - let detail = fixture.detail().await; - let resource: homebased::resource::Resource = - serde_json::from_value(detail["resource"].clone()).unwrap(); - let request = RequestId::new(); - let task = TaskId::new(); - // the supervisor saved its fixed-ID route, then the send never reached the authority - fixture.supervisor.stop(); - fixture.authority.stop(); - let route = - OriginRoute::new_resource_background(homebased::submission::NewResourceBackgroundRoute { - request, - task, - callback: CallbackContext { - env: TaskEnv { - path: format!("{}:/bin:/usr/bin", fixture.callbacks.bin.display()), - home: fixture.callbacks.cwd.to_string_lossy().into_owned(), - }, - cwd: fixture.callbacks.cwd.clone(), - codex: fixture.callbacks.bin.join("codex").into(), - }, - spec: spec.clone(), - binding: homebased::resource::background_launch::BackgroundLaunchBinding { - assignment: - homebased::resource::background_launch::BackgroundSupervisorAssignment::of( - &resource, - ), - expected_state_revision: resource.state_revision, - }, - }) - .unwrap(); - let supervisor_db = fixture.supervisor.home.join("homebased.sqlite"); - Store::open(&supervisor_db) - .unwrap() - .insert_origin_route(&route) - .unwrap(); - let unresolved = || { - Store::open(&supervisor_db) - .unwrap() - .unknown_resource_background_routes() - .unwrap() - .len() - }; - - // startup recovery with the authority offline keeps the same unresolved route - fixture.supervisor.spawn(); - tokio::time::sleep(Duration::from_millis(500)).await; - assert_eq!(unresolved(), 1); - assert_eq!(task_count(&fixture.authority, task), 0); - - fixture.authority.spawn(); - wait_for_probe(&fixture.authority.address()).await; - fixture.supervisor.restart(); - assert!( - wait_for(Duration::from_secs(20), || unresolved() == 0).await, - "startup recovery did not resolve the saved route" - ); - assert_eq!(task_count(&fixture.authority, task), 1); - assert!( - wait_for(Duration::from_secs(20), || fixture.trainer.starts() == 1).await, - "the accepted launch did not start" - ); - - // a lost reply and a socket retry are answered by the same identity - let (status, body) = post_background(&fixture.authority, &fixture.wire_launch(&route)).await; - assert_eq!(status, 200, "{body}"); - assert_eq!(body["outcome"]["acceptance"]["type"], "existing", "{body}"); - let retried = fixture.launch(request, &spec).await.unwrap(); - assert_eq!(retried["outcome"]["type"], "existing", "{retried}"); - assert_eq!(retried["task_id"], serde_json::json!(task)); - - // an authority restart observes the running trainer and never relaunches it - fixture.authority.restart(); - wait_for_probe(&fixture.authority.address()).await; - let retried = fixture.launch(request, &spec).await.unwrap(); - assert_eq!(retried["outcome"]["type"], "existing", "{retried}"); - assert_eq!(task_count(&fixture.authority, task), 1); - assert_eq!(background_receipt_count(&fixture.authority), 1); - tokio::time::sleep(Duration::from_millis(500)).await; - assert_eq!(fixture.trainer.starts(), 1); - fs::write(&fixture.trainer.gate, b"").unwrap(); -} - -#[test] -fn remote_submit_with_a_missing_executor_cwd_is_a_typed_rejection() { - let origin = Daemon::start("missing-cwd-origin", true); - let executor = Daemon::start("remote-executor", true); - add_peer(&origin, &executor); - let mut spec = cli_remote_spec(&executor, vec!["/bin/echo", "hello"]); - spec.cwd = PathBuf::from("~/not-present"); - let request = RequestId::new(); - - // the executor refuses at acceptance, and a retry answers from the saved refusal - for _ in 0..2 { - let output = submit_file(&origin, &spec, request); - assert!(!output.status.success()); - let error: Value = serde_json::from_slice(&output.stderr).unwrap(); - assert_eq!(error["error"]["code"], "invalid_cwd", "{error}"); - assert_eq!(error["error"]["input"]["pointer"], "/cwd"); - assert_eq!(error["error"]["input"]["value"], "~/not-present"); - assert_eq!(error["error"]["input"]["problem"], "not_found"); + // the executor refuses at acceptance, and a retry answers from the saved refusal + for _ in 0..2 { + let output = submit_file(&origin, &spec, request); + assert!(!output.status.success()); + let error: Value = serde_json::from_slice(&output.stderr).unwrap(); + assert_eq!(error["error"]["code"], "invalid_cwd", "{error}"); + assert_eq!(error["error"]["input"]["pointer"], "/cwd"); + assert_eq!(error["error"]["input"]["value"], "~/not-present"); + assert_eq!(error["error"]["input"]["problem"], "not_found"); } let rejected: String = rusqlite::Connection::open(executor.home.join("homebased.sqlite")) .unwrap() @@ -4837,59 +2865,6 @@ fn remote_submit_with_a_missing_executor_cwd_is_a_typed_rejection() { assert!(rejected.contains("cwd_not_found"), "{rejected}"); } -#[tokio::test(flavor = "multi_thread")] -async fn remote_resource_request_with_a_missing_cwd_is_rejected_at_acceptance() { - let origin = Daemon::start("resource-missing-cwd-origin", true); - let authority = Daemon::start("resource-missing-cwd-authority", true); - add_peer(&origin, &authority); - add_peer(&authority, &origin); - let resource = ResourceId::new(); - seed_authority_resource(&authority, resource); - let mut spec = remote_spec(&authority, vec!["/bin/echo", "never-queued"]); - spec.cwd = PathBuf::from("/nonexistent/homebased-missing-cwd"); - let route = seed_resource_route( - &origin, - &authority, - resource, - RequestId::new(), - TaskId::new(), - &spec, - ); - - let (status, rejected) = post_resource_queue(&authority, &origin, &route).await; - assert_eq!(status, 200, "{rejected}"); - let receipt: ResourceQueueReceipt = - serde_json::from_value(rejected["receipt"].clone()).unwrap(); - let saved = Store::open(&origin.home.join("homebased.sqlite")) - .unwrap() - .resolve_resource_route(&receipt) - .unwrap(); - let SubmissionState::Resource { - phase: ResourceRoutePhase::Rejected { reason }, - .. - } = saved.submission - else { - panic!( - "the authority must refuse the request: {:?}", - saved.submission - ); - }; - assert_eq!(reason, "cwd_not_found"); - // the origin rebuilds the typed error from its own spec - let error = homebased::spec::HostInputRejection::parse(&reason) - .unwrap() - .into_error(&spec); - assert_eq!(error.code(), "invalid_cwd"); - - let queued: i64 = rusqlite::Connection::open(authority.home.join("homebased.sqlite")) - .unwrap() - .query_row("SELECT COUNT(*) FROM resource_requests", [], |row| { - row.get(0) - }) - .unwrap(); - assert_eq!(queued, 0); -} - /// Origin and executor daemons that know each other, with callbacks recorded on the origin fn dependency_fleet(name: &str) -> (Daemon, Daemon) { use std::os::unix::fs::PermissionsExt; diff --git a/tests/integration.rs b/tests/integration.rs index a73d157..8a521cc 100644 --- a/tests/integration.rs +++ b/tests/integration.rs @@ -2617,7 +2617,8 @@ fn cancel_on_terminal_task_is_idempotent() { let h = Harness::new(); let id = h.submit(&Harness::spec("claude", "quick")); h.wait_status(&id, "succeeded"); - let before = h.queue_messages().len(); + // the success callback can land after the status flips, so count from it + let before = h.wait_for_event(&id, "TASK_SUCCEEDED").len(); let out = h .cmd() .args(["--json", "task", "cancel", &id]) @@ -3777,29 +3778,8 @@ fn mistyped_thread_is_rejected_before_any_work_starts() { let mistyped = "01a0ab97-a7aa-7463-a5b0-8d500e40e4ff"; let mut spec = Harness::task_spec(&["/bin/echo", "hello"]); spec["thread"] = json!(mistyped); - let resource = uuid::Uuid::now_v7().to_string(); - let request = uuid::Uuid::now_v7().to_string(); - for args in [ - vec!["task", "submit"], - vec!["task", "submit", "--dry-run"], - vec![ - "resource", - "request", - "submit", - &resource, - "--request-id", - &request, - ], - vec![ - "resource", - "background", - "submit", - &resource, - "--request-id", - &request, - ], - ] { + for args in [vec!["task", "submit"], vec!["task", "submit", "--dry-run"]] { let (code, error) = submit_error(&h, &spec, &args, None); assert_eq!(code, 2, "{args:?} {error}"); assert_eq!(error["api_version"], 1); diff --git a/tests/message_receiver.rs b/tests/message_receiver.rs index 30ddf06..a556fa0 100644 --- a/tests/message_receiver.rs +++ b/tests/message_receiver.rs @@ -1,4 +1,4 @@ -//! Message and supervisor-notice receiver behavior through a Fleet-enabled daemon +//! Message receiver behavior through a Fleet-enabled daemon #![allow(clippy::unwrap_used, clippy::expect_used)] @@ -16,10 +16,6 @@ use homebased::domain::{API_VERSION, ThreadId}; use homebased::fleet::protocol::CLUSTER_PROTOCOL_VERSION; use homebased::machine::MachineId; use homebased::message::{MessageId, MessageRequest}; -use homebased::resource::{ - ActionId, AssignmentRevision, DeliveryAttemptId, LoanId, NoticeId, ResourceRevision, - SupervisorAddress, SupervisorNoticePayload, SupervisorNoticeRequest, -}; use homebased::store::Store; use serde_json::{Value, json}; use sha2::{Digest, Sha256}; @@ -122,10 +118,6 @@ impl Daemon { post_json(&self.address, "/v1/cluster/messages", body) } - fn send_notice(&self, body: &Value) -> HttpResponse { - post_json(&self.address, "/v1/cluster/resource-notices", body) - } - fn queue_calls(&self) -> Vec { fs::read_to_string(&self.queue_log) .unwrap_or_default() @@ -141,13 +133,6 @@ impl Daemon { .unwrap() } - fn notice_delivery( - &self, - attempt_id: DeliveryAttemptId, - ) -> homebased::message::MessageDelivery { - self.delivery(MessageId::from_uuid(attempt_id.as_uuid()).unwrap()) - } - fn restart(&mut self) { self.stop(); self.spawn(); @@ -251,32 +236,6 @@ fn request(id: MessageId, destination: MachineId, recipient: Value, body: &str) }) } -fn notice_request( - destination_machine: MachineId, - destination_thread: ThreadId, - attempt_id: DeliveryAttemptId, -) -> Value { - serde_json::to_value(SupervisorNoticeRequest { - api_version: API_VERSION, - protocol_version: CLUSTER_PROTOCOL_VERSION.0, - source_machine: MachineId::new(), - destination: SupervisorAddress { - machine: destination_machine, - thread: destination_thread, - }, - notice_id: NoticeId::new(), - loan_id: LoanId::new(), - action_id: ActionId::new(), - state_revision: ResourceRevision::new(9), - assignment_revision: AssignmentRevision::new(3), - attempt_id, - payload: SupervisorNoticePayload::ReleaseRequired { - task_id: homebased::domain::TaskId::new(), - }, - }) - .unwrap() -} - fn post_json(address: &str, path: &str, body: &Value) -> HttpResponse { let bytes = serde_json::to_vec(body).unwrap(); let request = format!( @@ -526,205 +485,3 @@ fn receiver_delivers_to_claude_sessions_without_codex_queue() { assert!(message.contains("no T3 thread owns it"), "{message}"); assert!(daemon.queue_calls().is_empty()); } - -#[test] -fn receiver_queues_resource_notice_by_attempt_on_the_exact_thread() { - let daemon = Daemon::start(); - let exact_cwd = daemon.user_home.join("exact-project"); - let newer_cwd = daemon.user_home.join("newer-project"); - fs::create_dir_all(&exact_cwd).unwrap(); - fs::create_dir_all(&newer_cwd).unwrap(); - - let exact_thread = ThreadId(Uuid::now_v7()); - let newer_thread = ThreadId(Uuid::now_v7()); - write_session(&daemon.user_home, "notice-exact", exact_thread, &exact_cwd); - write_session(&daemon.user_home, "notice-newer", newer_thread, &newer_cwd); - - let wrong_machine_attempt = DeliveryAttemptId::new(); - let wrong_machine = notice_request(MachineId::new(), exact_thread, wrong_machine_attempt); - let mismatch = daemon.send_notice(&wrong_machine); - assert_eq!(mismatch.status, 409, "{:?}", mismatch.body); - assert_eq!(mismatch.body["error"]["code"], "machine_identity_mismatch"); - assert!( - daemon - .notice_delivery(wrong_machine_attempt) - .attempt - .is_none() - ); - - let missing_thread_attempt = DeliveryAttemptId::new(); - let missing_thread = notice_request( - daemon.machine_id(), - ThreadId(Uuid::now_v7()), - missing_thread_attempt, - ); - let missing = daemon.send_notice(&missing_thread); - assert_eq!(missing.status, 404, "{:?}", missing.body); - assert_eq!(missing.body["error"]["code"], "agent_thread_not_found"); - assert!( - daemon - .notice_delivery(missing_thread_attempt) - .attempt - .is_none() - ); - assert!(daemon.queue_calls().is_empty()); - - let attempt_id = DeliveryAttemptId::new(); - let notice = notice_request(daemon.machine_id(), exact_thread, attempt_id); - let response = daemon.send_notice(¬ice); - assert_eq!(response.status, 200, "{:?}", response.body); - assert_eq!(response.body["receipt"]["attempt_id"], json!(attempt_id)); - assert_eq!( - response.body["receipt"]["destination_thread"], - exact_thread.to_string() - ); - let first_queue_line = daemon.queue_calls().remove(0); - assert!(first_queue_line.contains(&format!( - "--thread {exact_thread} --message HOMEBASED_RESOURCE_NOTICE " - ))); - assert!(!first_queue_line.contains("HOMEBASED_MESSAGE ")); - assert!(!first_queue_line.contains("HOMEBASED_EVENT ")); - let queued: Value = serde_json::from_str( - first_queue_line - .split_once("--message HOMEBASED_RESOURCE_NOTICE ") - .unwrap() - .1, - ) - .unwrap(); - assert_eq!(queued["attempt_id"], json!(attempt_id)); - assert!(queued.get("notice_id").is_some()); - assert!(queued.get("loan_id").is_some()); - assert!(queued.get("action_id").is_some()); - assert!(queued.get("state_revision").is_some()); - assert!(queued.get("assignment_revision").is_some()); - assert_eq!(queued["destination"]["thread"], json!(exact_thread)); - assert!(queued.get("delivery").is_none()); - - let saved_receipt = response.body["receipt"].clone(); - let duplicate = daemon.send_notice(¬ice); - assert_eq!(duplicate.status, 200, "{:?}", duplicate.body); - assert_eq!(duplicate.body["receipt"], saved_receipt); - assert_eq!(daemon.queue_calls().len(), 1); - - let mut conflicting = notice.clone(); - conflicting["payload"] = json!({ - "type": "attention_required", - "reason": "different content for the same attempt", - }); - let conflict = daemon.send_notice(&conflicting); - assert_eq!(conflict.status, 409, "{:?}", conflict.body); - assert_eq!(conflict.body["error"]["code"], "message_conflict"); - assert_eq!(daemon.queue_calls().len(), 1); - - let retargeted_attempt = DeliveryAttemptId::new(); - let mut retargeted = notice.clone(); - retargeted["attempt_id"] = json!(retargeted_attempt); - retargeted["assignment_revision"] = json!(4); - retargeted["destination"]["thread"] = json!(newer_thread); - let retargeted_response = daemon.send_notice(&retargeted); - assert_eq!( - retargeted_response.status, 200, - "{:?}", - retargeted_response.body - ); - assert_eq!( - retargeted_response.body["receipt"]["notice_id"], - notice["notice_id"] - ); - assert_eq!( - retargeted_response.body["receipt"]["attempt_id"], - json!(retargeted_attempt) - ); - assert_eq!( - retargeted_response.body["receipt"]["destination_thread"], - newer_thread.to_string() - ); - assert!(daemon.queue_calls()[1].contains(&format!( - "--thread {newer_thread} --message HOMEBASED_RESOURCE_NOTICE " - ))); - - let direct = request( - MessageId::new(), - daemon.machine_id(), - json!({"kind": "thread", "thread": exact_thread}), - "Keep direct message delivery unchanged", - ); - let direct_response = daemon.send(&direct); - assert_eq!(direct_response.status, 200, "{:?}", direct_response.body); - let queue_calls = daemon.queue_calls(); - assert!(queue_calls[2].contains("HOMEBASED_MESSAGE ")); - assert!(!queue_calls[2].contains("HOMEBASED_RESOURCE_NOTICE ")); - - let mut notice_source = request( - MessageId::new(), - daemon.machine_id(), - json!({"kind": "thread", "thread": exact_thread}), - "A notice source cannot use the direct-message route", - ); - notice_source["source"] = json!({ - "kind": "resource_notice", - "machine": daemon.machine_id(), - "notice_id": NoticeId::new(), - }); - let rejected_source = daemon.send(¬ice_source); - assert_eq!(rejected_source.status, 400, "{:?}", rejected_source.body); - assert_eq!(rejected_source.body["error"]["code"], "message_invalid"); - assert_eq!(daemon.queue_calls().len(), 3); - - let mut unknown = notice_request(daemon.machine_id(), exact_thread, DeliveryAttemptId::new()); - unknown["unknown"] = json!(true); - assert!((400..500).contains(&daemon.send_notice(&unknown).status)); - - let mut invalid = notice_request(daemon.machine_id(), exact_thread, DeliveryAttemptId::new()); - invalid["attempt_id"] = json!(Uuid::nil()); - // a nil identity cannot decode, so the body is refused before validation - assert_eq!(daemon.send_notice(&invalid).status, 422); - assert_eq!(daemon.queue_calls().len(), 3); -} - -#[test] -fn failed_notice_attempt_is_retained_and_same_attempt_can_retry() { - let daemon = Daemon::start(); - let exact_cwd = daemon.user_home.join("exact-project"); - let other_cwd = daemon.user_home.join("other-project"); - fs::create_dir_all(&exact_cwd).unwrap(); - fs::create_dir_all(&other_cwd).unwrap(); - let exact_thread = ThreadId(Uuid::now_v7()); - let other_thread = ThreadId(Uuid::now_v7()); - write_session( - &daemon.user_home, - "failed-notice-exact", - exact_thread, - &exact_cwd, - ); - write_session( - &daemon.user_home, - "failed-notice-other", - other_thread, - &other_cwd, - ); - - let attempt_id = DeliveryAttemptId::new(); - let request = notice_request(daemon.machine_id(), exact_thread, attempt_id); - fs::write(&daemon.queue_fail_marker, "fail").unwrap(); - let failed = daemon.send_notice(&request); - assert_eq!(failed.status, 503, "{:?}", failed.body); - assert_eq!(failed.body["error"]["code"], "message_delivery_failed"); - let delivery = daemon.notice_delivery(attempt_id); - assert!(delivery.attempt.is_some()); - assert!(delivery.receipt.is_none()); - assert_eq!(daemon.queue_calls().len(), 1); - - fs::remove_file(&daemon.queue_fail_marker).unwrap(); - let retried = daemon.send_notice(&request); - assert_eq!(retried.status, 200, "{:?}", retried.body); - assert_eq!( - retried.body["receipt"]["destination_thread"], - exact_thread.to_string() - ); - assert_eq!(daemon.queue_calls().len(), 2); - assert!(daemon.queue_calls()[1].contains(&format!( - "--thread {exact_thread} --message HOMEBASED_RESOURCE_NOTICE " - ))); - assert!(daemon.notice_delivery(attempt_id).receipt.is_some()); -} diff --git a/tests/message_sender.rs b/tests/message_sender.rs index af00b1f..3ada1e2 100644 --- a/tests/message_sender.rs +++ b/tests/message_sender.rs @@ -954,5 +954,12 @@ fn worker_message_reaches_a_running_claude_worker_on_its_execution_machine() { .unwrap() .contains("homebased task followup") ); - assert_eq!(receiver.queue_calls().len(), 1); + // the finished task's own callback also goes through `codex queue`, at a time + // the test does not control, so count only the worker message + let messages = receiver + .queue_calls() + .into_iter() + .filter(|call| call.contains("HOMEBASED_MESSAGE")) + .count(); + assert_eq!(messages, 1); } diff --git a/web/src/lib/api.ts b/web/src/lib/api.ts index fec2654..60d464b 100644 --- a/web/src/lib/api.ts +++ b/web/src/lib/api.ts @@ -593,23 +593,6 @@ export function fetchContentOrigin(): Promise { return getJson('/files/origin', ContentOriginSchema); } -/** - * Return an unknown JSON value for a feature client to validate with Effect Schema. - * The path stays relative so browser calls use the exact same origin as the dashboard. - */ -export function getJsonBody(path: string): Promise { - return requestJson(path, { method: 'GET' }); -} - -/** Return an unknown JSON value for a feature client to validate with Effect Schema. */ -export function postJsonBody(path: string, body: unknown): Promise { - return requestJson(path, { - method: 'POST', - headers: { 'content-type': 'application/json' }, - body: JSON.stringify(body) - }); -} - /** * Absolute URL on the content origin for a UTF-8 filesystem path. Uses the * current browser hostname so loopback, LAN, and Tailscale clients match. diff --git a/web/src/lib/components/ExpandToggle.svelte b/web/src/lib/components/ExpandToggle.svelte deleted file mode 100644 index 83969ad..0000000 --- a/web/src/lib/components/ExpandToggle.svelte +++ /dev/null @@ -1,35 +0,0 @@ - - - - diff --git a/web/src/lib/components/ResourceQueuePanel.svelte b/web/src/lib/components/ResourceQueuePanel.svelte deleted file mode 100644 index b085a26..0000000 --- a/web/src/lib/components/ResourceQueuePanel.svelte +++ /dev/null @@ -1,344 +0,0 @@ - - -{#snippet iconButton(Icon: typeof X, label: string, disabled: boolean, onclick: () => void)} - -{/snippet} - -{#snippet jobThread(ref: ThreadRef | null)} - {#if ref} - - {/if} -{/snippet} - -{#snippet holder(label: string, task: ResourceTaskSummary, authority: string)} - {@const machine = runsOn(task, authority)} - {@const peerHref = peerTaskHref(machine, task.id)} -
- {label} -
- {#if machine?.location.type === 'local'} - {task.display_name} - {:else if peerHref} - {task.display_name} - {:else} - {task.display_name} - {/if} -
- {@render jobThread(resourceTaskThread(task, authority))} -
- {#if isProcessStatus(task.status)} - - {:else} - {task.status} - {/if} - {#if task.created_at} - - {/if} -
-
-{/snippet} - -
- {#each queues as item (item.resource.id)} - {@const authority = machineName(item.resource.authority_machine)} - {@const operationState = operationManager.stateFor(item.resource.id)} - {@const resourceActionDisabled = - resourceStore.error !== null || operationState.operation !== null || operationState.busy} - -
-
-
- -
- {#if operationState.operation || operationState.error || operationState.message} -
- {operationMessage(operationState)} - {#if operationState.operation} - - {:else if operationState.error} - - {/if} -
- {/if} - {#if item.current} - {@render holder('now', item.current, item.resource.authority_machine)} - {/if} - {#if item.background} - {@render holder('background', item.background, item.resource.authority_machine)} - {/if} - {#if !item.current && !item.background} -

Nothing holds this resource

- {/if} - {#if item.returnPending} -

{item.returnPending}

- {/if} - -
- - queue {item.queue.length} - - {#if item.queue.length === 0} -

Empty

- {:else} -
    - {#each item.queue as request, index (request.request_id)} - {@const origin = machineName(request.origin_machine)} - {@const confirming = isConfirming(item.resource.id, request.request_id)} -
  1. - - {index + 1} - - - - {request.display_name} - - {@render jobThread(requestThread(request))} - - {#if request.origin_machine !== item.resource.authority_machine} - {origin} - {/if} -
    - {@render iconButton( - ArrowUp, - 'Move up', - index === 0 || resourceActionDisabled, - () => moveQueued(item, request.request_id, index, 'up') - )} - {@render iconButton( - ArrowDown, - 'Move down', - index === item.queue.length - 1 || resourceActionDisabled, - () => moveQueued(item, request.request_id, index, 'down') - )} - {#if confirming} - Cancel? - - - {:else} - {@render iconButton( - X, - 'Cancel queued request', - resourceActionDisabled, - () => { - confirmingCancel = { - resourceId: item.resource.id, - requestId: request.request_id - }; - } - )} - {/if} -
    -
  2. - {/each} -
- {/if} -
-
-
- {/each} -
diff --git a/web/src/lib/components/TaskList.svelte b/web/src/lib/components/TaskList.svelte index b0aeafb..6a3d6d3 100644 --- a/web/src/lib/components/TaskList.svelte +++ b/web/src/lib/components/TaskList.svelte @@ -34,8 +34,6 @@ threadTitle: (ref: ThreadRef) => string | null; /** Rendered in place of rows when the list is empty. */ empty: Snippet; - /** Controls for a title bar; the bar shows only when there are some. */ - actions?: Snippet; class?: string; } @@ -48,7 +46,6 @@ onProject, threadTitle, empty, - actions, class: className }: Props = $props(); @@ -64,14 +61,6 @@
- {#if actions} -
- Tasks {tasks.length} - {@render actions()} -
- {/if}