From 41e8eb87522c6c52894550151104bb467c45a1ef Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:10:14 -0700 Subject: [PATCH 001/126] Add the filesystem snapshots configuration block with its parse refusals --- golem-debugging-service/src/config.rs | 1 + .../config/worker-executor.sample.env | 3 + .../config/worker-executor.toml | 15 + .../src/services/golem_config.rs | 347 +++++++++++++++++- 4 files changed, 364 insertions(+), 2 deletions(-) diff --git a/golem-debugging-service/src/config.rs b/golem-debugging-service/src/config.rs index 786571e83d..39432f2061 100644 --- a/golem-debugging-service/src/config.rs +++ b/golem-debugging-service/src/config.rs @@ -91,6 +91,7 @@ impl DebugConfig { public_worker_api: self.public_worker_api, memory: self.memory, filesystem_storage: Default::default(), + filesystem_snapshots: Default::default(), resource_usage_metering: Default::default(), rdbms: self.rdbms, resource_limits: self.resource_limits, diff --git a/golem-worker-executor/config/worker-executor.sample.env b/golem-worker-executor/config/worker-executor.sample.env index b3511b1cac..ef8278a3d0 100644 --- a/golem-worker-executor/config/worker-executor.sample.env +++ b/golem-worker-executor/config/worker-executor.sample.env @@ -38,6 +38,7 @@ GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_CAPACITY=1000 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_EVICTION_INTERVAL="1m" GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__NANOS=0 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__SECS=300 +GOLEM__FILESYSTEM_SNAPSHOTS__TYPE="Disabled" #GOLEM__FILESYSTEM_STORAGE__DETERMINISTIC_ROOT_DIR= #GOLEM__FILESYSTEM_STORAGE__MANAGED_XFS_ROOT_DIR= GOLEM__FILESYSTEM_STORAGE__CLEANUP_RETRY__MAX_ATTEMPTS=4 @@ -337,6 +338,7 @@ GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_CAPACITY=1000 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_EVICTION_INTERVAL="1m" GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__NANOS=0 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__SECS=300 +GOLEM__FILESYSTEM_SNAPSHOTS__TYPE="Disabled" #GOLEM__FILESYSTEM_STORAGE__DETERMINISTIC_ROOT_DIR= #GOLEM__FILESYSTEM_STORAGE__MANAGED_XFS_ROOT_DIR= GOLEM__FILESYSTEM_STORAGE__CLEANUP_RETRY__MAX_ATTEMPTS=4 @@ -614,6 +616,7 @@ GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_CAPACITY=1000 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_EVICTION_INTERVAL="1m" GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__NANOS=0 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__SECS=300 +GOLEM__FILESYSTEM_SNAPSHOTS__TYPE="Disabled" #GOLEM__FILESYSTEM_STORAGE__DETERMINISTIC_ROOT_DIR= #GOLEM__FILESYSTEM_STORAGE__MANAGED_XFS_ROOT_DIR= GOLEM__FILESYSTEM_STORAGE__CLEANUP_RETRY__MAX_ATTEMPTS=4 diff --git a/golem-worker-executor/config/worker-executor.toml b/golem-worker-executor/config/worker-executor.toml index 19ee85ad42..1d76c51c04 100644 --- a/golem-worker-executor/config/worker-executor.toml +++ b/golem-worker-executor/config/worker-executor.toml @@ -70,6 +70,11 @@ cache_eviction_interval = "1m" nanos = 0 secs = 300 +[filesystem_snapshots] +type = "Disabled" + +[filesystem_snapshots.config] + [filesystem_storage.cleanup_retry] max_attempts = 4 max_delay = "250ms" @@ -507,6 +512,11 @@ without_time = false # nanos = 0 # secs = 300 # +# [filesystem_snapshots] +# type = "Disabled" +# +# [filesystem_snapshots.config] +# # [filesystem_storage.cleanup_retry] # max_attempts = 4 # max_delay = "250ms" @@ -916,6 +926,11 @@ without_time = false # nanos = 0 # secs = 300 # +# [filesystem_snapshots] +# type = "Disabled" +# +# [filesystem_snapshots.config] +# # [filesystem_storage.cleanup_retry] # max_attempts = 4 # max_delay = "250ms" diff --git a/golem-worker-executor/src/services/golem_config.rs b/golem-worker-executor/src/services/golem_config.rs index 661194cf5c..0346bf94e2 100644 --- a/golem-worker-executor/src/services/golem_config.rs +++ b/golem-worker-executor/src/services/golem_config.rs @@ -89,6 +89,8 @@ pub struct GolemConfig { pub memory: MemoryConfig, pub filesystem_storage: FilesystemStorageConfig, #[serde(default)] + pub filesystem_snapshots: FilesystemSnapshotsConfig, + #[serde(default)] pub resource_usage_metering: ResourceUsageMeteringConfig, pub rdbms: RdbmsConfig, pub resource_limits: ResourceLimitsConfig, @@ -267,6 +269,12 @@ impl SafeDisplay for GolemConfig { "{}", self.filesystem_storage.to_safe_string_indented() ); + let _ = writeln!(&mut result, "filesystem snapshots:"); + let _ = writeln!( + &mut result, + "{}", + self.filesystem_snapshots.to_safe_string_indented() + ); let _ = writeln!(&mut result, "resource usage metering:"); let _ = writeln!( &mut result, @@ -388,6 +396,7 @@ impl Default for GolemConfig { public_worker_api: WorkerServiceGrpcConfig::default(), memory: MemoryConfig::default(), filesystem_storage: FilesystemStorageConfig::default(), + filesystem_snapshots: FilesystemSnapshotsConfig::default(), resource_usage_metering: ResourceUsageMeteringConfig::default(), rdbms: RdbmsConfig::default(), resource_limits: ResourceLimitsConfig::default(), @@ -2330,6 +2339,204 @@ impl SafeDisplay for FilesystemPressureConfig { } } +/// The default of [`FilesystemSnapshotStoreConfig::storage_call_deadline`]. +/// +/// On S3, with the retries of the S3 storage, a write of a pack took at most 1.7 s with eight saves +/// at the same time. A ranged read of a pack took at most 1.5 s under the CPU request of an +/// executor. Keep the value at least 10 times the longest measured call. +pub const DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE: Duration = Duration::from_secs(30); + +/// The default of [`FilesystemSnapshotStoreConfig::restore_reader_threads`]. +const DEFAULT_FILESYSTEM_SNAPSHOT_RESTORE_READER_THREADS: usize = 4; + +/// The default of [`FilesystemSnapshotStoreConfig::save_threads`]. +const DEFAULT_FILESYSTEM_SNAPSHOT_SAVE_THREADS: usize = 4; + +/// Tells whether the executor keeps filesystem snapshots, and gives the settings of the store. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(tag = "type", content = "config")] +pub enum FilesystemSnapshotsConfig { + Disabled(FilesystemSnapshotsDisabledConfig), + Managed(FilesystemSnapshotStoreConfig), +} + +impl Default for FilesystemSnapshotsConfig { + fn default() -> Self { + Self::Disabled(FilesystemSnapshotsDisabledConfig {}) + } +} + +impl SafeDisplay for FilesystemSnapshotsConfig { + fn to_safe_string(&self) -> String { + let mut result = String::new(); + match self { + Self::Disabled(_) => { + let _ = writeln!(&mut result, "disabled"); + } + Self::Managed(store) => { + let _ = writeln!(&mut result, "managed:"); + let _ = writeln!(&mut result, "{}", store.to_safe_string_indented()); + } + } + result + } +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct FilesystemSnapshotsDisabledConfig {} + +/// The settings of the store of filesystem snapshots. +#[derive(Clone, Debug, Serialize)] +pub struct FilesystemSnapshotStoreConfig { + /// The key that encrypts each repository, as 128 hex characters. The key is a secret. + repository_key: FilesystemSnapshotRepositoryKey, + /// The longest time that one blob storage call of the store waits for an answer. + #[serde(with = "humantime_serde")] + storage_call_deadline: Duration, + /// The number of threads that read packs in a restore. + restore_reader_threads: NonZeroUsize, + /// The number of threads of each parallel stage of a save. + save_threads: NonZeroUsize, +} + +#[derive(Deserialize)] +struct RawFilesystemSnapshotStoreConfig { + repository_key: String, + #[serde( + with = "humantime_serde", + default = "default_filesystem_snapshot_storage_call_deadline" + )] + storage_call_deadline: Duration, + #[serde(default = "default_filesystem_snapshot_restore_reader_threads")] + restore_reader_threads: usize, + #[serde(default = "default_filesystem_snapshot_save_threads")] + save_threads: usize, +} + +fn default_filesystem_snapshot_storage_call_deadline() -> Duration { + DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE +} + +fn default_filesystem_snapshot_restore_reader_threads() -> usize { + DEFAULT_FILESYSTEM_SNAPSHOT_RESTORE_READER_THREADS +} + +fn default_filesystem_snapshot_save_threads() -> usize { + DEFAULT_FILESYSTEM_SNAPSHOT_SAVE_THREADS +} + +impl FilesystemSnapshotStoreConfig { + pub fn new( + repository_key: &str, + storage_call_deadline: Duration, + restore_reader_threads: usize, + save_threads: usize, + ) -> Result { + let repository_key = FilesystemSnapshotRepositoryKey::parse(repository_key)?; + if storage_call_deadline.is_zero() { + return Err("storage_call_deadline must be greater than zero".to_string()); + } + let restore_reader_threads = NonZeroUsize::new(restore_reader_threads) + .ok_or_else(|| "restore_reader_threads must be greater than zero".to_string())?; + let save_threads = NonZeroUsize::new(save_threads) + .ok_or_else(|| "save_threads must be greater than zero".to_string())?; + Ok(Self { + repository_key, + storage_call_deadline, + restore_reader_threads, + save_threads, + }) + } + + pub fn repository_key(&self) -> &FilesystemSnapshotRepositoryKey { + &self.repository_key + } + + pub const fn storage_call_deadline(&self) -> Duration { + self.storage_call_deadline + } + + pub const fn restore_reader_threads(&self) -> NonZeroUsize { + self.restore_reader_threads + } + + pub const fn save_threads(&self) -> NonZeroUsize { + self.save_threads + } +} + +impl<'de> Deserialize<'de> for FilesystemSnapshotStoreConfig { + fn deserialize(deserializer: D) -> Result + where + D: Deserializer<'de>, + { + let raw = RawFilesystemSnapshotStoreConfig::deserialize(deserializer)?; + Self::new( + &raw.repository_key, + raw.storage_call_deadline, + raw.restore_reader_threads, + raw.save_threads, + ) + .map_err(D::Error::custom) + } +} + +impl SafeDisplay for FilesystemSnapshotStoreConfig { + fn to_safe_string(&self) -> String { + let mut result = String::new(); + let _ = writeln!(&mut result, "repository key: ****"); + let _ = writeln!( + &mut result, + "storage call deadline: {:?}", + self.storage_call_deadline + ); + let _ = writeln!( + &mut result, + "restore reader threads: {}", + self.restore_reader_threads + ); + let _ = writeln!(&mut result, "save threads: {}", self.save_threads); + result + } +} + +/// The key that encrypts the repositories of filesystem snapshots. It has 64 bytes. +#[derive(Clone, PartialEq, Eq)] +pub struct FilesystemSnapshotRepositoryKey([u8; 64]); + +impl FilesystemSnapshotRepositoryKey { + /// Gives the key from its 128 hex characters. + pub fn parse(text: &str) -> Result { + if text.is_empty() { + return Err("repository_key must not be empty".to_string()); + } + hex::decode(text) + .ok() + .and_then(|bytes| <[u8; 64]>::try_from(bytes).ok()) + .map(Self) + .ok_or_else(|| "repository_key must be 128 hex characters".to_string()) + } + + pub const fn bytes(&self) -> &[u8; 64] { + &self.0 + } +} + +impl std::fmt::Debug for FilesystemSnapshotRepositoryKey { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + formatter.write_str("FilesystemSnapshotRepositoryKey(****)") + } +} + +impl Serialize for FilesystemSnapshotRepositoryKey { + fn serialize(&self, serializer: S) -> Result + where + S: serde::Serializer, + { + serializer.serialize_str(&hex::encode(self.0)) + } +} + #[derive(Clone, Debug, Serialize)] pub struct FilesystemObjectLimitPolicyConfig { /// Number of filesystem objects granted per GiB of allocated storage. @@ -2663,9 +2870,13 @@ pub fn make_config_loader() -> ConfigLoader { #[cfg(test)] mod tests { - use super::{DurableStreamConfig, InvocationResultsConfig, Limits}; + use super::{ + DurableStreamConfig, FilesystemSnapshotStoreConfig, FilesystemSnapshotsConfig, GolemConfig, + InvocationResultsConfig, Limits, + }; use golem_common::SafeDisplay; - use serde_json::Value; + use serde_json::{Value, json}; + use std::time::Duration; use test_r::test; #[test] @@ -2741,4 +2952,136 @@ mod tests { assert!(serde_json::from_value::(serialized).is_err()); } + const KEY: &str = "000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f\ + 202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f"; + + fn managed(config: Value) -> Result { + serde_json::from_value::(json!({ + "type": "Managed", + "config": config, + })) + .map_err(|error| error.to_string()) + .and_then(|parsed| match parsed { + FilesystemSnapshotsConfig::Managed(store) => Ok(store), + FilesystemSnapshotsConfig::Disabled(_) => Err("disabled".to_string()), + }) + } + + fn refusal(config: Value) -> String { + managed(config).unwrap_err() + } + + #[test] + fn filesystem_snapshots_are_disabled_by_default() { + let config = GolemConfig::default().filesystem_snapshots; + + assert!( + matches!(config, FilesystemSnapshotsConfig::Disabled(_)), + "{config:?}" + ); + } + + #[test] + fn filesystem_snapshots_managed_config_reads_the_key_the_deadline_and_the_thread_counts() { + let store = managed(json!({ + "repository_key": KEY, + "storage_call_deadline": "45s", + "restore_reader_threads": 7, + "save_threads": 3, + })) + .unwrap(); + + assert_eq!( + ( + store.repository_key().bytes().to_vec(), + store.storage_call_deadline(), + store.restore_reader_threads().get(), + store.save_threads().get(), + ), + ((0..64).collect::>(), Duration::from_secs(45), 7, 3) + ); + } + + #[test] + fn filesystem_snapshots_managed_config_gives_the_defaults_of_the_fields_it_does_not_set() { + let store = managed(json!({ "repository_key": KEY })).unwrap(); + + assert_eq!( + ( + store.storage_call_deadline(), + store.restore_reader_threads().get(), + store.save_threads().get(), + ), + (Duration::from_secs(30), 4, 4) + ); + } + + #[test] + fn filesystem_snapshots_managed_config_refuses_a_zero_deadline() { + assert!( + refusal(json!({ "repository_key": KEY, "storage_call_deadline": "0s" })) + .contains("storage_call_deadline must be greater than zero") + ); + } + + #[test] + fn filesystem_snapshots_managed_config_refuses_zero_restore_reader_threads() { + assert!( + refusal(json!({ "repository_key": KEY, "restore_reader_threads": 0 })) + .contains("restore_reader_threads must be greater than zero") + ); + } + + #[test] + fn filesystem_snapshots_managed_config_refuses_zero_save_threads() { + assert!( + refusal(json!({ "repository_key": KEY, "save_threads": 0 })) + .contains("save_threads must be greater than zero") + ); + } + + #[test] + fn filesystem_snapshots_managed_config_refuses_an_empty_short_or_non_hex_key() { + let non_hex = format!("{}g", &KEY[..127]); + let cases = [ + (json!({ "repository_key": "" }), "must not be empty"), + ( + json!({ "repository_key": &KEY[..126] }), + "must be 128 hex characters", + ), + ( + json!({ "repository_key": format!("{KEY}00") }), + "must be 128 hex characters", + ), + ( + json!({ "repository_key": non_hex }), + "must be 128 hex characters", + ), + (json!({}), "missing field `repository_key`"), + ]; + + let unexpected = cases + .into_iter() + .map(|(config, reason)| (refusal(config), reason)) + .filter(|(error, reason)| !error.contains(reason)) + .collect::>(); + + assert_eq!(unexpected, Vec::<(String, &str)>::new()); + } + + #[test] + fn filesystem_snapshots_managed_config_hides_the_key() { + let store = managed(json!({ "repository_key": KEY })).unwrap(); + let shown = FilesystemSnapshotsConfig::Managed(store.clone()).to_safe_string(); + let debugged = format!("{store:?}"); + + assert_eq!( + ( + shown.contains("0001020304"), + debugged.contains("0001020304"), + shown.contains("repository key: ****"), + ), + (false, false, true) + ); + } } From e6cf655fb729d81ee35bfcf7039643dffb701b2e Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:10:14 -0700 Subject: [PATCH 002/126] Give the executor image the POSIX time zone UTC0 --- golem-worker-executor/docker/Dockerfile | 4 +++ .../src/filesystem_snapshot/mod.rs | 2 ++ .../filesystem_snapshot/time_zone_tests.rs | 33 +++++++++++++++++++ 3 files changed, 39 insertions(+) create mode 100644 golem-worker-executor/src/filesystem_snapshot/time_zone_tests.rs diff --git a/golem-worker-executor/docker/Dockerfile b/golem-worker-executor/docker/Dockerfile index 11fe1b2e8f..61eace6855 100644 --- a/golem-worker-executor/docker/Dockerfile +++ b/golem-worker-executor/docker/Dockerfile @@ -16,6 +16,10 @@ LABEL cloud.golem.cpu.features="neon,crc,lse,aes,sha2,dotprod" FROM platform-${TARGETARCH} AS final +# The image has no time zone database. jiff reads this POSIX rule without one, so the snapshot +# store writes no time zone warning for each timestamp. +ENV TZ=UTC0 + WORKDIR /app COPY /target/$RUST_TARGET/release/worker-executor ./ COPY /golem-worker-executor/config/worker-executor.toml ./config/worker-executor.toml diff --git a/golem-worker-executor/src/filesystem_snapshot/mod.rs b/golem-worker-executor/src/filesystem_snapshot/mod.rs index f9a6d904f3..9f1abd7681 100644 --- a/golem-worker-executor/src/filesystem_snapshot/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/mod.rs @@ -31,6 +31,8 @@ pub(crate) mod benchmark; mod contract_tests; mod memory; mod rustic; +#[cfg(test)] +mod time_zone_tests; #[allow(unused_imports)] pub(crate) use memory::InMemorySnapshotStore; diff --git a/golem-worker-executor/src/filesystem_snapshot/time_zone_tests.rs b/golem-worker-executor/src/filesystem_snapshot/time_zone_tests.rs new file mode 100644 index 0000000000..76f13f62fc --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/time_zone_tests.rs @@ -0,0 +1,33 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The image of the executor must give rustic a time zone. rustic asks for the time zone of the +//! system for each timestamp that a save writes, and it logs a warning when it finds none. + +use test_r::test; + +const DOCKERFILE: &str = include_str!("../../docker/Dockerfile"); + +#[test] +fn the_executor_image_sets_a_posix_utc_time_zone() { + let final_stage = DOCKERFILE + .rsplit_once("\nFROM ") + .map(|(_, stage)| stage) + .unwrap_or_default(); + + assert!( + final_stage.lines().any(|line| line.trim() == "ENV TZ=UTC0"), + "the final stage of the executor image does not set `ENV TZ=UTC0`:\n{final_stage}" + ); +} From 7845f9b7108cf13df80f4453055f904ff8521fc2 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:10:24 -0700 Subject: [PATCH 003/126] Classify a failed rustic operation as a snapshot store error --- .../src/filesystem_snapshot/rustic/fault.rs | 298 ++++++++++++++++++ .../src/filesystem_snapshot/rustic/mod.rs | 2 + 2 files changed, 300 insertions(+) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs new file mode 100644 index 0000000000..95a73d4718 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs @@ -0,0 +1,298 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The errors that the backend puts into the chain of a rustic error, and the classification of a +//! failed operation as a [`SnapshotStoreError`]. +//! +//! rustic gives no public kind of an error. So the classification reads the chain of sources: the +//! markers of this module, the name errors of the blob storage, and the I/O errors. + +use crate::filesystem_snapshot::SnapshotStoreError; +use golem_service_base::storage::blob::BlobNameError; +use std::error::Error; +use std::fmt::{Display, Formatter}; + +/// A blob storage call of the backend that failed, got no answer within its deadline, or did not +/// start because its operation was cancelled. The source is the failure. +#[derive(Debug)] +pub(super) struct BlobCallFailed { + failure: anyhow::Error, +} + +impl BlobCallFailed { + pub(super) fn new(failure: anyhow::Error) -> Self { + Self { failure } + } +} + +impl Display for BlobCallFailed { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("the blob storage call gave an error") + } +} + +impl Error for BlobCallFailed { + fn source(&self) -> Option<&(dyn Error + 'static)> { + Some(self.failure.as_ref()) + } +} + +/// The operation of the backend was cancelled, so the backend made no more calls. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct OperationCancelled; + +impl Display for OperationCancelled { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("the filesystem snapshot operation was cancelled") + } +} + +impl Error for OperationCancelled {} + +/// Another writer made the config file of the repository first. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct ConfigExists; + +impl Display for ConfigExists { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("another writer made the config file of the repository first") + } +} + +impl Error for ConfigExists {} + +/// Tells whether an error in the chain is [`ConfigExists`]. +pub(super) fn is_config_exists(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| error.is::()) +} + +/// The kind of operation that failed. It tells where an I/O error came from. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum Operation { + /// Reads a tree from the local filesystem and writes it into the repository. + Save, + /// Reads the repository and writes a tree into the local filesystem. + Restore, + /// Reads or changes only the repository. + Repository, +} + +/// Gives the error of the store for an operation that failed with the error. +/// +/// - A failed blob storage call gives `Storage`. It is retryable unless a name error of the blob +/// storage caused it, because a name error is the same for each new try. +/// - An I/O error gives `Source` in a save and `Destination` in a restore, with the kind of that +/// I/O error. +/// - Each other error gives `Storage` that is not retryable in a save, and `Corrupt` in the other +/// operations, because the repository gave data that rustic refused. +pub(super) fn classify(operation: Operation, error: anyhow::Error) -> SnapshotStoreError { + let from_storage = chain(error.as_ref()).any(|error| error.is::()); + let io_kind = chain(error.as_ref()) + .find_map(|error| error.downcast_ref::()) + .map(std::io::Error::kind); + let permanent = chain(error.as_ref()).any(|error| error.is::()); + match (from_storage, io_kind, operation) { + (true, _, _) => SnapshotStoreError::Storage { + retryable: !permanent, + source: error, + }, + (false, Some(kind), Operation::Save) => { + SnapshotStoreError::Source(std::io::Error::new(kind, error_text(&error))) + } + (false, Some(kind), Operation::Restore) => { + SnapshotStoreError::Destination(std::io::Error::new(kind, error_text(&error))) + } + (false, _, Operation::Save) => SnapshotStoreError::Storage { + retryable: false, + source: error, + }, + (false, _, Operation::Restore | Operation::Repository) => { + SnapshotStoreError::Corrupt(error) + } + } +} + +/// Gives the text of the error with the text of each of its sources. +fn error_text(error: &anyhow::Error) -> String { + format!("{error:#}") +} + +/// Gives the error and each error in its chain of sources. +fn chain<'a>(error: &'a (dyn Error + 'static)) -> impl Iterator { + std::iter::successors(Some(error), |&error| error.source()) +} + +#[cfg(test)] +mod tests { + use super::{ + BlobCallFailed, ConfigExists, Operation, OperationCancelled, classify, is_config_exists, + }; + use crate::filesystem_snapshot::SnapshotStoreError; + use golem_service_base::storage::blob::BlobNameError; + use pretty_assertions::assert_eq; + use rustic_core::{ErrorKind, RusticError}; + use std::io; + use test_r::test; + + /// Gives a rustic error whose source is the error, as rustic gives it to the store. + fn rustic(source: impl std::error::Error + Send + Sync + 'static) -> anyhow::Error { + anyhow::Error::new(RusticError::with_source( + ErrorKind::Backend, + "the operation failed", + source, + )) + } + + fn storage_failure(failure: anyhow::Error) -> anyhow::Error { + rustic(BlobCallFailed::new(failure)) + } + + /// Gives the variant of the error, whether it is retryable, and the kind of its I/O error. + fn shape(error: &SnapshotStoreError) -> (&'static str, Option, Option) { + match error { + SnapshotStoreError::NotFound => ("NotFound", None, None), + SnapshotStoreError::AlreadyExists => ("AlreadyExists", None, None), + SnapshotStoreError::Source(error) => ("Source", None, Some(error.kind())), + SnapshotStoreError::Destination(error) => ("Destination", None, Some(error.kind())), + SnapshotStoreError::Storage { retryable, .. } => ("Storage", Some(*retryable), None), + SnapshotStoreError::Corrupt(_) => ("Corrupt", None, None), + } + } + + #[test] + fn a_failed_blob_storage_call_gives_a_retryable_storage_error_in_each_operation() { + let shapes = + [Operation::Save, Operation::Restore, Operation::Repository].map(|operation| { + shape(&classify( + operation, + storage_failure(anyhow::anyhow!("the bucket is gone")), + )) + }); + + assert_eq!(shapes, [("Storage", Some(true), None); 3]); + } + + #[test] + fn a_blob_storage_call_that_holds_an_io_error_is_still_a_storage_error() { + let failure = anyhow::Error::new(io::Error::new(io::ErrorKind::StorageFull, "no space")); + + assert_eq!( + shape(&classify(Operation::Restore, storage_failure(failure))), + ("Storage", Some(true), None) + ); + } + + #[test] + fn a_name_error_of_the_blob_storage_is_not_retryable() { + let failure = anyhow::Error::new(BlobNameError::NoName { + path: std::path::PathBuf::new(), + }); + + assert_eq!( + shape(&classify(Operation::Repository, storage_failure(failure))), + ("Storage", Some(false), None) + ); + } + + #[test] + fn a_cancelled_operation_gives_a_retryable_storage_error_in_each_operation() { + let shapes = + [Operation::Save, Operation::Restore, Operation::Repository].map(|operation| { + shape(&classify( + operation, + storage_failure(anyhow::Error::new(OperationCancelled)), + )) + }); + + assert_eq!(shapes, [("Storage", Some(true), None); 3]); + } + + #[test] + fn an_io_error_of_a_save_gives_source_with_its_kind() { + let error = rustic(io::Error::new(io::ErrorKind::PermissionDenied, "locked")); + + let classified = classify(Operation::Save, error); + + assert_eq!( + ( + shape(&classified), + classified.to_string().contains("locked") + ), + ( + ("Source", None, Some(io::ErrorKind::PermissionDenied)), + true + ) + ); + } + + #[test] + fn a_full_volume_during_a_restore_gives_destination() { + let error = rustic(io::Error::new(io::ErrorKind::StorageFull, "no space")); + + assert_eq!( + shape(&classify(Operation::Restore, error)), + ("Destination", None, Some(io::ErrorKind::StorageFull)) + ); + } + + #[test] + fn a_metadata_error_of_a_restore_gives_destination() { + let error = rustic(io::Error::new( + io::ErrorKind::PermissionDenied, + "setting extended attributes failed", + )); + + assert_eq!( + shape(&classify(Operation::Restore, error)), + ("Destination", None, Some(io::ErrorKind::PermissionDenied)) + ); + } + + #[test] + fn an_error_without_storage_or_io_is_corrupt_when_it_reads_and_not_retryable_in_a_save() { + let refused = || { + anyhow::Error::new(RusticError::new( + ErrorKind::Cryptography, + "the data failed its check", + )) + }; + + assert_eq!( + [ + shape(&classify(Operation::Restore, refused())), + shape(&classify(Operation::Repository, refused())), + shape(&classify(Operation::Save, refused())), + ], + [ + ("Corrupt", None, None), + ("Corrupt", None, None), + ("Storage", Some(false), None), + ] + ); + } + + #[test] + fn the_config_marker_is_found_in_the_chain() { + let exists = storage_failure(anyhow::Error::new(ConfigExists)); + let other = storage_failure(anyhow::anyhow!("the bucket is gone")); + + assert_eq!( + ( + is_config_exists(exists.as_ref()), + is_config_exists(other.as_ref()) + ), + (true, false) + ); + } +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index bc9b4a00f6..9b30b446f8 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -19,6 +19,8 @@ //! module. mod backend; +#[cfg_attr(not(test), allow(dead_code))] +mod fault; #[cfg(test)] mod holding; From 13f0854f742e561c9bb0aea94ef4f5d6cc599515 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:10:24 -0700 Subject: [PATCH 004/126] Cancel, stage and write conditionally in the rustic backend, and publish a staged snapshot file --- .../src/filesystem_snapshot/rustic/backend.rs | 126 +++++-- .../rustic/backend/tests.rs | 235 ++++++++++++- .../src/filesystem_snapshot/rustic/mod.rs | 4 + .../src/filesystem_snapshot/rustic/publish.rs | 166 +++++++++ .../rustic/publish/tests.rs | 218 ++++++++++++ .../filesystem_snapshot/rustic/scripted.rs | 319 ++++++++++++++++++ 6 files changed, 1039 insertions(+), 29 deletions(-) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index 856eb0dcee..8d99fc90f5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -17,10 +17,12 @@ //! The backend keeps the files of one repository in one blob storage namespace, with the paths of //! the restic repository format. rustic calls the backend from threads outside the async runtime, //! and each call waits for the blob storage on the runtime that the backend holds. Each call waits -//! for at most a deadline. +//! for at most a deadline, and a cancelled operation makes no more calls. +use super::fault::{BlobCallFailed, ConfigExists, OperationCancelled}; +use super::publish::{SnapshotStage, StagedSnapshot}; use bytes::Bytes; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace, PutIfAbsent}; use rustic_core::{ BytesList, ErrorKind, FileType, Id, ReadBackend, RusticError, RusticResult, WriteBackend, }; @@ -29,6 +31,7 @@ use std::path::{Path, PathBuf}; use std::sync::Arc; use std::time::Duration; use tokio::runtime::Handle; +use tokio_util::sync::CancellationToken; /// The target label of each blob storage call of the backend. const TARGET_LABEL: &str = "filesystem_snapshot"; @@ -77,12 +80,17 @@ impl StorageCall { /// A call that gets no answer from the blob storage within the deadline gives an error. The /// runtime must be a multi-thread runtime, because on a `current_thread` runtime `Handle::block_on` /// does not drive the timer of the deadline. +/// +/// The config file and the index files are written only when their path has no blob. A snapshot +/// file goes into the stage of the backend when it has one, and the backend does not write it. #[derive(Debug)] pub(super) struct BlobBackend { storage: Arc, namespace: BlobStorageNamespace, runtime: Handle, deadline: Duration, + cancel: CancellationToken, + stage: Option>, } impl BlobBackend { @@ -99,23 +107,70 @@ impl BlobBackend { namespace, runtime, deadline, + cancel: CancellationToken::new(), + stage: None, + } + } + + /// Gives the backend with the token of its operation. When the token is cancelled, a call that + /// has not started gives an error at once, and a call that runs stops and gives an error. + #[cfg_attr(not(test), allow(dead_code))] + pub(super) fn cancelled_by(self, cancel: CancellationToken) -> Self { + Self { cancel, ..self } + } + + /// Gives the backend with a stage for the snapshot file of a save. + #[cfg_attr(not(test), allow(dead_code))] + pub(super) fn staging_in(self, stage: Arc) -> Self { + Self { + stage: Some(stage), + ..self } } /// Waits for one call on the blob storage, and gives its result as a rustic result. /// /// Each call of the backend on the blob storage goes through this function. A call that gives - /// no answer within the deadline gives an error, the same as a call that failed. + /// no answer within the deadline gives an error, the same as a call that failed. So does a + /// call of a cancelled operation. fn request( &self, call: StorageCall, path: &Path, future: impl Future>, ) -> RusticResult { + if self.cancel.is_cancelled() { + return Err(storage_error( + call, + path, + anyhow::Error::new(OperationCancelled), + )); + } self.runtime - .block_on(answer_within(self.deadline, future)) + .block_on(async { + tokio::select! { + biased; + answer = answer_within(self.deadline, future) => answer, + () = self.cancel.cancelled() => Err(anyhow::Error::new(OperationCancelled)), + } + }) .map_err(|error| storage_error(call, path, error)) } + + /// Writes the content at the path only when the path has no blob, and gives whether it wrote. + fn write_if_absent(&self, path: &Path, content: &[u8]) -> RusticResult { + self.request( + StorageCall::Write, + path, + self.storage.put_raw_if_absent( + TARGET_LABEL, + StorageCall::Write.label(), + self.namespace.clone(), + path, + content, + ), + ) + } } /// Gives the output of the future, or an error when the future gives no output within the deadline. @@ -124,7 +179,7 @@ impl BlobBackend { /// thread without a runtime context can wait for the result with `Handle::block_on`. A future that /// is ready at its first poll always gives its output. At the deadline, the function drops the /// future and gives an error whose root cause is tokio's `Elapsed`. -async fn answer_within( +pub(super) async fn answer_within( deadline: Duration, future: impl Future>, ) -> anyhow::Result { @@ -248,25 +303,33 @@ impl WriteBackend for BlobBackend { ) -> RusticResult<()> { let path = file_path(tpe, id)?; let parts = content.into_vec(); - let joined; - let data: &[u8] = match parts.as_slice() { - [part] => part, - parts => { - joined = join(parts); - &joined - } + let content = match parts.as_slice() { + [part] => part.clone(), + parts => Bytes::from(join(parts)), }; - self.request( - StorageCall::Write, - &path, - self.storage.put_raw( - TARGET_LABEL, - StorageCall::Write.label(), - self.namespace.clone(), + match (tpe, &self.stage) { + (FileType::Snapshot, Some(stage)) => stage + .keep(StagedSnapshot { path, content }) + .map_err(|staged| second_snapshot(&staged.path)), + (FileType::Config, _) => match self.write_if_absent(&path, &content)? { + PutIfAbsent::Written => Ok(()), + PutIfAbsent::AlreadyExists => Err(config_exists(&path)), + }, + // The name of an index file is the hash of its content, so a blob at the path holds + // the same content. + (FileType::Index, _) => self.write_if_absent(&path, &content).map(|_| ()), + _ => self.request( + StorageCall::Write, &path, - data, + self.storage.put_raw( + TARGET_LABEL, + StorageCall::Write.label(), + self.namespace.clone(), + &path, + &content, + ), ), - ) + } } fn remove(&self, tpe: FileType, id: &Id, _cacheable: bool) -> RusticResult<()> { @@ -342,12 +405,31 @@ fn missing_file(path: &Path) -> Box { .attach_context("path", path.display().to_string()) } +/// The error of a config file that another writer made first. +fn config_exists(path: &Path) -> Box { + RusticError::with_source( + ErrorKind::Backend, + "The blob storage already holds the config file `{path}`.", + ConfigExists, + ) + .attach_context("path", path.display().to_string()) +} + +/// The error of a second snapshot file in one stage. +fn second_snapshot(path: &Path) -> Box { + RusticError::new( + ErrorKind::Internal, + "The stage already holds a snapshot file, so it cannot keep `{path}`.", + ) + .attach_context("path", path.display().to_string()) +} + /// The error of a blob storage call that failed. fn storage_error(call: StorageCall, path: &Path, error: anyhow::Error) -> Box { RusticError::with_source( ErrorKind::Backend, "The blob storage call `{call}` failed at `{path}`.", - error, + BlobCallFailed::new(error), ) .attach_context("call", call.label()) .attach_context("path", path.display().to_string()) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index 95877bd875..66fa7afa3d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -18,8 +18,12 @@ //! runtime that the backend holds, as the threads of rustic are not. use super::super::STORAGE_CALL_DEADLINE; +use super::super::fault::{Operation, OperationCancelled, classify, is_config_exists}; use super::super::holding::{holding_storage, reached_deadline}; +use super::super::publish::{SnapshotStage, StagedSnapshot}; +use super::super::scripted::{Script, ScriptedBlobStorage}; use super::{BlobBackend, file_size}; +use crate::filesystem_snapshot::SnapshotStoreError; use anyhow::anyhow; use async_trait::async_trait; use bytes::Bytes; @@ -37,6 +41,7 @@ use std::sync::Arc; use std::time::{Duration, Instant}; use test_r::test; use tokio::runtime::Runtime; +use tokio_util::sync::CancellationToken; use uuid::Uuid; /// The longest time that a test waits for the calls on the backend. @@ -473,6 +478,218 @@ fn a_thread_that_is_not_a_thread_of_the_runtime_can_call_the_backend() { assert_eq!(read.ok().flatten(), Some(Bytes::from_static(b"index"))); } +#[test] +fn a_cancelled_backend_makes_no_storage_call() { + let runtime = Runtime::new().unwrap(); + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let cancel = CancellationToken::new(); + let backend = BlobBackend::new( + storage.clone(), + new_namespace(), + runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .cancelled_by(cancel.clone()); + cancel.cancel(); + + let cancelled = [ + backend.list_with_size(FileType::Config).err(), + backend.list_with_size(FileType::Pack).err(), + backend.read_full(FileType::Pack, &id("ab")).err(), + backend + .read_partial(FileType::Pack, &id("ab"), false, 0, 1) + .err(), + backend + .write_bytes(FileType::Pack, &id("ab"), false, bytes("pack")) + .err(), + backend.remove(FileType::Pack, &id("ab"), false).err(), + ] + .map(|error| error.is_some_and(|error| was_cancelled(&error))); + + assert_eq!((cancelled, storage.calls()), ([true; 6], Vec::new())); +} + +#[test] +fn a_cancel_ends_a_call_that_runs() { + let runtime = Runtime::new().unwrap(); + let (storage, _gate, _dropped) = + holding_storage(Arc::new(InMemoryBlobStorage::new()), |_, _| true); + let cancel = CancellationToken::new(); + let backend = BlobBackend::new( + storage, + new_namespace(), + runtime.handle().clone(), + Duration::from_secs(60), + ) + .cancelled_by(cancel.clone()); + runtime.spawn(async move { + tokio::time::sleep(Duration::from_millis(100)).await; + cancel.cancel(); + }); + + let outcome = within_limit(move || { + backend + .read_full(FileType::Pack, &id("ab")) + .err() + .map(|error| (was_cancelled(&error), reached_deadline(&*error))) + }); + + assert_eq!(outcome, Some(Some((true, false)))); +} + +#[test] +fn a_backend_whose_token_is_not_cancelled_answers() { + let fixture = Fixture::new(); + let backend = BlobBackend::new( + fixture.storage.clone(), + fixture.namespace.clone(), + fixture.runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .cancelled_by(CancellationToken::new()); + + let read = backend + .write_bytes(FileType::Pack, &id("ab"), false, bytes("pack")) + .and_then(|()| backend.read_full(FileType::Pack, &id("ab"))); + + assert_eq!(read.ok(), Some(Bytes::from_static(b"pack"))); +} + +#[test] +fn the_config_file_is_written_only_when_the_repository_has_none() { + let fixture = Fixture::new(); + + let first = + fixture + .backend + .write_bytes(FileType::Config, &Id::default(), false, bytes("first")); + let second = + fixture + .backend + .write_bytes(FileType::Config, &Id::default(), false, bytes("second")); + + assert_eq!( + ( + first.is_ok(), + second + .as_ref() + .is_err_and(|error| is_config_exists(&**error)), + fixture.stored(), + ), + (true, true, vec![("config".to_string(), 5)]) + ); +} + +#[test] +fn an_index_file_that_is_there_is_kept_and_its_write_succeeds() { + let fixture = Fixture::new(); + let path = format!("index/{}", "ab".repeat(32)); + fixture.put(&path, b"kept"); + + let written = + fixture + .backend + .write_bytes(FileType::Index, &id("ab"), false, bytes("replacement")); + let read = fixture.backend.read_full(FileType::Index, &id("ab")); + + assert_eq!( + (written.is_ok(), read.ok()), + (true, Some(Bytes::from_static(b"kept"))) + ); +} + +#[test] +fn a_pack_file_that_is_there_is_written_again() { + let fixture = Fixture::new(); + fixture.put(&format!("data/ab/{}", "ab".repeat(32)), b"old"); + + let written = fixture + .backend + .write_bytes(FileType::Pack, &id("ab"), false, bytes("new")); + let read = fixture.backend.read_full(FileType::Pack, &id("ab")); + + assert_eq!( + (written.is_ok(), read.ok()), + (true, Some(Bytes::from_static(b"new"))) + ); +} + +#[test] +fn a_backend_with_a_stage_keeps_the_snapshot_file_and_does_not_write_it() { + let fixture = Fixture::new(); + let stage = Arc::new(SnapshotStage::default()); + let backend = BlobBackend::new( + fixture.storage.clone(), + fixture.namespace.clone(), + fixture.runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .staging_in(stage.clone()); + let content = [Bytes::from_static(b"snap"), Bytes::from_static(b"shot")] + .into_iter() + .fold(BytesList::default(), |mut content, part| { + content.add(part); + content + }); + + let kept = backend.write_bytes(FileType::Snapshot, &id("cd"), false, content); + let second = backend.write_bytes(FileType::Snapshot, &id("ef"), false, bytes("other")); + let pack = backend.write_bytes(FileType::Pack, &id("ab"), false, bytes("pack")); + + assert_eq!( + ( + kept.is_ok(), + second.is_err(), + pack.is_ok(), + stage.take(), + fixture.stored(), + ), + ( + true, + true, + true, + Some(StagedSnapshot { + path: PathBuf::from(format!("snapshots/{}", "cd".repeat(32))).into_boxed_path(), + content: Bytes::from_static(b"snapshot"), + }), + vec![(format!("data/ab/{}", "ab".repeat(32)), 4)] + ) + ); +} + +#[test] +fn each_failed_call_is_a_storage_failure_to_the_classification() { + let runtime = Runtime::new().unwrap(); + let backend = BlobBackend::new( + Arc::new(FailingBlobStorage), + new_namespace(), + runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ); + + let classified = [ + backend.list_with_size(FileType::Pack).err(), + backend.read_full(FileType::Pack, &id("ab")).err(), + backend + .write_bytes(FileType::Pack, &id("ab"), false, bytes("pack")) + .err(), + ] + .map(|error| { + error.map(|error| { + matches!( + classify(Operation::Restore, anyhow::Error::new(error)), + SnapshotStoreError::Storage { + retryable: true, + .. + } + ) + }) + }); + + assert_eq!(classified, [Some(true); 3]); +} + #[test] fn a_size_that_does_not_fit_in_32_bits_is_an_error() { let path = Path::new("data/ab/file"); @@ -494,14 +711,18 @@ fn error_text(result: RusticResult) -> String { } } -/// Gives the text of the error, with the text of its source. +/// Gives the text of the error, with the text of each error in its chain of sources. fn text_of(error: &RusticError) -> String { - format!( - "{error} {}", - std::error::Error::source(error) - .map(ToString::to_string) - .unwrap_or_default() - ) + std::iter::successors(std::error::Error::source(error), |error| error.source()) + .fold(error.to_string(), |text, source| format!("{text} {source}")) +} + +/// Tells whether the error or an error in its chain of sources is [`OperationCancelled`]. +fn was_cancelled(error: &RusticError) -> bool { + std::iter::successors(Some(error as &(dyn std::error::Error + 'static)), |error| { + error.source() + }) + .any(|error| error.is::()) } /// A blob storage that fails every call. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 9b30b446f8..1e877f390d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -21,10 +21,14 @@ mod backend; #[cfg_attr(not(test), allow(dead_code))] mod fault; +#[cfg_attr(not(test), allow(dead_code))] +mod publish; #[cfg(test)] mod holding; #[cfg(test)] +mod scripted; +#[cfg(test)] mod tests; use super::{SnapshotName, SnapshotScope}; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs new file mode 100644 index 0000000000..e8d48455ca --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs @@ -0,0 +1,166 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The publish of a snapshot file. +//! +//! In a save of the store, the backend keeps the snapshot file in a [`SnapshotStage`] and does not +//! write it. The save writes it later with [`publish`], after the blocking work returns. That +//! write is the step that makes the snapshot visible. A publish that fails, or that the caller +//! drops, deletes the file again, because a write that the storage received can still complete. + +use super::backend::answer_within; +use bytes::Bytes; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use std::path::Path; +use std::sync::{Arc, Mutex, PoisonError}; +use std::time::Duration; +use tokio::runtime::Handle; +use tokio_util::task::TaskTracker; +use tracing::warn; + +/// The target label of each blob storage call of a publish. +const TARGET_LABEL: &str = "filesystem_snapshot"; + +/// A snapshot file that the backend kept and did not write. +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct StagedSnapshot { + /// The path of the file, relative to the root of the namespace. + pub(super) path: Box, + pub(super) content: Bytes, +} + +/// The place where the backend of one save keeps its snapshot file. +#[derive(Debug, Default)] +pub(super) struct SnapshotStage(Mutex>); + +impl SnapshotStage { + /// Keeps the file. A stage holds one file, so a second file gives it back as the error. + pub(super) fn keep(&self, staged: StagedSnapshot) -> Result<(), StagedSnapshot> { + let mut slot = self.0.lock().unwrap_or_else(PoisonError::into_inner); + match *slot { + Some(_) => Err(staged), + None => { + *slot = Some(staged); + Ok(()) + } + } + } + + /// Takes the file out of the stage. + pub(super) fn take(&self) -> Option { + self.0.lock().unwrap_or_else(PoisonError::into_inner).take() + } +} + +/// The snapshot files of one scope: the storage, the namespace of the scope, and the deadline of +/// each call. +#[derive(Clone, Debug)] +pub(super) struct SnapshotFiles { + pub(super) storage: Arc, + pub(super) namespace: BlobStorageNamespace, + pub(super) deadline: Duration, +} + +/// Writes the staged file, and so makes the snapshot visible. +/// +/// The file is written only when the path has no blob. The name of a snapshot file is the hash of +/// its content, so a blob at the path already holds this content, and the call succeeds. +/// +/// When the write fails, the call deletes the path before it gives the error. When the caller +/// drops the call during the write, a task of `tracker` deletes the path. +pub(super) async fn publish( + files: &SnapshotFiles, + staged: &StagedSnapshot, + tracker: &TaskTracker, +) -> anyhow::Result<()> { + let mut retraction = RetractOnDrop { + files: files.clone(), + path: staged.path.clone(), + tracker: tracker.clone(), + armed: true, + }; + let written = answer_within( + files.deadline, + files.storage.put_raw_if_absent( + TARGET_LABEL, + "publish", + files.namespace.clone(), + &staged.path, + &staged.content, + ), + ) + .await; + retraction.armed = false; + match written { + Ok(_) => Ok(()), + Err(error) => { + retract_or_warn(files, &staged.path).await; + Err(error) + } + } +} + +/// Deletes the snapshot file at the path. A path without a blob gives success. +pub(super) async fn retract(files: &SnapshotFiles, path: &Path) -> anyhow::Result<()> { + answer_within( + files.deadline, + files + .storage + .delete(TARGET_LABEL, "retract", files.namespace.clone(), path), + ) + .await +} + +async fn retract_or_warn(files: &SnapshotFiles, path: &Path) { + if let Err(error) = retract(files, path).await { + warn!( + path = %path.display(), + error = %format!("{error:#}"), + "Failed to delete a filesystem snapshot file whose publish did not finish" + ); + } +} + +/// Deletes the path in a task of the tracker when it is dropped while it is armed. +struct RetractOnDrop { + files: SnapshotFiles, + path: Box, + tracker: TaskTracker, + armed: bool, +} + +impl Drop for RetractOnDrop { + fn drop(&mut self) { + if !self.armed { + return; + } + let files = self.files.clone(); + let path = self.path.clone(); + match Handle::try_current() { + Ok(runtime) => { + self.tracker.spawn_on( + async move { retract_or_warn(&files, &path).await }, + &runtime, + ); + } + Err(_) => warn!( + path = %path.display(), + "Failed to delete a dropped filesystem snapshot file, because no runtime runs" + ), + } + } +} + +#[cfg(test)] +mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs new file mode 100644 index 0000000000..4042fee100 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -0,0 +1,218 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use super::super::holding::reached_deadline; +use super::super::scripted::{Script, ScriptedBlobStorage}; +use super::{SnapshotFiles, SnapshotStage, StagedSnapshot, publish, retract}; +use bytes::Bytes; +use futures::FutureExt; +use golem_common::model::environment::EnvironmentId; +use golem_service_base::storage::blob::memory::InMemoryBlobStorage; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use pretty_assertions::assert_eq; +use std::path::{Path, PathBuf}; +use std::sync::Arc; +use std::time::Duration; +use test_r::test; +use tokio_util::task::TaskTracker; +use uuid::Uuid; + +/// The longest time that a test waits for the tasks of a tracker. +const LIMIT: Duration = Duration::from_secs(10); + +const SNAPSHOT_PATH: &str = + "snapshots/cdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcd"; + +fn staged() -> StagedSnapshot { + StagedSnapshot { + path: PathBuf::from(SNAPSHOT_PATH).into_boxed_path(), + content: Bytes::from_static(b"snapshot"), + } +} + +/// Gives the snapshot files of a new namespace over a storage whose script for the publish is +/// `publish`, and the in-memory storage below it. +fn files( + publish: Script, + deadline: Duration, +) -> ( + SnapshotFiles, + Arc, + Arc, +) { + let inner = Arc::new(InMemoryBlobStorage::new()); + let storage = ScriptedBlobStorage::new(inner.clone(), move |op_label, _| { + if op_label == "publish" { + publish + } else { + Script::Pass + } + }); + ( + SnapshotFiles { + storage: storage.clone(), + namespace: BlobStorageNamespace::InitialAgentFiles { + environment_id: EnvironmentId(Uuid::new_v4()), + }, + deadline, + }, + storage, + inner, + ) +} + +/// Gives the content of the snapshot file, when the storage holds it. +async fn stored(files: &SnapshotFiles, inner: &InMemoryBlobStorage) -> Option> { + inner + .get_raw( + "test", + "test", + files.namespace.clone(), + Path::new(SNAPSHOT_PATH), + ) + .await + .unwrap() +} + +#[test] +async fn a_publish_writes_the_staged_file() { + let (files, _, inner) = files(Script::Pass, Duration::from_secs(2)); + + let published = publish(&files, &staged(), &TaskTracker::new()).await; + + assert_eq!( + (published.is_ok(), stored(&files, &inner).await), + (true, Some(b"snapshot".to_vec())) + ); +} + +#[test] +async fn a_publish_of_a_file_that_is_there_succeeds_and_keeps_the_file() { + let (files, _, inner) = files(Script::Pass, Duration::from_secs(2)); + let tracker = TaskTracker::new(); + + let first = publish(&files, &staged(), &tracker).await; + let second = publish(&files, &staged(), &tracker).await; + + assert_eq!( + (first.is_ok(), second.is_ok(), stored(&files, &inner).await), + (true, true, Some(b"snapshot".to_vec())) + ); +} + +#[test] +async fn a_publish_whose_answer_is_lost_deletes_the_file_and_gives_the_error() { + let (files, storage, inner) = files(Script::LoseTheAnswer, Duration::from_secs(2)); + + let published = publish(&files, &staged(), &TaskTracker::new()).await; + + assert_eq!( + ( + published.map_err(|error| error.to_string()), + stored(&files, &inner).await, + storage.calls(), + ), + ( + Err("the answer of the call was lost".to_string()), + None, + vec![ + ("publish", SNAPSHOT_PATH.to_string()), + ("retract", SNAPSHOT_PATH.to_string()), + ] + ) + ); +} + +#[test] +async fn a_publish_that_reaches_the_deadline_deletes_the_file_that_the_storage_wrote() { + let (files, _, inner) = files(Script::NeverAnswer, Duration::from_millis(100)); + + let published = publish(&files, &staged(), &TaskTracker::new()).await; + + assert_eq!( + ( + published + .as_ref() + .is_err_and(|error| reached_deadline(error.as_ref())), + stored(&files, &inner).await + ), + (true, None) + ); +} + +#[test] +async fn a_publish_that_the_caller_drops_deletes_the_file_in_a_task_of_the_tracker() { + let (files, _, inner) = files(Script::NeverAnswer, Duration::from_secs(60)); + let tracker = TaskTracker::new(); + + let dropped = publish(&files, &staged(), &tracker).now_or_never(); + let written_before_the_drop = stored(&files, &inner).await; + tracker.close(); + let waited = tokio::time::timeout(LIMIT, tracker.wait()).await; + + assert_eq!( + ( + dropped.is_none(), + written_before_the_drop, + waited.is_ok(), + stored(&files, &inner).await + ), + (true, Some(b"snapshot".to_vec()), true, None) + ); +} + +#[test] +async fn a_publish_that_returns_keeps_the_file_when_the_tasks_of_the_tracker_end() { + let (files, _, inner) = files(Script::Pass, Duration::from_secs(2)); + let tracker = TaskTracker::new(); + + let published = publish(&files, &staged(), &tracker).await; + tracker.close(); + let waited = tokio::time::timeout(LIMIT, tracker.wait()).await; + + assert_eq!( + ( + published.is_ok(), + waited.is_ok(), + stored(&files, &inner).await + ), + (true, true, Some(b"snapshot".to_vec())) + ); +} + +#[test] +async fn a_retract_of_a_path_without_a_file_succeeds() { + let (files, _, _) = files(Script::Pass, Duration::from_secs(2)); + + assert!(retract(&files, Path::new(SNAPSHOT_PATH)).await.is_ok()); +} + +#[test] +fn a_stage_keeps_one_file_until_it_is_taken() { + let stage = SnapshotStage::default(); + let other = StagedSnapshot { + content: Bytes::from_static(b"other"), + ..staged() + }; + + let first = stage.keep(staged()); + let second = stage.keep(other.clone()); + let taken = stage.take(); + let taken_again = stage.take(); + + assert_eq!( + (first, second, taken, taken_again), + (Ok(()), Err(other), Some(staged()), None) + ); +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs new file mode 100644 index 0000000000..ddb80b5e04 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -0,0 +1,319 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! A blob storage for tests that records each call and follows a script for each call. +//! +//! This module is test code, and it compiles only for tests. + +use async_trait::async_trait; +use bytes::Bytes; +use futures::stream::BoxStream; +use golem_service_base::replayable_stream::ErasedReplayableStream; +use golem_service_base::storage::blob::memory::InMemoryBlobStorage; +use golem_service_base::storage::blob::{ + BlobMetadata, BlobStorage, BlobStorageNamespace, ExistsResult, ListedBlob, PutIfAbsent, +}; +use std::fmt::{Debug, Formatter}; +use std::future::Future; +use std::path::{Path, PathBuf}; +use std::sync::{Arc, Mutex, PoisonError}; + +/// What the storage does with one call. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) enum Script { + /// Passes the call to the in-memory storage. + Pass, + /// Gives an error and does not pass the call. + Refuse, + /// Passes the call, and then gives an error in place of its answer. + LoseTheAnswer, + /// Passes the call, and then never answers. + NeverAnswer, +} + +/// A rule that gives the script of a call from its operation label and its path. +type Rule = Box Script + Send + Sync>; + +/// A blob storage that records the operation label and the path of each call, and does with each +/// call what its rule gives. +pub(super) struct ScriptedBlobStorage { + inner: Arc, + rule: Rule, + calls: Mutex)>>, +} + +impl ScriptedBlobStorage { + pub(super) fn new( + inner: Arc, + rule: impl Fn(&str, &Path) -> Script + Send + Sync + 'static, + ) -> Arc { + Arc::new(Self { + inner, + rule: Box::new(rule), + calls: Mutex::new(Vec::new()), + }) + } + + /// Gives the operation label and the path of each call, in the order of the calls. + pub(super) fn calls(&self) -> Vec<(&'static str, String)> { + self.calls + .lock() + .unwrap_or_else(PoisonError::into_inner) + .iter() + .map(|(op_label, path)| (*op_label, path.display().to_string())) + .collect() + } + + async fn answer( + &self, + op_label: &'static str, + path: &Path, + call: impl Future>, + ) -> anyhow::Result { + self.calls + .lock() + .unwrap_or_else(PoisonError::into_inner) + .push((op_label, path.into())); + match (self.rule)(op_label, path) { + Script::Pass => call.await, + Script::Refuse => Err(anyhow::anyhow!("the storage refused the call")), + Script::LoseTheAnswer => { + call.await?; + Err(anyhow::anyhow!("the answer of the call was lost")) + } + Script::NeverAnswer => { + call.await?; + std::future::pending().await + } + } + } +} + +impl Debug for ScriptedBlobStorage { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("ScriptedBlobStorage") + } +} + +#[async_trait] +impl BlobStorage for ScriptedBlobStorage { + async fn get_raw( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result>> { + self.answer( + op_label, + path, + self.inner.get_raw(target_label, op_label, namespace, path), + ) + .await + } + + async fn get_stream( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result>>> { + self.answer( + op_label, + path, + self.inner + .get_stream(target_label, op_label, namespace, path), + ) + .await + } + + async fn get_raw_slice( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + start: u64, + end: u64, + ) -> anyhow::Result>> { + self.answer( + op_label, + path, + self.inner + .get_raw_slice(target_label, op_label, namespace, path, start, end), + ) + .await + } + + async fn get_metadata( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result> { + self.answer( + op_label, + path, + self.inner + .get_metadata(target_label, op_label, namespace, path), + ) + .await + } + + async fn put_raw( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + data: &[u8], + ) -> anyhow::Result<()> { + self.answer( + op_label, + path, + self.inner + .put_raw(target_label, op_label, namespace, path, data), + ) + .await + } + + async fn put_raw_if_absent( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + data: &[u8], + ) -> anyhow::Result { + self.answer( + op_label, + path, + self.inner + .put_raw_if_absent(target_label, op_label, namespace, path, data), + ) + .await + } + + async fn put_stream( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + stream: &dyn ErasedReplayableStream>, Error = anyhow::Error>, + ) -> anyhow::Result<()> { + self.answer( + op_label, + path, + self.inner + .put_stream(target_label, op_label, namespace, path, stream), + ) + .await + } + + async fn delete( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result<()> { + self.answer( + op_label, + path, + self.inner.delete(target_label, op_label, namespace, path), + ) + .await + } + + async fn create_dir( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result<()> { + self.answer( + op_label, + path, + self.inner + .create_dir(target_label, op_label, namespace, path), + ) + .await + } + + async fn list_dir( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result> { + self.answer( + op_label, + path, + self.inner.list_dir(target_label, op_label, namespace, path), + ) + .await + } + + async fn list_blobs_below( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result> { + self.answer( + op_label, + path, + self.inner + .list_blobs_below(target_label, op_label, namespace, path), + ) + .await + } + + async fn delete_dir( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result { + self.answer( + op_label, + path, + self.inner + .delete_dir(target_label, op_label, namespace, path), + ) + .await + } + + async fn exists( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result { + self.answer( + op_label, + path, + self.inner.exists(target_label, op_label, namespace, path), + ) + .await + } +} From 9d167c83aa06a2b691f6d3f5ccec135e5863722d Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:10:24 -0700 Subject: [PATCH 005/126] Add the prune ledger of a snapshot scope and the rule that makes a prune due --- .../src/filesystem_snapshot/rustic/mod.rs | 2 + .../src/filesystem_snapshot/rustic/prune.rs | 294 ++++++++++++++++++ 2 files changed, 296 insertions(+) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 1e877f390d..1ac0a6aa06 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -22,6 +22,8 @@ mod backend; #[cfg_attr(not(test), allow(dead_code))] mod fault; #[cfg_attr(not(test), allow(dead_code))] +mod prune; +#[cfg_attr(not(test), allow(dead_code))] mod publish; #[cfg(test)] diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs new file mode 100644 index 0000000000..f2510a8b93 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -0,0 +1,294 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! When a delete of the store prunes the repository of its scope. +//! +//! The scope keeps a small ledger blob next to the files of the repository. The ledger holds the +//! packed bytes that deleted snapshots added since the last prune, the time of the last prune, and +//! whether that prune marked packs that a later prune removes. The ledger is advice: two deletes at +//! the same time can lose a count, and that only makes a prune come later. + +use super::backend::answer_within; +use golem_common::model::Timestamp; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use serde::{Deserialize, Serialize}; +use std::path::Path; +use std::time::Duration; +use tracing::warn; + +/// The target label of each blob storage call on the ledger. +const TARGET_LABEL: &str = "filesystem_snapshot"; + +/// The path of the ledger blob, relative to the root of the namespace of the scope. +pub(super) const LEDGER_PATH: &str = "golem/prune-ledger"; + +/// What the scope did since its last prune. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] +pub(super) struct PruneLedger { + /// The packed bytes that the deleted snapshots added, since the last prune. + pub(super) freed_bytes: u64, + /// The time of the last prune, in milliseconds since the Unix epoch. + pub(super) last_prune_millis: Option, + /// Whether the last prune marked packs that a later prune removes. + pub(super) awaiting_removal: bool, +} + +impl PruneLedger { + /// Gives the ledger after a delete of snapshots that added `bytes` packed bytes. + pub(super) fn with_deleted(self, bytes: u64) -> Self { + Self { + freed_bytes: self.freed_bytes.saturating_add(bytes), + ..self + } + } + + /// Gives the ledger after a prune at `now` that marked packs or not. + pub(super) fn after_prune(now: Timestamp, marked_packs: bool) -> Self { + Self { + freed_bytes: 0, + last_prune_millis: Some(now.to_millis()), + awaiting_removal: marked_packs, + } + } +} + +/// Tells whether a prune is due at `now`. +/// +/// A prune is due when the grace period passed since the last prune, and the freed bytes reach the +/// threshold or the last prune marked packs. A threshold of zero counts as one byte, so a prune +/// never runs for a scope that freed nothing and marked nothing. +pub(super) fn prune_due( + ledger: &PruneLedger, + now: Timestamp, + threshold: u64, + grace: Duration, +) -> bool { + let grace_passed = ledger.last_prune_millis.is_none_or(|last| { + now.to_millis() >= last.saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) + }); + let work = ledger.freed_bytes >= threshold.max(1) || ledger.awaiting_removal; + grace_passed && work +} + +/// Reads the ledger of the scope. A scope without a ledger gives an empty ledger, and so does a +/// ledger that does not parse, because the ledger is advice. +pub(super) async fn read_ledger( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, + deadline: Duration, +) -> anyhow::Result { + let content = answer_within( + deadline, + storage.get_raw( + TARGET_LABEL, + "read_ledger", + namespace.clone(), + Path::new(LEDGER_PATH), + ), + ) + .await?; + Ok(content.map_or_else(PruneLedger::default, |content| { + serde_json::from_slice(&content).unwrap_or_else(|error| { + warn!( + error = %error, + "The prune ledger of a filesystem snapshot scope does not parse, so it starts again" + ); + PruneLedger::default() + }) + })) +} + +/// Writes the ledger of the scope over the ledger that was there. +pub(super) async fn write_ledger( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, + deadline: Duration, + ledger: &PruneLedger, +) -> anyhow::Result<()> { + let content = serde_json::to_vec(ledger)?; + answer_within( + deadline, + storage.put_raw( + TARGET_LABEL, + "write_ledger", + namespace.clone(), + Path::new(LEDGER_PATH), + &content, + ), + ) + .await +} + +#[cfg(test)] +mod tests { + use super::{LEDGER_PATH, PruneLedger, prune_due, read_ledger, write_ledger}; + use golem_common::model::Timestamp; + use golem_common::model::environment::EnvironmentId; + use golem_service_base::storage::blob::memory::InMemoryBlobStorage; + use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; + use pretty_assertions::assert_eq; + use std::path::Path; + use std::time::Duration; + use test_r::test; + use uuid::Uuid; + + const MIB: u64 = 1024 * 1024; + const THRESHOLD: u64 = 64 * MIB; + const GRACE: Duration = Duration::from_secs(3600); + const DEADLINE: Duration = Duration::from_secs(2); + + fn at(millis: u64) -> Timestamp { + Timestamp::from(millis) + } + + fn ledger( + freed_bytes: u64, + last_prune_millis: Option, + awaiting_removal: bool, + ) -> PruneLedger { + PruneLedger { + freed_bytes, + last_prune_millis, + awaiting_removal, + } + } + + fn new_namespace() -> BlobStorageNamespace { + BlobStorageNamespace::InitialAgentFiles { + environment_id: EnvironmentId(Uuid::new_v4()), + } + } + + #[test] + fn a_prune_is_due_when_the_freed_bytes_reach_the_threshold() { + let now = at(10_000_000); + + assert_eq!( + [ + prune_due(&ledger(THRESHOLD - 1, None, false), now, THRESHOLD, GRACE), + prune_due(&ledger(THRESHOLD, None, false), now, THRESHOLD, GRACE), + prune_due(&ledger(THRESHOLD + 1, None, false), now, THRESHOLD, GRACE), + ], + [false, true, true] + ); + } + + #[test] + fn no_second_prune_runs_within_the_grace_period() { + let last = 1_000_000; + let grace_millis = 3_600_000; + let full = |now| { + prune_due( + &ledger(THRESHOLD, Some(last), true), + at(now), + THRESHOLD, + GRACE, + ) + }; + + assert_eq!( + [ + full(last), + full(last + grace_millis - 1), + full(last + grace_millis), + full(last + grace_millis + 1), + ], + [false, false, true, true] + ); + } + + #[test] + fn marked_packs_make_a_prune_due_after_the_grace_period_without_freed_bytes() { + let last = 1_000_000; + let after_grace = at(last + 3_600_000); + + assert_eq!( + [ + prune_due(&ledger(0, Some(last), true), after_grace, THRESHOLD, GRACE), + prune_due(&ledger(0, Some(last), false), after_grace, THRESHOLD, GRACE), + ], + [true, false] + ); + } + + #[test] + fn a_zero_threshold_prunes_after_each_delete_that_freed_bytes() { + let now = at(10_000_000); + + assert_eq!( + [ + prune_due(&ledger(0, None, false), now, 0, Duration::ZERO), + prune_due(&ledger(1, None, false), now, 0, Duration::ZERO), + prune_due(&ledger(1, Some(10_000_000), false), now, 0, Duration::ZERO), + ], + [false, true, true] + ); + } + + #[test] + fn a_delete_adds_its_bytes_and_a_prune_starts_the_ledger_again() { + let deleted = ledger(5, Some(7), true) + .with_deleted(10) + .with_deleted(u64::MAX); + + assert_eq!( + ( + deleted, + PruneLedger::after_prune(at(42), true), + PruneLedger::after_prune(at(43), false) + ), + ( + ledger(u64::MAX, Some(7), true), + ledger(0, Some(42), true), + ledger(0, Some(43), false) + ) + ); + } + + #[test] + async fn the_ledger_is_written_and_read_back() { + let storage = InMemoryBlobStorage::new(); + let namespace = new_namespace(); + let written = ledger(123, Some(456), true); + + let before = read_ledger(&storage, &namespace, DEADLINE).await.unwrap(); + write_ledger(&storage, &namespace, DEADLINE, &written) + .await + .unwrap(); + let after = read_ledger(&storage, &namespace, DEADLINE).await.unwrap(); + + assert_eq!((before, after), (PruneLedger::default(), written)); + } + + #[test] + async fn a_ledger_that_does_not_parse_reads_as_an_empty_ledger() { + let storage = InMemoryBlobStorage::new(); + let namespace = new_namespace(); + storage + .put_raw( + "test", + "test", + namespace.clone(), + Path::new(LEDGER_PATH), + b"not json", + ) + .await + .unwrap(); + + assert_eq!( + read_ledger(&storage, &namespace, DEADLINE).await.unwrap(), + PruneLedger::default() + ); + } +} From a358148f3724a8eff610410a0635e891dbd6dd58 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:10:25 -0700 Subject: [PATCH 006/126] Copy and delete a snapshot scope on its blobs --- .../src/filesystem_snapshot/rustic/mod.rs | 2 + .../src/filesystem_snapshot/rustic/scope.rs | 173 +++++++++++ .../filesystem_snapshot/rustic/scope/tests.rs | 269 ++++++++++++++++++ 3 files changed, 444 insertions(+) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 1ac0a6aa06..67e2552bc5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -25,6 +25,8 @@ mod fault; mod prune; #[cfg_attr(not(test), allow(dead_code))] mod publish; +#[cfg_attr(not(test), allow(dead_code))] +mod scope; #[cfg(test)] mod holding; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs new file mode 100644 index 0000000000..30f125eca9 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs @@ -0,0 +1,173 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The copy and the delete of a whole scope, on the blobs of its repository. +//! +//! These operations do not read the repository format. They only know the directories of the +//! repository, its config file, and the ledger directory of the store. + +use super::backend::answer_within; +use super::prune::LEDGER_PATH; +use futures::{StreamExt, TryStreamExt, stream}; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace, PutIfAbsent}; +use std::path::{Path, PathBuf}; +use std::time::Duration; + +/// The target label of each blob storage call on a scope. +const TARGET_LABEL: &str = "filesystem_snapshot"; + +/// The path of the config file of a repository. +const CONFIG_PATH: &str = "config"; + +/// The directories of a repository in the order of a listing. A save writes them in the reverse +/// order, so each snapshot file in a listing has its index files and packs in the later listings. +const LISTING_ORDER: [&str; 4] = ["snapshots", "index", "keys", "data"]; + +/// The directories of a repository in the order of the writes of a copy: packs, index files, keys, +/// then snapshot files. So a snapshot file in the target always has its data. +const COPY_ORDER: [&str; 4] = ["data", "index", "keys", "snapshots"]; + +/// Copies the repository of the namespace `from` into the empty namespace `to`. +/// +/// A namespace without a config file holds no repository, so the call copies nothing. The call +/// writes the config file of `to` last, so `to` holds a repository only when all its blobs are +/// there. It does not copy the ledger. A blob that is gone when the call reads it was deleted +/// after the listing, and the call does not copy it. +pub(super) async fn copy_scope( + storage: &dyn BlobStorage, + from: &BlobStorageNamespace, + to: &BlobStorageNamespace, + deadline: Duration, +) -> anyhow::Result<()> { + let Some(config) = answer_within( + deadline, + storage.get_raw( + TARGET_LABEL, + "copy_read", + from.clone(), + Path::new(CONFIG_PATH), + ), + ) + .await? + else { + return Ok(()); + }; + let listed = stream::iter(LISTING_ORDER) + .then(|directory| async move { + answer_within( + deadline, + storage.list_blobs_below( + TARGET_LABEL, + "copy_list", + from.clone(), + Path::new(directory), + ), + ) + .await + .map(|blobs| (directory, blobs)) + }) + .try_collect::>() + .await?; + let paths = COPY_ORDER + .iter() + .flat_map(|directory| { + listed + .iter() + .filter(move |(listed_directory, _)| listed_directory == directory) + .flat_map(|(_, blobs)| blobs.iter().map(|blob| blob.path.clone())) + }) + .collect::>(); + stream::iter(paths.iter().map(Ok)) + .try_for_each(|path| copy_blob(storage, from, to, path, deadline)) + .await?; + answer_within( + deadline, + storage.put_raw_if_absent( + TARGET_LABEL, + "copy_write", + to.clone(), + Path::new(CONFIG_PATH), + &config, + ), + ) + .await + .map(|_: PutIfAbsent| ()) +} + +async fn copy_blob( + storage: &dyn BlobStorage, + from: &BlobStorageNamespace, + to: &BlobStorageNamespace, + path: &Path, + deadline: Duration, +) -> anyhow::Result<()> { + let content = answer_within( + deadline, + storage.get_raw(TARGET_LABEL, "copy_read", from.clone(), path), + ) + .await?; + match content { + Some(content) => { + answer_within( + deadline, + storage.put_raw(TARGET_LABEL, "copy_write", to.clone(), path, &content), + ) + .await + } + None => Ok(()), + } +} + +/// Deletes the repository of the namespace, and the ledger of the store. +/// +/// The call deletes the config file first, so the namespace holds no repository from that step on. +/// Then it deletes each directory. A namespace that holds nothing gives success. +pub(super) async fn delete_scope( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, + deadline: Duration, +) -> anyhow::Result<()> { + answer_within( + deadline, + storage.delete( + TARGET_LABEL, + "delete_scope", + namespace.clone(), + Path::new(CONFIG_PATH), + ), + ) + .await?; + let ledger_directory = Path::new(LEDGER_PATH) + .parent() + .map(Path::to_path_buf) + .unwrap_or_default(); + let directories = LISTING_ORDER + .iter() + .map(PathBuf::from) + .chain(std::iter::once(ledger_directory)) + .collect::>(); + stream::iter(directories.iter().map(Ok)) + .try_for_each(|directory| async move { + answer_within( + deadline, + storage.delete_dir(TARGET_LABEL, "delete_scope", namespace.clone(), directory), + ) + .await + .map(|_| ()) + }) + .await +} + +#[cfg(test)] +mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs new file mode 100644 index 0000000000..90b940d21c --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs @@ -0,0 +1,269 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use super::super::scripted::{Script, ScriptedBlobStorage}; +use super::{copy_scope, delete_scope}; +use golem_common::model::environment::EnvironmentId; +use golem_service_base::storage::blob::memory::InMemoryBlobStorage; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use pretty_assertions::assert_eq; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; +use test_r::test; +use uuid::Uuid; + +const DEADLINE: Duration = Duration::from_secs(2); + +/// The blobs of a small repository, with the ledger of the store. +const REPOSITORY: [(&str, &str); 6] = [ + ("config", "config"), + ("data/ab/abab", "pack"), + ("golem/prune-ledger", "ledger"), + ("index/cdcd", "index"), + ("keys/efef", "key"), + ("snapshots/0101", "snapshot"), +]; + +fn new_namespace() -> BlobStorageNamespace { + BlobStorageNamespace::InitialAgentFiles { + environment_id: EnvironmentId(Uuid::new_v4()), + } +} + +async fn put_all( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, + blobs: &[(&str, &str)], +) { + futures::future::join_all(blobs.iter().map(|(path, content)| { + storage.put_raw( + "test", + "test", + namespace.clone(), + Path::new(path), + content.as_bytes(), + ) + })) + .await + .into_iter() + .collect::>>() + .unwrap(); +} + +/// Gives the path and the content of each blob of the namespace, in the order of the paths. +async fn stored( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, +) -> Vec<(String, String)> { + let listed = storage + .list_blobs_below("test", "test", namespace.clone(), Path::new("")) + .await + .unwrap(); + let mut blobs = futures::future::join_all(listed.iter().map(|blob| async { + let content = storage + .get_raw("test", "test", namespace.clone(), &blob.path) + .await + .unwrap() + .unwrap(); + ( + blob.path.display().to_string(), + String::from_utf8(content).unwrap(), + ) + })) + .await; + blobs.sort(); + blobs +} + +fn owned(blobs: &[(&str, &str)]) -> Vec<(String, String)> { + blobs + .iter() + .map(|(path, content)| (path.to_string(), content.to_string())) + .collect() +} + +#[test] +async fn a_copy_gives_the_target_each_blob_of_the_repository_and_not_the_ledger() { + let storage = InMemoryBlobStorage::new(); + let (from, to) = (new_namespace(), new_namespace()); + put_all(&storage, &from, &REPOSITORY).await; + + copy_scope(&storage, &from, &to, DEADLINE).await.unwrap(); + + assert_eq!( + (stored(&storage, &to).await, stored(&storage, &from).await), + ( + owned( + &REPOSITORY + .into_iter() + .filter(|(path, _)| !path.starts_with("golem/")) + .collect::>() + ), + owned(&REPOSITORY) + ) + ); +} + +#[test] +async fn a_copy_writes_the_packs_the_index_files_the_keys_the_snapshot_files_and_then_the_config() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let (from, to) = (new_namespace(), new_namespace()); + put_all(&*storage, &from, &REPOSITORY).await; + + copy_scope(&*storage, &from, &to, DEADLINE).await.unwrap(); + + assert_eq!( + storage + .calls() + .into_iter() + .filter(|(op_label, _)| *op_label == "copy_write") + .map(|(_, path)| path) + .collect::>(), + vec![ + "data/ab/abab", + "index/cdcd", + "keys/efef", + "snapshots/0101", + "config" + ] + ); +} + +#[test] +async fn a_copy_lists_the_snapshot_files_before_the_index_files_the_keys_and_the_packs() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let (from, to) = (new_namespace(), new_namespace()); + put_all(&*storage, &from, &REPOSITORY).await; + + copy_scope(&*storage, &from, &to, DEADLINE).await.unwrap(); + + assert_eq!( + storage + .calls() + .into_iter() + .filter(|(op_label, _)| *op_label == "copy_list") + .map(|(_, path)| path) + .collect::>(), + vec!["snapshots", "index", "keys", "data"] + ); +} + +#[test] +async fn a_copy_of_a_namespace_without_a_config_copies_nothing() { + let storage = InMemoryBlobStorage::new(); + let (from, to) = (new_namespace(), new_namespace()); + put_all( + &storage, + &from, + &REPOSITORY + .into_iter() + .filter(|(path, _)| *path != "config") + .collect::>(), + ) + .await; + + copy_scope(&storage, &from, &to, DEADLINE).await.unwrap(); + + assert_eq!(stored(&storage, &to).await, Vec::<(String, String)>::new()); +} + +#[test] +async fn a_copy_that_fails_gives_the_error_and_the_target_has_no_config() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "copy_read" && path.starts_with("index") { + Script::Refuse + } else { + Script::Pass + } + }); + let (from, to) = (new_namespace(), new_namespace()); + put_all(&*storage, &from, &REPOSITORY).await; + + let copied = copy_scope(&*storage, &from, &to, DEADLINE).await; + + assert_eq!( + ( + copied.is_err(), + stored(&*storage, &to) + .await + .into_iter() + .map(|(path, _)| path) + .collect::>() + ), + (true, vec!["data/ab/abab".to_string()]) + ); +} + +#[test] +async fn a_deleted_scope_holds_no_blob_and_another_scope_keeps_its_blobs() { + let storage = InMemoryBlobStorage::new(); + let (deleted, kept) = (new_namespace(), new_namespace()); + put_all(&storage, &deleted, &REPOSITORY).await; + put_all(&storage, &kept, &REPOSITORY).await; + + delete_scope(&storage, &deleted, DEADLINE).await.unwrap(); + + assert_eq!( + ( + stored(&storage, &deleted).await, + stored(&storage, &kept).await + ), + (Vec::new(), owned(&REPOSITORY)) + ); +} + +#[test] +async fn a_delete_of_a_scope_deletes_the_config_first() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let namespace = new_namespace(); + put_all(&*storage, &namespace, &REPOSITORY).await; + + delete_scope(&*storage, &namespace, DEADLINE).await.unwrap(); + + assert_eq!( + storage + .calls() + .into_iter() + .filter(|(op_label, _)| *op_label == "delete_scope") + .map(|(_, path)| path) + .collect::>(), + vec!["config", "snapshots", "index", "keys", "data", "golem"] + ); +} + +#[test] +async fn a_delete_of_an_unused_scope_succeeds_and_can_run_again() { + let storage = InMemoryBlobStorage::new(); + let namespace = new_namespace(); + + let first = delete_scope(&storage, &namespace, DEADLINE).await; + put_all(&storage, &namespace, &REPOSITORY).await; + let second = delete_scope(&storage, &namespace, DEADLINE).await; + let third = delete_scope(&storage, &namespace, DEADLINE).await; + + assert_eq!( + ( + first.is_ok(), + second.is_ok(), + third.is_ok(), + stored(&storage, &namespace).await + ), + (true, true, true, Vec::new()) + ); +} From 96d550aeb75843ccd9c24ec32ce2b42239f792f7 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:34:48 -0700 Subject: [PATCH 007/126] Hold the delete of a dropped publish until the test reads the file --- .../rustic/publish/tests.rs | 39 ++++++++++++------- .../filesystem_snapshot/rustic/scripted.rs | 14 +++++++ 2 files changed, 39 insertions(+), 14 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs index 4042fee100..a61fc61724 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -42,9 +42,10 @@ fn staged() -> StagedSnapshot { } /// Gives the snapshot files of a new namespace over a storage whose script for the publish is -/// `publish`, and the in-memory storage below it. +/// `publish` and for the delete is `retract`, and the in-memory storage below it. fn files( publish: Script, + retract: Script, deadline: Duration, ) -> ( SnapshotFiles, @@ -52,12 +53,10 @@ fn files( Arc, ) { let inner = Arc::new(InMemoryBlobStorage::new()); - let storage = ScriptedBlobStorage::new(inner.clone(), move |op_label, _| { - if op_label == "publish" { - publish - } else { - Script::Pass - } + let storage = ScriptedBlobStorage::new(inner.clone(), move |op_label, _| match op_label { + "publish" => publish, + "retract" => retract, + _ => Script::Pass, }); ( SnapshotFiles { @@ -87,7 +86,7 @@ async fn stored(files: &SnapshotFiles, inner: &InMemoryBlobStorage) -> Option, rule: Rule, calls: Mutex)>>, + gate: CancellationToken, } impl ScriptedBlobStorage { @@ -62,9 +66,15 @@ impl ScriptedBlobStorage { inner, rule: Box::new(rule), calls: Mutex::new(Vec::new()), + gate: CancellationToken::new(), }) } + /// Lets each call that waits for the gate, and each later such call, go on. + pub(super) fn open_gate(&self) { + self.gate.cancel(); + } + /// Gives the operation label and the path of each call, in the order of the calls. pub(super) fn calls(&self) -> Vec<(&'static str, String)> { self.calls @@ -96,6 +106,10 @@ impl ScriptedBlobStorage { call.await?; std::future::pending().await } + Script::WaitForGate => { + self.gate.cancelled().await; + call.await + } } } } From ba222da8cc2a24c3003f8ed3d98eef3cb800ed12 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:34:49 -0700 Subject: [PATCH 008/126] Write the blobs of a scope copy in the reverse order of the listing --- .../src/filesystem_snapshot/rustic/scope.rs | 17 +++++------------ .../filesystem_snapshot/rustic/scope/tests.rs | 9 ++++++--- 2 files changed, 11 insertions(+), 15 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs index 30f125eca9..d80ebe91f6 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs @@ -32,12 +32,10 @@ const CONFIG_PATH: &str = "config"; /// The directories of a repository in the order of a listing. A save writes them in the reverse /// order, so each snapshot file in a listing has its index files and packs in the later listings. +/// A copy writes them in the reverse order too, so a snapshot file in the target always has its +/// data. const LISTING_ORDER: [&str; 4] = ["snapshots", "index", "keys", "data"]; -/// The directories of a repository in the order of the writes of a copy: packs, index files, keys, -/// then snapshot files. So a snapshot file in the target always has its data. -const COPY_ORDER: [&str; 4] = ["data", "index", "keys", "snapshots"]; - /// Copies the repository of the namespace `from` into the empty namespace `to`. /// /// A namespace without a config file holds no repository, so the call copies nothing. The call @@ -75,18 +73,13 @@ pub(super) async fn copy_scope( ), ) .await - .map(|blobs| (directory, blobs)) }) .try_collect::>() .await?; - let paths = COPY_ORDER + let paths = listed .iter() - .flat_map(|directory| { - listed - .iter() - .filter(move |(listed_directory, _)| listed_directory == directory) - .flat_map(|(_, blobs)| blobs.iter().map(|blob| blob.path.clone())) - }) + .rev() + .flat_map(|blobs| blobs.iter().map(|blob| blob.path.clone())) .collect::>(); stream::iter(paths.iter().map(Ok)) .try_for_each(|path| copy_blob(storage, from, to, path, deadline)) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs index 90b940d21c..aaff157543 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs @@ -117,7 +117,7 @@ async fn a_copy_gives_the_target_each_blob_of_the_repository_and_not_the_ledger( } #[test] -async fn a_copy_writes_the_packs_the_index_files_the_keys_the_snapshot_files_and_then_the_config() { +async fn a_copy_writes_the_packs_the_keys_the_index_files_the_snapshot_files_and_then_the_config() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); let (from, to) = (new_namespace(), new_namespace()); @@ -134,8 +134,8 @@ async fn a_copy_writes_the_packs_the_index_files_the_keys_the_snapshot_files_and .collect::>(), vec![ "data/ab/abab", - "index/cdcd", "keys/efef", + "index/cdcd", "snapshots/0101", "config" ] @@ -205,7 +205,10 @@ async fn a_copy_that_fails_gives_the_error_and_the_target_has_no_config() { .map(|(path, _)| path) .collect::>() ), - (true, vec!["data/ab/abab".to_string()]) + ( + true, + vec!["data/ab/abab".to_string(), "keys/efef".to_string()] + ) ); } From ac83e53f046aebbd0b499cd09428a36c17493e64 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:34:49 -0700 Subject: [PATCH 009/126] Separate the key constant of the configuration tests --- golem-worker-executor/src/services/golem_config.rs | 1 + 1 file changed, 1 insertion(+) diff --git a/golem-worker-executor/src/services/golem_config.rs b/golem-worker-executor/src/services/golem_config.rs index 0346bf94e2..82e667787f 100644 --- a/golem-worker-executor/src/services/golem_config.rs +++ b/golem-worker-executor/src/services/golem_config.rs @@ -2952,6 +2952,7 @@ mod tests { assert!(serde_json::from_value::(serialized).is_err()); } + const KEY: &str = "000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f\ 202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f"; From 6b3f43102bff4ca288b7ded436f304641b38ee59 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:16:14 -0700 Subject: [PATCH 010/126] Take the storage call deadline of the rustic bridge from the configuration default --- .../src/filesystem_snapshot/benchmark/mod.rs | 3 ++- .../src/filesystem_snapshot/rustic/backend/tests.rs | 2 +- .../src/filesystem_snapshot/rustic/mod.rs | 9 --------- .../src/filesystem_snapshot/rustic/tests.rs | 6 +++--- 4 files changed, 6 insertions(+), 14 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs index 97ace67da1..7b05b1ad9e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs @@ -39,9 +39,10 @@ mod volume; use super::rustic::{ ChangeDetection, Chunking, Compression, InspectReport, PhaseTime, Repository, RepositoryKey, - RepositorySettings, STORAGE_CALL_DEADLINE, SaveSettings, + RepositorySettings, SaveSettings, }; use super::{SnapshotName, SnapshotScope}; +use crate::services::golem_config::DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE as STORAGE_CALL_DEADLINE; use agents::{AgentStorage, FIRST_AGENT}; use golem_common::model::environment::EnvironmentId; use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index 66fa7afa3d..cb52483c36 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -17,13 +17,13 @@ //! Each test calls the backend from the thread of the test. That thread is not a thread of the //! runtime that the backend holds, as the threads of rustic are not. -use super::super::STORAGE_CALL_DEADLINE; use super::super::fault::{Operation, OperationCancelled, classify, is_config_exists}; use super::super::holding::{holding_storage, reached_deadline}; use super::super::publish::{SnapshotStage, StagedSnapshot}; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::{BlobBackend, file_size}; use crate::filesystem_snapshot::SnapshotStoreError; +use crate::services::golem_config::DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE as STORAGE_CALL_DEADLINE; use anyhow::anyhow; use async_trait::async_trait; use bytes::Bytes; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 67e2552bc5..5fe655745c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -57,15 +57,6 @@ use std::sync::Arc; use std::time::{Duration, Instant}; use tokio::runtime::Handle; -/// The longest time that one call of a repository waits for the blob storage. -/// -/// A call that gets no answer within this time fails, and its operation fails with it. The value -/// stops a call that does not return. It is not a limit for a slow call. On S3, with the retries of -/// the S3 storage, a write of a pack took at most 1.7 s with eight saves at the same time. A ranged -/// read of a pack took at most 1.5 s under the CPU request of an executor. Keep the value at least -/// 10 times the longest measured call. -pub(super) const STORAGE_CALL_DEADLINE: Duration = Duration::from_secs(30); - /// The key that encrypts a repository. /// /// The key has 64 bytes: 32 bytes of the AES-256 key, then 16 bytes of the number `k` and 16 diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index 4ed61a7a94..d94c238915 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -22,15 +22,15 @@ use super::backend::BlobBackend; use super::holding::{holding_storage, reached_deadline}; use super::{ ChangeDetection, Chunking, Compression, OperationPhase, PruneSettings, RepackLimits, - Repository, RepositoryKey, RepositorySettings, STORAGE_CALL_DEADLINE, SaveSettings, - backup_options, config_options, open_existing, prune_options, repository_options, run_blocking, - unopened, + Repository, RepositoryKey, RepositorySettings, SaveSettings, backup_options, config_options, + open_existing, prune_options, repository_options, run_blocking, unopened, }; use crate::filesystem_snapshot::contract_tests::fixture::{ Scratch, Spec, fixture, listing, write_tree, }; use crate::filesystem_snapshot::contract_tests::new_scope; use crate::filesystem_snapshot::{SnapshotName, SnapshotScope}; +use crate::services::golem_config::DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE as STORAGE_CALL_DEADLINE; use anyhow::Context; use async_trait::async_trait; use bytes::Bytes; From 56157e3fafa6cc1921bc05e3feb63cbc39e16461 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:16:20 -0700 Subject: [PATCH 011/126] Open the repository of the writer that makes the config file first --- .../src/filesystem_snapshot/rustic/mod.rs | 23 ++++++++++++------- 1 file changed, 15 insertions(+), 8 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 5fe655745c..f6bf801017 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -604,24 +604,31 @@ fn forget( } /// Opens the repository, or makes it when the scope has none. +/// +/// When another writer makes the config file of the repository first, the call opens the +/// repository of that writer. fn open_or_create( backend: Arc, key: &RepositoryKey, settings: &RepositorySettings, ) -> RusticResult<(RusticRepository, OperationPhase)> { - let repository = unopened(backend)?; + let repository = unopened(backend.clone())?; let credentials = Credentials::Masterkey(key.master_key()); match repository.config_id()? { Some(_) => repository .open(&credentials) .map(|repository| (repository, OperationPhase::Open)), - None => repository - .init( - &credentials, - &KeyOptions::default(), - &config_options(settings), - ) - .map(|repository| (repository, OperationPhase::Create)), + None => match repository.init( + &credentials, + &KeyOptions::default(), + &config_options(settings), + ) { + Ok(repository) => Ok((repository, OperationPhase::Create)), + Err(error) if fault::is_config_exists(&*error) => unopened(backend)? + .open(&credentials) + .map(|repository| (repository, OperationPhase::Open)), + Err(error) => Err(error), + }, } } From a9f8ea192a55fddcbc386fbd3a81c402d618b371 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:16:20 -0700 Subject: [PATCH 012/126] Add the rustic filesystem snapshot store --- .../src/filesystem_snapshot/mod.rs | 2 + .../src/filesystem_snapshot/rustic/backend.rs | 20 +- .../src/filesystem_snapshot/rustic/fault.rs | 32 + .../src/filesystem_snapshot/rustic/mod.rs | 43 +- .../src/filesystem_snapshot/rustic/store.rs | 706 +++++++++++ .../filesystem_snapshot/rustic/store/tests.rs | 1116 +++++++++++++++++ 6 files changed, 1904 insertions(+), 15 deletions(-) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/store.rs create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/mod.rs b/golem-worker-executor/src/filesystem_snapshot/mod.rs index 9f1abd7681..b4354f2a47 100644 --- a/golem-worker-executor/src/filesystem_snapshot/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/mod.rs @@ -36,6 +36,8 @@ mod time_zone_tests; #[allow(unused_imports)] pub(crate) use memory::InMemorySnapshotStore; +#[allow(unused_imports)] +pub(crate) use rustic::RusticSnapshotStore; /// The place of the filesystem snapshots of one agent. /// diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index 8d99fc90f5..0c64a04aeb 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -19,7 +19,7 @@ //! and each call waits for the blob storage on the runtime that the backend holds. Each call waits //! for at most a deadline, and a cancelled operation makes no more calls. -use super::fault::{BlobCallFailed, ConfigExists, OperationCancelled}; +use super::fault::{BlobCallFailed, ConfigExists, FileMissing, OperationCancelled}; use super::publish::{SnapshotStage, StagedSnapshot}; use bytes::Bytes; use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace, PutIfAbsent}; @@ -32,6 +32,7 @@ use std::sync::Arc; use std::time::Duration; use tokio::runtime::Handle; use tokio_util::sync::CancellationToken; +use tokio_util::task::task_tracker::TaskTrackerToken; /// The target label of each blob storage call of the backend. const TARGET_LABEL: &str = "filesystem_snapshot"; @@ -91,6 +92,8 @@ pub(super) struct BlobBackend { deadline: Duration, cancel: CancellationToken, stage: Option>, + /// Counts the backend as work of a tracker, until the last owner drops the backend. + _tracked: Option, } impl BlobBackend { @@ -109,18 +112,17 @@ impl BlobBackend { deadline, cancel: CancellationToken::new(), stage: None, + _tracked: None, } } /// Gives the backend with the token of its operation. When the token is cancelled, a call that /// has not started gives an error at once, and a call that runs stops and gives an error. - #[cfg_attr(not(test), allow(dead_code))] pub(super) fn cancelled_by(self, cancel: CancellationToken) -> Self { Self { cancel, ..self } } /// Gives the backend with a stage for the snapshot file of a save. - #[cfg_attr(not(test), allow(dead_code))] pub(super) fn staging_in(self, stage: Arc) -> Self { Self { stage: Some(stage), @@ -128,6 +130,15 @@ impl BlobBackend { } } + /// Gives the backend with a token of a task tracker. The tracker counts the backend until the + /// last owner drops it, for example a thread of rustic. + pub(super) fn tracked_by(self, token: TaskTrackerToken) -> Self { + Self { + _tracked: Some(token), + ..self + } + } + /// Waits for one call on the blob storage, and gives its result as a rustic result. /// /// Each call of the backend on the blob storage goes through this function. A call that gives @@ -398,9 +409,10 @@ fn join(parts: &[Bytes]) -> Box<[u8]> { /// The error of a file that the blob storage does not hold. fn missing_file(path: &Path) -> Box { - RusticError::new( + RusticError::with_source( ErrorKind::Backend, "The blob storage holds no file at `{path}`.", + FileMissing, ) .attach_context("path", path.display().to_string()) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs index 95a73d4718..0c273adf74 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs @@ -72,6 +72,24 @@ impl Display for ConfigExists { impl Error for ConfigExists {} +/// The blob storage holds no file at the path that rustic reads, for example because a delete +/// removed it after a listing. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct FileMissing; + +impl Display for FileMissing { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("the blob storage holds no file at the path") + } +} + +impl Error for FileMissing {} + +/// Tells whether an error in the chain is [`FileMissing`]. +pub(super) fn is_file_missing(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| error.is::()) +} + /// Tells whether an error in the chain is [`ConfigExists`]. pub(super) fn is_config_exists(error: &(dyn Error + 'static)) -> bool { chain(error).any(|error| error.is::()) @@ -123,6 +141,20 @@ pub(super) fn classify(operation: Operation, error: anyhow::Error) -> SnapshotSt } } +/// Tells whether an error in the chain is a failed blob storage call. +pub(super) fn is_storage_failure(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| error.is::()) +} + +/// Gives the error of the store for a blob storage call that the store made without rustic. It is +/// retryable unless a name error of the blob storage caused it. +pub(super) fn storage_failure(error: anyhow::Error) -> SnapshotStoreError { + classify( + Operation::Repository, + anyhow::Error::new(BlobCallFailed::new(error)), + ) +} + /// Gives the text of the error with the text of each of its sources. fn error_text(error: &anyhow::Error) -> String { format!("{error:#}") diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index f6bf801017..6be31c2ff0 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -19,14 +19,13 @@ //! module. mod backend; -#[cfg_attr(not(test), allow(dead_code))] mod fault; -#[cfg_attr(not(test), allow(dead_code))] mod prune; -#[cfg_attr(not(test), allow(dead_code))] mod publish; -#[cfg_attr(not(test), allow(dead_code))] mod scope; +mod store; + +pub(crate) use store::RusticSnapshotStore; #[cfg(test)] mod holding; @@ -512,29 +511,51 @@ fn restore( let Some(snapshot) = snapshot else { return Ok(None); }; + let restored = restore_snapshot( + repository, + &snapshot, + into, + &RestoreOptions::default().reader_threads(reader_threads), + )?; + Ok(Some(RestoreReport { + phases: [open, lookup] + .into_iter() + .chain(restored.phases.iter().copied()) + .collect(), + ..restored + })) +} + +/// Writes the tree of the snapshot into the empty directory `into`. The phases of the result are +/// the index load, the plan and the writes. +fn restore_snapshot( + repository: RusticRepository, + snapshot: &SnapshotFile, + into: &Path, + options: &RestoreOptions, +) -> anyhow::Result { let (repository, index) = timed(OperationPhase::IndexLoad, || repository.to_indexed())?; let into = into .to_str() .context("the directory of a restore must have a UTF-8 path")?; let destination = LocalDestination::new(into, false, false)?; - let node = repository.node_from_snapshot_and_path(&snapshot, "")?; + let node = repository.node_from_snapshot_and_path(snapshot, "")?; let entries = repository.ls(&node, &LsOptions::default())?; - let options = RestoreOptions::default().reader_threads(reader_threads); let (plan, planning) = timed(OperationPhase::RestorePlan, || { - repository.prepare_restore(&options, entries.clone(), &destination, false) + repository.prepare_restore(options, entries.clone(), &destination, false) })?; let files = plan.stats.files.restore; let dirs = plan.stats.dirs.restore; let bytes = plan.restore_size; let ((), writing) = timed(OperationPhase::Restore, || { - repository.restore(plan, &options, entries, &destination) + repository.restore(plan, options, entries, &destination) })?; - Ok(Some(RestoreReport { + Ok(RestoreReport { files, dirs, bytes, - phases: Box::new([open, lookup, index, planning, writing]), - })) + phases: Box::new([index, planning, writing]), + }) } fn prune( diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs new file mode 100644 index 0000000000..705d2de252 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -0,0 +1,706 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The filesystem snapshot store over the rustic repositories of the scopes. +//! +//! A save makes the snapshot visible in one step: the backend keeps the snapshot file of the +//! backup, and the save writes that file after the blocking work returns. Each operation has a +//! cancellation token that its drop cancels, so the threads of a dropped operation stop at their +//! next storage call. The store counts each blocking task and each backend in a task tracker, and +//! [`RusticSnapshotStore::shut_down`] waits for them. + +use super::backend::BlobBackend; +use super::fault::{Operation, classify, is_file_missing, is_storage_failure, storage_failure}; +use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; +use super::publish::{SnapshotFiles, SnapshotStage, StagedSnapshot, publish}; +use super::scope::{copy_scope, delete_scope}; +use super::{ + ChangeDetection, PruneSettings, RepackLimits, RepositoryKey, RepositorySettings, SaveSettings, + backup_options, open_existing, open_or_create, prune, restore_snapshot, run_blocking, +}; +use crate::filesystem_snapshot::{ + FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, + newest_first, snapshot_time, +}; +use crate::services::golem_config::FilesystemSnapshotStoreConfig; +use anyhow::Context; +use async_trait::async_trait; +use golem_common::model::Timestamp; +use golem_service_base::storage::blob::BlobStorage; +use rustic_core::jiff::tz::TimeZone; +use rustic_core::jiff::{Timestamp as SnapshotTime, Zoned}; +use rustic_core::repofile::{SnapshotFile, SnapshotId}; +use rustic_core::{ + BackupOptions, DevIdOption, LocalSourceSaveOptions, Open, PathList, + Repository as RusticRepository, RestoreOptions, SnapshotOptions, +}; +use serde::{Deserialize, Serialize}; +use std::num::NonZeroUsize; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; +use tokio::runtime::Handle; +use tokio_util::sync::{CancellationToken, DropGuard}; +use tokio_util::task::TaskTracker; + +/// The packed bytes that deleted snapshots must free before a delete prunes the scope. +const PRUNE_THRESHOLD_BYTES: u64 = 64 * 1024 * 1024; + +/// How long a pack that a prune marks stays before a later prune deletes it. It is also the +/// shortest time between two prunes of one scope. It must be longer than the longest save and the +/// longest restore. +const PRUNE_GRACE: Duration = Duration::from_secs(3600); + +/// The settings of the store: the rustic settings of each operation, and the prune threshold. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) struct StorePolicy { + /// The longest time that one blob storage call waits for an answer. + pub(super) deadline: Duration, + /// The settings of a repository that a save makes. + pub(super) repository: RepositorySettings, + pub(super) save: SaveSettings, + /// The number of threads that read packs in a restore. + pub(super) restore_reader_threads: NonZeroUsize, + /// The settings of a prune. `keep_delete` is also the shortest time between two prunes. + pub(super) prune: PruneSettings, + /// The packed bytes that deleted snapshots must free before a delete prunes. + pub(super) prune_threshold: u64, +} + +impl StorePolicy { + /// Gives the policy with the values of the configuration. + pub(super) fn from_config(config: &FilesystemSnapshotStoreConfig) -> Self { + Self { + deadline: config.storage_call_deadline(), + repository: RepositorySettings::DEFAULT, + save: SaveSettings { + threads: Some(config.save_threads()), + detection: ChangeDetection::Ctime, + }, + restore_reader_threads: config.restore_reader_threads(), + prune: PruneSettings { + fast_repack: false, + keep_delete: PRUNE_GRACE, + repack: RepackLimits::Rustic, + }, + prune_threshold: PRUNE_THRESHOLD_BYTES, + } + } +} + +/// The options of a save of the store: the options of the bridge, and a save that cannot read an +/// entry fails before it writes the snapshot file. A save records no device id, so a restore gives +/// each name of a hard-linked file as its own file. +fn store_backup_options(policy: &StorePolicy) -> BackupOptions { + backup_options(&policy.save) + .fail_on_read_error(true) + .ignore_save_opts(LocalSourceSaveOptions::default().set_devid(DevIdOption::No)) +} + +/// The options of a restore of the store. A metadata error fails the restore. The restore does not +/// set the owner, because a snapshot does not keep it. +fn store_restore_options(policy: &StorePolicy) -> RestoreOptions { + RestoreOptions::default() + .reader_threads(Some(policy.restore_reader_threads)) + .fail_on_metadata_error(true) + .no_ownership(true) +} + +/// A filesystem snapshot store that keeps one rustic repository for each scope in blob storage. +pub(crate) struct RusticSnapshotStore { + storage: Arc, + key: RepositoryKey, + policy: StorePolicy, + /// The parent of the token of each operation. + root: CancellationToken, + /// Counts the blocking tasks, the backends and the deletes of dropped publishes. + tracker: TaskTracker, +} + +impl RusticSnapshotStore { + /// Gives the store over the blob storage, with the key and the values of the configuration. + pub(crate) fn new( + storage: Arc, + config: &FilesystemSnapshotStoreConfig, + ) -> Self { + Self::with_policy( + storage, + RepositoryKey::new(*config.repository_key().bytes()), + StorePolicy::from_config(config), + ) + } + + pub(super) fn with_policy( + storage: Arc, + key: RepositoryKey, + policy: StorePolicy, + ) -> Self { + Self { + storage, + key, + policy, + root: CancellationToken::new(), + tracker: TaskTracker::new(), + } + } + + /// Stops each operation at its next storage call, and waits until no blocking task and no + /// backend of the store remains. After the call, each operation gives `Storage`. + /// + /// The runtime must not drop before the call returns, because a storage call that waits on the + /// runtime after its time driver stops aborts the process. + pub(crate) async fn shut_down(&self) { + self.root.cancel(); + self.tracker.close(); + self.tracker.wait().await; + } + + /// Gives the number of blocking tasks, backends and deletes of the store that have not ended. + #[cfg(test)] + pub(super) fn work_in_flight(&self) -> usize { + self.tracker.len() + } + + /// Starts an operation. The token of the operation is cancelled when the guard drops. + fn start(&self) -> Result<(CancellationToken, DropGuard), SnapshotStoreError> { + if self.root.is_cancelled() { + return Err(SnapshotStoreError::Storage { + retryable: false, + source: anyhow::anyhow!("the filesystem snapshot store is shut down"), + }); + } + let token = self.root.child_token(); + Ok((token.clone(), token.drop_guard())) + } + + /// Gives a backend over the repository of the scope for the operation with the token. + fn backend( + &self, + scope: &SnapshotScope, + token: &CancellationToken, + ) -> Result { + let runtime = Handle::try_current() + .context("a filesystem snapshot operation needs an async runtime") + .map_err(|source| SnapshotStoreError::Storage { + retryable: false, + source, + })?; + Ok(BlobBackend::new( + self.storage.clone(), + scope.0.clone(), + runtime, + self.policy.deadline, + ) + .cancelled_by(token.clone()) + .tracked_by(self.tracker.token())) + } + + /// Runs the task on a blocking thread that the tracker counts, and classifies its error. + async fn blocking( + &self, + operation: Operation, + task: impl FnOnce() -> anyhow::Result + Send + 'static, + ) -> Result { + let tracked = self.tracker.token(); + run_blocking(move || { + let _tracked = tracked; + task() + }) + .await + .map_err(|error| classify(operation, error)) + } + + fn files(&self, scope: &SnapshotScope) -> SnapshotFiles { + SnapshotFiles { + storage: self.storage.clone(), + namespace: scope.0.clone(), + deadline: self.policy.deadline, + } + } + + /// Adds the freed bytes to the ledger of the scope, and prunes the repository when a prune is + /// due. The ledger keeps the freed bytes before the prune starts, so a delete that runs again + /// after a failed prune prunes again. + async fn prune_when_due( + &self, + scope: &SnapshotScope, + token: &CancellationToken, + freed: u64, + ) -> Result<(), SnapshotStoreError> { + let deadline = self.policy.deadline; + let ledger = read_ledger(&*self.storage, &scope.0, deadline) + .await + .map_err(storage_failure)? + .with_deleted(freed); + if freed > 0 { + write_ledger(&*self.storage, &scope.0, deadline, &ledger) + .await + .map_err(storage_failure)?; + } + let now = Timestamp::now_utc(); + if !prune_due( + &ledger, + now, + self.policy.prune_threshold, + self.policy.prune.keep_delete, + ) { + return Ok(()); + } + let backend = Arc::new(self.backend(scope, token)?); + let key = self.key.clone(); + let settings = self.policy.prune; + let report = self + .blocking(Operation::Repository, move || { + prune(backend, &key, &settings) + }) + .await?; + let marked_packs = report.is_some_and(|report| { + report.packs_unused + report.packs_repacked + report.marked_packs_kept > 0 + }); + write_ledger( + &*self.storage, + &scope.0, + deadline, + &PruneLedger::after_prune(now, marked_packs), + ) + .await + .map_err(storage_failure) + } +} + +#[async_trait] +impl FilesystemSnapshotStore for RusticSnapshotStore { + async fn save( + &self, + scope: &SnapshotScope, + name: &SnapshotName, + tree: &Path, + ) -> Result { + let (token, _guard) = self.start()?; + check_tree(tree).await?; + let stage = Arc::new(SnapshotStage::default()); + let backend = Arc::new(self.backend(scope, &token)?.staging_in(stage.clone())); + let key = self.key.clone(); + let policy = self.policy; + let name = name.clone(); + let tree: Box = tree.into(); + let staged = self + .blocking(Operation::Save, move || { + stage_save(backend, &stage, &key, &policy, &name, &tree) + }) + .await?; + let (staged, info) = staged.ok_or(SnapshotStoreError::AlreadyExists)?; + publish(&self.files(scope), &staged, &self.tracker) + .await + .map_err(storage_failure)?; + Ok(info) + } + + async fn restore( + &self, + scope: &SnapshotScope, + name: &SnapshotName, + into: &Path, + ) -> Result { + let (token, _guard) = self.start()?; + check_destination(into).await?; + let backend = Arc::new(self.backend(scope, &token)?); + let key = self.key.clone(); + let options = store_restore_options(&self.policy); + let name = name.clone(); + let into: Box = into.into(); + self.blocking(Operation::Restore, move || { + let Some(repository) = open_existing(backend, &key)? else { + return Ok(Lookup::Missing); + }; + match lookup(scope_snapshots(&repository)?, &name) { + Lookup::Found(snapshot, info) => { + restore_snapshot(repository, &snapshot, &into, &options)?; + Ok(Lookup::Found(snapshot, info)) + } + other => Ok(other), + } + }) + .await? + .into_info()? + .ok_or(SnapshotStoreError::NotFound) + } + + async fn stat( + &self, + scope: &SnapshotScope, + name: &SnapshotName, + ) -> Result, SnapshotStoreError> { + let (token, _guard) = self.start()?; + let backend = Arc::new(self.backend(scope, &token)?); + let key = self.key.clone(); + let name = name.clone(); + self.blocking(Operation::Repository, move || { + Ok(match open_existing(backend, &key)? { + Some(repository) => lookup(scope_snapshots(&repository)?, &name), + None => Lookup::Missing, + }) + }) + .await? + .into_info() + } + + async fn list( + &self, + scope: &SnapshotScope, + ) -> Result, SnapshotStoreError> { + let (token, _guard) = self.start()?; + let backend = Arc::new(self.backend(scope, &token)?); + let key = self.key.clone(); + self.blocking(Operation::Repository, move || { + Ok(match open_existing(backend, &key)? { + Some(repository) => newest_first(listed(scope_snapshots(&repository)?)), + None => Box::default(), + }) + }) + .await + } + + async fn delete( + &self, + scope: &SnapshotScope, + name: &SnapshotName, + ) -> Result<(), SnapshotStoreError> { + let (token, _guard) = self.start()?; + let backend = Arc::new(self.backend(scope, &token)?); + let key = self.key.clone(); + let name = name.clone(); + let freed = self + .blocking(Operation::Repository, move || { + let Some(repository) = open_existing(backend, &key)? else { + return Ok(None); + }; + let named = scope_snapshots(&repository)? + .readable + .into_iter() + .filter(|snapshot| snapshot.label == name.as_str()) + .collect::>(); + let ids = named + .iter() + .map(|snapshot| snapshot.id) + .collect::>(); + repository.delete_snapshots(&ids)?; + Ok(Some(named.iter().map(added_packed_bytes).sum::())) + }) + .await?; + match freed { + Some(freed) => self.prune_when_due(scope, &token, freed).await, + None => Ok(()), + } + } + + async fn delete_scope(&self, scope: &SnapshotScope) -> Result<(), SnapshotStoreError> { + let _operation = self.start()?; + delete_scope(&*self.storage, &scope.0, self.policy.deadline) + .await + .map_err(storage_failure) + } + + async fn copy_scope( + &self, + from: &SnapshotScope, + to: &SnapshotScope, + ) -> Result<(), SnapshotStoreError> { + let _operation = self.start()?; + copy_scope(&*self.storage, &from.0, &to.0, self.policy.deadline) + .await + .map_err(storage_failure) + } +} + +/// What the tree of a snapshot holds. The store keeps it as the description of the snapshot, +/// because rustic counts a symlink as a file. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] +struct TreeContent { + files: u64, + bytes: u64, +} + +/// The snapshot files of a scope that rustic could read, and whether a file failed its integrity +/// check. A file that a delete removed after the listing is in neither. +struct ScopeSnapshots { + readable: Vec, + unreadable: bool, +} + +/// What a lookup of a name found. +enum Lookup { + Found(Box, SnapshotInfo), + Missing, + Corrupt(anyhow::Error), +} + +impl Lookup { + fn into_info(self) -> Result, SnapshotStoreError> { + match self { + Self::Found(_, info) => Ok(Some(info)), + Self::Missing => Ok(None), + Self::Corrupt(error) => Err(SnapshotStoreError::Corrupt(error)), + } + } +} + +/// Checks that the tree of a save is a directory at an absolute path. +async fn check_tree(tree: &Path) -> Result<(), SnapshotStoreError> { + let metadata = tokio::fs::metadata(tree) + .await + .map_err(SnapshotStoreError::Source)?; + if !tree.is_absolute() || !metadata.is_dir() { + return Err(SnapshotStoreError::Source(std::io::Error::new( + std::io::ErrorKind::NotADirectory, + format!( + "the tree {} is not a directory at an absolute path", + tree.display() + ), + ))); + } + Ok(()) +} + +/// Checks that the directory of a restore is an empty directory with a UTF-8 path. +async fn check_destination(into: &Path) -> Result<(), SnapshotStoreError> { + let refused = |kind, reason: &str| { + SnapshotStoreError::Destination(std::io::Error::new( + kind, + format!("the directory {} {reason}", into.display()), + )) + }; + let metadata = tokio::fs::metadata(into) + .await + .map_err(SnapshotStoreError::Destination)?; + if !metadata.is_dir() { + return Err(refused( + std::io::ErrorKind::NotADirectory, + "is not a directory", + )); + } + if into.to_str().is_none() { + return Err(refused( + std::io::ErrorKind::InvalidInput, + "does not have a UTF-8 path", + )); + } + let first = tokio::fs::read_dir(into) + .await + .map_err(SnapshotStoreError::Destination)? + .next_entry() + .await + .map_err(SnapshotStoreError::Destination)?; + match first { + Some(_) => Err(refused( + std::io::ErrorKind::DirectoryNotEmpty, + "is not empty", + )), + None => Ok(()), + } +} + +/// Backs up the tree with the snapshot file in the stage, and gives the staged file with the info +/// of the snapshot. The result is `None` when a snapshot of the scope already has the name. +fn stage_save( + backend: Arc, + stage: &SnapshotStage, + key: &RepositoryKey, + policy: &StorePolicy, + name: &SnapshotName, + tree: &Path, +) -> anyhow::Result> { + let (repository, _) = open_or_create(backend, key, &policy.repository)?; + let before = scope_snapshots(&repository)?; + if has_name(&before, name) { + return Ok(None); + } + let newest = before + .readable + .iter() + .filter_map(snapshot_info) + .map(|info| info.created_at) + .max(); + let created_at = snapshot_time(whole_millis_from(Timestamp::now_utc()), newest); + // rustic strips the root from the path of each entry. The canonical root is the path that the + // walk of rustic gives, also when the caller gives a path through a symlink such as + // `/proc/self/fd/N`. + let tree = std::fs::canonicalize(tree)?; + let content = tree_content(&tree)?; + let snapshot = SnapshotOptions::default() + .label(name.as_str().to_string()) + .time(snapshot_zoned(created_at)?) + .description(serde_json::to_string(&content)?) + .to_snapshot()?; + let repository = repository.to_indexed_ids()?; + repository.backup( + &store_backup_options(policy), + &PathList::from_iter(Some(tree)), + snapshot, + )?; + let staged = stage + .take() + .context("the backup gave no snapshot file to the stage")?; + if has_name(&scope_snapshots(&repository)?, name) { + return Ok(None); + } + Ok(Some(( + staged, + SnapshotInfo { + created_at, + files: content.files, + bytes: content.bytes, + }, + ))) +} + +/// Reads each snapshot file of the repository. +/// +/// A failed storage call fails the read. A file that the storage no longer holds is left out, +/// because a delete removed it after the listing. Each other failure counts as a file that failed +/// its integrity check. +fn scope_snapshots(repository: &RusticRepository) -> anyhow::Result { + repository.list::()?.try_fold( + ScopeSnapshots { + readable: Vec::new(), + unreadable: false, + }, + |mut found, id| match repository.get_file::(&id) { + Ok(mut snapshot) => { + snapshot.id = id; + found.readable.push(snapshot); + Ok(found) + } + Err(error) if is_storage_failure(&*error) => Err(anyhow::Error::from(error)), + Err(error) if is_file_missing(&*error) => Ok(found), + Err(_) => Ok(ScopeSnapshots { + unreadable: true, + ..found + }), + }, + ) +} + +fn has_name(found: &ScopeSnapshots, name: &SnapshotName) -> bool { + found + .readable + .iter() + .any(|snapshot| snapshot.label == name.as_str()) +} + +/// Finds the snapshot with the name. Of the snapshot files with the name, the one with the least +/// time and id wins. When no file has the name and a file failed its integrity check, the result is +/// `Corrupt`, because that file can have the name. +fn lookup(found: ScopeSnapshots, name: &SnapshotName) -> Lookup { + let winner = found + .readable + .into_iter() + .filter(|snapshot| snapshot.label == name.as_str()) + .min_by_key(|snapshot| (snapshot.time.timestamp(), snapshot.id)); + match (winner, found.unreadable) { + (Some(snapshot), _) => match snapshot_info(&snapshot) { + Some(info) => Lookup::Found(Box::new(snapshot), info), + None => Lookup::Corrupt(anyhow::anyhow!( + "the snapshot {} does not describe its tree", + snapshot.id + )), + }, + (None, true) => Lookup::Corrupt(anyhow::anyhow!( + "a snapshot file of the scope failed its integrity check" + )), + (None, false) => Lookup::Missing, + } +} + +/// Gives the name and the info of each snapshot whose label is a name and whose description +/// parses. Of the files with one name, only the one that [`lookup`] takes stays. +fn listed(found: ScopeSnapshots) -> Vec<(SnapshotName, SnapshotInfo)> { + let mut snapshots = found.readable; + snapshots.sort_by_key(|snapshot| { + ( + snapshot.label.clone(), + snapshot.time.timestamp(), + snapshot.id, + ) + }); + snapshots.dedup_by(|later, first| later.label == first.label); + snapshots + .iter() + .filter_map(|snapshot| { + SnapshotName::new(&snapshot.label) + .ok() + .zip(snapshot_info(snapshot)) + }) + .collect() +} + +/// Gives the info of a snapshot from its time and its description. +fn snapshot_info(snapshot: &SnapshotFile) -> Option { + let content = serde_json::from_str::(snapshot.description.as_deref()?).ok()?; + let created_at = u64::try_from(snapshot.time.timestamp().as_millisecond()).ok()?; + Some(SnapshotInfo { + created_at: Timestamp::from(created_at), + files: content.files, + bytes: content.bytes, + }) +} + +/// Gives the packed bytes that the save of the snapshot added to the repository. +fn added_packed_bytes(snapshot: &SnapshotFile) -> u64 { + snapshot + .summary + .as_ref() + .map_or(0, |summary| summary.data_added_packed) +} + +/// Gives the first time in whole milliseconds that is not before the time. A snapshot keeps its +/// time in milliseconds, so the time of a save is not before the call. +fn whole_millis_from(time: Timestamp) -> Timestamp { + let truncated = Timestamp::from(time.to_millis()); + if truncated < time { + Timestamp::from(time.to_millis().saturating_add(1)) + } else { + truncated + } +} + +/// Gives the time as the time of a snapshot, in UTC. +fn snapshot_zoned(time: Timestamp) -> anyhow::Result { + Ok(SnapshotTime::from_millisecond(i64::try_from(time.to_millis())?)?.to_zoned(TimeZone::UTC)) +} + +/// Counts the names of the regular files below the root, and the sum of their sizes. The walk +/// reads the metadata of each entry and does not follow a symlink. +fn tree_content(root: &Path) -> std::io::Result { + fn walk(directory: &Path, content: TreeContent) -> std::io::Result { + std::fs::read_dir(directory)?.try_fold(content, |content, entry| { + let entry = entry?; + let kind = entry.file_type()?; + if kind.is_dir() { + walk(&entry.path(), content) + } else if kind.is_file() { + Ok(TreeContent { + files: content.files + 1, + bytes: content.bytes + entry.metadata()?.len(), + }) + } else { + Ok(content) + } + }) + } + walk(root, TreeContent { files: 0, bytes: 0 }) +} + +#[cfg(test)] +mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs new file mode 100644 index 0000000000..c6d7c85b16 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -0,0 +1,1116 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The rustic store through the interface of the store, on the in-memory blob storage. +//! +//! The contract suite runs on the store with the policy of the configuration. The other tests +//! give the store a short or a long deadline and a prune policy that the test controls. + +use super::super::prune::{PruneLedger, read_ledger}; +use super::super::scripted::{Script, ScriptedBlobStorage}; +use super::super::{PruneSettings, RepackLimits, RepositoryKey, open_existing}; +use super::{ + RusticSnapshotStore, StorePolicy, scope_snapshots, store_backup_options, store_restore_options, + whole_millis_from, +}; +use crate::filesystem_snapshot::contract_tests::fixture::{ + Listed, Scratch, Spec, fixture, listing, write_tree, +}; +use crate::filesystem_snapshot::contract_tests::{self, OpenStore, new_scope}; +use crate::filesystem_snapshot::{ + FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, +}; +use crate::services::golem_config::FilesystemSnapshotStoreConfig; +use futures::{FutureExt, StreamExt}; +use golem_service_base::storage::blob::memory::InMemoryBlobStorage; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use pretty_assertions::assert_eq; +use std::future::Future; +use std::num::NonZeroUsize; +use std::path::Path; +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; +use std::time::Duration; +use test_r::core::DynamicTestRegistration; +use test_r::{test, test_gen}; + +/// The longest time that a test waits for an operation or for the work of a store to end. +const LIMIT: Duration = Duration::from_secs(10); + +/// A deadline that no call of these tests reaches, so a held call ends only by a cancel. +const LONG_DEADLINE: Duration = Duration::from_secs(60); + +const KEY: &str = "000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f\ + 202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f"; + +fn config() -> FilesystemSnapshotStoreConfig { + FilesystemSnapshotStoreConfig::new(KEY, Duration::from_secs(30), 4, 3).unwrap() +} + +fn key() -> RepositoryKey { + RepositoryKey::new(*config().repository_key().bytes()) +} + +/// The policy of the configuration, with the deadline, the prune threshold and the grace period +/// of the test. +fn policy(deadline: Duration, prune_threshold: u64, grace: Duration) -> StorePolicy { + StorePolicy { + deadline, + prune: PruneSettings { + keep_delete: grace, + ..StorePolicy::from_config(&config()).prune + }, + prune_threshold, + ..StorePolicy::from_config(&config()) + } +} + +fn store(storage: Arc, policy: StorePolicy) -> Arc { + Arc::new(RusticSnapshotStore::with_policy(storage, key(), policy)) +} + +fn name(text: &str) -> SnapshotName { + SnapshotName::new(text).unwrap() +} + +/// Writes a tree of one file with the content into a new directory, and gives the directory. +fn one_file_tree(content: &str) -> Scratch { + let tree = Scratch::new(); + write_tree( + tree.path(), + &[( + "file.txt", + Spec::File { + content: Box::from(content.as_bytes()), + mode: 0o644, + }, + )], + ); + tree +} + +fn fixture_tree() -> Scratch { + let tree = Scratch::new(); + write_tree(tree.path(), &fixture()); + tree +} + +/// Restores the name into a new directory, and gives the listing of the directory. +async fn restored_listing( + store: &RusticSnapshotStore, + scope: &SnapshotScope, + name: &SnapshotName, +) -> Result, SnapshotStoreError> { + let into = Scratch::new(); + store.restore(scope, name, into.path()).await?; + Ok(listing(into.path())) +} + +async fn listed_names(store: &RusticSnapshotStore, scope: &SnapshotScope) -> Vec { + store + .list(scope) + .await + .unwrap() + .iter() + .map(|(name, _)| name.to_string()) + .collect() +} + +/// Gives the path of each blob of the namespace whose path starts with the prefix. +async fn blobs( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, + prefix: &str, +) -> Vec { + let mut paths = storage + .list_blobs_below("test", "test", namespace.clone(), Path::new("")) + .await + .unwrap() + .iter() + .map(|blob| blob.path.display().to_string()) + .filter(|path| path.starts_with(prefix)) + .collect::>(); + paths.sort(); + paths +} + +async fn ledger(storage: &dyn BlobStorage, scope: &SnapshotScope) -> PruneLedger { + read_ledger(storage, &scope.0, Duration::from_secs(2)) + .await + .unwrap() +} + +/// Waits until the condition holds, or until the limit ends. Gives whether the condition holds. +async fn eventually(condition: impl Fn() -> bool) -> bool { + tokio::time::timeout(LIMIT, async { + futures::stream::repeat(()) + .then(|()| tokio::time::sleep(Duration::from_millis(5))) + .take_while(|()| std::future::ready(!condition())) + .for_each(|()| std::future::ready(())) + .await + }) + .await + .is_ok() +} + +/// Runs the operation until the calls of the storage fulfil the condition, and then drops it. +/// Gives the output of the operation when it ends first. +async fn drop_when( + storage: &ScriptedBlobStorage, + condition: impl Fn(&[(&'static str, String)]) -> bool, + operation: impl Future, +) -> Option { + tokio::select! { + output = operation => Some(output), + _ = eventually(|| condition(&storage.calls())) => None, + } +} + +fn is_storage(error: &SnapshotStoreError, expected_retryable: bool) -> bool { + matches!(error, SnapshotStoreError::Storage { retryable, .. } if *retryable == expected_retryable) +} + +#[test_gen] +fn rustic_store_keeps_the_contract(r: &mut DynamicTestRegistration) { + contract_tests::register(r, || { + let storage: Arc = Arc::new(InMemoryBlobStorage::new()); + let open: OpenStore = Arc::new(move || { + Arc::new(RusticSnapshotStore::new(storage.clone(), &config())) + as Arc + }); + open + }); +} + +#[test] +fn the_policy_takes_the_configured_values_and_the_options_are_strict() { + let policy = StorePolicy::from_config(&config()); + let backup = store_backup_options(&policy); + let restore = store_restore_options(&policy); + + assert_eq!( + ( + policy.deadline, + policy.save.threads.map(NonZeroUsize::get), + policy.restore_reader_threads.get(), + policy.prune.keep_delete, + policy.prune.fast_repack, + policy.prune.repack, + policy.prune_threshold, + ), + ( + Duration::from_secs(30), + Some(3), + 4, + Duration::from_secs(3600), + false, + RepackLimits::Rustic, + 64 * 1024 * 1024, + ) + ); + assert_eq!( + ( + backup.fail_on_read_error, + backup.threads.map(NonZeroUsize::get), + backup.as_path.as_deref().map(Path::to_path_buf), + restore.fail_on_metadata_error, + restore.no_ownership, + restore.reader_threads.map(NonZeroUsize::get), + ), + ( + true, + Some(3), + Some(Path::new("/").to_path_buf()), + true, + true, + Some(4), + ) + ); +} + +#[test] +fn a_save_time_is_the_first_whole_millisecond_that_is_not_before_the_call() { + let now = golem_common::model::Timestamp::now_utc(); + let whole = golem_common::model::Timestamp::from(5_000); + let rounded = whole_millis_from(now); + + assert_eq!( + ( + whole_millis_from(whole), + rounded >= now, + rounded.to_millis() - now.to_millis() <= 1, + golem_common::model::Timestamp::from(rounded.to_millis()) == rounded, + ), + (whole, true, true, true) + ); +} + +#[cfg(target_os = "linux")] +#[test] +async fn a_tree_saved_through_a_proc_self_fd_path_is_stored_below_the_root() { + use std::os::fd::AsRawFd; + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + let directory = std::fs::File::open(tree.path()).unwrap(); + let through_fd = format!("/proc/self/fd/{}", directory.as_raw_fd()); + + store + .save(&scope, &name("p-fd"), Path::new(&through_fd)) + .await + .unwrap(); + let namespace = scope.0.clone(); + let paths = tokio::task::spawn_blocking(move || { + let backend = super::super::backend::BlobBackend::new( + storage, + namespace, + tokio::runtime::Handle::current(), + LONG_DEADLINE, + ); + let repository = open_existing(Arc::new(backend), &key()).unwrap().unwrap(); + scope_snapshots(&repository) + .unwrap() + .readable + .iter() + .map(|snapshot| snapshot.paths.to_string()) + .collect::>() + }) + .await + .unwrap(); + + assert_eq!( + ( + paths, + restored_listing(&store, &scope, &name("p-fd")) + .await + .unwrap() + ), + (vec!["/".to_string()], listing(tree.path())) + ); +} + +#[test] +async fn a_save_whose_index_write_fails_publishes_nothing_and_leaves_the_name_free() { + let refuse = Arc::new(AtomicBool::new(true)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) && op_label == "write" && path.starts_with("index") { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = fixture_tree(); + + let failed = store.save(&scope, &name("p-1"), tree.path()).await; + let stat = store.stat(&scope, &name("p-1")).await.unwrap(); + let names = listed_names(&store, &scope).await; + let restore = restored_listing(&store, &scope, &name("p-1")).await; + refuse.store(false, Ordering::SeqCst); + let saved_again = store.save(&scope, &name("p-1"), tree.path()).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!( + matches!(restore, Err(SnapshotStoreError::NotFound)), + "{restore:?}" + ); + assert_eq!( + ( + stat, + names, + saved_again.is_ok(), + restored_listing(&store, &scope, &name("p-1")) + .await + .unwrap() + ), + (None, Vec::::new(), true, listing(tree.path())) + ); +} + +#[test] +async fn a_publish_that_reaches_the_deadline_and_lands_late_publishes_nothing() { + let hang = Arc::new(AtomicBool::new(true)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let hang = hang.clone(); + move |op_label, _| { + if hang.load(Ordering::SeqCst) && op_label == "publish" { + Script::NeverAnswer + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(Duration::from_secs(1), u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("late"); + + let failed = store.save(&scope, &name("p-late"), tree.path()).await; + hang.store(false, Ordering::SeqCst); + let stat = store.stat(&scope, &name("p-late")).await.unwrap(); + let names = listed_names(&store, &scope).await; + let saved_again = store.save(&scope, &name("p-late"), tree.path()).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert_eq!( + ( + stat, + names, + saved_again.is_ok(), + storage + .calls() + .iter() + .filter(|(op_label, _)| *op_label == "retract") + .count() + ), + (None, Vec::::new(), true, 1) + ); +} + +#[test] +async fn a_second_save_of_an_unchanged_tree_writes_no_pack() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + + store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + let first = storage.calls().len(); + store.save(&scope, &name("p-2"), tree.path()).await.unwrap(); + let writes = storage.calls()[first..] + .iter() + .filter(|(op_label, _)| *op_label == "write" || *op_label == "publish") + .map(|(op_label, path)| { + ( + *op_label, + path.split('/').next().unwrap_or_default().to_string(), + ) + }) + .collect::>(); + + let restored = restored_listing(&store, &scope, &name("p-2")).await.ok(); + + assert_eq!( + (writes, restored), + ( + vec![("publish", "snapshots".to_string())], + Some(listing(tree.path())) + ) + ); +} + +#[test] +async fn two_stores_that_create_one_repository_at_the_same_time_both_save() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "write" && path == Path::new("config") { + Script::WaitForGate + } else { + Script::Pass + } + }); + let policy = policy(LONG_DEADLINE, u64::MAX, Duration::ZERO); + let (first, second) = ( + store(storage.clone(), policy), + store(storage.clone(), policy), + ); + let scope = new_scope(); + let (first_tree, second_tree) = (one_file_tree("first"), one_file_tree("second")); + let config_writes = || { + storage + .calls() + .iter() + .filter(|(op_label, path)| *op_label == "write" && path == "config") + .count() + }; + + let (first_name, second_name) = (name("p-first"), name("p-second")); + let (first_saved, second_saved, both_waited) = tokio::join!( + first.save(&scope, &first_name, first_tree.path()), + second.save(&scope, &second_name, second_tree.path()), + async { + let both = eventually(|| config_writes() == 2).await; + storage.open_gate(); + both + } + ); + let mut names = listed_names(&first, &scope).await; + names.sort(); + + assert_eq!( + ( + both_waited, + first_saved.map(|_| ()).map_err(|error| error.to_string()), + second_saved.map(|_| ()).map_err(|error| error.to_string()), + names, + restored_listing(&second, &scope, &name("p-first")) + .await + .unwrap(), + restored_listing(&first, &scope, &name("p-second")) + .await + .unwrap(), + ), + ( + true, + Ok(()), + Ok(()), + vec!["p-first".to_string(), "p-second".to_string()], + listing(first_tree.path()), + listing(second_tree.path()), + ) + ); +} + +#[test] +async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { + // The first index write after the arm waits at the gate. That is the index write of the + // second save, so its packs are in no index while the delete prunes. + let hold_next_index = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let hold_next_index = hold_next_index.clone(); + move |op_label, path| { + if op_label == "write" + && path.starts_with("index") + && hold_next_index.swap(false, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let scope = new_scope(); + let (old_tree, new_tree) = (one_file_tree("old"), fixture_tree()); + store + .save(&scope, &name("p-old"), old_tree.path()) + .await + .unwrap(); + hold_next_index.store(true, Ordering::SeqCst); + let index_writes = || { + storage + .calls() + .iter() + .filter(|(op_label, path)| *op_label == "write" && path.starts_with("index")) + .count() + }; + let before = index_writes(); + + let saving = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + let path = new_tree.path().to_path_buf(); + async move { store.save(&scope, &name("p-new"), &path).await } + }); + let held = eventually(|| index_writes() > before).await; + let deleted = store.delete(&scope, &name("p-old")).await; + let pruned_while_held = ledger(&*storage, &scope).await.last_prune_millis.is_some(); + storage.open_gate(); + let saved = saving.await.unwrap(); + let pruned_again = store.delete(&scope, &name("p-none")).await; + + assert_eq!( + ( + held, + deleted.is_ok(), + pruned_while_held, + saved.is_ok(), + pruned_again.is_ok(), + restored_listing(&store, &scope, &name("p-new")).await.ok(), + ), + (true, true, true, true, true, Some(listing(new_tree.path()))) + ); +} + +#[test] +async fn a_restore_whose_pack_reads_fail_gives_a_retryable_storage_error() { + let refuse = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) + && matches!(op_label, "read" | "read_range") + && path.starts_with("data") + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = fixture_tree(); + store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + refuse.store(true, Ordering::SeqCst); + + let restored = restored_listing(&store, &scope, &name("p-1")).await; + + assert!( + restored + .as_ref() + .is_err_and(|error| is_storage(error, true)), + "{restored:?}" + ); +} + +#[cfg(unix)] +#[test] +async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_denied() { + use std::os::unix::fs::PermissionsExt; + // SAFETY: `geteuid` has no preconditions. + if unsafe { libc::geteuid() } == 0 { + return; + } + let store = store( + Arc::new(InMemoryBlobStorage::new()), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("readable"); + let locked = tree.path().join("locked.txt"); + std::fs::write(&locked, b"locked").unwrap(); + std::fs::set_permissions(&locked, std::fs::Permissions::from_mode(0o000)).unwrap(); + + let saved = store.save(&scope, &name("p-locked"), tree.path()).await; + + assert!( + matches!(&saved, Err(SnapshotStoreError::Source(error)) if error.kind() == std::io::ErrorKind::PermissionDenied), + "{saved:?}" + ); +} + +#[cfg(target_os = "linux")] +#[test] +async fn a_restore_that_cannot_set_an_extended_attribute_gives_destination() { + // A file on tmpfs takes a user attribute of 6,000 bytes. A file on ext4 with 4 KiB blocks + // does not, so the restore cannot set it. The test checks nothing on a host where the source + // does not take the attribute or where the destination takes it. + let value = vec![b'a'; 6000]; + let set = |path: &Path| xattr_set(path, "user.golem-test", &value); + let Ok(source) = tempfile::tempdir_in("/dev/shm") else { + return; + }; + let file = source.path().join("file.txt"); + std::fs::write(&file, b"content").unwrap(); + let probe = Scratch::new(); + let probe_file = probe.path().join("probe"); + std::fs::write(&probe_file, b"probe").unwrap(); + if set(&file).is_err() || set(&probe_file).is_ok() { + return; + } + let store = store( + Arc::new(InMemoryBlobStorage::new()), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + store + .save(&scope, &name("p-xattr"), source.path()) + .await + .unwrap(); + + let restored = restored_listing(&store, &scope, &name("p-xattr")).await; + + assert!( + matches!(&restored, Err(SnapshotStoreError::Destination(_))), + "{restored:?}" + ); +} + +#[cfg(target_os = "linux")] +fn xattr_set(path: &Path, name: &str, value: &[u8]) -> std::io::Result<()> { + use std::os::unix::ffi::OsStrExt; + let path = std::ffi::CString::new(path.as_os_str().as_bytes())?; + let name = std::ffi::CString::new(name)?; + // SAFETY: the path and the name are NUL-terminated strings, and the value lives for the call. + let result = unsafe { + libc::setxattr( + path.as_ptr(), + name.as_ptr(), + value.as_ptr().cast(), + value.len(), + 0, + ) + }; + if result == 0 { + Ok(()) + } else { + Err(std::io::Error::last_os_error()) + } +} + +#[test] +async fn a_snapshot_file_that_fails_its_check_is_left_out_of_list_and_makes_an_unknown_name_corrupt() + { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path()) + .await + .unwrap(); + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new(&format!("snapshots/{}", "ab".repeat(32))), + b"not a snapshot", + ) + .await + .unwrap(); + + let names = listed_names(&store, &scope).await; + let kept = store.stat(&scope, &name("p-kept")).await; + let unknown = store.stat(&scope, &name("p-unknown")).await; + let restore_unknown = restored_listing(&store, &scope, &name("p-unknown")).await; + + assert!( + matches!(unknown, Err(SnapshotStoreError::Corrupt(_))), + "{unknown:?}" + ); + assert!( + matches!(restore_unknown, Err(SnapshotStoreError::Corrupt(_))), + "{restore_unknown:?}" + ); + assert_eq!( + (names, kept.map(|info| info.is_some()).ok()), + (vec!["p-kept".to_string()], Some(true)) + ); +} + +#[test] +async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_period() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path()) + .await + .unwrap(); + let packs_before = blobs(&*storage, &scope.0, "data/").await; + + store.delete(&scope, &name("p-deleted")).await.unwrap(); + let after_first = ledger(&*storage, &scope).await; + store.delete(&scope, &name("p-none")).await.unwrap(); + let packs_after = blobs(&*storage, &scope.0, "data/").await; + + assert_eq!( + ( + after_first.freed_bytes, + after_first.last_prune_millis.is_some(), + after_first.awaiting_removal, + packs_after.len() < packs_before.len(), + packs_after.iter().all(|pack| packs_before.contains(pack)), + restored_listing(&store, &scope, &name("p-kept")).await.ok(), + ), + (0, true, true, true, true, Some(listing(kept_tree.path()))) + ); +} + +#[test] +async fn a_delete_below_the_threshold_does_not_prune() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), one_file_tree("kept")); + store + .save(&scope, &name("p-deleted"), deleted_tree.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path()) + .await + .unwrap(); + let packs_before = blobs(&*storage, &scope.0, "data/").await; + + store.delete(&scope, &name("p-deleted")).await.unwrap(); + let after = ledger(&*storage, &scope).await; + + assert_eq!( + ( + after.freed_bytes > 0, + after.last_prune_millis, + blobs(&*storage, &scope.0, "data/").await, + ), + (true, None, packs_before) + ); +} + +#[test] +async fn no_second_prune_runs_within_the_grace_period() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, 1, Duration::from_secs(3600)), + ); + let scope = new_scope(); + let trees = [one_file_tree("a"), one_file_tree("b"), one_file_tree("c")]; + futures::stream::iter(["p-a", "p-b", "p-c"].into_iter().zip(&trees)) + .for_each(|(text, tree)| { + let store = store.clone(); + let scope = scope.clone(); + async move { + store.save(&scope, &name(text), tree.path()).await.unwrap(); + } + }) + .await; + + store.delete(&scope, &name("p-a")).await.unwrap(); + let after_first = ledger(&*storage, &scope).await; + store.delete(&scope, &name("p-b")).await.unwrap(); + let after_second = ledger(&*storage, &scope).await; + + assert_eq!( + ( + after_first.last_prune_millis.is_some(), + after_second.last_prune_millis == after_first.last_prune_millis, + after_second.freed_bytes > 0, + ), + (true, true, true) + ); +} + +#[test] +async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { + let refuse = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) + && matches!(op_label, "read" | "read_range") + && path.starts_with("data") + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path()) + .await + .unwrap(); + refuse.store(true, Ordering::SeqCst); + + let failed = store.delete(&scope, &name("p-deleted")).await; + let after_failure = ledger(&*storage, &scope).await; + refuse.store(false, Ordering::SeqCst); + let retried = store.delete(&scope, &name("p-deleted")).await; + let after_retry = ledger(&*storage, &scope).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert_eq!( + ( + after_failure.freed_bytes > 0, + after_failure.last_prune_millis, + retried.is_ok(), + after_retry.freed_bytes, + after_retry.last_prune_millis.is_some(), + ), + (true, None, true, 0, true) + ); +} + +#[test] +async fn a_deleted_scope_holds_no_blob() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let scope = new_scope(); + let (first, second) = (one_file_tree("first"), one_file_tree("second")); + store + .save(&scope, &name("p-1"), first.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), second.path()) + .await + .unwrap(); + store.delete(&scope, &name("p-1")).await.unwrap(); + let before = blobs(&*storage, &scope.0, "").await; + + store.delete_scope(&scope).await.unwrap(); + + assert_eq!( + ( + before.contains(&"golem/prune-ledger".to_string()), + blobs(&*storage, &scope.0, "").await + ), + (true, Vec::::new()) + ); +} + +#[test] +async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_name_free() { + // The first save counts the calls of a save. Each later round holds one of these calls: the + // call reaches the storage and never answers, as a write that S3 received and completes after + // the caller left. The round drops the save there, and waits until the store has no work. + let tree = one_file_tree("dropped"); + let other = one_file_tree("saved later"); + let counted = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + store( + counted.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ) + .save(&new_scope(), &name("p-dropped"), tree.path()) + .await + .unwrap(); + let calls = counted.calls().len(); + + let rounds = futures::stream::iter(1..=calls) + .then(|held| { + let (tree, other) = (tree.path().to_path_buf(), other.path().to_path_buf()); + async move { + let inner = Arc::new(InMemoryBlobStorage::new()); + let seen = Arc::new(AtomicUsize::new(0)); + let storage = ScriptedBlobStorage::new(inner.clone(), move |_, _| { + if seen.fetch_add(1, Ordering::SeqCst) + 1 == held { + Script::NeverAnswer + } else { + Script::Pass + } + }); + let policy = policy(LONG_DEADLINE, u64::MAX, Duration::ZERO); + let dropping = store(storage.clone(), policy); + let scope = new_scope(); + let ended = drop_when( + &storage, + |calls| calls.len() >= held, + dropping.save(&scope, &name("p-dropped"), &tree), + ) + .await; + let stopped = tokio::time::timeout(LIMIT, dropping.shut_down()) + .await + .is_ok(); + let later = store(inner, policy); + let stat = later.stat(&scope, &name("p-dropped")).await.ok().flatten(); + let names = listed_names(&later, &scope).await; + let saved_again = later.save(&scope, &name("p-dropped"), &other).await; + let restored = restored_listing(&later, &scope, &name("p-dropped")) + .await + .ok(); + ( + held, + ended.is_none(), + stopped, + stat, + names, + saved_again.is_ok(), + restored == Some(listing(&other)), + ) + } + }) + .collect::>() + .await; + + assert_eq!( + rounds, + (1..=calls) + .map(|held| (held, true, true, None, Vec::new(), true, true)) + .collect::, + Vec, + bool, + bool + )>>() + ); +} + +#[test] +async fn shut_down_ends_running_operations_before_it_returns() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "write" && path.starts_with("data") { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + let saving = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + let path = tree.path().to_path_buf(); + async move { store.save(&scope, &name("p-held"), &path).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, path)| *op_label == "write" && path.starts_with("data")) + }) + .await; + + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + let saved = tokio::time::timeout(LIMIT, saving).await; + let later = store.stat(&scope, &name("p-held")).await; + storage.open_gate(); + + assert!( + matches!(&saved, Ok(Ok(Err(error))) if is_storage(error, true)), + "{saved:?}" + ); + assert!( + later.as_ref().is_err_and(|error| is_storage(error, false)), + "{later:?}" + ); + assert_eq!((held, stopped, store.work_in_flight()), (true, true, 0)); +} + +/// The operation of the store that a test drops. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum Dropped { + Restore, + Stat, + List, + Delete, +} + +#[test] +async fn a_dropped_operation_stops_its_blocking_work() { + let outcomes = futures::stream::iter([ + Dropped::Restore, + Dropped::Stat, + Dropped::List, + Dropped::Delete, + ]) + .then(|dropped| async move { + let held_call = move |op_label: &str, path: &Path| match dropped { + Dropped::Restore => op_label == "read_range" && path.starts_with("data"), + Dropped::Stat | Dropped::List => op_label == "read" && path.starts_with("snapshots"), + Dropped::Delete => op_label == "delete" && path.starts_with("snapshots"), + }; + let hold = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let hold = hold.clone(); + move |op_label, path| { + if held_call(op_label, path) && hold.load(Ordering::SeqCst) { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + hold.store(true, Ordering::SeqCst); + let before = storage.calls().len(); + let reached = |calls: &[(&'static str, String)]| { + calls[before..] + .iter() + .any(|(op_label, path)| held_call(op_label, Path::new(path))) + }; + let into = Scratch::new(); + let ended = match dropped { + Dropped::Restore => { + drop_when( + &storage, + reached, + store.restore(&scope, &name("p-1"), into.path()).map(|_| ()), + ) + .await + } + Dropped::Stat => { + drop_when( + &storage, + reached, + store.stat(&scope, &name("p-1")).map(|_| ()), + ) + .await + } + Dropped::List => drop_when(&storage, reached, store.list(&scope).map(|_| ())).await, + Dropped::Delete => { + drop_when( + &storage, + reached, + store.delete(&scope, &name("p-1")).map(|_| ()), + ) + .await + } + }; + let held = reached(&storage.calls()); + let stopped = eventually(|| store.work_in_flight() == 0).await; + storage.open_gate(); + (dropped, held, ended.is_none(), stopped) + }) + .collect::>() + .await; + + assert_eq!( + outcomes, + [ + Dropped::Restore, + Dropped::Stat, + Dropped::List, + Dropped::Delete + ] + .map(|dropped| (dropped, true, true, true)) + ); +} From bba479a86f0aeeb49b4f1a279b35e03aca96cf1b Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 10:03:26 -0700 Subject: [PATCH 013/126] Decide from a pure function whether a prune leaves marked packs, and sort listed snapshots on borrowed labels --- .../filesystem_snapshot/rustic/scripted.rs | 20 ++++++++++--- .../src/filesystem_snapshot/rustic/store.rs | 30 ++++++++++++------- 2 files changed, 35 insertions(+), 15 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs index 348e5206da..8c02e95e1b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -43,6 +43,9 @@ pub(super) enum Script { NeverAnswer, /// Waits until the test opens the gate of the storage, and then passes the call. WaitForGate, + /// Gives no blob to a read of a whole blob, as a delete after a listing does. Each other call + /// passes. + Vanish, } /// A rule that gives the script of a call from its operation label and its path. @@ -85,16 +88,20 @@ impl ScriptedBlobStorage { .collect() } + fn record(&self, op_label: &'static str, path: &Path) { + self.calls + .lock() + .unwrap_or_else(PoisonError::into_inner) + .push((op_label, path.into())); + } + async fn answer( &self, op_label: &'static str, path: &Path, call: impl Future>, ) -> anyhow::Result { - self.calls - .lock() - .unwrap_or_else(PoisonError::into_inner) - .push((op_label, path.into())); + self.record(op_label, path); match (self.rule)(op_label, path) { Script::Pass => call.await, Script::Refuse => Err(anyhow::anyhow!("the storage refused the call")), @@ -110,6 +117,7 @@ impl ScriptedBlobStorage { self.gate.cancelled().await; call.await } + Script::Vanish => call.await, } } } @@ -129,6 +137,10 @@ impl BlobStorage for ScriptedBlobStorage { namespace: BlobStorageNamespace, path: &Path, ) -> anyhow::Result>> { + if (self.rule)(op_label, path) == Script::Vanish { + self.record(op_label, path); + return Ok(None); + } self.answer( op_label, path, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 705d2de252..59028be87c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -26,8 +26,9 @@ use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; use super::publish::{SnapshotFiles, SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; use super::{ - ChangeDetection, PruneSettings, RepackLimits, RepositoryKey, RepositorySettings, SaveSettings, - backup_options, open_existing, open_or_create, prune, restore_snapshot, run_blocking, + ChangeDetection, PruneReport, PruneSettings, RepackLimits, RepositoryKey, RepositorySettings, + SaveSettings, backup_options, open_existing, open_or_create, prune, restore_snapshot, + run_blocking, }; use crate::filesystem_snapshot::{ FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, @@ -265,9 +266,7 @@ impl RusticSnapshotStore { prune(backend, &key, &settings) }) .await?; - let marked_packs = report.is_some_and(|report| { - report.packs_unused + report.packs_repacked + report.marked_packs_kept > 0 - }); + let marked_packs = report.as_ref().is_some_and(leaves_marked_packs); write_ledger( &*self.storage, &scope.0, @@ -627,12 +626,12 @@ fn lookup(found: ScopeSnapshots, name: &SnapshotName) -> Lookup { /// parses. Of the files with one name, only the one that [`lookup`] takes stays. fn listed(found: ScopeSnapshots) -> Vec<(SnapshotName, SnapshotInfo)> { let mut snapshots = found.readable; - snapshots.sort_by_key(|snapshot| { - ( - snapshot.label.clone(), - snapshot.time.timestamp(), - snapshot.id, - ) + snapshots.sort_by(|left, right| { + (&left.label, left.time.timestamp(), left.id).cmp(&( + &right.label, + right.time.timestamp(), + right.id, + )) }); snapshots.dedup_by(|later, first| later.label == first.label); snapshots @@ -656,6 +655,15 @@ fn snapshot_info(snapshot: &SnapshotFile) -> Option { }) } +/// Tells whether a later prune removes packs that this prune leaves marked. +/// +/// A prune marks each pack that holds only unused blobs, and each pack that it repacks. A pack that +/// an earlier prune marked and whose time to stay is not over stays marked. A pack that no index +/// lists is also marked, but the report does not count it. A later due prune removes that pack. +fn leaves_marked_packs(report: &PruneReport) -> bool { + report.packs_unused > 0 || report.packs_repacked > 0 || report.marked_packs_kept > 0 +} + /// Gives the packed bytes that the save of the snapshot added to the repository. fn added_packed_bytes(snapshot: &SnapshotFile) -> u64 { snapshot From 4384f423e7df85489e4e2e2ca863615291a57415 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 10:03:26 -0700 Subject: [PATCH 014/126] Test the snapshot reads, the ledger, the tree check, the tracker and the restore phases of the rustic store --- .../rustic/backend/tests.rs | 18 ++ .../filesystem_snapshot/rustic/store/tests.rs | 241 +++++++++++++++++- .../src/filesystem_snapshot/rustic/tests.rs | 30 +++ 3 files changed, 286 insertions(+), 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index cb52483c36..e0ef93e606 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -478,6 +478,24 @@ fn a_thread_that_is_not_a_thread_of_the_runtime_can_call_the_backend() { assert_eq!(read.ok().flatten(), Some(Bytes::from_static(b"index"))); } +#[test] +fn a_tracked_backend_counts_in_its_tracker_until_it_drops() { + let fixture = Fixture::new(); + let tracker = tokio_util::task::TaskTracker::new(); + let backend = BlobBackend::new( + fixture.storage.clone(), + fixture.namespace.clone(), + fixture.runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .tracked_by(tracker.token()); + + let while_alive = tracker.len(); + drop(backend); + + assert_eq!((while_alive, tracker.len()), (1, 0)); +} + #[test] fn a_cancelled_backend_makes_no_storage_call() { let runtime = Runtime::new().unwrap(); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index c6d7c85b16..ed851d126b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -19,10 +19,10 @@ use super::super::prune::{PruneLedger, read_ledger}; use super::super::scripted::{Script, ScriptedBlobStorage}; -use super::super::{PruneSettings, RepackLimits, RepositoryKey, open_existing}; +use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; use super::{ - RusticSnapshotStore, StorePolicy, scope_snapshots, store_backup_options, store_restore_options, - whole_millis_from, + RusticSnapshotStore, StorePolicy, leaves_marked_packs, scope_snapshots, store_backup_options, + store_restore_options, whole_millis_from, }; use crate::filesystem_snapshot::contract_tests::fixture::{ Listed, Scratch, Spec, fixture, listing, write_tree, @@ -1114,3 +1114,238 @@ async fn a_dropped_operation_stops_its_blocking_work() { .map(|dropped| (dropped, true, true, true)) ); } + +/// Gives the snapshot files of the scope that the bridge reads, on a blocking thread. +async fn snapshot_files( + storage: Arc, + scope: &SnapshotScope, +) -> Vec { + let namespace = scope.0.clone(); + tokio::task::spawn_blocking(move || { + let backend = super::super::backend::BlobBackend::new( + storage, + namespace, + tokio::runtime::Handle::current(), + LONG_DEADLINE, + ); + let repository = open_existing(Arc::new(backend), &key()).unwrap().unwrap(); + scope_snapshots(&repository).unwrap().readable + }) + .await + .unwrap() +} + +#[test] +async fn a_failed_read_of_a_snapshot_file_fails_stat_and_list_with_a_retryable_storage_error() { + let refuse = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) && op_label == "read" && path.starts_with("snapshots") + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path()) + .await + .unwrap(); + refuse.store(true, Ordering::SeqCst); + + let stat = store.stat(&scope, &name("p-kept")).await; + let list = store.list(&scope).await; + + assert!( + stat.as_ref().is_err_and(|error| is_storage(error, true)), + "{stat:?}" + ); + assert!( + list.as_ref().is_err_and(|error| is_storage(error, true)), + "{list:?}" + ); +} + +#[test] +async fn a_snapshot_file_that_is_gone_after_the_listing_is_left_out() { + let gone = format!("snapshots/{}", "cd".repeat(32)); + let inner = Arc::new(InMemoryBlobStorage::new()); + let storage = ScriptedBlobStorage::new(inner.clone(), { + let gone = gone.clone(); + move |op_label, path| { + if op_label == "read" && path == Path::new(&gone) { + Script::Vanish + } else { + Script::Pass + } + } + }); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path()) + .await + .unwrap(); + inner + .put_raw("test", "test", scope.0.clone(), Path::new(&gone), b"listed") + .await + .unwrap(); + + let unknown = store.stat(&scope, &name("p-unknown")).await; + let names = listed_names(&store, &scope).await; + + assert!(matches!(unknown, Ok(None)), "{unknown:?}"); + assert_eq!(names, vec!["p-kept".to_string()]); +} + +#[test] +async fn a_delete_that_frees_nothing_writes_no_ledger() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path()) + .await + .unwrap(); + + store.delete(&scope, &name("p-unknown")).await.unwrap(); + + assert_eq!( + blobs(&*storage, &scope.0, "golem/").await, + Vec::::new() + ); +} + +#[test] +fn a_prune_leaves_marked_packs_when_it_marks_repacks_or_keeps_marked_packs() { + let report = |packs_unused, packs_repacked, marked_packs_kept| PruneReport { + packs_unused, + packs_repacked, + marked_packs_kept, + ..PruneReport::default() + }; + + assert_eq!( + [ + leaves_marked_packs(&report(0, 0, 0)), + leaves_marked_packs(&report(1, 0, 0)), + leaves_marked_packs(&report(0, 1, 0)), + leaves_marked_packs(&report(0, 0, 1)), + leaves_marked_packs(&PruneReport { + packs_used: 3, + marked_packs_deleted: 2, + ..PruneReport::default() + }), + ], + [false, true, true, true, false] + ); +} + +#[test] +async fn a_save_of_a_relative_directory_path_gives_source_and_publishes_nothing() { + // Cargo runs the tests in the directory of the crate, so the path names a directory. + let relative = Path::new("src/filesystem_snapshot/contract_tests"); + assert!( + relative.is_dir(), + "the test runs in the directory of the crate" + ); + let store = store( + Arc::new(InMemoryBlobStorage::new()), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + + let saved = store.save(&scope, &name("p-relative"), relative).await; + let names = listed_names(&store, &scope).await; + + assert!( + matches!(saved, Err(SnapshotStoreError::Source(_))), + "{saved:?}" + ); + assert_eq!(names, Vec::::new()); +} + +#[test] +async fn a_save_of_a_regular_file_gives_source_and_publishes_nothing() { + let tree = one_file_tree("a file, not a tree"); + let store = store( + Arc::new(InMemoryBlobStorage::new()), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + + let saved = store + .save(&scope, &name("p-file"), &tree.path().join("file.txt")) + .await; + let names = listed_names(&store, &scope).await; + + assert!( + matches!(saved, Err(SnapshotStoreError::Source(_))), + "{saved:?}" + ); + assert_eq!(names, Vec::::new()); +} + +#[test] +async fn the_ledger_counts_the_packed_bytes_that_the_deleted_snapshot_added() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + store + .save(&scope, &name("p-deleted"), tree.path()) + .await + .unwrap(); + let added = snapshot_files(storage.clone(), &scope) + .await + .iter() + .filter_map(|snapshot| snapshot.summary.as_ref()) + .map(|summary| summary.data_added_packed) + .sum::(); + + store.delete(&scope, &name("p-deleted")).await.unwrap(); + + assert_eq!( + (added > 1, ledger(&*storage, &scope).await.freed_bytes), + (true, added) + ); +} + +#[test] +async fn a_config_write_that_fails_gives_a_storage_error_with_that_failure() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "write" && path == Path::new("config") { + Script::Refuse + } else { + Script::Pass + } + }); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = one_file_tree("never saved"); + + let saved = store.save(&scope, &name("p-1"), tree.path()).await; + + assert!( + matches!( + &saved, + Err(SnapshotStoreError::Storage { retryable: true, source }) + if format!("{source:#}").contains("the storage refused the call") + ), + "{saved:?}" + ); +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index d94c238915..6d9fa71445 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -209,6 +209,36 @@ async fn a_saved_tree_comes_back_the_same() { ); } +#[test] +async fn a_restore_report_gives_each_phase_of_the_restore_in_order() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let repository = repository(&storage, &new_scope()); + let tree = fixture_tree(); + let into = Scratch::new(); + repository.save(&name("first"), tree.path()).await.unwrap(); + + let restored = repository + .restore(&name("first"), into.path(), None) + .await + .unwrap() + .unwrap(); + + assert_eq!( + restored + .phases + .iter() + .map(|phase| phase.phase) + .collect::>(), + vec![ + OperationPhase::Open, + OperationPhase::Lookup, + OperationPhase::IndexLoad, + OperationPhase::RestorePlan, + OperationPhase::Restore, + ] + ); +} + #[test] async fn a_second_save_has_the_first_as_parent_and_reads_only_the_changed_file() { let storage = Arc::new(InMemoryBlobStorage::new()); From 7c832e9d20f0c939c1877e628a2034181fa5cbee Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 10:20:34 -0700 Subject: [PATCH 015/126] Require that a refused tree leaves the scope without a blob --- .../filesystem_snapshot/rustic/store/tests.rs | 17 +++++++++-------- 1 file changed, 9 insertions(+), 8 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index ed851d126b..e429f0e3ae 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1252,34 +1252,36 @@ fn a_prune_leaves_marked_packs_when_it_marks_repacks_or_keeps_marked_packs() { } #[test] -async fn a_save_of_a_relative_directory_path_gives_source_and_publishes_nothing() { +async fn a_save_of_a_relative_directory_path_gives_source_and_writes_nothing() { // Cargo runs the tests in the directory of the crate, so the path names a directory. let relative = Path::new("src/filesystem_snapshot/contract_tests"); assert!( relative.is_dir(), "the test runs in the directory of the crate" ); + let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( - Arc::new(InMemoryBlobStorage::new()), + storage.clone(), policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), ); let scope = new_scope(); let saved = store.save(&scope, &name("p-relative"), relative).await; - let names = listed_names(&store, &scope).await; assert!( matches!(saved, Err(SnapshotStoreError::Source(_))), "{saved:?}" ); - assert_eq!(names, Vec::::new()); + assert_eq!(blobs(&*storage, &scope.0, "").await, Vec::::new()); } #[test] -async fn a_save_of_a_regular_file_gives_source_and_publishes_nothing() { +async fn a_save_of_a_regular_file_gives_source_and_writes_nothing() { + // The store refuses the tree before it makes a repository, so the scope stays unused. let tree = one_file_tree("a file, not a tree"); + let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( - Arc::new(InMemoryBlobStorage::new()), + storage.clone(), policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), ); let scope = new_scope(); @@ -1287,13 +1289,12 @@ async fn a_save_of_a_regular_file_gives_source_and_publishes_nothing() { let saved = store .save(&scope, &name("p-file"), &tree.path().join("file.txt")) .await; - let names = listed_names(&store, &scope).await; assert!( matches!(saved, Err(SnapshotStoreError::Source(_))), "{saved:?}" ); - assert_eq!(names, Vec::::new()); + assert_eq!(blobs(&*storage, &scope.0, "").await, Vec::::new()); } #[test] From 785cb0894eb5974da7c121f02a6a993c622f5c32 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 11:50:32 -0700 Subject: [PATCH 016/126] Wake the waits for gateway bodies when a body starts, and let the test timeout bound them --- .../src/gateway_server/tests.rs | 26 +++++++++---------- 1 file changed, 12 insertions(+), 14 deletions(-) diff --git a/golem-worker-service/src/gateway_server/tests.rs b/golem-worker-service/src/gateway_server/tests.rs index 0ed56f402b..ab432c8fb4 100644 --- a/golem-worker-service/src/gateway_server/tests.rs +++ b/golem-worker-service/src/gateway_server/tests.rs @@ -17,8 +17,6 @@ use tokio::sync::Notify; use super::run; -const CLEANUP_TIMEOUT: Duration = Duration::from_secs(2); - #[derive(Clone, Default)] struct LiveBodies { count: Arc, @@ -26,8 +24,10 @@ struct LiveBodies { } impl LiveBodies { + /// Counts one more live body, and wakes each wait for a count. fn guard(&self) -> BodyGuard { self.count.fetch_add(1, Ordering::SeqCst); + self.changed.notify_waiters(); BodyGuard(self.clone()) } @@ -35,20 +35,18 @@ impl LiveBodies { self.count.load(Ordering::SeqCst) } + /// Waits until the count is `expected`. The wait has no limit of its own; the timeout of each + /// test ends a wait that never ends. async fn wait_for(&self, expected: usize) { - tokio::time::timeout(CLEANUP_TIMEOUT, async { - loop { - let changed = self.changed.notified(); - tokio::pin!(changed); - changed.as_mut().enable(); - if self.count() == expected { - return; - } - changed.await; + loop { + let changed = self.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + if self.count() == expected { + return; } - }) - .await - .unwrap_or_else(|_| panic!("body count did not become {expected}; was {}", self.count())); + changed.await; + } } } From 94b2321a14da5d9ca4f359e4da26835a7f8a2edf Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:31:16 -0700 Subject: [PATCH 017/126] Try the dropped-save contract case in new scopes until a save does not end in its first poll --- .../filesystem_snapshot/contract_tests/mod.rs | 27 +++++++++++++------ 1 file changed, 19 insertions(+), 8 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs b/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs index 1d57d21b91..0b33ba5d99 100644 --- a/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs @@ -62,6 +62,10 @@ pub(crate) type OpenStore = Arc Arc + S type Case = fn(OpenStore) -> BoxFuture<'static, ()>; +/// The number of saves that the dropped-save case starts to find one that does not end in its +/// first poll. +const DROP_ATTEMPTS: usize = 20; + /// Each case of the contract, with its name. const CASES: &[(&str, Case)] = &[ ("a_saved_tree_comes_back_the_same", |open| { @@ -958,19 +962,26 @@ async fn a_dropped_save_publishes_nothing_and_leaves_the_name_free(open: OpenSto // interrupted save, so it publishes nothing and leaves the name free. This case checks at // once after the drop, and it cannot see a publish that comes much later. So an adapter that // runs its save in the background also proves in its own tests that a dropped save stops. + // A save can end in its first poll when a busy host runs its read before that poll, so the + // case tries saves in new scopes until one does not end in its first poll. let store = open(); - let scope = new_scope(); let dropped_name = name("p-dropped"); let tree = new_tree(&fixture()); let other = new_tree(&one_file("other tree")); - let dropped = store - .save(&scope, &dropped_name, tree.path()) - .now_or_never(); - assert!( - dropped.is_none(), - "the save returned in one poll, so the case cannot drop it before it returns: {dropped:?}" - ); + let scope = futures::stream::iter(0..DROP_ATTEMPTS) + .filter_map(|_| { + let scope = new_scope(); + let dropped = store + .save(&scope, &dropped_name, tree.path()) + .now_or_never(); + std::future::ready(dropped.is_none().then_some(scope)) + }) + .next() + .await + .unwrap_or_else(|| { + panic!("each of {DROP_ATTEMPTS} saves returned in one poll, so the case cannot drop one before it returns") + }); let stat = store.stat(&scope, &dropped_name).await.unwrap(); let restore = restored(&*store, &scope, &dropped_name).await; let names_after_the_drop = listed_names(&*store, &scope).await; From c73a5dcecca3ddaeb49b7324f220eb708e26d61a Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:55:48 -0700 Subject: [PATCH 018/126] Classify a prune error as a storage error, and test the real chain of a failed extended attribute --- .../src/filesystem_snapshot/rustic/fault.rs | 100 +++++++++++++++--- 1 file changed, 83 insertions(+), 17 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs index 0c273adf74..36c09adb04 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs @@ -104,18 +104,15 @@ pub(super) enum Operation { Restore, /// Reads or changes only the repository. Repository, + /// Removes the data that no snapshot uses. + Prune, } -/// Gives the error of the store for an operation that failed with the error. -/// -/// - A failed blob storage call gives `Storage`. It is retryable unless a name error of the blob -/// storage caused it, because a name error is the same for each new try. -/// - An I/O error gives `Source` in a save and `Destination` in a restore, with the kind of that -/// I/O error. -/// - Each other error gives `Storage` that is not retryable in a save, and `Corrupt` in the other -/// operations, because the repository gave data that rustic refused. +/// A failed storage call gives `Storage`, retryable unless a name error caused it. An I/O error +/// gives `Source` in a save and `Destination` in a restore. Each other error gives `Storage` that +/// is not retryable in a save or a prune, and `Corrupt` in a restore or a read of the repository. pub(super) fn classify(operation: Operation, error: anyhow::Error) -> SnapshotStoreError { - let from_storage = chain(error.as_ref()).any(|error| error.is::()); + let from_storage = is_storage_failure(error.as_ref()); let io_kind = chain(error.as_ref()) .find_map(|error| error.downcast_ref::()) .map(std::io::Error::kind); @@ -131,7 +128,7 @@ pub(super) fn classify(operation: Operation, error: anyhow::Error) -> SnapshotSt (false, Some(kind), Operation::Restore) => { SnapshotStoreError::Destination(std::io::Error::new(kind, error_text(&error))) } - (false, _, Operation::Save) => SnapshotStoreError::Storage { + (false, _, Operation::Save | Operation::Prune) => SnapshotStoreError::Storage { retryable: false, source: error, }, @@ -186,7 +183,7 @@ mod tests { )) } - fn storage_failure(failure: anyhow::Error) -> anyhow::Error { + fn failed_call(failure: anyhow::Error) -> anyhow::Error { rustic(BlobCallFailed::new(failure)) } @@ -208,7 +205,7 @@ mod tests { [Operation::Save, Operation::Restore, Operation::Repository].map(|operation| { shape(&classify( operation, - storage_failure(anyhow::anyhow!("the bucket is gone")), + failed_call(anyhow::anyhow!("the bucket is gone")), )) }); @@ -220,7 +217,7 @@ mod tests { let failure = anyhow::Error::new(io::Error::new(io::ErrorKind::StorageFull, "no space")); assert_eq!( - shape(&classify(Operation::Restore, storage_failure(failure))), + shape(&classify(Operation::Restore, failed_call(failure))), ("Storage", Some(true), None) ); } @@ -232,7 +229,7 @@ mod tests { }); assert_eq!( - shape(&classify(Operation::Repository, storage_failure(failure))), + shape(&classify(Operation::Repository, failed_call(failure))), ("Storage", Some(false), None) ); } @@ -243,7 +240,7 @@ mod tests { [Operation::Save, Operation::Restore, Operation::Repository].map(|operation| { shape(&classify( operation, - storage_failure(anyhow::Error::new(OperationCancelled)), + failed_call(anyhow::Error::new(OperationCancelled)), )) }); @@ -314,10 +311,79 @@ mod tests { ); } + /// The error below the rustic error of a restore that cannot set an extended attribute, as + /// the fork gives it on ext4 for a user attribute of 6,000 bytes. + #[derive(Debug)] + struct SettingXattrFailed(io::Error); + + impl std::fmt::Display for SettingXattrFailed { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + formatter, + "setting xattr `user.golem-test` on `\"/restore/file.txt\"` with `{:?}`", + self.0 + ) + } + } + + impl std::error::Error for SettingXattrFailed { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(&self.0) + } + } + + #[test] + fn the_real_chain_of_a_failed_extended_attribute_of_a_restore_gives_destination() { + let error = anyhow::Error::new(RusticError::with_source( + ErrorKind::InputOutput, + "The restore cannot set the extended attributes of `file.txt`.", + SettingXattrFailed(io::Error::from_raw_os_error(28)), + )); + + assert_eq!( + shape(&classify(Operation::Restore, error)), + ("Destination", None, Some(io::ErrorKind::StorageFull)) + ); + } + + #[test] + fn a_prune_error_keeps_the_retryable_flag_of_a_storage_failure_and_is_never_corrupt() { + let refused = anyhow::Error::new(RusticError::new( + ErrorKind::Internal, + "the pack has another size than the index says", + )); + + assert_eq!( + [ + shape(&classify(Operation::Prune, refused)), + shape(&classify( + Operation::Prune, + rustic(io::Error::new(io::ErrorKind::InvalidData, "bad pack")) + )), + shape(&classify( + Operation::Prune, + failed_call(anyhow::anyhow!("the bucket is gone")) + )), + shape(&classify( + Operation::Prune, + failed_call(anyhow::Error::new(BlobNameError::NoName { + path: std::path::PathBuf::new(), + })) + )), + ], + [ + ("Storage", Some(false), None), + ("Storage", Some(false), None), + ("Storage", Some(true), None), + ("Storage", Some(false), None), + ] + ); + } + #[test] fn the_config_marker_is_found_in_the_chain() { - let exists = storage_failure(anyhow::Error::new(ConfigExists)); - let other = storage_failure(anyhow::anyhow!("the bucket is gone")); + let exists = failed_call(anyhow::Error::new(ConfigExists)); + let other = failed_call(anyhow::anyhow!("the bucket is gone")); assert_eq!( ( From c0e489dc07be2fdbf986b58936f4d089f0efaaf8 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:55:48 -0700 Subject: [PATCH 019/126] Shorten the doc of the default storage call deadline --- golem-worker-executor/src/services/golem_config.rs | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/golem-worker-executor/src/services/golem_config.rs b/golem-worker-executor/src/services/golem_config.rs index 82e667787f..3ab3c64114 100644 --- a/golem-worker-executor/src/services/golem_config.rs +++ b/golem-worker-executor/src/services/golem_config.rs @@ -2339,11 +2339,8 @@ impl SafeDisplay for FilesystemPressureConfig { } } -/// The default of [`FilesystemSnapshotStoreConfig::storage_call_deadline`]. -/// -/// On S3, with the retries of the S3 storage, a write of a pack took at most 1.7 s with eight saves -/// at the same time. A ranged read of a pack took at most 1.5 s under the CPU request of an -/// executor. Keep the value at least 10 times the longest measured call. +/// The default of [`FilesystemSnapshotStoreConfig::storage_call_deadline`]. The slowest measured +/// call on S3 took 1.7 s, and the value stays at least 10 times the slowest measured call. pub const DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE: Duration = Duration::from_secs(30); /// The default of [`FilesystemSnapshotStoreConfig::restore_reader_threads`]. From 855fd18aa0dfa48a858894ea575f01974aac91a8 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:55:49 -0700 Subject: [PATCH 020/126] Pass the blobs of a scope as one value with one target label, share staged paths, and tighten the store tests --- .../src/filesystem_snapshot/rustic/backend.rs | 17 +- .../rustic/backend/tests.rs | 2 +- .../src/filesystem_snapshot/rustic/files.rs | 130 ++++++++++++++++ .../src/filesystem_snapshot/rustic/mod.rs | 1 + .../src/filesystem_snapshot/rustic/prune.rs | 111 ++++++-------- .../src/filesystem_snapshot/rustic/publish.rs | 55 ++----- .../rustic/publish/tests.rs | 7 +- .../src/filesystem_snapshot/rustic/scope.rs | 145 ++++-------------- .../filesystem_snapshot/rustic/scope/tests.rs | 68 +++++--- .../src/filesystem_snapshot/rustic/store.rs | 49 +++--- .../filesystem_snapshot/rustic/store/tests.rs | 114 +++++++++++--- 11 files changed, 386 insertions(+), 313 deletions(-) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/files.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index 0c64a04aeb..c0afa5cd55 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -20,6 +20,7 @@ //! for at most a deadline, and a cancelled operation makes no more calls. use super::fault::{BlobCallFailed, ConfigExists, FileMissing, OperationCancelled}; +use super::files::TARGET_LABEL; use super::publish::{SnapshotStage, StagedSnapshot}; use bytes::Bytes; use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace, PutIfAbsent}; @@ -34,9 +35,6 @@ use tokio::runtime::Handle; use tokio_util::sync::CancellationToken; use tokio_util::task::task_tracker::TaskTrackerToken; -/// The target label of each blob storage call of the backend. -const TARGET_LABEL: &str = "filesystem_snapshot"; - /// The path of the config file of a repository. const CONFIG_PATH: &str = "config"; @@ -139,11 +137,9 @@ impl BlobBackend { } } - /// Waits for one call on the blob storage, and gives its result as a rustic result. - /// - /// Each call of the backend on the blob storage goes through this function. A call that gives - /// no answer within the deadline gives an error, the same as a call that failed. So does a - /// call of a cancelled operation. + /// Waits for one call on the blob storage, which each call of the backend goes through. A call + /// without an answer within the deadline, or of a cancelled operation, gives an error, the same + /// as a call that failed. fn request( &self, call: StorageCall, @@ -320,7 +316,10 @@ impl WriteBackend for BlobBackend { }; match (tpe, &self.stage) { (FileType::Snapshot, Some(stage)) => stage - .keep(StagedSnapshot { path, content }) + .keep(StagedSnapshot { + path: Arc::from(path), + content, + }) .map_err(|staged| second_snapshot(&staged.path)), (FileType::Config, _) => match self.write_if_absent(&path, &content)? { PutIfAbsent::Written => Ok(()), diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index e0ef93e606..413763cdc5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -668,7 +668,7 @@ fn a_backend_with_a_stage_keeps_the_snapshot_file_and_does_not_write_it() { true, true, Some(StagedSnapshot { - path: PathBuf::from(format!("snapshots/{}", "cd".repeat(32))).into_boxed_path(), + path: Arc::from(PathBuf::from(format!("snapshots/{}", "cd".repeat(32)))), content: Bytes::from_static(b"snapshot"), }), vec![(format!("data/ab/{}", "ab".repeat(32)), 4)] diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs new file mode 100644 index 0000000000..297f5c5959 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs @@ -0,0 +1,130 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The blobs of the repository of one scope, and the blob storage calls of the store on them. +//! +//! Each call waits for at most the deadline of the scope. + +use super::backend::answer_within; +use golem_service_base::storage::blob::{ + BlobStorage, BlobStorageNamespace, ListedBlob, PutIfAbsent, +}; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; + +/// The target label of each blob storage call of the rustic store. +pub(super) const TARGET_LABEL: &str = "filesystem_snapshot"; + +/// The blobs of one scope: the storage, the namespace of the scope, and the deadline of each call. +#[derive(Clone, Debug)] +pub(super) struct SnapshotFiles { + pub(super) storage: Arc, + pub(super) namespace: BlobStorageNamespace, + pub(super) deadline: Duration, +} + +impl SnapshotFiles { + /// Gives the content of the blob at the path, or `None` when the path has no blob. + pub(super) async fn get( + &self, + op_label: &'static str, + path: &Path, + ) -> anyhow::Result>> { + answer_within( + self.deadline, + self.storage + .get_raw(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await + } + + /// Writes the content as the blob at the path, over the blob that was there. + pub(super) async fn put( + &self, + op_label: &'static str, + path: &Path, + content: &[u8], + ) -> anyhow::Result<()> { + answer_within( + self.deadline, + self.storage.put_raw( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + content, + ), + ) + .await + } + + /// Writes the content as the blob at the path only when the path has no blob. + pub(super) async fn put_if_absent( + &self, + op_label: &'static str, + path: &Path, + content: &[u8], + ) -> anyhow::Result { + answer_within( + self.deadline, + self.storage.put_raw_if_absent( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + content, + ), + ) + .await + } + + /// Deletes the blob at the path. A path without a blob gives success. + pub(super) async fn delete(&self, op_label: &'static str, path: &Path) -> anyhow::Result<()> { + answer_within( + self.deadline, + self.storage + .delete(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await + } + + /// Deletes the directory at the path and each blob below it. + pub(super) async fn delete_dir( + &self, + op_label: &'static str, + path: &Path, + ) -> anyhow::Result { + answer_within( + self.deadline, + self.storage + .delete_dir(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await + } + + /// Gives each blob below the path, at all depths, with its size. + pub(super) async fn list_below( + &self, + op_label: &'static str, + path: &Path, + ) -> anyhow::Result> { + answer_within( + self.deadline, + self.storage + .list_blobs_below(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await + } +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 6be31c2ff0..cd02c37a1d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -20,6 +20,7 @@ mod backend; mod fault; +mod files; mod prune; mod publish; mod scope; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index f2510a8b93..e412271a5d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -16,20 +16,16 @@ //! //! The scope keeps a small ledger blob next to the files of the repository. The ledger holds the //! packed bytes that deleted snapshots added since the last prune, the time of the last prune, and -//! whether that prune marked packs that a later prune removes. The ledger is advice: two deletes at -//! the same time can lose a count, and that only makes a prune come later. +//! whether that prune marked packs that a later prune removes. Two deletes at the same time can +//! lose a count. A lost count only delays a prune. -use super::backend::answer_within; +use super::files::SnapshotFiles; use golem_common::model::Timestamp; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; use serde::{Deserialize, Serialize}; use std::path::Path; use std::time::Duration; use tracing::warn; -/// The target label of each blob storage call on the ledger. -const TARGET_LABEL: &str = "filesystem_snapshot"; - /// The path of the ledger blob, relative to the root of the namespace of the scope. pub(super) const LEDGER_PATH: &str = "golem/prune-ledger"; @@ -38,8 +34,8 @@ pub(super) const LEDGER_PATH: &str = "golem/prune-ledger"; pub(super) struct PruneLedger { /// The packed bytes that the deleted snapshots added, since the last prune. pub(super) freed_bytes: u64, - /// The time of the last prune, in milliseconds since the Unix epoch. - pub(super) last_prune_millis: Option, + /// The time of the last prune. + pub(super) last_prune: Option, /// Whether the last prune marked packs that a later prune removes. pub(super) awaiting_removal: bool, } @@ -57,7 +53,7 @@ impl PruneLedger { pub(super) fn after_prune(now: Timestamp, marked_packs: bool) -> Self { Self { freed_bytes: 0, - last_prune_millis: Some(now.to_millis()), + last_prune: Some(now), awaiting_removal: marked_packs, } } @@ -74,30 +70,20 @@ pub(super) fn prune_due( threshold: u64, grace: Duration, ) -> bool { - let grace_passed = ledger.last_prune_millis.is_none_or(|last| { - now.to_millis() >= last.saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) + let grace_passed = ledger.last_prune.is_none_or(|last| { + now.to_millis() + >= last + .to_millis() + .saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) }); let work = ledger.freed_bytes >= threshold.max(1) || ledger.awaiting_removal; grace_passed && work } -/// Reads the ledger of the scope. A scope without a ledger gives an empty ledger, and so does a -/// ledger that does not parse, because the ledger is advice. -pub(super) async fn read_ledger( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - deadline: Duration, -) -> anyhow::Result { - let content = answer_within( - deadline, - storage.get_raw( - TARGET_LABEL, - "read_ledger", - namespace.clone(), - Path::new(LEDGER_PATH), - ), - ) - .await?; +/// Reads the ledger of the scope. A scope without a ledger, or with a ledger that does not parse, +/// gives an empty ledger, which only delays a prune. +pub(super) async fn read_ledger(files: &SnapshotFiles) -> anyhow::Result { + let content = files.get("read_ledger", Path::new(LEDGER_PATH)).await?; Ok(content.map_or_else(PruneLedger::default, |content| { serde_json::from_slice(&content).unwrap_or_else(|error| { warn!( @@ -111,34 +97,26 @@ pub(super) async fn read_ledger( /// Writes the ledger of the scope over the ledger that was there. pub(super) async fn write_ledger( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - deadline: Duration, + files: &SnapshotFiles, ledger: &PruneLedger, ) -> anyhow::Result<()> { let content = serde_json::to_vec(ledger)?; - answer_within( - deadline, - storage.put_raw( - TARGET_LABEL, - "write_ledger", - namespace.clone(), - Path::new(LEDGER_PATH), - &content, - ), - ) - .await + files + .put("write_ledger", Path::new(LEDGER_PATH), &content) + .await } #[cfg(test)] mod tests { + use super::super::files::SnapshotFiles; use super::{LEDGER_PATH, PruneLedger, prune_due, read_ledger, write_ledger}; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; + use golem_service_base::storage::blob::BlobStorageNamespace; use golem_service_base::storage::blob::memory::InMemoryBlobStorage; - use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; use pretty_assertions::assert_eq; use std::path::Path; + use std::sync::Arc; use std::time::Duration; use test_r::test; use uuid::Uuid; @@ -159,14 +137,18 @@ mod tests { ) -> PruneLedger { PruneLedger { freed_bytes, - last_prune_millis, + last_prune: last_prune_millis.map(Timestamp::from), awaiting_removal, } } - fn new_namespace() -> BlobStorageNamespace { - BlobStorageNamespace::InitialAgentFiles { - environment_id: EnvironmentId(Uuid::new_v4()), + fn new_files() -> SnapshotFiles { + SnapshotFiles { + storage: Arc::new(InMemoryBlobStorage::new()), + namespace: BlobStorageNamespace::InitialAgentFiles { + environment_id: EnvironmentId(Uuid::new_v4()), + }, + deadline: DEADLINE, } } @@ -257,38 +239,37 @@ mod tests { } #[test] - async fn the_ledger_is_written_and_read_back() { - let storage = InMemoryBlobStorage::new(); - let namespace = new_namespace(); - let written = ledger(123, Some(456), true); + async fn the_ledger_is_written_and_read_back_with_the_time_in_milliseconds() { + // The ledger keeps the time of the last prune as ISO 8601 text with milliseconds. + let files = new_files(); + let written = PruneLedger { + freed_bytes: 123, + last_prune: Some(Timestamp::from(Timestamp::now_utc().to_millis())), + awaiting_removal: true, + }; - let before = read_ledger(&storage, &namespace, DEADLINE).await.unwrap(); - write_ledger(&storage, &namespace, DEADLINE, &written) - .await - .unwrap(); - let after = read_ledger(&storage, &namespace, DEADLINE).await.unwrap(); + let before = read_ledger(&files).await.unwrap(); + write_ledger(&files, &written).await.unwrap(); + let after = read_ledger(&files).await.unwrap(); assert_eq!((before, after), (PruneLedger::default(), written)); } #[test] async fn a_ledger_that_does_not_parse_reads_as_an_empty_ledger() { - let storage = InMemoryBlobStorage::new(); - let namespace = new_namespace(); - storage + let files = new_files(); + files + .storage .put_raw( "test", "test", - namespace.clone(), + files.namespace.clone(), Path::new(LEDGER_PATH), b"not json", ) .await .unwrap(); - assert_eq!( - read_ledger(&storage, &namespace, DEADLINE).await.unwrap(), - PruneLedger::default() - ); + assert_eq!(read_ledger(&files).await.unwrap(), PruneLedger::default()); } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs index e8d48455ca..c2ea5c6e61 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs @@ -19,24 +19,20 @@ //! write is the step that makes the snapshot visible. A publish that fails, or that the caller //! drops, deletes the file again, because a write that the storage received can still complete. -use super::backend::answer_within; +use super::files::SnapshotFiles; use bytes::Bytes; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; use std::path::Path; use std::sync::{Arc, Mutex, PoisonError}; -use std::time::Duration; use tokio::runtime::Handle; use tokio_util::task::TaskTracker; use tracing::warn; -/// The target label of each blob storage call of a publish. -const TARGET_LABEL: &str = "filesystem_snapshot"; - /// A snapshot file that the backend kept and did not write. #[derive(Clone, Debug, PartialEq, Eq)] pub(super) struct StagedSnapshot { - /// The path of the file, relative to the root of the namespace. - pub(super) path: Box, + /// The path of the file, relative to the root of the namespace. The drop guard of a publish + /// and its delete task share it. + pub(super) path: Arc, pub(super) content: Bytes, } @@ -63,22 +59,9 @@ impl SnapshotStage { } } -/// The snapshot files of one scope: the storage, the namespace of the scope, and the deadline of -/// each call. -#[derive(Clone, Debug)] -pub(super) struct SnapshotFiles { - pub(super) storage: Arc, - pub(super) namespace: BlobStorageNamespace, - pub(super) deadline: Duration, -} - -/// Writes the staged file, and so makes the snapshot visible. -/// -/// The file is written only when the path has no blob. The name of a snapshot file is the hash of -/// its content, so a blob at the path already holds this content, and the call succeeds. -/// -/// When the write fails, the call deletes the path before it gives the error. When the caller -/// drops the call during the write, a task of `tracker` deletes the path. +/// Writes the staged file only when its path has no blob, which makes the snapshot visible. The +/// name is the hash of the content, so a blob at the path is this file. A failed write deletes the +/// path before the error returns, and a dropped write deletes it in a task of `tracker`. pub(super) async fn publish( files: &SnapshotFiles, staged: &StagedSnapshot, @@ -90,17 +73,9 @@ pub(super) async fn publish( tracker: tracker.clone(), armed: true, }; - let written = answer_within( - files.deadline, - files.storage.put_raw_if_absent( - TARGET_LABEL, - "publish", - files.namespace.clone(), - &staged.path, - &staged.content, - ), - ) - .await; + let written = files + .put_if_absent("publish", &staged.path, &staged.content) + .await; retraction.armed = false; match written { Ok(_) => Ok(()), @@ -113,13 +88,7 @@ pub(super) async fn publish( /// Deletes the snapshot file at the path. A path without a blob gives success. pub(super) async fn retract(files: &SnapshotFiles, path: &Path) -> anyhow::Result<()> { - answer_within( - files.deadline, - files - .storage - .delete(TARGET_LABEL, "retract", files.namespace.clone(), path), - ) - .await + files.delete("retract", path).await } async fn retract_or_warn(files: &SnapshotFiles, path: &Path) { @@ -135,7 +104,7 @@ async fn retract_or_warn(files: &SnapshotFiles, path: &Path) { /// Deletes the path in a task of the tracker when it is dropped while it is armed. struct RetractOnDrop { files: SnapshotFiles, - path: Box, + path: Arc, tracker: TaskTracker, armed: bool, } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs index a61fc61724..ec5723fe13 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -12,16 +12,17 @@ // See the License for the specific language governing permissions and // limitations under the License. +use super::super::files::SnapshotFiles; use super::super::holding::reached_deadline; use super::super::scripted::{Script, ScriptedBlobStorage}; -use super::{SnapshotFiles, SnapshotStage, StagedSnapshot, publish, retract}; +use super::{SnapshotStage, StagedSnapshot, publish, retract}; use bytes::Bytes; use futures::FutureExt; use golem_common::model::environment::EnvironmentId; use golem_service_base::storage::blob::memory::InMemoryBlobStorage; use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; use pretty_assertions::assert_eq; -use std::path::{Path, PathBuf}; +use std::path::Path; use std::sync::Arc; use std::time::Duration; use test_r::test; @@ -36,7 +37,7 @@ const SNAPSHOT_PATH: &str = fn staged() -> StagedSnapshot { StagedSnapshot { - path: PathBuf::from(SNAPSHOT_PATH).into_boxed_path(), + path: Arc::from(Path::new(SNAPSHOT_PATH)), content: Bytes::from_static(b"snapshot"), } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs index d80ebe91f6..996e605de2 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs @@ -17,63 +17,28 @@ //! These operations do not read the repository format. They only know the directories of the //! repository, its config file, and the ledger directory of the store. -use super::backend::answer_within; +use super::files::SnapshotFiles; use super::prune::LEDGER_PATH; use futures::{StreamExt, TryStreamExt, stream}; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace, PutIfAbsent}; -use std::path::{Path, PathBuf}; -use std::time::Duration; - -/// The target label of each blob storage call on a scope. -const TARGET_LABEL: &str = "filesystem_snapshot"; +use golem_service_base::storage::blob::PutIfAbsent; +use std::path::Path; /// The path of the config file of a repository. const CONFIG_PATH: &str = "config"; /// The directories of a repository in the order of a listing. A save writes them in the reverse -/// order, so each snapshot file in a listing has its index files and packs in the later listings. -/// A copy writes them in the reverse order too, so a snapshot file in the target always has its -/// data. +/// order, and so does a copy, so a snapshot file always has its data. const LISTING_ORDER: [&str; 4] = ["snapshots", "index", "keys", "data"]; -/// Copies the repository of the namespace `from` into the empty namespace `to`. -/// -/// A namespace without a config file holds no repository, so the call copies nothing. The call -/// writes the config file of `to` last, so `to` holds a repository only when all its blobs are -/// there. It does not copy the ledger. A blob that is gone when the call reads it was deleted -/// after the listing, and the call does not copy it. -pub(super) async fn copy_scope( - storage: &dyn BlobStorage, - from: &BlobStorageNamespace, - to: &BlobStorageNamespace, - deadline: Duration, -) -> anyhow::Result<()> { - let Some(config) = answer_within( - deadline, - storage.get_raw( - TARGET_LABEL, - "copy_read", - from.clone(), - Path::new(CONFIG_PATH), - ), - ) - .await? - else { +/// Copies the repository of `from` into the empty scope `to`, with the config file last, so `to` +/// holds a repository only when all its blobs are there. It copies nothing without a config file, +/// and it does not copy the ledger or a blob that a delete removes after the listing. +pub(super) async fn copy_scope(from: &SnapshotFiles, to: &SnapshotFiles) -> anyhow::Result<()> { + let Some(config) = from.get("copy_read", Path::new(CONFIG_PATH)).await? else { return Ok(()); }; let listed = stream::iter(LISTING_ORDER) - .then(|directory| async move { - answer_within( - deadline, - storage.list_blobs_below( - TARGET_LABEL, - "copy_list", - from.clone(), - Path::new(directory), - ), - ) - .await - }) + .then(|directory| from.list_below("copy_list", Path::new(directory))) .try_collect::>() .await?; let paths = listed @@ -82,84 +47,38 @@ pub(super) async fn copy_scope( .flat_map(|blobs| blobs.iter().map(|blob| blob.path.clone())) .collect::>(); stream::iter(paths.iter().map(Ok)) - .try_for_each(|path| copy_blob(storage, from, to, path, deadline)) + .try_for_each(|path| copy_blob(from, to, path)) .await?; - answer_within( - deadline, - storage.put_raw_if_absent( - TARGET_LABEL, - "copy_write", - to.clone(), - Path::new(CONFIG_PATH), - &config, - ), - ) - .await - .map(|_: PutIfAbsent| ()) + to.put_if_absent("copy_write", Path::new(CONFIG_PATH), &config) + .await + .map(|_: PutIfAbsent| ()) } -async fn copy_blob( - storage: &dyn BlobStorage, - from: &BlobStorageNamespace, - to: &BlobStorageNamespace, - path: &Path, - deadline: Duration, -) -> anyhow::Result<()> { - let content = answer_within( - deadline, - storage.get_raw(TARGET_LABEL, "copy_read", from.clone(), path), - ) - .await?; - match content { - Some(content) => { - answer_within( - deadline, - storage.put_raw(TARGET_LABEL, "copy_write", to.clone(), path, &content), - ) - .await - } +async fn copy_blob(from: &SnapshotFiles, to: &SnapshotFiles, path: &Path) -> anyhow::Result<()> { + match from.get("copy_read", path).await? { + Some(content) => to.put("copy_write", path, &content).await, None => Ok(()), } } -/// Deletes the repository of the namespace, and the ledger of the store. -/// -/// The call deletes the config file first, so the namespace holds no repository from that step on. -/// Then it deletes each directory. A namespace that holds nothing gives success. -pub(super) async fn delete_scope( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - deadline: Duration, -) -> anyhow::Result<()> { - answer_within( - deadline, - storage.delete( - TARGET_LABEL, - "delete_scope", - namespace.clone(), - Path::new(CONFIG_PATH), - ), +/// Deletes the repository and the ledger of the scope. The config file goes first, so the scope +/// holds no repository from that step on. A scope that holds nothing gives success. +pub(super) async fn delete_scope(files: &SnapshotFiles) -> anyhow::Result<()> { + files.delete("delete_scope", Path::new(CONFIG_PATH)).await?; + stream::iter( + LISTING_ORDER + .iter() + .map(Path::new) + .chain(Path::new(LEDGER_PATH).parent()) + .map(Ok), ) - .await?; - let ledger_directory = Path::new(LEDGER_PATH) - .parent() - .map(Path::to_path_buf) - .unwrap_or_default(); - let directories = LISTING_ORDER - .iter() - .map(PathBuf::from) - .chain(std::iter::once(ledger_directory)) - .collect::>(); - stream::iter(directories.iter().map(Ok)) - .try_for_each(|directory| async move { - answer_within( - deadline, - storage.delete_dir(TARGET_LABEL, "delete_scope", namespace.clone(), directory), - ) + .try_for_each(|directory| async move { + files + .delete_dir("delete_scope", directory) .await .map(|_| ()) - }) - .await + }) + .await } #[cfg(test)] diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs index aaff157543..229d3380a9 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs @@ -12,6 +12,7 @@ // See the License for the specific language governing permissions and // limitations under the License. +use super::super::files::SnapshotFiles; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::{copy_scope, delete_scope}; use golem_common::model::environment::EnvironmentId; @@ -36,6 +37,17 @@ const REPOSITORY: [(&str, &str); 6] = [ ("snapshots/0101", "snapshot"), ]; +fn files( + storage: &Arc, + namespace: &BlobStorageNamespace, +) -> SnapshotFiles { + SnapshotFiles { + storage: storage.clone(), + namespace: namespace.clone(), + deadline: DEADLINE, + } +} + fn new_namespace() -> BlobStorageNamespace { BlobStorageNamespace::InitialAgentFiles { environment_id: EnvironmentId(Uuid::new_v4()), @@ -96,14 +108,16 @@ fn owned(blobs: &[(&str, &str)]) -> Vec<(String, String)> { #[test] async fn a_copy_gives_the_target_each_blob_of_the_repository_and_not_the_ledger() { - let storage = InMemoryBlobStorage::new(); + let storage = Arc::new(InMemoryBlobStorage::new()); let (from, to) = (new_namespace(), new_namespace()); - put_all(&storage, &from, &REPOSITORY).await; + put_all(&*storage, &from, &REPOSITORY).await; - copy_scope(&storage, &from, &to, DEADLINE).await.unwrap(); + copy_scope(&files(&storage, &from), &files(&storage, &to)) + .await + .unwrap(); assert_eq!( - (stored(&storage, &to).await, stored(&storage, &from).await), + (stored(&*storage, &to).await, stored(&*storage, &from).await), ( owned( &REPOSITORY @@ -123,7 +137,9 @@ async fn a_copy_writes_the_packs_the_keys_the_index_files_the_snapshot_files_and let (from, to) = (new_namespace(), new_namespace()); put_all(&*storage, &from, &REPOSITORY).await; - copy_scope(&*storage, &from, &to, DEADLINE).await.unwrap(); + copy_scope(&files(&storage, &from), &files(&storage, &to)) + .await + .unwrap(); assert_eq!( storage @@ -149,7 +165,9 @@ async fn a_copy_lists_the_snapshot_files_before_the_index_files_the_keys_and_the let (from, to) = (new_namespace(), new_namespace()); put_all(&*storage, &from, &REPOSITORY).await; - copy_scope(&*storage, &from, &to, DEADLINE).await.unwrap(); + copy_scope(&files(&storage, &from), &files(&storage, &to)) + .await + .unwrap(); assert_eq!( storage @@ -164,10 +182,10 @@ async fn a_copy_lists_the_snapshot_files_before_the_index_files_the_keys_and_the #[test] async fn a_copy_of_a_namespace_without_a_config_copies_nothing() { - let storage = InMemoryBlobStorage::new(); + let storage = Arc::new(InMemoryBlobStorage::new()); let (from, to) = (new_namespace(), new_namespace()); put_all( - &storage, + &*storage, &from, &REPOSITORY .into_iter() @@ -176,9 +194,11 @@ async fn a_copy_of_a_namespace_without_a_config_copies_nothing() { ) .await; - copy_scope(&storage, &from, &to, DEADLINE).await.unwrap(); + copy_scope(&files(&storage, &from), &files(&storage, &to)) + .await + .unwrap(); - assert_eq!(stored(&storage, &to).await, Vec::<(String, String)>::new()); + assert_eq!(stored(&*storage, &to).await, Vec::<(String, String)>::new()); } #[test] @@ -194,7 +214,7 @@ async fn a_copy_that_fails_gives_the_error_and_the_target_has_no_config() { let (from, to) = (new_namespace(), new_namespace()); put_all(&*storage, &from, &REPOSITORY).await; - let copied = copy_scope(&*storage, &from, &to, DEADLINE).await; + let copied = copy_scope(&files(&storage, &from), &files(&storage, &to)).await; assert_eq!( ( @@ -214,17 +234,17 @@ async fn a_copy_that_fails_gives_the_error_and_the_target_has_no_config() { #[test] async fn a_deleted_scope_holds_no_blob_and_another_scope_keeps_its_blobs() { - let storage = InMemoryBlobStorage::new(); + let storage = Arc::new(InMemoryBlobStorage::new()); let (deleted, kept) = (new_namespace(), new_namespace()); - put_all(&storage, &deleted, &REPOSITORY).await; - put_all(&storage, &kept, &REPOSITORY).await; + put_all(&*storage, &deleted, &REPOSITORY).await; + put_all(&*storage, &kept, &REPOSITORY).await; - delete_scope(&storage, &deleted, DEADLINE).await.unwrap(); + delete_scope(&files(&storage, &deleted)).await.unwrap(); assert_eq!( ( - stored(&storage, &deleted).await, - stored(&storage, &kept).await + stored(&*storage, &deleted).await, + stored(&*storage, &kept).await ), (Vec::new(), owned(&REPOSITORY)) ); @@ -237,7 +257,7 @@ async fn a_delete_of_a_scope_deletes_the_config_first() { let namespace = new_namespace(); put_all(&*storage, &namespace, &REPOSITORY).await; - delete_scope(&*storage, &namespace, DEADLINE).await.unwrap(); + delete_scope(&files(&storage, &namespace)).await.unwrap(); assert_eq!( storage @@ -252,20 +272,20 @@ async fn a_delete_of_a_scope_deletes_the_config_first() { #[test] async fn a_delete_of_an_unused_scope_succeeds_and_can_run_again() { - let storage = InMemoryBlobStorage::new(); + let storage = Arc::new(InMemoryBlobStorage::new()); let namespace = new_namespace(); - let first = delete_scope(&storage, &namespace, DEADLINE).await; - put_all(&storage, &namespace, &REPOSITORY).await; - let second = delete_scope(&storage, &namespace, DEADLINE).await; - let third = delete_scope(&storage, &namespace, DEADLINE).await; + let first = delete_scope(&files(&storage, &namespace)).await; + put_all(&*storage, &namespace, &REPOSITORY).await; + let second = delete_scope(&files(&storage, &namespace)).await; + let third = delete_scope(&files(&storage, &namespace)).await; assert_eq!( ( first.is_ok(), second.is_ok(), third.is_ok(), - stored(&storage, &namespace).await + stored(&*storage, &namespace).await ), (true, true, true, Vec::new()) ); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 59028be87c..63849aa514 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -22,8 +22,9 @@ use super::backend::BlobBackend; use super::fault::{Operation, classify, is_file_missing, is_storage_failure, storage_failure}; +use super::files::SnapshotFiles; use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; -use super::publish::{SnapshotFiles, SnapshotStage, StagedSnapshot, publish}; +use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; use super::{ ChangeDetection, PruneReport, PruneSettings, RepackLimits, RepositoryKey, RepositorySettings, @@ -157,10 +158,8 @@ impl RusticSnapshotStore { } /// Stops each operation at its next storage call, and waits until no blocking task and no - /// backend of the store remains. After the call, each operation gives `Storage`. - /// - /// The runtime must not drop before the call returns, because a storage call that waits on the - /// runtime after its time driver stops aborts the process. + /// backend of the store remains; later operations give `Storage`. The runtime must not drop + /// before it returns, because a storage call after its time driver stops aborts the process. pub(crate) async fn shut_down(&self) { self.root.cancel(); self.tracker.close(); @@ -239,13 +238,13 @@ impl RusticSnapshotStore { token: &CancellationToken, freed: u64, ) -> Result<(), SnapshotStoreError> { - let deadline = self.policy.deadline; - let ledger = read_ledger(&*self.storage, &scope.0, deadline) + let files = self.files(scope); + let ledger = read_ledger(&files) .await .map_err(storage_failure)? .with_deleted(freed); if freed > 0 { - write_ledger(&*self.storage, &scope.0, deadline, &ledger) + write_ledger(&files, &ledger) .await .map_err(storage_failure)?; } @@ -262,19 +261,12 @@ impl RusticSnapshotStore { let key = self.key.clone(); let settings = self.policy.prune; let report = self - .blocking(Operation::Repository, move || { - prune(backend, &key, &settings) - }) + .blocking(Operation::Prune, move || prune(backend, &key, &settings)) .await?; let marked_packs = report.as_ref().is_some_and(leaves_marked_packs); - write_ledger( - &*self.storage, - &scope.0, - deadline, - &PruneLedger::after_prune(now, marked_packs), - ) - .await - .map_err(storage_failure) + write_ledger(&files, &PruneLedger::after_prune(now, marked_packs)) + .await + .map_err(storage_failure) } } @@ -406,7 +398,7 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { async fn delete_scope(&self, scope: &SnapshotScope) -> Result<(), SnapshotStoreError> { let _operation = self.start()?; - delete_scope(&*self.storage, &scope.0, self.policy.deadline) + delete_scope(&self.files(scope)) .await .map_err(storage_failure) } @@ -417,7 +409,7 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { to: &SnapshotScope, ) -> Result<(), SnapshotStoreError> { let _operation = self.start()?; - copy_scope(&*self.storage, &from.0, &to.0, self.policy.deadline) + copy_scope(&self.files(from), &self.files(to)) .await .map_err(storage_failure) } @@ -564,11 +556,8 @@ fn stage_save( ))) } -/// Reads each snapshot file of the repository. -/// -/// A failed storage call fails the read. A file that the storage no longer holds is left out, -/// because a delete removed it after the listing. Each other failure counts as a file that failed -/// its integrity check. +/// Reads each snapshot file of the repository. A failed storage call fails the read, a file that a +/// delete removed after the listing is left out, and each other failure counts as a failed check. fn scope_snapshots(repository: &RusticRepository) -> anyhow::Result { repository.list::()?.try_fold( ScopeSnapshots { @@ -655,11 +644,9 @@ fn snapshot_info(snapshot: &SnapshotFile) -> Option { }) } -/// Tells whether a later prune removes packs that this prune leaves marked. -/// -/// A prune marks each pack that holds only unused blobs, and each pack that it repacks. A pack that -/// an earlier prune marked and whose time to stay is not over stays marked. A pack that no index -/// lists is also marked, but the report does not count it. A later due prune removes that pack. +/// Tells whether a later prune removes packs that this prune leaves marked: unused packs, repacked +/// packs, and packs of an earlier prune whose grace period is not over. The report does not count a +/// marked pack that no index lists, so the next due prune removes it. fn leaves_marked_packs(report: &PruneReport) -> bool { report.packs_unused > 0 || report.packs_repacked > 0 || report.marked_packs_kept > 0 } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index e429f0e3ae..1dc3d1deb4 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -17,6 +17,7 @@ //! The contract suite runs on the store with the policy of the configuration. The other tests //! give the store a short or a long deadline and a prune policy that the test controls. +use super::super::files::SnapshotFiles; use super::super::prune::{PruneLedger, read_ledger}; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; @@ -145,10 +146,14 @@ async fn blobs( paths } -async fn ledger(storage: &dyn BlobStorage, scope: &SnapshotScope) -> PruneLedger { - read_ledger(storage, &scope.0, Duration::from_secs(2)) - .await - .unwrap() +async fn ledger(storage: &Arc, scope: &SnapshotScope) -> PruneLedger { + read_ledger(&SnapshotFiles { + storage: storage.clone(), + namespace: scope.0.clone(), + deadline: Duration::from_secs(2), + }) + .await + .unwrap() } /// Waits until the condition holds, or until the limit ends. Gives whether the condition holds. @@ -164,7 +169,7 @@ async fn eventually(condition: impl Fn() -> bool) -> bool { .is_ok() } -/// Runs the operation until the calls of the storage fulfil the condition, and then drops it. +/// Runs the operation until the calls of the storage match the condition, and then drops it. /// Gives the output of the operation when it ends first. async fn drop_when( storage: &ScriptedBlobStorage, @@ -535,7 +540,7 @@ async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { }); let held = eventually(|| index_writes() > before).await; let deleted = store.delete(&scope, &name("p-old")).await; - let pruned_while_held = ledger(&*storage, &scope).await.last_prune_millis.is_some(); + let pruned_while_held = ledger(&storage, &scope).await.last_prune.is_some(); storage.open_gate(); let saved = saving.await.unwrap(); let pruned_again = store.delete(&scope, &name("p-none")).await; @@ -590,11 +595,14 @@ async fn a_restore_whose_pack_reads_fail_gives_a_retryable_storage_error() { async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_denied() { use std::os::unix::fs::PermissionsExt; // SAFETY: `geteuid` has no preconditions. - if unsafe { libc::geteuid() } == 0 { - return; - } + let uid = unsafe { libc::geteuid() }; + assert_ne!( + uid, 0, + "this test needs a user other than root, because permissions do not stop root from a read" + ); + let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( - Arc::new(InMemoryBlobStorage::new()), + storage.clone(), policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), ); let scope = new_scope(); @@ -609,6 +617,10 @@ async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_d matches!(&saved, Err(SnapshotStoreError::Source(error)) if error.kind() == std::io::ErrorKind::PermissionDenied), "{saved:?}" ); + assert_eq!( + blobs(&*storage, &scope.0, "snapshots/").await, + Vec::::new() + ); } #[cfg(target_os = "linux")] @@ -620,6 +632,7 @@ async fn a_restore_that_cannot_set_an_extended_attribute_gives_destination() { let value = vec![b'a'; 6000]; let set = |path: &Path| xattr_set(path, "user.golem-test", &value); let Ok(source) = tempfile::tempdir_in("/dev/shm") else { + println!("SKIPPED: /dev/shm has no directory for the source of the test"); return; }; let file = source.path().join("file.txt"); @@ -627,7 +640,15 @@ async fn a_restore_that_cannot_set_an_extended_attribute_gives_destination() { let probe = Scratch::new(); let probe_file = probe.path().join("probe"); std::fs::write(&probe_file, b"probe").unwrap(); - if set(&file).is_err() || set(&probe_file).is_ok() { + if let Err(error) = set(&file) { + println!("SKIPPED: /dev/shm does not take a user attribute of 6,000 bytes: {error}"); + return; + } + if set(&probe_file).is_ok() { + println!( + "SKIPPED: the destination {} takes a user attribute of 6,000 bytes", + probe.path().display() + ); return; } let store = store( @@ -731,14 +752,14 @@ async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_per let packs_before = blobs(&*storage, &scope.0, "data/").await; store.delete(&scope, &name("p-deleted")).await.unwrap(); - let after_first = ledger(&*storage, &scope).await; + let after_first = ledger(&storage, &scope).await; store.delete(&scope, &name("p-none")).await.unwrap(); let packs_after = blobs(&*storage, &scope.0, "data/").await; assert_eq!( ( after_first.freed_bytes, - after_first.last_prune_millis.is_some(), + after_first.last_prune.is_some(), after_first.awaiting_removal, packs_after.len() < packs_before.len(), packs_after.iter().all(|pack| packs_before.contains(pack)), @@ -768,12 +789,12 @@ async fn a_delete_below_the_threshold_does_not_prune() { let packs_before = blobs(&*storage, &scope.0, "data/").await; store.delete(&scope, &name("p-deleted")).await.unwrap(); - let after = ledger(&*storage, &scope).await; + let after = ledger(&storage, &scope).await; assert_eq!( ( after.freed_bytes > 0, - after.last_prune_millis, + after.last_prune, blobs(&*storage, &scope.0, "data/").await, ), (true, None, packs_before) @@ -800,14 +821,14 @@ async fn no_second_prune_runs_within_the_grace_period() { .await; store.delete(&scope, &name("p-a")).await.unwrap(); - let after_first = ledger(&*storage, &scope).await; + let after_first = ledger(&storage, &scope).await; store.delete(&scope, &name("p-b")).await.unwrap(); - let after_second = ledger(&*storage, &scope).await; + let after_second = ledger(&storage, &scope).await; assert_eq!( ( - after_first.last_prune_millis.is_some(), - after_second.last_prune_millis == after_first.last_prune_millis, + after_first.last_prune.is_some(), + after_second.last_prune == after_first.last_prune, after_second.freed_bytes > 0, ), (true, true, true) @@ -844,10 +865,10 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { refuse.store(true, Ordering::SeqCst); let failed = store.delete(&scope, &name("p-deleted")).await; - let after_failure = ledger(&*storage, &scope).await; + let after_failure = ledger(&storage, &scope).await; refuse.store(false, Ordering::SeqCst); let retried = store.delete(&scope, &name("p-deleted")).await; - let after_retry = ledger(&*storage, &scope).await; + let after_retry = ledger(&storage, &scope).await; assert!( failed.as_ref().is_err_and(|error| is_storage(error, true)), @@ -856,10 +877,10 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { assert_eq!( ( after_failure.freed_bytes > 0, - after_failure.last_prune_millis, + after_failure.last_prune, retried.is_ok(), after_retry.freed_bytes, - after_retry.last_prune_millis.is_some(), + after_retry.last_prune.is_some(), ), (true, None, true, 0, true) ); @@ -1320,7 +1341,7 @@ async fn the_ledger_counts_the_packed_bytes_that_the_deleted_snapshot_added() { store.delete(&scope, &name("p-deleted")).await.unwrap(); assert_eq!( - (added > 1, ledger(&*storage, &scope).await.freed_bytes), + (added > 1, ledger(&storage, &scope).await.freed_bytes), (true, added) ); } @@ -1350,3 +1371,48 @@ async fn a_config_write_that_fails_gives_a_storage_error_with_that_failure() { "{saved:?}" ); } + +#[test] +async fn a_prune_that_fails_without_a_storage_failure_gives_storage_that_is_not_retryable() { + // Packs of zeros with the sizes of the index give the prune a decryption error, not a failed + // storage call. The forget before the prune has succeeded, so the delete gives `Storage`. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path()) + .await + .unwrap(); + let packs = storage + .list_blobs_below("test", "test", scope.0.clone(), Path::new("data")) + .await + .unwrap(); + futures::future::join_all(packs.iter().map(|pack| { + let zeros = vec![0; usize::try_from(pack.size).unwrap()]; + let storage = storage.clone(); + let namespace = scope.0.clone(); + async move { + storage + .put_raw("test", "test", namespace, &pack.path, &zeros) + .await + } + })) + .await + .into_iter() + .collect::>>() + .unwrap(); + + let deleted = store.delete(&scope, &name("p-deleted")).await; + + assert!( + deleted + .as_ref() + .is_err_and(|error| is_storage(error, false)), + "{deleted:?}" + ); +} From 224e1a4baa6802effa3bf1eec75be0137eae3c8f Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 21:30:06 -0700 Subject: [PATCH 021/126] Run the rule of the scripted test storage one time for each call --- .../filesystem_snapshot/rustic/scripted.rs | 37 ++++++++++++++----- 1 file changed, 27 insertions(+), 10 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs index 8c02e95e1b..0c4946aeee 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -100,9 +100,21 @@ impl ScriptedBlobStorage { op_label: &'static str, path: &Path, call: impl Future>, + ) -> anyhow::Result { + self.follow((self.rule)(op_label, path), op_label, path, call) + .await + } + + /// Records the call and does what the script says. The rule runs one time for each call. + async fn follow( + &self, + script: Script, + op_label: &'static str, + path: &Path, + call: impl Future>, ) -> anyhow::Result { self.record(op_label, path); - match (self.rule)(op_label, path) { + match script { Script::Pass => call.await, Script::Refuse => Err(anyhow::anyhow!("the storage refused the call")), Script::LoseTheAnswer => { @@ -137,16 +149,21 @@ impl BlobStorage for ScriptedBlobStorage { namespace: BlobStorageNamespace, path: &Path, ) -> anyhow::Result>> { - if (self.rule)(op_label, path) == Script::Vanish { - self.record(op_label, path); - return Ok(None); + match (self.rule)(op_label, path) { + Script::Vanish => { + self.record(op_label, path); + Ok(None) + } + script => { + self.follow( + script, + op_label, + path, + self.inner.get_raw(target_label, op_label, namespace, path), + ) + .await + } } - self.answer( - op_label, - path, - self.inner.get_raw(target_label, op_label, namespace, path), - ) - .await } async fn get_stream( From a49c636043d1ca21b6dd472ffe09546daa398652 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 21:30:06 -0700 Subject: [PATCH 022/126] Keep the tree packs of one operation in memory, with one read for each pack --- .../src/filesystem_snapshot/rustic/backend.rs | 47 +++- .../rustic/backend/kept.rs | 120 +++++++++ .../rustic/backend/tests.rs | 232 ++++++++++++++++++ .../src/filesystem_snapshot/rustic/tests.rs | 92 ++++++- 4 files changed, 487 insertions(+), 4 deletions(-) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index c0afa5cd55..8e66cd1214 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -23,7 +23,10 @@ use super::fault::{BlobCallFailed, ConfigExists, FileMissing, OperationCancelled use super::files::TARGET_LABEL; use super::publish::{SnapshotStage, StagedSnapshot}; use bytes::Bytes; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace, PutIfAbsent}; +use golem_service_base::storage::blob::{ + BlobRangeError, BlobStorage, BlobStorageNamespace, PutIfAbsent, +}; +use kept::KeptPacks; use rustic_core::{ BytesList, ErrorKind, FileType, Id, ReadBackend, RusticError, RusticResult, WriteBackend, }; @@ -38,6 +41,9 @@ use tokio_util::task::task_tracker::TaskTrackerToken; /// The path of the config file of a repository. const CONFIG_PATH: &str = "config"; +/// The largest number of bytes of tree packs that one backend keeps in memory. +const KEPT_PACKS_LIMIT: usize = 32 * 1024 * 1024; + /// A call that the backend makes on the blob storage. #[derive(Clone, Copy, Debug, PartialEq, Eq)] enum StorageCall { @@ -92,6 +98,8 @@ pub(super) struct BlobBackend { stage: Option>, /// Counts the backend as work of a tracker, until the last owner drops the backend. _tracked: Option, + /// The packs of tree blobs that the operation of the backend read. + kept: KeptPacks, } impl BlobBackend { @@ -111,6 +119,15 @@ impl BlobBackend { cancel: CancellationToken::new(), stage: None, _tracked: None, + kept: KeptPacks::new(KEPT_PACKS_LIMIT), + } + } + + /// Gives the backend with another limit of the bytes of the tree packs that it keeps. + pub(super) fn keeping_packs_up_to(self, limit: usize) -> Self { + Self { + kept: KeptPacks::new(limit), + ..self } } @@ -268,7 +285,7 @@ impl ReadBackend for BlobBackend { &self, tpe: FileType, id: &Id, - _cacheable: bool, + cacheable: bool, offset: u32, length: u32, ) -> RusticResult { @@ -276,6 +293,11 @@ impl ReadBackend for BlobBackend { let Some(last) = length.checked_sub(1) else { return Ok(Bytes::new()); }; + // rustic marks the reads of tree blobs as cacheable, and reads each tree blob on its own. + if cacheable && tpe == FileType::Pack { + let pack = self.kept.get_or_read(id, || self.read_full(tpe, id))?; + return range_of(&pack, &path, offset, last); + } let start = u64::from(offset); self.request( StorageCall::ReadRange, @@ -406,6 +428,25 @@ fn join(parts: &[Bytes]) -> Box<[u8]> { .into_boxed_slice() } +/// Gives the bytes from `offset` to `last` of the pack as a slice of the pack. A range outside the +/// pack gives the error of a ranged read outside a blob. +fn range_of(pack: &Bytes, path: &Path, offset: u32, last: u32) -> RusticResult { + let start = u64::from(offset); + let end = start + u64::from(last); + usize::try_from(start) + .ok() + .zip(usize::try_from(end).ok()) + .filter(|(_, end)| *end < pack.len()) + .map(|(start, end)| pack.slice(start..=end)) + .ok_or_else(|| { + storage_error( + StorageCall::ReadRange, + path, + anyhow::Error::new(BlobRangeError { start, end }), + ) + }) +} + /// The error of a file that the blob storage does not hold. fn missing_file(path: &Path) -> Box { RusticError::with_source( @@ -446,5 +487,7 @@ fn storage_error(call: StorageCall, path: &Path, error: anyhow::Error) -> Box, + /// Wakes the threads that wait for the read of a pack when that read ends. + read_ended: Condvar, +} + +#[derive(Default)] +struct State { + packs: HashMap, + bytes: usize, + reading: HashSet, +} + +impl KeptPacks { + /// Gives an empty set that keeps packs up to `limit` bytes in total. + pub(super) fn new(limit: usize) -> Self { + Self { + limit, + state: Mutex::default(), + read_ended: Condvar::new(), + } + } + + /// Gives the kept pack, or reads it with `read`. While one thread reads a pack, the other + /// threads that want it wait for that read, and then take the kept pack or read it again. A + /// pack is kept only when its read succeeds and it fits in the limit. + pub(super) fn get_or_read( + &self, + id: &Id, + read: impl FnOnce() -> RusticResult, + ) -> RusticResult { + let mut state = self + .read_ended + .wait_while(self.state(), |state| state.reading.contains(id)) + .unwrap_or_else(PoisonError::into_inner); + if let Some(pack) = state.packs.get(id) { + return Ok(pack.clone()); + } + state.reading.insert(*id); + drop(state); + let reading = Reading { + kept: self, + id: *id, + }; + let read = read(); + if let Ok(pack) = &read { + reading.keep(pack); + } + read + } + + fn state(&self) -> MutexGuard<'_, State> { + self.state.lock().unwrap_or_else(PoisonError::into_inner) + } +} + +impl Debug for KeptPacks { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + let state = self.state(); + formatter + .debug_struct("KeptPacks") + .field("limit", &self.limit) + .field("packs", &state.packs.len()) + .field("bytes", &state.bytes) + .finish() + } +} + +/// The read of one pack by one thread. Its drop ends the read and wakes the waiting threads, also +/// when the read fails or panics. +struct Reading<'a> { + kept: &'a KeptPacks, + id: Id, +} + +impl Reading<'_> { + /// Keeps the pack when it fits in the limit. + fn keep(&self, pack: &Bytes) { + let mut state = self.kept.state(); + let bytes = state.bytes.saturating_add(pack.len()); + if bytes <= self.kept.limit { + state.bytes = bytes; + state.packs.insert(self.id, pack.clone()); + } + } +} + +impl Drop for Reading<'_> { + fn drop(&mut self) { + self.kept.state().reading.remove(&self.id); + self.kept.read_ended.notify_all(); + } +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index 413763cdc5..994c7b88ea 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -478,6 +478,238 @@ fn a_thread_that_is_not_a_thread_of_the_runtime_can_call_the_backend() { assert_eq!(read.ok().flatten(), Some(Bytes::from_static(b"index"))); } +/// The content of the pack of the tests of the kept packs: 100 bytes, each its own offset. +fn pack_content() -> Vec { + (0..100).collect() +} + +/// A backend over a storage that holds one pack at the path of the id `ab`, with the rule of +/// the storage and the limit of the kept packs. The storage records each call. +struct PackFixture { + _runtime: Runtime, + storage: Arc, + backend: Arc, +} + +impl PackFixture { + fn new(limit: usize, rule: impl Fn(&str, &Path) -> Script + Send + Sync + 'static) -> Self { + let runtime = Runtime::new().unwrap(); + let inner = Arc::new(InMemoryBlobStorage::new()); + let namespace = new_namespace(); + runtime + .block_on(inner.put_raw( + "test", + "test", + namespace.clone(), + Path::new(&format!("data/ab/{}", "ab".repeat(32))), + &pack_content(), + )) + .unwrap(); + let storage = ScriptedBlobStorage::new(inner, rule); + let backend = BlobBackend::new( + storage.clone(), + namespace, + runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .keeping_packs_up_to(limit); + Self { + _runtime: runtime, + storage, + backend: Arc::new(backend), + } + } + + /// Gives the operation label of each call on the pack. + fn pack_calls(&self) -> Vec<&'static str> { + self.storage + .calls() + .into_iter() + .filter(|(_, path)| path.starts_with("data/")) + .map(|(op_label, _)| op_label) + .collect() + } +} + +/// Reads the range of the pack as a range of tree blobs, which rustic marks as cacheable. +fn tree_range(backend: &BlobBackend, offset: u32, length: u32) -> RusticResult { + backend.read_partial(FileType::Pack, &id("ab"), true, offset, length) +} + +#[test] +fn a_later_range_of_a_kept_pack_makes_no_storage_call() { + let fixture = PackFixture::new(1024, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + ( + tree_range(&backend, 0, 10).ok(), + tree_range(&backend, 20, 10).ok(), + tree_range(&backend, 90, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)), + Some(Bytes::from_iter(90..100)), + )), + vec!["read"] + ) + ); +} + +#[test] +fn a_range_that_is_not_cacheable_is_a_ranged_read_each_time() { + let fixture = PackFixture::new(1024, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + [(0, 10), (20, 10)].map(|(offset, length)| { + backend + .read_partial(FileType::Pack, &id("ab"), false, offset, length) + .ok() + }) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some([ + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)) + ]), + vec!["read_range", "read_range"] + ) + ); +} + +#[test] +fn two_threads_that_miss_one_pack_make_one_storage_read() { + // The first read waits at the gate. The second thread starts while it waits, and the gate + // opens only after the second thread had time to ask for the pack. + let fixture = PackFixture::new(1024, |op_label, _| { + if op_label == "read" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let first = std::thread::spawn({ + let backend = fixture.backend.clone(); + move || tree_range(&backend, 0, 10).ok() + }); + let first_read_started = (0..1000).any(|_| { + std::thread::sleep(Duration::from_millis(10)); + !fixture.pack_calls().is_empty() + }); + let second = std::thread::spawn({ + let backend = fixture.backend.clone(); + move || tree_range(&backend, 50, 10).ok() + }); + std::thread::sleep(Duration::from_millis(200)); + fixture.storage.open_gate(); + + assert_eq!( + ( + first_read_started, + first.join().ok().flatten(), + second.join().ok().flatten(), + fixture.pack_calls() + ), + ( + true, + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(50..60)), + vec!["read"] + ) + ); +} + +#[test] +fn a_pack_over_the_limit_is_read_again_at_its_next_range() { + let fixture = PackFixture::new(99, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + ( + tree_range(&backend, 0, 10).ok(), + tree_range(&backend, 20, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)) + )), + vec!["read", "read"] + ) + ); +} + +#[test] +fn a_failed_read_of_a_pack_is_not_kept_and_the_next_range_reads_again() { + let refused = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let fixture = PackFixture::new(1024, { + let refused = refused.clone(); + move |op_label, _| { + if op_label == "read" && !refused.swap(true, std::sync::atomic::Ordering::SeqCst) { + Script::Refuse + } else { + Script::Pass + } + } + }); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + ( + tree_range(&backend, 0, 10).is_err(), + tree_range(&backend, 20, 10).ok(), + tree_range(&backend, 40, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + true, + Some(Bytes::from_iter(20..30)), + Some(Bytes::from_iter(40..50)) + )), + vec!["read", "read"] + ) + ); +} + +#[test] +fn a_range_outside_a_kept_pack_gives_the_error_of_a_ranged_read_outside_a_blob() { + let fixture = PackFixture::new(1024, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let outside = within_limit(move || { + [(90, 11), (100, 1), (u32::MAX, 1)].map(|(offset, length)| { + tree_range(&backend, offset, length).err().map(|error| { + ( + text_of(&error).contains("is not in the blob"), + classify(Operation::Restore, anyhow::Error::new(error)) + .to_string() + .contains("storage"), + ) + }) + }) + }); + + assert_eq!(outside, Some([Some((true, true)); 3])); +} + #[test] fn a_tracked_backend_counts_in_its_tracker_until_it_drops() { let fixture = Fixture::new(); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index 6d9fa71445..a50db65f08 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -20,6 +20,7 @@ use super::backend::BlobBackend; use super::holding::{holding_storage, reached_deadline}; +use super::scripted::{Script, ScriptedBlobStorage}; use super::{ ChangeDetection, Chunking, Compression, OperationPhase, PruneSettings, RepackLimits, Repository, RepositoryKey, RepositorySettings, SaveSettings, backup_options, config_options, @@ -140,6 +141,52 @@ async fn data_packs(storage: &Arc, scope: &SnapshotScope) - .unwrap() } +/// Gives the id in hex of each pack of tree blobs in the repository of the scope. +async fn tree_packs(storage: &Arc, scope: &SnapshotScope) -> Box<[Box]> { + with_existing_repository( + storage.clone(), + scope, + STORAGE_CALL_DEADLINE, + |repository| { + let indexes = repository + .stream_files::()? + .collect::>>()?; + Ok(indexes + .into_iter() + .flat_map(|(_, index)| index.packs) + .filter(|pack| pack.blob_type() == BlobType::Tree) + .map(|pack| Box::from(pack.id.to_hex().as_str())) + .collect()) + }, + ) + .await + .unwrap() +} + +/// Writes a tree of `count` directories, each with one small file, into a new directory. +fn many_directories_tree(count: usize) -> Scratch { + let tree = Scratch::new(); + let names = (0..count) + .flat_map(|index| [format!("dir-{index}"), format!("dir-{index}/file.txt")]) + .collect::>(); + let entries = names + .iter() + .map(|name| { + let spec = if name.ends_with(".txt") { + Spec::File { + content: Box::from(name.as_bytes()), + mode: 0o644, + } + } else { + Spec::Directory { mode: 0o755 } + }; + (name.as_str(), spec) + }) + .collect::>(); + write_tree(tree.path(), &entries); + tree +} + /// Prunes the repository of the scope with the options, on a blocking thread. Each call on the /// storage waits for at most `deadline`. async fn prune( @@ -239,6 +286,46 @@ async fn a_restore_report_gives_each_phase_of_the_restore_in_order() { ); } +#[test] +async fn a_restore_reads_each_tree_pack_one_time_in_full_and_no_range_of_a_tree_pack() { + let inner = Arc::new(InMemoryBlobStorage::new()); + let scope = new_scope(); + let tree = many_directories_tree(60); + repository(&inner, &scope) + .save(&name("first"), tree.path()) + .await + .unwrap(); + let tree_packs = tree_packs(&inner, &scope).await; + let storage = ScriptedBlobStorage::new(inner.clone(), |_, _| Script::Pass); + let into = Scratch::new(); + + Repository::new(storage.clone(), scope.clone(), key(), STORAGE_CALL_DEADLINE) + .restore(&name("first"), into.path(), None) + .await + .unwrap(); + let calls_on_tree_packs = |op: &str| { + tree_packs + .iter() + .map(|pack| { + storage + .calls() + .iter() + .filter(|(op_label, path)| *op_label == op && path.ends_with(&**pack)) + .count() + }) + .collect::>() + }; + + assert_eq!( + ( + calls_on_tree_packs("read"), + calls_on_tree_packs("read_range").iter().sum::(), + listing(into.path()) + ), + (vec![1; tree_packs.len()], 0, listing(tree.path())) + ); +} + #[test] async fn a_second_save_has_the_first_as_parent_and_reads_only_the_changed_file() { let storage = Arc::new(InMemoryBlobStorage::new()); @@ -785,7 +872,7 @@ async fn a_restore_whose_data_pack_reads_get_no_answer_fails_and_stops_its_threa } #[test] -async fn a_prune_whose_pack_reads_get_no_answer_fails_and_stops_its_threads() { +async fn a_prune_whose_tree_pack_reads_get_no_answer_fails_and_stops_its_threads() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); let tree = fixture_tree(); @@ -793,8 +880,9 @@ async fn a_prune_whose_pack_reads_get_no_answer_fails_and_stops_its_threads() { .save(&name("first"), tree.path()) .await .unwrap(); + // A prune reads the trees of the snapshots, and each tree read is a full read of a pack. let (storage, _gate, dropped) = holding_storage(inner, |op_label, path| { - op_label == "read_range" && path.starts_with("data") + op_label == "read" && path.starts_with("data") }); let pruned = tokio::time::timeout( From ccd253e260ab8122bd82aae946271018ed0f9a32 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:38:09 -0700 Subject: [PATCH 023/126] Run saves and prunes at nice 19 on their own threads --- golem-worker-executor/Cargo.toml | 5 +- .../src/filesystem_snapshot/rustic/mod.rs | 1 + .../filesystem_snapshot/rustic/priority.rs | 109 +++++++++++ .../rustic/priority/tests.rs | 57 ++++++ .../src/filesystem_snapshot/rustic/store.rs | 9 +- .../filesystem_snapshot/rustic/store/tests.rs | 172 ++++++++++++++++++ 6 files changed, 348 insertions(+), 5 deletions(-) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs diff --git a/golem-worker-executor/Cargo.toml b/golem-worker-executor/Cargo.toml index cd1e5ff1ec..d54d18649a 100644 --- a/golem-worker-executor/Cargo.toml +++ b/golem-worker-executor/Cargo.toml @@ -13,7 +13,7 @@ autotests = false [features] test-utils = [] -fs-snapshot-benchmark = ["dep:clap", "dep:rayon"] +fs-snapshot-benchmark = ["dep:clap"] [lib] path = "src/lib.rs" @@ -96,7 +96,7 @@ prometheus = { workspace = true } prost = { workspace = true } prost-types = { workspace = true } rand = { workspace = true } -rayon = { workspace = true, optional = true } +rayon = { workspace = true } regex = { workspace = true } ringbuf = { workspace = true } rustic_core = { workspace = true } @@ -169,7 +169,6 @@ goldenfile = { workspace = true } pretty_assertions = { workspace = true, features = [ "unstable" ] } proptest = { workspace = true } rand = { workspace = true } -rayon = { workspace = true } redis = { workspace = true } serde_json = { workspace = true } test-r = { workspace = true } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index cd02c37a1d..ddcebdba84 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -21,6 +21,7 @@ mod backend; mod fault; mod files; +mod priority; mod prune; mod publish; mod scope; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs new file mode 100644 index 0000000000..31b746eaa4 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs @@ -0,0 +1,109 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Saves and prunes run at a low CPU priority, so the work of the agents comes first. +//! +//! A thread without privilege cannot raise its priority again after it lowers it. So the work runs +//! on a new thread that ends with the work, and no thread of a pool gets the low priority. + +use std::sync::{Arc, Mutex, PoisonError}; +use tracing::warn; + +/// The nice value of the threads of a save or a prune. +#[cfg(target_os = "linux")] +pub(super) const LOW_PRIORITY: i32 = 19; + +/// Runs the work at nice 19 on a new thread with the name, and waits for it. The threads that the +/// work starts get the same nice value. On a platform other than Linux the work runs as it is. +pub(super) fn at_low_priority( + name: &str, + work: impl FnOnce() -> anyhow::Result + Send + 'static, +) -> anyhow::Result { + #[cfg(target_os = "linux")] + { + // The global rayon pool of rustic starts at its first use. It starts here, so its threads + // keep the normal priority and a restore that uses them does not run at nice 19. + let _ = rayon::current_num_threads(); + on_own_thread(name, work, lower_own_priority) + } + #[cfg(not(target_os = "linux"))] + { + let _ = name; + work() + } +} + +/// Runs the work on a new thread that first calls `lower`, and waits for it. A failed `lower` or a +/// failed start of the thread gives a warning, and the work runs at the normal priority. +fn on_own_thread( + name: &str, + work: impl FnOnce() -> anyhow::Result + Send + 'static, + lower: fn() -> std::io::Result<()>, +) -> anyhow::Result { + // The work waits in a slot, so the calling thread can still run it when no thread starts. + let slot = Arc::new(Mutex::new(Some(work))); + let spawned = std::thread::Builder::new().name(name.to_string()).spawn({ + let slot = slot.clone(); + move || { + if let Err(error) = lower() { + warn!( + error = %error, + "Failed to lower the CPU priority of filesystem snapshot work, so it runs at the normal priority" + ); + } + take(&slot).map_or_else(|| Err(anyhow::anyhow!("the work was taken")), |work| work()) + } + }); + match spawned { + Ok(thread) => thread + .join() + .map_err(|_| anyhow::anyhow!("the thread of the filesystem snapshot work panicked"))?, + Err(error) => { + warn!( + error = %error, + "Failed to start a thread for filesystem snapshot work, so it runs at the normal priority" + ); + take(&slot).map_or_else(|| Err(anyhow::anyhow!("the work was taken")), |work| work()) + } + } +} + +fn take(slot: &Mutex>) -> Option { + slot.lock().unwrap_or_else(PoisonError::into_inner).take() +} + +/// Gives the calling thread the nice value 19. +#[cfg(target_os = "linux")] +fn lower_own_priority() -> std::io::Result<()> { + // SAFETY: `gettid` has no preconditions. + let thread = unsafe { libc::gettid() }; + let thread = libc::id_t::try_from(thread).map_err(std::io::Error::other)?; + // SAFETY: `setpriority` only reads its arguments. + let result = unsafe { libc::setpriority(libc::PRIO_PROCESS, thread, LOW_PRIORITY) }; + if result == 0 { + Ok(()) + } else { + Err(std::io::Error::last_os_error()) + } +} + +/// Gives the nice value of the calling thread. +#[cfg(all(test, target_os = "linux"))] +pub(super) fn own_nice() -> i32 { + // SAFETY: `gettid` has no preconditions, and `getpriority` only reads its arguments. + unsafe { libc::getpriority(libc::PRIO_PROCESS, libc::gettid() as libc::id_t) } +} + +#[cfg(all(test, target_os = "linux"))] +mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs new file mode 100644 index 0000000000..0a08a433a0 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs @@ -0,0 +1,57 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use super::{LOW_PRIORITY, at_low_priority, on_own_thread, own_nice}; +use pretty_assertions::assert_eq; +use test_r::test; + +#[test] +fn work_at_low_priority_runs_at_nice_19_on_its_own_thread_and_the_caller_keeps_its_priority() { + let before = own_nice(); + + let inside = at_low_priority("fs-snap-test", || { + Ok(( + own_nice(), + std::thread::current().name().map(str::to_string), + )) + }); + + assert_eq!( + (inside.ok(), own_nice()), + ( + Some((LOW_PRIORITY, Some("fs-snap-test".to_string()))), + before + ) + ); +} + +#[test] +fn work_whose_priority_cannot_be_lowered_still_runs_at_the_normal_priority() { + let before = own_nice(); + + let done = on_own_thread( + "fs-snap-test", + || Ok(own_nice()), + || Err(std::io::Error::other("the priority cannot change here")), + ); + + assert_eq!(done.ok(), Some(before)); +} + +#[test] +fn a_panic_of_the_work_gives_an_error() { + let done = on_own_thread::<()>("fs-snap-test", || panic!("the work panics"), || Ok(())); + + assert!(done.is_err()); +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 63849aa514..5c1a30a753 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -23,6 +23,7 @@ use super::backend::BlobBackend; use super::fault::{Operation, classify, is_file_missing, is_storage_failure, storage_failure}; use super::files::SnapshotFiles; +use super::priority::at_low_priority; use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -261,7 +262,9 @@ impl RusticSnapshotStore { let key = self.key.clone(); let settings = self.policy.prune; let report = self - .blocking(Operation::Prune, move || prune(backend, &key, &settings)) + .blocking(Operation::Prune, move || { + at_low_priority("fs-snap-prune", move || prune(backend, &key, &settings)) + }) .await?; let marked_packs = report.as_ref().is_some_and(leaves_marked_packs); write_ledger(&files, &PruneLedger::after_prune(now, marked_packs)) @@ -288,7 +291,9 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { let tree: Box = tree.into(); let staged = self .blocking(Operation::Save, move || { - stage_save(backend, &stage, &key, &policy, &name, &tree) + at_low_priority("fs-snap-save", move || { + stage_save(backend, &stage, &key, &policy, &name, &tree) + }) }) .await?; let (staged, info) = staged.ok_or(SnapshotStoreError::AlreadyExists)?; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 1dc3d1deb4..288a66fd16 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1416,3 +1416,175 @@ async fn a_prune_that_fails_without_a_storage_failure_gives_storage_that_is_not_ "{deleted:?}" ); } + +/// The operation label, the path and the nice value of the calling thread of each storage call. +#[cfg(target_os = "linux")] +type NiceCalls = Arc>>; + +/// A storage that records the nice value of the thread of each call. +#[cfg(target_os = "linux")] +fn nice_recording_storage() -> (Arc, NiceCalls) { + let calls = NiceCalls::default(); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let calls = calls.clone(); + move |op_label, path| { + calls + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .push(( + op_label.to_string(), + path.display().to_string(), + super::super::priority::own_nice(), + )); + Script::Pass + } + }); + (storage, calls) +} + +/// Takes the recorded calls with the operation label. +#[cfg(target_os = "linux")] +fn taken_calls(calls: &NiceCalls, op_label: &str) -> Vec<(String, i32)> { + std::mem::take( + &mut *calls + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner), + ) + .into_iter() + .filter(|(op, _, _)| op == op_label) + .map(|(_, path, nice)| (path, nice)) + .collect() +} + +#[cfg(target_os = "linux")] +#[test] +async fn the_writes_of_a_save_run_at_nice_19() { + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = fixture_tree(); + + store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + let writes = taken_calls(&calls, "write"); + + assert_eq!( + ( + writes.is_empty(), + writes + .iter() + .filter(|(_, nice)| *nice != 19) + .collect::>() + ), + (false, Vec::<&(String, i32)>::new()) + ); +} + +#[cfg(target_os = "linux")] +#[test] +async fn the_writes_of_a_prune_run_at_nice_19() { + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, 1, Duration::ZERO)); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path()) + .await + .unwrap(); + taken_calls(&calls, "write"); + + store.delete(&scope, &name("p-deleted")).await.unwrap(); + let writes = taken_calls(&calls, "write"); + + assert_eq!( + ( + writes.is_empty(), + writes + .iter() + .filter(|(_, nice)| *nice != 19) + .collect::>() + ), + (false, Vec::<&(String, i32)>::new()) + ); +} + +#[cfg(target_os = "linux")] +#[test] +async fn the_storage_calls_of_a_restore_run_at_the_nice_value_of_the_process() { + let process_nice = super::super::priority::own_nice(); + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = fixture_tree(); + store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + std::mem::take(&mut *calls.lock().unwrap()); + + let restored = restored_listing(&store, &scope, &name("p-1")).await; + let recorded = std::mem::take(&mut *calls.lock().unwrap()); + + assert_eq!( + ( + restored.ok(), + recorded.is_empty(), + recorded + .iter() + .filter(|(_, _, nice)| *nice != process_nice) + .collect::>() + ), + ( + Some(listing(tree.path())), + false, + Vec::<&(String, String, i32)>::new() + ) + ); +} + +#[cfg(target_os = "linux")] +#[test] +async fn after_saves_and_prunes_the_pools_keep_the_nice_value_of_the_process() { + // All tasks wait for each other, so each runs on its own thread of the blocking pool, and the + // idle threads that ran the saves and the prune are among them. + const TASKS: usize = 16; + let process_nice = super::super::priority::own_nice(); + let store = store( + Arc::new(InMemoryBlobStorage::new()), + policy(LONG_DEADLINE, 1, Duration::ZERO), + ); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path()) + .await + .unwrap(); + store.delete(&scope, &name("p-deleted")).await.unwrap(); + + let barrier = Arc::new(std::sync::Barrier::new(TASKS)); + let blocking = futures::future::join_all((0..TASKS).map(|_| { + let barrier = barrier.clone(); + tokio::task::spawn_blocking(move || { + barrier.wait(); + super::super::priority::own_nice() + }) + })) + .await + .into_iter() + .map(Result::unwrap) + .collect::>(); + let rayon = rayon::broadcast(|_| super::super::priority::own_nice()); + + assert_eq!( + ( + blocking.iter().all(|nice| *nice == process_nice), + rayon.iter().all(|nice| *nice == process_nice), + ), + (true, true), + "{blocking:?} {rayon:?}" + ); +} From a359a2bab6d2fd97b879bdb8180f9eb29044e06f Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:02:53 -0700 Subject: [PATCH 024/126] Give each save and prune a rayon pool of its own, and start the global rayon pool when the store is made --- .../filesystem_snapshot/rustic/priority.rs | 148 ++++++++++++------ .../rustic/priority/tests.rs | 78 ++++++--- .../src/filesystem_snapshot/rustic/store.rs | 14 +- .../filesystem_snapshot/rustic/store/tests.rs | 128 ++++++++++++++- 4 files changed, 293 insertions(+), 75 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs index 31b746eaa4..e03448bef9 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs @@ -15,72 +15,119 @@ //! Saves and prunes run at a low CPU priority, so the work of the agents comes first. //! //! A thread without privilege cannot raise its priority again after it lowers it. So the work runs -//! on a new thread that ends with the work, and no thread of a pool gets the low priority. +//! on a new thread, with a rayon pool of its own, and both end with the work. +use rayon::{ThreadPool, ThreadPoolBuildError, ThreadPoolBuilder}; +use std::num::NonZeroUsize; use std::sync::{Arc, Mutex, PoisonError}; use tracing::warn; /// The nice value of the threads of a save or a prune. -#[cfg(target_os = "linux")] pub(super) const LOW_PRIORITY: i32 = 19; -/// Runs the work at nice 19 on a new thread with the name, and waits for it. The threads that the -/// work starts get the same nice value. On a platform other than Linux the work runs as it is. -pub(super) fn at_low_priority( - name: &str, - work: impl FnOnce() -> anyhow::Result + Send + 'static, -) -> anyhow::Result { - #[cfg(target_os = "linux")] - { - // The global rayon pool of rustic starts at its first use. It starts here, so its threads - // keep the normal priority and a restore that uses them does not run at nice 19. - let _ = rayon::current_num_threads(); - on_own_thread(name, work, lower_own_priority) +/// How the store runs work at a low priority: the thread count of the rayon pool of the work, and +/// the two steps that a test can replace. +#[derive(Clone, Copy, Debug)] +pub(super) struct LowPriority { + /// The threads of the rayon pool of the work. `None` is the default count of rayon. + pub(super) threads: Option, + /// Gives the calling thread the low priority. + pub(super) lower: fn() -> std::io::Result<()>, + /// Builds the rayon pool of the work, with the name and the thread count. + pub(super) build_pool: + fn(&str, Option) -> Result, +} + +impl LowPriority { + /// Gives the steps of the platform, with a rayon pool of `threads` threads. + pub(super) fn new(threads: Option) -> Self { + Self { + threads, + lower: lower_own_priority, + build_pool, + } } - #[cfg(not(target_os = "linux"))] - { - let _ = name; - work() + + /// Runs the work at nice 19 on a new thread with the name, inside a new rayon pool, and waits + /// for it. The threads that the work starts get the same nice value. On a platform other than + /// Linux the work runs as it is. + pub(super) fn run( + self, + name: &str, + work: impl FnOnce() -> anyhow::Result + Send + 'static, + ) -> anyhow::Result { + if cfg!(target_os = "linux") { + self.on_own_thread(name, work) + } else { + work() + } } -} -/// Runs the work on a new thread that first calls `lower`, and waits for it. A failed `lower` or a -/// failed start of the thread gives a warning, and the work runs at the normal priority. -fn on_own_thread( - name: &str, - work: impl FnOnce() -> anyhow::Result + Send + 'static, - lower: fn() -> std::io::Result<()>, -) -> anyhow::Result { - // The work waits in a slot, so the calling thread can still run it when no thread starts. - let slot = Arc::new(Mutex::new(Some(work))); - let spawned = std::thread::Builder::new().name(name.to_string()).spawn({ - let slot = slot.clone(); - move || { - if let Err(error) = lower() { + /// Runs the work on a new thread that first lowers its priority and builds the pool. A failure + /// of a step gives a warning, and the work runs without that step. + fn on_own_thread( + self, + name: &str, + work: impl FnOnce() -> anyhow::Result + Send + 'static, + ) -> anyhow::Result { + // The work waits in a slot, so the calling thread can still run it when no thread starts. + let slot = Arc::new(Mutex::new(Some(work))); + let pool_name = name.to_string(); + let spawned = std::thread::Builder::new().name(name.to_string()).spawn({ + let slot = slot.clone(); + move || { + if let Err(error) = (self.lower)() { + warn!( + error = %error, + "Failed to lower the CPU priority of filesystem snapshot work, so it runs at the normal priority" + ); + } + match (self.build_pool)(&pool_name, self.threads) { + Ok(pool) => pool.install(|| run_taken(&slot)), + Err(error) => { + warn!( + error = %error, + "Failed to build the thread pool of filesystem snapshot work, so its parallel parts use the global pool" + ); + run_taken(&slot) + } + } + } + }); + match spawned { + Ok(thread) => thread.join().map_err(|_| { + anyhow::anyhow!("the thread of the filesystem snapshot work panicked") + })?, + Err(error) => { warn!( error = %error, - "Failed to lower the CPU priority of filesystem snapshot work, so it runs at the normal priority" + "Failed to start a thread for filesystem snapshot work, so it runs at the normal priority" ); + run_taken(&slot) } - take(&slot).map_or_else(|| Err(anyhow::anyhow!("the work was taken")), |work| work()) - } - }); - match spawned { - Ok(thread) => thread - .join() - .map_err(|_| anyhow::anyhow!("the thread of the filesystem snapshot work panicked"))?, - Err(error) => { - warn!( - error = %error, - "Failed to start a thread for filesystem snapshot work, so it runs at the normal priority" - ); - take(&slot).map_or_else(|| Err(anyhow::anyhow!("the work was taken")), |work| work()) } } } -fn take(slot: &Mutex>) -> Option { - slot.lock().unwrap_or_else(PoisonError::into_inner).take() +/// Takes the work out of the slot and runs it. +fn run_taken anyhow::Result>(slot: &Mutex>) -> anyhow::Result { + let work = slot.lock().unwrap_or_else(PoisonError::into_inner).take(); + work.map_or_else( + || Err(anyhow::anyhow!("the work already ran")), + |work| work(), + ) +} + +/// Builds a rayon pool whose threads get the nice value of the calling thread. +fn build_pool( + name: &str, + threads: Option, +) -> Result { + let name = name.to_string(); + ThreadPoolBuilder::new() + .num_threads(threads.map_or(0, NonZeroUsize::get)) + .thread_name(move |index| format!("{name}-{index}")) + .build() } /// Gives the calling thread the nice value 19. @@ -98,6 +145,11 @@ fn lower_own_priority() -> std::io::Result<()> { } } +#[cfg(not(target_os = "linux"))] +fn lower_own_priority() -> std::io::Result<()> { + Ok(()) +} + /// Gives the nice value of the calling thread. #[cfg(all(test, target_os = "linux"))] pub(super) fn own_nice() -> i32 { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs index 0a08a433a0..6610032d75 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs @@ -12,46 +12,88 @@ // See the License for the specific language governing permissions and // limitations under the License. -use super::{LOW_PRIORITY, at_low_priority, on_own_thread, own_nice}; +use super::{LOW_PRIORITY, LowPriority, own_nice}; use pretty_assertions::assert_eq; +use rayon::{ThreadPool, ThreadPoolBuildError, ThreadPoolBuilder}; +use std::num::NonZeroUsize; use test_r::test; +/// What the work sees: its nice value, its thread name, whether it runs in a rayon pool, and +/// the thread count of the current rayon pool. +fn seen() -> anyhow::Result<(i32, Option, bool, usize)> { + Ok(( + own_nice(), + std::thread::current().name().map(str::to_string), + rayon::current_thread_index().is_some(), + rayon::current_num_threads(), + )) +} + +/// A pool builder that cannot start a thread. +fn no_pool(_: &str, _: Option) -> Result { + ThreadPoolBuilder::new() + .num_threads(1) + .spawn_handler(|_| Err(std::io::Error::other("no thread can start here"))) + .build() +} + #[test] -fn work_at_low_priority_runs_at_nice_19_on_its_own_thread_and_the_caller_keeps_its_priority() { +fn work_at_low_priority_runs_at_nice_19_in_a_pool_of_its_own_and_the_caller_keeps_its_priority() { let before = own_nice(); - let inside = at_low_priority("fs-snap-test", || { - Ok(( - own_nice(), - std::thread::current().name().map(str::to_string), - )) - }); + let inside = LowPriority::new(NonZeroUsize::new(3)).run("fs-snap-test", seen); assert_eq!( - (inside.ok(), own_nice()), ( - Some((LOW_PRIORITY, Some("fs-snap-test".to_string()))), - before - ) + inside.ok().map(|(nice, name, in_pool, threads)| ( + nice, + name.is_some_and(|name| name.starts_with("fs-snap-test-")), + in_pool, + threads + )), + own_nice() + ), + (Some((LOW_PRIORITY, true, true, 3)), before) ); } #[test] fn work_whose_priority_cannot_be_lowered_still_runs_at_the_normal_priority() { let before = own_nice(); + let low_priority = LowPriority { + lower: || Err(std::io::Error::other("the priority cannot change here")), + ..LowPriority::new(NonZeroUsize::new(2)) + }; + + let inside = low_priority.run("fs-snap-test", seen); - let done = on_own_thread( - "fs-snap-test", - || Ok(own_nice()), - || Err(std::io::Error::other("the priority cannot change here")), + assert_eq!( + inside.ok().map(|(nice, _, in_pool, _)| (nice, in_pool)), + Some((before, true)) ); +} + +#[test] +fn work_without_its_pool_still_runs_at_nice_19_on_its_own_thread() { + let low_priority = LowPriority { + build_pool: no_pool, + ..LowPriority::new(NonZeroUsize::new(2)) + }; - assert_eq!(done.ok(), Some(before)); + let inside = low_priority.run("fs-snap-test", seen); + + assert_eq!( + inside + .ok() + .map(|(nice, name, in_pool, _)| (nice, name, in_pool)), + Some((LOW_PRIORITY, Some("fs-snap-test".to_string()), false)) + ); } #[test] fn a_panic_of_the_work_gives_an_error() { - let done = on_own_thread::<()>("fs-snap-test", || panic!("the work panics"), || Ok(())); + let done = LowPriority::new(NonZeroUsize::new(1)) + .run::<()>("fs-snap-test", || panic!("the work panics")); assert!(done.is_err()); } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 5c1a30a753..a95f140817 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -23,7 +23,7 @@ use super::backend::BlobBackend; use super::fault::{Operation, classify, is_file_missing, is_storage_failure, storage_failure}; use super::files::SnapshotFiles; -use super::priority::at_low_priority; +use super::priority::LowPriority; use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -129,6 +129,8 @@ pub(crate) struct RusticSnapshotStore { root: CancellationToken, /// Counts the blocking tasks, the backends and the deletes of dropped publishes. tracker: TaskTracker, + /// Runs saves and prunes at a low priority. + low_priority: LowPriority, } impl RusticSnapshotStore { @@ -149,12 +151,16 @@ impl RusticSnapshotStore { key: RepositoryKey, policy: StorePolicy, ) -> Self { + // The global rayon pool starts at its first use, and its threads keep the priority of the + // thread that starts it. It starts here, at the normal priority, before a save or a prune. + let _ = rayon::current_num_threads(); Self { storage, key, policy, root: CancellationToken::new(), tracker: TaskTracker::new(), + low_priority: LowPriority::new(policy.save.threads), } } @@ -261,9 +267,10 @@ impl RusticSnapshotStore { let backend = Arc::new(self.backend(scope, token)?); let key = self.key.clone(); let settings = self.policy.prune; + let low_priority = self.low_priority; let report = self .blocking(Operation::Prune, move || { - at_low_priority("fs-snap-prune", move || prune(backend, &key, &settings)) + low_priority.run("fs-snap-prune", move || prune(backend, &key, &settings)) }) .await?; let marked_packs = report.as_ref().is_some_and(leaves_marked_packs); @@ -289,9 +296,10 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { let policy = self.policy; let name = name.clone(); let tree: Box = tree.into(); + let low_priority = self.low_priority; let staged = self .blocking(Operation::Save, move || { - at_low_priority("fs-snap-save", move || { + low_priority.run("fs-snap-save", move || { stage_save(backend, &stage, &key, &policy, &name, &tree) }) }) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 288a66fd16..1cfa5704d6 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1417,9 +1417,10 @@ async fn a_prune_that_fails_without_a_storage_failure_gives_storage_that_is_not_ ); } -/// The operation label, the path and the nice value of the calling thread of each storage call. +/// The operation label, the path, and the name and the nice value of the calling thread of each +/// storage call. #[cfg(target_os = "linux")] -type NiceCalls = Arc>>; +type NiceCalls = Arc>>; /// A storage that records the nice value of the thread of each call. #[cfg(target_os = "linux")] @@ -1434,6 +1435,10 @@ fn nice_recording_storage() -> (Arc, NiceCalls) { .push(( op_label.to_string(), path.display().to_string(), + std::thread::current() + .name() + .unwrap_or_default() + .to_string(), super::super::priority::own_nice(), )); Script::Pass @@ -1451,8 +1456,8 @@ fn taken_calls(calls: &NiceCalls, op_label: &str) -> Vec<(String, i32)> { .unwrap_or_else(std::sync::PoisonError::into_inner), ) .into_iter() - .filter(|(op, _, _)| op == op_label) - .map(|(_, path, nice)| (path, nice)) + .filter(|(op, _, _, _)| op == op_label) + .map(|(_, path, _, nice)| (path, nice)) .collect() } @@ -1531,13 +1536,13 @@ async fn the_storage_calls_of_a_restore_run_at_the_nice_value_of_the_process() { recorded.is_empty(), recorded .iter() - .filter(|(_, _, nice)| *nice != process_nice) + .filter(|(_, _, _, nice)| *nice != process_nice) .collect::>() ), ( Some(listing(tree.path())), false, - Vec::<&(String, String, i32)>::new() + Vec::<&(String, String, String, i32)>::new() ) ); } @@ -1588,3 +1593,114 @@ async fn after_saves_and_prunes_the_pools_keep_the_nice_value_of_the_process() { "{blocking:?} {rayon:?}" ); } + +#[cfg(target_os = "linux")] +#[test] +async fn the_storage_calls_of_the_rayon_workers_of_a_prune_that_repacks_run_at_nice_19() { + // The deleted snapshot shares a pack with the kept one, so the prune repacks that pack. The + // prune reads the index files and repacks with rayon, on the workers of the pool of the prune. + let (storage, calls) = nice_recording_storage(); + let base = policy(LONG_DEADLINE, 1, Duration::ZERO); + let store = store( + storage, + StorePolicy { + prune: PruneSettings { + repack: RepackLimits::Unlimited, + ..base.prune + }, + ..base + }, + ); + let scope = new_scope(); + let file = |content: &[u8]| Spec::File { + content: Box::from(content), + mode: 0o644, + }; + let both = Scratch::new(); + write_tree( + both.path(), + &[ + ("kept.txt", file(b"kept content")), + ("deleted.txt", file(b"deleted content")), + ], + ); + let kept = Scratch::new(); + write_tree(kept.path(), &[("kept.txt", file(b"kept content"))]); + store + .save(&scope, &name("p-both"), both.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept.path()) + .await + .unwrap(); + std::mem::take(&mut *calls.lock().unwrap()); + + store.delete(&scope, &name("p-both")).await.unwrap(); + let recorded = std::mem::take(&mut *calls.lock().unwrap()); + let from_workers = recorded + .iter() + .filter(|(_, _, thread, _)| thread.starts_with("fs-snap-prune-")) + .collect::>(); + + assert_eq!( + ( + recorded + .iter() + .any(|(op, path, _, _)| op == "write" && path.starts_with("data/")), + from_workers.is_empty(), + from_workers + .iter() + .filter(|(_, _, _, nice)| *nice != 19) + .count(), + restored_listing(&store, &scope, &name("p-kept")).await.ok(), + ), + (true, false, 0, Some(listing(kept.path()))) + ); +} + +/// A pool builder that cannot start a thread. +#[cfg(target_os = "linux")] +fn no_pool( + _: &str, + _: Option, +) -> Result { + rayon::ThreadPoolBuilder::new() + .num_threads(1) + .spawn_handler(|_| Err(std::io::Error::other("no thread can start here"))) + .build() +} + +#[cfg(target_os = "linux")] +#[test] +async fn the_global_rayon_pool_keeps_the_nice_value_of_the_process_after_saves_without_their_pool() +{ + // Without its own pool, the rayon work of a save goes to the global pool from a thread at + // nice 19. The store starts the global pool when it is made, so its threads keep the normal + // priority also when a save is the first rayon work of the process. + let process_nice = super::super::priority::own_nice(); + let store = RusticSnapshotStore { + low_priority: super::super::priority::LowPriority { + build_pool: no_pool, + ..super::super::priority::LowPriority::new(NonZeroUsize::new(2)) + }, + ..RusticSnapshotStore::new(Arc::new(InMemoryBlobStorage::new()), &config()) + }; + let scope = new_scope(); + let (first, second) = (one_file_tree("first"), fixture_tree()); + store + .save(&scope, &name("p-1"), first.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), second.path()) + .await + .unwrap(); + + let global = rayon::broadcast(|_| super::super::priority::own_nice()); + + assert!( + global.iter().all(|nice| *nice == process_nice), + "{global:?}" + ); +} From f18d52a3f00d0268534b9f9a5bca12b91a5633ae Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:53:45 -0700 Subject: [PATCH 025/126] Check that lowering the own priority gives Ok and nice 19 --- .../filesystem_snapshot/rustic/priority/tests.rs | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs index 6610032d75..f9ea230354 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs @@ -12,7 +12,7 @@ // See the License for the specific language governing permissions and // limitations under the License. -use super::{LOW_PRIORITY, LowPriority, own_nice}; +use super::{LOW_PRIORITY, LowPriority, lower_own_priority, own_nice}; use pretty_assertions::assert_eq; use rayon::{ThreadPool, ThreadPoolBuildError, ThreadPoolBuilder}; use std::num::NonZeroUsize; @@ -97,3 +97,17 @@ fn a_panic_of_the_work_gives_an_error() { assert!(done.is_err()); } + +#[test] +fn lowering_the_own_priority_gives_ok_and_nice_19() { + // The thread cannot raise its priority again, so the test lowers a thread of its own. + let lowered = std::thread::spawn(|| { + ( + lower_own_priority().map_err(|error| error.to_string()), + own_nice(), + ) + }) + .join(); + + assert_eq!(lowered.ok(), Some((Ok(()), LOW_PRIORITY))); +} From fb620d13c5ad9d7b776906127735b777343090f7 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:14:02 -0700 Subject: [PATCH 026/126] Let the caller of a save name its parent and choose the change detection --- .../filesystem_snapshot/contract_tests/mod.rs | 193 +++++++++--- .../src/filesystem_snapshot/memory.rs | 5 +- .../src/filesystem_snapshot/memory/tests.rs | 13 +- .../src/filesystem_snapshot/mod.rs | 16 + .../src/filesystem_snapshot/rustic/store.rs | 84 +++-- .../filesystem_snapshot/rustic/store/tests.rs | 292 +++++++++++++++--- .../src/filesystem_snapshot/rustic/tests.rs | 8 +- 7 files changed, 480 insertions(+), 131 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs b/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs index 0b33ba5d99..bd4f6392bf 100644 --- a/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs @@ -40,7 +40,8 @@ pub(super) mod fixture; use super::{ - FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, + ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, + SnapshotStoreError, }; use fixture::{ Listed, Scratch, Spec, files_and_bytes, fixture, listing, one_file, pattern, write_tree, @@ -165,6 +166,12 @@ const CASES: &[(&str, Case)] = &[ "a_dropped_save_publishes_nothing_and_leaves_the_name_free", |open| a_dropped_save_publishes_nothing_and_leaves_the_name_free(open).boxed(), ), + ( + "a_save_with_a_truthful_parent_restores_the_same_tree_as_a_save_without_one", + |open| { + a_save_with_a_truthful_parent_restores_the_same_tree_as_a_save_without_one(open).boxed() + }, + ), ("no_method_blocks_the_runtime", |open| { no_method_blocks_the_runtime(open).boxed() }), @@ -250,7 +257,7 @@ async fn a_saved_tree_comes_back_the_same(open: OpenStore) { let tree = new_tree(&fixture()); let saved = store - .save(&scope, &name("p-fixture"), tree.path()) + .save(&scope, &name("p-fixture"), tree.path(), None) .await .unwrap(); let (restored, info) = restored(&*store, &scope, &name("p-fixture")).await.unwrap(); @@ -258,6 +265,67 @@ async fn a_saved_tree_comes_back_the_same(open: OpenStore) { assert_eq!((restored, info), (listing(tree.path()), saved)); } +async fn a_save_with_a_truthful_parent_restores_the_same_tree_as_a_save_without_one( + open: OpenStore, +) { + // Each changed file gets a new size, so the parent is truthful for both modes. + let store = open(); + let scope = new_scope(); + let file = |content: &str| Spec::File { + content: Box::from(content.as_bytes()), + mode: 0o644, + }; + let tree = new_tree(&[ + ("changed.txt", file("old")), + ("kept.txt", file("kept")), + ("removed.txt", file("removed")), + ]); + let parent = name("p-parent"); + store + .save(&scope, &parent, tree.path(), None) + .await + .unwrap(); + write_tree( + tree.path(), + &[ + ("changed.txt", file("new and longer")), + ("added.txt", file("added")), + ], + ); + std::fs::remove_file(tree.path().join("removed.txt")).unwrap(); + + store + .save(&scope, &name("p-none"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-size-mtime"), + tree.path(), + Some((&parent, ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + store + .save( + &scope, + &name("p-full"), + tree.path(), + Some((&parent, ChangeDetection::Full)), + ) + .await + .unwrap(); + let restored = [ + restored_listing(&*store, &scope, &name("p-none")).await, + restored_listing(&*store, &scope, &name("p-size-mtime")).await, + restored_listing(&*store, &scope, &name("p-full")).await, + ]; + + let expected = listing(tree.path()); + assert_eq!(restored, [expected.clone(), expected.clone(), expected]); +} + async fn each_name_of_a_hard_linked_file_comes_back_as_its_own_file(open: OpenStore) { let store = open(); let scope = new_scope(); @@ -270,7 +338,7 @@ async fn each_name_of_a_hard_linked_file_comes_back_as_its_own_file(open: OpenSt let into = Scratch::new(); store - .save(&scope, &name("p-linked"), tree.path()) + .save(&scope, &name("p-linked"), tree.path(), None) .await .unwrap(); store @@ -301,7 +369,7 @@ async fn save_stat_list_and_restore_give_the_same_info(open: OpenStore) { let before = Timestamp::now_utc(); let saved = store - .save(&scope, &name("p-info"), tree.path()) + .save(&scope, &name("p-info"), tree.path(), None) .await .unwrap(); let after = Timestamp::now_utc(); @@ -334,7 +402,7 @@ async fn a_save_leaves_the_tree_as_it_was(open: OpenStore) { let before = listing(tree.path()); store - .save(&scope, &name("p-source"), tree.path()) + .save(&scope, &name("p-source"), tree.path(), None) .await .unwrap(); @@ -347,11 +415,13 @@ async fn a_name_in_use_gives_already_exists_and_changes_nothing(open: OpenStore) let first = new_tree(&one_file("first")); let second = new_tree(&one_file("second tree")); let saved = store - .save(&scope, &name("p-taken"), first.path()) + .save(&scope, &name("p-taken"), first.path(), None) .await .unwrap(); - let again = store.save(&scope, &name("p-taken"), second.path()).await; + let again = store + .save(&scope, &name("p-taken"), second.path(), None) + .await; assert!( matches!(again, Err(SnapshotStoreError::AlreadyExists)), @@ -382,12 +452,14 @@ async fn a_tree_that_cannot_be_read_gives_source_and_publishes_nothing(open: Ope std::fs::write(&file, b"not a directory").unwrap(); let tree = new_tree(&one_file("real")); - let from_missing = store.save(&scope, &name("p-unread"), &missing).await; - let from_file = store.save(&scope, &name("p-unread"), &file).await; + let from_missing = store.save(&scope, &name("p-unread"), &missing, None).await; + let from_file = store.save(&scope, &name("p-unread"), &file, None).await; let stat = store.stat(&scope, &name("p-unread")).await.unwrap(); let names = listed_names(&*store, &scope).await; let restore = restored(&*store, &scope, &name("p-unread")).await; - let later = store.save(&scope, &name("p-unread"), tree.path()).await; + let later = store + .save(&scope, &name("p-unread"), tree.path(), None) + .await; assert!( matches!(from_missing, Err(SnapshotStoreError::Source(_))), @@ -421,7 +493,9 @@ async fn an_entry_that_cannot_be_read_gives_source_and_publishes_nothing(open: O std::fs::write(&locked, b"locked").unwrap(); std::fs::set_permissions(&locked, std::fs::Permissions::from_mode(0o000)).unwrap(); - let saved = store.save(&scope, &name("p-locked"), tree.path()).await; + let saved = store + .save(&scope, &name("p-locked"), tree.path(), None) + .await; let stat = store.stat(&scope, &name("p-locked")).await.unwrap(); let names = listed_names(&*store, &scope).await; @@ -448,7 +522,7 @@ async fn sequential_saves_get_later_times_and_list_newest_first(open: OpenStore) let store = store.clone(); let scope = scope.clone(); let tree = tree.path().to_path_buf(); - async move { store.save(&scope, &name(text), &tree).await.unwrap() } + async move { store.save(&scope, &name(text), &tree, None).await.unwrap() } }) .collect::>() .await; @@ -478,7 +552,7 @@ async fn an_unknown_name_gives_not_found_every_time(open: OpenStore) { let used = new_scope(); let tree = new_tree(&one_file("other")); store - .save(&used, &name("p-other"), tree.path()) + .save(&used, &name("p-other"), tree.path(), None) .await .unwrap(); @@ -505,7 +579,7 @@ async fn a_restore_into_a_directory_that_is_not_empty_writes_nothing(open: OpenS let scope = new_scope(); let tree = new_tree(&one_file("content")); store - .save(&scope, &name("p-busy"), tree.path()) + .save(&scope, &name("p-busy"), tree.path(), None) .await .unwrap(); let into = new_tree(&[( @@ -532,7 +606,7 @@ async fn a_restore_into_a_path_that_is_not_a_directory_writes_nothing(open: Open let scope = new_scope(); let tree = new_tree(&one_file("content")); store - .save(&scope, &name("p-nowhere"), tree.path()) + .save(&scope, &name("p-nowhere"), tree.path(), None) .await .unwrap(); let parent = Scratch::new(); @@ -562,11 +636,11 @@ async fn a_deleted_name_stops_resolving_at_once(open: OpenStore) { let scope = new_scope(); let tree = new_tree(&one_file("deleted")); store - .save(&scope, &name("p-deleted"), tree.path()) + .save(&scope, &name("p-deleted"), tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), tree.path()) + .save(&scope, &name("p-kept"), tree.path(), None) .await .unwrap(); @@ -587,7 +661,7 @@ async fn delete_is_idempotent(open: OpenStore) { let scope = new_scope(); let tree = new_tree(&one_file("twice")); store - .save(&scope, &name("p-twice"), tree.path()) + .save(&scope, &name("p-twice"), tree.path(), None) .await .unwrap(); @@ -611,15 +685,15 @@ async fn a_delete_keeps_every_other_snapshot(open: OpenStore) { let shared = new_tree(&fixture()); let other = new_tree(&one_file("other")); store - .save(&scope, &name("p-twin-1"), shared.path()) + .save(&scope, &name("p-twin-1"), shared.path(), None) .await .unwrap(); store - .save(&scope, &name("p-twin-2"), shared.path()) + .save(&scope, &name("p-twin-2"), shared.path(), None) .await .unwrap(); store - .save(&scope, &name("p-other"), other.path()) + .save(&scope, &name("p-other"), other.path(), None) .await .unwrap(); @@ -644,7 +718,7 @@ async fn a_restore_that_races_a_delete_of_its_name_gives_a_whole_tree_or_nothing let scope = new_scope(); let tree = new_tree(&fixture()); store - .save(&scope, &name("p-raced"), tree.path()) + .save(&scope, &name("p-raced"), tree.path(), None) .await .unwrap(); @@ -668,11 +742,11 @@ async fn a_restore_during_a_delete_of_another_name_gives_the_whole_tree(open: Op let tree = new_tree(&fixture()); let other = new_tree(&fixture()); store - .save(&scope, &name("p-restored"), tree.path()) + .save(&scope, &name("p-restored"), tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-deleted"), other.path()) + .save(&scope, &name("p-deleted"), other.path(), None) .await .unwrap(); @@ -695,14 +769,17 @@ async fn a_save_a_restore_and_a_delete_in_one_scope_run_at_the_same_time(open: O let second = new_tree(&one_file("second")); let third = new_tree(&one_file("third")); let (first_name, second_name, third_name) = (name("p-1"), name("p-2"), name("p-3")); - store.save(&scope, &first_name, first.path()).await.unwrap(); store - .save(&scope, &second_name, second.path()) + .save(&scope, &first_name, first.path(), None) + .await + .unwrap(); + store + .save(&scope, &second_name, second.path(), None) .await .unwrap(); let (saved, restore, deleted) = futures::join!( - store.save(&scope, &third_name, third.path()), + store.save(&scope, &third_name, third.path(), None), restored(&*store, &scope, &first_name), store.delete(&scope, &second_name) ); @@ -728,14 +805,23 @@ async fn a_deleted_scope_is_as_unused_as_before_its_first_save(open: OpenStore) let scope = new_scope(); let old = new_tree(&one_file("old")); let new = new_tree(&one_file("new tree")); - store.save(&scope, &name("p-1"), old.path()).await.unwrap(); - store.save(&scope, &name("p-2"), old.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), old.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), old.path(), None) + .await + .unwrap(); store.delete_scope(&scope).await.unwrap(); let names = listed_names(&*store, &scope).await; let stat = store.stat(&scope, &name("p-1")).await.unwrap(); let restore = restored(&*store, &scope, &name("p-2")).await; - store.save(&scope, &name("p-1"), new.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), new.path(), None) + .await + .unwrap(); assert!(is_not_found(&restore), "{restore:?}"); assert_eq!( @@ -754,10 +840,13 @@ async fn delete_scope_is_idempotent_and_keeps_other_scopes(open: OpenStore) { let kept = new_scope(); let tree = new_tree(&one_file("kept")); store - .save(&deleted, &name("p-1"), tree.path()) + .save(&deleted, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save(&kept, &name("p-1"), tree.path(), None) .await .unwrap(); - store.save(&kept, &name("p-1"), tree.path()).await.unwrap(); let results = [ store.delete_scope(&new_scope()).await.is_ok(), @@ -787,9 +876,12 @@ async fn a_copied_scope_has_the_same_names_infos_and_trees(open: OpenStore) { let to = new_scope(); let first = new_tree(&fixture()); let second = new_tree(&one_file("second")); - store.save(&from, &name("p-1"), first.path()).await.unwrap(); store - .save(&from, &name("u-2"), second.path()) + .save(&from, &name("p-1"), first.path(), None) + .await + .unwrap(); + store + .save(&from, &name("u-2"), second.path(), None) .await .unwrap(); let source = store.list(&from).await.unwrap(); @@ -822,12 +914,21 @@ async fn copied_scopes_are_independent(open: OpenStore) { let to = new_scope(); let tree = new_tree(&one_file("copied")); let later = new_tree(&one_file("later")); - store.save(&from, &name("p-1"), tree.path()).await.unwrap(); - store.save(&from, &name("p-2"), tree.path()).await.unwrap(); + store + .save(&from, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save(&from, &name("p-2"), tree.path(), None) + .await + .unwrap(); store.copy_scope(&from, &to).await.unwrap(); store.delete(&from, &name("p-1")).await.unwrap(); - store.save(&to, &name("p-3"), later.path()).await.unwrap(); + store + .save(&to, &name("p-3"), later.path(), None) + .await + .unwrap(); let target_after_source_delete = restored_listing(&*store, &to, &name("p-1")).await; let source_names = listed_names(&*store, &from).await; store.delete_scope(&to).await.unwrap(); @@ -855,7 +956,7 @@ async fn a_copy_of_an_unused_scope_leaves_the_target_unused(open: OpenStore) { let to = new_scope(); let tree = new_tree(&one_file("other scope")); store - .save(&other, &name("p-other"), tree.path()) + .save(&other, &name("p-other"), tree.path(), None) .await .unwrap(); @@ -871,11 +972,11 @@ async fn one_name_in_two_scopes_gives_two_snapshots(open: OpenStore) { let first = new_tree(&one_file("first scope")); let second = new_tree(&one_file("second scope")); store - .save(&first_scope, &name("p-same"), first.path()) + .save(&first_scope, &name("p-same"), first.path(), None) .await .unwrap(); store - .save(&second_scope, &name("p-same"), second.path()) + .save(&second_scope, &name("p-same"), second.path(), None) .await .unwrap(); let first_restored = restored_listing(&*store, &first_scope, &name("p-same")).await; @@ -904,7 +1005,7 @@ async fn a_save_through_one_store_resolves_through_another(open: OpenStore) { let tree = new_tree(&fixture()); let saved = writer - .save(&scope, &name("p-shared"), tree.path()) + .save(&scope, &name("p-shared"), tree.path(), None) .await .unwrap(); @@ -933,8 +1034,8 @@ async fn two_stores_save_into_a_new_scope_at_the_same_time(open: OpenStore) { let (first_name, second_name) = (name("p-first"), name("p-second")); let (first_saved, second_saved) = futures::join!( - first.save(&scope, &first_name, first_tree.path()), - second.save(&scope, &second_name, second_tree.path()) + first.save(&scope, &first_name, first_tree.path(), None), + second.save(&scope, &second_name, second_tree.path(), None) ); let mut names = listed_names(&*first, &scope).await; names.sort(); @@ -973,7 +1074,7 @@ async fn a_dropped_save_publishes_nothing_and_leaves_the_name_free(open: OpenSto .filter_map(|_| { let scope = new_scope(); let dropped = store - .save(&scope, &dropped_name, tree.path()) + .save(&scope, &dropped_name, tree.path(), None) .now_or_never(); std::future::ready(dropped.is_none().then_some(scope)) }) @@ -985,7 +1086,7 @@ async fn a_dropped_save_publishes_nothing_and_leaves_the_name_free(open: OpenSto let stat = store.stat(&scope, &dropped_name).await.unwrap(); let restore = restored(&*store, &scope, &dropped_name).await; let names_after_the_drop = listed_names(&*store, &scope).await; - let saved_again = store.save(&scope, &dropped_name, other.path()).await; + let saved_again = store.save(&scope, &dropped_name, other.path(), None).await; assert!(is_not_found(&restore), "{restore:?}"); assert!(saved_again.is_ok(), "{saved_again:?}"); @@ -1041,7 +1142,7 @@ async fn no_method_blocks_the_runtime(open: OpenStore) { let before_save = ticks.load(Ordering::SeqCst); store - .save(&scope, &name("p-large"), &tree_path) + .save(&scope, &name("p-large"), &tree_path, None) .await .unwrap(); let during_save = counted(before_save); diff --git a/golem-worker-executor/src/filesystem_snapshot/memory.rs b/golem-worker-executor/src/filesystem_snapshot/memory.rs index b4fe682a18..3816efb695 100644 --- a/golem-worker-executor/src/filesystem_snapshot/memory.rs +++ b/golem-worker-executor/src/filesystem_snapshot/memory.rs @@ -25,8 +25,8 @@ mod tree; mod tests; use super::{ - FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, - newest_first, snapshot_time, + ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, + SnapshotStoreError, newest_first, snapshot_time, }; use async_trait::async_trait; use golem_common::model::Timestamp; @@ -120,6 +120,7 @@ impl FilesystemSnapshotStore for InMemorySnapshotStore { scope: &SnapshotScope, name: &SnapshotName, tree: &Path, + _parent: Option<(&SnapshotName, ChangeDetection)>, ) -> Result { let snapshots = self.snapshots_of(scope); if found(&snapshots, name).is_some() { diff --git a/golem-worker-executor/src/filesystem_snapshot/memory/tests.rs b/golem-worker-executor/src/filesystem_snapshot/memory/tests.rs index af377f4d88..d71a58668b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/memory/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/memory/tests.rs @@ -57,7 +57,12 @@ async fn a_save_after_a_snapshot_from_a_clock_that_is_ahead_gets_a_later_time() let tree = tempfile::tempdir().unwrap(); let saved = store - .save(&scope, &SnapshotName::new("p-next").unwrap(), tree.path()) + .save( + &scope, + &SnapshotName::new("p-next").unwrap(), + tree.path(), + None, + ) .await .unwrap(); let listed = store.list(&scope).await.unwrap(); @@ -97,8 +102,8 @@ async fn of_two_saves_of_one_name_at_the_same_time_one_wins() { let second_tree = tree_with("second tree"); let (first_saved, second_saved) = futures::join!( - first.save(&scope, &name, first_tree.path()), - second.save(&scope, &name, second_tree.path()) + first.save(&scope, &name, first_tree.path(), None), + second.save(&scope, &name, second_tree.path(), None) ); let into = tempfile::tempdir().unwrap(); first.restore(&scope, &name, into.path()).await.unwrap(); @@ -145,7 +150,7 @@ mod unix { let scope = new_scope(); let name = SnapshotName::new("p-socket").unwrap(); - let saved = store.save(&scope, &name, tree.path()).await; + let saved = store.save(&scope, &name, tree.path(), None).await; let listed = store.list(&scope).await.unwrap(); assert!( diff --git a/golem-worker-executor/src/filesystem_snapshot/mod.rs b/golem-worker-executor/src/filesystem_snapshot/mod.rs index b4354f2a47..5cbfe09faf 100644 --- a/golem-worker-executor/src/filesystem_snapshot/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/mod.rs @@ -121,6 +121,16 @@ pub(crate) struct SnapshotInfo { pub bytes: u64, } +/// How a save with a parent finds the files that did not change since the parent. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) enum ChangeDetection { + /// A file whose size and modification time equal those of the same path in the parent keeps + /// the content of the parent, and the save does not read it. + SizeMtime, + /// The save reads every file. + Full, +} + /// Why a call on a [`FilesystemSnapshotStore`] failed. #[derive(Debug)] pub(crate) enum SnapshotStoreError { @@ -210,11 +220,17 @@ pub(crate) trait FilesystemSnapshotStore: Send + Sync { /// The result gives the number of files, the size of the tree, and the time of the /// snapshot. That time is later than the time of each snapshot that the scope held when the /// save started. + /// + /// `parent` names the snapshot that the save compares with. With `SizeMtime`, a file whose + /// size and modification time equal those of the same path in the parent keeps the content of + /// the parent, and the save does not read it. With `Full`, the save reads every file. A parent + /// that the scope does not hold gives a save that reads every file. async fn save( &self, scope: &SnapshotScope, name: &SnapshotName, tree: &Path, + parent: Option<(&SnapshotName, ChangeDetection)>, ) -> Result; /// Rebuilds a saved tree in the empty directory `into`. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index a95f140817..b90dbb8262 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -28,13 +28,13 @@ use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; use super::{ - ChangeDetection, PruneReport, PruneSettings, RepackLimits, RepositoryKey, RepositorySettings, - SaveSettings, backup_options, open_existing, open_or_create, prune, restore_snapshot, - run_blocking, + ChangeDetection as RusticChangeDetection, PruneReport, PruneSettings, RepackLimits, + RepositoryKey, RepositorySettings, SaveSettings, backup_options, open_existing, open_or_create, + prune, restore_snapshot, run_blocking, }; use crate::filesystem_snapshot::{ - FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, - newest_first, snapshot_time, + ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, + SnapshotStoreError, newest_first, snapshot_time, }; use crate::services::golem_config::FilesystemSnapshotStoreConfig; use anyhow::Context; @@ -72,7 +72,9 @@ pub(super) struct StorePolicy { pub(super) deadline: Duration, /// The settings of a repository that a save makes. pub(super) repository: RepositorySettings, - pub(super) save: SaveSettings, + /// The number of threads of each parallel stage of a save. `None` is the number of CPUs that + /// the process can use. + pub(super) save_threads: Option, /// The number of threads that read packs in a restore. pub(super) restore_reader_threads: NonZeroUsize, /// The settings of a prune. `keep_delete` is also the shortest time between two prunes. @@ -87,10 +89,7 @@ impl StorePolicy { Self { deadline: config.storage_call_deadline(), repository: RepositorySettings::DEFAULT, - save: SaveSettings { - threads: Some(config.save_threads()), - detection: ChangeDetection::Ctime, - }, + save_threads: Some(config.save_threads()), restore_reader_threads: config.restore_reader_threads(), prune: PruneSettings { fast_repack: false, @@ -105,8 +104,34 @@ impl StorePolicy { /// The options of a save of the store: the options of the bridge, and a save that cannot read an /// entry fails before it writes the snapshot file. A save records no device id, so a restore gives /// each name of a hard-linked file as its own file. -fn store_backup_options(policy: &StorePolicy) -> BackupOptions { - backup_options(&policy.save) +/// +/// With a parent and `SizeMtime`, rustic compares each file with the parent that the id names, by +/// size and modification time. Without a parent, or with `Full`, rustic uses no parent and reads +/// every file. +fn store_backup_options( + policy: &StorePolicy, + parent: Option<(SnapshotId, ChangeDetection)>, +) -> BackupOptions { + let settings = |detection| SaveSettings { + threads: policy.save_threads, + detection, + }; + let options = match parent { + Some((id, ChangeDetection::SizeMtime)) => { + let options = backup_options(&settings(RusticChangeDetection::SizeMtime)); + let parent_opts = options + .parent_opts + .clone() + .parents(vec![id.to_hex().to_string()]); + options.parent_opts(parent_opts) + } + None | Some((_, ChangeDetection::Full)) => { + let options = backup_options(&settings(RusticChangeDetection::Ctime)); + let parent_opts = options.parent_opts.clone().force(true); + options.parent_opts(parent_opts) + } + }; + options .fail_on_read_error(true) .ignore_save_opts(LocalSourceSaveOptions::default().set_devid(DevIdOption::No)) } @@ -160,7 +185,7 @@ impl RusticSnapshotStore { policy, root: CancellationToken::new(), tracker: TaskTracker::new(), - low_priority: LowPriority::new(policy.save.threads), + low_priority: LowPriority::new(policy.save_threads), } } @@ -287,6 +312,7 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { scope: &SnapshotScope, name: &SnapshotName, tree: &Path, + parent: Option<(&SnapshotName, ChangeDetection)>, ) -> Result { let (token, _guard) = self.start()?; check_tree(tree).await?; @@ -296,11 +322,12 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { let policy = self.policy; let name = name.clone(); let tree: Box = tree.into(); + let parent = parent.map(|(parent, detection)| (parent.clone(), detection)); let low_priority = self.low_priority; let staged = self .blocking(Operation::Save, move || { low_priority.run("fs-snap-save", move || { - stage_save(backend, &stage, &key, &policy, &name, &tree) + stage_save(backend, &stage, &key, &policy, &name, &tree, parent) }) }) .await?; @@ -516,7 +543,9 @@ async fn check_destination(into: &Path) -> Result<(), SnapshotStoreError> { } /// Backs up the tree with the snapshot file in the stage, and gives the staged file with the info -/// of the snapshot. The result is `None` when a snapshot of the scope already has the name. +/// of the snapshot. The result is `None` when a snapshot of the scope already has the name. The +/// parent is the snapshot that [`lookup`] finds for its name, and a name without a snapshot gives +/// no parent. fn stage_save( backend: Arc, stage: &SnapshotStage, @@ -524,12 +553,16 @@ fn stage_save( policy: &StorePolicy, name: &SnapshotName, tree: &Path, + parent: Option<(SnapshotName, ChangeDetection)>, ) -> anyhow::Result> { let (repository, _) = open_or_create(backend, key, &policy.repository)?; let before = scope_snapshots(&repository)?; if has_name(&before, name) { return Ok(None); } + let parent = parent.and_then(|(parent, detection)| { + named(&before.readable, &parent).map(|snapshot| (snapshot.id, detection)) + }); let newest = before .readable .iter() @@ -549,7 +582,7 @@ fn stage_save( .to_snapshot()?; let repository = repository.to_indexed_ids()?; repository.backup( - &store_backup_options(policy), + &store_backup_options(policy, parent), &PathList::from_iter(Some(tree)), snapshot, )?; @@ -604,14 +637,9 @@ fn has_name(found: &ScopeSnapshots, name: &SnapshotName) -> bool { /// time and id wins. When no file has the name and a file failed its integrity check, the result is /// `Corrupt`, because that file can have the name. fn lookup(found: ScopeSnapshots, name: &SnapshotName) -> Lookup { - let winner = found - .readable - .into_iter() - .filter(|snapshot| snapshot.label == name.as_str()) - .min_by_key(|snapshot| (snapshot.time.timestamp(), snapshot.id)); - match (winner, found.unreadable) { - (Some(snapshot), _) => match snapshot_info(&snapshot) { - Some(info) => Lookup::Found(Box::new(snapshot), info), + match (named(&found.readable, name), found.unreadable) { + (Some(snapshot), _) => match snapshot_info(snapshot) { + Some(info) => Lookup::Found(Box::new(snapshot.clone()), info), None => Lookup::Corrupt(anyhow::anyhow!( "the snapshot {} does not describe its tree", snapshot.id @@ -624,6 +652,14 @@ fn lookup(found: ScopeSnapshots, name: &SnapshotName) -> Lookup { } } +/// Of the snapshot files with the name, gives the one with the least time and id. +fn named<'a>(snapshots: &'a [SnapshotFile], name: &SnapshotName) -> Option<&'a SnapshotFile> { + snapshots + .iter() + .filter(|snapshot| snapshot.label == name.as_str()) + .min_by_key(|snapshot| (snapshot.time.timestamp(), snapshot.id)) +} + /// Gives the name and the info of each snapshot whose label is a name and whose description /// parses. Of the files with one name, only the one that [`lookup`] takes stays. fn listed(found: ScopeSnapshots) -> Vec<(SnapshotName, SnapshotInfo)> { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 1cfa5704d6..16b7c1e7ae 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -20,6 +20,7 @@ use super::super::files::SnapshotFiles; use super::super::prune::{PruneLedger, read_ledger}; use super::super::scripted::{Script, ScriptedBlobStorage}; +use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; use super::{ RusticSnapshotStore, StorePolicy, leaves_marked_packs, scope_snapshots, store_backup_options, @@ -30,13 +31,15 @@ use crate::filesystem_snapshot::contract_tests::fixture::{ }; use crate::filesystem_snapshot::contract_tests::{self, OpenStore, new_scope}; use crate::filesystem_snapshot::{ - FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, + ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, + SnapshotStoreError, }; use crate::services::golem_config::FilesystemSnapshotStoreConfig; use futures::{FutureExt, StreamExt}; use golem_service_base::storage::blob::memory::InMemoryBlobStorage; use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; use pretty_assertions::assert_eq; +use rustic_core::repofile::SnapshotId; use std::future::Future; use std::num::NonZeroUsize; use std::path::Path; @@ -201,13 +204,13 @@ fn rustic_store_keeps_the_contract(r: &mut DynamicTestRegistration) { #[test] fn the_policy_takes_the_configured_values_and_the_options_are_strict() { let policy = StorePolicy::from_config(&config()); - let backup = store_backup_options(&policy); + let backup = store_backup_options(&policy, None); let restore = store_restore_options(&policy); assert_eq!( ( policy.deadline, - policy.save.threads.map(NonZeroUsize::get), + policy.save_threads.map(NonZeroUsize::get), policy.restore_reader_threads.get(), policy.prune.keep_delete, policy.prune.fast_repack, @@ -276,7 +279,7 @@ async fn a_tree_saved_through_a_proc_self_fd_path_is_stored_below_the_root() { let through_fd = format!("/proc/self/fd/{}", directory.as_raw_fd()); store - .save(&scope, &name("p-fd"), Path::new(&through_fd)) + .save(&scope, &name("p-fd"), Path::new(&through_fd), None) .await .unwrap(); let namespace = scope.0.clone(); @@ -326,12 +329,12 @@ async fn a_save_whose_index_write_fails_publishes_nothing_and_leaves_the_name_fr let scope = new_scope(); let tree = fixture_tree(); - let failed = store.save(&scope, &name("p-1"), tree.path()).await; + let failed = store.save(&scope, &name("p-1"), tree.path(), None).await; let stat = store.stat(&scope, &name("p-1")).await.unwrap(); let names = listed_names(&store, &scope).await; let restore = restored_listing(&store, &scope, &name("p-1")).await; refuse.store(false, Ordering::SeqCst); - let saved_again = store.save(&scope, &name("p-1"), tree.path()).await; + let saved_again = store.save(&scope, &name("p-1"), tree.path(), None).await; assert!( failed.as_ref().is_err_and(|error| is_storage(error, true)), @@ -374,11 +377,11 @@ async fn a_publish_that_reaches_the_deadline_and_lands_late_publishes_nothing() let scope = new_scope(); let tree = one_file_tree("late"); - let failed = store.save(&scope, &name("p-late"), tree.path()).await; + let failed = store.save(&scope, &name("p-late"), tree.path(), None).await; hang.store(false, Ordering::SeqCst); let stat = store.stat(&scope, &name("p-late")).await.unwrap(); let names = listed_names(&store, &scope).await; - let saved_again = store.save(&scope, &name("p-late"), tree.path()).await; + let saved_again = store.save(&scope, &name("p-late"), tree.path(), None).await; assert!( failed.as_ref().is_err_and(|error| is_storage(error, true)), @@ -410,9 +413,15 @@ async fn a_second_save_of_an_unchanged_tree_writes_no_pack() { let scope = new_scope(); let tree = fixture_tree(); - store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); let first = storage.calls().len(); - store.save(&scope, &name("p-2"), tree.path()).await.unwrap(); + store + .save(&scope, &name("p-2"), tree.path(), None) + .await + .unwrap(); let writes = storage.calls()[first..] .iter() .filter(|(op_label, _)| *op_label == "write" || *op_label == "publish") @@ -462,8 +471,8 @@ async fn two_stores_that_create_one_repository_at_the_same_time_both_save() { let (first_name, second_name) = (name("p-first"), name("p-second")); let (first_saved, second_saved, both_waited) = tokio::join!( - first.save(&scope, &first_name, first_tree.path()), - second.save(&scope, &second_name, second_tree.path()), + first.save(&scope, &first_name, first_tree.path(), None), + second.save(&scope, &second_name, second_tree.path(), None), async { let both = eventually(|| config_writes() == 2).await; storage.open_gate(); @@ -519,7 +528,7 @@ async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { let scope = new_scope(); let (old_tree, new_tree) = (one_file_tree("old"), fixture_tree()); store - .save(&scope, &name("p-old"), old_tree.path()) + .save(&scope, &name("p-old"), old_tree.path(), None) .await .unwrap(); hold_next_index.store(true, Ordering::SeqCst); @@ -536,7 +545,7 @@ async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { let store = store.clone(); let scope = scope.clone(); let path = new_tree.path().to_path_buf(); - async move { store.save(&scope, &name("p-new"), &path).await } + async move { store.save(&scope, &name("p-new"), &path, None).await } }); let held = eventually(|| index_writes() > before).await; let deleted = store.delete(&scope, &name("p-old")).await; @@ -577,7 +586,10 @@ async fn a_restore_whose_pack_reads_fail_gives_a_retryable_storage_error() { let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); - store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); refuse.store(true, Ordering::SeqCst); let restored = restored_listing(&store, &scope, &name("p-1")).await; @@ -611,7 +623,9 @@ async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_d std::fs::write(&locked, b"locked").unwrap(); std::fs::set_permissions(&locked, std::fs::Permissions::from_mode(0o000)).unwrap(); - let saved = store.save(&scope, &name("p-locked"), tree.path()).await; + let saved = store + .save(&scope, &name("p-locked"), tree.path(), None) + .await; assert!( matches!(&saved, Err(SnapshotStoreError::Source(error)) if error.kind() == std::io::ErrorKind::PermissionDenied), @@ -657,7 +671,7 @@ async fn a_restore_that_cannot_set_an_extended_attribute_gives_destination() { ); let scope = new_scope(); store - .save(&scope, &name("p-xattr"), source.path()) + .save(&scope, &name("p-xattr"), source.path(), None) .await .unwrap(); @@ -702,7 +716,7 @@ async fn a_snapshot_file_that_fails_its_check_is_left_out_of_list_and_makes_an_u let scope = new_scope(); let tree = one_file_tree("kept"); store - .save(&scope, &name("p-kept"), tree.path()) + .save(&scope, &name("p-kept"), tree.path(), None) .await .unwrap(); storage @@ -742,11 +756,11 @@ async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_per let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store - .save(&scope, &name("p-deleted"), deleted_tree.path()) + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept_tree.path()) + .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); let packs_before = blobs(&*storage, &scope.0, "data/").await; @@ -779,11 +793,11 @@ async fn a_delete_below_the_threshold_does_not_prune() { let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), one_file_tree("kept")); store - .save(&scope, &name("p-deleted"), deleted_tree.path()) + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept_tree.path()) + .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); let packs_before = blobs(&*storage, &scope.0, "data/").await; @@ -815,7 +829,10 @@ async fn no_second_prune_runs_within_the_grace_period() { let store = store.clone(); let scope = scope.clone(); async move { - store.save(&scope, &name(text), tree.path()).await.unwrap(); + store + .save(&scope, &name(text), tree.path(), None) + .await + .unwrap(); } }) .await; @@ -855,11 +872,11 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store - .save(&scope, &name("p-deleted"), deleted_tree.path()) + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept_tree.path()) + .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); refuse.store(true, Ordering::SeqCst); @@ -893,11 +910,11 @@ async fn a_deleted_scope_holds_no_blob() { let scope = new_scope(); let (first, second) = (one_file_tree("first"), one_file_tree("second")); store - .save(&scope, &name("p-1"), first.path()) + .save(&scope, &name("p-1"), first.path(), None) .await .unwrap(); store - .save(&scope, &name("p-2"), second.path()) + .save(&scope, &name("p-2"), second.path(), None) .await .unwrap(); store.delete(&scope, &name("p-1")).await.unwrap(); @@ -927,7 +944,7 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam counted.clone(), policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), ) - .save(&new_scope(), &name("p-dropped"), tree.path()) + .save(&new_scope(), &name("p-dropped"), tree.path(), None) .await .unwrap(); let calls = counted.calls().len(); @@ -951,7 +968,7 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam let ended = drop_when( &storage, |calls| calls.len() >= held, - dropping.save(&scope, &name("p-dropped"), &tree), + dropping.save(&scope, &name("p-dropped"), &tree, None), ) .await; let stopped = tokio::time::timeout(LIMIT, dropping.shut_down()) @@ -960,7 +977,7 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam let later = store(inner, policy); let stat = later.stat(&scope, &name("p-dropped")).await.ok().flatten(); let names = listed_names(&later, &scope).await; - let saved_again = later.save(&scope, &name("p-dropped"), &other).await; + let saved_again = later.save(&scope, &name("p-dropped"), &other, None).await; let restored = restored_listing(&later, &scope, &name("p-dropped")) .await .ok(); @@ -1014,7 +1031,7 @@ async fn shut_down_ends_running_operations_before_it_returns() { let store = store.clone(); let scope = scope.clone(); let path = tree.path().to_path_buf(); - async move { store.save(&scope, &name("p-held"), &path).await } + async move { store.save(&scope, &name("p-held"), &path, None).await } }); let held = eventually(|| { storage @@ -1080,7 +1097,10 @@ async fn a_dropped_operation_stops_its_blocking_work() { ); let scope = new_scope(); let tree = fixture_tree(); - store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); hold.store(true, Ordering::SeqCst); let before = storage.calls().len(); let reached = |calls: &[(&'static str, String)]| { @@ -1156,6 +1176,168 @@ async fn snapshot_files( .unwrap() } +/// Gives, for the snapshot with the name, the numbers of new, changed and unmodified files that its +/// save counted, and the id of its parent. +fn read_counts( + files: &[rustic_core::repofile::SnapshotFile], + name: &str, +) -> Option<((u64, u64, u64), Option)> { + let snapshot = files.iter().find(|snapshot| snapshot.label == name)?; + let summary = snapshot.summary.as_ref()?; + Some(( + ( + summary.files_new, + summary.files_changed, + summary.files_unmodified, + ), + snapshot.parent, + )) +} + +fn id_of(files: &[rustic_core::repofile::SnapshotFile], name: &str) -> Option { + files + .iter() + .find(|snapshot| snapshot.label == name) + .map(|snapshot| snapshot.id) +} + +#[test] +async fn a_size_and_mtime_save_of_a_copied_tree_reads_no_unchanged_file() { + // A copy gives each file a new inode and a new change time, and keeps its size and its + // modification time, as a capture does. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = three_file_tree(); + let copy = Scratch::new(); + wait_past_change_times(&entries(tree.path())); + copy_flat_tree(tree.path(), copy.path()); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-2"), + copy.path(), + Some((&name("p-1"), ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + let files = snapshot_files(storage, &scope).await; + + assert_eq!( + read_counts(&files, "p-2"), + Some(((0, 0, 3), id_of(&files, "p-1"))) + ); +} + +#[test] +async fn a_full_save_reads_each_file() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = three_file_tree(); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-2"), + tree.path(), + Some((&name("p-1"), ChangeDetection::Full)), + ) + .await + .unwrap(); + let files = snapshot_files(storage, &scope).await; + + assert_eq!(read_counts(&files, "p-2"), Some(((3, 0, 0), None))); +} + +#[test] +async fn the_parent_of_a_save_is_the_named_snapshot_also_when_a_newer_snapshot_exists() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = three_file_tree(); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-3"), + tree.path(), + Some((&name("p-1"), ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + let files = snapshot_files(storage, &scope).await; + + assert_eq!( + ( + read_counts(&files, "p-3"), + id_of(&files, "p-1") == id_of(&files, "p-2") + ), + (Some(((0, 0, 3), id_of(&files, "p-1"))), false) + ); +} + +#[test] +async fn a_save_without_a_parent_or_with_a_parent_that_the_scope_does_not_hold_reads_each_file() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = three_file_tree(); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-3"), + tree.path(), + Some((&name("p-missing"), ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + let files = snapshot_files(storage, &scope).await; + + assert_eq!( + (read_counts(&files, "p-2"), read_counts(&files, "p-3")), + (Some(((3, 0, 0), None)), Some(((3, 0, 0), None))) + ); +} + #[test] async fn a_failed_read_of_a_snapshot_file_fails_stat_and_list_with_a_retryable_storage_error() { let refuse = Arc::new(AtomicBool::new(false)); @@ -1174,7 +1356,7 @@ async fn a_failed_read_of_a_snapshot_file_fails_stat_and_list_with_a_retryable_s let scope = new_scope(); let tree = one_file_tree("kept"); store - .save(&scope, &name("p-kept"), tree.path()) + .save(&scope, &name("p-kept"), tree.path(), None) .await .unwrap(); refuse.store(true, Ordering::SeqCst); @@ -1210,7 +1392,7 @@ async fn a_snapshot_file_that_is_gone_after_the_listing_is_left_out() { let scope = new_scope(); let tree = one_file_tree("kept"); store - .save(&scope, &name("p-kept"), tree.path()) + .save(&scope, &name("p-kept"), tree.path(), None) .await .unwrap(); inner @@ -1235,7 +1417,7 @@ async fn a_delete_that_frees_nothing_writes_no_ledger() { let scope = new_scope(); let tree = one_file_tree("kept"); store - .save(&scope, &name("p-kept"), tree.path()) + .save(&scope, &name("p-kept"), tree.path(), None) .await .unwrap(); @@ -1287,7 +1469,9 @@ async fn a_save_of_a_relative_directory_path_gives_source_and_writes_nothing() { ); let scope = new_scope(); - let saved = store.save(&scope, &name("p-relative"), relative).await; + let saved = store + .save(&scope, &name("p-relative"), relative, None) + .await; assert!( matches!(saved, Err(SnapshotStoreError::Source(_))), @@ -1308,7 +1492,7 @@ async fn a_save_of_a_regular_file_gives_source_and_writes_nothing() { let scope = new_scope(); let saved = store - .save(&scope, &name("p-file"), &tree.path().join("file.txt")) + .save(&scope, &name("p-file"), &tree.path().join("file.txt"), None) .await; assert!( @@ -1328,7 +1512,7 @@ async fn the_ledger_counts_the_packed_bytes_that_the_deleted_snapshot_added() { let scope = new_scope(); let tree = fixture_tree(); store - .save(&scope, &name("p-deleted"), tree.path()) + .save(&scope, &name("p-deleted"), tree.path(), None) .await .unwrap(); let added = snapshot_files(storage.clone(), &scope) @@ -1360,7 +1544,7 @@ async fn a_config_write_that_fails_gives_a_storage_error_with_that_failure() { let scope = new_scope(); let tree = one_file_tree("never saved"); - let saved = store.save(&scope, &name("p-1"), tree.path()).await; + let saved = store.save(&scope, &name("p-1"), tree.path(), None).await; assert!( matches!( @@ -1381,11 +1565,11 @@ async fn a_prune_that_fails_without_a_storage_failure_gives_storage_that_is_not_ let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store - .save(&scope, &name("p-deleted"), deleted_tree.path()) + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept_tree.path()) + .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); let packs = storage @@ -1469,7 +1653,10 @@ async fn the_writes_of_a_save_run_at_nice_19() { let scope = new_scope(); let tree = fixture_tree(); - store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); let writes = taken_calls(&calls, "write"); assert_eq!( @@ -1492,11 +1679,11 @@ async fn the_writes_of_a_prune_run_at_nice_19() { let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store - .save(&scope, &name("p-deleted"), deleted_tree.path()) + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept_tree.path()) + .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); taken_calls(&calls, "write"); @@ -1524,7 +1711,10 @@ async fn the_storage_calls_of_a_restore_run_at_the_nice_value_of_the_process() { let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); - store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); std::mem::take(&mut *calls.lock().unwrap()); let restored = restored_listing(&store, &scope, &name("p-1")).await; @@ -1561,11 +1751,11 @@ async fn after_saves_and_prunes_the_pools_keep_the_nice_value_of_the_process() { let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store - .save(&scope, &name("p-deleted"), deleted_tree.path()) + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept_tree.path()) + .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); store.delete(&scope, &name("p-deleted")).await.unwrap(); @@ -1627,11 +1817,11 @@ async fn the_storage_calls_of_the_rayon_workers_of_a_prune_that_repacks_run_at_n let kept = Scratch::new(); write_tree(kept.path(), &[("kept.txt", file(b"kept content"))]); store - .save(&scope, &name("p-both"), both.path()) + .save(&scope, &name("p-both"), both.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept.path()) + .save(&scope, &name("p-kept"), kept.path(), None) .await .unwrap(); std::mem::take(&mut *calls.lock().unwrap()); @@ -1689,11 +1879,11 @@ async fn the_global_rayon_pool_keeps_the_nice_value_of_the_process_after_saves_w let scope = new_scope(); let (first, second) = (one_file_tree("first"), fixture_tree()); store - .save(&scope, &name("p-1"), first.path()) + .save(&scope, &name("p-1"), first.path(), None) .await .unwrap(); store - .save(&scope, &name("p-2"), second.path()) + .save(&scope, &name("p-2"), second.path(), None) .await .unwrap(); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index a50db65f08..d10f6697e4 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -967,7 +967,7 @@ fn set_modified(path: &Path, time: std::time::SystemTime) { /// Copies each file of the flat tree `from` into the new directory `to`, with its modification /// time. Each copy is a new inode with a new change time, as a capture gives. -fn copy_flat_tree(from: &Path, to: &Path) { +pub(super) fn copy_flat_tree(from: &Path, to: &Path) { std::fs::read_dir(from).unwrap().for_each(|entry| { let entry = entry.unwrap(); let target = to.join(entry.file_name()); @@ -992,7 +992,7 @@ const CHANGE_TIME_WAIT: Duration = Duration::from_secs(10); /// changes within one tick get the same change time. The wait changes a probe file in its own /// directory until the change time of the probe is later than the latest change time of the /// files. It fails the test when that does not happen within [`CHANGE_TIME_WAIT`]. -fn wait_past_change_times(files: &[PathBuf]) { +pub(super) fn wait_past_change_times(files: &[PathBuf]) { let latest = files.iter().map(|file| changed_at(file)).max().unwrap(); let probe_directory = Scratch::new(); let probe = probe_directory.path().join("probe"); @@ -1010,7 +1010,7 @@ fn wait_past_change_times(files: &[PathBuf]) { } /// Gives the path of each entry of the directory, in the order of the names. -fn entries(directory: &Path) -> Vec { +pub(super) fn entries(directory: &Path) -> Vec { let mut paths = std::fs::read_dir(directory) .unwrap() .map(|entry| entry.unwrap().path()) @@ -1020,7 +1020,7 @@ fn entries(directory: &Path) -> Vec { } /// Writes a tree of three files into a new directory, and gives the directory. -fn three_file_tree() -> Scratch { +pub(super) fn three_file_tree() -> Scratch { let tree = Scratch::new(); ["a.txt", "b.txt", "c.txt"] .iter() From 8be6236c78bd74555d31496670037017d88f4996 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 10:01:18 -0700 Subject: [PATCH 027/126] Shorten the docs of the change detection and the save options, and return the options from the match --- .../src/filesystem_snapshot/mod.rs | 5 ++-- .../src/filesystem_snapshot/rustic/store.rs | 29 +++++++++---------- 2 files changed, 15 insertions(+), 19 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/mod.rs b/golem-worker-executor/src/filesystem_snapshot/mod.rs index 5cbfe09faf..a3b0ad53d5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/mod.rs @@ -124,10 +124,9 @@ pub(crate) struct SnapshotInfo { /// How a save with a parent finds the files that did not change since the parent. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub(crate) enum ChangeDetection { - /// A file whose size and modification time equal those of the same path in the parent keeps - /// the content of the parent, and the save does not read it. + /// Compares each file with the parent by size and modification time. SizeMtime, - /// The save reads every file. + /// Reads every file. Full, } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index b90dbb8262..2351129eee 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -101,24 +101,24 @@ impl StorePolicy { } } -/// The options of a save of the store: the options of the bridge, and a save that cannot read an -/// entry fails before it writes the snapshot file. A save records no device id, so a restore gives -/// each name of a hard-linked file as its own file. -/// -/// With a parent and `SizeMtime`, rustic compares each file with the parent that the id names, by -/// size and modification time. Without a parent, or with `Full`, rustic uses no parent and reads +/// The options of a save of the store: a failed read of an entry fails the save, and no device id +/// is kept. `SizeMtime` compares with the parent that the id names, and each other case reads /// every file. fn store_backup_options( policy: &StorePolicy, parent: Option<(SnapshotId, ChangeDetection)>, ) -> BackupOptions { - let settings = |detection| SaveSettings { - threads: policy.save_threads, - detection, + let base = |detection| { + backup_options(&SaveSettings { + threads: policy.save_threads, + detection, + }) + .fail_on_read_error(true) + .ignore_save_opts(LocalSourceSaveOptions::default().set_devid(DevIdOption::No)) }; - let options = match parent { + match parent { Some((id, ChangeDetection::SizeMtime)) => { - let options = backup_options(&settings(RusticChangeDetection::SizeMtime)); + let options = base(RusticChangeDetection::SizeMtime); let parent_opts = options .parent_opts .clone() @@ -126,14 +126,11 @@ fn store_backup_options( options.parent_opts(parent_opts) } None | Some((_, ChangeDetection::Full)) => { - let options = backup_options(&settings(RusticChangeDetection::Ctime)); + let options = base(RusticChangeDetection::Ctime); let parent_opts = options.parent_opts.clone().force(true); options.parent_opts(parent_opts) } - }; - options - .fail_on_read_error(true) - .ignore_save_opts(LocalSourceSaveOptions::default().set_devid(DevIdOption::No)) + } } /// The options of a restore of the store. A metadata error fails the restore. The restore does not From b5c2b090e606d6e780694ccfddc634cf2110638a Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 10:01:18 -0700 Subject: [PATCH 028/126] Pass the name of low-priority work as a static str, and read the thread id of the test nice with try_from --- .../src/filesystem_snapshot/rustic/priority.rs | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs index e03448bef9..71be890c12 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs @@ -35,7 +35,7 @@ pub(super) struct LowPriority { pub(super) lower: fn() -> std::io::Result<()>, /// Builds the rayon pool of the work, with the name and the thread count. pub(super) build_pool: - fn(&str, Option) -> Result, + fn(&'static str, Option) -> Result, } impl LowPriority { @@ -53,7 +53,7 @@ impl LowPriority { /// Linux the work runs as it is. pub(super) fn run( self, - name: &str, + name: &'static str, work: impl FnOnce() -> anyhow::Result + Send + 'static, ) -> anyhow::Result { if cfg!(target_os = "linux") { @@ -67,12 +67,11 @@ impl LowPriority { /// of a step gives a warning, and the work runs without that step. fn on_own_thread( self, - name: &str, + name: &'static str, work: impl FnOnce() -> anyhow::Result + Send + 'static, ) -> anyhow::Result { // The work waits in a slot, so the calling thread can still run it when no thread starts. let slot = Arc::new(Mutex::new(Some(work))); - let pool_name = name.to_string(); let spawned = std::thread::Builder::new().name(name.to_string()).spawn({ let slot = slot.clone(); move || { @@ -82,7 +81,7 @@ impl LowPriority { "Failed to lower the CPU priority of filesystem snapshot work, so it runs at the normal priority" ); } - match (self.build_pool)(&pool_name, self.threads) { + match (self.build_pool)(name, self.threads) { Ok(pool) => pool.install(|| run_taken(&slot)), Err(error) => { warn!( @@ -120,10 +119,9 @@ fn run_taken anyhow::Result>(slot: &Mutex>) -> an /// Builds a rayon pool whose threads get the nice value of the calling thread. fn build_pool( - name: &str, + name: &'static str, threads: Option, ) -> Result { - let name = name.to_string(); ThreadPoolBuilder::new() .num_threads(threads.map_or(0, NonZeroUsize::get)) .thread_name(move |index| format!("{name}-{index}")) @@ -153,8 +151,10 @@ fn lower_own_priority() -> std::io::Result<()> { /// Gives the nice value of the calling thread. #[cfg(all(test, target_os = "linux"))] pub(super) fn own_nice() -> i32 { - // SAFETY: `gettid` has no preconditions, and `getpriority` only reads its arguments. - unsafe { libc::getpriority(libc::PRIO_PROCESS, libc::gettid() as libc::id_t) } + // SAFETY: `gettid` has no preconditions. + let thread = libc::id_t::try_from(unsafe { libc::gettid() }).unwrap(); + // SAFETY: `getpriority` only reads its arguments. + unsafe { libc::getpriority(libc::PRIO_PROCESS, thread) } } #[cfg(all(test, target_os = "linux"))] From f7af20e758e0072b2b5957eb5e1d7f0a93b7ebe7 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 10:03:12 -0700 Subject: [PATCH 029/126] Check that a failed read of the snapshot files while a save finds its parent gives storage and publishes nothing --- .../filesystem_snapshot/rustic/store/tests.rs | 52 +++++++++++++++++++ 1 file changed, 52 insertions(+) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 16b7c1e7ae..4f7b3d2422 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1338,6 +1338,58 @@ async fn a_save_without_a_parent_or_with_a_parent_that_the_scope_does_not_hold_r ); } +#[test] +async fn a_save_whose_read_of_the_snapshot_files_fails_while_it_finds_the_parent_gives_storage_and_publishes_nothing() + { + let refuse = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) && op_label == "read" && path.starts_with("snapshots") + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("parent"); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + refuse.store(true, Ordering::SeqCst); + let before = storage.calls().len(); + + let saved = store + .save( + &scope, + &name("p-2"), + tree.path(), + Some((&name("p-1"), ChangeDetection::SizeMtime)), + ) + .await; + let publishes = storage.calls()[before..] + .iter() + .filter(|(op_label, _)| *op_label == "publish") + .count(); + refuse.store(false, Ordering::SeqCst); + + assert!( + saved.as_ref().is_err_and(|error| is_storage(error, true)), + "{saved:?}" + ); + assert_eq!( + (publishes, listed_names(&store, &scope).await), + (0, vec!["p-1".to_string()]) + ); +} + #[test] async fn a_failed_read_of_a_snapshot_file_fails_stat_and_list_with_a_retryable_storage_error() { let refuse = Arc::new(AtomicBool::new(false)); From ff8047efced28e6745eb5d1d530d50c5df4ebee7 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 10:07:22 -0700 Subject: [PATCH 030/126] Check every storage call of the work of a save and of a prune at nice 19, reads included --- .../filesystem_snapshot/rustic/store/tests.rs | 77 ++++++++++++------- 1 file changed, 51 insertions(+), 26 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 4f7b3d2422..6b51ff640a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1683,49 +1683,76 @@ fn nice_recording_storage() -> (Arc, NiceCalls) { (storage, calls) } -/// Takes the recorded calls with the operation label. +/// Takes the recorded calls as the operation label, the path and the nice value. #[cfg(target_os = "linux")] -fn taken_calls(calls: &NiceCalls, op_label: &str) -> Vec<(String, i32)> { +fn taken_calls(calls: &NiceCalls) -> Vec<(String, String, i32)> { std::mem::take( &mut *calls .lock() .unwrap_or_else(std::sync::PoisonError::into_inner), ) .into_iter() - .filter(|(op, _, _, _)| op == op_label) - .map(|(_, path, _, nice)| (path, nice)) + .map(|(op_label, path, _, nice)| (op_label, path, nice)) .collect() } +/// Gives the operation labels of the calls, and each call that does not run at nice 19. +#[cfg(target_os = "linux")] +fn labels_and_calls_not_at_nice_19( + calls: &[(String, String, i32)], +) -> (Vec<&str>, Vec<&(String, String, i32)>) { + let mut labels = calls + .iter() + .map(|(op_label, _, _)| op_label.as_str()) + .collect::>(); + labels.sort_unstable(); + labels.dedup(); + let not_at_nice_19 = calls.iter().filter(|(_, _, nice)| *nice != 19).collect(); + (labels, not_at_nice_19) +} + #[cfg(target_os = "linux")] #[test] -async fn the_writes_of_a_save_run_at_nice_19() { +async fn the_storage_calls_of_a_save_run_at_nice_19() { + // The publish of the snapshot file runs on the async runtime after the work, so it keeps + // the normal priority. The second save reads the config, the index, the snapshot files and + // the trees of its parent, and writes the added file. let (storage, calls) = nice_recording_storage(); let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); - store .save(&scope, &name("p-1"), tree.path(), None) .await .unwrap(); - let writes = taken_calls(&calls, "write"); + std::fs::write(tree.path().join("added.txt"), b"added").unwrap(); + taken_calls(&calls); + + store + .save( + &scope, + &name("p-2"), + tree.path(), + Some((&name("p-1"), ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + let work = taken_calls(&calls) + .into_iter() + .filter(|(op_label, _, _)| op_label != "publish") + .collect::>(); assert_eq!( - ( - writes.is_empty(), - writes - .iter() - .filter(|(_, nice)| *nice != 19) - .collect::>() - ), - (false, Vec::<&(String, i32)>::new()) + labels_and_calls_not_at_nice_19(&work), + (vec!["list", "read", "stat", "write"], Vec::new()) ); } #[cfg(target_os = "linux")] #[test] -async fn the_writes_of_a_prune_run_at_nice_19() { +async fn the_storage_calls_of_a_prune_run_at_nice_19() { + // The forget of a delete runs before the ledger read at the normal priority, and the ledger + // read and writes run on the async runtime. The prune runs between them. let (storage, calls) = nice_recording_storage(); let store = store(storage, policy(LONG_DEADLINE, 1, Duration::ZERO)); let scope = new_scope(); @@ -1738,20 +1765,18 @@ async fn the_writes_of_a_prune_run_at_nice_19() { .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); - taken_calls(&calls, "write"); + taken_calls(&calls); store.delete(&scope, &name("p-deleted")).await.unwrap(); - let writes = taken_calls(&calls, "write"); + let prune = taken_calls(&calls) + .into_iter() + .skip_while(|(op_label, _, _)| op_label != "read_ledger") + .filter(|(op_label, _, _)| op_label != "read_ledger" && op_label != "write_ledger") + .collect::>(); assert_eq!( - ( - writes.is_empty(), - writes - .iter() - .filter(|(_, nice)| *nice != 19) - .collect::>() - ), - (false, Vec::<&(String, i32)>::new()) + labels_and_calls_not_at_nice_19(&prune), + (vec!["delete", "list", "read", "stat", "write"], Vec::new()) ); } From 361a1d2cfd690116b20293e6597253399b0826bd Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 22:21:52 -0700 Subject: [PATCH 031/126] Set the approved defaults of the filesystem snapshot store: a 60 s deadline, 6 restore readers and 2 save threads --- golem-worker-executor/src/services/golem_config.rs | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/services/golem_config.rs b/golem-worker-executor/src/services/golem_config.rs index 3ab3c64114..0b3a388b95 100644 --- a/golem-worker-executor/src/services/golem_config.rs +++ b/golem-worker-executor/src/services/golem_config.rs @@ -2341,13 +2341,13 @@ impl SafeDisplay for FilesystemPressureConfig { /// The default of [`FilesystemSnapshotStoreConfig::storage_call_deadline`]. The slowest measured /// call on S3 took 1.7 s, and the value stays at least 10 times the slowest measured call. -pub const DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE: Duration = Duration::from_secs(30); +pub const DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE: Duration = Duration::from_secs(60); /// The default of [`FilesystemSnapshotStoreConfig::restore_reader_threads`]. -const DEFAULT_FILESYSTEM_SNAPSHOT_RESTORE_READER_THREADS: usize = 4; +const DEFAULT_FILESYSTEM_SNAPSHOT_RESTORE_READER_THREADS: usize = 6; /// The default of [`FilesystemSnapshotStoreConfig::save_threads`]. -const DEFAULT_FILESYSTEM_SNAPSHOT_SAVE_THREADS: usize = 4; +const DEFAULT_FILESYSTEM_SNAPSHOT_SAVE_THREADS: usize = 2; /// Tells whether the executor keeps filesystem snapshots, and gives the settings of the store. #[derive(Clone, Debug, Serialize, Deserialize)] @@ -3010,7 +3010,7 @@ mod tests { store.restore_reader_threads().get(), store.save_threads().get(), ), - (Duration::from_secs(30), 4, 4) + (Duration::from_secs(60), 6, 2) ); } From 3f8546400b0a0f6ba043280159b870f57850402b Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 22:21:52 -0700 Subject: [PATCH 032/126] Keep marked packs for 15 minutes and repack fast in a prune of the store --- golem-worker-executor/src/filesystem_snapshot/rustic/store.rs | 4 ++-- .../src/filesystem_snapshot/rustic/store/tests.rs | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 2351129eee..524ae4c539 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -63,7 +63,7 @@ const PRUNE_THRESHOLD_BYTES: u64 = 64 * 1024 * 1024; /// How long a pack that a prune marks stays before a later prune deletes it. It is also the /// shortest time between two prunes of one scope. It must be longer than the longest save and the /// longest restore. -const PRUNE_GRACE: Duration = Duration::from_secs(3600); +const PRUNE_GRACE: Duration = Duration::from_secs(15 * 60); /// The settings of the store: the rustic settings of each operation, and the prune threshold. #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -92,7 +92,7 @@ impl StorePolicy { save_threads: Some(config.save_threads()), restore_reader_threads: config.restore_reader_threads(), prune: PruneSettings { - fast_repack: false, + fast_repack: true, keep_delete: PRUNE_GRACE, repack: RepackLimits::Rustic, }, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 6b51ff640a..6c6a9e85e7 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -221,8 +221,8 @@ fn the_policy_takes_the_configured_values_and_the_options_are_strict() { Duration::from_secs(30), Some(3), 4, - Duration::from_secs(3600), - false, + Duration::from_secs(15 * 60), + true, RepackLimits::Rustic, 64 * 1024 * 1024, ) From 1fdc959be3d051f9df71977acfec8e5acaf7a400 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 23:04:33 -0700 Subject: [PATCH 033/126] Prune a scope when its deleted snapshots free 10 percent of the size of its packs --- .../src/filesystem_snapshot/rustic/prune.rs | 144 ++++++++++---- .../src/filesystem_snapshot/rustic/store.rs | 32 +-- .../filesystem_snapshot/rustic/store/tests.rs | 186 ++++++++++++++---- 3 files changed, 270 insertions(+), 92 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index e412271a5d..8b8cd5f8ce 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -59,25 +59,61 @@ impl PruneLedger { } } +/// The path of the packs of a repository, relative to the root of the namespace of the scope. +const DATA_PATH: &str = "data"; + +/// A share of the size of a repository, in percent. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) struct Percent(pub(super) u16); + +impl Percent { + /// Gives this share of the bytes, rounded down. + fn of(self, bytes: u64) -> u64 { + u64::try_from(u128::from(bytes) * u128::from(self.0) / 100).unwrap_or(u64::MAX) + } +} + +/// Tells whether the grace period passed at `now` since the last prune. +fn grace_passed(ledger: &PruneLedger, now: Timestamp, grace: Duration) -> bool { + ledger.last_prune.is_none_or(|last| { + now.to_millis() + >= last + .to_millis() + .saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) + }) +} + +/// Tells whether [`prune_due`] needs the size of the repository at `now`. Only freed bytes after +/// the grace period, without marked packs, need it. +pub(super) fn needs_repository_size(ledger: &PruneLedger, now: Timestamp, grace: Duration) -> bool { + grace_passed(ledger, now, grace) && ledger.freed_bytes > 0 && !ledger.awaiting_removal +} + /// Tells whether a prune is due at `now`. /// /// A prune is due when the grace period passed since the last prune, and the freed bytes reach the -/// threshold or the last prune marked packs. A threshold of zero counts as one byte, so a prune -/// never runs for a scope that freed nothing and marked nothing. +/// threshold share of `repository_bytes` or the last prune marked packs. A threshold of zero bytes +/// counts as one byte, so a prune never runs for a scope that freed nothing and marked nothing. pub(super) fn prune_due( ledger: &PruneLedger, now: Timestamp, - threshold: u64, + repository_bytes: u64, + threshold: Percent, grace: Duration, ) -> bool { - let grace_passed = ledger.last_prune.is_none_or(|last| { - now.to_millis() - >= last - .to_millis() - .saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) - }); - let work = ledger.freed_bytes >= threshold.max(1) || ledger.awaiting_removal; - grace_passed && work + let work = + ledger.freed_bytes >= threshold.of(repository_bytes).max(1) || ledger.awaiting_removal; + grace_passed(ledger, now, grace) && work +} + +/// Gives the size of the repository of the scope: the sum of the sizes of its packs. +pub(super) async fn repository_bytes(files: &SnapshotFiles) -> anyhow::Result { + Ok(files + .list_below("list_data", Path::new(DATA_PATH)) + .await? + .iter() + .map(|blob| blob.size) + .fold(0, u64::saturating_add)) } /// Reads the ledger of the scope. A scope without a ledger, or with a ledger that does not parse, @@ -109,7 +145,10 @@ pub(super) async fn write_ledger( #[cfg(test)] mod tests { use super::super::files::SnapshotFiles; - use super::{LEDGER_PATH, PruneLedger, prune_due, read_ledger, write_ledger}; + use super::{ + LEDGER_PATH, Percent, PruneLedger, needs_repository_size, prune_due, read_ledger, + write_ledger, + }; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; use golem_service_base::storage::blob::BlobStorageNamespace; @@ -121,9 +160,9 @@ mod tests { use test_r::test; use uuid::Uuid; - const MIB: u64 = 1024 * 1024; - const THRESHOLD: u64 = 64 * MIB; - const GRACE: Duration = Duration::from_secs(3600); + const TEN_PERCENT: Percent = Percent(10); + const GRACE: Duration = Duration::from_secs(15 * 60); + const GRACE_MILLIS: u64 = 15 * 60 * 1000; const DEADLINE: Duration = Duration::from_secs(2); fn at(millis: u64) -> Timestamp { @@ -153,28 +192,39 @@ mod tests { } #[test] - fn a_prune_is_due_when_the_freed_bytes_reach_the_threshold() { + fn a_prune_is_due_when_the_freed_bytes_reach_ten_percent_of_the_repository() { let now = at(10_000_000); + let due = |freed, repository_bytes| { + prune_due( + &ledger(freed, None, false), + now, + repository_bytes, + TEN_PERCENT, + GRACE, + ) + }; assert_eq!( [ - prune_due(&ledger(THRESHOLD - 1, None, false), now, THRESHOLD, GRACE), - prune_due(&ledger(THRESHOLD, None, false), now, THRESHOLD, GRACE), - prune_due(&ledger(THRESHOLD + 1, None, false), now, THRESHOLD, GRACE), + due(99, 1000), + due(100, 1000), + due(101, 1000), + due(0, 0), + due(1, 0), ], - [false, true, true] + [false, true, true, false, true] ); } #[test] fn no_second_prune_runs_within_the_grace_period() { let last = 1_000_000; - let grace_millis = 3_600_000; let full = |now| { prune_due( - &ledger(THRESHOLD, Some(last), true), + &ledger(1000, Some(last), true), at(now), - THRESHOLD, + 1000, + TEN_PERCENT, GRACE, ) }; @@ -182,9 +232,9 @@ mod tests { assert_eq!( [ full(last), - full(last + grace_millis - 1), - full(last + grace_millis), - full(last + grace_millis + 1), + full(last + GRACE_MILLIS - 1), + full(last + GRACE_MILLIS), + full(last + GRACE_MILLIS + 1), ], [false, false, true, true] ); @@ -193,28 +243,50 @@ mod tests { #[test] fn marked_packs_make_a_prune_due_after_the_grace_period_without_freed_bytes() { let last = 1_000_000; - let after_grace = at(last + 3_600_000); + let after_grace = at(last + GRACE_MILLIS); + let due = |awaiting_removal| { + prune_due( + &ledger(0, Some(last), awaiting_removal), + after_grace, + 1000, + TEN_PERCENT, + GRACE, + ) + }; + + assert_eq!([due(true), due(false)], [true, false]); + } + + #[test] + fn a_zero_threshold_prunes_after_each_delete_that_freed_bytes() { + let now = at(10_000_000); + let due = |ledger| prune_due(&ledger, now, 1000, Percent(0), Duration::ZERO); assert_eq!( [ - prune_due(&ledger(0, Some(last), true), after_grace, THRESHOLD, GRACE), - prune_due(&ledger(0, Some(last), false), after_grace, THRESHOLD, GRACE), + due(ledger(0, None, false)), + due(ledger(1, None, false)), + due(ledger(1, Some(10_000_000), false)), ], - [true, false] + [false, true, true] ); } #[test] - fn a_zero_threshold_prunes_after_each_delete_that_freed_bytes() { - let now = at(10_000_000); + fn only_freed_bytes_after_the_grace_period_without_marked_packs_need_the_repository_size() { + let last = 1_000_000; + let needs = |freed, awaiting_removal, now| { + needs_repository_size(&ledger(freed, Some(last), awaiting_removal), at(now), GRACE) + }; assert_eq!( [ - prune_due(&ledger(0, None, false), now, 0, Duration::ZERO), - prune_due(&ledger(1, None, false), now, 0, Duration::ZERO), - prune_due(&ledger(1, Some(10_000_000), false), now, 0, Duration::ZERO), + needs(1, false, last + GRACE_MILLIS), + needs(1, false, last + GRACE_MILLIS - 1), + needs(0, false, last + GRACE_MILLIS), + needs(1, true, last + GRACE_MILLIS), ], - [false, true, true] + [true, false, false, false] ); } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 524ae4c539..4490ec1c0a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -24,7 +24,10 @@ use super::backend::BlobBackend; use super::fault::{Operation, classify, is_file_missing, is_storage_failure, storage_failure}; use super::files::SnapshotFiles; use super::priority::LowPriority; -use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; +use super::prune::{ + Percent, PruneLedger, needs_repository_size, prune_due, read_ledger, repository_bytes, + write_ledger, +}; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; use super::{ @@ -57,8 +60,9 @@ use tokio::runtime::Handle; use tokio_util::sync::{CancellationToken, DropGuard}; use tokio_util::task::TaskTracker; -/// The packed bytes that deleted snapshots must free before a delete prunes the scope. -const PRUNE_THRESHOLD_BYTES: u64 = 64 * 1024 * 1024; +/// The share of the size of the repository that deleted snapshots must free before a delete prunes +/// the scope. +const PRUNE_THRESHOLD: Percent = Percent(10); /// How long a pack that a prune marks stays before a later prune deletes it. It is also the /// shortest time between two prunes of one scope. It must be longer than the longest save and the @@ -79,8 +83,9 @@ pub(super) struct StorePolicy { pub(super) restore_reader_threads: NonZeroUsize, /// The settings of a prune. `keep_delete` is also the shortest time between two prunes. pub(super) prune: PruneSettings, - /// The packed bytes that deleted snapshots must free before a delete prunes. - pub(super) prune_threshold: u64, + /// The share of the size of the repository that deleted snapshots must free before a delete + /// prunes. + pub(super) prune_threshold: Percent, } impl StorePolicy { @@ -96,7 +101,7 @@ impl StorePolicy { keep_delete: PRUNE_GRACE, repack: RepackLimits::Rustic, }, - prune_threshold: PRUNE_THRESHOLD_BYTES, + prune_threshold: PRUNE_THRESHOLD, } } } @@ -260,7 +265,7 @@ impl RusticSnapshotStore { /// Adds the freed bytes to the ledger of the scope, and prunes the repository when a prune is /// due. The ledger keeps the freed bytes before the prune starts, so a delete that runs again - /// after a failed prune prunes again. + /// after a failed prune prunes again. It lists the packs only when their size can make a prune due. async fn prune_when_due( &self, scope: &SnapshotScope, @@ -278,12 +283,13 @@ impl RusticSnapshotStore { .map_err(storage_failure)?; } let now = Timestamp::now_utc(); - if !prune_due( - &ledger, - now, - self.policy.prune_threshold, - self.policy.prune.keep_delete, - ) { + let grace = self.policy.prune.keep_delete; + let size = if needs_repository_size(&ledger, now, grace) { + repository_bytes(&files).await.map_err(storage_failure)? + } else { + 0 + }; + if !prune_due(&ledger, now, size, self.policy.prune_threshold, grace) { return Ok(()); } let backend = Arc::new(self.backend(scope, token)?); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 6c6a9e85e7..0144dae408 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -18,7 +18,7 @@ //! give the store a short or a long deadline and a prune policy that the test controls. use super::super::files::SnapshotFiles; -use super::super::prune::{PruneLedger, read_ledger}; +use super::super::prune::{Percent, PruneLedger, read_ledger}; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; @@ -52,6 +52,12 @@ use test_r::{test, test_gen}; /// The longest time that a test waits for an operation or for the work of a store to end. const LIMIT: Duration = Duration::from_secs(10); +/// A prune threshold that a delete never reaches. +const NEVER: Percent = Percent(u16::MAX); + +/// A prune threshold of zero bytes, so each delete that frees bytes prunes. +const ALWAYS: Percent = Percent(0); + /// A deadline that no call of these tests reaches, so a held call ends only by a cancel. const LONG_DEADLINE: Duration = Duration::from_secs(60); @@ -68,7 +74,7 @@ fn key() -> RepositoryKey { /// The policy of the configuration, with the deadline, the prune threshold and the grace period /// of the test. -fn policy(deadline: Duration, prune_threshold: u64, grace: Duration) -> StorePolicy { +fn policy(deadline: Duration, prune_threshold: Percent, grace: Duration) -> StorePolicy { StorePolicy { deadline, prune: PruneSettings { @@ -224,7 +230,7 @@ fn the_policy_takes_the_configured_values_and_the_options_are_strict() { Duration::from_secs(15 * 60), true, RepackLimits::Rustic, - 64 * 1024 * 1024, + Percent(10), ) ); assert_eq!( @@ -271,7 +277,7 @@ async fn a_tree_saved_through_a_proc_self_fd_path_is_stored_below_the_root() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = fixture_tree(); @@ -325,7 +331,7 @@ async fn a_save_whose_index_write_fails_publishes_nothing_and_leaves_the_name_fr } } }); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); @@ -372,7 +378,7 @@ async fn a_publish_that_reaches_the_deadline_and_lands_late_publishes_nothing() }); let store = store( storage.clone(), - policy(Duration::from_secs(1), u64::MAX, Duration::ZERO), + policy(Duration::from_secs(1), NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = one_file_tree("late"); @@ -408,7 +414,7 @@ async fn a_second_save_of_an_unchanged_tree_writes_no_pack() { ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = fixture_tree(); @@ -454,7 +460,7 @@ async fn two_stores_that_create_one_repository_at_the_same_time_both_save() { Script::Pass } }); - let policy = policy(LONG_DEADLINE, u64::MAX, Duration::ZERO); + let policy = policy(LONG_DEADLINE, NEVER, Duration::ZERO); let (first, second) = ( store(storage.clone(), policy), store(storage.clone(), policy), @@ -524,7 +530,10 @@ async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { } } }); - let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); let scope = new_scope(); let (old_tree, new_tree) = (one_file_tree("old"), fixture_tree()); store @@ -583,7 +592,7 @@ async fn a_restore_whose_pack_reads_fail_gives_a_retryable_storage_error() { } } }); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); store @@ -615,7 +624,7 @@ async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_d let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = one_file_tree("readable"); @@ -667,7 +676,7 @@ async fn a_restore_that_cannot_set_an_extended_attribute_gives_destination() { } let store = store( Arc::new(InMemoryBlobStorage::new()), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); store @@ -711,7 +720,7 @@ async fn a_snapshot_file_that_fails_its_check_is_left_out_of_list_and_makes_an_u let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = one_file_tree("kept"); @@ -752,7 +761,10 @@ async fn a_snapshot_file_that_fails_its_check_is_left_out_of_list_and_makes_an_u #[test] async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_period() { let storage = Arc::new(InMemoryBlobStorage::new()); - let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store @@ -783,12 +795,89 @@ async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_per ); } +/// Counts the listings of the packs among the recorded calls. +fn data_listings(calls: &[(&'static str, String)]) -> usize { + calls + .iter() + .filter(|(op_label, _)| *op_label == "list_data") + .count() +} + +#[test] +async fn a_delete_within_the_grace_period_does_not_list_the_packs() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + let (first, second) = (one_file_tree("first"), one_file_tree("second")); + store + .save(&scope, &name("p-1"), first.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), second.path(), None) + .await + .unwrap(); + let before_first = storage.calls().len(); + + store.delete(&scope, &name("p-1")).await.unwrap(); + let before_second = storage.calls().len(); + store.delete(&scope, &name("p-2")).await.unwrap(); + let calls = storage.calls(); + + assert_eq!( + ( + data_listings(&calls[before_first..before_second]), + data_listings(&calls[before_second..]), + ledger(&storage, &scope).await.freed_bytes > 0, + ), + (1, 0, true) + ); +} + +#[test] +async fn a_failed_listing_of_the_packs_gives_storage_and_records_no_prune() { + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "list_data" { + Script::Refuse + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), one_file_tree("kept")); + store + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path(), None) + .await + .unwrap(); + + let deleted = store.delete(&scope, &name("p-deleted")).await; + let after = ledger(&storage, &scope).await; + + assert!( + deleted.as_ref().is_err_and(|error| is_storage(error, true)), + "{deleted:?}" + ); + assert_eq!((after.freed_bytes > 0, after.last_prune), (true, None)); +} + #[test] async fn a_delete_below_the_threshold_does_not_prune() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), one_file_tree("kept")); @@ -820,7 +909,7 @@ async fn no_second_prune_runs_within_the_grace_period() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, 1, Duration::from_secs(3600)), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), ); let scope = new_scope(); let trees = [one_file_tree("a"), one_file_tree("b"), one_file_tree("c")]; @@ -868,7 +957,10 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { } } }); - let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store @@ -906,7 +998,10 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { #[test] async fn a_deleted_scope_holds_no_blob() { let storage = Arc::new(InMemoryBlobStorage::new()); - let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); let scope = new_scope(); let (first, second) = (one_file_tree("first"), one_file_tree("second")); store @@ -942,7 +1037,7 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); store( counted.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ) .save(&new_scope(), &name("p-dropped"), tree.path(), None) .await @@ -962,7 +1057,7 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam Script::Pass } }); - let policy = policy(LONG_DEADLINE, u64::MAX, Duration::ZERO); + let policy = policy(LONG_DEADLINE, NEVER, Duration::ZERO); let dropping = store(storage.clone(), policy); let scope = new_scope(); let ended = drop_when( @@ -1023,7 +1118,7 @@ async fn shut_down_ends_running_operations_before_it_returns() { }); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = fixture_tree(); @@ -1093,7 +1188,7 @@ async fn a_dropped_operation_stops_its_blocking_work() { }); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = fixture_tree(); @@ -1208,7 +1303,7 @@ async fn a_size_and_mtime_save_of_a_copied_tree_reads_no_unchanged_file() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = three_file_tree(); @@ -1242,7 +1337,7 @@ async fn a_full_save_reads_each_file() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = three_file_tree(); @@ -1270,7 +1365,7 @@ async fn the_parent_of_a_save_is_the_named_snapshot_also_when_a_newer_snapshot_e let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = three_file_tree(); @@ -1308,7 +1403,7 @@ async fn a_save_without_a_parent_or_with_a_parent_that_the_scope_does_not_hold_r let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = three_file_tree(); @@ -1355,7 +1450,7 @@ async fn a_save_whose_read_of_the_snapshot_files_fails_while_it_finds_the_parent }); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = one_file_tree("parent"); @@ -1404,7 +1499,7 @@ async fn a_failed_read_of_a_snapshot_file_fails_stat_and_list_with_a_retryable_s } } }); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = one_file_tree("kept"); store @@ -1440,7 +1535,7 @@ async fn a_snapshot_file_that_is_gone_after_the_listing_is_left_out() { } } }); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = one_file_tree("kept"); store @@ -1464,7 +1559,7 @@ async fn a_delete_that_frees_nothing_writes_no_ledger() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = one_file_tree("kept"); @@ -1517,7 +1612,7 @@ async fn a_save_of_a_relative_directory_path_gives_source_and_writes_nothing() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); @@ -1539,7 +1634,7 @@ async fn a_save_of_a_regular_file_gives_source_and_writes_nothing() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); @@ -1559,7 +1654,7 @@ async fn the_ledger_counts_the_packed_bytes_that_the_deleted_snapshot_added() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = fixture_tree(); @@ -1592,7 +1687,7 @@ async fn a_config_write_that_fails_gives_a_storage_error_with_that_failure() { Script::Pass } }); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = one_file_tree("never saved"); @@ -1613,7 +1708,10 @@ async fn a_prune_that_fails_without_a_storage_failure_gives_storage_that_is_not_ // Packs of zeros with the sizes of the index give the prune a decryption error, not a failed // storage call. The forget before the prune has succeeded, so the delete gives `Storage`. let storage = Arc::new(InMemoryBlobStorage::new()); - let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store @@ -1718,7 +1816,7 @@ async fn the_storage_calls_of_a_save_run_at_nice_19() { // the normal priority. The second save reads the config, the index, the snapshot files and // the trees of its parent, and writes the added file. let (storage, calls) = nice_recording_storage(); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); store @@ -1751,10 +1849,10 @@ async fn the_storage_calls_of_a_save_run_at_nice_19() { #[cfg(target_os = "linux")] #[test] async fn the_storage_calls_of_a_prune_run_at_nice_19() { - // The forget of a delete runs before the ledger read at the normal priority, and the ledger - // read and writes run on the async runtime. The prune runs between them. + // The forget of a delete runs before the ledger read at the normal priority. The ledger calls + // and the listing of the packs run on the async runtime, and the prune runs after them. let (storage, calls) = nice_recording_storage(); - let store = store(storage, policy(LONG_DEADLINE, 1, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, ALWAYS, Duration::ZERO)); let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store @@ -1771,7 +1869,9 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { let prune = taken_calls(&calls) .into_iter() .skip_while(|(op_label, _, _)| op_label != "read_ledger") - .filter(|(op_label, _, _)| op_label != "read_ledger" && op_label != "write_ledger") + .filter(|(op_label, _, _)| { + !["read_ledger", "write_ledger", "list_data"].contains(&op_label.as_str()) + }) .collect::>(); assert_eq!( @@ -1785,7 +1885,7 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { async fn the_storage_calls_of_a_restore_run_at_the_nice_value_of_the_process() { let process_nice = super::super::priority::own_nice(); let (storage, calls) = nice_recording_storage(); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); store @@ -1823,7 +1923,7 @@ async fn after_saves_and_prunes_the_pools_keep_the_nice_value_of_the_process() { let process_nice = super::super::priority::own_nice(); let store = store( Arc::new(InMemoryBlobStorage::new()), - policy(LONG_DEADLINE, 1, Duration::ZERO), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), ); let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); @@ -1867,7 +1967,7 @@ async fn the_storage_calls_of_the_rayon_workers_of_a_prune_that_repacks_run_at_n // The deleted snapshot shares a pack with the kept one, so the prune repacks that pack. The // prune reads the index files and repacks with rayon, on the workers of the pool of the prune. let (storage, calls) = nice_recording_storage(); - let base = policy(LONG_DEADLINE, 1, Duration::ZERO); + let base = policy(LONG_DEADLINE, ALWAYS, Duration::ZERO); let store = store( storage, StorePolicy { From 2eb1a1668a960beaedca3988c6fb282970383503 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 23:11:30 -0700 Subject: [PATCH 034/126] Check the listing of the packs against the grace period alone, with a ledger that holds no marked packs --- .../filesystem_snapshot/rustic/store/tests.rs | 44 ++++++++++++++++--- 1 file changed, 37 insertions(+), 7 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 0144dae408..e08862fb16 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -18,7 +18,7 @@ //! give the store a short or a long deadline and a prune policy that the test controls. use super::super::files::SnapshotFiles; -use super::super::prune::{Percent, PruneLedger, read_ledger}; +use super::super::prune::{Percent, PruneLedger, read_ledger, write_ledger}; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; @@ -803,14 +803,37 @@ fn data_listings(calls: &[(&'static str, String)]) -> usize { .count() } +/// Writes a ledger with freed bytes, no marked packs, and a last prune at the time. +async fn set_last_prune( + storage: &Arc, + scope: &SnapshotScope, + last_prune: golem_common::model::Timestamp, +) { + let ledger = PruneLedger { + freed_bytes: 1, + last_prune: Some(last_prune), + awaiting_removal: false, + }; + write_ledger( + &SnapshotFiles { + storage: storage.clone(), + namespace: scope.0.clone(), + deadline: Duration::from_secs(2), + }, + &ledger, + ) + .await + .unwrap(); +} + #[test] async fn a_delete_within_the_grace_period_does_not_list_the_packs() { + // The ledger has no marked packs, so only the grace period keeps the first delete from a + // listing. The second delete comes after the grace period and lists the packs one time. + let grace = Duration::from_secs(3600); let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); - let store = store( - storage.clone(), - policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), - ); + let store = store(storage.clone(), policy(LONG_DEADLINE, NEVER, grace)); let scope = new_scope(); let (first, second) = (one_file_tree("first"), one_file_tree("second")); store @@ -821,10 +844,18 @@ async fn a_delete_within_the_grace_period_does_not_list_the_packs() { .save(&scope, &name("p-2"), second.path(), None) .await .unwrap(); + let now = golem_common::model::Timestamp::now_utc(); + set_last_prune(&storage, &scope, now).await; let before_first = storage.calls().len(); store.delete(&scope, &name("p-1")).await.unwrap(); let before_second = storage.calls().len(); + set_last_prune( + &storage, + &scope, + golem_common::model::Timestamp::from(now.to_millis().saturating_sub(2 * 3_600_000)), + ) + .await; store.delete(&scope, &name("p-2")).await.unwrap(); let calls = storage.calls(); @@ -832,9 +863,8 @@ async fn a_delete_within_the_grace_period_does_not_list_the_packs() { ( data_listings(&calls[before_first..before_second]), data_listings(&calls[before_second..]), - ledger(&storage, &scope).await.freed_bytes > 0, ), - (1, 0, true) + (0, 1) ); } From 73bcc7a51eaa49d042ae1664d5badabd1fc6146e Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 03:22:28 -0700 Subject: [PATCH 035/126] Record the packs that no index lists as marked packs of a prune --- .../src/filesystem_snapshot/rustic/mod.rs | 3 + .../src/filesystem_snapshot/rustic/store.rs | 9 ++- .../filesystem_snapshot/rustic/store/tests.rs | 55 +++++++++++++++++++ 3 files changed, 64 insertions(+), 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index ddcebdba84..594499f248 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -293,6 +293,8 @@ pub(super) struct PruneReport { pub(super) marked_bytes_deleted: u64, /// The packs that an earlier prune marked and that stay marked. pub(super) marked_packs_kept: u64, + /// The packs that no index lists. The prune marks each of them. + pub(super) packs_unindexed: u64, /// The bytes of the blobs that snapshots use. pub(super) bytes_used: u64, /// The bytes of the blobs that no snapshot uses. @@ -772,6 +774,7 @@ fn prune_report(stats: &PruneStats) -> PruneReport { marked_packs_deleted: stats.packs_to_delete.remove, marked_bytes_deleted: stats.size_to_delete.remove, marked_packs_kept: stats.packs_to_delete.keep, + packs_unindexed: stats.packs_unref, bytes_used: blobs.used, bytes_unused: blobs.unused, bytes_removed: blobs.remove, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 4490ec1c0a..85ebbda7df 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -697,10 +697,13 @@ fn snapshot_info(snapshot: &SnapshotFile) -> Option { } /// Tells whether a later prune removes packs that this prune leaves marked: unused packs, repacked -/// packs, and packs of an earlier prune whose grace period is not over. The report does not count a -/// marked pack that no index lists, so the next due prune removes it. +/// packs, packs that no index lists, and packs of an earlier prune whose grace period is not over. +/// A marked pack is in an index after the prune, so a later prune does not count it as unindexed. fn leaves_marked_packs(report: &PruneReport) -> bool { - report.packs_unused > 0 || report.packs_repacked > 0 || report.marked_packs_kept > 0 + report.packs_unused > 0 + || report.packs_repacked > 0 + || report.packs_unindexed > 0 + || report.marked_packs_kept > 0 } /// Gives the packed bytes that the save of the snapshot added to the repository. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index e08862fb16..a066273d19 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1606,6 +1606,61 @@ async fn a_delete_that_frees_nothing_writes_no_ledger() { ); } +#[test] +fn a_prune_that_marks_only_a_pack_that_no_index_lists_leaves_marked_packs() { + assert!(leaves_marked_packs(&PruneReport { + packs_used: 3, + packs_unindexed: 1, + ..PruneReport::default() + })); +} + +#[test] +async fn a_due_prune_that_marks_a_pack_that_no_index_lists_records_the_marked_pack() { + // The pack of the kept snapshot stays in use, so the pack that no index lists is the only pack + // that the prune marks. The ledger holds freed bytes from an earlier delete, so a delete of an + // unknown name prunes. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path(), None) + .await + .unwrap(); + let unindexed = format!("data/ab/{}", "ab".repeat(32)); + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new(&unindexed), + b"a pack that no index lists", + ) + .await + .unwrap(); + set_last_prune(&storage, &scope, golem_common::model::Timestamp::from(0)).await; + + store.delete(&scope, &name("p-unknown")).await.unwrap(); + let after = ledger(&storage, &scope).await; + + assert_eq!( + ( + after.freed_bytes, + after.last_prune.is_some_and(|last| last.to_millis() > 0), + after.awaiting_removal, + blobs(&*storage, &scope.0, "data/") + .await + .contains(&unindexed), + restored_listing(&store, &scope, &name("p-kept")).await.ok(), + ), + (0, true, true, true, Some(listing(tree.path()))) + ); +} + #[test] fn a_prune_leaves_marked_packs_when_it_marks_repacks_or_keeps_marked_packs() { let report = |packs_unused, packs_repacked, marked_packs_kept| PruneReport { From 1e8c0c6d57440365ad18ec783edf7b99966e2bf2 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 03:27:30 -0700 Subject: [PATCH 036/126] Stop the blob calls of an operation at shut down, and let shut down wait for a publish --- .../src/filesystem_snapshot/rustic/files.rs | 78 +++++++----- .../src/filesystem_snapshot/rustic/prune.rs | 1 + .../rustic/publish/tests.rs | 1 + .../filesystem_snapshot/rustic/scope/tests.rs | 1 + .../src/filesystem_snapshot/rustic/store.rs | 27 ++-- .../filesystem_snapshot/rustic/store/tests.rs | 118 ++++++++++++++++++ 6 files changed, 183 insertions(+), 43 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs index 297f5c5959..8d8119ba69 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs @@ -14,36 +14,55 @@ //! The blobs of the repository of one scope, and the blob storage calls of the store on them. //! -//! Each call waits for at most the deadline of the scope. +//! Each call waits for at most the deadline of the scope, and a cancel of its operation ends it. use super::backend::answer_within; +use super::fault::OperationCancelled; use golem_service_base::storage::blob::{ BlobStorage, BlobStorageNamespace, ListedBlob, PutIfAbsent, }; use std::path::Path; use std::sync::Arc; use std::time::Duration; +use tokio_util::sync::CancellationToken; /// The target label of each blob storage call of the rustic store. pub(super) const TARGET_LABEL: &str = "filesystem_snapshot"; -/// The blobs of one scope: the storage, the namespace of the scope, and the deadline of each call. +/// The blobs of one scope: the storage, the namespace of the scope, the deadline of each call, and +/// the token of the operation. #[derive(Clone, Debug)] pub(super) struct SnapshotFiles { pub(super) storage: Arc, pub(super) namespace: BlobStorageNamespace, pub(super) deadline: Duration, + pub(super) cancel: CancellationToken, } impl SnapshotFiles { + /// Waits for one call within the deadline. A call of a cancelled operation does not start, and + /// a cancel ends a running call. Both give an error. + async fn answer( + &self, + future: impl Future>, + ) -> anyhow::Result { + if self.cancel.is_cancelled() { + return Err(anyhow::Error::new(OperationCancelled)); + } + tokio::select! { + biased; + answer = answer_within(self.deadline, future) => answer, + () = self.cancel.cancelled() => Err(anyhow::Error::new(OperationCancelled)), + } + } + /// Gives the content of the blob at the path, or `None` when the path has no blob. pub(super) async fn get( &self, op_label: &'static str, path: &Path, ) -> anyhow::Result>> { - answer_within( - self.deadline, + self.answer( self.storage .get_raw(TARGET_LABEL, op_label, self.namespace.clone(), path), ) @@ -57,16 +76,13 @@ impl SnapshotFiles { path: &Path, content: &[u8], ) -> anyhow::Result<()> { - answer_within( - self.deadline, - self.storage.put_raw( - TARGET_LABEL, - op_label, - self.namespace.clone(), - path, - content, - ), - ) + self.answer(self.storage.put_raw( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + content, + )) .await } @@ -77,23 +93,19 @@ impl SnapshotFiles { path: &Path, content: &[u8], ) -> anyhow::Result { - answer_within( - self.deadline, - self.storage.put_raw_if_absent( - TARGET_LABEL, - op_label, - self.namespace.clone(), - path, - content, - ), - ) + self.answer(self.storage.put_raw_if_absent( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + content, + )) .await } /// Deletes the blob at the path. A path without a blob gives success. pub(super) async fn delete(&self, op_label: &'static str, path: &Path) -> anyhow::Result<()> { - answer_within( - self.deadline, + self.answer( self.storage .delete(TARGET_LABEL, op_label, self.namespace.clone(), path), ) @@ -106,8 +118,7 @@ impl SnapshotFiles { op_label: &'static str, path: &Path, ) -> anyhow::Result { - answer_within( - self.deadline, + self.answer( self.storage .delete_dir(TARGET_LABEL, op_label, self.namespace.clone(), path), ) @@ -120,11 +131,12 @@ impl SnapshotFiles { op_label: &'static str, path: &Path, ) -> anyhow::Result> { - answer_within( - self.deadline, - self.storage - .list_blobs_below(TARGET_LABEL, op_label, self.namespace.clone(), path), - ) + self.answer(self.storage.list_blobs_below( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + )) .await } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 8b8cd5f8ce..90ec11fd06 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -188,6 +188,7 @@ mod tests { environment_id: EnvironmentId(Uuid::new_v4()), }, deadline: DEADLINE, + cancel: tokio_util::sync::CancellationToken::new(), } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs index ec5723fe13..7acfc8d002 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -66,6 +66,7 @@ fn files( environment_id: EnvironmentId(Uuid::new_v4()), }, deadline, + cancel: tokio_util::sync::CancellationToken::new(), }, storage, inner, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs index 229d3380a9..692a5fc642 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs @@ -45,6 +45,7 @@ fn files( storage: storage.clone(), namespace: namespace.clone(), deadline: DEADLINE, + cancel: tokio_util::sync::CancellationToken::new(), } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 85ebbda7df..32e9e3c3d5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -191,9 +191,10 @@ impl RusticSnapshotStore { } } - /// Stops each operation at its next storage call, and waits until no blocking task and no - /// backend of the store remains; later operations give `Storage`. The runtime must not drop - /// before it returns, because a storage call after its time driver stops aborts the process. + /// Cancels each operation, so each running storage call ends and no new call starts, and later + /// operations give `Storage`. A publish is not cancelled. The call waits until no blocking task, + /// backend, publish or delete of a dropped publish remains. The runtime must not drop before it + /// returns, because a storage call after its time driver stops aborts the process. pub(crate) async fn shut_down(&self) { self.root.cancel(); self.tracker.close(); @@ -255,11 +256,13 @@ impl RusticSnapshotStore { .map_err(|error| classify(operation, error)) } - fn files(&self, scope: &SnapshotScope) -> SnapshotFiles { + /// Gives the blobs of the scope for the operation with the token. + fn files(&self, scope: &SnapshotScope, token: &CancellationToken) -> SnapshotFiles { SnapshotFiles { storage: self.storage.clone(), namespace: scope.0.clone(), deadline: self.policy.deadline, + cancel: token.clone(), } } @@ -272,7 +275,7 @@ impl RusticSnapshotStore { token: &CancellationToken, freed: u64, ) -> Result<(), SnapshotStoreError> { - let files = self.files(scope); + let files = self.files(scope, token); let ledger = read_ledger(&files) .await .map_err(storage_failure)? @@ -335,7 +338,11 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { }) .await?; let (staged, info) = staged.ok_or(SnapshotStoreError::AlreadyExists)?; - publish(&self.files(scope), &staged, &self.tracker) + // The publish is the commit point, so no cancel ends it. The tracker counts it, so + // `shut_down` waits for it, and the deadline limits that wait. + let files = self.files(scope, &CancellationToken::new()); + self.tracker + .track_future(publish(&files, &staged, &self.tracker)) .await .map_err(storage_failure)?; Ok(info) @@ -440,8 +447,8 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { } async fn delete_scope(&self, scope: &SnapshotScope) -> Result<(), SnapshotStoreError> { - let _operation = self.start()?; - delete_scope(&self.files(scope)) + let (token, _guard) = self.start()?; + delete_scope(&self.files(scope, &token)) .await .map_err(storage_failure) } @@ -451,8 +458,8 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { from: &SnapshotScope, to: &SnapshotScope, ) -> Result<(), SnapshotStoreError> { - let _operation = self.start()?; - copy_scope(&self.files(from), &self.files(to)) + let (token, _guard) = self.start()?; + copy_scope(&self.files(from, &token), &self.files(to, &token)) .await .map_err(storage_failure) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index a066273d19..ab82f96dee 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -160,6 +160,7 @@ async fn ledger(storage: &Arc, scope: &SnapshotScop storage: storage.clone(), namespace: scope.0.clone(), deadline: Duration::from_secs(2), + cancel: tokio_util::sync::CancellationToken::new(), }) .await .unwrap() @@ -819,6 +820,7 @@ async fn set_last_prune( storage: storage.clone(), namespace: scope.0.clone(), deadline: Duration::from_secs(2), + cancel: tokio_util::sync::CancellationToken::new(), }, &ledger, ) @@ -1136,6 +1138,122 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam ); } +#[test] +async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_call() { + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "copy_list" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let (from, to) = (new_scope(), new_scope()); + let tree = one_file_tree("copied"); + store + .save(&from, &name("p-1"), tree.path(), None) + .await + .unwrap(); + let copying = tokio::spawn({ + let store = store.clone(); + let (from, to) = (from.clone(), to.clone()); + async move { store.copy_scope(&from, &to).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "copy_list") + }) + .await; + + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + let calls_at_stop = storage.calls().len(); + storage.open_gate(); + let copied = tokio::time::timeout(LIMIT, copying).await; + + assert!( + matches!(&copied, Ok(Ok(Err(error))) if is_storage(error, true)), + "{copied:?}" + ); + assert_eq!( + (held, stopped, storage.calls().len()), + (true, true, calls_at_stop) + ); +} + +#[test] +async fn a_publish_held_at_its_storage_call_keeps_shut_down_waiting_until_it_ends() { + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "publish" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("published"); + let saving = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + let path = tree.path().to_path_buf(); + async move { store.save(&scope, &name("p-held"), &path, None).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "publish") + }) + .await; + + let shutting = store.shut_down(); + tokio::pin!(shutting); + let waited = tokio::time::timeout(Duration::from_millis(200), &mut shutting) + .await + .is_err(); + storage.open_gate(); + let stopped = tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + let saved = tokio::time::timeout(LIMIT, saving).await; + + assert!(matches!(&saved, Ok(Ok(Ok(_)))), "{saved:?}"); + assert_eq!((held, waited, stopped), (true, true, true)); +} + +#[test] +async fn delete_scope_and_copy_scope_after_shut_down_give_storage() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let (scope, other) = (new_scope(), new_scope()); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store.shut_down().await; + + let deleted = store.delete_scope(&scope).await; + let copied = store.copy_scope(&scope, &other).await; + + assert!( + deleted + .as_ref() + .is_err_and(|error| is_storage(error, false)), + "{deleted:?}" + ); + assert!( + copied.as_ref().is_err_and(|error| is_storage(error, false)), + "{copied:?}" + ); +} + #[test] async fn shut_down_ends_running_operations_before_it_returns() { let storage = From 79fdad4ca79bce526a17b6ee0809fd7a61e500b8 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 03:28:13 -0700 Subject: [PATCH 037/126] Keep one constant for the path of the config file of a repository --- .../src/filesystem_snapshot/rustic/backend.rs | 2 +- golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs | 4 +--- 2 files changed, 2 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index 8e66cd1214..cd994935bc 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -39,7 +39,7 @@ use tokio_util::sync::CancellationToken; use tokio_util::task::task_tracker::TaskTrackerToken; /// The path of the config file of a repository. -const CONFIG_PATH: &str = "config"; +pub(super) const CONFIG_PATH: &str = "config"; /// The largest number of bytes of tree packs that one backend keeps in memory. const KEPT_PACKS_LIMIT: usize = 32 * 1024 * 1024; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs index 996e605de2..ba992a8c6b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs @@ -17,15 +17,13 @@ //! These operations do not read the repository format. They only know the directories of the //! repository, its config file, and the ledger directory of the store. +use super::backend::CONFIG_PATH; use super::files::SnapshotFiles; use super::prune::LEDGER_PATH; use futures::{StreamExt, TryStreamExt, stream}; use golem_service_base::storage::blob::PutIfAbsent; use std::path::Path; -/// The path of the config file of a repository. -const CONFIG_PATH: &str = "config"; - /// The directories of a repository in the order of a listing. A save writes them in the reverse /// order, and so does a copy, so a snapshot file always has its data. const LISTING_ORDER: [&str; 4] = ["snapshots", "index", "keys", "data"]; From ad5036612561cfd3b3749f07fdbaa989762e7c30 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 03:36:27 -0700 Subject: [PATCH 038/126] Poll the shut down of the publish test again only when its first wait timed out --- .../src/filesystem_snapshot/rustic/store/tests.rs | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index ab82f96dee..0af6fdef00 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1220,7 +1220,9 @@ async fn a_publish_held_at_its_storage_call_keeps_shut_down_waiting_until_it_end .await .is_err(); storage.open_gate(); - let stopped = tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + // A finished future must not be polled again, so the second wait runs only after a first wait + // that timed out. + let stopped = !waited || tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); let saved = tokio::time::timeout(LIMIT, saving).await; assert!(matches!(&saved, Ok(Ok(Ok(_)))), "{saved:?}"); From 66c9ee7b3309f03ecb79efa0660d2074a5d06507 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 03:53:07 -0700 Subject: [PATCH 039/126] Publish nothing when a save reaches its publish after shut down --- .../src/filesystem_snapshot/rustic/store.rs | 56 ++++++++++++++----- .../filesystem_snapshot/rustic/store/tests.rs | 49 +++++++++++++++- 2 files changed, 90 insertions(+), 15 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 32e9e3c3d5..5b2dea6451 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -154,10 +154,31 @@ pub(crate) struct RusticSnapshotStore { policy: StorePolicy, /// The parent of the token of each operation. root: CancellationToken, - /// Counts the blocking tasks, the backends and the deletes of dropped publishes. + /// Counts the blocking tasks, the backends, the publishes and the deletes of dropped publishes. tracker: TaskTracker, /// Runs saves and prunes at a low priority. low_priority: LowPriority, + /// Holds a save after its blocking work and before its publish, when a test sets it. + #[cfg(test)] + pub(super) publish_gate: Option>, +} + +/// A gate that holds a save after its blocking work and before its publish. +#[cfg(test)] +#[derive(Debug, Default)] +pub(super) struct PublishGate { + /// Notified when a save reaches the gate. + pub(super) reached: tokio::sync::Notify, + /// Lets the save go on. + pub(super) open: tokio::sync::Notify, +} + +/// The error of an operation of a store that is shut down. +fn shut_down_error() -> SnapshotStoreError { + SnapshotStoreError::Storage { + retryable: false, + source: anyhow::anyhow!("the filesystem snapshot store is shut down"), + } } impl RusticSnapshotStore { @@ -188,12 +209,15 @@ impl RusticSnapshotStore { root: CancellationToken::new(), tracker: TaskTracker::new(), low_priority: LowPriority::new(policy.save_threads), + #[cfg(test)] + publish_gate: None, } } /// Cancels each operation, so each running storage call ends and no new call starts, and later - /// operations give `Storage`. A publish is not cancelled. The call waits until no blocking task, - /// backend, publish or delete of a dropped publish remains. The runtime must not drop before it + /// operations give `Storage`. A publish that starts before the cancel runs to its end. A save + /// that reaches its publish after the cancel publishes nothing and gives `Storage`. The call + /// waits until no blocking task, backend, publish or delete of a dropped publish remains. The runtime must not drop before it /// returns, because a storage call after its time driver stops aborts the process. pub(crate) async fn shut_down(&self) { self.root.cancel(); @@ -210,10 +234,7 @@ impl RusticSnapshotStore { /// Starts an operation. The token of the operation is cancelled when the guard drops. fn start(&self) -> Result<(CancellationToken, DropGuard), SnapshotStoreError> { if self.root.is_cancelled() { - return Err(SnapshotStoreError::Storage { - retryable: false, - source: anyhow::anyhow!("the filesystem snapshot store is shut down"), - }); + return Err(shut_down_error()); } let token = self.root.child_token(); Ok((token.clone(), token.drop_guard())) @@ -338,13 +359,22 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { }) .await?; let (staged, info) = staged.ok_or(SnapshotStoreError::AlreadyExists)?; - // The publish is the commit point, so no cancel ends it. The tracker counts it, so - // `shut_down` waits for it, and the deadline limits that wait. + #[cfg(test)] + if let Some(gate) = &self.publish_gate { + gate.reached.notify_one(); + gate.open.notified().await; + } + // The publish is the commit point, so no cancel ends it. The tracker counts it from here, + // so a `shut_down` that has not cancelled yet waits for it, and the deadline limits that wait. let files = self.files(scope, &CancellationToken::new()); - self.tracker - .track_future(publish(&files, &staged, &self.tracker)) - .await - .map_err(storage_failure)?; + let publishing = self + .tracker + .track_future(publish(&files, &staged, &self.tracker)); + if self.root.is_cancelled() { + // No snapshot file is written. A later prune marks the packs of the save. + return Err(shut_down_error()); + } + publishing.await.map_err(storage_failure)?; Ok(info) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 0af6fdef00..07ea2c6510 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -23,8 +23,8 @@ use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; use super::{ - RusticSnapshotStore, StorePolicy, leaves_marked_packs, scope_snapshots, store_backup_options, - store_restore_options, whole_millis_from, + PublishGate, RusticSnapshotStore, StorePolicy, leaves_marked_packs, scope_snapshots, + store_backup_options, store_restore_options, whole_millis_from, }; use crate::filesystem_snapshot::contract_tests::fixture::{ Listed, Scratch, Spec, fixture, listing, write_tree, @@ -1229,6 +1229,51 @@ async fn a_publish_held_at_its_storage_call_keeps_shut_down_waiting_until_it_end assert_eq!((held, waited, stopped), (true, true, true)); } +#[test] +async fn a_save_that_reaches_its_publish_after_shut_down_publishes_nothing() { + // The gate holds the save after its blocking work, so the tracker is empty and `shut_down` + // returns before the publish starts. + let storage = Arc::new(InMemoryBlobStorage::new()); + let gate = Arc::new(PublishGate::default()); + let store = Arc::new(RusticSnapshotStore { + publish_gate: Some(gate.clone()), + ..RusticSnapshotStore::with_policy( + storage.clone(), + key(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ) + }); + let scope = new_scope(); + let tree = one_file_tree("late"); + let saving = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + let path = tree.path().to_path_buf(); + async move { store.save(&scope, &name("p-late"), &path, None).await } + }); + let reached = tokio::time::timeout(LIMIT, gate.reached.notified()) + .await + .is_ok(); + + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + gate.open.notify_one(); + let saved = tokio::time::timeout(LIMIT, saving).await; + + assert!( + matches!(&saved, Ok(Ok(Err(error))) if is_storage(error, false)), + "{saved:?}" + ); + assert_eq!( + ( + reached, + stopped, + blobs(&*storage, &scope.0, "snapshots/").await, + store.work_in_flight(), + ), + (true, true, Vec::::new(), 0) + ); +} + #[test] async fn delete_scope_and_copy_scope_after_shut_down_give_storage() { let storage = Arc::new(InMemoryBlobStorage::new()); From 2a11af91f977341d3b4ebf28c3061a95b5b8df4c Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 03:55:16 -0700 Subject: [PATCH 040/126] Wrap the doc of the shut down of the store --- .../src/filesystem_snapshot/rustic/store.rs | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 5b2dea6451..697ba735d7 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -217,8 +217,9 @@ impl RusticSnapshotStore { /// Cancels each operation, so each running storage call ends and no new call starts, and later /// operations give `Storage`. A publish that starts before the cancel runs to its end. A save /// that reaches its publish after the cancel publishes nothing and gives `Storage`. The call - /// waits until no blocking task, backend, publish or delete of a dropped publish remains. The runtime must not drop before it - /// returns, because a storage call after its time driver stops aborts the process. + /// waits until no blocking task, backend, publish or delete of a dropped publish remains. The + /// runtime must not drop before it returns, because a storage call after its time driver stops + /// aborts the process. pub(crate) async fn shut_down(&self) { self.root.cancel(); self.tracker.close(); From a5a20474402f09aaf6b465bc06c38243d4c9f6c1 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 04:57:25 -0700 Subject: [PATCH 041/126] Check that a cancel ends a running blob call of a copy --- .../src/filesystem_snapshot/rustic/store/tests.rs | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 07ea2c6510..49b4d8b119 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1157,7 +1157,7 @@ async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_cal .save(&from, &name("p-1"), tree.path(), None) .await .unwrap(); - let copying = tokio::spawn({ + let mut copying = tokio::spawn({ let store = store.clone(); let (from, to) = (from.clone(), to.clone()); async move { store.copy_scope(&from, &to).await } @@ -1170,17 +1170,19 @@ async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_cal }) .await; + // The gate stays closed until the end, so only the cancel can end the held call. let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); let calls_at_stop = storage.calls().len(); + let copied = tokio::time::timeout(LIMIT, &mut copying).await; + let calls_after_copy = storage.calls().len(); storage.open_gate(); - let copied = tokio::time::timeout(LIMIT, copying).await; assert!( matches!(&copied, Ok(Ok(Err(error))) if is_storage(error, true)), "{copied:?}" ); assert_eq!( - (held, stopped, storage.calls().len()), + (held, stopped, calls_after_copy), (true, true, calls_at_stop) ); } From 754f1f9495f338cc32b24514d120e30e35982bfb Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 05:04:57 -0700 Subject: [PATCH 042/126] Check that a blob call of a cancelled operation does not start --- .../filesystem_snapshot/rustic/store/tests.rs | 23 +++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 49b4d8b119..b935b01f98 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1138,6 +1138,29 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam ); } +#[test] +async fn a_blob_call_of_a_cancelled_operation_does_not_start() { + // The in-memory storage answers at the first poll, so only the check before the call keeps + // the call from the storage. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let cancel = tokio_util::sync::CancellationToken::new(); + cancel.cancel(); + let files = SnapshotFiles { + storage: storage.clone(), + namespace: new_scope().0, + deadline: LONG_DEADLINE, + cancel, + }; + + let read = files + .get("read_ledger", Path::new("golem/prune-ledger")) + .await; + + assert!(read.is_err(), "{read:?}"); + assert_eq!(storage.calls(), Vec::new()); +} + #[test] async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_call() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { From f8e94e6bd628eadc308942768e75e894b473e8e5 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 10:31:06 -0700 Subject: [PATCH 043/126] Give each waiting test of the rustic store a time limit --- .../rustic/backend/tests.rs | 27 ++++++++++++------- .../rustic/publish/tests.rs | 4 ++- .../filesystem_snapshot/rustic/store/tests.rs | 12 ++++++++- .../src/filesystem_snapshot/rustic/tests.rs | 7 ++++- 4 files changed, 37 insertions(+), 13 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index 994c7b88ea..18402a6557 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -124,9 +124,17 @@ fn bytes(text: &str) -> BytesList { /// Runs the calls on a new thread, which is not a thread of a runtime, and gives their result. /// `None` means that the calls did not end within the limit. fn within_limit(calls: impl FnOnce() -> T + Send + 'static) -> Option { + on_own_thread(calls).recv_timeout(LIMIT).ok() +} + +/// Starts the calls on a new thread, which is not a thread of a runtime. The receiver gets their +/// result, so a test can wait for it with a limit. +fn on_own_thread( + calls: impl FnOnce() -> T + Send + 'static, +) -> std::sync::mpsc::Receiver { let (sender, receiver) = std::sync::mpsc::channel(); std::thread::spawn(move || sender.send(calls())); - receiver.recv_timeout(LIMIT).ok() + receiver } #[test] @@ -464,18 +472,17 @@ fn a_thread_that_is_not_a_thread_of_the_runtime_can_call_the_backend() { let fixture = Fixture::new(); let backend = Arc::new(fixture.backend); - let read = std::thread::spawn({ + let read = within_limit({ let backend = backend.clone(); move || { backend .write_bytes(FileType::Index, &id("ab"), false, bytes("index")) .and_then(|()| backend.read_full(FileType::Index, &id("ab"))) + .ok() } - }) - .join() - .map(|read| read.ok()); + }); - assert_eq!(read.ok().flatten(), Some(Bytes::from_static(b"index"))); + assert_eq!(read.flatten(), Some(Bytes::from_static(b"index"))); } /// The content of the pack of the tests of the kept packs: 100 bytes, each its own offset. @@ -598,7 +605,7 @@ fn two_threads_that_miss_one_pack_make_one_storage_read() { Script::Pass } }); - let first = std::thread::spawn({ + let first = on_own_thread({ let backend = fixture.backend.clone(); move || tree_range(&backend, 0, 10).ok() }); @@ -606,7 +613,7 @@ fn two_threads_that_miss_one_pack_make_one_storage_read() { std::thread::sleep(Duration::from_millis(10)); !fixture.pack_calls().is_empty() }); - let second = std::thread::spawn({ + let second = on_own_thread({ let backend = fixture.backend.clone(); move || tree_range(&backend, 50, 10).ok() }); @@ -616,8 +623,8 @@ fn two_threads_that_miss_one_pack_make_one_storage_read() { assert_eq!( ( first_read_started, - first.join().ok().flatten(), - second.join().ok().flatten(), + first.recv_timeout(LIMIT).ok().flatten(), + second.recv_timeout(LIMIT).ok().flatten(), fixture.pack_calls() ), ( diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs index 7acfc8d002..b6aeac819b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -25,7 +25,7 @@ use pretty_assertions::assert_eq; use std::path::Path; use std::sync::Arc; use std::time::Duration; -use test_r::test; +use test_r::{test, timeout}; use tokio_util::task::TaskTracker; use uuid::Uuid; @@ -137,6 +137,7 @@ async fn a_publish_whose_answer_is_lost_deletes_the_file_and_gives_the_error() { } #[test] +#[timeout("60s")] async fn a_publish_that_reaches_the_deadline_deletes_the_file_that_the_storage_wrote() { let (files, _, inner) = files( Script::NeverAnswer, @@ -158,6 +159,7 @@ async fn a_publish_that_reaches_the_deadline_deletes_the_file_that_the_storage_w } #[test] +#[timeout("60s")] async fn a_publish_that_the_caller_drops_deletes_the_file_in_a_task_of_the_tracker() { // The delete waits for the gate, so the test reads the file that the dropped write left // before the task of the tracker deletes it. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index b935b01f98..fa7df64f6e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -47,7 +47,7 @@ use std::sync::Arc; use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; use std::time::Duration; use test_r::core::DynamicTestRegistration; -use test_r::{test, test_gen}; +use test_r::{test, test_gen, timeout}; /// The longest time that a test waits for an operation or for the work of a store to end. const LIMIT: Duration = Duration::from_secs(10); @@ -365,6 +365,7 @@ async fn a_save_whose_index_write_fails_publishes_nothing_and_leaves_the_name_fr } #[test] +#[timeout("60s")] async fn a_publish_that_reaches_the_deadline_and_lands_late_publishes_nothing() { let hang = Arc::new(AtomicBool::new(true)); let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { @@ -452,6 +453,7 @@ async fn a_second_save_of_an_unchanged_tree_writes_no_pack() { } #[test] +#[timeout("60s")] async fn two_stores_that_create_one_repository_at_the_same_time_both_save() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { @@ -514,6 +516,7 @@ async fn two_stores_that_create_one_repository_at_the_same_time_both_save() { } #[test] +#[timeout("60s")] async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { // The first index write after the arm waits at the gate. That is the index write of the // second save, so its packs are in no index while the delete prunes. @@ -1059,6 +1062,7 @@ async fn a_deleted_scope_holds_no_blob() { } #[test] +#[timeout("60s")] async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_name_free() { // The first save counts the calls of a save. Each later round holds one of these calls: the // call reaches the storage and never answers, as a write that S3 received and completes after @@ -1162,6 +1166,7 @@ async fn a_blob_call_of_a_cancelled_operation_does_not_start() { } #[test] +#[timeout("60s")] async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_call() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { if op_label == "copy_list" { @@ -1211,6 +1216,7 @@ async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_cal } #[test] +#[timeout("60s")] async fn a_publish_held_at_its_storage_call_keeps_shut_down_waiting_until_it_ends() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { if op_label == "publish" { @@ -1255,6 +1261,7 @@ async fn a_publish_held_at_its_storage_call_keeps_shut_down_waiting_until_it_end } #[test] +#[timeout("60s")] async fn a_save_that_reaches_its_publish_after_shut_down_publishes_nothing() { // The gate holds the save after its blocking work, so the tracker is empty and `shut_down` // returns before the publish starts. @@ -1327,6 +1334,7 @@ async fn delete_scope_and_copy_scope_after_shut_down_give_storage() { } #[test] +#[timeout("60s")] async fn shut_down_ends_running_operations_before_it_returns() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { @@ -1382,6 +1390,7 @@ enum Dropped { } #[test] +#[timeout("60s")] async fn a_dropped_operation_stops_its_blocking_work() { let outcomes = futures::stream::iter([ Dropped::Restore, @@ -2191,6 +2200,7 @@ async fn the_storage_calls_of_a_restore_run_at_the_nice_value_of_the_process() { #[cfg(target_os = "linux")] #[test] +#[timeout("60s")] async fn after_saves_and_prunes_the_pools_keep_the_nice_value_of_the_process() { // All tasks wait for each other, so each runs on its own thread of the blocking pool, and the // idle threads that ran the saves and the prune are among them. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index d10f6697e4..f367be34f8 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -49,7 +49,7 @@ use std::path::{Path, PathBuf}; use std::sync::Arc; use std::sync::atomic::{AtomicUsize, Ordering}; use std::time::Duration; -use test_r::test; +use test_r::{test, timeout}; use tokio::runtime::Handle; use tokio::sync::{Notify, oneshot, watch}; use tokio::time::error::Elapsed; @@ -477,6 +477,7 @@ async fn each_scope_is_its_own_repository() { } #[test] +#[timeout("60s")] async fn a_restore_reads_data_on_at_most_its_reader_threads() { // Each save adds one pack with the data of its new file. The restore of the second snapshot // reads the data of both packs, one read for each pack. @@ -749,6 +750,7 @@ impl BlobStorage for OverlapCountingStorage { } #[test] +#[timeout("60s")] async fn a_save_whose_pack_write_gets_no_answer_fails_with_no_snapshot_and_its_threads_stop() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -774,6 +776,7 @@ async fn a_save_whose_pack_write_gets_no_answer_fails_with_no_snapshot_and_its_t } #[test] +#[timeout("60s")] async fn a_save_whose_pack_writes_answer_before_the_deadline_succeeds_and_restores() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -815,6 +818,7 @@ async fn a_save_whose_pack_writes_answer_before_the_deadline_succeeds_and_restor } #[test] +#[timeout("60s")] async fn a_restore_whose_data_pack_reads_get_no_answer_fails_and_stops_its_threads() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -872,6 +876,7 @@ async fn a_restore_whose_data_pack_reads_get_no_answer_fails_and_stops_its_threa } #[test] +#[timeout("60s")] async fn a_prune_whose_tree_pack_reads_get_no_answer_fails_and_stops_its_threads() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); From c6214409d15bf4f5b5133b8e2138a28f187bf377 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 10:33:47 -0700 Subject: [PATCH 044/126] Wait for a blob call of the store and of the backend in one function --- .../src/filesystem_snapshot/rustic/backend.rs | 33 +++++++++++-------- .../src/filesystem_snapshot/rustic/files.rs | 12 ++----- 2 files changed, 21 insertions(+), 24 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index cd994935bc..406bd18f91 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -163,21 +163,8 @@ impl BlobBackend { path: &Path, future: impl Future>, ) -> RusticResult { - if self.cancel.is_cancelled() { - return Err(storage_error( - call, - path, - anyhow::Error::new(OperationCancelled), - )); - } self.runtime - .block_on(async { - tokio::select! { - biased; - answer = answer_within(self.deadline, future) => answer, - () = self.cancel.cancelled() => Err(anyhow::Error::new(OperationCancelled)), - } - }) + .block_on(answer_or_cancel(self.deadline, &self.cancel, future)) .map_err(|error| storage_error(call, path, error)) } @@ -216,6 +203,24 @@ pub(super) async fn answer_within( }) } +/// Gives the output of the future within the deadline, or an error when the operation of the token +/// is cancelled. A call of a cancelled operation does not start, and a cancel ends a call that +/// runs. +pub(super) async fn answer_or_cancel( + deadline: Duration, + cancel: &CancellationToken, + future: impl Future>, +) -> anyhow::Result { + if cancel.is_cancelled() { + return Err(anyhow::Error::new(OperationCancelled)); + } + tokio::select! { + biased; + answer = answer_within(deadline, future) => answer, + () = cancel.cancelled() => Err(anyhow::Error::new(OperationCancelled)), + } +} + impl ReadBackend for BlobBackend { fn location(&self) -> String { format!("golem-blob-storage:{:?}", self.namespace) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs index 8d8119ba69..e60817137c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs @@ -16,8 +16,7 @@ //! //! Each call waits for at most the deadline of the scope, and a cancel of its operation ends it. -use super::backend::answer_within; -use super::fault::OperationCancelled; +use super::backend::answer_or_cancel; use golem_service_base::storage::blob::{ BlobStorage, BlobStorageNamespace, ListedBlob, PutIfAbsent, }; @@ -46,14 +45,7 @@ impl SnapshotFiles { &self, future: impl Future>, ) -> anyhow::Result { - if self.cancel.is_cancelled() { - return Err(anyhow::Error::new(OperationCancelled)); - } - tokio::select! { - biased; - answer = answer_within(self.deadline, future) => answer, - () = self.cancel.cancelled() => Err(anyhow::Error::new(OperationCancelled)), - } + answer_or_cancel(self.deadline, &self.cancel, future).await } /// Gives the content of the blob at the path, or `None` when the path has no blob. From c3347804d5f08209469de107c41d38bd6f786324 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 10:48:58 -0700 Subject: [PATCH 045/126] Take a claim before a prune, so concurrent deletes make one prune --- .../src/filesystem_snapshot/rustic/prune.rs | 179 ++++++++++++++++- .../src/filesystem_snapshot/rustic/store.rs | 35 +++- .../filesystem_snapshot/rustic/store/tests.rs | 186 +++++++++++++++++- 3 files changed, 383 insertions(+), 17 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 90ec11fd06..f84b25977e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -18,17 +18,26 @@ //! packed bytes that deleted snapshots added since the last prune, the time of the last prune, and //! whether that prune marked packs that a later prune removes. Two deletes at the same time can //! lose a count. A lost count only delays a prune. +//! +//! A delete whose prune is due takes a claim before it prunes, so two deletes that read the same +//! ledger make one prune. The claims of a ledger are in one directory, named by the time of the last +//! prune in that ledger. use super::files::SnapshotFiles; +use futures::{StreamExt, TryStreamExt, stream}; use golem_common::model::Timestamp; +use golem_service_base::storage::blob::PutIfAbsent; use serde::{Deserialize, Serialize}; -use std::path::Path; +use std::path::{Path, PathBuf}; use std::time::Duration; use tracing::warn; /// The path of the ledger blob, relative to the root of the namespace of the scope. pub(super) const LEDGER_PATH: &str = "golem/prune-ledger"; +/// The directory of the prune claims, relative to the root of the namespace of the scope. +const CLAIMS_PATH: &str = "golem/prune-claims"; + /// What the scope did since its last prune. #[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] pub(super) struct PruneLedger { @@ -75,12 +84,17 @@ impl Percent { /// Tells whether the grace period passed at `now` since the last prune. fn grace_passed(ledger: &PruneLedger, now: Timestamp, grace: Duration) -> bool { - ledger.last_prune.is_none_or(|last| { - now.to_millis() - >= last - .to_millis() - .saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) - }) + ledger + .last_prune + .is_none_or(|last| passed_since(last, now, grace)) +} + +/// Tells whether the grace period passed at `now` since the time. +fn passed_since(time: Timestamp, now: Timestamp, grace: Duration) -> bool { + now.to_millis() + >= time + .to_millis() + .saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) } /// Tells whether [`prune_due`] needs the size of the repository at `now`. Only freed bytes after @@ -142,12 +156,132 @@ pub(super) async fn write_ledger( .await } +/// A claim of a prune that a listing found: its number, and its time when its content parses. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) struct ListedClaim { + pub(super) number: u64, + pub(super) claimed_at: Option, +} + +/// What a delete whose prune is due does with the claims of its ledger. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) enum ClaimChoice { + /// Take the claim with the number, and prune when the write of the claim succeeds. + Claim(u64), + /// Another prune holds the claims of the ledger, so do not prune. + Held, +} + +/// Chooses the claim of a delete from the claims of its ledger. The newest claim holds the ledger +/// until the grace period passed since its time. A claim whose content does not parse is old. +pub(super) fn next_claim(claims: &[ListedClaim], now: Timestamp, grace: Duration) -> ClaimChoice { + match claims.iter().max_by_key(|claim| claim.number) { + None => ClaimChoice::Claim(0), + Some(newest) + if newest + .claimed_at + .is_some_and(|at| !passed_since(at, now, grace)) => + { + ClaimChoice::Held + } + Some(newest) => ClaimChoice::Claim(newest.number.saturating_add(1)), + } +} + +/// Gives the directory of the claims of the ledger: the time of its last prune in milliseconds, or +/// `none`. +pub(super) fn claims_directory(ledger: &PruneLedger) -> PathBuf { + let generation = ledger + .last_prune + .map_or_else(|| "none".to_string(), |last| last.to_millis().to_string()); + Path::new(CLAIMS_PATH).join(generation) +} + +/// Lists the claims in the directory. A name that is not a number is not a claim, and a claim +/// that a delete removed after the listing counts as old. +pub(super) async fn list_claims( + files: &SnapshotFiles, + directory: &Path, +) -> anyhow::Result> { + let listed = files.list_below("list_claims", directory).await?; + let numbered = listed + .iter() + .filter_map(|blob| { + let number = blob.path.file_name()?.to_str()?.parse::().ok()?; + Some((number, blob.path.clone())) + }) + .collect::>(); + stream::iter(numbered) + .then(|(number, path)| async move { + let content = files.get("read_claim", &path).await?; + Ok::<_, anyhow::Error>(ListedClaim { + number, + claimed_at: content.as_deref().and_then(parse_claim), + }) + }) + .try_collect() + .await +} + +/// Writes the claim with the number, and tells whether this call wrote it. +pub(super) async fn take_claim( + files: &SnapshotFiles, + directory: &Path, + number: u64, + now: Timestamp, +) -> anyhow::Result { + let content = now.to_millis().to_string(); + let written = files + .put_if_absent( + "write_claim", + &directory.join(number.to_string()), + content.as_bytes(), + ) + .await?; + Ok(written == PutIfAbsent::Written) +} + +/// Deletes the claim with the number. A failure gives a warning, because a claim only delays a +/// prune until its grace period passed. +pub(super) async fn release_claim(files: &SnapshotFiles, directory: &Path, number: u64) { + if let Err(error) = files + .delete("delete_claim", &directory.join(number.to_string())) + .await + { + warn!( + error = %format!("{error:#}"), + "Failed to delete the prune claim of a filesystem snapshot scope" + ); + } +} + +/// Deletes the claims of a ledger after its prune. A failure gives a warning, because a claim only +/// delays a prune until its grace period passed. +pub(super) async fn end_claims(files: &SnapshotFiles, directory: &Path) { + if let Err(error) = files.delete_dir("delete_claims", directory).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete the prune claims of a filesystem snapshot scope" + ); + } +} + +/// Reads the time of a claim, in milliseconds. +fn parse_claim(content: &[u8]) -> Option { + std::str::from_utf8(content) + .ok()? + .trim() + .parse::() + .ok() + .map(Timestamp::from) +} + #[cfg(test)] mod tests { use super::super::files::SnapshotFiles; use super::{ - LEDGER_PATH, Percent, PruneLedger, needs_repository_size, prune_due, read_ledger, - write_ledger, + ClaimChoice, LEDGER_PATH, ListedClaim, Percent, PruneLedger, needs_repository_size, + next_claim, prune_due, read_ledger, write_ledger, }; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; @@ -291,6 +425,33 @@ mod tests { ); } + #[test] + fn the_newest_claim_holds_a_ledger_until_the_grace_period_passed_since_its_time() { + let now = 10_000_000; + let claim = |number, claimed_at: Option| ListedClaim { + number, + claimed_at: claimed_at.map(Timestamp::from), + }; + let choose = |claims: &[ListedClaim]| next_claim(claims, at(now), GRACE); + + assert_eq!( + [ + choose(&[]), + choose(&[claim(0, Some(now - GRACE_MILLIS)), claim(1, Some(now - 1))]), + choose(&[claim(1, Some(now)), claim(2, Some(now - GRACE_MILLIS))]), + choose(&[claim(4, None)]), + choose(&[claim(0, Some(now - 1)), claim(3, None)]), + ], + [ + ClaimChoice::Claim(0), + ClaimChoice::Held, + ClaimChoice::Claim(3), + ClaimChoice::Claim(5), + ClaimChoice::Claim(4), + ] + ); + } + #[test] fn a_delete_adds_its_bytes_and_a_prune_starts_the_ledger_again() { let deleted = ledger(5, Some(7), true) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 697ba735d7..e8637c7802 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -25,8 +25,9 @@ use super::fault::{Operation, classify, is_file_missing, is_storage_failure, sto use super::files::SnapshotFiles; use super::priority::LowPriority; use super::prune::{ - Percent, PruneLedger, needs_repository_size, prune_due, read_ledger, repository_bytes, - write_ledger, + ClaimChoice, Percent, PruneLedger, claims_directory, end_claims, list_claims, + needs_repository_size, next_claim, prune_due, read_ledger, release_claim, repository_bytes, + take_claim, write_ledger, }; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -291,6 +292,8 @@ impl RusticSnapshotStore { /// Adds the freed bytes to the ledger of the scope, and prunes the repository when a prune is /// due. The ledger keeps the freed bytes before the prune starts, so a delete that runs again /// after a failed prune prunes again. It lists the packs only when their size can make a prune due. + /// A due prune runs only after the delete takes a claim of its ledger. A failed prune deletes + /// the claim, and a prune that succeeds deletes each claim of its ledger. async fn prune_when_due( &self, scope: &SnapshotScope, @@ -317,19 +320,41 @@ impl RusticSnapshotStore { if !prune_due(&ledger, now, size, self.policy.prune_threshold, grace) { return Ok(()); } + let claims = claims_directory(&ledger); + let listed = list_claims(&files, &claims) + .await + .map_err(storage_failure)?; + let ClaimChoice::Claim(number) = next_claim(&listed, now, grace) else { + return Ok(()); + }; + if !take_claim(&files, &claims, number, now) + .await + .map_err(storage_failure)? + { + return Ok(()); + } let backend = Arc::new(self.backend(scope, token)?); let key = self.key.clone(); let settings = self.policy.prune; let low_priority = self.low_priority; - let report = self + let pruned = self .blocking(Operation::Prune, move || { low_priority.run("fs-snap-prune", move || prune(backend, &key, &settings)) }) - .await?; + .await; + let report = match pruned { + Ok(report) => report, + Err(error) => { + release_claim(&files, &claims, number).await; + return Err(error); + } + }; let marked_packs = report.as_ref().is_some_and(leaves_marked_packs); write_ledger(&files, &PruneLedger::after_prune(now, marked_packs)) .await - .map_err(storage_failure) + .map_err(storage_failure)?; + end_claims(&files, &claims).await; + Ok(()) } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index fa7df64f6e..83ebd818d0 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -907,6 +907,177 @@ async fn a_failed_listing_of_the_packs_gives_storage_and_records_no_prune() { assert_eq!((after.freed_bytes > 0, after.last_prune), (true, None)); } +/// Counts the prunes among the recorded calls. A prune lists the packs, and no other step of a +/// delete makes that call. +fn prunes(calls: &[(&'static str, String)]) -> usize { + calls + .iter() + .filter(|(op_label, path)| *op_label == "list" && path == "data") + .count() +} + +/// Saves a tree of one file with the content under each name. +async fn save_each(store: &RusticSnapshotStore, scope: &SnapshotScope, names: &[&str]) { + futures::stream::iter(names) + .for_each(|text| async move { + let tree = one_file_tree(text); + store + .save(scope, &name(text), tree.path(), None) + .await + .unwrap(); + }) + .await; +} + +#[test] +#[timeout("60s")] +async fn two_deletes_that_read_the_same_ledger_make_one_prune() { + // The first read after the first claim is the start of the first prune. The gate holds it, + // so the second delete reads the ledger that the first delete read. + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, _| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "read" + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let first = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let first_held = eventually(|| held.load(Ordering::SeqCst)).await; + + let second = store.delete(&scope, &name("p-2")).await; + storage.open_gate(); + let first = tokio::time::timeout(LIMIT, first).await; + let calls = storage.calls(); + + assert!(matches!(first, Ok(Ok(Ok(())))), "{first:?}"); + assert!(second.is_ok(), "{second:?}"); + assert_eq!( + ( + first_held, + prunes(&calls), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, 1, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() { + // The first call after the first claim is the start of the first prune, and it fails before + // the prune lists the packs. So only the retry counts as a prune. + let claimed = Arc::new(AtomicBool::new(false)); + let refused = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, refused) = (claimed.clone(), refused.clone()); + move |op_label, _| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + Script::Pass + } else if claimed.load(Ordering::SeqCst) && !refused.swap(true, Ordering::SeqCst) { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + let retried = store.delete(&scope, &name("p-1")).await; + let after = ledger(&storage, &scope).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!(retried.is_ok(), "{retried:?}"); + assert_eq!( + ( + claims_after_failure, + after.last_prune.is_some(), + prunes(&storage.calls()) + ), + (Vec::::new(), true, 1) + ); +} + +#[test] +#[timeout("60s")] +async fn a_claim_older_than_the_grace_period_does_not_block_a_prune() { + let grace = Duration::from_secs(3600); + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let old = golem_common::model::Timestamp::now_utc() + .to_millis() + .saturating_sub(2 * 3_600_000); + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-claims/none/0"), + old.to_string().as_bytes(), + ) + .await + .unwrap(); + + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert!(ledger(&storage, &scope).await.last_prune.is_some()); +} + +#[test] +#[timeout("60s")] +async fn a_prune_that_succeeds_deletes_the_claims_of_its_ledger() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert_eq!( + ( + ledger(&storage, &scope).await.last_prune.is_some(), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, Vec::::new()) + ); +} + #[test] async fn a_delete_below_the_threshold_does_not_prune() { let storage = Arc::new(InMemoryBlobStorage::new()); @@ -2133,8 +2304,8 @@ async fn the_storage_calls_of_a_save_run_at_nice_19() { #[cfg(target_os = "linux")] #[test] async fn the_storage_calls_of_a_prune_run_at_nice_19() { - // The forget of a delete runs before the ledger read at the normal priority. The ledger calls - // and the listing of the packs run on the async runtime, and the prune runs after them. + // The forget of a delete runs before the ledger read at the normal priority. The ledger calls, + // the listing of the packs and the claim calls run on the async runtime. let (storage, calls) = nice_recording_storage(); let store = store(storage, policy(LONG_DEADLINE, ALWAYS, Duration::ZERO)); let scope = new_scope(); @@ -2154,7 +2325,16 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { .into_iter() .skip_while(|(op_label, _, _)| op_label != "read_ledger") .filter(|(op_label, _, _)| { - !["read_ledger", "write_ledger", "list_data"].contains(&op_label.as_str()) + ![ + "read_ledger", + "write_ledger", + "list_data", + "list_claims", + "read_claim", + "write_claim", + "delete_claims", + ] + .contains(&op_label.as_str()) }) .collect::>(); From 9c940503c37264f943db8d7c4bae926a687c7430 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 10:49:20 -0700 Subject: [PATCH 046/126] Say that the prune threshold is 10 percent of the size, rounded down --- .../src/filesystem_snapshot/rustic/prune.rs | 7 ++++--- .../src/filesystem_snapshot/rustic/store.rs | 2 +- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index f84b25977e..88ac3bc1e1 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -106,8 +106,9 @@ pub(super) fn needs_repository_size(ledger: &PruneLedger, now: Timestamp, grace: /// Tells whether a prune is due at `now`. /// /// A prune is due when the grace period passed since the last prune, and the freed bytes reach the -/// threshold share of `repository_bytes` or the last prune marked packs. A threshold of zero bytes -/// counts as one byte, so a prune never runs for a scope that freed nothing and marked nothing. +/// threshold share of `repository_bytes`, rounded down to a whole byte, or the last prune marked +/// packs. A threshold of zero bytes counts as one byte, so a prune never runs for a scope that +/// freed nothing and marked nothing. pub(super) fn prune_due( ledger: &PruneLedger, now: Timestamp, @@ -327,7 +328,7 @@ mod tests { } #[test] - fn a_prune_is_due_when_the_freed_bytes_reach_ten_percent_of_the_repository() { + fn a_prune_is_due_when_the_freed_bytes_reach_ten_percent_of_the_repository_rounded_down() { let now = at(10_000_000); let due = |freed, repository_bytes| { prune_due( diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index e8637c7802..acfc3a3684 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -62,7 +62,7 @@ use tokio_util::sync::{CancellationToken, DropGuard}; use tokio_util::task::TaskTracker; /// The share of the size of the repository that deleted snapshots must free before a delete prunes -/// the scope. +/// the scope. The threshold is 10% of the size, rounded down to a whole byte, so about 10%. const PRUNE_THRESHOLD: Percent = Percent(10); /// How long a pack that a prune marks stays before a later prune deletes it. It is also the From d83cd13269ec1faa0611d1ca4504993bb4aa96aa Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 10:51:46 -0700 Subject: [PATCH 047/126] Check that a claim is taken one time and listed with its time --- .../src/filesystem_snapshot/rustic/prune.rs | 30 +++++++++++++++++-- 1 file changed, 28 insertions(+), 2 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 88ac3bc1e1..fb8795a02b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -281,8 +281,8 @@ fn parse_claim(content: &[u8]) -> Option { mod tests { use super::super::files::SnapshotFiles; use super::{ - ClaimChoice, LEDGER_PATH, ListedClaim, Percent, PruneLedger, needs_repository_size, - next_claim, prune_due, read_ledger, write_ledger, + ClaimChoice, LEDGER_PATH, ListedClaim, Percent, PruneLedger, claims_directory, list_claims, + needs_repository_size, next_claim, prune_due, read_ledger, take_claim, write_ledger, }; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; @@ -453,6 +453,32 @@ mod tests { ); } + #[test] + async fn a_claim_is_taken_one_time_and_listed_with_its_time() { + let files = new_files(); + let directory = claims_directory(&ledger(1, Some(42), false)); + let now = at(10_000_000); + + let first = take_claim(&files, &directory, 0, now).await.unwrap(); + let again = take_claim(&files, &directory, 0, at(20_000_000)) + .await + .unwrap(); + let listed = list_claims(&files, &directory).await.unwrap(); + + assert_eq!( + (directory.display().to_string(), first, again, listed), + ( + "golem/prune-claims/42".to_string(), + true, + false, + vec![ListedClaim { + number: 0, + claimed_at: Some(now) + }] + ) + ); + } + #[test] fn a_delete_adds_its_bytes_and_a_prune_starts_the_ledger_again() { let deleted = ledger(5, Some(7), true) From 5e67f23f2a9d0fdc83f51037d19e70f9ad52e58d Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 11:10:48 -0700 Subject: [PATCH 048/126] Prune only when the ledger did not change after the claim --- .../src/filesystem_snapshot/rustic/store.rs | 18 +++++++- .../filesystem_snapshot/rustic/store/tests.rs | 45 +++++++++++++++++++ 2 files changed, 61 insertions(+), 2 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index acfc3a3684..289eb584c8 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -292,8 +292,9 @@ impl RusticSnapshotStore { /// Adds the freed bytes to the ledger of the scope, and prunes the repository when a prune is /// due. The ledger keeps the freed bytes before the prune starts, so a delete that runs again /// after a failed prune prunes again. It lists the packs only when their size can make a prune due. - /// A due prune runs only after the delete takes a claim of its ledger. A failed prune deletes - /// the claim, and a prune that succeeds deletes each claim of its ledger. + /// A due prune runs only after the delete takes a claim of its ledger, and only when the ledger + /// did not change after the claim. A failed prune deletes the claim, and a prune that succeeds + /// deletes each claim of its ledger. async fn prune_when_due( &self, scope: &SnapshotScope, @@ -333,6 +334,19 @@ impl RusticSnapshotStore { { return Ok(()); } + // A prune writes its ledger before it deletes the claims, so a delete that claims in a + // directory that such a prune removed sees the new ledger here. + match read_ledger(&files).await { + Ok(again) if claims_directory(&again) == claims => {} + Ok(_) => { + release_claim(&files, &claims, number).await; + return Ok(()); + } + Err(error) => { + release_claim(&files, &claims, number).await; + return Err(storage_failure(error)); + } + } let backend = Arc::new(self.backend(scope, token)?); let key = self.key.clone(); let settings = self.policy.prune; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 83ebd818d0..3dd5a17846 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -982,6 +982,51 @@ async fn two_deletes_that_read_the_same_ledger_make_one_prune() { ); } +#[test] +#[timeout("60s")] +async fn a_delete_that_claims_after_another_prune_removed_the_claims_does_not_prune() { + // The gate holds the first delete after its ledger read and before its listing of the claims. + // The second delete prunes to its end, so the first delete claims in a removed directory. + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let held = held.clone(); + move |op_label, _| { + if op_label == "list_claims" && !held.swap(true, Ordering::SeqCst) { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let late = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let late_held = eventually(|| held.load(Ordering::SeqCst)).await; + + let pruned = store.delete(&scope, &name("p-2")).await; + storage.open_gate(); + let late = tokio::time::timeout(LIMIT, late).await; + + assert!(pruned.is_ok(), "{pruned:?}"); + assert!(matches!(late, Ok(Ok(Ok(())))), "{late:?}"); + assert_eq!( + ( + late_held, + prunes(&storage.calls()), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, 1, Vec::::new()) + ); +} + #[test] #[timeout("60s")] async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() { From bc746e2cd1412cea721fab534b31f53b29ece467 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 12:16:22 -0700 Subject: [PATCH 049/126] Check that a failed prune and a failed second read each delete the claim --- .../filesystem_snapshot/rustic/store/tests.rs | 57 ++++++++++++++++++- 1 file changed, 54 insertions(+), 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 3dd5a17846..7fd5332540 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1029,9 +1029,9 @@ async fn a_delete_that_claims_after_another_prune_removed_the_claims_does_not_pr #[test] #[timeout("60s")] -async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() { - // The first call after the first claim is the start of the first prune, and it fails before - // the prune lists the packs. So only the retry counts as a prune. +async fn a_failed_second_read_of_the_ledger_deletes_the_claim_and_a_retry_of_the_delete_prunes() { + // The first call after the first claim is the second read of the ledger, and it fails. So the + // first delete does not prune, and only the retry counts as a prune. let claimed = Arc::new(AtomicBool::new(false)); let refused = Arc::new(AtomicBool::new(false)); let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { @@ -1074,6 +1074,57 @@ async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() ); } +#[test] +#[timeout("60s")] +async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() { + // The first listing of the packs by a prune after the first claim fails. Only a prune lists + // the packs with that call, so the second read of the ledger passes and the prune fails. + let claimed = Arc::new(AtomicBool::new(false)); + let refused = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, refused) = (claimed.clone(), refused.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !refused.swap(true, Ordering::SeqCst) + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + let retried = store.delete(&scope, &name("p-1")).await; + let after = ledger(&storage, &scope).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!(retried.is_ok(), "{retried:?}"); + assert_eq!( + ( + claims_after_failure, + refused.load(Ordering::SeqCst), + after.last_prune.is_some() + ), + (Vec::::new(), true, true) + ); +} + #[test] #[timeout("60s")] async fn a_claim_older_than_the_grace_period_does_not_block_a_prune() { From d9f87193a2b22fb286425b09b9e2fda5d74aa9af Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 15:12:43 -0700 Subject: [PATCH 050/126] Keep the freed bytes of each delete in a record of its own, so no delete writes back an old ledger --- .../src/filesystem_snapshot/rustic/prune.rs | 200 +++++++++++++----- .../src/filesystem_snapshot/rustic/store.rs | 34 +-- .../filesystem_snapshot/rustic/store/tests.rs | 176 +++++++++++++-- 3 files changed, 330 insertions(+), 80 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index fb8795a02b..415fbcadf9 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -15,9 +15,9 @@ //! When a delete of the store prunes the repository of its scope. //! //! The scope keeps a small ledger blob next to the files of the repository. The ledger holds the -//! packed bytes that deleted snapshots added since the last prune, the time of the last prune, and -//! whether that prune marked packs that a later prune removes. Two deletes at the same time can -//! lose a count. A lost count only delays a prune. +//! time of the last prune, and whether that prune marked packs that a later prune removes. Only the +//! delete that holds the claim of a prune writes it. Each delete that freed bytes writes a record +//! of its own, with the count in the name, and a prune that succeeds deletes the records it counted. //! //! A delete whose prune is due takes a claim before it prunes, so two deletes that read the same //! ledger make one prune. The claims of a ledger are in one directory, named by the time of the last @@ -26,7 +26,7 @@ use super::files::SnapshotFiles; use futures::{StreamExt, TryStreamExt, stream}; use golem_common::model::Timestamp; -use golem_service_base::storage::blob::PutIfAbsent; +use golem_service_base::storage::blob::{ListedBlob, PutIfAbsent}; use serde::{Deserialize, Serialize}; use std::path::{Path, PathBuf}; use std::time::Duration; @@ -38,11 +38,12 @@ pub(super) const LEDGER_PATH: &str = "golem/prune-ledger"; /// The directory of the prune claims, relative to the root of the namespace of the scope. const CLAIMS_PATH: &str = "golem/prune-claims"; +/// The directory of the records of freed bytes, relative to the root of the namespace of the scope. +const FREED_PATH: &str = "golem/prune-freed"; + /// What the scope did since its last prune. #[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] pub(super) struct PruneLedger { - /// The packed bytes that the deleted snapshots added, since the last prune. - pub(super) freed_bytes: u64, /// The time of the last prune. pub(super) last_prune: Option, /// Whether the last prune marked packs that a later prune removes. @@ -50,18 +51,9 @@ pub(super) struct PruneLedger { } impl PruneLedger { - /// Gives the ledger after a delete of snapshots that added `bytes` packed bytes. - pub(super) fn with_deleted(self, bytes: u64) -> Self { - Self { - freed_bytes: self.freed_bytes.saturating_add(bytes), - ..self - } - } - /// Gives the ledger after a prune at `now` that marked packs or not. pub(super) fn after_prune(now: Timestamp, marked_packs: bool) -> Self { Self { - freed_bytes: 0, last_prune: Some(now), awaiting_removal: marked_packs, } @@ -99,8 +91,13 @@ fn passed_since(time: Timestamp, now: Timestamp, grace: Duration) -> bool { /// Tells whether [`prune_due`] needs the size of the repository at `now`. Only freed bytes after /// the grace period, without marked packs, need it. -pub(super) fn needs_repository_size(ledger: &PruneLedger, now: Timestamp, grace: Duration) -> bool { - grace_passed(ledger, now, grace) && ledger.freed_bytes > 0 && !ledger.awaiting_removal +pub(super) fn needs_repository_size( + ledger: &PruneLedger, + freed_bytes: u64, + now: Timestamp, + grace: Duration, +) -> bool { + grace_passed(ledger, now, grace) && freed_bytes > 0 && !ledger.awaiting_removal } /// Tells whether a prune is due at `now`. @@ -111,13 +108,13 @@ pub(super) fn needs_repository_size(ledger: &PruneLedger, now: Timestamp, grace: /// freed nothing and marked nothing. pub(super) fn prune_due( ledger: &PruneLedger, + freed_bytes: u64, now: Timestamp, repository_bytes: u64, threshold: Percent, grace: Duration, ) -> bool { - let work = - ledger.freed_bytes >= threshold.of(repository_bytes).max(1) || ledger.awaiting_removal; + let work = freed_bytes >= threshold.of(repository_bytes).max(1) || ledger.awaiting_removal; grace_passed(ledger, now, grace) && work } @@ -157,6 +154,71 @@ pub(super) async fn write_ledger( .await } +/// The freed bytes of the records that a listing found, and the paths of the records it counted. +#[derive(Clone, Debug, Default, PartialEq, Eq)] +pub(super) struct FreedRecords { + pub(super) bytes: u64, + pub(super) counted: Vec>, +} + +/// Reads the freed bytes from the name of a record, `-`. +pub(super) fn parse_freed(name: &str) -> Option { + let (bytes, unique) = name.split_once('-')?; + if unique.is_empty() { + return None; + } + bytes.parse().ok() +} + +/// Sums the freed bytes of the listed records. A record whose name does not parse counts as zero +/// bytes, and it is not counted, so a prune leaves it in place. +pub(super) fn count_freed(listed: &[ListedBlob]) -> FreedRecords { + listed + .iter() + .filter_map(|blob| { + let bytes = parse_freed(blob.path.file_name()?.to_str()?)?; + Some((bytes, blob.path.clone())) + }) + .fold(FreedRecords::default(), |records, (bytes, path)| { + let mut counted = records.counted; + counted.push(path); + FreedRecords { + bytes: records.bytes.saturating_add(bytes), + counted, + } + }) +} + +/// Writes a record of the freed bytes of one delete. +pub(super) async fn record_freed(files: &SnapshotFiles, bytes: u64) -> anyhow::Result<()> { + let path = Path::new(FREED_PATH).join(format!("{bytes}-{}", uuid::Uuid::new_v4())); + files.put("write_freed", &path, &[]).await +} + +/// Lists the records of freed bytes, and sums them. +pub(super) async fn list_freed(files: &SnapshotFiles) -> anyhow::Result { + Ok(count_freed( + &files + .list_below("list_freed", Path::new(FREED_PATH)) + .await?, + )) +} + +/// Deletes the counted records after a prune. A failure gives a warning, because a record that +/// stays only makes the next prune come earlier. +pub(super) async fn remove_freed(files: &SnapshotFiles, records: &FreedRecords) { + stream::iter(&records.counted) + .for_each(|path| async move { + if let Err(error) = files.delete("delete_freed", path).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete a record of freed bytes of a filesystem snapshot scope" + ); + } + }) + .await; +} + /// A claim of a prune that a listing found: its number, and its time when its content parses. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub(super) struct ListedClaim { @@ -281,12 +343,14 @@ fn parse_claim(content: &[u8]) -> Option { mod tests { use super::super::files::SnapshotFiles; use super::{ - ClaimChoice, LEDGER_PATH, ListedClaim, Percent, PruneLedger, claims_directory, list_claims, - needs_repository_size, next_claim, prune_due, read_ledger, take_claim, write_ledger, + ClaimChoice, FREED_PATH, FreedRecords, LEDGER_PATH, ListedClaim, Percent, PruneLedger, + claims_directory, count_freed, list_claims, list_freed, needs_repository_size, next_claim, + parse_freed, prune_due, read_ledger, record_freed, take_claim, write_ledger, }; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; use golem_service_base::storage::blob::BlobStorageNamespace; + use golem_service_base::storage::blob::ListedBlob; use golem_service_base::storage::blob::memory::InMemoryBlobStorage; use pretty_assertions::assert_eq; use std::path::Path; @@ -304,13 +368,8 @@ mod tests { Timestamp::from(millis) } - fn ledger( - freed_bytes: u64, - last_prune_millis: Option, - awaiting_removal: bool, - ) -> PruneLedger { + fn ledger(last_prune_millis: Option, awaiting_removal: bool) -> PruneLedger { PruneLedger { - freed_bytes, last_prune: last_prune_millis.map(Timestamp::from), awaiting_removal, } @@ -332,7 +391,8 @@ mod tests { let now = at(10_000_000); let due = |freed, repository_bytes| { prune_due( - &ledger(freed, None, false), + &ledger(None, false), + freed, now, repository_bytes, TEN_PERCENT, @@ -357,7 +417,8 @@ mod tests { let last = 1_000_000; let full = |now| { prune_due( - &ledger(1000, Some(last), true), + &ledger(Some(last), true), + 1000, at(now), 1000, TEN_PERCENT, @@ -382,7 +443,8 @@ mod tests { let after_grace = at(last + GRACE_MILLIS); let due = |awaiting_removal| { prune_due( - &ledger(0, Some(last), awaiting_removal), + &ledger(Some(last), awaiting_removal), + 0, after_grace, 1000, TEN_PERCENT, @@ -396,13 +458,13 @@ mod tests { #[test] fn a_zero_threshold_prunes_after_each_delete_that_freed_bytes() { let now = at(10_000_000); - let due = |ledger| prune_due(&ledger, now, 1000, Percent(0), Duration::ZERO); + let due = |ledger, freed| prune_due(&ledger, freed, now, 1000, Percent(0), Duration::ZERO); assert_eq!( [ - due(ledger(0, None, false)), - due(ledger(1, None, false)), - due(ledger(1, Some(10_000_000), false)), + due(ledger(None, false), 0), + due(ledger(None, false), 1), + due(ledger(Some(10_000_000), false), 1), ], [false, true, true] ); @@ -412,7 +474,7 @@ mod tests { fn only_freed_bytes_after_the_grace_period_without_marked_packs_need_the_repository_size() { let last = 1_000_000; let needs = |freed, awaiting_removal, now| { - needs_repository_size(&ledger(freed, Some(last), awaiting_removal), at(now), GRACE) + needs_repository_size(&ledger(Some(last), awaiting_removal), freed, at(now), GRACE) }; assert_eq!( @@ -456,7 +518,7 @@ mod tests { #[test] async fn a_claim_is_taken_one_time_and_listed_with_its_time() { let files = new_files(); - let directory = claims_directory(&ledger(1, Some(42), false)); + let directory = claims_directory(&ledger(Some(42), false)); let now = at(10_000_000); let first = take_claim(&files, &directory, 0, now).await.unwrap(); @@ -480,31 +542,73 @@ mod tests { } #[test] - fn a_delete_adds_its_bytes_and_a_prune_starts_the_ledger_again() { - let deleted = ledger(5, Some(7), true) - .with_deleted(10) - .with_deleted(u64::MAX); - + fn a_prune_starts_the_ledger_again() { assert_eq!( ( - deleted, PruneLedger::after_prune(at(42), true), PruneLedger::after_prune(at(43), false) ), - ( - ledger(u64::MAX, Some(7), true), - ledger(0, Some(42), true), - ledger(0, Some(43), false) - ) + (ledger(Some(42), true), ledger(Some(43), false)) + ); + } + + #[test] + fn the_freed_bytes_of_a_record_are_the_number_before_the_first_dash() { + assert_eq!( + [ + parse_freed("123-0f4e"), + parse_freed("0-a-b"), + parse_freed("123"), + parse_freed("123-"), + parse_freed("x-0f4e"), + parse_freed("-0f4e"), + ], + [Some(123), Some(0), None, None, None, None] + ); + } + + #[test] + fn the_records_that_parse_are_summed_and_counted_and_the_others_stay() { + let blob = |name: &str| ListedBlob { + path: Path::new(FREED_PATH).join(name).into(), + size: 0, + }; + + let records = count_freed(&[ + blob("5-a"), + blob("not-a-count"), + blob(&format!("{}-b", u64::MAX)), + blob("7"), + ]); + + assert_eq!( + records, + FreedRecords { + bytes: u64::MAX, + counted: vec![ + Path::new(FREED_PATH).join("5-a").into(), + Path::new(FREED_PATH).join(format!("{}-b", u64::MAX)).into(), + ], + } ); } + #[test] + async fn a_record_of_freed_bytes_is_written_and_listed() { + let files = new_files(); + + record_freed(&files, 40).await.unwrap(); + record_freed(&files, 2).await.unwrap(); + let listed = list_freed(&files).await.unwrap(); + + assert_eq!((listed.bytes, listed.counted.len()), (42, 2)); + } + #[test] async fn the_ledger_is_written_and_read_back_with_the_time_in_milliseconds() { // The ledger keeps the time of the last prune as ISO 8601 text with milliseconds. let files = new_files(); let written = PruneLedger { - freed_bytes: 123, last_prune: Some(Timestamp::from(Timestamp::now_utc().to_millis())), awaiting_removal: true, }; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 289eb584c8..a9cf4ac388 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -25,9 +25,9 @@ use super::fault::{Operation, classify, is_file_missing, is_storage_failure, sto use super::files::SnapshotFiles; use super::priority::LowPriority; use super::prune::{ - ClaimChoice, Percent, PruneLedger, claims_directory, end_claims, list_claims, - needs_repository_size, next_claim, prune_due, read_ledger, release_claim, repository_bytes, - take_claim, write_ledger, + ClaimChoice, Percent, PruneLedger, claims_directory, end_claims, list_claims, list_freed, + needs_repository_size, next_claim, prune_due, read_ledger, record_freed, release_claim, + remove_freed, repository_bytes, take_claim, write_ledger, }; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -289,9 +289,9 @@ impl RusticSnapshotStore { } } - /// Adds the freed bytes to the ledger of the scope, and prunes the repository when a prune is - /// due. The ledger keeps the freed bytes before the prune starts, so a delete that runs again - /// after a failed prune prunes again. It lists the packs only when their size can make a prune due. + /// Writes a record of the freed bytes, and prunes the repository when a prune is due. The + /// record stays until a prune succeeds, so a delete that runs again after a failed prune prunes + /// again. It lists the packs only when their size can make a prune due. /// A due prune runs only after the delete takes a claim of its ledger, and only when the ledger /// did not change after the claim. A failed prune deletes the claim, and a prune that succeeds /// deletes each claim of its ledger. @@ -302,23 +302,26 @@ impl RusticSnapshotStore { freed: u64, ) -> Result<(), SnapshotStoreError> { let files = self.files(scope, token); - let ledger = read_ledger(&files) - .await - .map_err(storage_failure)? - .with_deleted(freed); if freed > 0 { - write_ledger(&files, &ledger) - .await - .map_err(storage_failure)?; + record_freed(&files, freed).await.map_err(storage_failure)?; } + let ledger = read_ledger(&files).await.map_err(storage_failure)?; + let records = list_freed(&files).await.map_err(storage_failure)?; let now = Timestamp::now_utc(); let grace = self.policy.prune.keep_delete; - let size = if needs_repository_size(&ledger, now, grace) { + let size = if needs_repository_size(&ledger, records.bytes, now, grace) { repository_bytes(&files).await.map_err(storage_failure)? } else { 0 }; - if !prune_due(&ledger, now, size, self.policy.prune_threshold, grace) { + if !prune_due( + &ledger, + records.bytes, + now, + size, + self.policy.prune_threshold, + grace, + ) { return Ok(()); } let claims = claims_directory(&ledger); @@ -367,6 +370,7 @@ impl RusticSnapshotStore { write_ledger(&files, &PruneLedger::after_prune(now, marked_packs)) .await .map_err(storage_failure)?; + remove_freed(&files, &records).await; end_claims(&files, &claims).await; Ok(()) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 7fd5332540..b819cc83d7 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -18,7 +18,7 @@ //! give the store a short or a long deadline and a prune policy that the test controls. use super::super::files::SnapshotFiles; -use super::super::prune::{Percent, PruneLedger, read_ledger, write_ledger}; +use super::super::prune::{Percent, PruneLedger, count_freed, read_ledger, write_ledger}; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; @@ -166,6 +166,20 @@ async fn ledger(storage: &Arc, scope: &SnapshotScop .unwrap() } +/// Gives the sum of the records of freed bytes of the scope. +async fn freed(storage: &Arc, scope: &SnapshotScope) -> u64 { + let listed = storage + .list_blobs_below( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-freed"), + ) + .await + .unwrap(); + count_freed(&listed).bytes +} + /// Waits until the condition holds, or until the limit ends. Gives whether the condition holds. async fn eventually(condition: impl Fn() -> bool) -> bool { tokio::time::timeout(LIMIT, async { @@ -783,12 +797,13 @@ async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_per store.delete(&scope, &name("p-deleted")).await.unwrap(); let after_first = ledger(&storage, &scope).await; + let after_first_freed = freed(&storage, &scope).await; store.delete(&scope, &name("p-none")).await.unwrap(); let packs_after = blobs(&*storage, &scope.0, "data/").await; assert_eq!( ( - after_first.freed_bytes, + after_first_freed, after_first.last_prune.is_some(), after_first.awaiting_removal, packs_after.len() < packs_before.len(), @@ -807,14 +822,24 @@ fn data_listings(calls: &[(&'static str, String)]) -> usize { .count() } -/// Writes a ledger with freed bytes, no marked packs, and a last prune at the time. +/// Writes a ledger with no marked packs and a last prune at the time, and a record of one freed +/// byte. async fn set_last_prune( storage: &Arc, scope: &SnapshotScope, last_prune: golem_common::model::Timestamp, ) { + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-freed/1-test"), + b"", + ) + .await + .unwrap(); let ledger = PruneLedger { - freed_bytes: 1, last_prune: Some(last_prune), awaiting_removal: false, }; @@ -899,12 +924,13 @@ async fn a_failed_listing_of_the_packs_gives_storage_and_records_no_prune() { let deleted = store.delete(&scope, &name("p-deleted")).await; let after = ledger(&storage, &scope).await; + let after_freed = freed(&storage, &scope).await; assert!( deleted.as_ref().is_err_and(|error| is_storage(error, true)), "{deleted:?}" ); - assert_eq!((after.freed_bytes > 0, after.last_prune), (true, None)); + assert_eq!((after_freed > 0, after.last_prune), (true, None)); } /// Counts the prunes among the recorded calls. A prune lists the packs, and no other step of a @@ -929,6 +955,115 @@ async fn save_each(store: &RusticSnapshotStore, scope: &SnapshotScope, names: &[ .await; } +#[test] +#[timeout("60s")] +async fn a_delete_that_paused_after_its_ledger_read_does_not_put_back_the_old_ledger() { + // The gate holds the first delete after its record write and its ledger read. The second + // delete prunes to its end. The first delete then goes on with the ledger that it read. + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let held = held.clone(); + move |op_label, _| { + if op_label == "list_freed" && !held.swap(true, Ordering::SeqCst) { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let paused = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let paused_held = eventually(|| held.load(Ordering::SeqCst)).await; + + let pruned = store.delete(&scope, &name("p-2")).await; + let after_prune = ledger(&storage, &scope).await; + storage.open_gate(); + let paused = tokio::time::timeout(LIMIT, paused).await; + let after_all = ledger(&storage, &scope).await; + + assert!(pruned.is_ok(), "{pruned:?}"); + assert!(matches!(paused, Ok(Ok(Ok(())))), "{paused:?}"); + assert_eq!( + ( + paused_held, + prunes(&storage.calls()), + after_prune.last_prune.is_some(), + after_all, + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, 1, true, after_prune, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_record_that_a_delete_adds_during_a_prune_stays_for_the_next_prune() { + // The gate holds the prune at its listing of the packs, and a record comes in meanwhile. + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let pruning = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let prune_held = eventually(|| held.load(Ordering::SeqCst)).await; + + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-freed/7-late"), + b"", + ) + .await + .unwrap(); + storage.open_gate(); + let pruned = tokio::time::timeout(LIMIT, pruning).await; + + assert!(matches!(pruned, Ok(Ok(Ok(())))), "{pruned:?}"); + assert_eq!( + ( + prune_held, + blobs(&*storage, &scope.0, "golem/prune-freed/").await, + freed(&storage, &scope).await, + ), + (true, vec!["golem/prune-freed/7-late".to_string()], 7) + ); +} + #[test] #[timeout("60s")] async fn two_deletes_that_read_the_same_ledger_make_one_prune() { @@ -1154,7 +1289,7 @@ async fn a_claim_older_than_the_grace_period_does_not_block_a_prune() { #[test] #[timeout("60s")] -async fn a_prune_that_succeeds_deletes_the_claims_of_its_ledger() { +async fn a_prune_that_succeeds_deletes_the_claims_and_the_counted_records_of_freed_bytes() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), @@ -1169,8 +1304,9 @@ async fn a_prune_that_succeeds_deletes_the_claims_of_its_ledger() { ( ledger(&storage, &scope).await.last_prune.is_some(), blobs(&*storage, &scope.0, "golem/prune-claims/").await, + blobs(&*storage, &scope.0, "golem/prune-freed/").await, ), - (true, Vec::::new()) + (true, Vec::::new(), Vec::::new()) ); } @@ -1195,10 +1331,11 @@ async fn a_delete_below_the_threshold_does_not_prune() { store.delete(&scope, &name("p-deleted")).await.unwrap(); let after = ledger(&storage, &scope).await; + let after_freed = freed(&storage, &scope).await; assert_eq!( ( - after.freed_bytes > 0, + after_freed > 0, after.last_prune, blobs(&*storage, &scope.0, "data/").await, ), @@ -1232,12 +1369,13 @@ async fn no_second_prune_runs_within_the_grace_period() { let after_first = ledger(&storage, &scope).await; store.delete(&scope, &name("p-b")).await.unwrap(); let after_second = ledger(&storage, &scope).await; + let after_second_freed = freed(&storage, &scope).await; assert_eq!( ( after_first.last_prune.is_some(), after_second.last_prune == after_first.last_prune, - after_second.freed_bytes > 0, + after_second_freed > 0, ), (true, true, true) ); @@ -1277,9 +1415,11 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { let failed = store.delete(&scope, &name("p-deleted")).await; let after_failure = ledger(&storage, &scope).await; + let after_failure_freed = freed(&storage, &scope).await; refuse.store(false, Ordering::SeqCst); let retried = store.delete(&scope, &name("p-deleted")).await; let after_retry = ledger(&storage, &scope).await; + let after_retry_freed = freed(&storage, &scope).await; assert!( failed.as_ref().is_err_and(|error| is_storage(error, true)), @@ -1287,10 +1427,10 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { ); assert_eq!( ( - after_failure.freed_bytes > 0, + after_failure_freed > 0, after_failure.last_prune, retried.is_ok(), - after_retry.freed_bytes, + after_retry_freed, after_retry.last_prune.is_some(), ), (true, None, true, 0, true) @@ -2112,10 +2252,11 @@ async fn a_due_prune_that_marks_a_pack_that_no_index_lists_records_the_marked_pa store.delete(&scope, &name("p-unknown")).await.unwrap(); let after = ledger(&storage, &scope).await; + let after_freed = freed(&storage, &scope).await; assert_eq!( ( - after.freed_bytes, + after_freed, after.last_prune.is_some_and(|last| last.to_millis() > 0), after.awaiting_removal, blobs(&*storage, &scope.0, "data/") @@ -2222,10 +2363,7 @@ async fn the_ledger_counts_the_packed_bytes_that_the_deleted_snapshot_added() { store.delete(&scope, &name("p-deleted")).await.unwrap(); - assert_eq!( - (added > 1, ledger(&storage, &scope).await.freed_bytes), - (true, added) - ); + assert_eq!((added > 1, freed(&storage, &scope).await), (true, added)); } #[test] @@ -2401,7 +2539,8 @@ async fn the_storage_calls_of_a_save_run_at_nice_19() { #[test] async fn the_storage_calls_of_a_prune_run_at_nice_19() { // The forget of a delete runs before the ledger read at the normal priority. The ledger calls, - // the listing of the packs and the claim calls run on the async runtime. + // the calls of the freed records, the listing of the packs and the claim calls run on the + // async runtime. let (storage, calls) = nice_recording_storage(); let store = store(storage, policy(LONG_DEADLINE, ALWAYS, Duration::ZERO)); let scope = new_scope(); @@ -2429,6 +2568,9 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { "read_claim", "write_claim", "delete_claims", + "write_freed", + "list_freed", + "delete_freed", ] .contains(&op_label.as_str()) }) From 29fb495636c80d60a522bdf64069fb3a7a75fa99 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 15:34:16 -0700 Subject: [PATCH 051/126] Keep the prune ledger as entries that are only added, and take the newest --- .../src/filesystem_snapshot/rustic/prune.rs | 265 +++++++++++++----- .../src/filesystem_snapshot/rustic/scope.rs | 4 +- .../src/filesystem_snapshot/rustic/store.rs | 8 +- .../filesystem_snapshot/rustic/store/tests.rs | 146 ++++++++-- 4 files changed, 334 insertions(+), 89 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 415fbcadf9..49c86f336a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -14,9 +14,10 @@ //! When a delete of the store prunes the repository of its scope. //! -//! The scope keeps a small ledger blob next to the files of the repository. The ledger holds the -//! time of the last prune, and whether that prune marked packs that a later prune removes. Only the -//! delete that holds the claim of a prune writes it. Each delete that freed bytes writes a record +//! The scope keeps a ledger next to the files of the repository: an entry for each prune, whose +//! name holds the time at which the prune ended and whether it marked packs that a later prune +//! removes. The newest entry is the ledger. Only the delete that holds the claim of a prune writes +//! an entry, and an entry is never written over. Each delete that freed bytes writes a record //! of its own, with the count in the name, and a prune that succeeds deletes the records it counted. //! //! A delete whose prune is due takes a claim before it prunes, so two deletes that read the same @@ -27,13 +28,16 @@ use super::files::SnapshotFiles; use futures::{StreamExt, TryStreamExt, stream}; use golem_common::model::Timestamp; use golem_service_base::storage::blob::{ListedBlob, PutIfAbsent}; -use serde::{Deserialize, Serialize}; use std::path::{Path, PathBuf}; use std::time::Duration; use tracing::warn; -/// The path of the ledger blob, relative to the root of the namespace of the scope. -pub(super) const LEDGER_PATH: &str = "golem/prune-ledger"; +/// The directory of the ledger entries, relative to the root of the namespace of the scope. +pub(super) const LEDGERS_PATH: &str = "golem/prune-ledgers"; + +/// How far the clock of another host can be ahead of the local clock. A time that is further +/// ahead counts as missing. +pub(super) const CLOCK_SKEW_MARGIN: Duration = Duration::from_secs(2 * 60); /// The directory of the prune claims, relative to the root of the namespace of the scope. const CLAIMS_PATH: &str = "golem/prune-claims"; @@ -42,7 +46,7 @@ const CLAIMS_PATH: &str = "golem/prune-claims"; const FREED_PATH: &str = "golem/prune-freed"; /// What the scope did since its last prune. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] pub(super) struct PruneLedger { /// The time of the last prune. pub(super) last_prune: Option, @@ -50,16 +54,6 @@ pub(super) struct PruneLedger { pub(super) awaiting_removal: bool, } -impl PruneLedger { - /// Gives the ledger after a prune at `now` that marked packs or not. - pub(super) fn after_prune(now: Timestamp, marked_packs: bool) -> Self { - Self { - last_prune: Some(now), - awaiting_removal: marked_packs, - } - } -} - /// The path of the packs of a repository, relative to the root of the namespace of the scope. const DATA_PATH: &str = "data"; @@ -128,30 +122,115 @@ pub(super) async fn repository_bytes(files: &SnapshotFiles) -> anyhow::Result anyhow::Result { - let content = files.get("read_ledger", Path::new(LEDGER_PATH)).await?; - Ok(content.map_or_else(PruneLedger::default, |content| { - serde_json::from_slice(&content).unwrap_or_else(|error| { - warn!( - error = %error, - "The prune ledger of a filesystem snapshot scope does not parse, so it starts again" - ); - PruneLedger::default() +/// Reads a ledger entry name, `-<0|1>-`, as the time of the prune and whether +/// it marked packs. +pub(super) fn parse_ledger_entry(name: &str) -> Option { + let mut parts = name.splitn(3, '-'); + let ended = parts.next()?.parse::().ok()?; + let awaiting_removal = match parts.next()? { + "0" => false, + "1" => true, + _ => return None, + }; + parts.next().filter(|unique| !unique.is_empty())?; + Some(PruneLedger { + last_prune: Some(Timestamp::from(ended)), + awaiting_removal, + }) +} + +/// Tells whether the time is more than [`CLOCK_SKEW_MARGIN`] after `now`. +fn beyond_margin(time: Timestamp, now: Timestamp) -> bool { + time.to_millis() + > now + .to_millis() + .saturating_add(u64::try_from(CLOCK_SKEW_MARGIN.as_millis()).unwrap_or(u64::MAX)) +} + +/// Gives the ledger from the listed entries: the entry with the greatest time. A name that does not +/// parse and a time more than the margin after `now` are left out. No entry gives the default. +pub(super) fn newest_ledger(listed: &[ListedBlob], now: Timestamp) -> PruneLedger { + listed + .iter() + .filter_map(|blob| parse_ledger_entry(blob.path.file_name()?.to_str()?)) + .filter(|entry| { + entry + .last_prune + .is_some_and(|ended| !beyond_margin(ended, now)) + }) + .max_by_key(|entry| (entry.last_prune, entry.awaiting_removal)) + .unwrap_or_default() +} + +/// Gives the paths of the listed entries whose time is before `ended`, in whole milliseconds as +/// an entry name holds it. +pub(super) fn older_entries(listed: &[ListedBlob], ended: Timestamp) -> Vec> { + listed + .iter() + .filter(|blob| { + blob.path + .file_name() + .and_then(|name| name.to_str()) + .and_then(parse_ledger_entry) + .and_then(|entry| entry.last_prune) + .is_some_and(|time| time.to_millis() < ended.to_millis()) }) - })) + .map(|blob| blob.path.clone()) + .collect() } -/// Writes the ledger of the scope over the ledger that was there. +/// Reads the ledger of the scope from one listing of its entries. No read of content is needed. +pub(super) async fn read_ledger(files: &SnapshotFiles) -> anyhow::Result { + let listed = files + .list_below("read_ledger", Path::new(LEDGERS_PATH)) + .await?; + Ok(newest_ledger(&listed, Timestamp::now_utc())) +} + +/// Writes a new ledger entry for a prune that ended at `ended`. pub(super) async fn write_ledger( files: &SnapshotFiles, - ledger: &PruneLedger, + ended: Timestamp, + awaiting_removal: bool, ) -> anyhow::Result<()> { - let content = serde_json::to_vec(ledger)?; + let name = format!( + "{}-{}-{}", + ended.to_millis(), + u8::from(awaiting_removal), + uuid::Uuid::new_v4() + ); files - .put("write_ledger", Path::new(LEDGER_PATH), &content) + .put_if_absent("write_ledger", &Path::new(LEDGERS_PATH).join(name), &[]) .await + .map(|_| ()) +} + +/// Deletes each ledger entry that is older than the entry of the prune that ended at `ended`. A +/// failure gives a warning, because an older entry is never the newest. +pub(super) async fn remove_older_ledgers(files: &SnapshotFiles, ended: Timestamp) { + let listed = match files + .list_below("list_ledgers", Path::new(LEDGERS_PATH)) + .await + { + Ok(listed) => listed, + Err(error) => { + warn!( + error = %format!("{error:#}"), + "Failed to list the prune ledger entries of a filesystem snapshot scope" + ); + return; + } + }; + stream::iter(older_entries(&listed, ended)) + .for_each(|path| async move { + if let Err(error) = files.delete("delete_ledger", &path).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete an old prune ledger entry of a filesystem snapshot scope" + ); + } + }) + .await; } /// The freed bytes of the records that a listing found, and the paths of the records it counted. @@ -343,9 +422,10 @@ fn parse_claim(content: &[u8]) -> Option { mod tests { use super::super::files::SnapshotFiles; use super::{ - ClaimChoice, FREED_PATH, FreedRecords, LEDGER_PATH, ListedClaim, Percent, PruneLedger, - claims_directory, count_freed, list_claims, list_freed, needs_repository_size, next_claim, - parse_freed, prune_due, read_ledger, record_freed, take_claim, write_ledger, + CLOCK_SKEW_MARGIN, ClaimChoice, FREED_PATH, FreedRecords, LEDGERS_PATH, ListedClaim, + Percent, PruneLedger, claims_directory, count_freed, list_claims, list_freed, + needs_repository_size, newest_ledger, next_claim, older_entries, parse_freed, + parse_ledger_entry, prune_due, read_ledger, record_freed, take_claim, write_ledger, }; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; @@ -542,13 +622,79 @@ mod tests { } #[test] - fn a_prune_starts_the_ledger_again() { + fn a_ledger_entry_name_gives_the_end_time_and_the_marked_packs() { assert_eq!( - ( - PruneLedger::after_prune(at(42), true), - PruneLedger::after_prune(at(43), false) + [ + parse_ledger_entry("42-1-a"), + parse_ledger_entry("43-0-a-b"), + parse_ledger_entry("42-2-a"), + parse_ledger_entry("42-1-"), + parse_ledger_entry("42-1"), + parse_ledger_entry("x-1-a"), + ], + [ + Some(ledger(Some(42), true)), + Some(ledger(Some(43), false)), + None, + None, + None, + None + ] + ); + } + + #[test] + fn the_newest_entry_within_the_margin_is_the_ledger() { + let now = 10_000_000; + let margin = u64::try_from(CLOCK_SKEW_MARGIN.as_millis()).unwrap(); + let entry = |name: &str| ListedBlob { + path: Path::new(LEDGERS_PATH).join(name).into(), + size: 0, + }; + let newest = |names: &[&str]| { + newest_ledger( + &names.iter().map(|name| entry(name)).collect::>(), + at(now), + ) + }; + + assert_eq!( + [ + newest(&[]), + newest(&["5-0-a", "9-1-b", "7-0-c"]), + newest(&["5-0-a", "not-an-entry"]), + newest(&["5-0-a", &format!("{}-1-b", now + margin)]), + newest(&["5-0-a", &format!("{}-1-b", now + margin + 1)]), + ], + [ + PruneLedger::default(), + ledger(Some(9), true), + ledger(Some(5), false), + ledger(Some(now + margin), true), + ledger(Some(5), false), + ] + ); + } + + #[test] + fn only_entries_before_the_end_of_a_prune_are_older() { + let entry = |name: &str| ListedBlob { + path: Path::new(LEDGERS_PATH).join(name).into(), + size: 0, + }; + + assert_eq!( + older_entries( + &[ + entry("5-0-a"), + entry("9-1-own"), + entry("9-0-same-time"), + entry("12-0-newer"), + entry("bad"), + ], + at(9) ), - (ledger(Some(42), true), ledger(Some(43), false)) + vec![Path::new(LEDGERS_PATH).join("5-0-a").into_boxed_path()] ); } @@ -605,36 +751,23 @@ mod tests { } #[test] - async fn the_ledger_is_written_and_read_back_with_the_time_in_milliseconds() { - // The ledger keeps the time of the last prune as ISO 8601 text with milliseconds. + async fn a_written_entry_is_the_ledger_that_a_read_gives() { let files = new_files(); - let written = PruneLedger { - last_prune: Some(Timestamp::from(Timestamp::now_utc().to_millis())), - awaiting_removal: true, - }; + let ended = Timestamp::from(Timestamp::now_utc().to_millis()); let before = read_ledger(&files).await.unwrap(); - write_ledger(&files, &written).await.unwrap(); + write_ledger(&files, ended, true).await.unwrap(); let after = read_ledger(&files).await.unwrap(); - assert_eq!((before, after), (PruneLedger::default(), written)); - } - - #[test] - async fn a_ledger_that_does_not_parse_reads_as_an_empty_ledger() { - let files = new_files(); - files - .storage - .put_raw( - "test", - "test", - files.namespace.clone(), - Path::new(LEDGER_PATH), - b"not json", + assert_eq!( + (before, after), + ( + PruneLedger::default(), + PruneLedger { + last_prune: Some(ended), + awaiting_removal: true + } ) - .await - .unwrap(); - - assert_eq!(read_ledger(&files).await.unwrap(), PruneLedger::default()); + ); } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs index ba992a8c6b..e60f507c61 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs @@ -19,7 +19,7 @@ use super::backend::CONFIG_PATH; use super::files::SnapshotFiles; -use super::prune::LEDGER_PATH; +use super::prune::LEDGERS_PATH; use futures::{StreamExt, TryStreamExt, stream}; use golem_service_base::storage::blob::PutIfAbsent; use std::path::Path; @@ -67,7 +67,7 @@ pub(super) async fn delete_scope(files: &SnapshotFiles) -> anyhow::Result<()> { LISTING_ORDER .iter() .map(Path::new) - .chain(Path::new(LEDGER_PATH).parent()) + .chain(Path::new(LEDGERS_PATH).parent()) .map(Ok), ) .try_for_each(|directory| async move { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index a9cf4ac388..5b0e63cef1 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -25,9 +25,9 @@ use super::fault::{Operation, classify, is_file_missing, is_storage_failure, sto use super::files::SnapshotFiles; use super::priority::LowPriority; use super::prune::{ - ClaimChoice, Percent, PruneLedger, claims_directory, end_claims, list_claims, list_freed, + ClaimChoice, Percent, claims_directory, end_claims, list_claims, list_freed, needs_repository_size, next_claim, prune_due, read_ledger, record_freed, release_claim, - remove_freed, repository_bytes, take_claim, write_ledger, + remove_freed, remove_older_ledgers, repository_bytes, take_claim, write_ledger, }; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -367,9 +367,11 @@ impl RusticSnapshotStore { } }; let marked_packs = report.as_ref().is_some_and(leaves_marked_packs); - write_ledger(&files, &PruneLedger::after_prune(now, marked_packs)) + let ended = Timestamp::now_utc(); + write_ledger(&files, ended, marked_packs) .await .map_err(storage_failure)?; + remove_older_ledgers(&files, ended).await; remove_freed(&files, &records).await; end_claims(&files, &claims).await; Ok(()) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index b819cc83d7..14f4e58ed8 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -18,7 +18,7 @@ //! give the store a short or a long deadline and a prune policy that the test controls. use super::super::files::SnapshotFiles; -use super::super::prune::{Percent, PruneLedger, count_freed, read_ledger, write_ledger}; +use super::super::prune::{LEDGERS_PATH, Percent, PruneLedger, count_freed, read_ledger}; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; @@ -822,8 +822,8 @@ fn data_listings(calls: &[(&'static str, String)]) -> usize { .count() } -/// Writes a ledger with no marked packs and a last prune at the time, and a record of one freed -/// byte. +/// Makes the ledger one entry with no marked packs and a last prune at the time, and writes a +/// record of one freed byte. async fn set_last_prune( storage: &Arc, scope: &SnapshotScope, @@ -839,21 +839,34 @@ async fn set_last_prune( ) .await .unwrap(); - let ledger = PruneLedger { - last_prune: Some(last_prune), - awaiting_removal: false, - }; - write_ledger( - &SnapshotFiles { - storage: storage.clone(), - namespace: scope.0.clone(), - deadline: Duration::from_secs(2), - cancel: tokio_util::sync::CancellationToken::new(), - }, - &ledger, + storage + .delete_dir("test", "test", scope.0.clone(), Path::new(LEDGERS_PATH)) + .await + .unwrap(); + put_ledger_entry( + storage, + scope, + &format!("{}-0-test", last_prune.to_millis()), ) - .await - .unwrap(); + .await; +} + +/// Writes a ledger entry with the name. +async fn put_ledger_entry( + storage: &Arc, + scope: &SnapshotScope, + name: &str, +) { + storage + .put_raw( + "test", + "test", + scope.0.clone(), + &Path::new(LEDGERS_PATH).join(name), + b"", + ) + .await + .unwrap(); } #[test] @@ -1064,6 +1077,99 @@ async fn a_record_that_a_delete_adds_during_a_prune_stays_for_the_next_prune() { ); } +#[test] +async fn a_late_older_ledger_entry_does_not_win() { + // The newer entry is inside the grace period, so a delete does not prune. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let now = golem_common::model::Timestamp::now_utc().to_millis(); + let newer = now.saturating_sub(60_000); + put_ledger_entry(&storage, &scope, &format!("{newer}-0-newer")).await; + put_ledger_entry( + &storage, + &scope, + &format!("{}-1-older", now.saturating_sub(2 * 3_600_000)), + ) + .await; + + let read = ledger(&storage, &scope).await; + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert_eq!( + ( + read.last_prune.map(|last| last.to_millis()), + read.awaiting_removal, + prunes(&storage.calls()), + ), + (Some(newer), false, 0) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_deletes_the_older_ledger_entries_and_keeps_a_newer_one() { + // The gate holds the prune at its listing of the packs. An older and a newer entry come in + // meanwhile. + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let pruning = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let prune_held = eventually(|| held.load(Ordering::SeqCst)).await; + + let newer = format!( + "{}-0-newer", + golem_common::model::Timestamp::now_utc().to_millis() + 60_000 + ); + put_ledger_entry(&storage, &scope, "1000-0-older").await; + put_ledger_entry(&storage, &scope, &newer).await; + storage.open_gate(); + let pruned = tokio::time::timeout(LIMIT, pruning).await; + let entries = blobs(&*storage, &scope.0, "golem/prune-ledgers/").await; + + assert!(matches!(pruned, Ok(Ok(Ok(())))), "{pruned:?}"); + assert_eq!( + ( + prune_held, + entries.len(), + entries.contains(&format!("golem/prune-ledgers/{newer}")), + entries.contains(&"golem/prune-ledgers/1000-0-older".to_string()), + ), + (true, 2, true, false) + ); +} + #[test] #[timeout("60s")] async fn two_deletes_that_read_the_same_ledger_make_one_prune() { @@ -1461,7 +1567,9 @@ async fn a_deleted_scope_holds_no_blob() { assert_eq!( ( - before.contains(&"golem/prune-ledger".to_string()), + before + .iter() + .any(|path| path.starts_with("golem/prune-ledgers/")), blobs(&*storage, &scope.0, "").await ), (true, Vec::::new()) @@ -2571,6 +2679,8 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { "write_freed", "list_freed", "delete_freed", + "list_ledgers", + "delete_ledger", ] .contains(&op_label.as_str()) }) From 9358c84baafccb7d5f7578ed1d0011f176fc9bc5 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 15:42:46 -0700 Subject: [PATCH 052/126] Count each blob call of the store in its tracker, so shut down waits for it --- .../src/filesystem_snapshot/rustic/files.rs | 13 ++++-- .../src/filesystem_snapshot/rustic/prune.rs | 1 + .../rustic/publish/tests.rs | 1 + .../filesystem_snapshot/rustic/scope/tests.rs | 1 + .../src/filesystem_snapshot/rustic/store.rs | 11 +++-- .../filesystem_snapshot/rustic/store/tests.rs | 44 +++++++++++++++++++ 6 files changed, 63 insertions(+), 8 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs index e60817137c..0ad802b598 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs @@ -24,28 +24,33 @@ use std::path::Path; use std::sync::Arc; use std::time::Duration; use tokio_util::sync::CancellationToken; +use tokio_util::task::TaskTracker; /// The target label of each blob storage call of the rustic store. pub(super) const TARGET_LABEL: &str = "filesystem_snapshot"; -/// The blobs of one scope: the storage, the namespace of the scope, the deadline of each call, and -/// the token of the operation. +/// The blobs of one scope: the storage, the namespace of the scope, the deadline of each call, the +/// token of the operation, and the tracker of the store, which counts each call. #[derive(Clone, Debug)] pub(super) struct SnapshotFiles { pub(super) storage: Arc, pub(super) namespace: BlobStorageNamespace, pub(super) deadline: Duration, pub(super) cancel: CancellationToken, + pub(super) tracker: TaskTracker, } impl SnapshotFiles { /// Waits for one call within the deadline. A call of a cancelled operation does not start, and - /// a cancel ends a running call. Both give an error. + /// a cancel ends a running call. Both give an error. The tracker counts the call before the + /// check of the cancel, so a shut down either stops the call or waits for it. async fn answer( &self, future: impl Future>, ) -> anyhow::Result { - answer_or_cancel(self.deadline, &self.cancel, future).await + self.tracker + .track_future(answer_or_cancel(self.deadline, &self.cancel, future)) + .await } /// Gives the content of the blob at the path, or `None` when the path has no blob. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 49c86f336a..f9b183ab7a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -463,6 +463,7 @@ mod tests { }, deadline: DEADLINE, cancel: tokio_util::sync::CancellationToken::new(), + tracker: tokio_util::task::TaskTracker::new(), } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs index b6aeac819b..22d1360e0e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -67,6 +67,7 @@ fn files( }, deadline, cancel: tokio_util::sync::CancellationToken::new(), + tracker: tokio_util::task::TaskTracker::new(), }, storage, inner, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs index 692a5fc642..c419279073 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs @@ -46,6 +46,7 @@ fn files( namespace: namespace.clone(), deadline: DEADLINE, cancel: tokio_util::sync::CancellationToken::new(), + tracker: tokio_util::task::TaskTracker::new(), } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 5b0e63cef1..3d1db6df78 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -155,7 +155,8 @@ pub(crate) struct RusticSnapshotStore { policy: StorePolicy, /// The parent of the token of each operation. root: CancellationToken, - /// Counts the blocking tasks, the backends, the publishes and the deletes of dropped publishes. + /// Counts the blocking tasks, the backends, the blob calls of the store, the publishes and the + /// deletes of dropped publishes. tracker: TaskTracker, /// Runs saves and prunes at a low priority. low_priority: LowPriority, @@ -218,9 +219,10 @@ impl RusticSnapshotStore { /// Cancels each operation, so each running storage call ends and no new call starts, and later /// operations give `Storage`. A publish that starts before the cancel runs to its end. A save /// that reaches its publish after the cancel publishes nothing and gives `Storage`. The call - /// waits until no blocking task, backend, publish or delete of a dropped publish remains. The - /// runtime must not drop before it returns, because a storage call after its time driver stops - /// aborts the process. + /// waits until no blocking task, backend, blob call of the store, publish or delete of a + /// dropped publish remains. A blob call that is not polled holds the wait until it is polled + /// again, and then it ends at once. The runtime must not drop before it returns, because a + /// storage call after its time driver stops aborts the process. pub(crate) async fn shut_down(&self) { self.root.cancel(); self.tracker.close(); @@ -286,6 +288,7 @@ impl RusticSnapshotStore { namespace: scope.0.clone(), deadline: self.policy.deadline, cancel: token.clone(), + tracker: self.tracker.clone(), } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 14f4e58ed8..be68172e22 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -161,6 +161,7 @@ async fn ledger(storage: &Arc, scope: &SnapshotScop namespace: scope.0.clone(), deadline: Duration::from_secs(2), cancel: tokio_util::sync::CancellationToken::new(), + tracker: tokio_util::task::TaskTracker::new(), }) .await .unwrap() @@ -1670,6 +1671,7 @@ async fn a_blob_call_of_a_cancelled_operation_does_not_start() { namespace: new_scope().0, deadline: LONG_DEADLINE, cancel, + tracker: tokio_util::task::TaskTracker::new(), }; let read = files @@ -1680,6 +1682,48 @@ async fn a_blob_call_of_a_cancelled_operation_does_not_start() { assert_eq!(storage.calls(), Vec::new()); } +#[test] +#[timeout("60s")] +async fn shut_down_waits_for_a_blob_call_of_the_store_that_is_not_polled() { + // The test polls the scope delete one time, so its first blob call waits at the gate, and + // then the test does not poll it again. Only the tracker makes the shut down wait for it. + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "delete_scope" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let deleting = store.delete_scope(&scope); + tokio::pin!(deleting); + let pending = futures::poll!(&mut deleting).is_pending(); + + let shutting = store.shut_down(); + tokio::pin!(shutting); + let waited = tokio::time::timeout(Duration::from_millis(200), &mut shutting) + .await + .is_err(); + let deleted = tokio::time::timeout(LIMIT, &mut deleting).await; + // A finished future must not be polled again, so the second wait runs only after a first wait + // that timed out. + let stopped = !waited || tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + storage.open_gate(); + + assert!( + matches!(&deleted, Ok(Err(error)) if is_storage(error, true)), + "{deleted:?}" + ); + assert_eq!( + (pending, waited, stopped, store.work_in_flight()), + (true, true, true, 0) + ); +} + #[test] #[timeout("60s")] async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_call() { From 268be3416039ec65613a7c5215824d3d245cb05a Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 15:48:57 -0700 Subject: [PATCH 053/126] Add a margin for clock skew to the grace period and to the age of a claim --- .../src/filesystem_snapshot/rustic/prune.rs | 63 ++++++++++++++----- .../filesystem_snapshot/rustic/store/tests.rs | 29 ++++++++- 2 files changed, 75 insertions(+), 17 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index f9b183ab7a..6d67acdb41 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -68,11 +68,13 @@ impl Percent { } } -/// Tells whether the grace period passed at `now` since the last prune. +/// Tells whether the grace period and the margin for clock skew passed at `now` since the last +/// prune. A time more than the margin after `now` counts as missing. fn grace_passed(ledger: &PruneLedger, now: Timestamp, grace: Duration) -> bool { ledger .last_prune - .is_none_or(|last| passed_since(last, now, grace)) + .filter(|last| !beyond_margin(*last, now)) + .is_none_or(|last| passed_since(last, now, grace.saturating_add(CLOCK_SKEW_MARGIN))) } /// Tells whether the grace period passed at `now` since the time. @@ -315,14 +317,17 @@ pub(super) enum ClaimChoice { } /// Chooses the claim of a delete from the claims of its ledger. The newest claim holds the ledger -/// until the grace period passed since its time. A claim whose content does not parse is old. +/// until the grace period and the margin for clock skew passed since its time. A claim whose +/// content does not parse, or whose time is more than the margin after `now`, is old. pub(super) fn next_claim(claims: &[ListedClaim], now: Timestamp, grace: Duration) -> ClaimChoice { + let held_until = grace.saturating_add(CLOCK_SKEW_MARGIN); match claims.iter().max_by_key(|claim| claim.number) { None => ClaimChoice::Claim(0), Some(newest) if newest .claimed_at - .is_some_and(|at| !passed_since(at, now, grace)) => + .filter(|at| !beyond_margin(*at, now)) + .is_some_and(|at| !passed_since(at, now, held_until)) => { ClaimChoice::Held } @@ -441,7 +446,8 @@ mod tests { const TEN_PERCENT: Percent = Percent(10); const GRACE: Duration = Duration::from_secs(15 * 60); - const GRACE_MILLIS: u64 = 15 * 60 * 1000; + /// The grace period and the margin for clock skew, in milliseconds. + const HELD_MILLIS: u64 = 15 * 60 * 1000 + 2 * 60 * 1000; const DEADLINE: Duration = Duration::from_secs(2); fn at(millis: u64) -> Timestamp { @@ -510,9 +516,9 @@ mod tests { assert_eq!( [ full(last), - full(last + GRACE_MILLIS - 1), - full(last + GRACE_MILLIS), - full(last + GRACE_MILLIS + 1), + full(last + HELD_MILLIS - 1), + full(last + HELD_MILLIS), + full(last + HELD_MILLIS + 1), ], [false, false, true, true] ); @@ -521,7 +527,7 @@ mod tests { #[test] fn marked_packs_make_a_prune_due_after_the_grace_period_without_freed_bytes() { let last = 1_000_000; - let after_grace = at(last + GRACE_MILLIS); + let after_grace = at(last + HELD_MILLIS); let due = |awaiting_removal| { prune_due( &ledger(Some(last), awaiting_removal), @@ -545,7 +551,7 @@ mod tests { [ due(ledger(None, false), 0), due(ledger(None, false), 1), - due(ledger(Some(10_000_000), false), 1), + due(ledger(Some(10_000_000 - 120_000), false), 1), ], [false, true, true] ); @@ -560,15 +566,40 @@ mod tests { assert_eq!( [ - needs(1, false, last + GRACE_MILLIS), - needs(1, false, last + GRACE_MILLIS - 1), - needs(0, false, last + GRACE_MILLIS), - needs(1, true, last + GRACE_MILLIS), + needs(1, false, last + HELD_MILLIS), + needs(1, false, last + HELD_MILLIS - 1), + needs(0, false, last + HELD_MILLIS), + needs(1, true, last + HELD_MILLIS), ], [true, false, false, false] ); } + #[test] + fn a_time_more_than_the_margin_ahead_counts_as_missing() { + let now = 10_000_000; + let margin = u64::try_from(CLOCK_SKEW_MARGIN.as_millis()).unwrap(); + let due = |last| prune_due(&ledger(Some(last), false), 1, at(now), 0, Percent(0), GRACE); + let claim = |claimed_at| { + next_claim( + &[ListedClaim { + number: 0, + claimed_at: Some(at(claimed_at)), + }], + at(now), + GRACE, + ) + }; + + assert_eq!( + ( + [due(now + margin), due(now + margin + 1)], + [claim(now + margin), claim(now + margin + 1)] + ), + ([false, true], [ClaimChoice::Held, ClaimChoice::Claim(1)]) + ); + } + #[test] fn the_newest_claim_holds_a_ledger_until_the_grace_period_passed_since_its_time() { let now = 10_000_000; @@ -581,8 +612,8 @@ mod tests { assert_eq!( [ choose(&[]), - choose(&[claim(0, Some(now - GRACE_MILLIS)), claim(1, Some(now - 1))]), - choose(&[claim(1, Some(now)), claim(2, Some(now - GRACE_MILLIS))]), + choose(&[claim(0, Some(now - HELD_MILLIS)), claim(1, Some(now - 1))]), + choose(&[claim(1, Some(now)), claim(2, Some(now - HELD_MILLIS))]), choose(&[claim(4, None)]), choose(&[claim(0, Some(now - 1)), claim(3, None)]), ], diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index be68172e22..a129752839 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -18,7 +18,9 @@ //! give the store a short or a long deadline and a prune policy that the test controls. use super::super::files::SnapshotFiles; -use super::super::prune::{LEDGERS_PATH, Percent, PruneLedger, count_freed, read_ledger}; +use super::super::prune::{ + CLOCK_SKEW_MARGIN, LEDGERS_PATH, Percent, PruneLedger, count_freed, read_ledger, +}; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; @@ -799,6 +801,8 @@ async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_per store.delete(&scope, &name("p-deleted")).await.unwrap(); let after_first = ledger(&storage, &scope).await; let after_first_freed = freed(&storage, &scope).await; + // The margin for clock skew keeps the next prune back, so the ledger moves back by it. + age_ledger(&storage, &scope, &after_first).await; store.delete(&scope, &name("p-none")).await.unwrap(); let packs_after = blobs(&*storage, &scope.0, "data/").await; @@ -852,6 +856,29 @@ async fn set_last_prune( .await; } +/// Makes the ledger one entry with the marked packs of `ledger` and a time before it by the margin +/// for clock skew. +async fn age_ledger( + storage: &Arc, + scope: &SnapshotScope, + ledger: &PruneLedger, +) { + let margin = u64::try_from(CLOCK_SKEW_MARGIN.as_millis()).unwrap(); + let aged = ledger + .last_prune + .map_or(0, |last| last.to_millis().saturating_sub(margin + 1)); + storage + .delete_dir("test", "test", scope.0.clone(), Path::new(LEDGERS_PATH)) + .await + .unwrap(); + put_ledger_entry( + storage, + scope, + &format!("{aged}-{}-aged", u8::from(ledger.awaiting_removal)), + ) + .await; +} + /// Writes a ledger entry with the name. async fn put_ledger_entry( storage: &Arc, From 0501348f69c2158cd8dca25b75b77472cc019517 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 15:56:44 -0700 Subject: [PATCH 054/126] Check the margin for clock skew at the end of the grace period --- .../src/filesystem_snapshot/rustic/prune.rs | 29 ++++++++++++++++--- 1 file changed, 25 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 6d67acdb41..12afc1f4d4 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -576,7 +576,8 @@ mod tests { } #[test] - fn a_time_more_than_the_margin_ahead_counts_as_missing() { + fn the_margin_extends_the_grace_period_and_a_time_more_than_the_margin_ahead_counts_as_missing() + { let now = 10_000_000; let margin = u64::try_from(CLOCK_SKEW_MARGIN.as_millis()).unwrap(); let due = |last| prune_due(&ledger(Some(last), false), 1, at(now), 0, Percent(0), GRACE); @@ -591,12 +592,32 @@ mod tests { ) }; + let grace = u64::try_from(GRACE.as_millis()).unwrap(); + assert_eq!( ( - [due(now + margin), due(now + margin + 1)], - [claim(now + margin), claim(now + margin + 1)] + [ + due(now - grace), + due(now - grace - margin), + due(now + margin), + due(now + margin + 1) + ], + [ + claim(now - grace), + claim(now - grace - margin), + claim(now + margin), + claim(now + margin + 1) + ] ), - ([false, true], [ClaimChoice::Held, ClaimChoice::Claim(1)]) + ( + [false, true, false, true], + [ + ClaimChoice::Held, + ClaimChoice::Claim(1), + ClaimChoice::Held, + ClaimChoice::Claim(1) + ] + ) ); } From 9e8e0761efe1f5e74a429aaea5eae6c754311f24 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 16:04:11 -0700 Subject: [PATCH 055/126] Write a live claim again while its prune runs --- .../src/filesystem_snapshot/rustic/prune.rs | 36 ++++++++ .../src/filesystem_snapshot/rustic/store.rs | 23 ++++-- .../filesystem_snapshot/rustic/store/tests.rs | 82 +++++++++++++++++++ 3 files changed, 133 insertions(+), 8 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 12afc1f4d4..d7cb0cc0e6 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -388,6 +388,42 @@ pub(super) async fn take_claim( Ok(written == PutIfAbsent::Written) } +/// Gives the time between two writes of a live claim: a fourth of the grace period, or a fourth of +/// the margin for clock skew when the grace period is zero. +pub(super) fn refresh_period(grace: Duration) -> Duration { + if grace.is_zero() { + CLOCK_SKEW_MARGIN / 4 + } else { + grace / 4 + } +} + +/// Writes the claim with the number again, with the current time, at each period, until the +/// caller drops the future. A failed write gives a warning. +pub(super) async fn keep_claim_fresh( + files: &SnapshotFiles, + directory: &Path, + number: u64, + period: Duration, +) { + let path = directory.join(number.to_string()); + stream::repeat(()) + .then(|()| tokio::time::sleep(period)) + .for_each(|()| { + let path = &path; + async move { + let content = Timestamp::now_utc().to_millis().to_string(); + if let Err(error) = files.put("refresh_claim", path, content.as_bytes()).await { + warn!( + error = %format!("{error:#}"), + "Failed to write the prune claim of a filesystem snapshot scope again" + ); + } + } + }) + .await; +} + /// Deletes the claim with the number. A failure gives a warning, because a claim only delays a /// prune until its grace period passed. pub(super) async fn release_claim(files: &SnapshotFiles, directory: &Path, number: u64) { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 3d1db6df78..f6de73711e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -25,9 +25,9 @@ use super::fault::{Operation, classify, is_file_missing, is_storage_failure, sto use super::files::SnapshotFiles; use super::priority::LowPriority; use super::prune::{ - ClaimChoice, Percent, claims_directory, end_claims, list_claims, list_freed, - needs_repository_size, next_claim, prune_due, read_ledger, record_freed, release_claim, - remove_freed, remove_older_ledgers, repository_bytes, take_claim, write_ledger, + ClaimChoice, Percent, claims_directory, end_claims, keep_claim_fresh, list_claims, list_freed, + needs_repository_size, next_claim, prune_due, read_ledger, record_freed, refresh_period, + release_claim, remove_freed, remove_older_ledgers, repository_bytes, take_claim, write_ledger, }; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -43,6 +43,7 @@ use crate::filesystem_snapshot::{ use crate::services::golem_config::FilesystemSnapshotStoreConfig; use anyhow::Context; use async_trait::async_trait; +use futures::future::{self, Either}; use golem_common::model::Timestamp; use golem_service_base::storage::blob::BlobStorage; use rustic_core::jiff::tz::TimeZone; @@ -55,6 +56,7 @@ use rustic_core::{ use serde::{Deserialize, Serialize}; use std::num::NonZeroUsize; use std::path::Path; +use std::pin::pin; use std::sync::Arc; use std::time::Duration; use tokio::runtime::Handle; @@ -357,11 +359,16 @@ impl RusticSnapshotStore { let key = self.key.clone(); let settings = self.policy.prune; let low_priority = self.low_priority; - let pruned = self - .blocking(Operation::Prune, move || { - low_priority.run("fs-snap-prune", move || prune(backend, &key, &settings)) - }) - .await; + let pruning = self.blocking(Operation::Prune, move || { + low_priority.run("fs-snap-prune", move || prune(backend, &key, &settings)) + }); + // The claim is written again while the prune runs, so a prune slower than the grace + // period keeps its claim. The writes stop when the prune ends. + let refreshing = keep_claim_fresh(&files, &claims, number, refresh_period(grace)); + let pruned = match future::select(pin!(pruning), pin!(refreshing)).await { + Either::Left((pruned, _)) => pruned, + Either::Right(((), pruning)) => pruning.await, + }; let report = match pruned { Ok(report) => report, Err(error) => { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index a129752839..653567f1be 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1198,6 +1198,87 @@ async fn a_prune_deletes_the_older_ledger_entries_and_keeps_a_newer_one() { ); } +/// Gives the time in the one claim of the scope, when the scope has one claim that parses. +async fn claim_time(storage: &ScriptedBlobStorage, scope: &SnapshotScope) -> Option { + let claims = blobs(storage, &scope.0, "golem/prune-claims/").await; + let [claim] = claims.as_slice() else { + return None; + }; + let content = storage + .get_raw("test", "test", scope.0.clone(), Path::new(claim)) + .await + .ok()??; + std::str::from_utf8(&content).ok()?.parse().ok() +} + +#[test] +#[timeout("60s")] +async fn a_prune_slower_than_the_grace_period_keeps_its_claim_fresh() { + // The gate holds the prune at its listing of the packs for longer than the grace period. + let grace = Duration::from_millis(400); + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let pruning = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let prune_held = eventually(|| held.load(Ordering::SeqCst)).await; + let first = claim_time(&storage, &scope).await.unwrap_or(u64::MAX); + let wanted = first.saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)); + + let refreshed = tokio::time::timeout( + LIMIT, + futures::stream::repeat(()) + .then(|()| async { + tokio::time::sleep(Duration::from_millis(20)).await; + claim_time(&storage, &scope).await + }) + .filter(|time| std::future::ready(time.is_some_and(|time| time >= wanted))) + .boxed() + .next(), + ) + .await + .is_ok(); + let second = store.delete(&scope, &name("p-2")).await; + let prunes_while_held = prunes(&storage.calls()); + storage.open_gate(); + let pruned = tokio::time::timeout(LIMIT, pruning).await; + + assert!(second.is_ok(), "{second:?}"); + assert!(matches!(pruned, Ok(Ok(Ok(())))), "{pruned:?}"); + assert_eq!( + ( + prune_held, + refreshed, + prunes_while_held, + prunes(&storage.calls()), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, true, 1, 1, Vec::::new()) + ); +} + #[test] #[timeout("60s")] async fn two_deletes_that_read_the_same_ledger_make_one_prune() { @@ -2752,6 +2833,7 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { "delete_freed", "list_ledgers", "delete_ledger", + "refresh_claim", ] .contains(&op_label.as_str()) }) From b4ec7cae5b69b26526d53bfeeffad06c2ba3aa93 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 17:11:17 -0700 Subject: [PATCH 056/126] Delete the claims of each old ledger after a prune --- .../src/filesystem_snapshot/rustic/prune.rs | 79 ++++++++++++++++--- .../src/filesystem_snapshot/rustic/store.rs | 11 ++- .../filesystem_snapshot/rustic/store/tests.rs | 35 ++++++++ 3 files changed, 109 insertions(+), 16 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index d7cb0cc0e6..c125637de3 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -438,15 +438,44 @@ pub(super) async fn release_claim(files: &SnapshotFiles, directory: &Path, numbe } } -/// Deletes the claims of a ledger after its prune. A failure gives a warning, because a claim only -/// delays a prune until its grace period passed. -pub(super) async fn end_claims(files: &SnapshotFiles, directory: &Path) { - if let Err(error) = files.delete_dir("delete_claims", directory).await { - warn!( - error = %format!("{error:#}"), - "Failed to delete the prune claims of a filesystem snapshot scope" - ); - } +/// Gives each claim directory of the listed claims, other than the directory to keep. +pub(super) fn claim_directories_except(listed: &[ListedBlob], keep: &Path) -> Vec { + listed + .iter() + .filter_map(|blob| blob.path.parent().map(Path::to_path_buf)) + .filter(|directory| directory != keep) + .collect::>() + .into_iter() + .collect() +} + +/// Deletes each claim directory other than the directory of the new ledger. It never deletes the +/// directory of all claims, so a live claim of the new ledger stays. A failure gives a warning, +/// because a claim only delays a prune until its grace period passed. +pub(super) async fn remove_old_claims(files: &SnapshotFiles, keep: &Path) { + let listed = match files + .list_below("list_claim_directories", Path::new(CLAIMS_PATH)) + .await + { + Ok(listed) => listed, + Err(error) => { + warn!( + error = %format!("{error:#}"), + "Failed to list the prune claims of a filesystem snapshot scope" + ); + return; + } + }; + stream::iter(claim_directories_except(&listed, keep)) + .for_each(|directory| async move { + if let Err(error) = files.delete_dir("delete_claims", &directory).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete the prune claims of a filesystem snapshot scope" + ); + } + }) + .await; } /// Reads the time of a claim, in milliseconds. @@ -463,10 +492,11 @@ fn parse_claim(content: &[u8]) -> Option { mod tests { use super::super::files::SnapshotFiles; use super::{ - CLOCK_SKEW_MARGIN, ClaimChoice, FREED_PATH, FreedRecords, LEDGERS_PATH, ListedClaim, - Percent, PruneLedger, claims_directory, count_freed, list_claims, list_freed, - needs_repository_size, newest_ledger, next_claim, older_entries, parse_freed, - parse_ledger_entry, prune_due, read_ledger, record_freed, take_claim, write_ledger, + CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, FREED_PATH, FreedRecords, LEDGERS_PATH, + ListedClaim, Percent, PruneLedger, claim_directories_except, claims_directory, count_freed, + list_claims, list_freed, needs_repository_size, newest_ledger, next_claim, older_entries, + parse_freed, parse_ledger_entry, prune_due, read_ledger, record_freed, take_claim, + write_ledger, }; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; @@ -710,6 +740,29 @@ mod tests { ); } + #[test] + fn each_claim_directory_other_than_the_new_one_is_old() { + let claim = |directory: &str, number: &str| ListedBlob { + path: Path::new(CLAIMS_PATH).join(directory).join(number).into(), + size: 0, + }; + let directory = |name: &str| Path::new(CLAIMS_PATH).join(name); + + assert_eq!( + claim_directories_except( + &[ + claim("none", "0"), + claim("100", "0"), + claim("100", "1"), + claim("200", "3"), + claim("300", "0"), + ], + &directory("300") + ), + vec![directory("100"), directory("200"), directory("none")] + ); + } + #[test] fn a_ledger_entry_name_gives_the_end_time_and_the_marked_packs() { assert_eq!( diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index f6de73711e..ca85db61c0 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -25,9 +25,10 @@ use super::fault::{Operation, classify, is_file_missing, is_storage_failure, sto use super::files::SnapshotFiles; use super::priority::LowPriority; use super::prune::{ - ClaimChoice, Percent, claims_directory, end_claims, keep_claim_fresh, list_claims, list_freed, + ClaimChoice, Percent, PruneLedger, claims_directory, keep_claim_fresh, list_claims, list_freed, needs_repository_size, next_claim, prune_due, read_ledger, record_freed, refresh_period, - release_claim, remove_freed, remove_older_ledgers, repository_bytes, take_claim, write_ledger, + release_claim, remove_freed, remove_old_claims, remove_older_ledgers, repository_bytes, + take_claim, write_ledger, }; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -383,7 +384,11 @@ impl RusticSnapshotStore { .map_err(storage_failure)?; remove_older_ledgers(&files, ended).await; remove_freed(&files, &records).await; - end_claims(&files, &claims).await; + let new_claims = claims_directory(&PruneLedger { + last_prune: Some(ended), + awaiting_removal: marked_packs, + }); + remove_old_claims(&files, &new_claims).await; Ok(()) } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 653567f1be..15f0a5b428 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1279,6 +1279,40 @@ async fn a_prune_slower_than_the_grace_period_keeps_its_claim_fresh() { ); } +#[test] +#[timeout("60s")] +async fn a_prune_deletes_the_claims_of_old_ledgers() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + futures::stream::iter(["golem/prune-claims/100/0", "golem/prune-claims/200/4"]) + .for_each(|path| { + let storage = storage.clone(); + let scope = scope.clone(); + async move { + storage + .put_raw("test", "test", scope.0.clone(), Path::new(path), b"100") + .await + .unwrap(); + } + }) + .await; + + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert_eq!( + ( + ledger(&storage, &scope).await.last_prune.is_some(), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, Vec::::new()) + ); +} + #[test] #[timeout("60s")] async fn two_deletes_that_read_the_same_ledger_make_one_prune() { @@ -2834,6 +2868,7 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { "list_ledgers", "delete_ledger", "refresh_claim", + "list_claim_directories", ] .contains(&op_label.as_str()) }) From 547deb46d0924e96f7d9667cdc8a582de67e401a Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 17:14:02 -0700 Subject: [PATCH 057/126] Write the record of freed bytes before the forget of a delete --- .../src/filesystem_snapshot/rustic/store.rs | 33 +++++++++++-------- .../filesystem_snapshot/rustic/store/tests.rs | 33 +++++++++++++++++++ 2 files changed, 53 insertions(+), 13 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index ca85db61c0..e7bdb98d53 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -295,9 +295,8 @@ impl RusticSnapshotStore { } } - /// Writes a record of the freed bytes, and prunes the repository when a prune is due. The - /// record stays until a prune succeeds, so a delete that runs again after a failed prune prunes - /// again. It lists the packs only when their size can make a prune due. + /// Prunes the repository when a prune is due. The records of freed bytes stay until a prune + /// succeeds, so a delete that runs again after a failed prune prunes again. It lists the packs only when their size can make a prune due. /// A due prune runs only after the delete takes a claim of its ledger, and only when the ledger /// did not change after the claim. A failed prune deletes the claim, and a prune that succeeds /// deletes each claim of its ledger. @@ -305,12 +304,8 @@ impl RusticSnapshotStore { &self, scope: &SnapshotScope, token: &CancellationToken, - freed: u64, ) -> Result<(), SnapshotStoreError> { let files = self.files(scope, token); - if freed > 0 { - record_freed(&files, freed).await.map_err(storage_failure)?; - } let ledger = read_ledger(&files).await.map_err(storage_failure)?; let records = list_freed(&files).await.map_err(storage_failure)?; let now = Timestamp::now_utc(); @@ -513,7 +508,7 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { let backend = Arc::new(self.backend(scope, &token)?); let key = self.key.clone(); let name = name.clone(); - let freed = self + let found = self .blocking(Operation::Repository, move || { let Some(repository) = open_existing(backend, &key)? else { return Ok(None); @@ -527,14 +522,26 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { .iter() .map(|snapshot| snapshot.id) .collect::>(); - repository.delete_snapshots(&ids)?; - Ok(Some(named.iter().map(added_packed_bytes).sum::())) + let freed = named.iter().map(added_packed_bytes).sum::(); + Ok(Some((repository, ids, freed))) }) .await?; - match freed { - Some(freed) => self.prune_when_due(scope, &token, freed).await, - None => Ok(()), + let Some((repository, ids, freed)) = found else { + return Ok(()); + }; + // The record comes before the forget, so a stop between the two cannot lose the bytes. A + // forget that then fails gives its error, and the record stays, which only brings a prune + // earlier. + let files = self.files(scope, &token); + if freed > 0 { + record_freed(&files, freed).await.map_err(storage_failure)?; } + self.blocking(Operation::Repository, move || { + repository.delete_snapshots(&ids)?; + Ok(()) + }) + .await?; + self.prune_when_due(scope, &token).await } async fn delete_scope(&self, scope: &SnapshotScope) -> Result<(), SnapshotStoreError> { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 15f0a5b428..d49ba19603 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1313,6 +1313,39 @@ async fn a_prune_deletes_the_claims_of_old_ledgers() { ); } +#[test] +async fn a_forget_that_fails_after_the_record_write_leaves_the_record() { + // The forget deletes the snapshot file, and the storage refuses that call. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "delete" && path.starts_with("snapshots") { + Script::Refuse + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + + assert!( + deleted.as_ref().is_err_and(|error| is_storage(error, true)), + "{deleted:?}" + ); + assert_eq!( + ( + freed(&storage, &scope).await > 0, + listed_names(&store, &scope).await + ), + (true, vec!["p-1".to_string()]) + ); +} + #[test] #[timeout("60s")] async fn two_deletes_that_read_the_same_ledger_make_one_prune() { From e69f9b5f410d5da04b6df3859b8a31227a845d5d Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 17:37:29 -0700 Subject: [PATCH 058/126] Check that two deletes make at most one prune in each order of their blob calls --- .../filesystem_snapshot/rustic/scripted.rs | 44 +++ .../filesystem_snapshot/rustic/store/tests.rs | 94 ++++- .../rustic/store/tests/sweep.rs | 335 ++++++++++++++++++ 3 files changed, 472 insertions(+), 1 deletion(-) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs index 0c4946aeee..9215eb0eed 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -27,7 +27,9 @@ use golem_service_base::storage::blob::{ use std::fmt::{Debug, Formatter}; use std::future::Future; use std::path::{Path, PathBuf}; +use std::sync::atomic::{AtomicUsize, Ordering}; use std::sync::{Arc, Mutex, PoisonError}; +use tokio::sync::Semaphore; use tokio_util::sync::CancellationToken; /// What the storage does with one call. @@ -46,6 +48,9 @@ pub(super) enum Script { /// Gives no blob to a read of a whole blob, as a delete after a listing does. Each other call /// passes. Vanish, + /// Waits until the test gives the storage one step, and then passes the call, or refuses it + /// when `refuse` is true. + Step { refuse: bool }, } /// A rule that gives the script of a call from its operation label and its path. @@ -58,6 +63,12 @@ pub(super) struct ScriptedBlobStorage { rule: Rule, calls: Mutex)>>, gate: CancellationToken, + /// The steps that the test gave and that no call took yet. + steps: Semaphore, + /// The calls that wait for a step. + waiting: AtomicUsize, + /// The calls that took a step and ended. + stepped: AtomicUsize, } impl ScriptedBlobStorage { @@ -70,6 +81,9 @@ impl ScriptedBlobStorage { rule: Box::new(rule), calls: Mutex::new(Vec::new()), gate: CancellationToken::new(), + steps: Semaphore::new(0), + waiting: AtomicUsize::new(0), + stepped: AtomicUsize::new(0), }) } @@ -78,6 +92,21 @@ impl ScriptedBlobStorage { self.gate.cancel(); } + /// Lets one call that waits for a step, now or later, go on. + pub(super) fn step(&self) { + self.steps.add_permits(1); + } + + /// Gives the number of calls that wait for a step. + pub(super) fn waiting_steps(&self) -> usize { + self.waiting.load(Ordering::SeqCst) + } + + /// Gives the number of calls that took a step and ended. + pub(super) fn stepped(&self) -> usize { + self.stepped.load(Ordering::SeqCst) + } + /// Gives the operation label and the path of each call, in the order of the calls. pub(super) fn calls(&self) -> Vec<(&'static str, String)> { self.calls @@ -130,6 +159,21 @@ impl ScriptedBlobStorage { call.await } Script::Vanish => call.await, + Script::Step { refuse } => { + self.waiting.fetch_add(1, Ordering::SeqCst); + let permit = self.steps.acquire().await; + self.waiting.fetch_sub(1, Ordering::SeqCst); + if let Ok(permit) = permit { + permit.forget(); + } + let answer = if refuse { + Err(anyhow::anyhow!("the storage refused the call")) + } else { + call.await + }; + self.stepped.fetch_add(1, Ordering::SeqCst); + answer + } } } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index d49ba19603..7b9f593e34 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -37,7 +37,7 @@ use crate::filesystem_snapshot::{ SnapshotStoreError, }; use crate::services::golem_config::FilesystemSnapshotStoreConfig; -use futures::{FutureExt, StreamExt}; +use futures::{FutureExt, StreamExt, TryStreamExt}; use golem_service_base::storage::blob::memory::InMemoryBlobStorage; use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; use pretty_assertions::assert_eq; @@ -3105,3 +3105,95 @@ async fn the_global_rayon_pool_keeps_the_nice_value_of_the_process_after_saves_w "{global:?}" ); } + +mod sweep; + +/// The largest number of steps of one turn. A delete takes at most 13 steps of the protocol, so a +/// turn of 14 steps runs a delete to its end. +const SWEEP_TURN: usize = 14; + +/// The number of random orders that the property test tries. +const SWEEP_CASES: u32 = 1000; + +/// Saves the two snapshots that each case deletes, in a scope that each case copies. +async fn prepared_scope() -> (Arc, SnapshotScope) { + let shared = Arc::new(InMemoryBlobStorage::new()); + let prepared = new_scope(); + save_each( + &store(shared.clone(), policy(LONG_DEADLINE, NEVER, Duration::ZERO)), + &prepared, + &["p-1", "p-2"], + ) + .await; + (shared, prepared) +} + +#[test] +#[timeout("60s")] +async fn two_deletes_make_at_most_one_prune_in_each_order_with_up_to_two_switches() { + // Each order where one delete runs some steps, the other runs some steps, and then each runs + // to its end. This holds each pause of one delete while the other runs. + let (shared, prepared) = prepared_scope().await; + let schedules = (0..2).flat_map(|first| { + (0..=SWEEP_TURN).flat_map(move |one| { + (0..=SWEEP_TURN).map(move |two| sweep::Schedule { + first, + turns: vec![one, two], + fail: None, + }) + }) + }); + let started = std::time::Instant::now(); + + let cases = futures::stream::iter(schedules) + .then(|schedule| { + let (shared, prepared) = (&shared, &prepared); + async move { sweep::run_case(shared, prepared, &schedule).await } + }) + .try_fold(0usize, |cases, _| async move { Ok(cases + 1) }) + .await; + + println!("{cases:?} orders in {:?}", started.elapsed()); + assert!(cases.is_ok(), "{cases:?}"); +} + +#[test] +#[timeout("60s")] +async fn two_deletes_make_at_most_one_prune_in_random_orders_with_a_failed_call() { + // The test generates the order of the steps and a call that fails, and shrinks a failing case + // to the shortest order. The seed is fixed, and no file keeps a failing case. + use proptest::prelude::{Strategy, prop}; + use proptest::test_runner::{Config, RngAlgorithm, TestRng, TestRunner}; + let (shared, prepared) = prepared_scope().await; + let runtime = tokio::runtime::Handle::current(); + let strategy = ( + 0usize..2, + prop::collection::vec(0usize..=SWEEP_TURN, 0..=8), + prop::option::of((0usize..2, 0usize..SWEEP_TURN)), + ) + .prop_map(|(first, turns, fail)| sweep::Schedule { first, turns, fail }); + let started = std::time::Instant::now(); + + let outcome = tokio::task::spawn_blocking(move || { + let mut runner = TestRunner::new_with_rng( + Config { + cases: SWEEP_CASES, + failure_persistence: None, + ..Config::default() + }, + TestRng::deterministic_rng(RngAlgorithm::ChaCha), + ); + runner + .run(&strategy, |schedule| { + runtime + .block_on(sweep::run_case(&shared, &prepared, &schedule)) + .map(|_| ()) + .map_err(proptest::test_runner::TestCaseError::fail) + }) + .map_err(|error| error.to_string()) + }) + .await; + + println!("{SWEEP_CASES} random orders in {:?}", started.elapsed()); + assert!(matches!(outcome, Ok(Ok(()))), "{outcome:?}"); +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs new file mode 100644 index 0000000000..c775ecabab --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs @@ -0,0 +1,335 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Two deletes of one scope whose prunes are both due, in each order of their blob calls. +//! +//! Each delete has its own store over its own scripted storage, and the two storages share one +//! in-memory storage, as two executors share a bucket. Each blob call of the prune protocol is a +//! step: it waits until the test gives its delete one step. So the test sets the order of the +//! calls of the two deletes, and nothing else changes it. + +use super::*; +use futures::TryStreamExt; +use tokio::task::JoinHandle; + +/// The operation labels of the blob calls of the prune protocol. The listing of the packs by a +/// prune is a step too, and it is the start of the prune. +const STEP_LABELS: &[&str] = &[ + "write_freed", + "read_ledger", + "list_freed", + "list_data", + "list_claims", + "read_claim", + "write_claim", + "refresh_claim", + "write_ledger", + "list_ledgers", + "delete_ledger", + "delete_freed", + "delete_claim", + "list_claim_directories", + "delete_claims", +]; + +/// The most steps that one delete takes before the test gives up on it. +const MOST_STEPS: usize = 64; + +fn is_step(op_label: &str, path: &Path) -> bool { + STEP_LABELS.contains(&op_label) || (op_label == "list" && path == Path::new("data")) +} + +fn is_prune_start(op_label: &str, path: &str) -> bool { + op_label == "list" && path == "data" +} + +/// The order of the steps of one case: the delete that goes first, and the numbers of steps that +/// the deletes take in turn. After the listed turns, the delete whose turn is next runs to its +/// end, and then the other one does. `fail` refuses one step of one delete. +#[derive(Clone, Debug)] +pub(super) struct Schedule { + pub(super) first: usize, + pub(super) turns: Vec, + pub(super) fail: Option<(usize, usize)>, +} + +/// One step that a delete took, in the order of all steps of the case. +#[derive(Clone, Debug)] +struct Step { + delete: usize, + op_label: &'static str, + path: String, + refused: bool, +} + +/// One delete of the case: its scripted storage and its task. +struct Delete { + storage: Arc, + task: JoinHandle>, + taken: usize, +} + +impl Delete { + fn finished(&self) -> bool { + self.task.is_finished() + } +} + +/// Waits until the condition holds, and gives false when it does not hold within [`LIMIT`]. The +/// wait yields to the runtime between two checks, so a step ends as soon as it can. +async fn until(condition: impl Fn() -> bool) -> bool { + tokio::time::timeout(LIMIT, async { + futures::stream::repeat(()) + .then(|()| tokio::task::yield_now()) + .take_while(|()| std::future::ready(!condition())) + .for_each(|()| std::future::ready(())) + .await + }) + .await + .is_ok() +} + +/// Waits until the delete waits for a step or ended. +async fn settle(delete: &Delete) -> bool { + until(|| delete.storage.waiting_steps() > 0 || delete.finished()).await +} + +/// Gives the delete one step and waits until the step ended and the delete waits again or ended. +async fn take_step( + deletes: &mut [Delete; 2], + who: usize, + fail: Option<(usize, usize)>, + log: &mut Vec, +) -> Result<(), String> { + let delete = &mut deletes[who]; + if delete.finished() { + return Ok(()); + } + let calls = delete.storage.calls(); + let (op_label, path) = calls + .iter() + .rev() + .find(|(op_label, path)| is_step(op_label, Path::new(path))) + .cloned() + .ok_or_else(|| format!("delete {who} waits for a step with no step call"))?; + let before = delete.storage.stepped(); + delete.storage.step(); + let storage = delete.storage.clone(); + if !until(|| storage.stepped() > before).await { + return Err(format!( + "the step {op_label} {path} of delete {who} did not end" + )); + } + log.push(Step { + delete: who, + op_label, + path, + refused: fail == Some((who, delete.taken)), + }); + delete.taken += 1; + if delete.taken > MOST_STEPS { + return Err(format!("delete {who} took more than {MOST_STEPS} steps")); + } + if settle(delete).await { + Ok(()) + } else { + Err(format!("delete {who} did not reach its next step")) + } +} + +/// Gives the delete steps until it ends. +async fn run_to_end( + deletes: &mut [Delete; 2], + who: usize, + fail: Option<(usize, usize)>, + log: &mut Vec, +) -> Result<(), String> { + futures::stream::iter(0..=MOST_STEPS) + .map(Ok) + .try_fold((deletes, log), |(deletes, log), _| async move { + take_step(deletes, who, fail, log).await?; + Ok::<_, String>((deletes, log)) + }) + .await + .map(|_| ()) +} + +/// What a case found. +#[derive(Debug)] +pub(super) struct Found { + pub(super) prunes: usize, + pub(super) refused: bool, + pub(super) steps: [usize; 2], +} + +/// Runs the two deletes of a copy of the prepared scope in the order of the schedule, and checks +/// the rules of the prune protocol. +pub(super) async fn run_case( + shared: &Arc, + prepared: &SnapshotScope, + schedule: &Schedule, +) -> Result { + let scope = new_scope(); + store(shared.clone(), policy(LONG_DEADLINE, NEVER, Duration::ZERO)) + .copy_scope(prepared, &scope) + .await + .map_err(|error| format!("the copy of the prepared scope failed: {error}"))?; + let deletes = [0, 1].map(|who| { + let counter = Arc::new(AtomicUsize::new(0)); + let fail = schedule.fail; + let storage = ScriptedBlobStorage::new(shared.clone(), move |op_label, path| { + if is_step(op_label, path) { + let number = counter.fetch_add(1, Ordering::SeqCst); + Script::Step { + refuse: fail == Some((who, number)), + } + } else { + Script::Pass + } + }); + let deleting = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = scope.clone(); + let task = + tokio::spawn(async move { deleting.delete(&scope, &name(["p-1", "p-2"][who])).await }); + Delete { + storage, + task, + taken: 0, + } + }); + let mut deletes = deletes; + let mut log = Vec::new(); + if !(settle(&deletes[0]).await && settle(&deletes[1]).await) { + return Err("a delete did not reach its first step".to_string()); + } + let (next, deletes, log) = futures::stream::iter(schedule.turns.iter().enumerate()) + .map(Ok) + .try_fold( + (schedule.first, &mut deletes, &mut log), + |(_, deletes, log), (turn, steps)| async move { + let who = (schedule.first + turn) % 2; + futures::stream::iter(0..*steps) + .map(Ok) + .try_fold((deletes, log), |(deletes, log), _| async move { + take_step(deletes, who, schedule.fail, log).await?; + Ok::<_, String>((deletes, log)) + }) + .await + .map(|(deletes, log)| ((who + 1) % 2, deletes, log)) + }, + ) + .await?; + run_to_end(deletes, next, schedule.fail, log).await?; + run_to_end(deletes, (next + 1) % 2, schedule.fail, log).await?; + let results = + futures::future::join_all(deletes.iter_mut().map(|delete| &mut delete.task)).await; + check(shared, &scope, schedule, log, results).await +} + +/// Checks the rules on the end state of a case. +async fn check( + shared: &Arc, + scope: &SnapshotScope, + schedule: &Schedule, + log: &[Step], + results: Vec, tokio::task::JoinError>>, +) -> Result { + let refused = log.iter().any(|step| step.refused); + let prune_starts = log + .iter() + .enumerate() + .filter(|(_, step)| !step.refused && is_prune_start(step.op_label, &step.path)) + .collect::>(); + let prunes = prune_starts.len(); + let order = || { + log.iter() + .map(|step| { + format!( + "{}:{}{}", + step.delete, + step.op_label, + if step.refused { "!" } else { "" } + ) + }) + .collect::>() + .join(" ") + }; + let fail = |rule: &str| Err(format!("{rule}; schedule {schedule:?}; steps {}", order())); + if prunes > 1 { + return fail("more than one prune ran"); + } + if !refused && prunes != 1 { + return fail("no prune ran, although a prune was due and no call failed"); + } + if !refused && results.iter().any(|result| !matches!(result, Ok(Ok(())))) { + return fail(&format!( + "a delete failed with no refused call: {results:?}" + )); + } + let claims = blobs(&**shared, &scope.0, "golem/prune-claims/").await; + if !refused && !claims.is_empty() { + return fail(&format!("claims stay: {claims:?}")); + } + let records = blobs(&**shared, &scope.0, "golem/prune-freed/").await; + let entries = blobs(&**shared, &scope.0, "golem/prune-ledgers/").await; + let ledger = ledger(shared, scope).await; + if let Some((start, pruner)) = prune_starts + .first() + .map(|(index, step)| (*index, step.delete)) + { + let wrote_ledger = log[start..] + .iter() + .find(|step| step.delete == pruner && step.op_label == "write_ledger" && !step.refused); + if let Some(written) = wrote_ledger { + let written_ms = Path::new(&written.path) + .file_name() + .and_then(|name| name.to_str()) + .and_then(|name| name.split('-').next()) + .and_then(|ms| ms.parse::().ok()); + if ledger.last_prune.map(|last| last.to_millis()) != written_ms { + return fail(&format!( + "the ledger {ledger:?} is not the entry of the prune {written_ms:?}; entries {entries:?}" + )); + } + } + let counted_at = log[..start] + .iter() + .rposition(|step| step.delete == pruner && step.op_label == "list_freed"); + let written_records = log + .iter() + .enumerate() + .filter(|(_, step)| step.op_label == "write_freed" && !step.refused) + .collect::>(); + let lost = written_records + .iter() + .filter(|(index, _)| counted_at.is_none_or(|counted| *index > counted)) + .filter(|(_, step)| !records.contains(&step.path)) + .map(|(_, step)| step.path.clone()) + .collect::>(); + if !lost.is_empty() { + return fail(&format!( + "records that the prune did not count are gone: {lost:?}" + )); + } + } + let steps = [0, 1].map(|who| log.iter().filter(|step| step.delete == who).count()); + Ok(Found { + prunes, + refused, + steps, + }) +} From e4b52c93d810edf4731239aa185928ac5817e091 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 21:09:39 -0700 Subject: [PATCH 059/126] Release the claim of a prune on each error before the prune ran --- .../src/filesystem_snapshot/rustic/store.rs | 118 +++++++++++++----- .../filesystem_snapshot/rustic/store/tests.rs | 113 +++++++++++++++++ 2 files changed, 197 insertions(+), 34 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index e7bdb98d53..141cea7bb3 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -166,6 +166,9 @@ pub(crate) struct RusticSnapshotStore { /// Holds a save after its blocking work and before its publish, when a test sets it. #[cfg(test)] pub(super) publish_gate: Option>, + /// Makes each backend build fail while a test sets it. + #[cfg(test)] + pub(super) refuse_backends: Arc, } /// A gate that holds a save after its blocking work and before its publish. @@ -178,6 +181,20 @@ pub(super) struct PublishGate { pub(super) open: tokio::sync::Notify, } +/// The claim of a prune that a delete holds. +struct Claim<'a> { + directory: &'a Path, + number: u64, +} + +/// How the work of a delete that holds a claim ended without an error. +enum ClaimOutcome { + /// The prune ran, and it marked packs or not. + Pruned { marked_packs: bool }, + /// Another prune ended after the claim, so this delete did not prune. + Superseded, +} + /// The error of an operation of a store that is shut down. fn shut_down_error() -> SnapshotStoreError { SnapshotStoreError::Storage { @@ -216,6 +233,8 @@ impl RusticSnapshotStore { low_priority: LowPriority::new(policy.save_threads), #[cfg(test)] publish_gate: None, + #[cfg(test)] + refuse_backends: Arc::default(), } } @@ -253,6 +272,16 @@ impl RusticSnapshotStore { scope: &SnapshotScope, token: &CancellationToken, ) -> Result { + #[cfg(test)] + if self + .refuse_backends + .load(std::sync::atomic::Ordering::SeqCst) + { + return Err(SnapshotStoreError::Storage { + retryable: true, + source: anyhow::anyhow!("the test refuses to build a backend"), + }); + } let runtime = Handle::try_current() .context("a filesystem snapshot operation needs an async runtime") .map_err(|source| SnapshotStoreError::Storage { @@ -298,8 +327,11 @@ impl RusticSnapshotStore { /// Prunes the repository when a prune is due. The records of freed bytes stay until a prune /// succeeds, so a delete that runs again after a failed prune prunes again. It lists the packs only when their size can make a prune due. /// A due prune runs only after the delete takes a claim of its ledger, and only when the ledger - /// did not change after the claim. A failed prune deletes the claim, and a prune that succeeds - /// deletes each claim of its ledger. + /// did not change after the claim. After an error before the prune ran, the claim is deleted, + /// so a retry of the delete prunes again, and a claim write that the storage completes after + /// that delete can delay that prune by up to the grace period. After a prune whose ledger write + /// failed, the claim stays, so the next prune waits up to the grace period. A prune that + /// succeeds deletes each claim of its ledger. async fn prune_when_due( &self, scope: &SnapshotScope, @@ -338,41 +370,23 @@ impl RusticSnapshotStore { { return Ok(()); } - // A prune writes its ledger before it deletes the claims, so a delete that claims in a - // directory that such a prune removed sees the new ledger here. - match read_ledger(&files).await { - Ok(again) if claims_directory(&again) == claims => {} - Ok(_) => { - release_claim(&files, &claims, number).await; - return Ok(()); - } - Err(error) => { - release_claim(&files, &claims, number).await; - return Err(storage_failure(error)); - } - } - let backend = Arc::new(self.backend(scope, token)?); - let key = self.key.clone(); - let settings = self.policy.prune; - let low_priority = self.low_priority; - let pruning = self.blocking(Operation::Prune, move || { - low_priority.run("fs-snap-prune", move || prune(backend, &key, &settings)) - }); - // The claim is written again while the prune runs, so a prune slower than the grace - // period keeps its claim. The writes stop when the prune ends. - let refreshing = keep_claim_fresh(&files, &claims, number, refresh_period(grace)); - let pruned = match future::select(pin!(pruning), pin!(refreshing)).await { - Either::Left((pruned, _)) => pruned, - Either::Right(((), pruning)) => pruning.await, + let claim = Claim { + directory: &claims, + number, }; - let report = match pruned { - Ok(report) => report, - Err(error) => { - release_claim(&files, &claims, number).await; - return Err(error); + let outcome = self + .prune_with_claim(scope, token, &files, &claim, grace) + .await; + // The claim is released unless the prune ran, so a retry of the delete prunes again. After + // a prune, the claim stays also when the ledger write fails, so no second prune runs close + // to the first one. + let marked_packs = match outcome { + Ok(ClaimOutcome::Pruned { marked_packs }) => marked_packs, + other => { + release_claim(&files, claim.directory, claim.number).await; + return other.map(|_| ()); } }; - let marked_packs = report.as_ref().is_some_and(leaves_marked_packs); let ended = Timestamp::now_utc(); write_ledger(&files, ended, marked_packs) .await @@ -386,6 +400,42 @@ impl RusticSnapshotStore { remove_old_claims(&files, &new_claims).await; Ok(()) } + + /// Prunes while the delete holds the claim. It gives `Superseded` when another prune ended after + /// the claim, and an error when no prune ran. + async fn prune_with_claim( + &self, + scope: &SnapshotScope, + token: &CancellationToken, + files: &SnapshotFiles, + claim: &Claim<'_>, + grace: Duration, + ) -> Result { + // A prune writes its ledger before it deletes the claims, so a delete that claims in a + // directory that such a prune removed sees the new ledger here. + let again = read_ledger(files).await.map_err(storage_failure)?; + if claims_directory(&again) != claim.directory { + return Ok(ClaimOutcome::Superseded); + } + let backend = Arc::new(self.backend(scope, token)?); + let key = self.key.clone(); + let settings = self.policy.prune; + let low_priority = self.low_priority; + let pruning = self.blocking(Operation::Prune, move || { + low_priority.run("fs-snap-prune", move || prune(backend, &key, &settings)) + }); + // The claim is written again while the prune runs, so a prune slower than the grace + // period keeps its claim. The writes stop when the prune ends. + let refreshing = + keep_claim_fresh(files, claim.directory, claim.number, refresh_period(grace)); + let report = match future::select(pin!(pruning), pin!(refreshing)).await { + Either::Left((pruned, _)) => pruned, + Either::Right(((), pruning)) => pruning.await, + }?; + Ok(ClaimOutcome::Pruned { + marked_packs: report.as_ref().is_some_and(leaves_marked_packs), + }) + } } #[async_trait] diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 7b9f593e34..e9324322a2 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1313,6 +1313,119 @@ async fn a_prune_deletes_the_claims_of_old_ledgers() { ); } +#[test] +#[timeout("60s")] +async fn a_failed_ledger_write_after_a_prune_keeps_the_claim_so_no_second_prune_runs() { + // The first prune runs and its ledger write fails. A delete right after it finds the claim + // and does not prune. When the claim is older than the grace period and the margin, the next + // delete prunes. + let refused = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refused = refused.clone(); + move |op_label, _| { + if op_label == "write_ledger" && !refused.swap(true, Ordering::SeqCst) { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2", "p-3"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + let second = store.delete(&scope, &name("p-2")).await; + let prunes_after_second = prunes(&storage.calls()); + let stale = golem_common::model::Timestamp::now_utc() + .to_millis() + .saturating_sub(3_600_000 + 2 * 60_000 + 1); + futures::stream::iter(&claims_after_failure) + .for_each(|claim| { + let (storage, scope) = (&storage, &scope); + async move { + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new(claim), + stale.to_string().as_bytes(), + ) + .await + .unwrap(); + } + }) + .await; + let third = store.delete(&scope, &name("p-3")).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!(second.is_ok(), "{second:?}"); + assert!(third.is_ok(), "{third:?}"); + assert_eq!( + ( + claims_after_failure.len(), + prunes_after_second, + prunes(&storage.calls()), + ledger(&storage, &scope).await.last_prune.is_some(), + ), + (1, 1, 2, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_backend_that_does_not_build_after_the_claim_releases_it() { + // The first claim write makes the next backend build fail, and that build is the one of the + // prune. + let refuse_backends = Arc::new(AtomicBool::new(false)); + let armed = Arc::new(AtomicBool::new(true)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse_backends = refuse_backends.clone(); + move |op_label, _| { + if op_label == "write_claim" && armed.swap(false, Ordering::SeqCst) { + refuse_backends.store(true, Ordering::SeqCst); + } + Script::Pass + } + }); + let store = Arc::new(RusticSnapshotStore { + refuse_backends: refuse_backends.clone(), + ..RusticSnapshotStore::with_policy( + storage.clone(), + key(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ) + }); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + refuse_backends.store(false, Ordering::SeqCst); + let retried = store.delete(&scope, &name("p-2")).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!(retried.is_ok(), "{retried:?}"); + assert_eq!( + ( + claims_after_failure, + ledger(&storage, &scope).await.last_prune.is_some(), + ), + (Vec::::new(), true) + ); +} + #[test] async fn a_forget_that_fails_after_the_record_write_leaves_the_record() { // The forget deletes the snapshot file, and the storage refuses that call. From 012214e6f0ee43af0137133fa6ae589c16f8f8f6 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 21:17:25 -0700 Subject: [PATCH 060/126] Count the prune step and its refresh in the tracker, and stop the refresh on cancel --- .../src/filesystem_snapshot/rustic/prune.rs | 30 +++++++- .../src/filesystem_snapshot/rustic/store.rs | 16 ++-- .../filesystem_snapshot/rustic/store/tests.rs | 74 +++++++++++++++++++ 3 files changed, 111 insertions(+), 9 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index c125637de3..a213cd5273 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -399,7 +399,8 @@ pub(super) fn refresh_period(grace: Duration) -> Duration { } /// Writes the claim with the number again, with the current time, at each period, until the -/// caller drops the future. A failed write gives a warning. +/// caller drops the future or the operation of the files is cancelled. A failed write gives a +/// warning. pub(super) async fn keep_claim_fresh( files: &SnapshotFiles, directory: &Path, @@ -409,6 +410,7 @@ pub(super) async fn keep_claim_fresh( let path = directory.join(number.to_string()); stream::repeat(()) .then(|()| tokio::time::sleep(period)) + .take_until(files.cancel.cancelled()) .for_each(|()| { let path = &path; async move { @@ -494,9 +496,9 @@ mod tests { use super::{ CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, FREED_PATH, FreedRecords, LEDGERS_PATH, ListedClaim, Percent, PruneLedger, claim_directories_except, claims_directory, count_freed, - list_claims, list_freed, needs_repository_size, newest_ledger, next_claim, older_entries, - parse_freed, parse_ledger_entry, prune_due, read_ledger, record_freed, take_claim, - write_ledger, + keep_claim_fresh, list_claims, list_freed, needs_repository_size, newest_ledger, + next_claim, older_entries, parse_freed, parse_ledger_entry, prune_due, read_ledger, + record_freed, take_claim, write_ledger, }; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; @@ -714,6 +716,26 @@ mod tests { ); } + #[test] + async fn the_refresh_of_a_claim_ends_when_its_operation_is_cancelled() { + let files = new_files(); + files.cancel.cancel(); + + let ended = tokio::time::timeout( + Duration::from_secs(10), + keep_claim_fresh( + &files, + &claims_directory(&ledger(None, false)), + 0, + Duration::from_secs(3600), + ), + ) + .await + .is_ok(); + + assert!(ended); + } + #[test] async fn a_claim_is_taken_one_time_and_listed_with_its_time() { let files = new_files(); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 141cea7bb3..6930835042 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -425,13 +425,19 @@ impl RusticSnapshotStore { low_priority.run("fs-snap-prune", move || prune(backend, &key, &settings)) }); // The claim is written again while the prune runs, so a prune slower than the grace - // period keeps its claim. The writes stop when the prune ends. + // period keeps its claim. The writes stop when the prune ends or the operation is + // cancelled. The tracker counts the whole step, so no timer of it runs after a shut down. let refreshing = keep_claim_fresh(files, claim.directory, claim.number, refresh_period(grace)); - let report = match future::select(pin!(pruning), pin!(refreshing)).await { - Either::Left((pruned, _)) => pruned, - Either::Right(((), pruning)) => pruning.await, - }?; + let report = self + .tracker + .track_future(async { + match future::select(pin!(pruning), pin!(refreshing)).await { + Either::Left((pruned, _)) => pruned, + Either::Right(((), pruning)) => pruning.await, + } + }) + .await?; Ok(ClaimOutcome::Pruned { marked_packs: report.as_ref().is_some_and(leaves_marked_packs), }) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index e9324322a2..d6efe22509 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -2012,6 +2012,80 @@ async fn shut_down_waits_for_a_blob_call_of_the_store_that_is_not_polled() { ); } +#[test] +#[timeout("60s")] +async fn shut_down_waits_for_the_step_of_a_prune_and_its_refresh_that_is_not_polled() { + // The test drives the delete until its prune waits at the gate and its refresh runs, and then + // does not poll it. Only the tracker of the step makes the shut down wait for it. + let grace = Duration::from_millis(400); + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleted_name = name("p-1"); + let deleting = store.delete(&scope, &deleted_name); + tokio::pin!(deleting); + let reached = tokio::select! { + biased; + () = async { + eventually(|| held.load(Ordering::SeqCst)).await; + } => true, + _ = &mut deleting => false, + }; + + let shutting = store.shut_down(); + tokio::pin!(shutting); + let waited = tokio::time::timeout(Duration::from_millis(200), &mut shutting) + .await + .is_err(); + let deleted = tokio::time::timeout(LIMIT, &mut deleting).await; + // A finished future must not be polled again, so the second wait runs only after a first wait + // that timed out. + let stopped = !waited || tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + let refreshes = |calls: &[(&'static str, String)]| { + calls + .iter() + .filter(|(op_label, _)| *op_label == "refresh_claim") + .count() + }; + let at_stop = refreshes(&storage.calls()); + tokio::time::sleep(grace).await; + storage.open_gate(); + + assert!( + matches!(&deleted, Ok(Err(error)) if is_storage(error, true)), + "{deleted:?}" + ); + assert_eq!( + ( + reached, + waited, + stopped, + store.work_in_flight(), + refreshes(&storage.calls()) - at_stop, + ), + (true, true, true, 0, 0) + ); +} + #[test] #[timeout("60s")] async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_call() { From a2cd19c8f12227bc87b48da11605dc251200329a Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 21:23:17 -0700 Subject: [PATCH 061/126] Fail a copy when a prune removed a file that the copy listed --- .../src/filesystem_snapshot/rustic/scope.rs | 14 ++- .../filesystem_snapshot/rustic/store/tests.rs | 90 +++++++++++++++++++ 2 files changed, 102 insertions(+), 2 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs index e60f507c61..13a736bdd4 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs @@ -26,7 +26,10 @@ use std::path::Path; /// The directories of a repository in the order of a listing. A save writes them in the reverse /// order, and so does a copy, so a snapshot file always has its data. -const LISTING_ORDER: [&str; 4] = ["snapshots", "index", "keys", "data"]; +const LISTING_ORDER: [&str; 4] = [SNAPSHOTS_PATH, "index", "keys", "data"]; + +/// The directory of the snapshot files of a repository. +const SNAPSHOTS_PATH: &str = "snapshots"; /// Copies the repository of `from` into the empty scope `to`, with the config file last, so `to` /// holds a repository only when all its blobs are there. It copies nothing without a config file, @@ -55,7 +58,14 @@ pub(super) async fn copy_scope(from: &SnapshotFiles, to: &SnapshotFiles) -> anyh async fn copy_blob(from: &SnapshotFiles, to: &SnapshotFiles, path: &Path) -> anyhow::Result<()> { match from.get("copy_read", path).await? { Some(content) => to.put("copy_write", path, &content).await, - None => Ok(()), + // A delete removed the snapshot file after the listing, so the copy leaves it out. + None if path.starts_with(SNAPSHOTS_PATH) => Ok(()), + // A prune removed a file that the copy listed, so the copy can be incomplete. It fails + // before the config write, so the target holds no repository, and a new copy can succeed. + None => Err(anyhow::anyhow!( + "the blob {} that the copy listed is gone, because a prune removed it", + path.display() + )), } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index d6efe22509..97bd67c12a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -2086,6 +2086,96 @@ async fn shut_down_waits_for_the_step_of_a_prune_and_its_refresh_that_is_not_pol ); } +#[test] +#[timeout("60s")] +async fn a_copy_fails_when_a_prune_removed_an_index_file_that_it_listed() { + // The gate holds the read of the listed index file, and the test deletes that file meanwhile, + // as a prune does. + let inner = Arc::new(InMemoryBlobStorage::new()); + let storage = ScriptedBlobStorage::new(inner.clone(), |op_label, path| { + if op_label == "copy_read" && path.starts_with("index") { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let (from, to) = (new_scope(), new_scope()); + save_each(&store, &from, &["p-1"]).await; + let copying = tokio::spawn({ + let store = store.clone(); + let (from, to) = (from.clone(), to.clone()); + async move { store.copy_scope(&from, &to).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, path)| *op_label == "copy_read" && path.starts_with("index")) + }) + .await; + let index_files = blobs(&*inner, &from.0, "index/").await; + + futures::stream::iter(&index_files) + .for_each(|path| { + let (inner, from) = (&inner, &from); + async move { + inner + .delete("test", "test", from.0.clone(), Path::new(path)) + .await + .unwrap(); + } + }) + .await; + storage.open_gate(); + let copied = tokio::time::timeout(LIMIT, copying).await; + + assert!( + matches!(&copied, Ok(Ok(Err(error))) if is_storage(error, true)), + "{copied:?}" + ); + assert_eq!( + ( + held, + index_files.is_empty(), + blobs(&*inner, &to.0, "config").await + ), + (true, false, Vec::::new()) + ); +} + +#[test] +async fn a_copy_leaves_out_a_snapshot_file_that_a_delete_removed_after_the_listing() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "copy_read" && path.starts_with("snapshots") { + Script::Vanish + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let (from, to) = (new_scope(), new_scope()); + save_each(&store, &from, &["p-1"]).await; + + let copied = store.copy_scope(&from, &to).await; + + assert!(copied.is_ok(), "{copied:?}"); + assert_eq!( + ( + blobs(&*storage, &to.0, "config").await.len(), + blobs(&*storage, &to.0, "snapshots/").await + ), + (1, Vec::::new()) + ); +} + #[test] #[timeout("60s")] async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_call() { From a6696934d5fedd7bd59b139e81afcacc5b5b0798 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 21:29:34 -0700 Subject: [PATCH 062/126] Keep the lists that a prune and a copy build one time as boxed slices --- .../src/filesystem_snapshot/rustic/prune.rs | 65 +++++++++++-------- .../src/filesystem_snapshot/rustic/scope.rs | 3 +- 2 files changed, 39 insertions(+), 29 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index a213cd5273..3b57627fd6 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -166,7 +166,7 @@ pub(super) fn newest_ledger(listed: &[ListedBlob], now: Timestamp) -> PruneLedge /// Gives the paths of the listed entries whose time is before `ended`, in whole milliseconds as /// an entry name holds it. -pub(super) fn older_entries(listed: &[ListedBlob], ended: Timestamp) -> Vec> { +pub(super) fn older_entries(listed: &[ListedBlob], ended: Timestamp) -> Box<[Box]> { listed .iter() .filter(|blob| { @@ -239,7 +239,7 @@ pub(super) async fn remove_older_ledgers(files: &SnapshotFiles, ended: Timestamp #[derive(Clone, Debug, Default, PartialEq, Eq)] pub(super) struct FreedRecords { pub(super) bytes: u64, - pub(super) counted: Vec>, + pub(super) counted: Box<[Box]>, } /// Reads the freed bytes from the name of a record, `-`. @@ -254,20 +254,20 @@ pub(super) fn parse_freed(name: &str) -> Option { /// Sums the freed bytes of the listed records. A record whose name does not parse counts as zero /// bytes, and it is not counted, so a prune leaves it in place. pub(super) fn count_freed(listed: &[ListedBlob]) -> FreedRecords { - listed + let parsed = listed .iter() .filter_map(|blob| { let bytes = parse_freed(blob.path.file_name()?.to_str()?)?; Some((bytes, blob.path.clone())) }) - .fold(FreedRecords::default(), |records, (bytes, path)| { - let mut counted = records.counted; - counted.push(path); - FreedRecords { - bytes: records.bytes.saturating_add(bytes), - counted, - } - }) + .collect::>(); + FreedRecords { + bytes: parsed + .iter() + .map(|(bytes, _)| *bytes) + .fold(0, u64::saturating_add), + counted: parsed.iter().map(|(_, path)| path.clone()).collect(), + } } /// Writes a record of the freed bytes of one delete. @@ -349,25 +349,26 @@ pub(super) fn claims_directory(ledger: &PruneLedger) -> PathBuf { pub(super) async fn list_claims( files: &SnapshotFiles, directory: &Path, -) -> anyhow::Result> { +) -> anyhow::Result> { let listed = files.list_below("list_claims", directory).await?; let numbered = listed .iter() .filter_map(|blob| { let number = blob.path.file_name()?.to_str()?.parse::().ok()?; - Some((number, blob.path.clone())) + Some((number, &blob.path)) }) - .collect::>(); - stream::iter(numbered) + .collect::>(); + stream::iter(numbered.iter()) .then(|(number, path)| async move { - let content = files.get("read_claim", &path).await?; + let content = files.get("read_claim", path).await?; Ok::<_, anyhow::Error>(ListedClaim { - number, + number: *number, claimed_at: content.as_deref().and_then(parse_claim), }) }) - .try_collect() + .try_collect::>() .await + .map(Vec::into_boxed_slice) } /// Writes the claim with the number, and tells whether this call wrote it. @@ -441,14 +442,15 @@ pub(super) async fn release_claim(files: &SnapshotFiles, directory: &Path, numbe } /// Gives each claim directory of the listed claims, other than the directory to keep. -pub(super) fn claim_directories_except(listed: &[ListedBlob], keep: &Path) -> Vec { - listed +pub(super) fn claim_directories_except(listed: &[ListedBlob], keep: &Path) -> Box<[PathBuf]> { + let mut directories = listed .iter() .filter_map(|blob| blob.path.parent().map(Path::to_path_buf)) .filter(|directory| directory != keep) - .collect::>() - .into_iter() - .collect() + .collect::>(); + directories.sort(); + directories.dedup(); + directories.into_boxed_slice() } /// Deletes each claim directory other than the directory of the new ledger. It never deletes the @@ -749,7 +751,12 @@ mod tests { let listed = list_claims(&files, &directory).await.unwrap(); assert_eq!( - (directory.display().to_string(), first, again, listed), + ( + directory.display().to_string(), + first, + again, + listed.to_vec() + ), ( "golem/prune-claims/42".to_string(), true, @@ -780,7 +787,8 @@ mod tests { claim("300", "0"), ], &directory("300") - ), + ) + .to_vec(), vec![directory("100"), directory("200"), directory("none")] ); } @@ -857,7 +865,8 @@ mod tests { entry("bad"), ], at(9) - ), + ) + .to_vec(), vec![Path::new(LEDGERS_PATH).join("5-0-a").into_boxed_path()] ); } @@ -895,10 +904,10 @@ mod tests { records, FreedRecords { bytes: u64::MAX, - counted: vec![ + counted: Box::new([ Path::new(FREED_PATH).join("5-a").into(), Path::new(FREED_PATH).join(format!("{}-b", u64::MAX)).into(), - ], + ]), } ); } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs index 13a736bdd4..17365f4996 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs @@ -41,7 +41,8 @@ pub(super) async fn copy_scope(from: &SnapshotFiles, to: &SnapshotFiles) -> anyh let listed = stream::iter(LISTING_ORDER) .then(|directory| from.list_below("copy_list", Path::new(directory))) .try_collect::>() - .await?; + .await? + .into_boxed_slice(); let paths = listed .iter() .rev() From 00a56a5b78c616f8b6858eced783258eca83edf5 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 22:04:47 -0700 Subject: [PATCH 063/126] Count a record of freed bytes only when its snapshot files are gone --- .../src/filesystem_snapshot/rustic/prune.rs | 194 ++++++++++++---- .../src/filesystem_snapshot/rustic/store.rs | 8 +- .../filesystem_snapshot/rustic/store/tests.rs | 208 +++++++++++++++++- 3 files changed, 354 insertions(+), 56 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 3b57627fd6..ec4c1245d6 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -28,6 +28,7 @@ use super::files::SnapshotFiles; use futures::{StreamExt, TryStreamExt, stream}; use golem_common::model::Timestamp; use golem_service_base::storage::blob::{ListedBlob, PutIfAbsent}; +use std::collections::HashSet; use std::path::{Path, PathBuf}; use std::time::Duration; use tracing::warn; @@ -235,54 +236,122 @@ pub(super) async fn remove_older_ledgers(files: &SnapshotFiles, ended: Timestamp .await; } -/// The freed bytes of the records that a listing found, and the paths of the records it counted. +/// The freed bytes of the settled records, and the paths of those records. #[derive(Clone, Debug, Default, PartialEq, Eq)] pub(super) struct FreedRecords { pub(super) bytes: u64, pub(super) counted: Box<[Box]>, } -/// Reads the freed bytes from the name of a record, `-`. -pub(super) fn parse_freed(name: &str) -> Option { - let (bytes, unique) = name.split_once('-')?; - if unique.is_empty() { - return None; - } - bytes.parse().ok() +/// A record of freed bytes that a delete wrote: its path, the bytes in its name, and the ids of +/// the snapshot files of that delete, when its content parses. +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct FreedRecord { + pub(super) path: Box, + pub(super) bytes: u64, + pub(super) snapshots: Option>, +} + +/// The directory of the snapshot files of a repository. +const SNAPSHOTS_PATH: &str = "snapshots"; + +/// Gives the content of a record: the id of each snapshot file of the delete, one on each line. +pub(super) fn record_content(snapshots: &[String]) -> String { + snapshots.join("\n") +} + +/// Reads the snapshot ids from the content of a record. Each line must be an id of 64 hex +/// characters, and an empty content has no id. +pub(super) fn parse_record(content: &[u8]) -> Option> { + let text = std::str::from_utf8(content).ok()?; + text.lines() + .filter(|line| !line.is_empty()) + .map(|line| { + (line.len() == 64 && line.bytes().all(|byte| byte.is_ascii_hexdigit())) + .then(|| line.to_string()) + }) + .collect() } -/// Sums the freed bytes of the listed records. A record whose name does not parse counts as zero -/// bytes, and it is not counted, so a prune leaves it in place. -pub(super) fn count_freed(listed: &[ListedBlob]) -> FreedRecords { - let parsed = listed +/// Gives the settled records and the sum of their bytes. A record is settled when none of its +/// snapshot files exists. Any other record counts as zero bytes and stays, and so does a record +/// whose content does not parse. +pub(super) fn settle(records: &[FreedRecord], existing: &HashSet) -> FreedRecords { + let settled = records .iter() - .filter_map(|blob| { - let bytes = parse_freed(blob.path.file_name()?.to_str()?)?; - Some((bytes, blob.path.clone())) + .filter(|record| { + record + .snapshots + .as_ref() + .is_some_and(|snapshots| snapshots.iter().all(|id| !existing.contains(id))) }) .collect::>(); FreedRecords { - bytes: parsed + bytes: settled .iter() - .map(|(bytes, _)| *bytes) + .map(|record| record.bytes) .fold(0, u64::saturating_add), - counted: parsed.iter().map(|(_, path)| path.clone()).collect(), + counted: settled.iter().map(|record| record.path.clone()).collect(), } } -/// Writes a record of the freed bytes of one delete. -pub(super) async fn record_freed(files: &SnapshotFiles, bytes: u64) -> anyhow::Result<()> { +/// Reads the freed bytes from the name of a record, `-`. +pub(super) fn parse_freed(name: &str) -> Option { + let (bytes, unique) = name.split_once('-')?; + if unique.is_empty() { + return None; + } + bytes.parse().ok() +} + +/// Writes a record of the freed bytes of one delete, with the ids of its snapshot files. +pub(super) async fn record_freed( + files: &SnapshotFiles, + bytes: u64, + snapshots: &[String], +) -> anyhow::Result<()> { let path = Path::new(FREED_PATH).join(format!("{bytes}-{}", uuid::Uuid::new_v4())); - files.put("write_freed", &path, &[]).await + files + .put("write_freed", &path, record_content(snapshots).as_bytes()) + .await } -/// Lists the records of freed bytes, and sums them. +/// Lists and reads the records of freed bytes, lists the snapshot files one time, and gives the +/// settled records. A name that does not parse counts as zero bytes and stays, and a record that +/// a prune deleted after the listing is left out. Without records, no snapshot file is listed. pub(super) async fn list_freed(files: &SnapshotFiles) -> anyhow::Result { - Ok(count_freed( - &files - .list_below("list_freed", Path::new(FREED_PATH)) - .await?, - )) + let listed = files + .list_below("list_freed", Path::new(FREED_PATH)) + .await?; + let named = listed + .iter() + .filter_map(|blob| { + let bytes = parse_freed(blob.path.file_name()?.to_str()?)?; + Some((bytes, &blob.path)) + }) + .collect::>(); + if named.is_empty() { + return Ok(FreedRecords::default()); + } + let records = stream::iter(named.iter()) + .then(|(bytes, path)| async move { + let content = files.get("read_freed", path).await?; + Ok::<_, anyhow::Error>(content.map(|content| FreedRecord { + path: (*path).clone(), + bytes: *bytes, + snapshots: parse_record(&content), + })) + }) + .try_filter_map(|record| std::future::ready(Ok(record))) + .try_collect::>() + .await?; + let existing = files + .list_below("list_snapshots", Path::new(SNAPSHOTS_PATH)) + .await? + .iter() + .filter_map(|blob| Some(blob.path.file_name()?.to_str()?.to_string())) + .collect::>(); + Ok(settle(&records, &existing)) } /// Deletes the counted records after a prune. A failure gives a warning, because a record that @@ -496,11 +565,11 @@ fn parse_claim(content: &[u8]) -> Option { mod tests { use super::super::files::SnapshotFiles; use super::{ - CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, FREED_PATH, FreedRecords, LEDGERS_PATH, - ListedClaim, Percent, PruneLedger, claim_directories_except, claims_directory, count_freed, - keep_claim_fresh, list_claims, list_freed, needs_repository_size, newest_ledger, - next_claim, older_entries, parse_freed, parse_ledger_entry, prune_due, read_ledger, - record_freed, take_claim, write_ledger, + CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, FREED_PATH, FreedRecord, FreedRecords, + LEDGERS_PATH, ListedClaim, Percent, PruneLedger, claim_directories_except, + claims_directory, keep_claim_fresh, list_claims, list_freed, needs_repository_size, + newest_ledger, next_claim, older_entries, parse_freed, parse_ledger_entry, parse_record, + prune_due, read_ledger, record_content, record_freed, settle, take_claim, write_ledger, }; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; @@ -887,37 +956,68 @@ mod tests { } #[test] - fn the_records_that_parse_are_summed_and_counted_and_the_others_stay() { - let blob = |name: &str| ListedBlob { + fn a_record_counts_only_when_each_of_its_snapshot_files_is_gone() { + let id = |digit: char| std::iter::repeat_n(digit, 64).collect::(); + let record = |name: &str, bytes: u64, snapshots: Option>| FreedRecord { path: Path::new(FREED_PATH).join(name).into(), - size: 0, + bytes, + snapshots: snapshots.map(Vec::into_boxed_slice), }; - - let records = count_freed(&[ - blob("5-a"), - blob("not-a-count"), - blob(&format!("{}-b", u64::MAX)), - blob("7"), - ]); + let existing = [id('b')].into_iter().collect(); + + let settled = settle( + &[ + record("5-gone", 5, Some(vec![id('a')])), + record("7-kept", 7, Some(vec![id('a'), id('b')])), + record("11-empty", 11, Some(Vec::new())), + record("13-bad", 13, None), + record(&format!("{}-max", u64::MAX), u64::MAX, Some(vec![id('c')])), + ], + &existing, + ); assert_eq!( - records, + settled, FreedRecords { bytes: u64::MAX, counted: Box::new([ - Path::new(FREED_PATH).join("5-a").into(), - Path::new(FREED_PATH).join(format!("{}-b", u64::MAX)).into(), + Path::new(FREED_PATH).join("5-gone").into(), + Path::new(FREED_PATH).join("11-empty").into(), + Path::new(FREED_PATH) + .join(format!("{}-max", u64::MAX)) + .into(), ]), } ); } + #[test] + fn the_content_of_a_record_holds_one_snapshot_id_on_each_line() { + let id = |digit: char| std::iter::repeat_n(digit, 64).collect::(); + let ids = [id('a'), id('b')]; + + assert_eq!( + [ + parse_record(record_content(&ids).as_bytes()), + parse_record(b""), + parse_record(b"not an id"), + parse_record(&[0xff, 0xfe]), + ], + [ + Some(Box::new(ids.clone()) as Box<[String]>), + Some(Box::new([]) as Box<[String]>), + None, + None + ] + ); + } + #[test] async fn a_record_of_freed_bytes_is_written_and_listed() { let files = new_files(); - record_freed(&files, 40).await.unwrap(); - record_freed(&files, 2).await.unwrap(); + record_freed(&files, 40, &[]).await.unwrap(); + record_freed(&files, 2, &[]).await.unwrap(); let listed = list_freed(&files).await.unwrap(); assert_eq!((listed.bytes, listed.counted.len()), (42, 2)); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 6930835042..b79e35bb9a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -590,7 +590,13 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { // earlier. let files = self.files(scope, &token); if freed > 0 { - record_freed(&files, freed).await.map_err(storage_failure)?; + let snapshots = ids + .iter() + .map(|id| id.to_hex().to_string()) + .collect::>(); + record_freed(&files, freed, &snapshots) + .await + .map_err(storage_failure)?; } self.blocking(Operation::Repository, move || { repository.delete_snapshots(&ids)?; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 97bd67c12a..760ef314eb 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -19,7 +19,7 @@ use super::super::files::SnapshotFiles; use super::super::prune::{ - CLOCK_SKEW_MARGIN, LEDGERS_PATH, Percent, PruneLedger, count_freed, read_ledger, + CLOCK_SKEW_MARGIN, LEDGERS_PATH, Percent, PruneLedger, parse_freed, read_ledger, }; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; @@ -169,7 +169,7 @@ async fn ledger(storage: &Arc, scope: &SnapshotScop .unwrap() } -/// Gives the sum of the records of freed bytes of the scope. +/// Gives the sum of the bytes in the names of the records of freed bytes of the scope. async fn freed(storage: &Arc, scope: &SnapshotScope) -> u64 { let listed = storage .list_blobs_below( @@ -180,7 +180,10 @@ async fn freed(storage: &Arc, scope: &SnapshotScope ) .await .unwrap(); - count_freed(&listed).bytes + listed + .iter() + .filter_map(|blob| parse_freed(blob.path.file_name()?.to_str()?)) + .fold(0, u64::saturating_add) } /// Waits until the condition holds, or until the limit ends. Gives whether the condition holds. @@ -801,7 +804,7 @@ async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_per store.delete(&scope, &name("p-deleted")).await.unwrap(); let after_first = ledger(&storage, &scope).await; let after_first_freed = freed(&storage, &scope).await; - // The margin for clock skew keeps the next prune back, so the ledger moves back by it. + // The margin for clock skew keeps the next prune back, so the ledger moves back. age_ledger(&storage, &scope, &after_first).await; store.delete(&scope, &name("p-none")).await.unwrap(); let packs_after = blobs(&*storage, &scope.0, "data/").await; @@ -856,17 +859,17 @@ async fn set_last_prune( .await; } -/// Makes the ledger one entry with the marked packs of `ledger` and a time before it by the margin -/// for clock skew. +/// Makes the ledger one entry with the marked packs of `ledger` and a time before it by two hours, +/// which is more than the grace period of each test and the margin for clock skew. async fn age_ledger( storage: &Arc, scope: &SnapshotScope, ledger: &PruneLedger, ) { - let margin = u64::try_from(CLOCK_SKEW_MARGIN.as_millis()).unwrap(); + let back = u64::try_from(CLOCK_SKEW_MARGIN.as_millis()).unwrap() + 2 * 3_600_000; let aged = ledger .last_prune - .map_or(0, |last| last.to_millis().saturating_sub(margin + 1)); + .map_or(0, |last| last.to_millis().saturating_sub(back)); storage .delete_dir("test", "test", scope.0.clone(), Path::new(LEDGERS_PATH)) .await @@ -1459,6 +1462,193 @@ async fn a_forget_that_fails_after_the_record_write_leaves_the_record() { ); } +/// Tells whether the call is the forget of a delete: the delete of a snapshot file. +fn is_forget(op_label: &str, path: &Path) -> bool { + op_label == "delete" && path.starts_with("snapshots") +} + +/// Gives the path of each record of freed bytes of the scope. +async fn records(storage: &InMemoryBlobStorage, scope: &SnapshotScope) -> Vec { + blobs(storage, &scope.0, "golem/prune-freed/").await +} + +#[test] +#[timeout("60s")] +async fn a_prune_keeps_the_record_of_a_delete_that_has_not_forgotten_its_snapshot() { + // The first delete writes its record and waits at its forget. The second delete prunes and + // must not count that record. After the forget, the next due prune counts it. + let inner = Arc::new(InMemoryBlobStorage::new()); + let plain = store( + inner.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&plain, &scope, &["p-1", "p-2"]).await; + let held = ScriptedBlobStorage::new(inner.clone(), |op_label, path| { + if is_forget(op_label, path) { + Script::WaitForGate + } else { + Script::Pass + } + }); + let pausing = store( + held.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let paused = tokio::spawn({ + let scope = scope.clone(); + async move { pausing.delete(&scope, &name("p-1")).await } + }); + let at_forget = eventually(|| { + held.calls() + .iter() + .any(|(op_label, path)| is_forget(op_label, Path::new(path))) + }) + .await; + let paused_record = records(&inner, &scope).await; + + plain.delete(&scope, &name("p-2")).await.unwrap(); + let after_prune = records(&inner, &scope).await; + let pruned = ledger(&inner, &scope).await; + age_ledger(&inner, &scope, &pruned).await; + held.open_gate(); + let resumed = tokio::time::timeout(LIMIT, paused).await; + + assert!(matches!(resumed, Ok(Ok(Ok(())))), "{resumed:?}"); + assert_eq!( + ( + at_forget, + paused_record.len(), + after_prune == paused_record, + pruned.last_prune.is_some(), + records(&inner, &scope).await, + ledger(&inner, &scope).await.last_prune > pruned.last_prune, + ), + (true, 1, true, true, Vec::::new(), true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_forget_that_lands_while_a_prune_runs_keeps_its_record_for_the_next_prune() { + // The prune of the second delete checks the records before the first delete forgets. The + // forget lands while that prune runs, so the record was not settled at the check, and it must + // stay for the next prune. + let inner = Arc::new(InMemoryBlobStorage::new()); + let plain = store( + inner.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&plain, &scope, &["p-1", "p-2", "p-3"]).await; + let forgetting = ScriptedBlobStorage::new(inner.clone(), |op_label, path| { + if is_forget(op_label, path) { + Script::Step { refuse: false } + } else { + Script::Pass + } + }); + let claimed = Arc::new(AtomicBool::new(false)); + let pruning = ScriptedBlobStorage::new(inner.clone(), { + let claimed = claimed.clone(); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" && path == Path::new("data") && claimed.load(Ordering::SeqCst) { + Script::Step { refuse: false } + } else { + Script::Pass + } + } + }); + let first = tokio::spawn({ + let deleting = store( + forgetting.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = scope.clone(); + async move { deleting.delete(&scope, &name("p-1")).await } + }); + let first_at_forget = eventually(|| forgetting.waiting_steps() > 0).await; + let first_record = records(&inner, &scope).await; + let second = tokio::spawn({ + let deleting = store( + pruning.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = scope.clone(); + async move { deleting.delete(&scope, &name("p-2")).await } + }); + let second_at_prune = eventually(|| pruning.waiting_steps() > 0).await; + + forgetting.step(); + let first = tokio::time::timeout(LIMIT, first).await; + pruning.step(); + let second = tokio::time::timeout(LIMIT, second).await; + let after_prune = records(&inner, &scope).await; + let pruned = ledger(&inner, &scope).await; + age_ledger(&inner, &scope, &pruned).await; + plain.delete(&scope, &name("p-3")).await.unwrap(); + + assert!(matches!(first, Ok(Ok(Ok(())))), "{first:?}"); + assert!(matches!(second, Ok(Ok(Ok(())))), "{second:?}"); + assert_eq!( + ( + first_at_forget, + second_at_prune, + first_record.len(), + after_prune == first_record, + pruned.last_prune.is_some(), + records(&inner, &scope).await, + ), + (true, true, 1, true, true, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_record_whose_snapshot_still_exists_counts_nothing_and_does_not_make_a_prune_due() { + let inner = Arc::new(InMemoryBlobStorage::new()); + let store = store( + inner.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1"]).await; + let snapshot = snapshot_files(inner.clone(), &scope) + .await + .first() + .map(|snapshot| snapshot.id.to_hex().to_string()) + .unwrap_or_default(); + inner + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-freed/1000000000-kept"), + snapshot.as_bytes(), + ) + .await + .unwrap(); + + store.delete(&scope, &name("p-unknown")).await.unwrap(); + store.delete(&scope, &name("p-unknown")).await.unwrap(); + + assert_eq!( + ( + snapshot.len(), + ledger(&inner, &scope).await.last_prune, + records(&inner, &scope).await, + ), + ( + 64, + None, + vec!["golem/prune-freed/1000000000-kept".to_string()] + ) + ); +} + #[test] #[timeout("60s")] async fn two_deletes_that_read_the_same_ledger_make_one_prune() { @@ -3179,6 +3369,8 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { "delete_ledger", "refresh_claim", "list_claim_directories", + "read_freed", + "list_snapshots", ] .contains(&op_label.as_str()) }) From 1522ca60e325d2bdd670f5a0899624ee2de875b7 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 22:43:31 -0700 Subject: [PATCH 064/126] Run the forget and the restore of the store in rayon pools of their own --- .../filesystem_snapshot/rustic/priority.rs | 43 +++++++++--- .../src/filesystem_snapshot/rustic/store.rs | 34 +++++++--- .../filesystem_snapshot/rustic/store/tests.rs | 67 +++++++++++++++++++ 3 files changed, 123 insertions(+), 21 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs index 71be890c12..d7b9dbf966 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs @@ -81,16 +81,7 @@ impl LowPriority { "Failed to lower the CPU priority of filesystem snapshot work, so it runs at the normal priority" ); } - match (self.build_pool)(name, self.threads) { - Ok(pool) => pool.install(|| run_taken(&slot)), - Err(error) => { - warn!( - error = %error, - "Failed to build the thread pool of filesystem snapshot work, so its parallel parts use the global pool" - ); - run_taken(&slot) - } - } + self.in_own_pool(name, || run_taken(&slot)) } }); match spawned { @@ -108,6 +99,38 @@ impl LowPriority { } } +impl LowPriority { + /// Runs the work at the normal priority on the calling thread, inside a new rayon pool with + /// the name, and gives its result. So the rayon work of one operation does not wait for the + /// rayon work of another operation on the global pool. + pub(super) fn run_at_normal_priority( + self, + name: &'static str, + work: impl FnOnce() -> anyhow::Result + Send, + ) -> anyhow::Result { + self.in_own_pool(name, work) + } + + /// Runs the work inside a new rayon pool. A pool that does not build gives a warning, and the + /// work runs without it. + fn in_own_pool( + self, + name: &'static str, + work: impl FnOnce() -> anyhow::Result + Send, + ) -> anyhow::Result { + match (self.build_pool)(name, self.threads) { + Ok(pool) => pool.install(work), + Err(error) => { + warn!( + error = %error, + "Failed to build the thread pool of filesystem snapshot work, so its parallel parts use the global pool" + ); + work() + } + } + } +} + /// Takes the work out of the slot and runs it. fn run_taken anyhow::Result>(slot: &Mutex>) -> anyhow::Result { let work = slot.lock().unwrap_or_else(PoisonError::into_inner).take(); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index b79e35bb9a..092588db9b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -503,17 +503,25 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { let options = store_restore_options(&self.policy); let name = name.clone(); let into: Box = into.into(); + // The index load of a restore uses rayon, so the restore runs in a pool of its own, with the + // reader threads of a restore. + let pool = LowPriority { + threads: Some(self.policy.restore_reader_threads), + ..self.low_priority + }; self.blocking(Operation::Restore, move || { - let Some(repository) = open_existing(backend, &key)? else { - return Ok(Lookup::Missing); - }; - match lookup(scope_snapshots(&repository)?, &name) { - Lookup::Found(snapshot, info) => { - restore_snapshot(repository, &snapshot, &into, &options)?; - Ok(Lookup::Found(snapshot, info)) + pool.run_at_normal_priority("fs-snap-restore", move || { + let Some(repository) = open_existing(backend, &key)? else { + return Ok(Lookup::Missing); + }; + match lookup(scope_snapshots(&repository)?, &name) { + Lookup::Found(snapshot, info) => { + restore_snapshot(repository, &snapshot, &into, &options)?; + Ok(Lookup::Found(snapshot, info)) + } + other => Ok(other), } - other => Ok(other), - } + }) }) .await? .into_info()? @@ -598,9 +606,13 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { .await .map_err(storage_failure)?; } + // The forget deletes the snapshot files with rayon, so it runs in a pool of its own. + let pool = self.low_priority; self.blocking(Operation::Repository, move || { - repository.delete_snapshots(&ids)?; - Ok(()) + pool.run_at_normal_priority("fs-snap-delete", move || { + repository.delete_snapshots(&ids)?; + Ok(()) + }) }) .await?; self.prune_when_due(scope, &token).await diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 760ef314eb..35be37af04 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -3277,6 +3277,73 @@ fn taken_calls(calls: &NiceCalls) -> Vec<(String, String, i32)> { .collect() } +/// Takes the recorded calls with the operation label and a path below the directory, as the path +/// and the name of the thread of each call. +#[cfg(target_os = "linux")] +fn taken_threads(calls: &NiceCalls, op_label: &str, directory: &str) -> Vec<(String, String)> { + std::mem::take( + &mut *calls + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner), + ) + .into_iter() + .filter(|(op, path, _, _)| op == op_label && path.starts_with(directory)) + .map(|(_, path, thread, _)| (path, thread)) + .collect() +} + +#[cfg(target_os = "linux")] +#[test] +async fn the_forget_of_a_delete_runs_its_storage_calls_in_a_rayon_pool_of_its_own() { + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1"]).await; + taken_calls(&calls); + + store.delete(&scope, &name("p-1")).await.unwrap(); + let forgets = taken_threads(&calls, "delete", "snapshots/"); + + assert_eq!( + ( + forgets.len(), + forgets + .iter() + .filter(|(_, thread)| !thread.starts_with("fs-snap-delete-")) + .collect::>() + ), + (1, Vec::<&(String, String)>::new()) + ); +} + +#[cfg(target_os = "linux")] +#[test] +async fn the_index_load_of_a_restore_runs_its_storage_calls_in_a_rayon_pool_of_its_own() { + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1"]).await; + taken_calls(&calls); + + let into = Scratch::new(); + store + .restore(&scope, &name("p-1"), into.path()) + .await + .unwrap(); + let index_reads = taken_threads(&calls, "read", "index/"); + + assert_eq!( + ( + index_reads.is_empty(), + index_reads + .iter() + .filter(|(_, thread)| !thread.starts_with("fs-snap-restore-")) + .collect::>() + ), + (false, Vec::<&(String, String)>::new()) + ); +} + /// Gives the operation labels of the calls, and each call that does not run at nice 19. #[cfg(target_os = "linux")] fn labels_and_calls_not_at_nice_19( From 894fc7e9b194e6b8ecca396521dacc2df6089a38 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 22:56:14 -0700 Subject: [PATCH 065/126] Sweep the orders of two deletes with the forget, late calls and the rule that no freed bytes are lost --- .../filesystem_snapshot/rustic/scripted.rs | 118 +++++- .../filesystem_snapshot/rustic/store/tests.rs | 29 +- .../rustic/store/tests/sweep.rs | 362 ++++++++++++------ 3 files changed, 375 insertions(+), 134 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs index 9215eb0eed..e8f3a33769 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -49,8 +49,9 @@ pub(super) enum Script { /// passes. Vanish, /// Waits until the test gives the storage one step, and then passes the call, or refuses it - /// when `refuse` is true. - Step { refuse: bool }, + /// when `refuse` is true. A `late` write or delete gives an error at its step, as a call that + /// got no answer within its deadline, and it reaches the storage when the test lands it. + Step { refuse: bool, late: bool }, } /// A rule that gives the script of a call from its operation label and its path. @@ -69,6 +70,10 @@ pub(super) struct ScriptedBlobStorage { waiting: AtomicUsize, /// The calls that took a step and ended. stepped: AtomicUsize, + /// The landings that the test gave and that no late call took yet. + landings: Arc, + /// The late calls that reached the storage. + landed: Arc, } impl ScriptedBlobStorage { @@ -84,6 +89,8 @@ impl ScriptedBlobStorage { steps: Semaphore::new(0), waiting: AtomicUsize::new(0), stepped: AtomicUsize::new(0), + landings: Arc::new(Semaphore::new(0)), + landed: Arc::new(AtomicUsize::new(0)), }) } @@ -107,6 +114,45 @@ impl ScriptedBlobStorage { self.stepped.load(Ordering::SeqCst) } + /// Lets one late call, now or later, reach the storage. + pub(super) fn land_late(&self) { + self.landings.add_permits(1); + } + + /// Gives the number of late calls that reached the storage. + pub(super) fn landed(&self) -> usize { + self.landed.load(Ordering::SeqCst) + } + + /// Takes one step for a late call: the caller gets an error, and the call reaches the storage + /// in a task when the test lands it. + async fn late( + &self, + op_label: &'static str, + path: &Path, + landing: impl Future> + Send + 'static, + ) -> anyhow::Result { + self.record(op_label, path); + self.waiting.fetch_add(1, Ordering::SeqCst); + let permit = self.steps.acquire().await; + self.waiting.fetch_sub(1, Ordering::SeqCst); + if let Ok(permit) = permit { + permit.forget(); + } + let (landings, landed) = (self.landings.clone(), self.landed.clone()); + tokio::spawn(async move { + if let Ok(permit) = landings.acquire().await { + permit.forget(); + } + let _ = landing.await; + landed.fetch_add(1, Ordering::SeqCst); + }); + self.stepped.fetch_add(1, Ordering::SeqCst); + Err(anyhow::anyhow!( + "the call got no answer within its deadline" + )) + } + /// Gives the operation label and the path of each call, in the order of the calls. pub(super) fn calls(&self) -> Vec<(&'static str, String)> { self.calls @@ -159,7 +205,7 @@ impl ScriptedBlobStorage { call.await } Script::Vanish => call.await, - Script::Step { refuse } => { + Script::Step { refuse, .. } => { self.waiting.fetch_add(1, Ordering::SeqCst); let permit = self.steps.acquire().await; self.waiting.fetch_sub(1, Ordering::SeqCst); @@ -268,7 +314,20 @@ impl BlobStorage for ScriptedBlobStorage { path: &Path, data: &[u8], ) -> anyhow::Result<()> { - self.answer( + let script = (self.rule)(op_label, path); + if let Script::Step { late: true, .. } = script { + let (inner, owned_path, owned_data) = + (self.inner.clone(), path.to_path_buf(), data.to_vec()); + return self + .late(op_label, path, async move { + inner + .put_raw(target_label, op_label, namespace, &owned_path, &owned_data) + .await + }) + .await; + } + self.follow( + script, op_label, path, self.inner @@ -285,7 +344,27 @@ impl BlobStorage for ScriptedBlobStorage { path: &Path, data: &[u8], ) -> anyhow::Result { - self.answer( + let script = (self.rule)(op_label, path); + if let Script::Step { late: true, .. } = script { + let (inner, owned_path, owned_data) = + (self.inner.clone(), path.to_path_buf(), data.to_vec()); + return self + .late(op_label, path, async move { + inner + .put_raw_if_absent( + target_label, + op_label, + namespace, + &owned_path, + &owned_data, + ) + .await + .map(|_| ()) + }) + .await; + } + self.follow( + script, op_label, path, self.inner @@ -318,7 +397,19 @@ impl BlobStorage for ScriptedBlobStorage { namespace: BlobStorageNamespace, path: &Path, ) -> anyhow::Result<()> { - self.answer( + let script = (self.rule)(op_label, path); + if let Script::Step { late: true, .. } = script { + let (inner, owned_path) = (self.inner.clone(), path.to_path_buf()); + return self + .late(op_label, path, async move { + inner + .delete(target_label, op_label, namespace, &owned_path) + .await + }) + .await; + } + self.follow( + script, op_label, path, self.inner.delete(target_label, op_label, namespace, path), @@ -380,7 +471,20 @@ impl BlobStorage for ScriptedBlobStorage { namespace: BlobStorageNamespace, path: &Path, ) -> anyhow::Result { - self.answer( + let script = (self.rule)(op_label, path); + if let Script::Step { late: true, .. } = script { + let (inner, owned_path) = (self.inner.clone(), path.to_path_buf()); + return self + .late(op_label, path, async move { + inner + .delete_dir(target_label, op_label, namespace, &owned_path) + .await + .map(|_| ()) + }) + .await; + } + self.follow( + script, op_label, path, self.inner diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 35be37af04..985433f7de 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1543,7 +1543,10 @@ async fn a_forget_that_lands_while_a_prune_runs_keeps_its_record_for_the_next_pr save_each(&plain, &scope, &["p-1", "p-2", "p-3"]).await; let forgetting = ScriptedBlobStorage::new(inner.clone(), |op_label, path| { if is_forget(op_label, path) { - Script::Step { refuse: false } + Script::Step { + refuse: false, + late: false, + } } else { Script::Pass } @@ -1556,7 +1559,10 @@ async fn a_forget_that_lands_while_a_prune_runs_keeps_its_record_for_the_next_pr claimed.store(true, Ordering::SeqCst); } if op_label == "list" && path == Path::new("data") && claimed.load(Ordering::SeqCst) { - Script::Step { refuse: false } + Script::Step { + refuse: false, + late: false, + } } else { Script::Pass } @@ -3644,9 +3650,9 @@ async fn the_global_rayon_pool_keeps_the_nice_value_of_the_process_after_saves_w mod sweep; -/// The largest number of steps of one turn. A delete takes at most 13 steps of the protocol, so a -/// turn of 14 steps runs a delete to its end. -const SWEEP_TURN: usize = 14; +/// The largest number of steps of one turn. A delete takes at most 15 steps of the protocol, so a +/// turn of 16 steps runs a delete to its end. +const SWEEP_TURN: usize = 16; /// The number of random orders that the property test tries. const SWEEP_CASES: u32 = 1000; @@ -3676,6 +3682,7 @@ async fn two_deletes_make_at_most_one_prune_in_each_order_with_up_to_two_switche first, turns: vec![one, two], fail: None, + late: None, }) }) }); @@ -3696,8 +3703,8 @@ async fn two_deletes_make_at_most_one_prune_in_each_order_with_up_to_two_switche #[test] #[timeout("60s")] async fn two_deletes_make_at_most_one_prune_in_random_orders_with_a_failed_call() { - // The test generates the order of the steps and a call that fails, and shrinks a failing case - // to the shortest order. The seed is fixed, and no file keeps a failing case. + // The test generates the order of the steps, a call that fails, and a call that gets no answer + // and reaches the storage later, and shrinks a failing case to the shortest order. The seed is fixed, and no file keeps a failing case. use proptest::prelude::{Strategy, prop}; use proptest::test_runner::{Config, RngAlgorithm, TestRng, TestRunner}; let (shared, prepared) = prepared_scope().await; @@ -3706,8 +3713,14 @@ async fn two_deletes_make_at_most_one_prune_in_random_orders_with_a_failed_call( 0usize..2, prop::collection::vec(0usize..=SWEEP_TURN, 0..=8), prop::option::of((0usize..2, 0usize..SWEEP_TURN)), + prop::option::of((0usize..2, 0usize..SWEEP_TURN, 0usize..8)), ) - .prop_map(|(first, turns, fail)| sweep::Schedule { first, turns, fail }); + .prop_map(|(first, turns, fail, late)| sweep::Schedule { + first, + turns, + fail, + late, + }); let started = std::time::Instant::now(); let outcome = tokio::task::spawn_blocking(move || { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs index c775ecabab..b17dbdcdc6 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs @@ -24,9 +24,12 @@ use futures::TryStreamExt; use tokio::task::JoinHandle; /// The operation labels of the blob calls of the prune protocol. The listing of the packs by a -/// prune is a step too, and it is the start of the prune. +/// prune is a step too, and it is the start of the prune. The delete of a snapshot file is the +/// forget of a delete. const STEP_LABELS: &[&str] = &[ "write_freed", + "read_freed", + "list_snapshots", "read_ledger", "list_freed", "list_data", @@ -47,7 +50,18 @@ const STEP_LABELS: &[&str] = &[ const MOST_STEPS: usize = 64; fn is_step(op_label: &str, path: &Path) -> bool { - STEP_LABELS.contains(&op_label) || (op_label == "list" && path == Path::new("data")) + STEP_LABELS.contains(&op_label) + || (op_label == "list" && path == Path::new("data")) + || is_forget(op_label, path) +} + +/// Tells whether the call writes or deletes, so that it can reach the storage late. +fn can_be_late(op_label: &str) -> bool { + op_label.starts_with("write_") || op_label.starts_with("delete") || op_label == "refresh_claim" +} + +fn is_forget(op_label: &str, path: &Path) -> bool { + op_label == "delete" && path.starts_with("snapshots") } fn is_prune_start(op_label: &str, path: &str) -> bool { @@ -56,21 +70,28 @@ fn is_prune_start(op_label: &str, path: &str) -> bool { /// The order of the steps of one case: the delete that goes first, and the numbers of steps that /// the deletes take in turn. After the listed turns, the delete whose turn is next runs to its -/// end, and then the other one does. `fail` refuses one step of one delete. +/// end, and then the other one does. `fail` refuses one step of one delete. `late` makes one +/// write or delete of one delete give no answer at its step, and reach the storage after the +/// given number of further steps of the case. #[derive(Clone, Debug)] pub(super) struct Schedule { pub(super) first: usize, pub(super) turns: Vec, pub(super) fail: Option<(usize, usize)>, + pub(super) late: Option<(usize, usize, usize)>, } -/// One step that a delete took, in the order of all steps of the case. +/// One step that a delete took, or a late call that reached the storage, in the order of all +/// steps of the case. #[derive(Clone, Debug)] struct Step { delete: usize, op_label: &'static str, path: String, - refused: bool, + /// The call gave an error to its caller: it was refused, or it was late. + failed: bool, + /// The call reached the storage: a step that passed, or the landing of a late call. + effect: bool, } /// One delete of the case: its scripted storage and its task. @@ -86,6 +107,16 @@ impl Delete { } } +/// The state of a case while it runs. +struct Case { + deletes: [Delete; 2], + log: Vec, + schedule: Schedule, + /// A late call that has not reached the storage: its delete, the log length at which it + /// lands, and its step. + pending: Option<(usize, usize, Step)>, +} + /// Waits until the condition holds, and gives false when it does not hold within [`LIMIT`]. The /// wait yields to the runtime between two checks, so a step ends as soon as it can. async fn until(condition: impl Fn() -> bool) -> bool { @@ -105,71 +136,108 @@ async fn settle(delete: &Delete) -> bool { until(|| delete.storage.waiting_steps() > 0 || delete.finished()).await } -/// Gives the delete one step and waits until the step ended and the delete waits again or ended. -async fn take_step( - deletes: &mut [Delete; 2], - who: usize, - fail: Option<(usize, usize)>, - log: &mut Vec, -) -> Result<(), String> { - let delete = &mut deletes[who]; - if delete.finished() { - return Ok(()); - } - let calls = delete.storage.calls(); - let (op_label, path) = calls - .iter() - .rev() - .find(|(op_label, path)| is_step(op_label, Path::new(path))) - .cloned() - .ok_or_else(|| format!("delete {who} waits for a step with no step call"))?; - let before = delete.storage.stepped(); - delete.storage.step(); - let storage = delete.storage.clone(); - if !until(|| storage.stepped() > before).await { - return Err(format!( - "the step {op_label} {path} of delete {who} did not end" - )); - } - log.push(Step { - delete: who, - op_label, - path, - refused: fail == Some((who, delete.taken)), - }); - delete.taken += 1; - if delete.taken > MOST_STEPS { - return Err(format!("delete {who} took more than {MOST_STEPS} steps")); - } - if settle(delete).await { +impl Case { + /// Lands the pending late call when its time came, or when `now` is true. + async fn land(&mut self, now: bool) -> Result<(), String> { + let due = self + .pending + .as_ref() + .is_some_and(|(_, at, _)| now || self.log.len() >= *at); + if !due { + return Ok(()); + } + let Some((who, _, step)) = self.pending.take() else { + return Ok(()); + }; + let storage = self.deletes[who].storage.clone(); + let before = storage.landed(); + storage.land_late(); + if !until(|| storage.landed() > before).await { + return Err(format!( + "the late call {} {} did not land", + step.op_label, step.path + )); + } + self.log.push(Step { + failed: false, + effect: true, + ..step + }); Ok(()) - } else { - Err(format!("delete {who} did not reach its next step")) } -} -/// Gives the delete steps until it ends. -async fn run_to_end( - deletes: &mut [Delete; 2], - who: usize, - fail: Option<(usize, usize)>, - log: &mut Vec, -) -> Result<(), String> { - futures::stream::iter(0..=MOST_STEPS) - .map(Ok) - .try_fold((deletes, log), |(deletes, log), _| async move { - take_step(deletes, who, fail, log).await?; - Ok::<_, String>((deletes, log)) - }) - .await - .map(|_| ()) + /// Gives the delete one step and waits until the step ended and the delete waits again or + /// ended. + async fn take_step(&mut self, who: usize) -> Result<(), String> { + if self.deletes[who].finished() { + return Ok(()); + } + let calls = self.deletes[who].storage.calls(); + let (op_label, path) = calls + .iter() + .rev() + .find(|(op_label, path)| is_step(op_label, Path::new(path))) + .cloned() + .ok_or_else(|| format!("delete {who} waits for a step with no step call"))?; + let number = self.deletes[who].taken; + let refused = self.schedule.fail == Some((who, number)); + let late = self.schedule.late.filter(|(late_who, late_step, _)| { + (*late_who, *late_step) == (who, number) && can_be_late(op_label) + }); + let storage = self.deletes[who].storage.clone(); + let before = storage.stepped(); + storage.step(); + if !until(|| storage.stepped() > before).await { + return Err(format!( + "the step {op_label} {path} of delete {who} did not end" + )); + } + let step = Step { + delete: who, + op_label, + path, + failed: refused || late.is_some(), + effect: !refused && late.is_none(), + }; + self.log.push(step.clone()); + if let Some((_, _, delay)) = late { + self.pending = Some((who, self.log.len() + delay, step)); + } + self.deletes[who].taken += 1; + if self.deletes[who].taken > MOST_STEPS { + return Err(format!("delete {who} took more than {MOST_STEPS} steps")); + } + self.land(false).await?; + if settle(&self.deletes[who]).await { + Ok(()) + } else { + let calls = self.deletes[who].storage.calls(); + let last = calls.iter().rev().take(6).collect::>(); + Err(format!( + "delete {who} did not reach its next step; its last calls {last:?}; the other delete waits {}", + self.deletes[(who + 1) % 2].storage.waiting_steps() + )) + } + } + + /// Gives the delete the steps. + async fn take_steps(&mut self, who: usize, steps: usize) -> Result<(), String> { + futures::stream::iter(0..steps) + .map(Ok) + .try_fold(self, |case, _| async move { + case.take_step(who).await?; + Ok::<_, String>(case) + }) + .await + .map(|_| ()) + } } /// What a case found. #[derive(Debug)] pub(super) struct Found { pub(super) prunes: usize, - pub(super) refused: bool, + pub(super) failed: bool, pub(super) steps: [usize; 2], } @@ -187,12 +255,16 @@ pub(super) async fn run_case( .map_err(|error| format!("the copy of the prepared scope failed: {error}"))?; let deletes = [0, 1].map(|who| { let counter = Arc::new(AtomicUsize::new(0)); - let fail = schedule.fail; + let (fail, late) = (schedule.fail, schedule.late); let storage = ScriptedBlobStorage::new(shared.clone(), move |op_label, path| { if is_step(op_label, path) { let number = counter.fetch_add(1, Ordering::SeqCst); Script::Step { refuse: fail == Some((who, number)), + late: can_be_late(op_label) + && late.is_some_and(|(late_who, late_step, _)| { + (late_who, late_step) == (who, number) + }), } } else { Script::Pass @@ -211,33 +283,66 @@ pub(super) async fn run_case( taken: 0, } }); - let mut deletes = deletes; - let mut log = Vec::new(); - if !(settle(&deletes[0]).await && settle(&deletes[1]).await) { + let mut case = Case { + deletes, + log: Vec::new(), + schedule: schedule.clone(), + pending: None, + }; + if !(settle(&case.deletes[0]).await && settle(&case.deletes[1]).await) { return Err("a delete did not reach its first step".to_string()); } - let (next, deletes, log) = futures::stream::iter(schedule.turns.iter().enumerate()) - .map(Ok) - .try_fold( - (schedule.first, &mut deletes, &mut log), - |(_, deletes, log), (turn, steps)| async move { - let who = (schedule.first + turn) % 2; - futures::stream::iter(0..*steps) - .map(Ok) - .try_fold((deletes, log), |(deletes, log), _| async move { - take_step(deletes, who, schedule.fail, log).await?; - Ok::<_, String>((deletes, log)) - }) - .await - .map(|(deletes, log)| ((who + 1) % 2, deletes, log)) - }, - ) - .await?; - run_to_end(deletes, next, schedule.fail, log).await?; - run_to_end(deletes, (next + 1) % 2, schedule.fail, log).await?; + let turns = schedule.turns.clone(); + let ran = run_turns(&mut case, schedule, &turns).await; + if let Err(error) = ran { + return Err(format!( + "{error}; schedule {schedule:?}; steps {}", + order(&case.log) + )); + } let results = - futures::future::join_all(deletes.iter_mut().map(|delete| &mut delete.task)).await; - check(shared, &scope, schedule, log, results).await + futures::future::join_all(case.deletes.iter_mut().map(|delete| &mut delete.task)).await; + check(shared, &scope, schedule, &case.log, results).await +} + +/// Runs the turns of the schedule, then each delete to its end, and then lands a late call that +/// did not land yet. +async fn run_turns(case: &mut Case, schedule: &Schedule, turns: &[usize]) -> Result<(), String> { + let next = futures::stream::iter(turns.iter().enumerate()) + .map(Ok) + .try_fold(&mut *case, |case, (turn, steps)| async move { + case.take_steps((schedule.first + turn) % 2, *steps).await?; + Ok::<_, String>(case) + }) + .await + .map(|_| (schedule.first + turns.len()) % 2)?; + case.take_steps(next, MOST_STEPS + 1).await?; + case.take_steps((next + 1) % 2, MOST_STEPS + 1).await?; + case.land(true).await +} + +/// Gives the steps of the log as text: the delete and the operation label, with `!` for a call +/// that gave an error. +fn order(log: &[Step]) -> String { + log.iter() + .map(|step| { + let mark = match (step.failed, step.effect) { + (true, _) => "!", + (false, true) => "", + (false, false) => "?", + }; + format!("{}:{}{}", step.delete, step.op_label, mark) + }) + .collect::>() + .join(" ") +} + +/// Gives the index in the log at which the delete forgot its snapshot, when its forget reached +/// the storage. +fn forgotten_at(log: &[Step], who: usize) -> Option { + log.iter().position(|step| { + step.delete == who && step.effect && is_forget(step.op_label, Path::new(&step.path)) + }) } /// Checks the rules on the end state of a case. @@ -248,52 +353,70 @@ async fn check( log: &[Step], results: Vec, tokio::task::JoinError>>, ) -> Result { - let refused = log.iter().any(|step| step.refused); + let failed = log.iter().any(|step| step.failed); let prune_starts = log .iter() .enumerate() - .filter(|(_, step)| !step.refused && is_prune_start(step.op_label, &step.path)) + .filter(|(_, step)| step.effect && is_prune_start(step.op_label, &step.path)) .collect::>(); let prunes = prune_starts.len(); - let order = || { - log.iter() - .map(|step| { - format!( - "{}:{}{}", - step.delete, - step.op_label, - if step.refused { "!" } else { "" } - ) - }) - .collect::>() - .join(" ") + let fail = |rule: &str| { + Err(format!( + "{rule}; schedule {schedule:?}; steps {}", + order(log) + )) }; - let fail = |rule: &str| Err(format!("{rule}; schedule {schedule:?}; steps {}", order())); if prunes > 1 { return fail("more than one prune ran"); } - if !refused && prunes != 1 { + if !failed && prunes != 1 { return fail("no prune ran, although a prune was due and no call failed"); } - if !refused && results.iter().any(|result| !matches!(result, Ok(Ok(())))) { - return fail(&format!( - "a delete failed with no refused call: {results:?}" - )); + if !failed && results.iter().any(|result| !matches!(result, Ok(Ok(())))) { + return fail(&format!("a delete failed with no failed call: {results:?}")); } let claims = blobs(&**shared, &scope.0, "golem/prune-claims/").await; - if !refused && !claims.is_empty() { + if !failed && !claims.is_empty() { return fail(&format!("claims stay: {claims:?}")); } let records = blobs(&**shared, &scope.0, "golem/prune-freed/").await; let entries = blobs(&**shared, &scope.0, "golem/prune-ledgers/").await; let ledger = ledger(shared, scope).await; + let written = |path: &str| { + log.iter() + .find(|step| step.op_label == "write_freed" && step.effect && step.path == path) + .map(|step| step.delete) + }; + let lost = log + .iter() + .enumerate() + .filter(|(_, step)| step.op_label == "delete_freed" && step.effect) + .filter_map(|(deleted_at, step)| { + let pruner = step.delete; + let decided_at = log[..deleted_at].iter().rposition(|earlier| { + earlier.delete == pruner + && matches!( + earlier.op_label, + "list_freed" | "read_freed" | "list_snapshots" + ) + })?; + let writer = written(&step.path)?; + let settled = forgotten_at(log, writer).is_some_and(|forgot| forgot < decided_at); + (!settled).then(|| step.path.clone()) + }) + .collect::>(); + if !lost.is_empty() { + return fail(&format!( + "a prune deleted a record whose snapshot was not gone at its count: {lost:?}" + )); + } if let Some((start, pruner)) = prune_starts .first() .map(|(index, step)| (*index, step.delete)) { let wrote_ledger = log[start..] .iter() - .find(|step| step.delete == pruner && step.op_label == "write_ledger" && !step.refused); + .find(|step| step.delete == pruner && step.op_label == "write_ledger" && step.effect); if let Some(written) = wrote_ledger { let written_ms = Path::new(&written.path) .file_name() @@ -306,30 +429,31 @@ async fn check( )); } } - let counted_at = log[..start] - .iter() - .rposition(|step| step.delete == pruner && step.op_label == "list_freed"); - let written_records = log + let counted_at = log[..start].iter().rposition(|step| { + step.delete == pruner + && matches!( + step.op_label, + "list_freed" | "read_freed" | "list_snapshots" + ) + }); + let gone = log .iter() .enumerate() - .filter(|(_, step)| step.op_label == "write_freed" && !step.refused) - .collect::>(); - let lost = written_records - .iter() + .filter(|(_, step)| step.op_label == "write_freed" && step.effect) .filter(|(index, _)| counted_at.is_none_or(|counted| *index > counted)) .filter(|(_, step)| !records.contains(&step.path)) .map(|(_, step)| step.path.clone()) .collect::>(); - if !lost.is_empty() { + if !gone.is_empty() { return fail(&format!( - "records that the prune did not count are gone: {lost:?}" + "records that the prune did not count are gone: {gone:?}" )); } } let steps = [0, 1].map(|who| log.iter().filter(|step| step.delete == who).count()); Ok(Found { prunes, - refused, + failed, steps, }) } From beefc8650b98c311e26a450cb30f65459aec9933 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 23:21:13 -0700 Subject: [PATCH 066/126] Write each record of freed bytes if absent, and count a record without a snapshot id as nothing --- .../src/filesystem_snapshot/rustic/prune.rs | 31 ++++++++-------- .../filesystem_snapshot/rustic/scripted.rs | 13 ++++++- .../filesystem_snapshot/rustic/store/tests.rs | 36 +++++++++++++++++-- 3 files changed, 63 insertions(+), 17 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index ec4c1245d6..a553637226 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -261,7 +261,8 @@ pub(super) fn record_content(snapshots: &[String]) -> String { } /// Reads the snapshot ids from the content of a record. Each line must be an id of 64 hex -/// characters, and an empty content has no id. +/// characters. A content without an id does not parse, because a reader can see a record that a +/// write has not filled yet. pub(super) fn parse_record(content: &[u8]) -> Option> { let text = std::str::from_utf8(content).ok()?; text.lines() @@ -270,20 +271,20 @@ pub(super) fn parse_record(content: &[u8]) -> Option> { (line.len() == 64 && line.bytes().all(|byte| byte.is_ascii_hexdigit())) .then(|| line.to_string()) }) - .collect() + .collect::>>() + .filter(|snapshots| !snapshots.is_empty()) } -/// Gives the settled records and the sum of their bytes. A record is settled when none of its -/// snapshot files exists. Any other record counts as zero bytes and stays, and so does a record -/// whose content does not parse. +/// Gives the settled records and the sum of their bytes. A record is settled when it names at +/// least one snapshot file and none of them exists. Any other record counts as zero bytes and +/// stays, and so does a record whose content does not parse. pub(super) fn settle(records: &[FreedRecord], existing: &HashSet) -> FreedRecords { let settled = records .iter() .filter(|record| { - record - .snapshots - .as_ref() - .is_some_and(|snapshots| snapshots.iter().all(|id| !existing.contains(id))) + record.snapshots.as_ref().is_some_and(|snapshots| { + !snapshots.is_empty() && snapshots.iter().all(|id| !existing.contains(id)) + }) }) .collect::>(); FreedRecords { @@ -311,9 +312,11 @@ pub(super) async fn record_freed( snapshots: &[String], ) -> anyhow::Result<()> { let path = Path::new(FREED_PATH).join(format!("{bytes}-{}", uuid::Uuid::new_v4())); + // The name is unique, so `AlreadyExists` means that an earlier try of this call wrote it. files - .put("write_freed", &path, record_content(snapshots).as_bytes()) + .put_if_absent("write_freed", &path, record_content(snapshots).as_bytes()) .await + .map(|_| ()) } /// Lists and reads the records of freed bytes, lists the snapshot files one time, and gives the @@ -982,7 +985,6 @@ mod tests { bytes: u64::MAX, counted: Box::new([ Path::new(FREED_PATH).join("5-gone").into(), - Path::new(FREED_PATH).join("11-empty").into(), Path::new(FREED_PATH) .join(format!("{}-max", u64::MAX)) .into(), @@ -1005,7 +1007,7 @@ mod tests { ], [ Some(Box::new(ids.clone()) as Box<[String]>), - Some(Box::new([]) as Box<[String]>), + None, None, None ] @@ -1016,8 +1018,9 @@ mod tests { async fn a_record_of_freed_bytes_is_written_and_listed() { let files = new_files(); - record_freed(&files, 40, &[]).await.unwrap(); - record_freed(&files, 2, &[]).await.unwrap(); + let gone = ["0".repeat(64)]; + record_freed(&files, 40, &gone).await.unwrap(); + record_freed(&files, 2, &gone).await.unwrap(); let listed = list_freed(&files).await.unwrap(); assert_eq!((listed.bytes, listed.counted.len()), (42, 2)); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs index e8f3a33769..de6f3f5355 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -48,6 +48,9 @@ pub(super) enum Script { /// Gives no blob to a read of a whole blob, as a delete after a listing does. Each other call /// passes. Vanish, + /// Passes the call, and then answers a write if absent with `AlreadyExists`, as a new try of a + /// call whose first answer was lost does. Each other call passes. + AnswerAlreadyExists, /// Waits until the test gives the storage one step, and then passes the call, or refuses it /// when `refuse` is true. A `late` write or delete gives an error at its step, as a call that /// got no answer within its deadline, and it reaches the storage when the test lands it. @@ -204,7 +207,7 @@ impl ScriptedBlobStorage { self.gate.cancelled().await; call.await } - Script::Vanish => call.await, + Script::Vanish | Script::AnswerAlreadyExists => call.await, Script::Step { refuse, .. } => { self.waiting.fetch_add(1, Ordering::SeqCst); let permit = self.steps.acquire().await; @@ -363,6 +366,14 @@ impl BlobStorage for ScriptedBlobStorage { }) .await; } + if script == Script::AnswerAlreadyExists { + self.record(op_label, path); + let _: PutIfAbsent = self + .inner + .put_raw_if_absent(target_label, op_label, namespace, path, data) + .await?; + return Ok(PutIfAbsent::AlreadyExists); + } self.follow( script, op_label, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 985433f7de..32d245854a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -51,6 +51,9 @@ use std::time::Duration; use test_r::core::DynamicTestRegistration; use test_r::{test, test_gen, timeout}; +/// The id of a snapshot file that no scope holds, as the content of a record of freed bytes. +const GONE_SNAPSHOT: &str = "0000000000000000000000000000000000000000000000000000000000000000"; + /// The longest time that a test waits for an operation or for the work of a store to end. const LIMIT: Duration = Duration::from_secs(10); @@ -843,7 +846,7 @@ async fn set_last_prune( "test", scope.0.clone(), Path::new("golem/prune-freed/1-test"), - b"", + GONE_SNAPSHOT.as_bytes(), ) .await .unwrap(); @@ -1090,7 +1093,7 @@ async fn a_record_that_a_delete_adds_during_a_prune_stays_for_the_next_prune() { "test", scope.0.clone(), Path::new("golem/prune-freed/7-late"), - b"", + GONE_SNAPSHOT.as_bytes(), ) .await .unwrap(); @@ -1655,6 +1658,35 @@ async fn a_record_whose_snapshot_still_exists_counts_nothing_and_does_not_make_a ); } +#[test] +async fn a_record_write_that_answers_already_exists_counts_as_written() { + // A new try of a record write whose first answer was lost finds the record of the first try. + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "write_freed" { + Script::AnswerAlreadyExists + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + freed(&storage, &scope).await > 0, + listed_names(&store, &scope).await + ), + (true, Vec::::new()) + ); +} + #[test] #[timeout("60s")] async fn two_deletes_that_read_the_same_ledger_make_one_prune() { From decf1ab112ecfb7c2701b3d9da9fdc629bf88069 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 23:51:14 -0700 Subject: [PATCH 067/126] Keep the times of prune claims in marker names, and keep the claim of a prune that started --- .../src/filesystem_snapshot/rustic/prune.rs | 302 ++++++++++++------ .../src/filesystem_snapshot/rustic/store.rs | 73 +++-- .../filesystem_snapshot/rustic/store/tests.rs | 229 ++++++++++--- 3 files changed, 434 insertions(+), 170 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index a553637226..7f581be483 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -372,11 +372,37 @@ pub(super) async fn remove_freed(files: &SnapshotFiles, records: &FreedRecords) .await; } -/// A claim of a prune that a listing found: its number, and its time when its content parses. +/// An entry of a claim directory that a listing found: a claim ``, or a marker +/// `@-` with the time at which a delete wrote it. #[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) struct ListedClaim { - pub(super) number: u64, - pub(super) claimed_at: Option, +pub(super) enum ClaimEntry { + Claim(u64), + Marker(u64, Timestamp), +} + +impl ClaimEntry { + fn number(self) -> u64 { + match self { + Self::Claim(number) | Self::Marker(number, _) => number, + } + } +} + +/// Reads the name of an entry of a claim directory. +pub(super) fn parse_claim_entry(name: &str) -> Option { + match name.split_once('@') { + None => name.parse().ok().map(ClaimEntry::Claim), + Some((number, rest)) => { + let (millis, unique) = rest.split_once('-')?; + if unique.is_empty() { + return None; + } + Some(ClaimEntry::Marker( + number.parse().ok()?, + Timestamp::from(millis.parse::().ok()?), + )) + } + } } /// What a delete whose prune is due does with the claims of its ledger. @@ -388,23 +414,34 @@ pub(super) enum ClaimChoice { Held, } -/// Chooses the claim of a delete from the claims of its ledger. The newest claim holds the ledger -/// until the grace period and the margin for clock skew passed since its time. A claim whose -/// content does not parse, or whose time is more than the margin after `now`, is old. -pub(super) fn next_claim(claims: &[ListedClaim], now: Timestamp, grace: Duration) -> ClaimChoice { - let held_until = grace.saturating_add(CLOCK_SKEW_MARGIN); - match claims.iter().max_by_key(|claim| claim.number) { - None => ClaimChoice::Claim(0), - Some(newest) - if newest - .claimed_at - .filter(|at| !beyond_margin(*at, now)) - .is_some_and(|at| !passed_since(at, now, held_until)) => - { - ClaimChoice::Held - } - Some(newest) => ClaimChoice::Claim(newest.number.saturating_add(1)), +/// Gives how long a marker holds the claims of its ledger: the grace period, but at least the time +/// between two refreshes and two margins for clock skew, and then one more margin. +pub(super) fn claim_hold(grace: Duration) -> Duration { + grace + .max(refresh_period(grace).saturating_add(CLOCK_SKEW_MARGIN.saturating_mul(2))) + .saturating_add(CLOCK_SKEW_MARGIN) +} + +/// Chooses the claim of a delete from the entries of the claim directory of its ledger. Any marker +/// younger than the hold holds the ledger, whatever its number. A marker whose time is more than +/// the margin after `now` counts as missing, and a claim without a young marker is old. Otherwise +/// the delete takes the number after the largest one, or 0. +pub(super) fn next_claim(entries: &[ClaimEntry], now: Timestamp, grace: Duration) -> ClaimChoice { + let hold = claim_hold(grace); + let held = entries.iter().any(|entry| match entry { + ClaimEntry::Marker(_, at) => !beyond_margin(*at, now) && !passed_since(*at, now, hold), + ClaimEntry::Claim(_) => false, + }); + if held { + return ClaimChoice::Held; } + entries + .iter() + .map(|entry| entry.number()) + .max() + .map_or(ClaimChoice::Claim(0), |largest| { + ClaimChoice::Claim(largest.saturating_add(1)) + }) } /// Gives the directory of the claims of the ledger: the time of its last prune in milliseconds, or @@ -416,53 +453,64 @@ pub(super) fn claims_directory(ledger: &PruneLedger) -> PathBuf { Path::new(CLAIMS_PATH).join(generation) } -/// Lists the claims in the directory. A name that is not a number is not a claim, and a claim -/// that a delete removed after the listing counts as old. +/// Lists the claims and the markers in the directory, from their names. A name that does not +/// parse is left out. pub(super) async fn list_claims( files: &SnapshotFiles, directory: &Path, -) -> anyhow::Result> { - let listed = files.list_below("list_claims", directory).await?; - let numbered = listed +) -> anyhow::Result> { + Ok(files + .list_below("list_claims", directory) + .await? .iter() - .filter_map(|blob| { - let number = blob.path.file_name()?.to_str()?.parse::().ok()?; - Some((number, &blob.path)) - }) - .collect::>(); - stream::iter(numbered.iter()) - .then(|(number, path)| async move { - let content = files.get("read_claim", path).await?; - Ok::<_, anyhow::Error>(ListedClaim { - number: *number, - claimed_at: content.as_deref().and_then(parse_claim), - }) - }) - .try_collect::>() - .await - .map(Vec::into_boxed_slice) + .filter_map(|blob| parse_claim_entry(blob.path.file_name()?.to_str()?)) + .collect()) +} + +/// Writes a marker of the claim with the number, with the time, and gives its path. The name is +/// unique, so `AlreadyExists` means that an earlier try of this call wrote it. +pub(super) async fn write_marker( + files: &SnapshotFiles, + op_label: &'static str, + directory: &Path, + number: u64, + time: Timestamp, +) -> anyhow::Result { + let path = directory.join(format!( + "{number}@{}-{}", + time.to_millis(), + uuid::Uuid::new_v4() + )); + let _: PutIfAbsent = files.put_if_absent(op_label, &path, &[]).await?; + Ok(path) } -/// Writes the claim with the number, and tells whether this call wrote it. +/// Writes the first marker of the claim with the number, then takes the claim, and tells whether +/// this delete holds it. A delete that loses the claim deletes its marker. pub(super) async fn take_claim( files: &SnapshotFiles, directory: &Path, number: u64, now: Timestamp, ) -> anyhow::Result { - let content = now.to_millis().to_string(); + let marker = write_marker(files, "write_marker", directory, number, now).await?; let written = files - .put_if_absent( - "write_claim", - &directory.join(number.to_string()), - content.as_bytes(), - ) + .put_if_absent("write_claim", &directory.join(number.to_string()), &[]) .await?; - Ok(written == PutIfAbsent::Written) + if written == PutIfAbsent::Written { + return Ok(true); + } + if let Err(error) = files.delete("delete_marker", &marker).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete the marker of a prune claim that a filesystem snapshot delete lost" + ); + } + Ok(false) } -/// Gives the time between two writes of a live claim: a fourth of the grace period, or a fourth of -/// the margin for clock skew when the grace period is zero. +/// Gives the time between two markers of a live claim: a fourth of the grace period, or a fourth +/// of the margin for clock skew when the grace period is zero. pub(super) fn refresh_period(grace: Duration) -> Duration { if grace.is_zero() { CLOCK_SKEW_MARGIN / 4 @@ -471,41 +519,56 @@ pub(super) fn refresh_period(grace: Duration) -> Duration { } } -/// Writes the claim with the number again, with the current time, at each period, until the -/// caller drops the future or the operation of the files is cancelled. A failed write gives a -/// warning. +/// Writes a new marker of the claim with the number at each period, until the caller drops the +/// future or the operation of the files is cancelled. A failed write gives a warning. pub(super) async fn keep_claim_fresh( files: &SnapshotFiles, directory: &Path, number: u64, period: Duration, ) { - let path = directory.join(number.to_string()); stream::repeat(()) .then(|()| tokio::time::sleep(period)) .take_until(files.cancel.cancelled()) - .for_each(|()| { - let path = &path; - async move { - let content = Timestamp::now_utc().to_millis().to_string(); - if let Err(error) = files.put("refresh_claim", path, content.as_bytes()).await { - warn!( - error = %format!("{error:#}"), - "Failed to write the prune claim of a filesystem snapshot scope again" - ); - } + .for_each(|()| async move { + let written = write_marker( + files, + "refresh_claim", + directory, + number, + Timestamp::now_utc(), + ) + .await; + if let Err(error) = written { + warn!( + error = %format!("{error:#}"), + "Failed to write a new marker of the prune claim of a filesystem snapshot scope" + ); } }) .await; } -/// Deletes the claim with the number. A failure gives a warning, because a claim only delays a -/// prune until its grace period passed. +/// Deletes the claim with the number, and then its markers. A failure gives a warning, because a +/// claim only delays a prune until its hold passed. pub(super) async fn release_claim(files: &SnapshotFiles, directory: &Path, number: u64) { - if let Err(error) = files - .delete("delete_claim", &directory.join(number.to_string())) + let released = async { + files + .delete("delete_claim", &directory.join(number.to_string())) + .await?; + let listed = files.list_below("list_claims", directory).await?; + stream::iter(listed.iter().filter(|blob| { + blob.path + .file_name() + .and_then(|name| name.to_str()) + .and_then(parse_claim_entry) + .is_some_and(|entry| matches!(entry, ClaimEntry::Marker(own, _) if own == number)) + })) + .map(Ok) + .try_for_each(|blob| files.delete("delete_marker", &blob.path)) .await - { + }; + if let Err(error) = released.await { warn!( error = %format!("{error:#}"), "Failed to delete the prune claim of a filesystem snapshot scope" @@ -554,25 +617,16 @@ pub(super) async fn remove_old_claims(files: &SnapshotFiles, keep: &Path) { .await; } -/// Reads the time of a claim, in milliseconds. -fn parse_claim(content: &[u8]) -> Option { - std::str::from_utf8(content) - .ok()? - .trim() - .parse::() - .ok() - .map(Timestamp::from) -} - #[cfg(test)] mod tests { use super::super::files::SnapshotFiles; use super::{ - CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, FREED_PATH, FreedRecord, FreedRecords, - LEDGERS_PATH, ListedClaim, Percent, PruneLedger, claim_directories_except, + CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, ClaimEntry, FREED_PATH, FreedRecord, + FreedRecords, LEDGERS_PATH, Percent, PruneLedger, claim_directories_except, claim_hold, claims_directory, keep_claim_fresh, list_claims, list_freed, needs_repository_size, - newest_ledger, next_claim, older_entries, parse_freed, parse_ledger_entry, parse_record, - prune_due, read_ledger, record_content, record_freed, settle, take_claim, write_ledger, + newest_ledger, next_claim, older_entries, parse_claim_entry, parse_freed, + parse_ledger_entry, parse_record, prune_due, read_ledger, record_content, record_freed, + settle, take_claim, write_ledger, }; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; @@ -725,10 +779,7 @@ mod tests { let due = |last| prune_due(&ledger(Some(last), false), 1, at(now), 0, Percent(0), GRACE); let claim = |claimed_at| { next_claim( - &[ListedClaim { - number: 0, - claimed_at: Some(at(claimed_at)), - }], + &[ClaimEntry::Claim(0), ClaimEntry::Marker(0, at(claimed_at))], at(now), GRACE, ) @@ -764,32 +815,72 @@ mod tests { } #[test] - fn the_newest_claim_holds_a_ledger_until_the_grace_period_passed_since_its_time() { + fn any_young_marker_holds_a_ledger_and_a_claim_without_a_marker_is_old() { let now = 10_000_000; - let claim = |number, claimed_at: Option| ListedClaim { - number, - claimed_at: claimed_at.map(Timestamp::from), - }; - let choose = |claims: &[ListedClaim]| next_claim(claims, at(now), GRACE); + let claim = ClaimEntry::Claim; + let marker = |number, millis| ClaimEntry::Marker(number, at(millis)); + let choose = |entries: &[ClaimEntry]| next_claim(entries, at(now), GRACE); assert_eq!( [ choose(&[]), - choose(&[claim(0, Some(now - HELD_MILLIS)), claim(1, Some(now - 1))]), - choose(&[claim(1, Some(now)), claim(2, Some(now - HELD_MILLIS))]), - choose(&[claim(4, None)]), - choose(&[claim(0, Some(now - 1)), claim(3, None)]), + choose(&[claim(0), claim(1), marker(0, now - 1)]), + choose(&[claim(0), claim(1), marker(1, now - HELD_MILLIS)]), + choose(&[claim(4)]), + choose(&[marker(2, now - 1)]), + choose(&[claim(0), claim(3), marker(0, now - HELD_MILLIS)]), ], [ ClaimChoice::Claim(0), ClaimChoice::Held, - ClaimChoice::Claim(3), + ClaimChoice::Claim(2), ClaimChoice::Claim(5), + ClaimChoice::Held, ClaimChoice::Claim(4), ] ); } + #[test] + fn the_hold_is_at_least_the_refresh_period_and_two_margins_and_then_one_margin() { + let margin = CLOCK_SKEW_MARGIN; + + assert_eq!( + [ + claim_hold(GRACE), + claim_hold(Duration::from_secs(60)), + claim_hold(Duration::ZERO), + ], + [ + GRACE + margin, + Duration::from_secs(15) + margin * 3, + margin / 4 + margin * 3, + ] + ); + } + + #[test] + fn a_claim_entry_name_is_a_number_or_a_number_with_a_time_and_a_unique_part() { + assert_eq!( + [ + parse_claim_entry("3"), + parse_claim_entry("3@42-a"), + parse_claim_entry("3@42-a-b"), + parse_claim_entry("3@42-"), + parse_claim_entry("3@x-a"), + parse_claim_entry("x"), + ], + [ + Some(ClaimEntry::Claim(3)), + Some(ClaimEntry::Marker(3, at(42))), + Some(ClaimEntry::Marker(3, at(42))), + None, + None, + None, + ] + ); + } + #[test] async fn the_refresh_of_a_claim_ends_when_its_operation_is_cancelled() { let files = new_files(); @@ -811,7 +902,7 @@ mod tests { } #[test] - async fn a_claim_is_taken_one_time_and_listed_with_its_time() { + async fn a_claim_is_taken_after_its_marker_and_a_loser_deletes_its_marker() { let files = new_files(); let directory = claims_directory(&ledger(Some(42), false)); let now = at(10_000_000); @@ -827,16 +918,17 @@ mod tests { directory.display().to_string(), first, again, - listed.to_vec() + listed.len(), + listed.contains(&ClaimEntry::Claim(0)), + listed.contains(&ClaimEntry::Marker(0, now)), ), ( "golem/prune-claims/42".to_string(), true, false, - vec![ListedClaim { - number: 0, - claimed_at: Some(now) - }] + 2, + true, + true ) ); } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 092588db9b..469ef700f5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -28,7 +28,7 @@ use super::prune::{ ClaimChoice, Percent, PruneLedger, claims_directory, keep_claim_fresh, list_claims, list_freed, needs_repository_size, next_claim, prune_due, read_ledger, record_freed, refresh_period, release_claim, remove_freed, remove_old_claims, remove_older_ledgers, repository_bytes, - take_claim, write_ledger, + take_claim, write_ledger, write_marker, }; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -63,6 +63,7 @@ use std::time::Duration; use tokio::runtime::Handle; use tokio_util::sync::{CancellationToken, DropGuard}; use tokio_util::task::TaskTracker; +use tracing::warn; /// The share of the size of the repository that deleted snapshots must free before a delete prunes /// the scope. The threshold is 10% of the size, rounded down to a whole byte, so about 10%. @@ -187,14 +188,6 @@ struct Claim<'a> { number: u64, } -/// How the work of a delete that holds a claim ended without an error. -enum ClaimOutcome { - /// The prune ran, and it marked packs or not. - Pruned { marked_packs: bool }, - /// Another prune ended after the claim, so this delete did not prune. - Superseded, -} - /// The error of an operation of a store that is shut down. fn shut_down_error() -> SnapshotStoreError { SnapshotStoreError::Storage { @@ -374,19 +367,33 @@ impl RusticSnapshotStore { directory: &claims, number, }; - let outcome = self - .prune_with_claim(scope, token, &files, &claim, grace) - .await; - // The claim is released unless the prune ran, so a retry of the delete prunes again. After - // a prune, the claim stays also when the ledger write fails, so no second prune runs close - // to the first one. - let marked_packs = match outcome { - Ok(ClaimOutcome::Pruned { marked_packs }) => marked_packs, + // Only an error before the prune starts releases the claim, so a retry of the delete + // prunes again. A prune that started can have marked packs, so its claim stays. + let backend = match self.prepare_prune(scope, token, &files, &claim).await { + Ok(Some(backend)) => backend, other => { release_claim(&files, claim.directory, claim.number).await; return other.map(|_| ()); } }; + let pruned = self.run_prune(backend, &files, &claim, grace).await; + // The final marker holds the claim for the grace period from the end of this prune, also + // when the prune or its ledger write fails. + if let Err(error) = write_marker( + &files, + "final_marker", + claim.directory, + claim.number, + Timestamp::now_utc(), + ) + .await + { + warn!( + error = %format!("{error:#}"), + "Failed to write the final marker of the prune claim of a filesystem snapshot scope" + ); + } + let marked_packs = pruned?; let ended = Timestamp::now_utc(); write_ledger(&files, ended, marked_packs) .await @@ -401,31 +408,41 @@ impl RusticSnapshotStore { Ok(()) } - /// Prunes while the delete holds the claim. It gives `Superseded` when another prune ended after - /// the claim, and an error when no prune ran. - async fn prune_with_claim( + /// Checks the ledger again and builds the backend of the prune. It gives `None` when another + /// prune ended after the claim. + async fn prepare_prune( &self, scope: &SnapshotScope, token: &CancellationToken, files: &SnapshotFiles, claim: &Claim<'_>, - grace: Duration, - ) -> Result { + ) -> Result>, SnapshotStoreError> { // A prune writes its ledger before it deletes the claims, so a delete that claims in a // directory that such a prune removed sees the new ledger here. let again = read_ledger(files).await.map_err(storage_failure)?; if claims_directory(&again) != claim.directory { - return Ok(ClaimOutcome::Superseded); + return Ok(None); } - let backend = Arc::new(self.backend(scope, token)?); + Ok(Some(Arc::new(self.backend(scope, token)?))) + } + + /// Runs the prune with a new marker of the claim at each refresh period, and tells whether the + /// prune leaves marked packs. + async fn run_prune( + &self, + backend: Arc, + files: &SnapshotFiles, + claim: &Claim<'_>, + grace: Duration, + ) -> Result { let key = self.key.clone(); let settings = self.policy.prune; let low_priority = self.low_priority; let pruning = self.blocking(Operation::Prune, move || { low_priority.run("fs-snap-prune", move || prune(backend, &key, &settings)) }); - // The claim is written again while the prune runs, so a prune slower than the grace - // period keeps its claim. The writes stop when the prune ends or the operation is + // The claim gets a new marker while the prune runs, so a prune slower than the grace + // period keeps its claim. The markers stop when the prune ends or the operation is // cancelled. The tracker counts the whole step, so no timer of it runs after a shut down. let refreshing = keep_claim_fresh(files, claim.directory, claim.number, refresh_period(grace)); @@ -438,9 +455,7 @@ impl RusticSnapshotStore { } }) .await?; - Ok(ClaimOutcome::Pruned { - marked_packs: report.as_ref().is_some_and(leaves_marked_packs), - }) + Ok(report.as_ref().is_some_and(leaves_marked_packs)) } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 32d245854a..4de48d693d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -19,7 +19,8 @@ use super::super::files::SnapshotFiles; use super::super::prune::{ - CLOCK_SKEW_MARGIN, LEDGERS_PATH, Percent, PruneLedger, parse_freed, read_ledger, + CLOCK_SKEW_MARGIN, ClaimEntry, LEDGERS_PATH, Percent, PruneLedger, parse_claim_entry, + parse_freed, read_ledger, }; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; @@ -1204,17 +1205,63 @@ async fn a_prune_deletes_the_older_ledger_entries_and_keeps_a_newer_one() { ); } -/// Gives the time in the one claim of the scope, when the scope has one claim that parses. +/// Gives the time of the newest marker of the claims of the scope. async fn claim_time(storage: &ScriptedBlobStorage, scope: &SnapshotScope) -> Option { - let claims = blobs(storage, &scope.0, "golem/prune-claims/").await; - let [claim] = claims.as_slice() else { - return None; - }; - let content = storage - .get_raw("test", "test", scope.0.clone(), Path::new(claim)) + blobs(storage, &scope.0, "golem/prune-claims/") .await - .ok()??; - std::str::from_utf8(&content).ok()?.parse().ok() + .iter() + .filter_map( + |path| match parse_claim_entry(Path::new(path).file_name()?.to_str()?)? { + ClaimEntry::Marker(_, at) => Some(at.to_millis()), + ClaimEntry::Claim(_) => None, + }, + ) + .max() +} + +/// Moves each marker of the claims of the scope back by two hours and the margin, so no marker +/// holds its ledger. +async fn age_claims(storage: &Arc, scope: &SnapshotScope) { + let stale = golem_common::model::Timestamp::now_utc() + .to_millis() + .saturating_sub(2 * 3_600_000 + 2 * 60_000); + let markers = storage + .list_blobs_below( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-claims"), + ) + .await + .unwrap() + .iter() + .filter_map(|blob| { + let entry = parse_claim_entry(blob.path.file_name()?.to_str()?)?; + match entry { + ClaimEntry::Marker(number, _) => Some((blob.path.clone(), number)), + ClaimEntry::Claim(_) => None, + } + }) + .collect::>(); + futures::stream::iter(markers) + .for_each(|(path, number)| { + let (storage, scope) = (storage.clone(), scope.clone()); + async move { + storage + .delete("test", "test", scope.0.clone(), &path) + .await + .unwrap(); + let aged = path + .parent() + .unwrap_or(Path::new("")) + .join(format!("{number}@{stale}-aged")); + storage + .put_raw("test", "test", scope.0.clone(), &aged, b"") + .await + .unwrap(); + } + }) + .await; } #[test] @@ -1347,26 +1394,7 @@ async fn a_failed_ledger_write_after_a_prune_keeps_the_claim_so_no_second_prune_ let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; let second = store.delete(&scope, &name("p-2")).await; let prunes_after_second = prunes(&storage.calls()); - let stale = golem_common::model::Timestamp::now_utc() - .to_millis() - .saturating_sub(3_600_000 + 2 * 60_000 + 1); - futures::stream::iter(&claims_after_failure) - .for_each(|claim| { - let (storage, scope) = (&storage, &scope); - async move { - storage - .put_raw( - "test", - "test", - scope.0.clone(), - Path::new(claim), - stale.to_string().as_bytes(), - ) - .await - .unwrap(); - } - }) - .await; + age_claims(&storage, &scope).await; let third = store.delete(&scope, &name("p-3")).await; assert!( @@ -1377,7 +1405,10 @@ async fn a_failed_ledger_write_after_a_prune_keeps_the_claim_so_no_second_prune_ assert!(third.is_ok(), "{third:?}"); assert_eq!( ( - claims_after_failure.len(), + claims_after_failure + .iter() + .filter(|path| !path.contains('@')) + .count(), prunes_after_second, prunes(&storage.calls()), ledger(&storage, &scope).await.last_prune.is_some(), @@ -1687,6 +1718,118 @@ async fn a_record_write_that_answers_already_exists_counts_as_written() { ); } +#[test] +#[timeout("60s")] +async fn a_delete_sees_the_marker_of_a_claim_that_is_not_taken_yet_and_does_not_prune() { + // The gate holds the second of the two writes of the first delete: its marker and its claim. + // The second delete runs meanwhile. + let inner = Arc::new(InMemoryBlobStorage::new()); + let writes = Arc::new(AtomicUsize::new(0)); + let first = ScriptedBlobStorage::new(inner.clone(), { + let writes = writes.clone(); + move |op_label, _| { + if matches!(op_label, "write_marker" | "write_claim") + && writes.fetch_add(1, Ordering::SeqCst) == 1 + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let second = ScriptedBlobStorage::new(inner.clone(), |_, _| Script::Pass); + let grace = Duration::from_secs(3600); + let scope = new_scope(); + save_each( + &store(inner.clone(), policy(LONG_DEADLINE, NEVER, grace)), + &scope, + &["p-1", "p-2"], + ) + .await; + let holding = tokio::spawn({ + let deleting = store(first.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = scope.clone(); + async move { deleting.delete(&scope, &name("p-1")).await } + }); + let held = eventually(|| writes.load(Ordering::SeqCst) >= 2).await; + + let seen = store(second.clone(), policy(LONG_DEADLINE, ALWAYS, grace)) + .delete(&scope, &name("p-2")) + .await; + first.open_gate(); + let holding = tokio::time::timeout(LIMIT, holding).await; + + assert!(seen.is_ok(), "{seen:?}"); + assert!(matches!(holding, Ok(Ok(Ok(())))), "{holding:?}"); + assert_eq!( + (held, prunes(&second.calls()), prunes(&first.calls())), + (true, 0, 1) + ); +} + +#[test] +#[timeout("60s")] +async fn a_claim_without_a_marker_does_not_unblock_a_live_holder() { + // The first delete holds claim 0 and waits at the start of its prune. A claim 1 without a + // marker is in the same directory. + let inner = Arc::new(InMemoryBlobStorage::new()); + let claimed = Arc::new(AtomicBool::new(false)); + let first = ScriptedBlobStorage::new(inner.clone(), { + let claimed = claimed.clone(); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" && path == Path::new("data") && claimed.load(Ordering::SeqCst) { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let second = ScriptedBlobStorage::new(inner.clone(), |_, _| Script::Pass); + let grace = Duration::from_secs(3600); + let scope = new_scope(); + save_each( + &store(inner.clone(), policy(LONG_DEADLINE, NEVER, grace)), + &scope, + &["p-1", "p-2"], + ) + .await; + let holding = tokio::spawn({ + let deleting = store(first.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = scope.clone(); + async move { deleting.delete(&scope, &name("p-1")).await } + }); + let held = eventually(|| { + first + .calls() + .iter() + .any(|(op_label, path)| *op_label == "list" && path == "data") + }) + .await; + inner + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-claims/none/1"), + b"", + ) + .await + .unwrap(); + + let seen = store(second.clone(), policy(LONG_DEADLINE, ALWAYS, grace)) + .delete(&scope, &name("p-2")) + .await; + first.open_gate(); + let holding = tokio::time::timeout(LIMIT, holding).await; + + assert!(seen.is_ok(), "{seen:?}"); + assert!(matches!(holding, Ok(Ok(Ok(())))), "{holding:?}"); + assert_eq!((held, prunes(&second.calls())), (true, 0)); +} + #[test] #[timeout("60s")] async fn two_deletes_that_read_the_same_ledger_make_one_prune() { @@ -1834,9 +1977,11 @@ async fn a_failed_second_read_of_the_ledger_deletes_the_claim_and_a_retry_of_the #[test] #[timeout("60s")] -async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() { +async fn a_prune_that_fails_keeps_its_claim_so_no_second_prune_runs_within_the_hold() { // The first listing of the packs by a prune after the first claim fails. Only a prune lists - // the packs with that call, so the second read of the ledger passes and the prune fails. + // the packs with that call, so the second read of the ledger passes and the prune fails after + // it started. Its claim stays, so a retry within the hold does not prune, and a retry after it + // does. let claimed = Arc::new(AtomicBool::new(false)); let refused = Arc::new(AtomicBool::new(false)); let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { @@ -1866,6 +2011,9 @@ async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() let failed = store.delete(&scope, &name("p-1")).await; let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; let retried = store.delete(&scope, &name("p-1")).await; + let after_retry = ledger(&storage, &scope).await; + age_claims(&storage, &scope).await; + let later = store.delete(&scope, &name("p-1")).await; let after = ledger(&storage, &scope).await; assert!( @@ -1873,13 +2021,18 @@ async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() "{failed:?}" ); assert!(retried.is_ok(), "{retried:?}"); + assert!(later.is_ok(), "{later:?}"); assert_eq!( ( - claims_after_failure, + claims_after_failure + .iter() + .filter(|path| !path.contains('@')) + .count(), refused.load(Ordering::SeqCst), + after_retry.last_prune, after.last_prune.is_some() ), - (Vec::::new(), true, true) + (1, true, None, true) ); } @@ -2040,6 +2193,8 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { let after_failure = ledger(&storage, &scope).await; let after_failure_freed = freed(&storage, &scope).await; refuse.store(false, Ordering::SeqCst); + // The prune started, so its claim stays until the hold passed. + age_claims(&storage, &scope).await; let retried = store.delete(&scope, &name("p-deleted")).await; let after_retry = ledger(&storage, &scope).await; let after_retry_freed = freed(&storage, &scope).await; @@ -3464,8 +3619,10 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { "write_ledger", "list_data", "list_claims", - "read_claim", + "write_marker", "write_claim", + "final_marker", + "delete_marker", "delete_claims", "write_freed", "list_freed", From 52b235260a15f901d317eba49d885b0a31ce39d6 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 00:07:40 -0700 Subject: [PATCH 068/126] Release a prune claim in a tracked task with a token of its own --- .../src/filesystem_snapshot/rustic/store.rs | 30 ++++- .../filesystem_snapshot/rustic/store/tests.rs | 124 ++++++++++++++++++ 2 files changed, 149 insertions(+), 5 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 469ef700f5..cce004bab5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -234,10 +234,10 @@ impl RusticSnapshotStore { /// Cancels each operation, so each running storage call ends and no new call starts, and later /// operations give `Storage`. A publish that starts before the cancel runs to its end. A save /// that reaches its publish after the cancel publishes nothing and gives `Storage`. The call - /// waits until no blocking task, backend, blob call of the store, publish or delete of a - /// dropped publish remains. A blob call that is not polled holds the wait until it is polled - /// again, and then it ends at once. The runtime must not drop before it returns, because a - /// storage call after its time driver stops aborts the process. + /// waits until no blocking task, backend, blob call of the store, publish, delete of a dropped + /// publish or release of a prune claim remains. A blob call that is not polled holds the wait + /// until it is polled again, and then it ends at once. The runtime must not drop before it + /// returns, because a storage call after its time driver stops aborts the process. pub(crate) async fn shut_down(&self) { self.root.cancel(); self.tracker.close(); @@ -372,7 +372,7 @@ impl RusticSnapshotStore { let backend = match self.prepare_prune(scope, token, &files, &claim).await { Ok(Some(backend)) => backend, other => { - release_claim(&files, claim.directory, claim.number).await; + self.release(&files, &claim).await; return other.map(|_| ()); } }; @@ -408,6 +408,26 @@ impl RusticSnapshotStore { Ok(()) } + /// Deletes the claim and then its markers in a task that the tracker counts. The task has a + /// token of its own, so a drop of the delete, a cancel or a shut down does not stop it. + async fn release(&self, files: &SnapshotFiles, claim: &Claim<'_>) { + let files = SnapshotFiles { + cancel: CancellationToken::new(), + ..files.clone() + }; + let directory: Box = claim.directory.into(); + let number = claim.number; + let releasing = self + .tracker + .spawn(async move { release_claim(&files, &directory, number).await }); + if let Err(error) = releasing.await { + warn!( + error = %error, + "The release of the prune claim of a filesystem snapshot scope did not end" + ); + } + } + /// Checks the ledger again and builds the backend of the prune. It gives `None` when another /// prune ended after the claim. async fn prepare_prune( diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 4de48d693d..b4759ca143 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1975,6 +1975,130 @@ async fn a_failed_second_read_of_the_ledger_deletes_the_claim_and_a_retry_of_the ); } +/// A storage that refuses the first call after the first claim write, which is the second read of +/// the ledger. It holds each claim delete, and each ledger read after the claim when `hold_read` is +/// true, until the gate opens. +fn failing_after_the_claim(hold_read: bool) -> Arc { + let claimed = Arc::new(AtomicBool::new(false)); + let refused = Arc::new(AtomicBool::new(false)); + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), move |op_label, _| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + Script::Pass + } else if op_label == "delete_claim" + || (hold_read && op_label == "read_ledger" && claimed.load(Ordering::SeqCst)) + { + Script::WaitForGate + } else if !hold_read + && claimed.load(Ordering::SeqCst) + && !refused.swap(true, Ordering::SeqCst) + { + Script::Refuse + } else { + Script::Pass + } + }) +} + +#[test] +#[timeout("60s")] +async fn a_delete_dropped_after_a_failed_second_read_still_releases_its_claim() { + // The second read of the ledger fails, and the gate holds the delete of the claim. The test + // drops the delete there, and then opens the gate. + let storage = failing_after_the_claim(false); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "delete_claim") + }) + .await; + + deleting.abort(); + let dropped = deleting.await; + let claims_at_drop = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + storage.open_gate(); + let ended = eventually(|| store.work_in_flight() == 0).await; + + assert!( + dropped.as_ref().is_err_and(|error| error.is_cancelled()), + "{dropped:?}" + ); + assert_eq!( + ( + held, + claims_at_drop.is_empty(), + ended, + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, false, true, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_shut_down_after_the_claim_still_releases_it() { + // The gate holds the second read of the ledger, and the shut down cancels that read. The + // gate stays shut for the read, and it opens only for the delete of the claim. + let storage = failing_after_the_claim(true); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .skip_while(|(op_label, _)| *op_label != "write_claim") + .any(|(op_label, _)| *op_label == "read_ledger") + }) + .await; + + let shutting_down = tokio::spawn({ + let store = store.clone(); + async move { store.shut_down().await } + }); + let reached_release = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "delete_claim") + }) + .await; + let claims_at_release = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + storage.open_gate(); + let shut_down = tokio::time::timeout(LIMIT, shutting_down).await; + let deleted = tokio::time::timeout(LIMIT, deleting).await; + + assert!(matches!(shut_down, Ok(Ok(()))), "{shut_down:?}"); + assert!(matches!(deleted, Ok(Ok(Err(_)))), "{deleted:?}"); + assert_eq!( + ( + held, + reached_release, + claims_at_release.is_empty(), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, true, false, Vec::::new()) + ); +} + #[test] #[timeout("60s")] async fn a_prune_that_fails_keeps_its_claim_so_no_second_prune_runs_within_the_hold() { From 7f65378dc4a7d3b474eef074c79c3f3aeb2c31ca Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 00:16:14 -0700 Subject: [PATCH 069/126] Find old claim directories with a listing of the directories and a listing of the blobs --- .../src/filesystem_snapshot/rustic/files.rs | 16 ++++- .../src/filesystem_snapshot/rustic/prune.rs | 67 +++++++++++++------ .../filesystem_snapshot/rustic/store/tests.rs | 48 +++++++++++++ .../rustic/store/tests/sweep.rs | 1 + 4 files changed, 110 insertions(+), 22 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs index 0ad802b598..8d389c6c9b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs @@ -20,7 +20,7 @@ use super::backend::answer_or_cancel; use golem_service_base::storage::blob::{ BlobStorage, BlobStorageNamespace, ListedBlob, PutIfAbsent, }; -use std::path::Path; +use std::path::{Path, PathBuf}; use std::sync::Arc; use std::time::Duration; use tokio_util::sync::CancellationToken; @@ -122,6 +122,20 @@ impl SnapshotFiles { .await } + /// Gives each blob directly below the path, and each directory that the storage keeps an + /// entry for below the path. + pub(super) async fn list_dir( + &self, + op_label: &'static str, + path: &Path, + ) -> anyhow::Result> { + self.answer( + self.storage + .list_dir(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await + } + /// Gives each blob below the path, at all depths, with its size. pub(super) async fn list_below( &self, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 7f581be483..8996b886e4 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -576,26 +576,44 @@ pub(super) async fn release_claim(files: &SnapshotFiles, directory: &Path, numbe } } -/// Gives each claim directory of the listed claims, other than the directory to keep. -pub(super) fn claim_directories_except(listed: &[ListedBlob], keep: &Path) -> Box<[PathBuf]> { +/// Gives the claim directory of each listed path below the directory of all claims, other than the +/// directory to keep. +pub(super) fn claim_directories_except( + listed: impl IntoIterator>, + keep: &Path, +) -> Box<[Box]> { + let claims = Path::new(CLAIMS_PATH); let mut directories = listed - .iter() - .filter_map(|blob| blob.path.parent().map(Path::to_path_buf)) - .filter(|directory| directory != keep) + .into_iter() + .filter_map(|path| { + let generation = path + .as_ref() + .strip_prefix(claims) + .ok()? + .components() + .next()?; + Some(claims.join(generation).into_boxed_path()) + }) + .filter(|directory| **directory != *keep) .collect::>(); directories.sort(); directories.dedup(); directories.into_boxed_slice() } -/// Deletes each claim directory other than the directory of the new ledger. It never deletes the -/// directory of all claims, so a live claim of the new ledger stays. A failure gives a warning, -/// because a claim only delays a prune until its grace period passed. +/// Deletes each claim directory other than the directory of the new ledger. A listing of the +/// directories finds an empty claim directory, and a listing of the blobs finds a claim directory +/// that the storage keeps no entry for. It never deletes the directory of all claims, so a live +/// claim of the new ledger stays. A failure gives a warning, because a claim only delays a prune +/// until its hold passed. pub(super) async fn remove_old_claims(files: &SnapshotFiles, keep: &Path) { - let listed = match files - .list_below("list_claim_directories", Path::new(CLAIMS_PATH)) - .await - { + let claims = Path::new(CLAIMS_PATH); + let listed = async { + let directories = files.list_dir("list_claim_directories", claims).await?; + let blobs = files.list_below("list_claim_blobs", claims).await?; + anyhow::Ok((directories, blobs)) + }; + let (directories, blobs) = match listed.await { Ok(listed) => listed, Err(error) => { warn!( @@ -605,7 +623,11 @@ pub(super) async fn remove_old_claims(files: &SnapshotFiles, keep: &Path) { return; } }; - stream::iter(claim_directories_except(&listed, keep)) + let listed = directories + .iter() + .map(AsRef::as_ref) + .chain(blobs.iter().map(|blob| blob.path.as_ref())); + stream::iter(claim_directories_except(listed, keep)) .for_each(|directory| async move { if let Err(error) = files.delete_dir("delete_claims", &directory).await { warn!( @@ -634,7 +656,7 @@ mod tests { use golem_service_base::storage::blob::ListedBlob; use golem_service_base::storage::blob::memory::InMemoryBlobStorage; use pretty_assertions::assert_eq; - use std::path::Path; + use std::path::{Path, PathBuf}; use std::sync::Arc; use std::time::Duration; use test_r::test; @@ -935,25 +957,28 @@ mod tests { #[test] fn each_claim_directory_other_than_the_new_one_is_old() { - let claim = |directory: &str, number: &str| ListedBlob { - path: Path::new(CLAIMS_PATH).join(directory).join(number).into(), - size: 0, - }; + let claim = + |directory: &str, number: &str| Path::new(CLAIMS_PATH).join(directory).join(number); let directory = |name: &str| Path::new(CLAIMS_PATH).join(name); assert_eq!( claim_directories_except( - &[ + [ claim("none", "0"), claim("100", "0"), - claim("100", "1"), + claim("100", "1@5-a"), claim("200", "3"), + directory("250"), + claim("260", "made/below"), claim("300", "0"), + directory("300"), + Path::new(LEDGERS_PATH).join("5-0-a"), + PathBuf::from(CLAIMS_PATH), ], &directory("300") ) .to_vec(), - vec![directory("100"), directory("200"), directory("none")] + ["100", "200", "250", "260", "none"].map(|name| directory(name).into_boxed_path()) ); } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index b4759ca143..4dfd94137f 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1366,6 +1366,53 @@ async fn a_prune_deletes_the_claims_of_old_ledgers() { ); } +#[test] +#[timeout("60s")] +async fn a_prune_deletes_an_empty_claim_directory_of_an_old_ledger() { + // A listing of the blobs does not find an empty directory, so only a listing of the + // directories finds it. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + storage + .create_dir( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-claims/100"), + ) + .await + .unwrap(); + let list_claims = || { + storage.list_dir( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-claims"), + ) + }; + let before = list_claims().await.unwrap(); + + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert_eq!( + ( + before, + ledger(&storage, &scope).await.last_prune.is_some(), + list_claims().await.unwrap(), + ), + ( + vec![std::path::PathBuf::from("golem/prune-claims/100")], + true, + Vec::::new() + ) + ); +} + #[test] #[timeout("60s")] async fn a_failed_ledger_write_after_a_prune_keeps_the_claim_so_no_second_prune_runs() { @@ -3755,6 +3802,7 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { "delete_ledger", "refresh_claim", "list_claim_directories", + "list_claim_blobs", "read_freed", "list_snapshots", ] diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs index b17dbdcdc6..ba2d1be087 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs @@ -43,6 +43,7 @@ const STEP_LABELS: &[&str] = &[ "delete_freed", "delete_claim", "list_claim_directories", + "list_claim_blobs", "delete_claims", ]; From 3d85e2799e4f0967eaa231d2a971951c4d30c9db Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 00:22:12 -0700 Subject: [PATCH 070/126] Check the tree of a save and the directory of a restore on a blocking thread --- .../src/filesystem_snapshot/rustic/store.rs | 52 +++++++++++++++---- 1 file changed, 42 insertions(+), 10 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index cce004bab5..f881fc4654 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -41,6 +41,7 @@ use crate::filesystem_snapshot::{ ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, newest_first, snapshot_time, }; +use crate::sandbox_filesystem::{NativeOperation, NativeStorageProfile, execute_native}; use crate::services::golem_config::FilesystemSnapshotStoreConfig; use anyhow::Context; use async_trait::async_trait; @@ -704,11 +705,35 @@ impl Lookup { } } +/// Runs the check of a local path on a blocking thread. A thread that fails gives the error of the +/// path. +async fn check_path( + operation: NativeOperation, + path: &Path, + error: fn(std::io::Error) -> SnapshotStoreError, + check: fn(&Path) -> Result<(), SnapshotStoreError>, +) -> Result<(), SnapshotStoreError> { + let path: Box = path.into(); + execute_native(NativeStorageProfile::Unknown, operation, move || { + check(&path) + }) + .await + .map_err(|failed| error(std::io::Error::other(failed)))? +} + /// Checks that the tree of a save is a directory at an absolute path. async fn check_tree(tree: &Path) -> Result<(), SnapshotStoreError> { - let metadata = tokio::fs::metadata(tree) - .await - .map_err(SnapshotStoreError::Source)?; + check_path( + NativeOperation::Metadata, + tree, + SnapshotStoreError::Source, + tree_is_valid, + ) + .await +} + +fn tree_is_valid(tree: &Path) -> Result<(), SnapshotStoreError> { + let metadata = std::fs::metadata(tree).map_err(SnapshotStoreError::Source)?; if !tree.is_absolute() || !metadata.is_dir() { return Err(SnapshotStoreError::Source(std::io::Error::new( std::io::ErrorKind::NotADirectory, @@ -723,15 +748,23 @@ async fn check_tree(tree: &Path) -> Result<(), SnapshotStoreError> { /// Checks that the directory of a restore is an empty directory with a UTF-8 path. async fn check_destination(into: &Path) -> Result<(), SnapshotStoreError> { + check_path( + NativeOperation::DirectoryEnumeration, + into, + SnapshotStoreError::Destination, + destination_is_valid, + ) + .await +} + +fn destination_is_valid(into: &Path) -> Result<(), SnapshotStoreError> { let refused = |kind, reason: &str| { SnapshotStoreError::Destination(std::io::Error::new( kind, format!("the directory {} {reason}", into.display()), )) }; - let metadata = tokio::fs::metadata(into) - .await - .map_err(SnapshotStoreError::Destination)?; + let metadata = std::fs::metadata(into).map_err(SnapshotStoreError::Destination)?; if !metadata.is_dir() { return Err(refused( std::io::ErrorKind::NotADirectory, @@ -744,11 +777,10 @@ async fn check_destination(into: &Path) -> Result<(), SnapshotStoreError> { "does not have a UTF-8 path", )); } - let first = tokio::fs::read_dir(into) - .await + let first = std::fs::read_dir(into) .map_err(SnapshotStoreError::Destination)? - .next_entry() - .await + .next() + .transpose() .map_err(SnapshotStoreError::Destination)?; match first { Some(_) => Err(refused( From 9f54901042e05c239556193a6f738ed0aa1dc02d Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 00:42:17 -0700 Subject: [PATCH 071/126] Release a prune claim by deleting the claim and its first marker by name, each also when the other delete fails --- .../src/filesystem_snapshot/rustic/prune.rs | 60 +++++++-------- .../src/filesystem_snapshot/rustic/store.rs | 18 +++-- .../filesystem_snapshot/rustic/store/tests.rs | 77 +++++++++++++++++++ 3 files changed, 116 insertions(+), 39 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 8996b886e4..94194777e8 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -485,20 +485,21 @@ pub(super) async fn write_marker( Ok(path) } -/// Writes the first marker of the claim with the number, then takes the claim, and tells whether -/// this delete holds it. A delete that loses the claim deletes its marker. +/// Writes the first marker of the claim with the number, then takes the claim, and gives the path +/// of that marker when this delete holds the claim. A delete that loses the claim deletes its +/// marker. pub(super) async fn take_claim( files: &SnapshotFiles, directory: &Path, number: u64, now: Timestamp, -) -> anyhow::Result { +) -> anyhow::Result> { let marker = write_marker(files, "write_marker", directory, number, now).await?; let written = files .put_if_absent("write_claim", &directory.join(number.to_string()), &[]) .await?; if written == PutIfAbsent::Written { - return Ok(true); + return Ok(Some(marker)); } if let Err(error) = files.delete("delete_marker", &marker).await { warn!( @@ -506,7 +507,7 @@ pub(super) async fn take_claim( "Failed to delete the marker of a prune claim that a filesystem snapshot delete lost" ); } - Ok(false) + Ok(None) } /// Gives the time between two markers of a live claim: a fourth of the grace period, or a fourth @@ -549,31 +550,26 @@ pub(super) async fn keep_claim_fresh( .await; } -/// Deletes the claim with the number, and then its markers. A failure gives a warning, because a -/// claim only delays a prune until its hold passed. -pub(super) async fn release_claim(files: &SnapshotFiles, directory: &Path, number: u64) { - let released = async { - files - .delete("delete_claim", &directory.join(number.to_string())) - .await?; - let listed = files.list_below("list_claims", directory).await?; - stream::iter(listed.iter().filter(|blob| { - blob.path - .file_name() - .and_then(|name| name.to_str()) - .and_then(parse_claim_entry) - .is_some_and(|entry| matches!(entry, ClaimEntry::Marker(own, _) if own == number)) - })) - .map(Ok) - .try_for_each(|blob| files.delete("delete_marker", &blob.path)) - .await - }; - if let Err(error) = released.await { - warn!( - error = %format!("{error:#}"), - "Failed to delete the prune claim of a filesystem snapshot scope" - ); - } +/// Deletes the claim with the number, and then its first marker by its path. It tries each delete +/// also when the other one fails, and each failure gives a warning. A claim that stays without +/// its marker is old, and a marker that stays only delays a prune until its hold passed. +pub(super) async fn release_claim( + files: &SnapshotFiles, + directory: &Path, + number: u64, + marker: &Path, +) { + let claim = directory.join(number.to_string()); + stream::iter([("delete_claim", claim.as_path()), ("delete_marker", marker)]) + .for_each(|(op_label, path)| async move { + if let Err(error) = files.delete(op_label, path).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete the prune claim of a filesystem snapshot scope" + ); + } + }) + .await; } /// Gives the claim directory of each listed path below the directory of all claims, other than the @@ -938,8 +934,8 @@ mod tests { assert_eq!( ( directory.display().to_string(), - first, - again, + first.is_some(), + again.is_some(), listed.len(), listed.contains(&ClaimEntry::Claim(0)), listed.contains(&ClaimEntry::Marker(0, now)), diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index f881fc4654..948c4df7fd 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -187,6 +187,8 @@ pub(super) struct PublishGate { struct Claim<'a> { directory: &'a Path, number: u64, + /// The first marker of the claim, the only marker before the prune starts. + marker: &'a Path, } /// The error of an operation of a store that is shut down. @@ -358,15 +360,16 @@ impl RusticSnapshotStore { let ClaimChoice::Claim(number) = next_claim(&listed, now, grace) else { return Ok(()); }; - if !take_claim(&files, &claims, number, now) + let Some(marker) = take_claim(&files, &claims, number, now) .await .map_err(storage_failure)? - { + else { return Ok(()); - } + }; let claim = Claim { directory: &claims, number, + marker: &marker, }; // Only an error before the prune starts releases the claim, so a retry of the delete // prunes again. A prune that started can have marked packs, so its claim stays. @@ -409,18 +412,19 @@ impl RusticSnapshotStore { Ok(()) } - /// Deletes the claim and then its markers in a task that the tracker counts. The task has a - /// token of its own, so a drop of the delete, a cancel or a shut down does not stop it. + /// Deletes the claim and then its first marker in a task that the tracker counts. The task has + /// a token of its own, so a drop of the delete, a cancel or a shut down does not stop it. async fn release(&self, files: &SnapshotFiles, claim: &Claim<'_>) { let files = SnapshotFiles { cancel: CancellationToken::new(), ..files.clone() }; - let directory: Box = claim.directory.into(); + let (directory, marker): (Box, Box) = + (claim.directory.into(), claim.marker.into()); let number = claim.number; let releasing = self .tracker - .spawn(async move { release_claim(&files, &directory, number).await }); + .spawn(async move { release_claim(&files, &directory, number, &marker).await }); if let Err(error) = releasing.await { warn!( error = %error, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 4dfd94137f..cb6cf90749 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -2146,6 +2146,83 @@ async fn a_shut_down_after_the_claim_still_releases_it() { ); } +/// A storage that refuses the first call after the first claim write, which is the second read of +/// the ledger, and each call of the release with the operation label. +fn refusing_the_release_call(release_label: &'static str) -> Arc { + let claimed = Arc::new(AtomicBool::new(false)); + let refused = Arc::new(AtomicBool::new(false)); + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), move |op_label, _| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + Script::Pass + } else if op_label == release_label + || (claimed.load(Ordering::SeqCst) && !refused.swap(true, Ordering::SeqCst)) + { + Script::Refuse + } else { + Script::Pass + } + }) +} + +#[test] +#[timeout("60s")] +async fn a_release_whose_claim_delete_is_refused_still_deletes_its_marker_and_the_next_delete_prunes() + { + let storage = refusing_the_release_call("delete_claim"); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + let retried = store.delete(&scope, &name("p-1")).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!(retried.is_ok(), "{retried:?}"); + assert_eq!( + ( + claims_after_failure, + prunes(&storage.calls()), + ledger(&storage, &scope).await.last_prune.is_some(), + ), + (vec!["golem/prune-claims/none/0".to_string()], 1, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_release_whose_marker_delete_is_refused_still_deletes_the_claim() { + let storage = refusing_the_release_call("delete_marker"); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert_eq!( + claims_after_failure + .iter() + .map(|path| path.starts_with("golem/prune-claims/none/0@")) + .collect::>(), + vec![true] + ); +} + #[test] #[timeout("60s")] async fn a_prune_that_fails_keeps_its_claim_so_no_second_prune_runs_within_the_hold() { From 81a6968780e361ce8f72fdc20c6113f54630746e Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 01:14:03 -0700 Subject: [PATCH 072/126] Keep snapshot ids, claim paths and listed directories in boxed forms --- .../src/filesystem_snapshot/rustic/files.rs | 14 +++--- .../src/filesystem_snapshot/rustic/prune.rs | 48 +++++++++++-------- .../src/filesystem_snapshot/rustic/store.rs | 4 +- 3 files changed, 38 insertions(+), 28 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs index 8d389c6c9b..a30d63da75 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs @@ -128,12 +128,14 @@ impl SnapshotFiles { &self, op_label: &'static str, path: &Path, - ) -> anyhow::Result> { - self.answer( - self.storage - .list_dir(TARGET_LABEL, op_label, self.namespace.clone(), path), - ) - .await + ) -> anyhow::Result]>> { + let listed = self + .answer( + self.storage + .list_dir(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await?; + Ok(listed.into_iter().map(PathBuf::into_boxed_path).collect()) } /// Gives each blob below the path, at all depths, with its size. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 94194777e8..b10342be58 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -29,7 +29,7 @@ use futures::{StreamExt, TryStreamExt, stream}; use golem_common::model::Timestamp; use golem_service_base::storage::blob::{ListedBlob, PutIfAbsent}; use std::collections::HashSet; -use std::path::{Path, PathBuf}; +use std::path::Path; use std::time::Duration; use tracing::warn; @@ -249,36 +249,36 @@ pub(super) struct FreedRecords { pub(super) struct FreedRecord { pub(super) path: Box, pub(super) bytes: u64, - pub(super) snapshots: Option>, + pub(super) snapshots: Option]>>, } /// The directory of the snapshot files of a repository. const SNAPSHOTS_PATH: &str = "snapshots"; /// Gives the content of a record: the id of each snapshot file of the delete, one on each line. -pub(super) fn record_content(snapshots: &[String]) -> String { +pub(super) fn record_content(snapshots: &[Box]) -> String { snapshots.join("\n") } /// Reads the snapshot ids from the content of a record. Each line must be an id of 64 hex /// characters. A content without an id does not parse, because a reader can see a record that a /// write has not filled yet. -pub(super) fn parse_record(content: &[u8]) -> Option> { +pub(super) fn parse_record(content: &[u8]) -> Option]>> { let text = std::str::from_utf8(content).ok()?; text.lines() .filter(|line| !line.is_empty()) .map(|line| { (line.len() == 64 && line.bytes().all(|byte| byte.is_ascii_hexdigit())) - .then(|| line.to_string()) + .then(|| line.into()) }) - .collect::>>() + .collect::]>>>() .filter(|snapshots| !snapshots.is_empty()) } /// Gives the settled records and the sum of their bytes. A record is settled when it names at /// least one snapshot file and none of them exists. Any other record counts as zero bytes and /// stays, and so does a record whose content does not parse. -pub(super) fn settle(records: &[FreedRecord], existing: &HashSet) -> FreedRecords { +pub(super) fn settle(records: &[FreedRecord], existing: &HashSet>) -> FreedRecords { let settled = records .iter() .filter(|record| { @@ -309,7 +309,7 @@ pub(super) fn parse_freed(name: &str) -> Option { pub(super) async fn record_freed( files: &SnapshotFiles, bytes: u64, - snapshots: &[String], + snapshots: &[Box], ) -> anyhow::Result<()> { let path = Path::new(FREED_PATH).join(format!("{bytes}-{}", uuid::Uuid::new_v4())); // The name is unique, so `AlreadyExists` means that an earlier try of this call wrote it. @@ -352,8 +352,8 @@ pub(super) async fn list_freed(files: &SnapshotFiles) -> anyhow::Result>(); + .filter_map(|blob| Some(blob.path.file_name()?.to_str()?.into())) + .collect::>>(); Ok(settle(&records, &existing)) } @@ -446,11 +446,11 @@ pub(super) fn next_claim(entries: &[ClaimEntry], now: Timestamp, grace: Duration /// Gives the directory of the claims of the ledger: the time of its last prune in milliseconds, or /// `none`. -pub(super) fn claims_directory(ledger: &PruneLedger) -> PathBuf { +pub(super) fn claims_directory(ledger: &PruneLedger) -> Box { let generation = ledger .last_prune .map_or_else(|| "none".to_string(), |last| last.to_millis().to_string()); - Path::new(CLAIMS_PATH).join(generation) + Path::new(CLAIMS_PATH).join(generation).into_boxed_path() } /// Lists the claims and the markers in the directory, from their names. A name that does not @@ -475,14 +475,14 @@ pub(super) async fn write_marker( directory: &Path, number: u64, time: Timestamp, -) -> anyhow::Result { +) -> anyhow::Result> { let path = directory.join(format!( "{number}@{}-{}", time.to_millis(), uuid::Uuid::new_v4() )); let _: PutIfAbsent = files.put_if_absent(op_label, &path, &[]).await?; - Ok(path) + Ok(path.into_boxed_path()) } /// Writes the first marker of the claim with the number, then takes the claim, and gives the path @@ -493,7 +493,7 @@ pub(super) async fn take_claim( directory: &Path, number: u64, now: Timestamp, -) -> anyhow::Result> { +) -> anyhow::Result>> { let marker = write_marker(files, "write_marker", directory, number, now).await?; let written = files .put_if_absent("write_claim", &directory.join(number.to_string()), &[]) @@ -1073,8 +1073,12 @@ mod tests { #[test] fn a_record_counts_only_when_each_of_its_snapshot_files_is_gone() { - let id = |digit: char| std::iter::repeat_n(digit, 64).collect::(); - let record = |name: &str, bytes: u64, snapshots: Option>| FreedRecord { + let id = |digit: char| { + std::iter::repeat_n(digit, 64) + .collect::() + .into_boxed_str() + }; + let record = |name: &str, bytes: u64, snapshots: Option>>| FreedRecord { path: Path::new(FREED_PATH).join(name).into(), bytes, snapshots: snapshots.map(Vec::into_boxed_slice), @@ -1108,7 +1112,11 @@ mod tests { #[test] fn the_content_of_a_record_holds_one_snapshot_id_on_each_line() { - let id = |digit: char| std::iter::repeat_n(digit, 64).collect::(); + let id = |digit: char| { + std::iter::repeat_n(digit, 64) + .collect::() + .into_boxed_str() + }; let ids = [id('a'), id('b')]; assert_eq!( @@ -1119,7 +1127,7 @@ mod tests { parse_record(&[0xff, 0xfe]), ], [ - Some(Box::new(ids.clone()) as Box<[String]>), + Some(Box::new(ids.clone()) as Box<[Box]>), None, None, None @@ -1131,7 +1139,7 @@ mod tests { async fn a_record_of_freed_bytes_is_written_and_listed() { let files = new_files(); - let gone = ["0".repeat(64)]; + let gone = ["0".repeat(64).into_boxed_str()]; record_freed(&files, 40, &gone).await.unwrap(); record_freed(&files, 2, &gone).await.unwrap(); let listed = list_freed(&files).await.unwrap(); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 948c4df7fd..bac990fe33 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -445,7 +445,7 @@ impl RusticSnapshotStore { // A prune writes its ledger before it deletes the claims, so a delete that claims in a // directory that such a prune removed sees the new ledger here. let again = read_ledger(files).await.map_err(storage_failure)?; - if claims_directory(&again) != claim.directory { + if *claims_directory(&again) != *claim.directory { return Ok(None); } Ok(Some(Arc::new(self.backend(scope, token)?))) @@ -640,7 +640,7 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { if freed > 0 { let snapshots = ids .iter() - .map(|id| id.to_hex().to_string()) + .map(|id| id.to_hex().to_string().into_boxed_str()) .collect::>(); record_freed(&files, freed, &snapshots) .await From 05278324194848302dd5b01cf3d575fcc85b7e5d Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 01:30:52 -0700 Subject: [PATCH 073/126] Plan a prune again when a concurrent forget removed a snapshot file, and release the claim when the attempts run out --- .../src/filesystem_snapshot/rustic/backend.rs | 2 +- .../src/filesystem_snapshot/rustic/fault.rs | 24 ++++- .../src/filesystem_snapshot/rustic/prune.rs | 54 ++++++----- .../src/filesystem_snapshot/rustic/store.rs | 90 +++++++++++++++---- .../filesystem_snapshot/rustic/store/tests.rs | 87 ++++++++++++++++++ 5 files changed, 215 insertions(+), 42 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index 406bd18f91..ef98482519 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -457,7 +457,7 @@ fn missing_file(path: &Path) -> Box { RusticError::with_source( ErrorKind::Backend, "The blob storage holds no file at `{path}`.", - FileMissing, + FileMissing { path: path.into() }, ) .attach_context("path", path.display().to_string()) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs index 36c09adb04..124ec28c5c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs @@ -18,10 +18,12 @@ //! rustic gives no public kind of an error. So the classification reads the chain of sources: the //! markers of this module, the name errors of the blob storage, and the I/O errors. +use super::prune::SNAPSHOTS_PATH; use crate::filesystem_snapshot::SnapshotStoreError; use golem_service_base::storage::blob::BlobNameError; use std::error::Error; use std::fmt::{Display, Formatter}; +use std::path::Path; /// A blob storage call of the backend that failed, got no answer within its deadline, or did not /// start because its operation was cancelled. The source is the failure. @@ -74,12 +76,19 @@ impl Error for ConfigExists {} /// The blob storage holds no file at the path that rustic reads, for example because a delete /// removed it after a listing. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(super) struct FileMissing; +#[derive(Debug, Clone, PartialEq, Eq)] +pub(super) struct FileMissing { + /// The path of the file, relative to the root of the repository. + pub(super) path: Box, +} impl Display for FileMissing { fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { - formatter.write_str("the blob storage holds no file at the path") + write!( + formatter, + "the blob storage holds no file at {}", + self.path.display() + ) } } @@ -90,6 +99,15 @@ pub(super) fn is_file_missing(error: &(dyn Error + 'static)) -> bool { chain(error).any(|error| error.is::()) } +/// Tells whether an error in the chain is [`FileMissing`] for a snapshot file. +pub(super) fn is_snapshot_missing(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| { + error + .downcast_ref::() + .is_some_and(|missing| missing.path.starts_with(SNAPSHOTS_PATH)) + }) +} + /// Tells whether an error in the chain is [`ConfigExists`]. pub(super) fn is_config_exists(error: &(dyn Error + 'static)) -> bool { chain(error).any(|error| error.is::()) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index b10342be58..843297d17e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -30,6 +30,7 @@ use golem_common::model::Timestamp; use golem_service_base::storage::blob::{ListedBlob, PutIfAbsent}; use std::collections::HashSet; use std::path::Path; +use std::sync::{Mutex, PoisonError}; use std::time::Duration; use tracing::warn; @@ -253,7 +254,7 @@ pub(super) struct FreedRecord { } /// The directory of the snapshot files of a repository. -const SNAPSHOTS_PATH: &str = "snapshots"; +pub(super) const SNAPSHOTS_PATH: &str = "snapshots"; /// Gives the content of a record: the id of each snapshot file of the delete, one on each line. pub(super) fn record_content(snapshots: &[Box]) -> String { @@ -521,18 +522,20 @@ pub(super) fn refresh_period(grace: Duration) -> Duration { } /// Writes a new marker of the claim with the number at each period, until the caller drops the -/// future or the operation of the files is cancelled. A failed write gives a warning. +/// future or the operation of the files is cancelled, and adds the path of each written marker to +/// `written`. A failed write gives a warning. pub(super) async fn keep_claim_fresh( files: &SnapshotFiles, directory: &Path, number: u64, period: Duration, + written: &Mutex>>, ) { stream::repeat(()) .then(|()| tokio::time::sleep(period)) .take_until(files.cancel.cancelled()) .for_each(|()| async move { - let written = write_marker( + let marker = write_marker( files, "refresh_claim", directory, @@ -540,36 +543,46 @@ pub(super) async fn keep_claim_fresh( Timestamp::now_utc(), ) .await; - if let Err(error) = written { - warn!( + match marker { + Ok(path) => written + .lock() + .unwrap_or_else(PoisonError::into_inner) + .push(path), + Err(error) => warn!( error = %format!("{error:#}"), "Failed to write a new marker of the prune claim of a filesystem snapshot scope" - ); + ), } }) .await; } -/// Deletes the claim with the number, and then its first marker by its path. It tries each delete -/// also when the other one fails, and each failure gives a warning. A claim that stays without -/// its marker is old, and a marker that stays only delays a prune until its hold passed. +/// Deletes the claim with the number, and then each of its markers by its path. It tries each +/// delete also when another one fails, and each failure gives a warning. A claim that stays +/// without its markers is old, and a marker that stays only delays a prune until its hold passed. pub(super) async fn release_claim( files: &SnapshotFiles, directory: &Path, number: u64, - marker: &Path, + markers: &[Box], ) { let claim = directory.join(number.to_string()); - stream::iter([("delete_claim", claim.as_path()), ("delete_marker", marker)]) - .for_each(|(op_label, path)| async move { - if let Err(error) = files.delete(op_label, path).await { - warn!( - error = %format!("{error:#}"), - "Failed to delete the prune claim of a filesystem snapshot scope" - ); - } - }) - .await; + stream::iter( + std::iter::once(("delete_claim", claim.as_path())).chain( + markers + .iter() + .map(|marker| ("delete_marker", marker.as_ref())), + ), + ) + .for_each(|(op_label, path)| async move { + if let Err(error) = files.delete(op_label, path).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete the prune claim of a filesystem snapshot scope" + ); + } + }) + .await; } /// Gives the claim directory of each listed path below the directory of all claims, other than the @@ -911,6 +924,7 @@ mod tests { &claims_directory(&ledger(None, false)), 0, Duration::from_secs(3600), + &std::sync::Mutex::default(), ), ) .await diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index bac990fe33..0225f8b7c6 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -21,7 +21,9 @@ //! [`RusticSnapshotStore::shut_down`] waits for them. use super::backend::BlobBackend; -use super::fault::{Operation, classify, is_file_missing, is_storage_failure, storage_failure}; +use super::fault::{ + Operation, classify, is_file_missing, is_snapshot_missing, is_storage_failure, storage_failure, +}; use super::files::SnapshotFiles; use super::priority::LowPriority; use super::prune::{ @@ -59,7 +61,7 @@ use serde::{Deserialize, Serialize}; use std::num::NonZeroUsize; use std::path::Path; use std::pin::pin; -use std::sync::Arc; +use std::sync::{Arc, Mutex, PoisonError}; use std::time::Duration; use tokio::runtime::Handle; use tokio_util::sync::{CancellationToken, DropGuard}; @@ -187,8 +189,24 @@ pub(super) struct PublishGate { struct Claim<'a> { directory: &'a Path, number: u64, - /// The first marker of the claim, the only marker before the prune starts. - marker: &'a Path, + /// The markers of the claim that this delete wrote: the first marker, and each new marker + /// while the prune runs. + markers: Mutex>>, +} + +/// The number of times that a prune plans again when a snapshot file that it listed is gone at +/// its read. +const PRUNE_ATTEMPTS: usize = 3; + +/// The error of a delete whose prune found a snapshot file gone at each attempt. A concurrent +/// delete removed the files, so a retry of the delete can prune. +fn snapshots_changed_error() -> SnapshotStoreError { + SnapshotStoreError::Storage { + retryable: true, + source: anyhow::anyhow!( + "a concurrent delete removed a snapshot file at each attempt of the prune" + ), + } } /// The error of an operation of a store that is shut down. @@ -369,7 +387,7 @@ impl RusticSnapshotStore { let claim = Claim { directory: &claims, number, - marker: &marker, + markers: Mutex::new(vec![marker]), }; // Only an error before the prune starts releases the claim, so a retry of the delete // prunes again. A prune that started can have marked packs, so its claim stays. @@ -380,7 +398,15 @@ impl RusticSnapshotStore { return other.map(|_| ()); } }; - let pruned = self.run_prune(backend, &files, &claim, grace).await; + // No attempt of a prune that found a snapshot file gone changed the repository, so no + // prune ran, and the claim goes. + let pruned = match self.run_prune(backend, &files, &claim, grace).await { + Ok(None) => { + self.release(&files, &claim).await; + return Err(snapshots_changed_error()); + } + other => other.map(|marked| marked.unwrap_or_default()), + }; // The final marker holds the claim for the grace period from the end of this prune, also // when the prune or its ledger write fails. if let Err(error) = write_marker( @@ -412,19 +438,23 @@ impl RusticSnapshotStore { Ok(()) } - /// Deletes the claim and then its first marker in a task that the tracker counts. The task has - /// a token of its own, so a drop of the delete, a cancel or a shut down does not stop it. + /// Deletes the claim and then its markers in a task that the tracker counts. The task has a + /// token of its own, so a drop of the delete, a cancel or a shut down does not stop it. async fn release(&self, files: &SnapshotFiles, claim: &Claim<'_>) { let files = SnapshotFiles { cancel: CancellationToken::new(), ..files.clone() }; - let (directory, marker): (Box, Box) = - (claim.directory.into(), claim.marker.into()); + let directory: Box = claim.directory.into(); + let markers = claim + .markers + .lock() + .unwrap_or_else(PoisonError::into_inner) + .clone(); let number = claim.number; let releasing = self .tracker - .spawn(async move { release_claim(&files, &directory, number, &marker).await }); + .spawn(async move { release_claim(&files, &directory, number, &markers).await }); if let Err(error) = releasing.await { warn!( error = %error, @@ -452,26 +482,44 @@ impl RusticSnapshotStore { } /// Runs the prune with a new marker of the claim at each refresh period, and tells whether the - /// prune leaves marked packs. + /// prune leaves marked packs. It gives `None` when each attempt found a snapshot file gone. async fn run_prune( &self, backend: Arc, files: &SnapshotFiles, claim: &Claim<'_>, grace: Duration, - ) -> Result { + ) -> Result, SnapshotStoreError> { let key = self.key.clone(); let settings = self.policy.prune; let low_priority = self.low_priority; + // The plan of a prune reads each snapshot file before the prune changes the repository. + // A forget of another delete can remove a listed file before its read, so the prune + // plans again from a new listing. let pruning = self.blocking(Operation::Prune, move || { - low_priority.run("fs-snap-prune", move || prune(backend, &key, &settings)) + low_priority.run("fs-snap-prune", move || { + Ok( + std::iter::repeat_with(|| prune(backend.clone(), &key, &settings)) + .take(PRUNE_ATTEMPTS) + .find(|attempt| { + !attempt + .as_ref() + .is_err_and(|error| is_snapshot_missing(&**error)) + }), + ) + }) }); // The claim gets a new marker while the prune runs, so a prune slower than the grace // period keeps its claim. The markers stop when the prune ends or the operation is // cancelled. The tracker counts the whole step, so no timer of it runs after a shut down. - let refreshing = - keep_claim_fresh(files, claim.directory, claim.number, refresh_period(grace)); - let report = self + let refreshing = keep_claim_fresh( + files, + claim.directory, + claim.number, + refresh_period(grace), + &claim.markers, + ); + let attempts = self .tracker .track_future(async { match future::select(pin!(pruning), pin!(refreshing)).await { @@ -480,7 +528,13 @@ impl RusticSnapshotStore { } }) .await?; - Ok(report.as_ref().is_some_and(leaves_marked_packs)) + attempts + .map(|report| { + report + .map(|report| report.as_ref().is_some_and(leaves_marked_packs)) + .map_err(|error| classify(Operation::Prune, error)) + }) + .transpose() } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index cb6cf90749..5b34f91d3d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -2223,6 +2223,93 @@ async fn a_release_whose_marker_delete_is_refused_still_deletes_the_claim() { ); } +/// A storage that gives no blob to the first `vanishing` reads of a snapshot file after the first +/// claim write, as a forget of another delete between the listing and the read does. It counts +/// those reads. +fn vanishing_snapshots_after_the_claim( + vanishing: usize, +) -> (Arc, Arc) { + let claimed = Arc::new(AtomicBool::new(false)); + let vanished = Arc::new(AtomicUsize::new(0)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let vanished = vanished.clone(); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + let snapshot_read = claimed.load(Ordering::SeqCst) + && path.starts_with("snapshots") + && !is_forget(op_label, path) + && op_label != "list"; + if snapshot_read + && vanished + .fetch_update(Ordering::SeqCst, Ordering::SeqCst, |count| { + (count < vanishing).then_some(count + 1) + }) + .is_ok() + { + Script::Vanish + } else { + Script::Pass + } + } + }); + (storage, vanished) +} + +#[test] +#[timeout("60s")] +async fn a_prune_plans_again_when_a_snapshot_file_is_gone_at_its_read_and_prunes_once() { + let (storage, vanished) = vanishing_snapshots_after_the_claim(1); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + vanished.load(Ordering::SeqCst), + prunes(&storage.calls()), + ledger(&storage, &scope).await.last_prune.is_some(), + ), + (1, 1, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_that_finds_a_snapshot_file_gone_at_each_attempt_releases_its_claim_and_gives_a_retryable_error() + { + let (storage, vanished) = vanishing_snapshots_after_the_claim(usize::MAX); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert_eq!( + ( + vanished.load(Ordering::SeqCst), + prunes(&storage.calls()), + ledger(&storage, &scope).await.last_prune.is_some(), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (3, 0, false, Vec::::new()) + ); +} + #[test] #[timeout("60s")] async fn a_prune_that_fails_keeps_its_claim_so_no_second_prune_runs_within_the_hold() { From d2d706850375376c4b351d37959a055b6b578df6 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 01:48:31 -0700 Subject: [PATCH 074/126] Sweep the marker steps of the claims, the refresh with a short grace period, and the rule that a started prune keeps its claim --- .../filesystem_snapshot/rustic/scripted.rs | 78 +++++-- .../filesystem_snapshot/rustic/store/tests.rs | 7 +- .../rustic/store/tests/sweep.rs | 220 +++++++++++++++--- 3 files changed, 251 insertions(+), 54 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs index de6f3f5355..9520c885fd 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -71,8 +71,10 @@ pub(super) struct ScriptedBlobStorage { steps: Semaphore, /// The calls that wait for a step. waiting: AtomicUsize, - /// The calls that took a step and ended. + /// The calls that took a step and ended, or that dropped after they took a step. stepped: AtomicUsize, + /// The operation label and the path of each call that took a step, in the order of the steps. + took: Mutex)>>, /// The landings that the test gave and that no late call took yet. landings: Arc, /// The late calls that reached the storage. @@ -92,6 +94,7 @@ impl ScriptedBlobStorage { steps: Semaphore::new(0), waiting: AtomicUsize::new(0), stepped: AtomicUsize::new(0), + took: Mutex::new(Vec::new()), landings: Arc::new(Semaphore::new(0)), landed: Arc::new(AtomicUsize::new(0)), }) @@ -117,6 +120,32 @@ impl ScriptedBlobStorage { self.stepped.load(Ordering::SeqCst) } + /// Gives the operation label and the path of each call that took a step, in the order of the + /// steps. Two calls can wait for a step at one time, and the one that waited first takes it. + pub(super) fn took(&self) -> Vec<(&'static str, String)> { + self.took + .lock() + .unwrap_or_else(PoisonError::into_inner) + .iter() + .map(|(op_label, path)| (*op_label, path.display().to_string())) + .collect() + } + + /// Waits until the test gives the call a step. The call counts as waiting until it takes the + /// step or drops, and it counts as stepped when the returned guard drops. + async fn wait_for_step(&self, op_label: &'static str, path: &Path) -> Stepped<'_> { + let waiting = Waiting::new(&self.waiting); + if let Ok(permit) = self.steps.acquire().await { + permit.forget(); + } + drop(waiting); + self.took + .lock() + .unwrap_or_else(PoisonError::into_inner) + .push((op_label, path.into())); + Stepped(&self.stepped) + } + /// Lets one late call, now or later, reach the storage. pub(super) fn land_late(&self) { self.landings.add_permits(1); @@ -136,12 +165,7 @@ impl ScriptedBlobStorage { landing: impl Future> + Send + 'static, ) -> anyhow::Result { self.record(op_label, path); - self.waiting.fetch_add(1, Ordering::SeqCst); - let permit = self.steps.acquire().await; - self.waiting.fetch_sub(1, Ordering::SeqCst); - if let Ok(permit) = permit { - permit.forget(); - } + let stepped = self.wait_for_step(op_label, path).await; let (landings, landed) = (self.landings.clone(), self.landed.clone()); tokio::spawn(async move { if let Ok(permit) = landings.acquire().await { @@ -150,7 +174,7 @@ impl ScriptedBlobStorage { let _ = landing.await; landed.fetch_add(1, Ordering::SeqCst); }); - self.stepped.fetch_add(1, Ordering::SeqCst); + drop(stepped); Err(anyhow::anyhow!( "the call got no answer within its deadline" )) @@ -209,24 +233,42 @@ impl ScriptedBlobStorage { } Script::Vanish | Script::AnswerAlreadyExists => call.await, Script::Step { refuse, .. } => { - self.waiting.fetch_add(1, Ordering::SeqCst); - let permit = self.steps.acquire().await; - self.waiting.fetch_sub(1, Ordering::SeqCst); - if let Ok(permit) = permit { - permit.forget(); - } - let answer = if refuse { + let _stepped = self.wait_for_step(op_label, path).await; + if refuse { Err(anyhow::anyhow!("the storage refused the call")) } else { call.await - }; - self.stepped.fetch_add(1, Ordering::SeqCst); - answer + } } } } } +/// Counts a call that waits for a step, until the guard drops. +struct Waiting<'a>(&'a AtomicUsize); + +impl<'a> Waiting<'a> { + fn new(waiting: &'a AtomicUsize) -> Self { + waiting.fetch_add(1, Ordering::SeqCst); + Self(waiting) + } +} + +impl Drop for Waiting<'_> { + fn drop(&mut self) { + self.0.fetch_sub(1, Ordering::SeqCst); + } +} + +/// Counts a call that took a step as stepped when the guard drops. +struct Stepped<'a>(&'a AtomicUsize); + +impl Drop for Stepped<'_> { + fn drop(&mut self) { + self.0.fetch_add(1, Ordering::SeqCst); + } +} + impl Debug for ScriptedBlobStorage { fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { formatter.write_str("ScriptedBlobStorage") diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 5b34f91d3d..497f1bb05c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -4175,9 +4175,10 @@ async fn the_global_rayon_pool_keeps_the_nice_value_of_the_process_after_saves_w mod sweep; -/// The largest number of steps of one turn. A delete takes at most 15 steps of the protocol, so a -/// turn of 16 steps runs a delete to its end. -const SWEEP_TURN: usize = 16; +/// The largest number of steps of one turn. A delete takes at most 21 steps of the protocol, and +/// its prune writes at least one new marker of its claim, so a turn of 22 steps runs a delete to +/// its end when its prune writes one new marker. +const SWEEP_TURN: usize = 22; /// The number of random orders that the property test tries. const SWEEP_CASES: u32 = 1000; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs index ba2d1be087..c70453b696 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs @@ -17,7 +17,9 @@ //! Each delete has its own store over its own scripted storage, and the two storages share one //! in-memory storage, as two executors share a bucket. Each blob call of the prune protocol is a //! step: it waits until the test gives its delete one step. So the test sets the order of the -//! calls of the two deletes, and nothing else changes it. +//! calls of the two deletes. The grace period is short, so a prune writes new markers of its claim +//! while it runs. A timer starts each such write, so its place in the order can change from run to +//! run, and the log of a case gives the call that took each step. use super::*; use futures::TryStreamExt; @@ -34,9 +36,11 @@ const STEP_LABELS: &[&str] = &[ "list_freed", "list_data", "list_claims", - "read_claim", + "write_marker", "write_claim", + "delete_marker", "refresh_claim", + "final_marker", "write_ledger", "list_ledgers", "delete_ledger", @@ -47,7 +51,8 @@ const STEP_LABELS: &[&str] = &[ "delete_claims", ]; -/// The most steps that one delete takes before the test gives up on it. +/// The most steps that one delete takes before the test gives up on it, without the writes of +/// new markers. Those writes end with the prune. const MOST_STEPS: usize = 64; fn is_step(op_label: &str, path: &Path) -> bool { @@ -58,9 +63,19 @@ fn is_step(op_label: &str, path: &Path) -> bool { /// Tells whether the call writes or deletes, so that it can reach the storage late. fn can_be_late(op_label: &str) -> bool { - op_label.starts_with("write_") || op_label.starts_with("delete") || op_label == "refresh_claim" + op_label.starts_with("write_") || op_label.starts_with("delete") || op_label == "final_marker" } +/// Tells whether the call writes a new marker of a claim while its prune runs. A timer starts +/// such a call, so the schedule neither counts it nor makes it fail. +fn is_refresh(op_label: &str) -> bool { + op_label == "refresh_claim" +} + +/// The grace period of the deletes of a case. The claim of a prune gets a new marker at each +/// quarter of it, so the markers come while the prune runs. +const SWEEP_GRACE: Duration = Duration::from_millis(16); + fn is_forget(op_label: &str, path: &Path) -> bool { op_label == "delete" && path.starts_with("snapshots") } @@ -70,10 +85,11 @@ fn is_prune_start(op_label: &str, path: &str) -> bool { } /// The order of the steps of one case: the delete that goes first, and the numbers of steps that -/// the deletes take in turn. After the listed turns, the delete whose turn is next runs to its -/// end, and then the other one does. `fail` refuses one step of one delete. `late` makes one -/// write or delete of one delete give no answer at its step, and reach the storage after the -/// given number of further steps of the case. +/// the deletes take in turn. A turn counts each step, also the write of a new marker. After the +/// listed turns, the delete whose turn is next runs to its end, and then the other one does. +/// `fail` refuses one step of one delete. `late` makes one write or delete of one delete give no +/// answer at its step, and reach the storage after the given number of further steps of the case. +/// The steps of `fail` and `late` are numbered without the writes of new markers. #[derive(Clone, Debug)] pub(super) struct Schedule { pub(super) first: usize, @@ -99,6 +115,7 @@ struct Step { struct Delete { storage: Arc, task: JoinHandle>, + /// The steps that the delete took, other than the writes of new markers. taken: usize, } @@ -132,9 +149,27 @@ async fn until(condition: impl Fn() -> bool) -> bool { .is_ok() } -/// Waits until the delete waits for a step or ended. +/// Waits until the delete waits for a step or ended. While the prune waits for the listing of +/// the packs, it also waits until the first new marker of the claim waits for a step, so each +/// prune writes a new marker right after that listing. async fn settle(delete: &Delete) -> bool { - until(|| delete.storage.waiting_steps() > 0 || delete.finished()).await + until(|| { + let waiting = delete.storage.waiting_steps(); + delete.finished() || waiting > usize::from(prune_waits(&delete.storage)) + }) + .await +} + +/// Tells whether the prune of the delete waits for its step: the storage has a call for the +/// listing of the packs that took no step yet. +fn prune_waits(storage: &ScriptedBlobStorage) -> bool { + let starts = |calls: Vec<(&'static str, String)>| { + calls + .iter() + .filter(|(op_label, path)| is_prune_start(op_label, path)) + .count() + }; + starts(storage.calls()) > starts(storage.took()) } impl Case { @@ -173,26 +208,32 @@ impl Case { if self.deletes[who].finished() { return Ok(()); } - let calls = self.deletes[who].storage.calls(); - let (op_label, path) = calls - .iter() - .rev() - .find(|(op_label, path)| is_step(op_label, Path::new(path))) - .cloned() - .ok_or_else(|| format!("delete {who} waits for a step with no step call"))?; - let number = self.deletes[who].taken; - let refused = self.schedule.fail == Some((who, number)); - let late = self.schedule.late.filter(|(late_who, late_step, _)| { - (*late_who, *late_step) == (who, number) && can_be_late(op_label) - }); let storage = self.deletes[who].storage.clone(); - let before = storage.stepped(); + let (before, taker) = (storage.stepped(), storage.took().len()); storage.step(); if !until(|| storage.stepped() > before).await { return Err(format!( - "the step {op_label} {path} of delete {who} did not end" + "a step of delete {who} did not end; its last calls {:?}", + storage.calls().iter().rev().take(6).collect::>() + )); + } + let took = storage.took(); + if took.len() != taker + 1 { + return Err(format!( + "one step of delete {who} was taken by {:?}", + took.get(taker..) )); } + let (op_label, path) = took + .get(taker) + .cloned() + .ok_or_else(|| format!("a step of delete {who} ended, and no call took it"))?; + let number = self.deletes[who].taken; + let counted = !is_refresh(op_label); + let refused = counted && self.schedule.fail == Some((who, number)); + let late = self.schedule.late.filter(|(late_who, late_step, _)| { + counted && (*late_who, *late_step) == (who, number) && can_be_late(op_label) + }); let step = Step { delete: who, op_label, @@ -204,7 +245,7 @@ impl Case { if let Some((_, _, delay)) = late { self.pending = Some((who, self.log.len() + delay, step)); } - self.deletes[who].taken += 1; + self.deletes[who].taken += usize::from(counted); if self.deletes[who].taken > MOST_STEPS { return Err(format!("delete {who} took more than {MOST_STEPS} steps")); } @@ -232,6 +273,20 @@ impl Case { .await .map(|_| ()) } + + /// Gives the delete steps until it ends. + async fn run_to_end(&mut self, who: usize) -> Result<(), String> { + futures::stream::unfold(self, |case| async move { + if case.deletes[who].finished() { + None + } else { + let stepped = case.take_step(who).await; + Some((stepped, case)) + } + }) + .try_for_each(|()| std::future::ready(Ok(()))) + .await + } } /// What a case found. @@ -258,7 +313,12 @@ pub(super) async fn run_case( let counter = Arc::new(AtomicUsize::new(0)); let (fail, late) = (schedule.fail, schedule.late); let storage = ScriptedBlobStorage::new(shared.clone(), move |op_label, path| { - if is_step(op_label, path) { + if is_refresh(op_label) { + Script::Step { + refuse: false, + late: false, + } + } else if is_step(op_label, path) { let number = counter.fetch_add(1, Ordering::SeqCst); Script::Step { refuse: fail == Some((who, number)), @@ -271,10 +331,7 @@ pub(super) async fn run_case( Script::Pass } }); - let deleting = store( - storage.clone(), - policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), - ); + let deleting = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, SWEEP_GRACE)); let scope = scope.clone(); let task = tokio::spawn(async move { deleting.delete(&scope, &name(["p-1", "p-2"][who])).await }); @@ -317,8 +374,8 @@ async fn run_turns(case: &mut Case, schedule: &Schedule, turns: &[usize]) -> Res }) .await .map(|_| (schedule.first + turns.len()) % 2)?; - case.take_steps(next, MOST_STEPS + 1).await?; - case.take_steps((next + 1) % 2, MOST_STEPS + 1).await?; + case.run_to_end(next).await?; + case.run_to_end((next + 1) % 2).await?; case.land(true).await } @@ -346,6 +403,89 @@ fn forgotten_at(log: &[Step], who: usize) -> Option { }) } +/// Gives each claim, or marker of a claim, that a delete wrote and that stays, when the first +/// failed call of that delete came after it took its claim and before it called for the listing of +/// the packs. A claim that the other delete wrote later at the same path is not the claim of the +/// delete. A blob whose own delete failed is left out, because no call can remove it then, and a +/// claim or a marker that stays only delays a prune. +fn kept_claims(log: &[Step], claims: &[String]) -> Vec { + [0, 1] + .into_iter() + .filter_map(|who| { + let own = |step: &Step| step.delete == who; + let claimed_at = log.iter().position(|step| { + own(step) && step.op_label == "write_claim" && step.effect && !step.failed + })?; + let failed_at = log.iter().position(|step| own(step) && step.failed)?; + let pruning = log[..=failed_at] + .iter() + .any(|step| own(step) && is_prune_start(step.op_label, &step.path)); + (claimed_at < failed_at && !pruning).then_some((who, claimed_at)) + }) + .flat_map(|(who, claimed_at)| { + let claim = &log[claimed_at].path; + let taken_again = log[claimed_at..].iter().any(|step| { + step.delete != who + && step.op_label == "write_claim" + && step.effect + && step.path == *claim + }); + let written = log + .iter() + .filter(|step| { + step.delete == who + && step.effect + && matches!( + step.op_label, + "write_marker" | "refresh_claim" | "final_marker" + ) + }) + .map(|step| step.path.clone()) + .chain((!taken_again).then(|| claim.clone())) + .collect::>(); + let refused = log + .iter() + .filter(|step| { + step.delete == who + && step.failed + && matches!(step.op_label, "delete_claim" | "delete_marker") + }) + .map(|step| step.path.clone()) + .collect::>(); + claims + .iter() + .filter(move |path| written.contains(path) && !refused.contains(path)) + .cloned() + .collect::>() + }) + .collect() +} + +/// Gives the claim of each delete whose prune started and that wrote no ledger entry, when that +/// claim is gone. A prune that started keeps its claim, so the next prune waits for the hold. +fn started_claims_gone(log: &[Step], claims: &[String]) -> Vec { + [0, 1] + .into_iter() + .filter_map(|who| { + let own = |step: &&Step| step.delete == who; + let claim = log + .iter() + .filter(own) + .find(|step| step.op_label == "write_claim" && step.effect && !step.failed)?; + let started = log + .iter() + .filter(own) + .any(|step| is_prune_start(step.op_label, &step.path)); + let ledger_written = log + .iter() + .filter(own) + .any(|step| step.op_label == "write_ledger" && step.effect); + (started && !ledger_written && !claims.contains(&claim.path)) + .then(|| claim.path.clone()) + }) + .collect() +} + /// Checks the rules on the end state of a case. async fn check( shared: &Arc, @@ -371,7 +511,9 @@ async fn check( return fail("more than one prune ran"); } if !failed && prunes != 1 { - return fail("no prune ran, although a prune was due and no call failed"); + return fail(&format!( + "no prune ran, although a prune was due and no call failed: {results:?}" + )); } if !failed && results.iter().any(|result| !matches!(result, Ok(Ok(())))) { return fail(&format!("a delete failed with no failed call: {results:?}")); @@ -380,6 +522,18 @@ async fn check( if !failed && !claims.is_empty() { return fail(&format!("claims stay: {claims:?}")); } + let released = kept_claims(log, &claims); + if !released.is_empty() { + return fail(&format!( + "a delete that failed after its claim and before its prune kept its claim: {released:?}" + )); + } + let dropped = started_claims_gone(log, &claims); + if !dropped.is_empty() { + return fail(&format!( + "a delete whose prune started and wrote no ledger lost its claim: {dropped:?}" + )); + } let records = blobs(&**shared, &scope.0, "golem/prune-freed/").await; let entries = blobs(&**shared, &scope.0, "golem/prune-ledgers/").await; let ledger = ledger(shared, scope).await; From acb42e678b1b4d50817980314d37c48d6b1360ed Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 02:07:00 -0700 Subject: [PATCH 075/126] Say in the doc of a prune that only an error before the prune started releases its claim --- .../src/filesystem_snapshot/rustic/store.rs | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 0225f8b7c6..6cdbf2d2b0 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -339,13 +339,16 @@ impl RusticSnapshotStore { } /// Prunes the repository when a prune is due. The records of freed bytes stay until a prune - /// succeeds, so a delete that runs again after a failed prune prunes again. It lists the packs only when their size can make a prune due. + /// succeeds, so a later prune counts them again. It lists the packs only when their size can + /// make a prune due. /// A due prune runs only after the delete takes a claim of its ledger, and only when the ledger /// did not change after the claim. After an error before the prune ran, the claim is deleted, /// so a retry of the delete prunes again, and a claim write that the storage completes after - /// that delete can delay that prune by up to the grace period. After a prune whose ledger write - /// failed, the claim stays, so the next prune waits up to the grace period. A prune that - /// succeeds deletes each claim of its ledger. + /// that delete can delay that prune by up to the grace period. A prune that found a snapshot + /// file gone at each attempt changed nothing, so it counts as an error before the prune ran. + /// After a prune that started, the claim stays on each outcome, also when the prune or its + /// ledger write fails, so the next prune waits the grace period from the end of this one. A + /// prune that succeeds deletes each claim of its ledger. async fn prune_when_due( &self, scope: &SnapshotScope, From 73190af2eb537179773129c107fc75a1dbc14bed Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 02:10:47 -0700 Subject: [PATCH 076/126] Open the gate of the single read test when the second thread waits for the pack, and not after a sleep --- .../rustic/backend/kept.rs | 25 ++++++++++++++++++- .../rustic/backend/tests.rs | 11 +++++--- 2 files changed, 32 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs index 92af19013d..f5c0d36a8d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs @@ -36,6 +36,9 @@ struct State { packs: HashMap, bytes: usize, reading: HashSet, + /// The threads that wait for the read of a pack by another thread. + #[cfg(test)] + waiters: usize, } impl KeptPacks { @@ -56,10 +59,24 @@ impl KeptPacks { id: &Id, read: impl FnOnce() -> RusticResult, ) -> RusticResult { + #[cfg(test)] + let mut counted = false; let mut state = self .read_ended - .wait_while(self.state(), |state| state.reading.contains(id)) + .wait_while(self.state(), |state| { + let waits = state.reading.contains(id); + #[cfg(test)] + if waits && !counted { + state.waiters += 1; + counted = true; + } + waits + }) .unwrap_or_else(PoisonError::into_inner); + #[cfg(test)] + if counted { + state.waiters -= 1; + } if let Some(pack) = state.packs.get(id) { return Ok(pack.clone()); } @@ -76,6 +93,12 @@ impl KeptPacks { read } + /// Gives the number of threads that wait for the read of a pack by another thread. + #[cfg(test)] + pub(super) fn waiters(&self) -> usize { + self.state().waiters + } + fn state(&self) -> MutexGuard<'_, State> { self.state.lock().unwrap_or_else(PoisonError::into_inner) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index 18402a6557..cfefc01d8b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -596,8 +596,8 @@ fn a_range_that_is_not_cacheable_is_a_ranged_read_each_time() { #[test] fn two_threads_that_miss_one_pack_make_one_storage_read() { - // The first read waits at the gate. The second thread starts while it waits, and the gate - // opens only after the second thread had time to ask for the pack. + // The first read waits at the gate. The gate opens only when the second thread waits for that + // read, or when the second thread reads the pack itself. let fixture = PackFixture::new(1024, |op_label, _| { if op_label == "read" { Script::WaitForGate @@ -617,17 +617,22 @@ fn two_threads_that_miss_one_pack_make_one_storage_read() { let backend = fixture.backend.clone(); move || tree_range(&backend, 50, 10).ok() }); - std::thread::sleep(Duration::from_millis(200)); + let second_asked = (0..1000).any(|_| { + std::thread::sleep(Duration::from_millis(10)); + fixture.backend.kept.waiters() == 1 || fixture.pack_calls().len() > 1 + }); fixture.storage.open_gate(); assert_eq!( ( first_read_started, + second_asked, first.recv_timeout(LIMIT).ok().flatten(), second.recv_timeout(LIMIT).ok().flatten(), fixture.pack_calls() ), ( + true, true, Some(Bytes::from_iter(0..10)), Some(Bytes::from_iter(50..60)), From 28a06262a40dcdd790e317ddf12fd43adda84900 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 02:15:47 -0700 Subject: [PATCH 077/126] Delete after a prune only the claim directories of older ledgers, and keep a newer one --- .../src/filesystem_snapshot/rustic/prune.rs | 49 ++++++++++++------- .../src/filesystem_snapshot/rustic/store.rs | 8 +-- .../filesystem_snapshot/rustic/store/tests.rs | 38 ++++++++++++++ 3 files changed, 70 insertions(+), 25 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 843297d17e..3313c52a5c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -585,11 +585,13 @@ pub(super) async fn release_claim( .await; } -/// Gives the claim directory of each listed path below the directory of all claims, other than the -/// directory to keep. -pub(super) fn claim_directories_except( +/// Gives the claim directory of each listed path below the directory of all claims whose ledger is +/// older than the ledger of `ended`: the directory `none`, and each directory whose time is before +/// `ended`. A newer directory can hold a live claim of a later prune, and a directory whose name is +/// not a time is not a directory of claims, so both stay. +pub(super) fn old_claim_directories( listed: impl IntoIterator>, - keep: &Path, + ended: Timestamp, ) -> Box<[Box]> { let claims = Path::new(CLAIMS_PATH); let mut directories = listed @@ -600,22 +602,28 @@ pub(super) fn claim_directories_except( .strip_prefix(claims) .ok()? .components() - .next()?; - Some(claims.join(generation).into_boxed_path()) + .next()? + .as_os_str() + .to_str()? + .to_owned(); + let old = generation == "none" + || generation + .parse::() + .is_ok_and(|millis| millis < ended.to_millis()); + old.then(|| claims.join(generation).into_boxed_path()) }) - .filter(|directory| **directory != *keep) .collect::>(); directories.sort(); directories.dedup(); directories.into_boxed_slice() } -/// Deletes each claim directory other than the directory of the new ledger. A listing of the +/// Deletes each claim directory of a ledger older than the ledger of `ended`. A listing of the /// directories finds an empty claim directory, and a listing of the blobs finds a claim directory -/// that the storage keeps no entry for. It never deletes the directory of all claims, so a live -/// claim of the new ledger stays. A failure gives a warning, because a claim only delays a prune -/// until its hold passed. -pub(super) async fn remove_old_claims(files: &SnapshotFiles, keep: &Path) { +/// that the storage keeps no entry for. It never deletes the directory of all claims or a newer +/// directory, so a live claim of a later ledger stays. A failure gives a warning, because a claim +/// only delays a prune until its hold passed. +pub(super) async fn remove_old_claims(files: &SnapshotFiles, ended: Timestamp) { let claims = Path::new(CLAIMS_PATH); let listed = async { let directories = files.list_dir("list_claim_directories", claims).await?; @@ -636,7 +644,7 @@ pub(super) async fn remove_old_claims(files: &SnapshotFiles, keep: &Path) { .iter() .map(AsRef::as_ref) .chain(blobs.iter().map(|blob| blob.path.as_ref())); - stream::iter(claim_directories_except(listed, keep)) + stream::iter(old_claim_directories(listed, ended)) .for_each(|directory| async move { if let Err(error) = files.delete_dir("delete_claims", &directory).await { warn!( @@ -653,9 +661,9 @@ mod tests { use super::super::files::SnapshotFiles; use super::{ CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, ClaimEntry, FREED_PATH, FreedRecord, - FreedRecords, LEDGERS_PATH, Percent, PruneLedger, claim_directories_except, claim_hold, - claims_directory, keep_claim_fresh, list_claims, list_freed, needs_repository_size, - newest_ledger, next_claim, older_entries, parse_claim_entry, parse_freed, + FreedRecords, LEDGERS_PATH, Percent, PruneLedger, claim_hold, claims_directory, + keep_claim_fresh, list_claims, list_freed, needs_repository_size, newest_ledger, + next_claim, old_claim_directories, older_entries, parse_claim_entry, parse_freed, parse_ledger_entry, parse_record, prune_due, read_ledger, record_content, record_freed, settle, take_claim, write_ledger, }; @@ -966,13 +974,13 @@ mod tests { } #[test] - fn each_claim_directory_other_than_the_new_one_is_old() { + fn a_claim_directory_is_old_when_it_is_none_or_its_time_is_before_the_new_ledger() { let claim = |directory: &str, number: &str| Path::new(CLAIMS_PATH).join(directory).join(number); let directory = |name: &str| Path::new(CLAIMS_PATH).join(name); assert_eq!( - claim_directories_except( + old_claim_directories( [ claim("none", "0"), claim("100", "0"), @@ -982,10 +990,13 @@ mod tests { claim("260", "made/below"), claim("300", "0"), directory("300"), + claim("301", "0"), + directory("400"), + claim("x1", "0"), Path::new(LEDGERS_PATH).join("5-0-a"), PathBuf::from(CLAIMS_PATH), ], - &directory("300") + Timestamp::from(300) ) .to_vec(), ["100", "200", "250", "260", "none"].map(|name| directory(name).into_boxed_path()) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 6cdbf2d2b0..54562199b5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -27,7 +27,7 @@ use super::fault::{ use super::files::SnapshotFiles; use super::priority::LowPriority; use super::prune::{ - ClaimChoice, Percent, PruneLedger, claims_directory, keep_claim_fresh, list_claims, list_freed, + ClaimChoice, Percent, claims_directory, keep_claim_fresh, list_claims, list_freed, needs_repository_size, next_claim, prune_due, read_ledger, record_freed, refresh_period, release_claim, remove_freed, remove_old_claims, remove_older_ledgers, repository_bytes, take_claim, write_ledger, write_marker, @@ -433,11 +433,7 @@ impl RusticSnapshotStore { .map_err(storage_failure)?; remove_older_ledgers(&files, ended).await; remove_freed(&files, &records).await; - let new_claims = claims_directory(&PruneLedger { - last_prune: Some(ended), - awaiting_removal: marked_packs, - }); - remove_old_claims(&files, &new_claims).await; + remove_old_claims(&files, ended).await; Ok(()) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 497f1bb05c..91c3077efa 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1366,6 +1366,44 @@ async fn a_prune_deletes_the_claims_of_old_ledgers() { ); } +#[test] +#[timeout("60s")] +async fn a_prune_keeps_the_claims_of_a_newer_ledger() { + // A newer claim directory can hold a live claim of a later prune, so the cleanup of a prune + // leaves it. Its time is far after the end of this prune. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let newer = "golem/prune-claims/99999999999999/0"; + let not_a_time = "golem/prune-claims/later/0"; + futures::stream::iter(["golem/prune-claims/100/0", newer, not_a_time]) + .for_each(|path| { + let storage = storage.clone(); + let scope = scope.clone(); + async move { + storage + .put_raw("test", "test", scope.0.clone(), Path::new(path), b"") + .await + .unwrap(); + } + }) + .await; + + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert_eq!( + ( + ledger(&storage, &scope).await.last_prune.is_some(), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, vec![newer.to_string(), not_a_time.to_string()]) + ); +} + #[test] #[timeout("60s")] async fn a_prune_deletes_an_empty_claim_directory_of_an_old_ledger() { From 31309a3fbb632d3d61c736da2018c990b3d2d62c Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 02:51:38 -0700 Subject: [PATCH 078/126] Fence a prune with a lease that each marker write moves, so a prune whose claim can go stale stops --- .../src/filesystem_snapshot/rustic/backend.rs | 75 +++++- .../src/filesystem_snapshot/rustic/fault.rs | 18 ++ .../src/filesystem_snapshot/rustic/prune.rs | 236 ++++++++++++++---- .../filesystem_snapshot/rustic/scripted.rs | 6 + .../src/filesystem_snapshot/rustic/store.rs | 34 ++- .../filesystem_snapshot/rustic/store/tests.rs | 136 +++++++++- 6 files changed, 442 insertions(+), 63 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index ef98482519..a2da8242cf 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -19,7 +19,7 @@ //! and each call waits for the blob storage on the runtime that the backend holds. Each call waits //! for at most a deadline, and a cancelled operation makes no more calls. -use super::fault::{BlobCallFailed, ConfigExists, FileMissing, OperationCancelled}; +use super::fault::{BlobCallFailed, ConfigExists, FileMissing, LeaseExpired, OperationCancelled}; use super::files::TARGET_LABEL; use super::publish::{SnapshotStage, StagedSnapshot}; use bytes::Bytes; @@ -32,8 +32,8 @@ use rustic_core::{ }; use std::future::Future; use std::path::{Path, PathBuf}; -use std::sync::Arc; -use std::time::Duration; +use std::sync::{Arc, Mutex, PoisonError}; +use std::time::{Duration, Instant}; use tokio::runtime::Handle; use tokio_util::sync::CancellationToken; use tokio_util::task::task_tracker::TaskTrackerToken; @@ -44,6 +44,35 @@ pub(super) const CONFIG_PATH: &str = "config"; /// The largest number of bytes of tree packs that one backend keeps in memory. const KEPT_PACKS_LIMIT: usize = 32 * 1024 * 1024; +/// The time until which a prune may make storage calls. A prune holds its claim until the time in +/// its newest marker plus the hold, as the other deletes see it. The lease ends before that, so a +/// prune stops before another delete can take its claim over. +#[derive(Debug)] +pub(super) struct Lease { + expiry: Mutex, +} + +impl Lease { + /// Gives a lease that ends at the instant. + pub(super) fn until(expiry: Instant) -> Self { + Self { + expiry: Mutex::new(expiry), + } + } + + /// Moves the end of the lease to the instant when that is later. A marker write that ends late + /// gives the instant from the start of the write, so it never moves the end beyond that. + pub(super) fn extend_to(&self, expiry: Instant) { + let mut current = self.expiry.lock().unwrap_or_else(PoisonError::into_inner); + *current = (*current).max(expiry); + } + + /// Gives the end of the lease. + pub(super) fn expiry(&self) -> Instant { + *self.expiry.lock().unwrap_or_else(PoisonError::into_inner) + } +} + /// A call that the backend makes on the blob storage. #[derive(Clone, Copy, Debug, PartialEq, Eq)] enum StorageCall { @@ -100,6 +129,9 @@ pub(super) struct BlobBackend { _tracked: Option, /// The packs of tree blobs that the operation of the backend read. kept: KeptPacks, + /// The lease of the prune of the backend. A backend without a lease has no limit other than + /// the deadline of each call. + lease: Option>, } impl BlobBackend { @@ -120,6 +152,7 @@ impl BlobBackend { stage: None, _tracked: None, kept: KeptPacks::new(KEPT_PACKS_LIMIT), + lease: None, } } @@ -145,6 +178,15 @@ impl BlobBackend { } } + /// Gives the backend with the lease of a prune. When the lease has run out, a call gives an + /// error at once, and a call that runs gives an error when the lease runs out. + pub(super) fn leased_by(self, lease: Arc) -> Self { + Self { + lease: Some(lease), + ..self + } + } + /// Gives the backend with a token of a task tracker. The tracker counts the backend until the /// last owner drops it, for example a thread of rustic. pub(super) fn tracked_by(self, token: TaskTrackerToken) -> Self { @@ -155,16 +197,22 @@ impl BlobBackend { } /// Waits for one call on the blob storage, which each call of the backend goes through. A call - /// without an answer within the deadline, or of a cancelled operation, gives an error, the same - /// as a call that failed. + /// without an answer within the deadline, of a cancelled operation, or of a backend whose lease + /// ran out, gives an error, the same as a call that failed. fn request( &self, call: StorageCall, path: &Path, future: impl Future>, ) -> RusticResult { + let answer = answer_or_cancel(self.deadline, &self.cancel, future); self.runtime - .block_on(answer_or_cancel(self.deadline, &self.cancel, future)) + .block_on(async { + match &self.lease { + None => answer.await, + Some(lease) => within_lease(lease, answer).await, + } + }) .map_err(|error| storage_error(call, path, error)) } @@ -184,6 +232,21 @@ impl BlobBackend { } } +/// Gives the output of the future, or [`LeaseExpired`] when the lease runs out first. A call does +/// not start when the lease has run out. +async fn within_lease( + lease: &Lease, + future: impl Future>, +) -> anyhow::Result { + let left = lease.expiry().saturating_duration_since(Instant::now()); + if left.is_zero() { + return Err(anyhow::Error::new(LeaseExpired)); + } + tokio::time::timeout(left, future) + .await + .unwrap_or_else(|_| Err(anyhow::Error::new(LeaseExpired))) +} + /// Gives the output of the future, or an error when the future gives no output within the deadline. /// /// The timer starts at the first poll of the returned future, in the runtime of that poll. Thus a diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs index 124ec28c5c..ad812d2364 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs @@ -62,6 +62,24 @@ impl Display for OperationCancelled { impl Error for OperationCancelled {} +/// The lease of the prune ran out, so the backend made no more calls. Another delete can then take +/// the claim of the prune, so the prune must not change the repository any more. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct LeaseExpired; + +impl Display for LeaseExpired { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("the lease of the prune claim ran out") + } +} + +impl Error for LeaseExpired {} + +/// Tells whether an error in the chain is [`LeaseExpired`]. +pub(super) fn is_lease_expired(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| error.is::()) +} + /// Another writer made the config file of the repository first. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub(super) struct ConfigExists; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 3313c52a5c..c38d9dc994 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -24,6 +24,7 @@ //! ledger make one prune. The claims of a ledger are in one directory, named by the time of the last //! prune in that ledger. +use super::backend::Lease; use super::files::SnapshotFiles; use futures::{StreamExt, TryStreamExt, stream}; use golem_common::model::Timestamp; @@ -31,7 +32,7 @@ use golem_service_base::storage::blob::{ListedBlob, PutIfAbsent}; use std::collections::HashSet; use std::path::Path; use std::sync::{Mutex, PoisonError}; -use std::time::Duration; +use std::time::{Duration, Instant}; use tracing::warn; /// The directory of the ledger entries, relative to the root of the namespace of the scope. @@ -415,20 +416,28 @@ pub(super) enum ClaimChoice { Held, } -/// Gives how long a marker holds the claims of its ledger: the grace period, but at least the time -/// between two refreshes and two margins for clock skew, and then one more margin. -pub(super) fn claim_hold(grace: Duration) -> Duration { - grace - .max(refresh_period(grace).saturating_add(CLOCK_SKEW_MARGIN.saturating_mul(2))) +/// Gives how long a marker write that succeeds lets the prune of its claim go on: the grace period +/// less one storage call deadline, but at least one deadline, so the lease is above zero for each +/// deadline above zero. +pub(super) fn lease_span(grace: Duration, deadline: Duration) -> Duration { + grace.saturating_sub(deadline).max(deadline) +} + +/// Gives how long a marker holds the claims of its ledger: the lease, then one margin for clock +/// skew, then one storage call deadline. A marker write that the storage received can still land up +/// to one deadline after the call gave up, and another delete sees the marker time with up to one +/// margin of skew. So a prune whose lease ran out stops before another delete can take its claim. +pub(super) fn claim_hold(grace: Duration, deadline: Duration) -> Duration { + lease_span(grace, deadline) .saturating_add(CLOCK_SKEW_MARGIN) + .saturating_add(deadline) } /// Chooses the claim of a delete from the entries of the claim directory of its ledger. Any marker /// younger than the hold holds the ledger, whatever its number. A marker whose time is more than /// the margin after `now` counts as missing, and a claim without a young marker is old. Otherwise /// the delete takes the number after the largest one, or 0. -pub(super) fn next_claim(entries: &[ClaimEntry], now: Timestamp, grace: Duration) -> ClaimChoice { - let hold = claim_hold(grace); +pub(super) fn next_claim(entries: &[ClaimEntry], now: Timestamp, hold: Duration) -> ClaimChoice { let held = entries.iter().any(|entry| match entry { ClaimEntry::Marker(_, at) => !beyond_margin(*at, now) && !passed_since(*at, now, hold), ClaimEntry::Claim(_) => false, @@ -486,16 +495,33 @@ pub(super) async fn write_marker( Ok(path.into_boxed_path()) } +/// Writes a marker of the claim with the number, and moves the end of the lease to `span` after +/// the start of the write when the write succeeds. +async fn write_leased_marker( + files: &SnapshotFiles, + op_label: &'static str, + directory: &Path, + number: u64, + lease: &Lease, + span: Duration, +) -> anyhow::Result> { + let started = Instant::now(); + let marker = write_marker(files, op_label, directory, number, Timestamp::now_utc()).await?; + lease.extend_to(started + span); + Ok(marker) +} + /// Writes the first marker of the claim with the number, then takes the claim, and gives the path /// of that marker when this delete holds the claim. A delete that loses the claim deletes its -/// marker. +/// marker. The marker write moves the end of the lease. pub(super) async fn take_claim( files: &SnapshotFiles, directory: &Path, number: u64, - now: Timestamp, + lease: &Lease, + span: Duration, ) -> anyhow::Result>> { - let marker = write_marker(files, "write_marker", directory, number, now).await?; + let marker = write_leased_marker(files, "write_marker", directory, number, lease, span).await?; let written = files .put_if_absent("write_claim", &directory.join(number.to_string()), &[]) .await?; @@ -512,37 +538,36 @@ pub(super) async fn take_claim( } /// Gives the time between two markers of a live claim: a fourth of the grace period, or a fourth -/// of the margin for clock skew when the grace period is zero. -pub(super) fn refresh_period(grace: Duration) -> Duration { - if grace.is_zero() { - CLOCK_SKEW_MARGIN / 4 +/// of the margin for clock skew when the grace period is zero, but at most a fourth of the lease, so +/// each lease has at least two refreshes. +pub(super) fn refresh_period(grace: Duration, deadline: Duration) -> Duration { + let base = if grace.is_zero() { + CLOCK_SKEW_MARGIN } else { - grace / 4 - } + grace + }; + (base / 4).min(lease_span(grace, deadline) / 4) } /// Writes a new marker of the claim with the number at each period, until the caller drops the /// future or the operation of the files is cancelled, and adds the path of each written marker to -/// `written`. A failed write gives a warning. +/// `written`. Each write that succeeds moves the end of the lease to `span` after its start. A +/// failed write gives a warning, and the next period tries again. pub(super) async fn keep_claim_fresh( files: &SnapshotFiles, directory: &Path, number: u64, period: Duration, written: &Mutex>>, + lease: &Lease, + span: Duration, ) { stream::repeat(()) .then(|()| tokio::time::sleep(period)) .take_until(files.cancel.cancelled()) .for_each(|()| async move { - let marker = write_marker( - files, - "refresh_claim", - directory, - number, - Timestamp::now_utc(), - ) - .await; + let marker = + write_leased_marker(files, "refresh_claim", directory, number, lease, span).await; match marker { Ok(path) => written .lock() @@ -659,16 +684,19 @@ pub(super) async fn remove_old_claims(files: &SnapshotFiles, ended: Timestamp) { #[cfg(test)] mod tests { use super::super::files::SnapshotFiles; + use super::super::scripted::{Script, ScriptedBlobStorage}; use super::{ CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, ClaimEntry, FREED_PATH, FreedRecord, - FreedRecords, LEDGERS_PATH, Percent, PruneLedger, claim_hold, claims_directory, - keep_claim_fresh, list_claims, list_freed, needs_repository_size, newest_ledger, - next_claim, old_claim_directories, older_entries, parse_claim_entry, parse_freed, - parse_ledger_entry, parse_record, prune_due, read_ledger, record_content, record_freed, - settle, take_claim, write_ledger, + FreedRecords, LEDGERS_PATH, Lease, Percent, PruneLedger, claim_hold, claims_directory, + keep_claim_fresh, lease_span, list_claims, list_freed, needs_repository_size, + newest_ledger, next_claim, old_claim_directories, older_entries, parse_claim_entry, + parse_freed, parse_ledger_entry, parse_record, prune_due, read_ledger, record_content, + record_freed, refresh_period, settle, take_claim, write_ledger, }; + use futures::StreamExt; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; + use golem_service_base::storage::blob::BlobStorage; use golem_service_base::storage::blob::BlobStorageNamespace; use golem_service_base::storage::blob::ListedBlob; use golem_service_base::storage::blob::memory::InMemoryBlobStorage; @@ -676,7 +704,8 @@ mod tests { use std::path::{Path, PathBuf}; use std::sync::Arc; use std::time::Duration; - use test_r::test; + use std::time::Instant; + use test_r::{test, timeout}; use uuid::Uuid; const TEN_PERCENT: Percent = Percent(10); @@ -697,8 +726,12 @@ mod tests { } fn new_files() -> SnapshotFiles { + files_over(Arc::new(InMemoryBlobStorage::new())) + } + + fn files_over(storage: Arc) -> SnapshotFiles { SnapshotFiles { - storage: Arc::new(InMemoryBlobStorage::new()), + storage, namespace: BlobStorageNamespace::InitialAgentFiles { environment_id: EnvironmentId(Uuid::new_v4()), }, @@ -820,7 +853,7 @@ mod tests { next_claim( &[ClaimEntry::Claim(0), ClaimEntry::Marker(0, at(claimed_at))], at(now), - GRACE, + claim_hold(GRACE, DEADLINE), ) }; @@ -858,7 +891,8 @@ mod tests { let now = 10_000_000; let claim = ClaimEntry::Claim; let marker = |number, millis| ClaimEntry::Marker(number, at(millis)); - let choose = |entries: &[ClaimEntry]| next_claim(entries, at(now), GRACE); + let choose = + |entries: &[ClaimEntry]| next_claim(entries, at(now), claim_hold(GRACE, DEADLINE)); assert_eq!( [ @@ -881,21 +915,70 @@ mod tests { } #[test] - fn the_hold_is_at_least_the_refresh_period_and_two_margins_and_then_one_margin() { + fn the_lease_is_the_grace_period_less_one_deadline_and_at_least_one_deadline() { + let second = Duration::from_secs(1); + let minute = Duration::from_secs(60); + + assert_eq!( + [ + lease_span(GRACE, minute), + lease_span(2 * minute, minute), + lease_span(2 * minute - second, minute), + lease_span(Duration::ZERO, minute), + lease_span(minute, Duration::ZERO), + lease_span(Duration::ZERO, second), + ], + [GRACE - minute, minute, minute, minute, minute, second] + ); + } + + #[test] + fn the_hold_is_the_lease_then_one_margin_then_one_deadline() { let margin = CLOCK_SKEW_MARGIN; + let minute = Duration::from_secs(60); assert_eq!( [ - claim_hold(GRACE), - claim_hold(Duration::from_secs(60)), - claim_hold(Duration::ZERO), + claim_hold(GRACE, minute), + claim_hold(GRACE, DEADLINE), + claim_hold(Duration::ZERO, minute), + claim_hold(Duration::ZERO, Duration::from_millis(1)), ], [ GRACE + margin, - Duration::from_secs(15) + margin * 3, - margin / 4 + margin * 3, + GRACE + margin, + minute + margin + minute, + Duration::from_millis(1) + margin + Duration::from_millis(1), + ] + ); + } + + #[test] + fn each_lease_has_at_least_two_refreshes() { + let minute = Duration::from_secs(60); + let cases = [ + (GRACE, minute), + (GRACE, DEADLINE), + (Duration::ZERO, minute), + (Duration::ZERO, Duration::from_millis(1)), + (Duration::from_millis(16), minute), + (Duration::from_secs(2), Duration::from_secs(1)), + ]; + + assert_eq!( + cases.map(|(grace, deadline)| refresh_period(grace, deadline)), + [ + (GRACE - minute) / 4, + (GRACE - DEADLINE) / 4, + minute / 4, + Duration::from_micros(250), + Duration::from_millis(4), + Duration::from_millis(250), ] ); + assert!(cases.iter().all(|(grace, deadline)| { + refresh_period(*grace, *deadline) * 2 <= lease_span(*grace, *deadline) + })); } #[test] @@ -933,6 +1016,8 @@ mod tests { 0, Duration::from_secs(3600), &std::sync::Mutex::default(), + &Lease::until(Instant::now()), + GRACE, ), ) .await @@ -941,14 +1026,71 @@ mod tests { assert!(ended); } + #[test] + #[timeout("60s")] + async fn a_refresh_that_ends_late_moves_the_lease_only_to_its_start_plus_the_span() { + // Each refresh write takes the delay. A lease from the end of the write would be later + // than a lease from its start by the delay. + let delay = Duration::from_millis(300); + let span = Duration::from_secs(1); + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), move |op_label, _| { + if op_label == "refresh_claim" { + Script::Delay(delay) + } else { + Script::Pass + } + }); + let files = files_over(storage); + let lease = Lease::until(Instant::now()); + let written = std::sync::Mutex::default(); + let directory = claims_directory(&ledger(None, false)); + let started = Instant::now(); + + let ended = tokio::select! { + () = keep_claim_fresh( + &files, + &directory, + 0, + Duration::from_millis(10), + &written, + &lease, + span, + ) => None, + ended = async { + futures::stream::repeat(()) + .then(|()| tokio::time::sleep(Duration::from_millis(5))) + .take_while(|()| { + std::future::ready( + written.lock().unwrap_or_else(std::sync::PoisonError::into_inner).is_empty(), + ) + }) + .for_each(|()| std::future::ready(())) + .await; + Instant::now() + } => Some(ended), + }; + + let ended = ended.unwrap(); + let expiry = lease.expiry(); + assert!(expiry >= started + span, "the lease did not move"); + assert!( + expiry + delay / 2 < ended + span, + "the lease moved to the end of the write" + ); + } + #[test] async fn a_claim_is_taken_after_its_marker_and_a_loser_deletes_its_marker() { let files = new_files(); let directory = claims_directory(&ledger(Some(42), false)); - let now = at(10_000_000); + let started = Instant::now(); - let first = take_claim(&files, &directory, 0, now).await.unwrap(); - let again = take_claim(&files, &directory, 0, at(20_000_000)) + let lease = Lease::until(started); + let first = take_claim(&files, &directory, 0, &lease, GRACE) + .await + .unwrap(); + let again = take_claim(&files, &directory, 0, &lease, GRACE) .await .unwrap(); let listed = list_claims(&files, &directory).await.unwrap(); @@ -960,7 +1102,10 @@ mod tests { again.is_some(), listed.len(), listed.contains(&ClaimEntry::Claim(0)), - listed.contains(&ClaimEntry::Marker(0, now)), + listed + .iter() + .any(|entry| matches!(entry, ClaimEntry::Marker(0, _))), + lease.expiry() >= started + GRACE, ), ( "golem/prune-claims/42".to_string(), @@ -968,6 +1113,7 @@ mod tests { false, 2, true, + true, true ) ); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs index 9520c885fd..aaa5d50908 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -45,6 +45,8 @@ pub(super) enum Script { NeverAnswer, /// Waits until the test opens the gate of the storage, and then passes the call. WaitForGate, + /// Waits for the time, and then passes the call. + Delay(std::time::Duration), /// Gives no blob to a read of a whole blob, as a delete after a listing does. Each other call /// passes. Vanish, @@ -232,6 +234,10 @@ impl ScriptedBlobStorage { call.await } Script::Vanish | Script::AnswerAlreadyExists => call.await, + Script::Delay(time) => { + tokio::time::sleep(time).await; + call.await + } Script::Step { refuse, .. } => { let _stepped = self.wait_for_step(op_label, path).await; if refuse { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 54562199b5..f2a47a5d6c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -20,17 +20,17 @@ //! next storage call. The store counts each blocking task and each backend in a task tracker, and //! [`RusticSnapshotStore::shut_down`] waits for them. -use super::backend::BlobBackend; +use super::backend::{BlobBackend, Lease}; use super::fault::{ Operation, classify, is_file_missing, is_snapshot_missing, is_storage_failure, storage_failure, }; use super::files::SnapshotFiles; use super::priority::LowPriority; use super::prune::{ - ClaimChoice, Percent, claims_directory, keep_claim_fresh, list_claims, list_freed, - needs_repository_size, next_claim, prune_due, read_ledger, record_freed, refresh_period, - release_claim, remove_freed, remove_old_claims, remove_older_ledgers, repository_bytes, - take_claim, write_ledger, write_marker, + ClaimChoice, Percent, claim_hold, claims_directory, keep_claim_fresh, lease_span, list_claims, + list_freed, needs_repository_size, next_claim, prune_due, read_ledger, record_freed, + refresh_period, release_claim, remove_freed, remove_old_claims, remove_older_ledgers, + repository_bytes, take_claim, write_ledger, write_marker, }; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -62,7 +62,7 @@ use std::num::NonZeroUsize; use std::path::Path; use std::pin::pin; use std::sync::{Arc, Mutex, PoisonError}; -use std::time::Duration; +use std::time::{Duration, Instant}; use tokio::runtime::Handle; use tokio_util::sync::{CancellationToken, DropGuard}; use tokio_util::task::TaskTracker; @@ -189,6 +189,10 @@ pub(super) struct PublishGate { struct Claim<'a> { directory: &'a Path, number: u64, + /// The lease of the prune. Each marker write that succeeds moves its end. + lease: Arc, + /// The time that a marker write that succeeds adds to the lease. + span: Duration, /// The markers of the claim that this delete wrote: the first marker, and each new marker /// while the prune runs. markers: Mutex>>, @@ -378,10 +382,14 @@ impl RusticSnapshotStore { let listed = list_claims(&files, &claims) .await .map_err(storage_failure)?; - let ClaimChoice::Claim(number) = next_claim(&listed, now, grace) else { + let deadline = self.policy.deadline; + let ClaimChoice::Claim(number) = next_claim(&listed, now, claim_hold(grace, deadline)) + else { return Ok(()); }; - let Some(marker) = take_claim(&files, &claims, number, now) + let lease = Arc::new(Lease::until(Instant::now())); + let span = lease_span(grace, deadline); + let Some(marker) = take_claim(&files, &claims, number, &lease, span) .await .map_err(storage_failure)? else { @@ -390,6 +398,8 @@ impl RusticSnapshotStore { let claim = Claim { directory: &claims, number, + lease, + span, markers: Mutex::new(vec![marker]), }; // Only an error before the prune starts releases the claim, so a retry of the delete @@ -477,7 +487,9 @@ impl RusticSnapshotStore { if *claims_directory(&again) != *claim.directory { return Ok(None); } - Ok(Some(Arc::new(self.backend(scope, token)?))) + Ok(Some(Arc::new( + self.backend(scope, token)?.leased_by(claim.lease.clone()), + ))) } /// Runs the prune with a new marker of the claim at each refresh period, and tells whether the @@ -515,8 +527,10 @@ impl RusticSnapshotStore { files, claim.directory, claim.number, - refresh_period(grace), + refresh_period(grace, self.policy.deadline), &claim.markers, + &claim.lease, + claim.span, ); let attempts = self .tracker diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 91c3077efa..a515193f84 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -17,10 +17,11 @@ //! The contract suite runs on the store with the policy of the configuration. The other tests //! give the store a short or a long deadline and a prune policy that the test controls. +use super::super::fault::is_lease_expired; use super::super::files::SnapshotFiles; use super::super::prune::{ - CLOCK_SKEW_MARGIN, ClaimEntry, LEDGERS_PATH, Percent, PruneLedger, parse_claim_entry, - parse_freed, read_ledger, + CLOCK_SKEW_MARGIN, ClaimChoice, ClaimEntry, LEDGERS_PATH, Percent, PruneLedger, claim_hold, + next_claim, parse_claim_entry, parse_freed, read_ledger, }; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; @@ -1332,6 +1333,137 @@ async fn a_prune_slower_than_the_grace_period_keeps_its_claim_fresh() { ); } +/// Tells whether the operation label is a call of a rustic backend. +fn is_backend_call(op_label: &str) -> bool { + matches!( + op_label, + "stat" | "list" | "read" | "read_range" | "write" | "delete" + ) +} + +/// Gives the entries of the claims of the scope. +async fn claim_entries(storage: &ScriptedBlobStorage, scope: &SnapshotScope) -> Vec { + blobs(storage, &scope.0, "golem/prune-claims/") + .await + .iter() + .filter_map(|path| parse_claim_entry(Path::new(path).file_name()?.to_str()?)) + .collect() +} + +#[test] +#[timeout("60s")] +async fn a_prune_whose_refreshes_fail_stops_when_its_lease_runs_out_and_keeps_its_claim() { + // A zero grace period and a deadline of 200 ms give a lease of 200 ms. The claim write and the + // second read of the ledger each take 150 ms, so the lease has run out when the prune makes its + // first call. Each refresh fails. + let deadline = Duration::from_millis(200); + let slow = Duration::from_millis(150); + let claimed = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let claimed = claimed.clone(); + move |op_label, _| match op_label { + "write_claim" => { + claimed.store(true, Ordering::SeqCst); + Script::Delay(slow) + } + "read_ledger" if claimed.load(Ordering::SeqCst) => Script::Delay(slow), + "refresh_claim" => Script::Refuse, + _ => Script::Pass, + } + }); + let store = store(storage.clone(), policy(deadline, ALWAYS, Duration::ZERO)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + let calls = storage.calls(); + let backend_calls_after_claim = calls + .iter() + .position(|(op_label, _)| *op_label == "write_claim") + .map(|claimed_at| { + calls[claimed_at..] + .iter() + .filter(|(op_label, _)| is_backend_call(op_label)) + .count() + }); + let entries = claim_entries(&storage, &scope).await; + let hold = claim_hold(Duration::ZERO, deadline); + let now = golem_common::model::Timestamp::now_utc(); + let after_hold = golem_common::model::Timestamp::from( + now.to_millis() + u64::try_from((hold + Duration::from_secs(1)).as_millis()).unwrap(), + ); + + assert!( + matches!( + &deleted, + Err(error @ SnapshotStoreError::Storage { source, .. }) + if is_storage(error, true) && is_lease_expired(source.as_ref()) + ), + "{deleted:?}" + ); + assert_eq!( + ( + backend_calls_after_claim, + entries.contains(&ClaimEntry::Claim(0)), + next_claim(&entries, now, hold), + next_claim(&entries, after_hold, hold), + ), + (Some(0), true, ClaimChoice::Held, ClaimChoice::Claim(1)) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_goes_on_after_one_failed_refresh_when_the_later_refreshes_succeed() { + // A zero grace period and a deadline of 200 ms give a lease of 200 ms and a refresh each 50 ms. + // Each call of the prune takes 40 ms, so the prune runs for more than one lease. The first + // refresh fails, and the later ones succeed. + let deadline = Duration::from_millis(200); + let claimed = Arc::new(AtomicBool::new(false)); + let refused = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, refused) = (claimed.clone(), refused.clone()); + move |op_label, _| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "refresh_claim" && !refused.swap(true, Ordering::SeqCst) { + Script::Refuse + } else if is_backend_call(op_label) && claimed.load(Ordering::SeqCst) { + Script::Delay(Duration::from_millis(40)) + } else { + Script::Pass + } + } + }); + let store = store(storage.clone(), policy(deadline, ALWAYS, Duration::ZERO)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + let calls = storage.calls(); + let backend_calls_after_claim = calls + .iter() + .position(|(op_label, _)| *op_label == "write_claim") + .map_or(0, |claimed_at| { + calls[claimed_at..] + .iter() + .filter(|(op_label, _)| is_backend_call(op_label)) + .count() + }); + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + refused.load(Ordering::SeqCst), + backend_calls_after_claim * 40 > 200, + prunes(&calls), + ledger(&storage, &scope).await.last_prune.is_some(), + ), + (true, true, 1, true) + ); +} + #[test] #[timeout("60s")] async fn a_prune_deletes_the_claims_of_old_ledgers() { From cd281303895e6c17797e39071264d9dd5b5cf950 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 07:25:57 -0700 Subject: [PATCH 079/126] Hold a prune claim in a guard from its first marker until the prune starts, and release it when the delete stops before that --- .../src/filesystem_snapshot/rustic/prune.rs | 87 +++++--- .../src/filesystem_snapshot/rustic/store.rs | 206 ++++++++++++++---- .../filesystem_snapshot/rustic/store/tests.rs | 123 +++++++++++ 3 files changed, 348 insertions(+), 68 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index c38d9dc994..bc0575aa28 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -477,8 +477,29 @@ pub(super) async fn list_claims( .collect()) } -/// Writes a marker of the claim with the number, with the time, and gives its path. The name is -/// unique, so `AlreadyExists` means that an earlier try of this call wrote it. +/// Gives a new path of a marker of the claim with the number, with the time. The name is unique. +pub(super) fn marker_path(directory: &Path, number: u64, time: Timestamp) -> Box { + directory + .join(format!( + "{number}@{}-{}", + time.to_millis(), + uuid::Uuid::new_v4() + )) + .into_boxed_path() +} + +/// Writes the marker at the path. The name is unique, so `AlreadyExists` means that an earlier try +/// of this call wrote it. +async fn write_marker_at( + files: &SnapshotFiles, + op_label: &'static str, + path: &Path, +) -> anyhow::Result<()> { + let _: PutIfAbsent = files.put_if_absent(op_label, path, &[]).await?; + Ok(()) +} + +/// Writes a marker of the claim with the number, with the time, and gives its path. pub(super) async fn write_marker( files: &SnapshotFiles, op_label: &'static str, @@ -486,13 +507,9 @@ pub(super) async fn write_marker( number: u64, time: Timestamp, ) -> anyhow::Result> { - let path = directory.join(format!( - "{number}@{}-{}", - time.to_millis(), - uuid::Uuid::new_v4() - )); - let _: PutIfAbsent = files.put_if_absent(op_label, &path, &[]).await?; - Ok(path.into_boxed_path()) + let path = marker_path(directory, number, time); + write_marker_at(files, op_label, &path).await?; + Ok(path) } /// Writes a marker of the claim with the number, and moves the end of the lease to `span` after @@ -511,30 +528,34 @@ async fn write_leased_marker( Ok(marker) } -/// Writes the first marker of the claim with the number, then takes the claim, and gives the path -/// of that marker when this delete holds the claim. A delete that loses the claim deletes its -/// marker. The marker write moves the end of the lease. +/// Writes the first marker of the claim with the number at the path `marker`, then takes the +/// claim, and tells whether this delete holds the claim. The caller makes the path before the +/// write, so a guard can delete the marker when the delete stops during the write. A delete that +/// loses the claim deletes its marker. The marker write moves the end of the lease. pub(super) async fn take_claim( files: &SnapshotFiles, directory: &Path, number: u64, + marker: &Path, lease: &Lease, span: Duration, -) -> anyhow::Result>> { - let marker = write_leased_marker(files, "write_marker", directory, number, lease, span).await?; +) -> anyhow::Result { + let started = Instant::now(); + write_marker_at(files, "write_marker", marker).await?; + lease.extend_to(started + span); let written = files .put_if_absent("write_claim", &directory.join(number.to_string()), &[]) .await?; if written == PutIfAbsent::Written { - return Ok(Some(marker)); + return Ok(true); } - if let Err(error) = files.delete("delete_marker", &marker).await { + if let Err(error) = files.delete("delete_marker", marker).await { warn!( error = %format!("{error:#}"), "Failed to delete the marker of a prune claim that a filesystem snapshot delete lost" ); } - Ok(None) + Ok(false) } /// Gives the time between two markers of a live claim: a fourth of the grace period, or a fourth @@ -582,22 +603,27 @@ pub(super) async fn keep_claim_fresh( .await; } -/// Deletes the claim with the number, and then each of its markers by its path. It tries each -/// delete also when another one fails, and each failure gives a warning. A claim that stays -/// without its markers is old, and a marker that stays only delays a prune until its hold passed. +/// Deletes the claim with the number when this delete took it, and then each of its markers by its +/// path. It tries each delete also when another one fails, and each failure gives a warning. A +/// claim that stays without its markers is old, and a marker that stays only delays a prune until +/// its hold passed. pub(super) async fn release_claim( files: &SnapshotFiles, directory: &Path, number: u64, + claimed: bool, markers: &[Box], ) { let claim = directory.join(number.to_string()); stream::iter( - std::iter::once(("delete_claim", claim.as_path())).chain( - markers - .iter() - .map(|marker| ("delete_marker", marker.as_ref())), - ), + claimed + .then_some(("delete_claim", claim.as_path())) + .into_iter() + .chain( + markers + .iter() + .map(|marker| ("delete_marker", marker.as_ref())), + ), ) .for_each(|(op_label, path)| async move { if let Err(error) = files.delete(op_label, path).await { @@ -688,7 +714,7 @@ mod tests { use super::{ CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, ClaimEntry, FREED_PATH, FreedRecord, FreedRecords, LEDGERS_PATH, Lease, Percent, PruneLedger, claim_hold, claims_directory, - keep_claim_fresh, lease_span, list_claims, list_freed, needs_repository_size, + keep_claim_fresh, lease_span, list_claims, list_freed, marker_path, needs_repository_size, newest_ledger, next_claim, old_claim_directories, older_entries, parse_claim_entry, parse_freed, parse_ledger_entry, parse_record, prune_due, read_ledger, record_content, record_freed, refresh_period, settle, take_claim, write_ledger, @@ -1087,10 +1113,11 @@ mod tests { let started = Instant::now(); let lease = Lease::until(started); - let first = take_claim(&files, &directory, 0, &lease, GRACE) + let marker = |time| marker_path(&directory, 0, Timestamp::from(time)); + let first = take_claim(&files, &directory, 0, &marker(1), &lease, GRACE) .await .unwrap(); - let again = take_claim(&files, &directory, 0, &lease, GRACE) + let again = take_claim(&files, &directory, 0, &marker(2), &lease, GRACE) .await .unwrap(); let listed = list_claims(&files, &directory).await.unwrap(); @@ -1098,8 +1125,8 @@ mod tests { assert_eq!( ( directory.display().to_string(), - first.is_some(), - again.is_some(), + first, + again, listed.len(), listed.contains(&ClaimEntry::Claim(0)), listed diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index f2a47a5d6c..791109f046 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -28,9 +28,9 @@ use super::files::SnapshotFiles; use super::priority::LowPriority; use super::prune::{ ClaimChoice, Percent, claim_hold, claims_directory, keep_claim_fresh, lease_span, list_claims, - list_freed, needs_repository_size, next_claim, prune_due, read_ledger, record_freed, - refresh_period, release_claim, remove_freed, remove_old_claims, remove_older_ledgers, - repository_bytes, take_claim, write_ledger, write_marker, + list_freed, marker_path, needs_repository_size, next_claim, prune_due, read_ledger, + record_freed, refresh_period, release_claim, remove_freed, remove_old_claims, + remove_older_ledgers, repository_bytes, take_claim, write_ledger, write_marker, }; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -61,11 +61,13 @@ use serde::{Deserialize, Serialize}; use std::num::NonZeroUsize; use std::path::Path; use std::pin::pin; +use std::sync::atomic::{AtomicBool, AtomicU8, Ordering}; use std::sync::{Arc, Mutex, PoisonError}; use std::time::{Duration, Instant}; use tokio::runtime::Handle; use tokio_util::sync::{CancellationToken, DropGuard}; use tokio_util::task::TaskTracker; +use tokio_util::task::task_tracker::TaskTrackerToken; use tracing::warn; /// The share of the size of the repository that deleted snapshots must free before a delete prunes @@ -193,9 +195,145 @@ struct Claim<'a> { lease: Arc, /// The time that a marker write that succeeds adds to the lease. span: Duration, - /// The markers of the claim that this delete wrote: the first marker, and each new marker - /// while the prune runs. + /// Releases the claim when the delete stops before its prune starts. + guard: ClaimGuard, +} + +/// The claim is written, and its prune did not start. +const CLAIM_PENDING: u8 = 0; +/// The rustic prune of the claim started, so the claim stays. +const CLAIM_STARTED: u8 = 1; +/// The claim was released, so its prune must not start. +const CLAIM_RELEASED: u8 = 2; + +/// Holds a prune claim from the write of its first marker until the rustic prune starts. When the +/// delete drops before that, the guard releases the claim in a task, and it moves its token of the +/// tracker into that task, so `shut_down` waits for the release. The prune and the guard change the +/// state from pending with one atomic step each, so only one of them wins: a released claim never +/// starts a prune, and a started prune never loses its claim. +struct ClaimGuard { + /// The blobs of the scope, with a token that nothing cancels, so a release also runs after a + /// cancel or a drop. + files: SnapshotFiles, + directory: Box, + number: u64, + /// Whether this delete wrote the claim. A delete that did not delete only its markers. + claimed: AtomicBool, + /// The markers of the claim that this delete wrote: the first marker, and each new marker while + /// the prune runs. markers: Mutex>>, + state: Arc, + tracked: Mutex>, +} + +impl ClaimGuard { + /// Gives the guard of the claim with the number and the path of its first marker, before the + /// write of that marker. + fn new( + files: &SnapshotFiles, + directory: &Path, + number: u64, + marker: Box, + tracked: TaskTrackerToken, + ) -> Self { + Self { + files: SnapshotFiles { + cancel: CancellationToken::new(), + ..files.clone() + }, + directory: directory.into(), + number, + claimed: AtomicBool::new(false), + markers: Mutex::new(vec![marker]), + state: Arc::new(AtomicU8::new(CLAIM_PENDING)), + tracked: Mutex::new(Some(tracked)), + } + } + + /// Makes the release task, and moves the token of the tracker into it. + fn spawn_release(&self) -> Option> { + let tracked = self + .tracked + .lock() + .unwrap_or_else(PoisonError::into_inner) + .take(); + let files = self.files.clone(); + let directory = self.directory.clone(); + let number = self.number; + let claimed = self.claimed.load(Ordering::SeqCst); + let markers = self + .markers + .lock() + .unwrap_or_else(PoisonError::into_inner) + .clone(); + match Handle::try_current() { + Ok(runtime) => Some(runtime.spawn(async move { + let _tracked = tracked; + release_claim(&files, &directory, number, claimed, &markers).await; + })), + Err(error) => { + warn!( + error = %error, + "The prune claim of a filesystem snapshot scope stays, because no runtime runs its release" + ); + None + } + } + } + + /// Releases the claim and waits for the release. + async fn release(&self) { + self.state.store(CLAIM_RELEASED, Ordering::SeqCst); + if let Some(releasing) = self.spawn_release() + && let Err(error) = releasing.await + { + warn!( + error = %error, + "The release of the prune claim of a filesystem snapshot scope did not end" + ); + } + } + + /// Ends the guard without a release: another delete took the claim, and this delete already + /// deleted its marker. + fn disarm(&self) { + self.state.store(CLAIM_RELEASED, Ordering::SeqCst); + drop( + self.tracked + .lock() + .unwrap_or_else(PoisonError::into_inner) + .take(), + ); + } +} + +impl Drop for ClaimGuard { + fn drop(&mut self) { + if self + .state + .compare_exchange( + CLAIM_PENDING, + CLAIM_RELEASED, + Ordering::SeqCst, + Ordering::SeqCst, + ) + .is_ok() + { + drop(self.spawn_release()); + } + } +} + +/// Moves the claim of the state from pending to started, and tells whether the prune may start. +fn start_prune(state: &AtomicU8) -> bool { + state + .compare_exchange( + CLAIM_PENDING, + CLAIM_STARTED, + Ordering::SeqCst, + Ordering::SeqCst, + ) + .is_ok() } /// The number of times that a prune plans again when a snapshot file that it listed is gone at @@ -389,25 +527,37 @@ impl RusticSnapshotStore { }; let lease = Arc::new(Lease::until(Instant::now())); let span = lease_span(grace, deadline); - let Some(marker) = take_claim(&files, &claims, number, &lease, span) + let marker = marker_path(&claims, number, Timestamp::now_utc()); + // The guard exists before the marker write, so a drop of the delete from here until the + // prune starts releases the claim. + let guard = ClaimGuard::new( + &files, + &claims, + number, + marker.clone(), + self.tracker.token(), + ); + let won = take_claim(&files, &claims, number, &marker, &lease, span) .await - .map_err(storage_failure)? - else { + .map_err(storage_failure)?; + if !won { + guard.disarm(); return Ok(()); - }; + } + guard.claimed.store(true, Ordering::SeqCst); let claim = Claim { directory: &claims, number, lease, span, - markers: Mutex::new(vec![marker]), + guard, }; // Only an error before the prune starts releases the claim, so a retry of the delete // prunes again. A prune that started can have marked packs, so its claim stays. let backend = match self.prepare_prune(scope, token, &files, &claim).await { Ok(Some(backend)) => backend, other => { - self.release(&files, &claim).await; + claim.guard.release().await; return other.map(|_| ()); } }; @@ -415,7 +565,7 @@ impl RusticSnapshotStore { // prune ran, and the claim goes. let pruned = match self.run_prune(backend, &files, &claim, grace).await { Ok(None) => { - self.release(&files, &claim).await; + claim.guard.release().await; return Err(snapshots_changed_error()); } other => other.map(|marked| marked.unwrap_or_default()), @@ -447,31 +597,6 @@ impl RusticSnapshotStore { Ok(()) } - /// Deletes the claim and then its markers in a task that the tracker counts. The task has a - /// token of its own, so a drop of the delete, a cancel or a shut down does not stop it. - async fn release(&self, files: &SnapshotFiles, claim: &Claim<'_>) { - let files = SnapshotFiles { - cancel: CancellationToken::new(), - ..files.clone() - }; - let directory: Box = claim.directory.into(); - let markers = claim - .markers - .lock() - .unwrap_or_else(PoisonError::into_inner) - .clone(); - let number = claim.number; - let releasing = self - .tracker - .spawn(async move { release_claim(&files, &directory, number, &markers).await }); - if let Err(error) = releasing.await { - warn!( - error = %error, - "The release of the prune claim of a filesystem snapshot scope did not end" - ); - } - } - /// Checks the ledger again and builds the backend of the prune. It gives `None` when another /// prune ended after the claim. async fn prepare_prune( @@ -504,11 +629,16 @@ impl RusticSnapshotStore { let key = self.key.clone(); let settings = self.policy.prune; let low_priority = self.low_priority; + let state = claim.guard.state.clone(); // The plan of a prune reads each snapshot file before the prune changes the repository. // A forget of another delete can remove a listed file before its read, so the prune // plans again from a new listing. let pruning = self.blocking(Operation::Prune, move || { low_priority.run("fs-snap-prune", move || { + // A delete that dropped before this point released the claim, so no prune runs. + if !start_prune(&state) { + return Ok(None); + } Ok( std::iter::repeat_with(|| prune(backend.clone(), &key, &settings)) .take(PRUNE_ATTEMPTS) @@ -528,7 +658,7 @@ impl RusticSnapshotStore { claim.directory, claim.number, refresh_period(grace, self.policy.deadline), - &claim.markers, + &claim.guard.markers, &claim.lease, claim.span, ); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index a515193f84..fe34a9e48d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -2262,6 +2262,129 @@ async fn a_delete_dropped_after_a_failed_second_read_still_releases_its_claim() ); } +/// A storage whose gate holds the second read of the ledger after the claim, and also each delete +/// of a claim or a marker when `hold_release` is true. +fn holding_after_the_claim(hold_release: bool) -> Arc { + let claimed = Arc::new(AtomicBool::new(false)); + ScriptedBlobStorage::new( + Arc::new(InMemoryBlobStorage::new()), + move |op_label, _| match op_label { + "write_claim" => { + claimed.store(true, Ordering::SeqCst); + Script::Pass + } + "read_ledger" if claimed.load(Ordering::SeqCst) => Script::WaitForGate, + "delete_claim" | "delete_marker" if hold_release => Script::WaitForGate, + _ => Script::Pass, + }, + ) +} + +#[test] +#[timeout("60s")] +async fn a_delete_dropped_after_its_claim_and_before_its_prune_releases_the_claim() { + // The gate holds the second read of the ledger. The test drops the delete there, before its + // prune starts. + let storage = holding_after_the_claim(false); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let claimed = |calls: &[(&'static str, String)]| { + calls + .iter() + .filter(|(op_label, _)| *op_label == "read_ledger") + .count() + >= 2 + }; + let held = eventually(|| claimed(&storage.calls())).await; + let claims_at_drop = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + + deleting.abort(); + let dropped = deleting.await; + let ended = eventually(|| store.work_in_flight() == 0).await; + let claims_after = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + storage.open_gate(); + + assert!( + dropped.as_ref().is_err_and(|error| error.is_cancelled()), + "{dropped:?}" + ); + assert_eq!( + ( + held, + claims_at_drop.len(), + ended, + prunes(&storage.calls()), + claims_after, + ), + (true, 2, true, 0, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn shut_down_waits_for_the_release_of_a_delete_dropped_after_its_claim() { + // The gate holds the second read of the ledger and each delete of the release. The test drops + // the delete at the read, so its guard starts the release, which waits at the gate. + let storage = holding_after_the_claim(true); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .filter(|(op_label, _)| *op_label == "read_ledger") + .count() + >= 2 + }) + .await; + deleting.abort(); + let _ = deleting.await; + let releasing = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "delete_claim") + }) + .await; + + let shutting = store.shut_down(); + tokio::pin!(shutting); + let waited = tokio::time::timeout(Duration::from_millis(200), &mut shutting) + .await + .is_err(); + storage.open_gate(); + // A finished future must not be polled again, so the second wait runs only after a first wait + // that timed out. + let stopped = !waited || tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + + assert_eq!( + ( + held, + releasing, + waited, + stopped, + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, true, true, true, Vec::::new()) + ); +} + #[test] #[timeout("60s")] async fn a_shut_down_after_the_claim_still_releases_it() { From 9b84b4acba2ee68849a8d12afe25d0d1cad23365 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 07:40:56 -0700 Subject: [PATCH 080/126] Sweep a drop of a delete between its claim and its prune --- .../filesystem_snapshot/rustic/store/tests.rs | 5 +- .../rustic/store/tests/sweep.rs | 54 ++++++++++++++++--- 2 files changed, 51 insertions(+), 8 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index fe34a9e48d..289a7e5c3c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -4502,6 +4502,7 @@ async fn two_deletes_make_at_most_one_prune_in_each_order_with_up_to_two_switche turns: vec![one, two], fail: None, late: None, + drop: None, }) }) }); @@ -4533,12 +4534,14 @@ async fn two_deletes_make_at_most_one_prune_in_random_orders_with_a_failed_call( prop::collection::vec(0usize..=SWEEP_TURN, 0..=8), prop::option::of((0usize..2, 0usize..SWEEP_TURN)), prop::option::of((0usize..2, 0usize..SWEEP_TURN, 0usize..8)), + prop::option::of((0usize..2, 0usize..SWEEP_TURN)), ) - .prop_map(|(first, turns, fail, late)| sweep::Schedule { + .prop_map(|(first, turns, fail, late, drop)| sweep::Schedule { first, turns, fail, late, + drop, }); let started = std::time::Instant::now(); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs index c70453b696..fea1ae0092 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs @@ -89,13 +89,15 @@ fn is_prune_start(op_label: &str, path: &str) -> bool { /// listed turns, the delete whose turn is next runs to its end, and then the other one does. /// `fail` refuses one step of one delete. `late` makes one write or delete of one delete give no /// answer at its step, and reach the storage after the given number of further steps of the case. -/// The steps of `fail` and `late` are numbered without the writes of new markers. +/// `drop` drops one delete right after one of its steps, as a caller that stops waiting does. The +/// steps of `fail`, `late` and `drop` are numbered without the writes of new markers. #[derive(Clone, Debug)] pub(super) struct Schedule { pub(super) first: usize, pub(super) turns: Vec, pub(super) fail: Option<(usize, usize)>, pub(super) late: Option<(usize, usize, usize)>, + pub(super) drop: Option<(usize, usize)>, } /// One step that a delete took, or a late call that reached the storage, in the order of all @@ -111,17 +113,23 @@ struct Step { effect: bool, } -/// One delete of the case: its scripted storage and its task. +/// One delete of the case: its store, its scripted storage and its task. struct Delete { + store: Arc, storage: Arc, task: JoinHandle>, /// The steps that the delete took, other than the writes of new markers. taken: usize, + /// Whether the case dropped the delete. A dropped delete makes a call only for the release of + /// its claim. + dropped: bool, } impl Delete { + /// Tells whether the delete ended and its store has no work left, such as the release of the + /// claim of a dropped delete. fn finished(&self) -> bool { - self.task.is_finished() + self.task.is_finished() && self.store.work_in_flight() == 0 } } @@ -208,6 +216,15 @@ impl Case { if self.deletes[who].finished() { return Ok(()); } + if self.deletes[who].dropped { + let delete = &self.deletes[who]; + if !until(|| delete.storage.waiting_steps() > 0 || delete.finished()).await { + return Err(format!("the dropped delete {who} did not end")); + } + if delete.finished() { + return Ok(()); + } + } let storage = self.deletes[who].storage.clone(); let (before, taker) = (storage.stepped(), storage.took().len()); storage.step(); @@ -250,6 +267,25 @@ impl Case { return Err(format!("delete {who} took more than {MOST_STEPS} steps")); } self.land(false).await?; + if counted && self.schedule.drop == Some((who, number)) { + self.deletes[who].task.abort(); + self.deletes[who].dropped = true; + // An abort only asks the task to end. Its waiting call counts as waiting until the + // task ends, so the case gives no step before that. + let task = &self.deletes[who].task; + if !until(|| task.is_finished()).await { + return Err(format!("the dropped delete {who} did not end")); + } + // The drop counts as a failed step of the delete, so the rules on a delete that + // stopped after its claim apply to it. + self.log.push(Step { + delete: who, + op_label: "drop", + path: String::new(), + failed: true, + effect: false, + }); + } if settle(&self.deletes[who]).await { Ok(()) } else { @@ -333,12 +369,16 @@ pub(super) async fn run_case( }); let deleting = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, SWEEP_GRACE)); let scope = scope.clone(); - let task = - tokio::spawn(async move { deleting.delete(&scope, &name(["p-1", "p-2"][who])).await }); + let task = tokio::spawn({ + let deleting = deleting.clone(); + async move { deleting.delete(&scope, &name(["p-1", "p-2"][who])).await } + }); Delete { + store: deleting, storage, task, taken: 0, + dropped: false, } }); let mut case = Case { @@ -404,8 +444,8 @@ fn forgotten_at(log: &[Step], who: usize) -> Option { } /// Gives each claim, or marker of a claim, that a delete wrote and that stays, when the first -/// failed call of that delete came after it took its claim and before it called for the listing of -/// the packs. A claim that the other delete wrote later at the same path is not the claim of the +/// failed call of that delete, or its drop, came after it took its claim and before it called for +/// the listing of the packs. A claim that the other delete wrote later at the same path is not the claim of the /// delete. A blob whose own delete failed is left out, because no call can remove it then, and a /// claim or a marker that stays only delays a prune. fn kept_claims(log: &[Step], claims: &[String]) -> Vec { From e590b31ebf9f60ac886dacc93bbac3d34cb57010 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 07:52:12 -0700 Subject: [PATCH 081/126] Read the instant of the lease and the time in the name of a marker at one moment --- .../src/filesystem_snapshot/rustic/prune.rs | 59 ++++++++++++++----- .../src/filesystem_snapshot/rustic/store.rs | 9 +-- 2 files changed, 49 insertions(+), 19 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index bc0575aa28..c395dc5476 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -477,6 +477,13 @@ pub(super) async fn list_claims( .collect()) } +/// Gives the time of a new marker: the instant, from which a write of the marker moves the lease, +/// and the wall time in the name of the marker, both read at one moment. So the lease never ends +/// later than the hold that other deletes read from the name. +pub(super) fn marker_time() -> (Instant, Timestamp) { + (Instant::now(), Timestamp::now_utc()) +} + /// Gives a new path of a marker of the claim with the number, with the time. The name is unique. pub(super) fn marker_path(directory: &Path, number: u64, time: Timestamp) -> Box { directory @@ -522,8 +529,8 @@ async fn write_leased_marker( lease: &Lease, span: Duration, ) -> anyhow::Result> { - let started = Instant::now(); - let marker = write_marker(files, op_label, directory, number, Timestamp::now_utc()).await?; + let (started, time) = marker_time(); + let marker = write_marker(files, op_label, directory, number, time).await?; lease.extend_to(started + span); Ok(marker) } @@ -531,16 +538,17 @@ async fn write_leased_marker( /// Writes the first marker of the claim with the number at the path `marker`, then takes the /// claim, and tells whether this delete holds the claim. The caller makes the path before the /// write, so a guard can delete the marker when the delete stops during the write. A delete that -/// loses the claim deletes its marker. The marker write moves the end of the lease. +/// loses the claim deletes its marker. The marker write moves the end of the lease from `started`, +/// the instant that [`marker_time`] gave with the time in the name of the marker. pub(super) async fn take_claim( files: &SnapshotFiles, directory: &Path, number: u64, marker: &Path, + started: Instant, lease: &Lease, span: Duration, ) -> anyhow::Result { - let started = Instant::now(); write_marker_at(files, "write_marker", marker).await?; lease.extend_to(started + span); let written = files @@ -714,10 +722,10 @@ mod tests { use super::{ CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, ClaimEntry, FREED_PATH, FreedRecord, FreedRecords, LEDGERS_PATH, Lease, Percent, PruneLedger, claim_hold, claims_directory, - keep_claim_fresh, lease_span, list_claims, list_freed, marker_path, needs_repository_size, - newest_ledger, next_claim, old_claim_directories, older_entries, parse_claim_entry, - parse_freed, parse_ledger_entry, parse_record, prune_due, read_ledger, record_content, - record_freed, refresh_period, settle, take_claim, write_ledger, + keep_claim_fresh, lease_span, list_claims, list_freed, marker_path, marker_time, + needs_repository_size, newest_ledger, next_claim, old_claim_directories, older_entries, + parse_claim_entry, parse_freed, parse_ledger_entry, parse_record, prune_due, read_ledger, + record_content, record_freed, refresh_period, settle, take_claim, write_ledger, }; use futures::StreamExt; use golem_common::model::Timestamp; @@ -1113,13 +1121,34 @@ mod tests { let started = Instant::now(); let lease = Lease::until(started); - let marker = |time| marker_path(&directory, 0, Timestamp::from(time)); - let first = take_claim(&files, &directory, 0, &marker(1), &lease, GRACE) - .await - .unwrap(); - let again = take_claim(&files, &directory, 0, &marker(2), &lease, GRACE) - .await - .unwrap(); + let marker = || { + let (at, time) = marker_time(); + (marker_path(&directory, 0, time), at) + }; + let (first_marker, first_at) = marker(); + let (second_marker, second_at) = marker(); + let first = take_claim( + &files, + &directory, + 0, + &first_marker, + first_at, + &lease, + GRACE, + ) + .await + .unwrap(); + let again = take_claim( + &files, + &directory, + 0, + &second_marker, + second_at, + &lease, + GRACE, + ) + .await + .unwrap(); let listed = list_claims(&files, &directory).await.unwrap(); assert_eq!( diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 791109f046..48ae0bd4d3 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -28,8 +28,8 @@ use super::files::SnapshotFiles; use super::priority::LowPriority; use super::prune::{ ClaimChoice, Percent, claim_hold, claims_directory, keep_claim_fresh, lease_span, list_claims, - list_freed, marker_path, needs_repository_size, next_claim, prune_due, read_ledger, - record_freed, refresh_period, release_claim, remove_freed, remove_old_claims, + list_freed, marker_path, marker_time, needs_repository_size, next_claim, prune_due, + read_ledger, record_freed, refresh_period, release_claim, remove_freed, remove_old_claims, remove_older_ledgers, repository_bytes, take_claim, write_ledger, write_marker, }; use super::publish::{SnapshotStage, StagedSnapshot, publish}; @@ -527,7 +527,8 @@ impl RusticSnapshotStore { }; let lease = Arc::new(Lease::until(Instant::now())); let span = lease_span(grace, deadline); - let marker = marker_path(&claims, number, Timestamp::now_utc()); + let (started, time) = marker_time(); + let marker = marker_path(&claims, number, time); // The guard exists before the marker write, so a drop of the delete from here until the // prune starts releases the claim. let guard = ClaimGuard::new( @@ -537,7 +538,7 @@ impl RusticSnapshotStore { marker.clone(), self.tracker.token(), ); - let won = take_claim(&files, &claims, number, &marker, &lease, span) + let won = take_claim(&files, &claims, number, &marker, started, &lease, span) .await .map_err(storage_failure)?; if !won { From 9912dc436ccf9b38a4504aaf87bfa5d6c5dc1bde Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 07:54:32 -0700 Subject: [PATCH 082/126] Cancel a held call only after the storage holds it, not after a sleep --- .../src/filesystem_snapshot/rustic/backend/tests.rs | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index cfefc01d8b..d971620253 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -775,8 +775,14 @@ fn a_cancelled_backend_makes_no_storage_call() { #[test] fn a_cancel_ends_a_call_that_runs() { let runtime = Runtime::new().unwrap(); - let (storage, _gate, _dropped) = - holding_storage(Arc::new(InMemoryBlobStorage::new()), |_, _| true); + let held = CancellationToken::new(); + let (storage, _gate, _dropped) = holding_storage(Arc::new(InMemoryBlobStorage::new()), { + let held = held.clone(); + move |_, _| { + held.cancel(); + true + } + }); let cancel = CancellationToken::new(); let backend = BlobBackend::new( storage, @@ -786,7 +792,7 @@ fn a_cancel_ends_a_call_that_runs() { ) .cancelled_by(cancel.clone()); runtime.spawn(async move { - tokio::time::sleep(Duration::from_millis(100)).await; + held.cancelled().await; cancel.cancel(); }); From bd36704fb45828bc2118ccd18492e593895f4791 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 07:59:37 -0700 Subject: [PATCH 083/126] Say that a prune whose attempts each found a snapshot file gone releases its claim after the start --- golem-worker-executor/src/filesystem_snapshot/rustic/store.rs | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 48ae0bd4d3..caf1692df9 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -210,7 +210,9 @@ const CLAIM_RELEASED: u8 = 2; /// delete drops before that, the guard releases the claim in a task, and it moves its token of the /// tracker into that task, so `shut_down` waits for the release. The prune and the guard change the /// state from pending with one atomic step each, so only one of them wins: a released claim never -/// starts a prune, and a started prune never loses its claim. +/// starts a prune, and a drop after the prune started does not release the claim. After the start, +/// only the delete releases the claim, when each attempt of the prune found a snapshot file gone, +/// because such a prune changed nothing and counts as a prune that did not run. struct ClaimGuard { /// The blobs of the scope, with a token that nothing cancels, so a release also runs after a /// cancel or a drop. From 5ab706cfd201e329a3aa45f84e66b7f07ff543fa Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 08:01:49 -0700 Subject: [PATCH 084/126] Read the instant of a marker before its wall time, and end its lease one millisecond before the hold math --- .../src/filesystem_snapshot/rustic/prune.rs | 55 ++++++++++++------- .../filesystem_snapshot/rustic/store/tests.rs | 8 +-- 2 files changed, 40 insertions(+), 23 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index c395dc5476..470aee3b00 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -416,19 +416,31 @@ pub(super) enum ClaimChoice { Held, } -/// Gives how long a marker write that succeeds lets the prune of its claim go on: the grace period -/// less one storage call deadline, but at least one deadline, so the lease is above zero for each -/// deadline above zero. -pub(super) fn lease_span(grace: Duration, deadline: Duration) -> Duration { +/// The unit of the time in the name of a marker. +const MARKER_TIME_UNIT: Duration = Duration::from_millis(1); + +/// Gives the time from the start of a marker write to the end of the lease that the marker gives, +/// as the other deletes count it from the time in the name: the grace period less one storage call +/// deadline, but at least one deadline. +fn lease_bound(grace: Duration, deadline: Duration) -> Duration { grace.saturating_sub(deadline).max(deadline) } -/// Gives how long a marker holds the claims of its ledger: the lease, then one margin for clock -/// skew, then one storage call deadline. A marker write that the storage received can still land up -/// to one deadline after the call gave up, and another delete sees the marker time with up to one -/// margin of skew. So a prune whose lease ran out stops before another delete can take its claim. +/// Gives how long a marker write that succeeds lets the prune of its claim go on: the lease bound +/// less one millisecond. The name of a marker keeps its time in whole milliseconds, so the time in +/// the name can be up to one millisecond before the time that the delete read. So the lease is above +/// zero for each deadline above one millisecond. +pub(super) fn lease_span(grace: Duration, deadline: Duration) -> Duration { + lease_bound(grace, deadline).saturating_sub(MARKER_TIME_UNIT) +} + +/// Gives how long a marker holds the claims of its ledger: the lease bound, then one margin for +/// clock skew, then one storage call deadline. A marker write that the storage received can still +/// land up to one deadline after the call gave up, and another delete sees the marker time with up +/// to one margin of skew. So a prune whose lease ran out stops before another delete can take its +/// claim. pub(super) fn claim_hold(grace: Duration, deadline: Duration) -> Duration { - lease_span(grace, deadline) + lease_bound(grace, deadline) .saturating_add(CLOCK_SKEW_MARGIN) .saturating_add(deadline) } @@ -478,10 +490,13 @@ pub(super) async fn list_claims( } /// Gives the time of a new marker: the instant, from which a write of the marker moves the lease, -/// and the wall time in the name of the marker, both read at one moment. So the lease never ends -/// later than the hold that other deletes read from the name. +/// and then the wall time in the name of the marker. The instant is read first, so the lease starts +/// no later than the time in the name, and it never ends later than the hold that other deletes +/// count from the name. pub(super) fn marker_time() -> (Instant, Timestamp) { - (Instant::now(), Timestamp::now_utc()) + let started = Instant::now(); + let time = Timestamp::now_utc(); + (started, time) } /// Gives a new path of a marker of the claim with the number, with the time. The name is unique. @@ -747,6 +762,7 @@ mod tests { /// The grace period and the margin for clock skew, in milliseconds. const HELD_MILLIS: u64 = 15 * 60 * 1000 + 2 * 60 * 1000; const DEADLINE: Duration = Duration::from_secs(2); + const MILLI: Duration = Duration::from_millis(1); fn at(millis: u64) -> Timestamp { Timestamp::from(millis) @@ -949,7 +965,8 @@ mod tests { } #[test] - fn the_lease_is_the_grace_period_less_one_deadline_and_at_least_one_deadline() { + fn the_lease_is_the_grace_period_less_one_deadline_and_at_least_one_deadline_less_one_millisecond() + { let second = Duration::from_secs(1); let minute = Duration::from_secs(60); @@ -962,7 +979,7 @@ mod tests { lease_span(minute, Duration::ZERO), lease_span(Duration::ZERO, second), ], - [GRACE - minute, minute, minute, minute, minute, second] + [GRACE - minute, minute, minute, minute, minute, second].map(|span| span - MILLI) ); } @@ -994,7 +1011,7 @@ mod tests { (GRACE, minute), (GRACE, DEADLINE), (Duration::ZERO, minute), - (Duration::ZERO, Duration::from_millis(1)), + (Duration::ZERO, 2 * MILLI), (Duration::from_millis(16), minute), (Duration::from_secs(2), Duration::from_secs(1)), ]; @@ -1002,12 +1019,12 @@ mod tests { assert_eq!( cases.map(|(grace, deadline)| refresh_period(grace, deadline)), [ - (GRACE - minute) / 4, - (GRACE - DEADLINE) / 4, - minute / 4, + (GRACE - minute - MILLI) / 4, + (GRACE - DEADLINE - MILLI) / 4, + (minute - MILLI) / 4, Duration::from_micros(250), Duration::from_millis(4), - Duration::from_millis(250), + Duration::from_micros(249_750), ] ); assert!(cases.iter().all(|(grace, deadline)| { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 289a7e5c3c..a8e35adecb 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1353,7 +1353,7 @@ async fn claim_entries(storage: &ScriptedBlobStorage, scope: &SnapshotScope) -> #[test] #[timeout("60s")] async fn a_prune_whose_refreshes_fail_stops_when_its_lease_runs_out_and_keeps_its_claim() { - // A zero grace period and a deadline of 200 ms give a lease of 200 ms. The claim write and the + // A zero grace period and a deadline of 200 ms give a lease of 199 ms. The claim write and the // second read of the ledger each take 150 ms, so the lease has run out when the prune makes its // first call. Each refresh fails. let deadline = Duration::from_millis(200); @@ -1415,9 +1415,9 @@ async fn a_prune_whose_refreshes_fail_stops_when_its_lease_runs_out_and_keeps_it #[test] #[timeout("60s")] async fn a_prune_goes_on_after_one_failed_refresh_when_the_later_refreshes_succeed() { - // A zero grace period and a deadline of 200 ms give a lease of 200 ms and a refresh each 50 ms. - // Each call of the prune takes 40 ms, so the prune runs for more than one lease. The first - // refresh fails, and the later ones succeed. + // A zero grace period and a deadline of 200 ms give a lease of 199 ms and a refresh about each + // 50 ms. Each call of the prune takes 40 ms, so the prune runs for more than one lease. The + // first refresh fails, and the later ones succeed. let deadline = Duration::from_millis(200); let claimed = Arc::new(AtomicBool::new(false)); let refused = Arc::new(AtomicBool::new(false)); From e9ea7e48bbbd1f5091654a89e5e5e0182602f5d0 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 08:03:25 -0700 Subject: [PATCH 085/126] Hold a claim for one more storage call deadline, for a final marker that lands late --- .../src/filesystem_snapshot/rustic/prune.rs | 33 +++++++++++-------- 1 file changed, 19 insertions(+), 14 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 470aee3b00..c52daea9c6 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -435,14 +435,16 @@ pub(super) fn lease_span(grace: Duration, deadline: Duration) -> Duration { } /// Gives how long a marker holds the claims of its ledger: the lease bound, then one margin for -/// clock skew, then one storage call deadline. A marker write that the storage received can still -/// land up to one deadline after the call gave up, and another delete sees the marker time with up -/// to one margin of skew. So a prune whose lease ran out stops before another delete can take its -/// claim. +/// clock skew, then two storage call deadlines. Another delete sees the marker time with up to one +/// margin of skew. A marker write that the storage received can still land up to one deadline after +/// the call gave up. A prune that its lease stopped writes its final marker after the lease ran +/// out, and that write can land up to one more deadline later. So a prune whose lease ran out stops +/// before another delete can take its claim, and the next prune waits a full hold from its final +/// marker. pub(super) fn claim_hold(grace: Duration, deadline: Duration) -> Duration { lease_bound(grace, deadline) .saturating_add(CLOCK_SKEW_MARGIN) - .saturating_add(deadline) + .saturating_add(deadline.saturating_mul(2)) } /// Chooses the claim of a delete from the entries of the claim directory of its ledger. Any marker @@ -761,6 +763,9 @@ mod tests { const GRACE: Duration = Duration::from_secs(15 * 60); /// The grace period and the margin for clock skew, in milliseconds. const HELD_MILLIS: u64 = 15 * 60 * 1000 + 2 * 60 * 1000; + /// The hold of a claim with the grace period and the deadline, in milliseconds: the grace + /// period less one deadline, then the margin, then two deadlines. + const HOLD_MILLIS: u64 = HELD_MILLIS + 2 * 1000; const DEADLINE: Duration = Duration::from_secs(2); const MILLI: Duration = Duration::from_millis(1); @@ -918,8 +923,8 @@ mod tests { due(now + margin + 1) ], [ - claim(now - grace), - claim(now - grace - margin), + claim(now - HOLD_MILLIS + 1), + claim(now - HOLD_MILLIS), claim(now + margin), claim(now + margin + 1) ] @@ -948,10 +953,10 @@ mod tests { [ choose(&[]), choose(&[claim(0), claim(1), marker(0, now - 1)]), - choose(&[claim(0), claim(1), marker(1, now - HELD_MILLIS)]), + choose(&[claim(0), claim(1), marker(1, now - HOLD_MILLIS)]), choose(&[claim(4)]), choose(&[marker(2, now - 1)]), - choose(&[claim(0), claim(3), marker(0, now - HELD_MILLIS)]), + choose(&[claim(0), claim(3), marker(0, now - HOLD_MILLIS)]), ], [ ClaimChoice::Claim(0), @@ -984,7 +989,7 @@ mod tests { } #[test] - fn the_hold_is_the_lease_then_one_margin_then_one_deadline() { + fn the_hold_is_the_lease_then_one_margin_then_two_deadlines() { let margin = CLOCK_SKEW_MARGIN; let minute = Duration::from_secs(60); @@ -996,10 +1001,10 @@ mod tests { claim_hold(Duration::ZERO, Duration::from_millis(1)), ], [ - GRACE + margin, - GRACE + margin, - minute + margin + minute, - Duration::from_millis(1) + margin + Duration::from_millis(1), + GRACE + margin + minute, + GRACE + margin + DEADLINE, + minute + margin + 2 * minute, + MILLI + margin + 2 * MILLI, ] ); } From b79f9057f103468bc456d51904386f8e7fa51da6 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 08:09:30 -0700 Subject: [PATCH 086/126] Start the lease of a prune at its first marker, and keep a lease that ran out from a refresh that started after its end --- .../src/filesystem_snapshot/rustic/backend.rs | 14 +- .../src/filesystem_snapshot/rustic/prune.rs | 192 +++++++++++------- .../src/filesystem_snapshot/rustic/store.rs | 11 +- .../filesystem_snapshot/rustic/store/tests.rs | 31 +++ 4 files changed, 168 insertions(+), 80 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index a2da8242cf..f44c2389c2 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -60,11 +60,17 @@ impl Lease { } } - /// Moves the end of the lease to the instant when that is later. A marker write that ends late - /// gives the instant from the start of the write, so it never moves the end beyond that. - pub(super) fn extend_to(&self, expiry: Instant) { + /// Moves the end of the lease to `span` after `started`, the start of a marker write that + /// succeeded, when that is later. A write that started at or after the end of the lease does + /// not move it, so a lease that ran out stays out: another delete can have taken the claim over + /// before the marker of that write was visible. A write that started before the end and ends + /// late can still move it, because no other delete can take the claim over before that marker + /// is visible. + pub(super) fn extend_from(&self, started: Instant, span: Duration) { let mut current = self.expiry.lock().unwrap_or_else(PoisonError::into_inner); - *current = (*current).max(expiry); + if started < *current { + *current = (*current).max(started + span); + } } /// Gives the end of the lease. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index c52daea9c6..2da763b737 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -548,31 +548,31 @@ async fn write_leased_marker( ) -> anyhow::Result> { let (started, time) = marker_time(); let marker = write_marker(files, op_label, directory, number, time).await?; - lease.extend_to(started + span); + lease.extend_from(started, span); Ok(marker) } /// Writes the first marker of the claim with the number at the path `marker`, then takes the -/// claim, and tells whether this delete holds the claim. The caller makes the path before the -/// write, so a guard can delete the marker when the delete stops during the write. A delete that -/// loses the claim deletes its marker. The marker write moves the end of the lease from `started`, -/// the instant that [`marker_time`] gave with the time in the name of the marker. +/// claim, and gives the lease of the prune when this delete holds the claim. The caller makes the +/// path before the write, so a guard can delete the marker when the delete stops during the write. +/// A delete that loses the claim deletes its marker. The lease starts with the marker write: it +/// ends `span` after `started`, the instant that [`marker_time`] gave with the time in the name of +/// the marker. pub(super) async fn take_claim( files: &SnapshotFiles, directory: &Path, number: u64, marker: &Path, started: Instant, - lease: &Lease, span: Duration, -) -> anyhow::Result { +) -> anyhow::Result> { write_marker_at(files, "write_marker", marker).await?; - lease.extend_to(started + span); + let lease = Lease::until(started + span); let written = files .put_if_absent("write_claim", &directory.join(number.to_string()), &[]) .await?; if written == PutIfAbsent::Written { - return Ok(true); + return Ok(Some(lease)); } if let Err(error) = files.delete("delete_marker", marker).await { warn!( @@ -580,7 +580,7 @@ pub(super) async fn take_claim( "Failed to delete the marker of a prune claim that a filesystem snapshot delete lost" ); } - Ok(false) + Ok(None) } /// Gives the time between two markers of a live claim: a fourth of the grace period, or a fourth @@ -734,6 +734,8 @@ pub(super) async fn remove_old_claims(files: &SnapshotFiles, ended: Timestamp) { #[cfg(test)] mod tests { + use super::super::backend::BlobBackend; + use super::super::fault::is_lease_expired; use super::super::files::SnapshotFiles; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::{ @@ -752,8 +754,10 @@ mod tests { use golem_service_base::storage::blob::ListedBlob; use golem_service_base::storage::blob::memory::InMemoryBlobStorage; use pretty_assertions::assert_eq; + use rustic_core::{FileType, ReadBackend}; use std::path::{Path, PathBuf}; use std::sync::Arc; + use std::sync::atomic::{AtomicBool, Ordering}; use std::time::Duration; use std::time::Instant; use test_r::{test, timeout}; @@ -1082,12 +1086,41 @@ mod tests { assert!(ended); } + /// Refreshes the claim at each period until the first refresh write succeeds, and gives the + /// instant after that write. It gives `None` when the refresh ends before a write succeeded. + async fn refresh_until_written( + files: &SnapshotFiles, + period: Duration, + lease: &Lease, + span: Duration, + ) -> Option { + let written = std::sync::Mutex::>>::default(); + let directory = claims_directory(&ledger(None, false)); + tokio::select! { + () = keep_claim_fresh(files, &directory, 0, period, &written, lease, span) => None, + ended = async { + futures::stream::repeat(()) + .then(|()| tokio::time::sleep(Duration::from_millis(5))) + .take_while(|()| { + std::future::ready( + written.lock().unwrap_or_else(std::sync::PoisonError::into_inner).is_empty(), + ) + }) + .for_each(|()| std::future::ready(())) + .await; + Instant::now() + } => Some(ended), + } + } + #[test] #[timeout("60s")] - async fn a_refresh_that_ends_late_moves_the_lease_only_to_its_start_plus_the_span() { - // Each refresh write takes the delay. A lease from the end of the write would be later - // than a lease from its start by the delay. - let delay = Duration::from_millis(300); + async fn a_refresh_that_starts_before_the_end_of_the_lease_and_ends_after_it_moves_the_lease_to_its_start_plus_the_span() + { + // The first refresh starts about 10 ms after the lease starts, before its end at 250 ms. + // Each refresh write takes the delay, so the write ends after the end of the lease. A lease + // from the end of the write would be later than a lease from its start by the delay. + let delay = Duration::from_millis(500); let span = Duration::from_secs(1); let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), move |op_label, _| { @@ -1098,37 +1131,19 @@ mod tests { } }); let files = files_over(storage); - let lease = Lease::until(Instant::now()); - let written = std::sync::Mutex::default(); - let directory = claims_directory(&ledger(None, false)); let started = Instant::now(); + let first_expiry = started + Duration::from_millis(250); + let lease = Lease::until(first_expiry); - let ended = tokio::select! { - () = keep_claim_fresh( - &files, - &directory, - 0, - Duration::from_millis(10), - &written, - &lease, - span, - ) => None, - ended = async { - futures::stream::repeat(()) - .then(|()| tokio::time::sleep(Duration::from_millis(5))) - .take_while(|()| { - std::future::ready( - written.lock().unwrap_or_else(std::sync::PoisonError::into_inner).is_empty(), - ) - }) - .for_each(|()| std::future::ready(())) - .await; - Instant::now() - } => Some(ended), - }; + let ended = refresh_until_written(&files, Duration::from_millis(10), &lease, span) + .await + .unwrap(); - let ended = ended.unwrap(); let expiry = lease.expiry(); + assert!( + ended > first_expiry, + "the write ended before the end of the lease" + ); assert!(expiry >= started + span, "the lease did not move"); assert!( expiry + delay / 2 < ended + span, @@ -1136,41 +1151,80 @@ mod tests { ); } + #[test] + #[timeout("60s")] + async fn a_refresh_that_starts_after_the_lease_ran_out_does_not_move_it_and_the_next_call_is_refused() + { + // The lease ends when the test starts. The first refresh write fails, and a later one + // succeeds, but each refresh starts after the end of the lease. Nothing calls the backend + // while the lease is out, and the first call after the refresh finds the lease still out. + let refused = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refused = refused.clone(); + move |op_label, _| { + if op_label == "refresh_claim" && !refused.swap(true, Ordering::SeqCst) { + Script::Refuse + } else { + Script::Pass + } + } + }); + let files = files_over(storage.clone()); + let expiry = Instant::now(); + let lease = Arc::new(Lease::until(expiry)); + + let refreshed = refresh_until_written( + &files, + Duration::from_millis(10), + &lease, + Duration::from_secs(3600), + ) + .await + .is_some(); + let backend = BlobBackend::new( + storage, + files.namespace.clone(), + tokio::runtime::Handle::current(), + DEADLINE, + ) + .leased_by(lease.clone()); + let listed = tokio::task::spawn_blocking(move || backend.list(FileType::Snapshot)) + .await + .unwrap(); + + assert_eq!( + ( + refreshed, + refused.load(Ordering::SeqCst), + lease.expiry() == expiry, + listed + .as_ref() + .err() + .is_some_and(|error| is_lease_expired(error)), + ), + (true, true, true, true), + "{listed:?}" + ); + } + #[test] async fn a_claim_is_taken_after_its_marker_and_a_loser_deletes_its_marker() { let files = new_files(); let directory = claims_directory(&ledger(Some(42), false)); - let started = Instant::now(); - - let lease = Lease::until(started); let marker = || { let (at, time) = marker_time(); (marker_path(&directory, 0, time), at) }; let (first_marker, first_at) = marker(); let (second_marker, second_at) = marker(); - let first = take_claim( - &files, - &directory, - 0, - &first_marker, - first_at, - &lease, - GRACE, - ) - .await - .unwrap(); - let again = take_claim( - &files, - &directory, - 0, - &second_marker, - second_at, - &lease, - GRACE, - ) - .await - .unwrap(); + let first = take_claim(&files, &directory, 0, &first_marker, first_at, GRACE) + .await + .unwrap() + .map(|lease| lease.expiry()); + let again = take_claim(&files, &directory, 0, &second_marker, second_at, GRACE) + .await + .unwrap() + .map(|lease| lease.expiry()); let listed = list_claims(&files, &directory).await.unwrap(); assert_eq!( @@ -1183,16 +1237,14 @@ mod tests { listed .iter() .any(|entry| matches!(entry, ClaimEntry::Marker(0, _))), - lease.expiry() >= started + GRACE, ), ( "golem/prune-claims/42".to_string(), - true, - false, + Some(first_at + GRACE), + None, 2, true, true, - true ) ); } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index caf1692df9..7cf9ed6547 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -63,7 +63,7 @@ use std::path::Path; use std::pin::pin; use std::sync::atomic::{AtomicBool, AtomicU8, Ordering}; use std::sync::{Arc, Mutex, PoisonError}; -use std::time::{Duration, Instant}; +use std::time::Duration; use tokio::runtime::Handle; use tokio_util::sync::{CancellationToken, DropGuard}; use tokio_util::task::TaskTracker; @@ -527,7 +527,6 @@ impl RusticSnapshotStore { else { return Ok(()); }; - let lease = Arc::new(Lease::until(Instant::now())); let span = lease_span(grace, deadline); let (started, time) = marker_time(); let marker = marker_path(&claims, number, time); @@ -540,18 +539,18 @@ impl RusticSnapshotStore { marker.clone(), self.tracker.token(), ); - let won = take_claim(&files, &claims, number, &marker, started, &lease, span) + let lease = take_claim(&files, &claims, number, &marker, started, span) .await .map_err(storage_failure)?; - if !won { + let Some(lease) = lease else { guard.disarm(); return Ok(()); - } + }; guard.claimed.store(true, Ordering::SeqCst); let claim = Claim { directory: &claims, number, - lease, + lease: Arc::new(lease), span, guard, }; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index a8e35adecb..4217ddd2ee 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1464,6 +1464,37 @@ async fn a_prune_goes_on_after_one_failed_refresh_when_the_later_refreshes_succe ); } +#[test] +#[timeout("60s")] +async fn the_lease_of_a_prune_starts_at_its_first_marker_so_a_prune_without_a_refresh_prunes() { + // A grace period of one hour gives a refresh period of fifteen minutes, so the prune ends + // before its first refresh, and only the first marker gives the lease. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + let calls = storage.calls(); + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + prunes(&calls), + calls + .iter() + .filter(|(op_label, _)| *op_label == "refresh_claim") + .count(), + ledger(&storage, &scope).await.last_prune.is_some(), + ), + (1, 0, true) + ); +} + #[test] #[timeout("60s")] async fn a_prune_deletes_the_claims_of_old_ledgers() { From bcdf25414f636334cd0d11d6870fca762e3fa364 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 08:15:28 -0700 Subject: [PATCH 087/126] Check the shut down before a delete builds its claim guard --- .../src/filesystem_snapshot/rustic/store.rs | 37 +++++++++----- .../filesystem_snapshot/rustic/store/tests.rs | 50 ++++++++++++++++++- 2 files changed, 73 insertions(+), 14 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 7cf9ed6547..af0c4decb9 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -171,19 +171,24 @@ pub(crate) struct RusticSnapshotStore { low_priority: LowPriority, /// Holds a save after its blocking work and before its publish, when a test sets it. #[cfg(test)] - pub(super) publish_gate: Option>, + pub(super) publish_gate: Option>, + /// Holds a delete after it chose its claim and before it builds its claim guard, when a test + /// sets it. + #[cfg(test)] + pub(super) claim_gate: Option>, /// Makes each backend build fail while a test sets it. #[cfg(test)] pub(super) refuse_backends: Arc, } -/// A gate that holds a save after its blocking work and before its publish. +/// A gate that holds an operation at one point, for example a save after its blocking work and +/// before its publish. #[cfg(test)] #[derive(Debug, Default)] -pub(super) struct PublishGate { - /// Notified when a save reaches the gate. +pub(super) struct StepGate { + /// Notified when the operation reaches the gate. pub(super) reached: tokio::sync::Notify, - /// Lets the save go on. + /// Lets the operation go on. pub(super) open: tokio::sync::Notify, } @@ -392,6 +397,8 @@ impl RusticSnapshotStore { #[cfg(test)] publish_gate: None, #[cfg(test)] + claim_gate: None, + #[cfg(test)] refuse_backends: Arc::default(), } } @@ -527,18 +534,24 @@ impl RusticSnapshotStore { else { return Ok(()); }; + #[cfg(test)] + if let Some(gate) = &self.claim_gate { + gate.reached.notify_one(); + gate.open.notified().await; + } + // The token of the tracker comes before the check of the cancel, so either the delete sees + // a shut down and makes no more storage calls, or `shut_down` waits for the guard and for + // each call that the guard makes. + let tracked = self.tracker.token(); + if self.root.is_cancelled() { + return Err(shut_down_error()); + } let span = lease_span(grace, deadline); let (started, time) = marker_time(); let marker = marker_path(&claims, number, time); // The guard exists before the marker write, so a drop of the delete from here until the // prune starts releases the claim. - let guard = ClaimGuard::new( - &files, - &claims, - number, - marker.clone(), - self.tracker.token(), - ); + let guard = ClaimGuard::new(&files, &claims, number, marker.clone(), tracked); let lease = take_claim(&files, &claims, number, &marker, started, span) .await .map_err(storage_failure)?; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 4217ddd2ee..3e3ff02a65 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -27,7 +27,7 @@ use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; use super::{ - PublishGate, RusticSnapshotStore, StorePolicy, leaves_marked_packs, scope_snapshots, + RusticSnapshotStore, StepGate, StorePolicy, leaves_marked_packs, scope_snapshots, store_backup_options, store_restore_options, whole_millis_from, }; use crate::filesystem_snapshot::contract_tests::fixture::{ @@ -3313,13 +3313,59 @@ async fn a_publish_held_at_its_storage_call_keeps_shut_down_waiting_until_it_end assert_eq!((held, waited, stopped), (true, true, true)); } +#[test] +#[timeout("60s")] +async fn a_shut_down_between_the_claim_listing_and_the_claim_guard_makes_no_storage_call_after_it_returns() + { + // The gate holds the delete after it listed the claims and before it builds its claim guard, + // so the tracker is empty and `shut_down` returns while the delete waits. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let gate = Arc::new(StepGate::default()); + let store = Arc::new(RusticSnapshotStore { + claim_gate: Some(gate.clone()), + ..RusticSnapshotStore::with_policy( + storage.clone(), + key(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ) + }); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let reached = tokio::time::timeout(LIMIT, gate.reached.notified()) + .await + .is_ok(); + + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + let calls_at_shut_down = storage.calls(); + gate.open.notify_one(); + let deleted = tokio::time::timeout(LIMIT, deleting).await; + let ended = eventually(|| store.work_in_flight() == 0).await; + + assert!( + matches!( + &deleted, + Ok(Ok(Err(error))) if is_storage(error, false) + ), + "{deleted:?}" + ); + assert_eq!( + (reached, stopped, ended, storage.calls()), + (true, true, true, calls_at_shut_down) + ); +} + #[test] #[timeout("60s")] async fn a_save_that_reaches_its_publish_after_shut_down_publishes_nothing() { // The gate holds the save after its blocking work, so the tracker is empty and `shut_down` // returns before the publish starts. let storage = Arc::new(InMemoryBlobStorage::new()); - let gate = Arc::new(PublishGate::default()); + let gate = Arc::new(StepGate::default()); let store = Arc::new(RusticSnapshotStore { publish_gate: Some(gate.clone()), ..RusticSnapshotStore::with_policy( From f9a36990cbdc65dc452fc6f9caa0d985b819b0d9 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 08:21:34 -0700 Subject: [PATCH 088/126] Write the final marker of a started prune through the claim guard, also on a shut down or a drop --- .../src/filesystem_snapshot/rustic/store.rs | 151 ++++++++++++------ .../filesystem_snapshot/rustic/store/tests.rs | 130 +++++++++++++++ .../rustic/store/tests/sweep.rs | 56 +++++++ 3 files changed, 285 insertions(+), 52 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index af0c4decb9..32ef5b6508 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -64,7 +64,7 @@ use std::pin::pin; use std::sync::atomic::{AtomicBool, AtomicU8, Ordering}; use std::sync::{Arc, Mutex, PoisonError}; use std::time::Duration; -use tokio::runtime::Handle; +use tokio::runtime::{Handle, TryCurrentError}; use tokio_util::sync::{CancellationToken, DropGuard}; use tokio_util::task::TaskTracker; use tokio_util::task::task_tracker::TaskTrackerToken; @@ -200,27 +200,32 @@ struct Claim<'a> { lease: Arc, /// The time that a marker write that succeeds adds to the lease. span: Duration, - /// Releases the claim when the delete stops before its prune starts. + /// Releases the claim when the delete stops before its prune starts, and writes the final + /// marker of a prune that started. guard: ClaimGuard, } /// The claim is written, and its prune did not start. const CLAIM_PENDING: u8 = 0; -/// The rustic prune of the claim started, so the claim stays. +/// The rustic prune of the claim started, so the claim stays, and it needs its final marker. const CLAIM_STARTED: u8 = 1; /// The claim was released, so its prune must not start. const CLAIM_RELEASED: u8 = 2; +/// The final marker of the prune that started is written. +const CLAIM_FINISHED: u8 = 3; -/// Holds a prune claim from the write of its first marker until the rustic prune starts. When the -/// delete drops before that, the guard releases the claim in a task, and it moves its token of the -/// tracker into that task, so `shut_down` waits for the release. The prune and the guard change the -/// state from pending with one atomic step each, so only one of them wins: a released claim never -/// starts a prune, and a drop after the prune started does not release the claim. After the start, -/// only the delete releases the claim, when each attempt of the prune found a snapshot file gone, -/// because such a prune changed nothing and counts as a prune that did not run. +/// Holds a prune claim from the write of its first marker until the final marker of its prune. +/// When the delete drops before the prune starts, the guard releases the claim in a task. When the +/// delete drops after the prune started and before the final marker is written, the guard writes +/// the final marker in a task. The guard moves its token of the tracker into that task, so +/// `shut_down` waits for it. The prune and the guard change the state from pending with one atomic +/// step each, so only one of them wins: a released claim never starts a prune, and a drop after the +/// prune started does not release the claim. After the start, only the delete releases the claim, +/// when each attempt of the prune found a snapshot file gone, because such a prune changed nothing +/// and counts as a prune that did not run. struct ClaimGuard { - /// The blobs of the scope, with a token that nothing cancels, so a release also runs after a - /// cancel or a drop. + /// The blobs of the scope, with a token that nothing cancels, so a release and a final marker + /// also run after a cancel or a drop. files: SnapshotFiles, directory: Box, number: u64, @@ -257,13 +262,26 @@ impl ClaimGuard { } } - /// Makes the release task, and moves the token of the tracker into it. - fn spawn_release(&self) -> Option> { + /// Spawns the work on the runtime, and moves the token of the tracker into it. + fn spawn_tracked( + &self, + work: impl Future + Send + 'static, + ) -> Result, TryCurrentError> { let tracked = self .tracked .lock() .unwrap_or_else(PoisonError::into_inner) .take(); + Handle::try_current().map(|runtime| { + runtime.spawn(async move { + let _tracked = tracked; + work.await; + }) + }) + } + + /// Makes the release task, and moves the token of the tracker into it. + fn spawn_release(&self) -> Option> { let files = self.files.clone(); let directory = self.directory.clone(); let number = self.number; @@ -273,18 +291,41 @@ impl ClaimGuard { .lock() .unwrap_or_else(PoisonError::into_inner) .clone(); - match Handle::try_current() { - Ok(runtime) => Some(runtime.spawn(async move { - let _tracked = tracked; - release_claim(&files, &directory, number, claimed, &markers).await; - })), - Err(error) => { - warn!( - error = %error, - "The prune claim of a filesystem snapshot scope stays, because no runtime runs its release" - ); - None - } + self.spawn_tracked(async move { + release_claim(&files, &directory, number, claimed, &markers).await; + }) + .inspect_err(|error| { + warn!( + error = %error, + "The prune claim of a filesystem snapshot scope stays, because no runtime runs its release" + ); + }) + .ok() + } + + /// Makes the task that writes the final marker, and moves the token of the tracker into it. + fn spawn_final_marker(&self) { + let files = self.files.clone(); + let directory = self.directory.clone(); + let number = self.number; + if let Err(error) = self.spawn_tracked(async move { + write_final_marker(&files, &directory, number).await; + }) { + warn!( + error = %error, + "The prune claim of a filesystem snapshot scope gets no final marker, because no runtime runs its write" + ); + } + } + + /// Writes the final marker of a prune that started, and then marks the guard finished. A guard + /// whose prune did not start writes nothing. When the write fails, the drop of the guard tries + /// it again. + async fn finish(&self) { + if self.state.load(Ordering::SeqCst) == CLAIM_STARTED + && write_final_marker(&self.files, &self.directory, self.number).await + { + self.state.store(CLAIM_FINISHED, Ordering::SeqCst); } } @@ -316,21 +357,39 @@ impl ClaimGuard { impl Drop for ClaimGuard { fn drop(&mut self) { - if self - .state - .compare_exchange( - CLAIM_PENDING, - CLAIM_RELEASED, - Ordering::SeqCst, - Ordering::SeqCst, - ) - .is_ok() - { + let moved = |from, to| { + self.state + .compare_exchange(from, to, Ordering::SeqCst, Ordering::SeqCst) + .is_ok() + }; + if moved(CLAIM_PENDING, CLAIM_RELEASED) { drop(self.spawn_release()); + } else if moved(CLAIM_STARTED, CLAIM_FINISHED) { + self.spawn_final_marker(); } } } +/// Writes the final marker of the claim with the number, and tells whether the write succeeded. A +/// failed write gives a warning. +async fn write_final_marker(files: &SnapshotFiles, directory: &Path, number: u64) -> bool { + write_marker( + files, + "final_marker", + directory, + number, + Timestamp::now_utc(), + ) + .await + .inspect_err(|error| { + warn!( + error = %format!("{error:#}"), + "Failed to write the final marker of the prune claim of a filesystem snapshot scope" + ); + }) + .is_ok() +} + /// Moves the claim of the state from pending to started, and tells whether the prune may start. fn start_prune(state: &AtomicU8) -> bool { state @@ -585,22 +644,10 @@ impl RusticSnapshotStore { } other => other.map(|marked| marked.unwrap_or_default()), }; - // The final marker holds the claim for the grace period from the end of this prune, also - // when the prune or its ledger write fails. - if let Err(error) = write_marker( - &files, - "final_marker", - claim.directory, - claim.number, - Timestamp::now_utc(), - ) - .await - { - warn!( - error = %format!("{error:#}"), - "Failed to write the final marker of the prune claim of a filesystem snapshot scope" - ); - } + // The final marker holds the claim for the hold from the end of this prune, also when the + // prune or its ledger write fails. It goes through the files of the guard, which no cancel + // ends, so a prune that a shut down stopped also gets it. + claim.guard.finish().await; let marked_packs = pruned?; let ended = Timestamp::now_utc(); write_ledger(&files, ended, marked_packs) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 3e3ff02a65..d883f6aa44 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -2416,6 +2416,136 @@ async fn shut_down_waits_for_the_release_of_a_delete_dropped_after_its_claim() { ); } +/// A storage whose gate holds the listing of the packs, which is the first call of a prune, and +/// whose final marker takes 200 ms. +fn holding_the_prune() -> Arc { + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "list" && path == Path::new("data") { + Script::WaitForGate + } else if op_label == "final_marker" { + Script::Delay(Duration::from_millis(200)) + } else { + Script::Pass + } + }) +} + +/// Gives the number of final marker writes in the calls. +fn final_markers(calls: &[(&'static str, String)]) -> usize { + calls + .iter() + .filter(|(op_label, _)| *op_label == "final_marker") + .count() +} + +/// Gives the number of markers of the claims of the scope. +async fn markers(storage: &ScriptedBlobStorage, scope: &SnapshotScope) -> usize { + claim_entries(storage, scope) + .await + .iter() + .filter(|entry| matches!(entry, ClaimEntry::Marker(..))) + .count() +} + +#[test] +#[timeout("60s")] +async fn a_shut_down_during_a_started_prune_still_writes_its_final_marker_and_waits_for_it() { + // The gate holds the first call of the prune, and the shut down cancels it. A grace period of + // one hour gives no refresh, so the claim has its first marker and its final marker. + let storage = holding_the_prune(); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let started = eventually(|| prunes(&storage.calls()) == 1).await; + + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + let markers_at_shut_down = markers(&storage, &scope).await; + storage.open_gate(); + let deleted = tokio::time::timeout(LIMIT, deleting).await; + + assert!(matches!(deleted, Ok(Ok(Err(_)))), "{deleted:?}"); + assert_eq!( + ( + started, + stopped, + markers_at_shut_down, + final_markers(&storage.calls()) + ), + (true, true, 2, 1) + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_dropped_during_a_started_prune_writes_its_final_marker() { + // The gate holds the first call of the prune, and the test drops the delete there. + let storage = holding_the_prune(); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let started = eventually(|| prunes(&storage.calls()) == 1).await; + + deleting.abort(); + let dropped = deleting.await; + let ended = eventually(|| store.work_in_flight() == 0).await; + let markers_after = markers(&storage, &scope).await; + storage.open_gate(); + + assert!( + dropped.as_ref().is_err_and(|error| error.is_cancelled()), + "{dropped:?}" + ); + assert_eq!( + ( + started, + ended, + markers_after, + final_markers(&storage.calls()) + ), + (true, true, 2, 1) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_writes_one_final_marker() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + let ended = eventually(|| store.work_in_flight() == 0).await; + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + ended, + prunes(&storage.calls()), + final_markers(&storage.calls()) + ), + (true, 1, 1) + ); +} + #[test] #[timeout("60s")] async fn a_shut_down_after_the_claim_still_releases_it() { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs index fea1ae0092..af98cf71c4 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs @@ -111,6 +111,8 @@ struct Step { failed: bool, /// The call reached the storage: a step that passed, or the landing of a late call. effect: bool, + /// The entry is the landing of a late call, not a call. + landed: bool, } /// One delete of the case: its store, its scripted storage and its task. @@ -205,6 +207,7 @@ impl Case { self.log.push(Step { failed: false, effect: true, + landed: true, ..step }); Ok(()) @@ -257,6 +260,7 @@ impl Case { path, failed: refused || late.is_some(), effect: !refused && late.is_none(), + landed: false, }; self.log.push(step.clone()); if let Some((_, _, delay)) = late { @@ -284,6 +288,7 @@ impl Case { path: String::new(), failed: true, effect: false, + landed: false, }); } if settle(&self.deletes[who]).await { @@ -526,6 +531,45 @@ fn started_claims_gone(log: &[Step], claims: &[String]) -> Vec { .collect() } +/// Gives each delete whose prune started and that made no call for its final marker. The prune +/// started when the delete called for the listing of the packs. A delete that called to delete its +/// claim released it, because each attempt of its prune found a snapshot file gone, so it writes +/// no final marker. +fn final_markers_missing(log: &[Step]) -> Vec { + [0, 1] + .into_iter() + .filter(|who| { + let own = log + .iter() + .filter(|step| step.delete == *who && !step.landed); + let (mut started, mut released, mut marked) = (false, false, false); + own.for_each(|step| { + started |= is_prune_start(step.op_label, &step.path); + released |= step.op_label == "delete_claim"; + marked |= step.op_label == "final_marker"; + }); + started && !released && !marked + }) + .collect() +} + +/// Gives each delete that called for a final marker after a final marker call of it that gave no +/// error. +fn final_markers_repeated(log: &[Step]) -> Vec { + [0, 1] + .into_iter() + .filter(|who| { + log.iter() + .filter(|step| { + step.delete == *who && !step.landed && step.op_label == "final_marker" + }) + .skip_while(|step| step.failed) + .nth(1) + .is_some() + }) + .collect() +} + /// Checks the rules on the end state of a case. async fn check( shared: &Arc, @@ -574,6 +618,18 @@ async fn check( "a delete whose prune started and wrote no ledger lost its claim: {dropped:?}" )); } + let unmarked = final_markers_missing(log); + if !unmarked.is_empty() { + return fail(&format!( + "a delete whose prune started and kept its claim made no final marker call: {unmarked:?}" + )); + } + let repeated = final_markers_repeated(log); + if !repeated.is_empty() { + return fail(&format!( + "a delete made a final marker call after one that succeeded: {repeated:?}" + )); + } let records = blobs(&**shared, &scope.0, "golem/prune-freed/").await; let entries = blobs(&**shared, &scope.0, "golem/prune-ledgers/").await; let ledger = ledger(shared, scope).await; From 2b8e7f733d6d076a1d4990159d32cdef991fa159 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 09:33:21 -0700 Subject: [PATCH 089/126] Name the hold of a claim as the delay of a late marker, and add the missing verb --- .../src/filesystem_snapshot/rustic/store.rs | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 32ef5b6508..89f2b21172 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -229,7 +229,8 @@ struct ClaimGuard { files: SnapshotFiles, directory: Box, number: u64, - /// Whether this delete wrote the claim. A delete that did not delete only its markers. + /// Whether this delete wrote the claim. A delete that did not write it deletes only its + /// markers. claimed: AtomicBool, /// The markers of the claim that this delete wrote: the first marker, and each new marker while /// the prune runs. @@ -554,10 +555,10 @@ impl RusticSnapshotStore { /// A due prune runs only after the delete takes a claim of its ledger, and only when the ledger /// did not change after the claim. After an error before the prune ran, the claim is deleted, /// so a retry of the delete prunes again, and a claim write that the storage completes after - /// that delete can delay that prune by up to the grace period. A prune that found a snapshot + /// that delete can delay that prune by up to the hold of a claim. A prune that found a snapshot /// file gone at each attempt changed nothing, so it counts as an error before the prune ran. /// After a prune that started, the claim stays on each outcome, also when the prune or its - /// ledger write fails, so the next prune waits the grace period from the end of this one. A + /// ledger write fails, so the next prune waits the hold of a claim from the end of this one. A /// prune that succeeds deletes each claim of its ledger. async fn prune_when_due( &self, From 8d38f4404894b78ab09d97ac8d94c03bcb660cbf Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 09:34:39 -0700 Subject: [PATCH 090/126] Limit each async storage test of the prune module to 60 s --- golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 2da763b737..99240ba86d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -1064,6 +1064,7 @@ mod tests { } #[test] + #[timeout("60s")] async fn the_refresh_of_a_claim_ends_when_its_operation_is_cancelled() { let files = new_files(); files.cancel.cancel(); @@ -1208,6 +1209,7 @@ mod tests { } #[test] + #[timeout("60s")] async fn a_claim_is_taken_after_its_marker_and_a_loser_deletes_its_marker() { let files = new_files(); let directory = claims_directory(&ledger(Some(42), false)); @@ -1437,6 +1439,7 @@ mod tests { } #[test] + #[timeout("60s")] async fn a_record_of_freed_bytes_is_written_and_listed() { let files = new_files(); @@ -1449,6 +1452,7 @@ mod tests { } #[test] + #[timeout("60s")] async fn a_written_entry_is_the_ledger_that_a_read_gives() { let files = new_files(); let ended = Timestamp::from(Timestamp::now_utc().to_millis()); From e7729c245f2e96c72a6bd3af338adb458e1e6b52 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 09:39:08 -0700 Subject: [PATCH 091/126] Read the clock for the claim choice after the claim listing --- .../src/filesystem_snapshot/rustic/store.rs | 24 ++++++- .../filesystem_snapshot/rustic/store/tests.rs | 65 +++++++++++++++++++ 2 files changed, 87 insertions(+), 2 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 89f2b21172..567474373c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -179,6 +179,10 @@ pub(crate) struct RusticSnapshotStore { /// Makes each backend build fail while a test sets it. #[cfg(test)] pub(super) refuse_backends: Arc, + /// The number of milliseconds that the clock of the prune decisions is ahead of the wall + /// clock. A test moves it to make time pass. + #[cfg(test)] + pub(super) clock_ahead: Arc, } /// A gate that holds an operation at one point, for example a save after its blocking work and @@ -460,6 +464,8 @@ impl RusticSnapshotStore { claim_gate: None, #[cfg(test)] refuse_backends: Arc::default(), + #[cfg(test)] + clock_ahead: Arc::default(), } } @@ -482,6 +488,16 @@ impl RusticSnapshotStore { self.tracker.len() } + /// Gives the time now, for a comparison with a time from storage. + fn now(&self) -> Timestamp { + let now = Timestamp::now_utc(); + #[cfg(test)] + let now = Timestamp::from( + now.to_millis() + self.clock_ahead.load(std::sync::atomic::Ordering::SeqCst), + ); + now + } + /// Starts an operation. The token of the operation is cancelled when the guard drops. fn start(&self) -> Result<(CancellationToken, DropGuard), SnapshotStoreError> { if self.root.is_cancelled() { @@ -568,7 +584,10 @@ impl RusticSnapshotStore { let files = self.files(scope, token); let ledger = read_ledger(&files).await.map_err(storage_failure)?; let records = list_freed(&files).await.map_err(storage_failure)?; - let now = Timestamp::now_utc(); + // Each comparison with a time from storage uses a clock reading from after the listing + // that gave that time. A listing can take up to one storage call deadline, and a stale + // reading can put a marker that another host wrote within the margin beyond the margin. + let now = self.now(); let grace = self.policy.prune.keep_delete; let size = if needs_repository_size(&ledger, records.bytes, now, grace) { repository_bytes(&files).await.map_err(storage_failure)? @@ -590,7 +609,8 @@ impl RusticSnapshotStore { .await .map_err(storage_failure)?; let deadline = self.policy.deadline; - let ClaimChoice::Claim(number) = next_claim(&listed, now, claim_hold(grace, deadline)) + let ClaimChoice::Claim(number) = + next_claim(&listed, self.now(), claim_hold(grace, deadline)) else { return Ok(()); }; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index d883f6aa44..f1d3602df9 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1464,6 +1464,71 @@ async fn a_prune_goes_on_after_one_failed_refresh_when_the_later_refreshes_succe ); } +#[test] +#[timeout("60s")] +async fn a_marker_ahead_within_the_margin_after_a_slow_claim_listing_holds_the_claim() { + // Another delete wrote a marker whose time is 130 s ahead of the clock at the start of the + // delete, beyond the margin of 120 s. The gate holds the claim listing while the clock goes + // 20 s on, so after the listing the marker is 110 s ahead, within the margin, and it holds. + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "list_claims" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let ahead = golem_common::model::Timestamp::now_utc().to_millis() + 130_000; + futures::stream::iter([ + "golem/prune-claims/none/0".to_string(), + format!("golem/prune-claims/none/0@{ahead}-{}", uuid::Uuid::new_v4()), + ]) + .for_each(|path| { + let (storage, scope) = (storage.clone(), scope.clone()); + async move { + storage + .put_raw("test", "test", scope.0.clone(), Path::new(&path), &[]) + .await + .unwrap(); + } + }) + .await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "list_claims") + }) + .await; + + store.clock_ahead.store(20_000, Ordering::SeqCst); + storage.open_gate(); + let deleted = tokio::time::timeout(LIMIT, deleting).await; + let calls = storage.calls(); + + assert!(matches!(deleted, Ok(Ok(Ok(())))), "{deleted:?}"); + assert_eq!( + ( + held, + prunes(&calls), + calls + .iter() + .filter(|(op_label, _)| *op_label == "write_claim") + .count(), + ), + (true, 0, 0) + ); +} + #[test] #[timeout("60s")] async fn the_lease_of_a_prune_starts_at_its_first_marker_so_a_prune_without_a_refresh_prunes() { From bf3700b3ad9491cd18184705434418a582f07490 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 09:52:44 -0700 Subject: [PATCH 092/126] Take back a step of a dropped delete whose call its cancel ended, and run that order as a fixed case --- .../filesystem_snapshot/rustic/scripted.rs | 12 ++++++++- .../filesystem_snapshot/rustic/store/tests.rs | 26 +++++++++++++++++++ .../rustic/store/tests/sweep.rs | 19 ++++++++++++++ 3 files changed, 56 insertions(+), 1 deletion(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs index aaa5d50908..831efb301f 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -112,6 +112,14 @@ impl ScriptedBlobStorage { self.steps.add_permits(1); } + /// Takes back one step that no call took, and tells whether one was there. + pub(super) fn take_back_step(&self) -> bool { + self.steps + .try_acquire() + .map(tokio::sync::SemaphorePermit::forget) + .is_ok() + } + /// Gives the number of calls that wait for a step. pub(super) fn waiting_steps(&self) -> usize { self.waiting.load(Ordering::SeqCst) @@ -140,11 +148,13 @@ impl ScriptedBlobStorage { if let Ok(permit) = self.steps.acquire().await { permit.forget(); } - drop(waiting); + // The call is in the steps that were taken before it stops waiting, so a test never sees + // a call that neither waits nor took its step. self.took .lock() .unwrap_or_else(PoisonError::into_inner) .push((op_label, path.into())); + drop(waiting); Stepped(&self.stepped) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index f1d3602df9..3c6e7c49d1 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -4745,6 +4745,9 @@ mod sweep; /// its end when its prune writes one new marker. const SWEEP_TURN: usize = 22; +/// The number of runs of the order with a drop at the first step. +const DROP_REPEATS: usize = 300; + /// The number of random orders that the property test tries. const SWEEP_CASES: u32 = 1000; @@ -4792,6 +4795,29 @@ async fn two_deletes_make_at_most_one_prune_in_each_order_with_up_to_two_switche assert!(cases.is_ok(), "{cases:?}"); } +#[test] +#[timeout("60s")] +async fn two_deletes_make_at_most_one_prune_when_one_is_dropped_at_its_first_step() { + // A drop right after the first step cancels the forget of the delete while its storage call + // can already wait for a step. The order ran into that race in a random case, so the test + // runs it many times. + let (shared, prepared) = prepared_scope().await; + let schedule = sweep::Schedule { + first: 0, + turns: vec![6, 1, 17, 14, 16, 2, 1, 12], + fail: Some((0, 17)), + late: None, + drop: Some((0, 0)), + }; + + let cases = futures::stream::iter(0..DROP_REPEATS) + .then(|_| sweep::run_case(&shared, &prepared, &schedule)) + .try_fold(0usize, |cases, _| async move { Ok(cases + 1) }) + .await; + + assert!(cases.is_ok(), "{cases:?}"); +} + #[test] #[timeout("60s")] async fn two_deletes_make_at_most_one_prune_in_random_orders_with_a_failed_call() { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs index af98cf71c4..980ee7fc02 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs @@ -231,6 +231,25 @@ impl Case { let storage = self.deletes[who].storage.clone(); let (before, taker) = (storage.stepped(), storage.took().len()); storage.step(); + // The drop cancels the operation of the delete, and a call that already waited for a step + // then ends without the step. So for a dropped delete, the step can stay with no call to + // take it, and the test takes it back. + let dropped = self.deletes[who].dropped; + let untaken = || { + dropped + && storage.stepped() == before + && storage.waiting_steps() == 0 + && storage.took().len() == taker + }; + if !until(|| storage.stepped() > before || untaken()).await { + return Err(format!( + "a step of the dropped delete {who} was neither taken nor left; its last calls {:?}", + storage.calls().iter().rev().take(6).collect::>() + )); + } + if untaken() && storage.take_back_step() { + return Ok(()); + } if !until(|| storage.stepped() > before).await { return Err(format!( "a step of delete {who} did not end; its last calls {:?}", From ce108bf01e92968884680d18bb1d4606897142ee Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:47:22 -0700 Subject: [PATCH 093/126] Fence the ledger write of a prune with its lease --- .../src/filesystem_snapshot/rustic/backend.rs | 2 +- .../filesystem_snapshot/rustic/scripted.rs | 6 ++ .../src/filesystem_snapshot/rustic/store.rs | 8 +- .../filesystem_snapshot/rustic/store/tests.rs | 89 +++++++++++++++++++ 4 files changed, 102 insertions(+), 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index f44c2389c2..bc1dee7dde 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -240,7 +240,7 @@ impl BlobBackend { /// Gives the output of the future, or [`LeaseExpired`] when the lease runs out first. A call does /// not start when the lease has run out. -async fn within_lease( +pub(super) async fn within_lease( lease: &Lease, future: impl Future>, ) -> anyhow::Result { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs index 831efb301f..732fbc2790 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -47,6 +47,8 @@ pub(super) enum Script { WaitForGate, /// Waits for the time, and then passes the call. Delay(std::time::Duration), + /// Waits for the time, and then gives an error and does not pass the call. + RefuseAfter(std::time::Duration), /// Gives no blob to a read of a whole blob, as a delete after a listing does. Each other call /// passes. Vanish, @@ -248,6 +250,10 @@ impl ScriptedBlobStorage { tokio::time::sleep(time).await; call.await } + Script::RefuseAfter(time) => { + tokio::time::sleep(time).await; + Err(anyhow::anyhow!("the storage refused the call")) + } Script::Step { refuse, .. } => { let _stepped = self.wait_for_step(op_label, path).await; if refuse { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 567474373c..a1e68273b0 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -20,7 +20,7 @@ //! next storage call. The store counts each blocking task and each backend in a task tracker, and //! [`RusticSnapshotStore::shut_down`] waits for them. -use super::backend::{BlobBackend, Lease}; +use super::backend::{BlobBackend, Lease, within_lease}; use super::fault::{ Operation, classify, is_file_missing, is_snapshot_missing, is_storage_failure, storage_failure, }; @@ -671,7 +671,11 @@ impl RusticSnapshotStore { claim.guard.finish().await; let marked_packs = pruned?; let ended = Timestamp::now_utc(); - write_ledger(&files, ended, marked_packs) + // The lease fences the ledger write as it fences the calls of rustic. A write that would + // start after the lease ran out is not sent, and a write that starts before is bounded by + // the time left, so the entry never lands after another delete can take the claim over. + // A prune whose ledger write the lease skipped keeps its claim and runs no cleanup. + within_lease(&claim.lease, write_ledger(&files, ended, marked_packs)) .await .map_err(storage_failure)?; remove_older_ledgers(&files, ended).await; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 3c6e7c49d1..0c363a44d5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1529,6 +1529,95 @@ async fn a_marker_ahead_within_the_margin_after_a_slow_claim_listing_holds_the_c ); } +#[test] +#[timeout("60s")] +async fn a_ledger_write_after_the_lease_ran_out_is_not_sent_and_the_claim_stays() { + // A zero grace period and a deadline of 1 s give a lease of 999 ms. Each refresh fails, and + // the final marker takes longer than its deadline, so it fails at 1 s, after the end of the + // lease. The prune itself ends well within the lease. + let deadline = Duration::from_secs(1); + let storage = + ScriptedBlobStorage::new( + Arc::new(InMemoryBlobStorage::new()), + |op_label, _| match op_label { + "refresh_claim" => Script::Refuse, + "final_marker" => Script::RefuseAfter(Duration::from_millis(1200)), + _ => Script::Pass, + }, + ); + let store = store(storage.clone(), policy(deadline, ALWAYS, Duration::ZERO)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + let calls = storage.calls(); + let entries = claim_entries(&storage, &scope).await; + + assert!( + matches!( + &deleted, + Err(SnapshotStoreError::Storage { source, .. }) if is_lease_expired(source.as_ref()) + ), + "{deleted:?}" + ); + assert_eq!( + ( + prunes(&calls), + calls + .iter() + .filter(|(op_label, _)| *op_label == "write_ledger") + .count(), + ledger(&storage, &scope).await.last_prune, + entries.contains(&ClaimEntry::Claim(0)), + ), + (1, 0, None, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_ledger_write_that_starts_within_the_lease_ends_at_the_end_of_the_lease() { + // A zero grace period and a deadline of 2 s give a lease of 1999 ms. The final marker takes + // 1.5 s, so the ledger write starts with about 0.5 s of the lease left. The ledger write takes + // 10 s, so the end of the lease ends it before its own deadline. + let deadline = Duration::from_secs(2); + let storage = + ScriptedBlobStorage::new( + Arc::new(InMemoryBlobStorage::new()), + |op_label, _| match op_label { + "final_marker" => Script::Delay(Duration::from_millis(1500)), + "write_ledger" => Script::Delay(Duration::from_secs(10)), + _ => Script::Pass, + }, + ); + let store = store(storage.clone(), policy(deadline, ALWAYS, Duration::ZERO)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + let calls = storage.calls(); + let entries = claim_entries(&storage, &scope).await; + + assert!( + matches!( + &deleted, + Err(SnapshotStoreError::Storage { source, .. }) if is_lease_expired(source.as_ref()) + ), + "{deleted:?}" + ); + assert_eq!( + ( + calls + .iter() + .filter(|(op_label, _)| *op_label == "write_ledger") + .count(), + ledger(&storage, &scope).await.last_prune, + entries.contains(&ClaimEntry::Claim(0)), + ), + (1, None, true) + ); +} + #[test] #[timeout("60s")] async fn the_lease_of_a_prune_starts_at_its_first_marker_so_a_prune_without_a_refresh_prunes() { From 0c2ca57bc0a86bb2760d5d7081b4b47aa8fe197f Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:50:04 -0700 Subject: [PATCH 094/126] Say that the next prune waits a full hold from the newest claim marker that was written --- .../src/filesystem_snapshot/rustic/store.rs | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index a1e68273b0..53dbca3d3c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -574,8 +574,10 @@ impl RusticSnapshotStore { /// that delete can delay that prune by up to the hold of a claim. A prune that found a snapshot /// file gone at each attempt changed nothing, so it counts as an error before the prune ran. /// After a prune that started, the claim stays on each outcome, also when the prune or its - /// ledger write fails, so the next prune waits the hold of a claim from the end of this one. A - /// prune that succeeds deletes each claim of its ledger. + /// ledger write fails, or when the lease skips its ledger write. So the next prune waits a full + /// hold from the newest claim marker that was written, which is the end of the prune unless the + /// final marker write failed. Two prunes never run at once; when writes fail, only the gap + /// between them can be shorter. A prune that succeeds deletes each claim of its ledger. async fn prune_when_due( &self, scope: &SnapshotScope, From 98676eba7288a2a3ef455df1cd8887e866a3784d Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:50:14 -0700 Subject: [PATCH 095/126] Name the calls that run after the cancel of a shut down, and the claim guards that the tracker counts --- .../src/filesystem_snapshot/rustic/store.rs | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 53dbca3d3c..8125d61817 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -164,8 +164,9 @@ pub(crate) struct RusticSnapshotStore { policy: StorePolicy, /// The parent of the token of each operation. root: CancellationToken, - /// Counts the blocking tasks, the backends, the blob calls of the store, the publishes and the - /// deletes of dropped publishes. + /// Counts the blocking tasks, the backends, the blob calls of the store, the publishes, the + /// deletes of dropped publishes, and the claim guards with their release and final-marker + /// tasks. tracker: TaskTracker, /// Runs saves and prunes at a low priority. low_priority: LowPriority, @@ -470,10 +471,12 @@ impl RusticSnapshotStore { } /// Cancels each operation, so each running storage call ends and no new call starts, and later - /// operations give `Storage`. A publish that starts before the cancel runs to its end. A save - /// that reaches its publish after the cancel publishes nothing and gives `Storage`. The call - /// waits until no blocking task, backend, blob call of the store, publish, delete of a dropped - /// publish or release of a prune claim remains. A blob call that is not polled holds the wait + /// operations give `Storage`. Some calls run after the cancel by design, because no cancel + /// ends them: a publish that started before the cancel runs to its end, and a claim guard + /// writes the release of its claim, or the final marker of a prune that started. A save that + /// reaches its publish after the cancel publishes nothing and gives `Storage`. The call waits + /// until no blocking task, backend, blob call of the store, publish, delete of a dropped + /// publish, claim guard, or release or final marker of a claim guard remains. A blob call that is not polled holds the wait /// until it is polled again, and then it ends at once. The runtime must not drop before it /// returns, because a storage call after its time driver stops aborts the process. pub(crate) async fn shut_down(&self) { From f701a8c6c893d20029718ca07b7934c2cae38659 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:50:24 -0700 Subject: [PATCH 096/126] Say that a claim guard releases its claim after the cancel, not that it writes the release --- golem-worker-executor/src/filesystem_snapshot/rustic/store.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 8125d61817..c58342f5b4 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -473,7 +473,7 @@ impl RusticSnapshotStore { /// Cancels each operation, so each running storage call ends and no new call starts, and later /// operations give `Storage`. Some calls run after the cancel by design, because no cancel /// ends them: a publish that started before the cancel runs to its end, and a claim guard - /// writes the release of its claim, or the final marker of a prune that started. A save that + /// releases its claim, or writes the final marker of a prune that started. A save that /// reaches its publish after the cancel publishes nothing and gives `Storage`. The call waits /// until no blocking task, backend, blob call of the store, publish, delete of a dropped /// publish, claim guard, or release or final marker of a claim guard remains. A blob call that is not polled holds the wait From c196a10defaf08c5e450e5e15a2091258ef6931c Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:50:47 -0700 Subject: [PATCH 097/126] Limit each async test of the store, scope, publish and rustic modules to 60 s --- .../rustic/publish/tests.rs | 5 +++ .../filesystem_snapshot/rustic/scope/tests.rs | 10 ++++- .../filesystem_snapshot/rustic/store/tests.rs | 41 +++++++++++++++++++ .../src/filesystem_snapshot/rustic/tests.rs | 15 +++++++ 4 files changed, 70 insertions(+), 1 deletion(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs index 22d1360e0e..bff3f14afb 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -88,6 +88,7 @@ async fn stored(files: &SnapshotFiles, inner: &InMemoryBlobStorage) -> Option Vec<(String, String)> { } #[test] +#[timeout("60s")] async fn a_copy_gives_the_target_each_blob_of_the_repository_and_not_the_ledger() { let storage = Arc::new(InMemoryBlobStorage::new()); let (from, to) = (new_namespace(), new_namespace()); @@ -133,6 +134,7 @@ async fn a_copy_gives_the_target_each_blob_of_the_repository_and_not_the_ledger( } #[test] +#[timeout("60s")] async fn a_copy_writes_the_packs_the_keys_the_index_files_the_snapshot_files_and_then_the_config() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); @@ -161,6 +163,7 @@ async fn a_copy_writes_the_packs_the_keys_the_index_files_the_snapshot_files_and } #[test] +#[timeout("60s")] async fn a_copy_lists_the_snapshot_files_before_the_index_files_the_keys_and_the_packs() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); @@ -183,6 +186,7 @@ async fn a_copy_lists_the_snapshot_files_before_the_index_files_the_keys_and_the } #[test] +#[timeout("60s")] async fn a_copy_of_a_namespace_without_a_config_copies_nothing() { let storage = Arc::new(InMemoryBlobStorage::new()); let (from, to) = (new_namespace(), new_namespace()); @@ -204,6 +208,7 @@ async fn a_copy_of_a_namespace_without_a_config_copies_nothing() { } #[test] +#[timeout("60s")] async fn a_copy_that_fails_gives_the_error_and_the_target_has_no_config() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { @@ -235,6 +240,7 @@ async fn a_copy_that_fails_gives_the_error_and_the_target_has_no_config() { } #[test] +#[timeout("60s")] async fn a_deleted_scope_holds_no_blob_and_another_scope_keeps_its_blobs() { let storage = Arc::new(InMemoryBlobStorage::new()); let (deleted, kept) = (new_namespace(), new_namespace()); @@ -253,6 +259,7 @@ async fn a_deleted_scope_holds_no_blob_and_another_scope_keeps_its_blobs() { } #[test] +#[timeout("60s")] async fn a_delete_of_a_scope_deletes_the_config_first() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); @@ -273,6 +280,7 @@ async fn a_delete_of_a_scope_deletes_the_config_first() { } #[test] +#[timeout("60s")] async fn a_delete_of_an_unused_scope_succeeds_and_can_run_again() { let storage = Arc::new(InMemoryBlobStorage::new()); let namespace = new_namespace(); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 0c363a44d5..d4634c15c9 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -298,6 +298,7 @@ fn a_save_time_is_the_first_whole_millisecond_that_is_not_before_the_call() { #[cfg(target_os = "linux")] #[test] +#[timeout("60s")] async fn a_tree_saved_through_a_proc_self_fd_path_is_stored_below_the_root() { use std::os::fd::AsRawFd; let storage = Arc::new(InMemoryBlobStorage::new()); @@ -345,6 +346,7 @@ async fn a_tree_saved_through_a_proc_self_fd_path_is_stored_below_the_root() { } #[test] +#[timeout("60s")] async fn a_save_whose_index_write_fails_publishes_nothing_and_leaves_the_name_free() { let refuse = Arc::new(AtomicBool::new(true)); let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { @@ -436,6 +438,7 @@ async fn a_publish_that_reaches_the_deadline_and_lands_late_publishes_nothing() } #[test] +#[timeout("60s")] async fn a_second_save_of_an_unchanged_tree_writes_no_pack() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); @@ -606,6 +609,7 @@ async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { } #[test] +#[timeout("60s")] async fn a_restore_whose_pack_reads_fail_gives_a_retryable_storage_error() { let refuse = Arc::new(AtomicBool::new(false)); let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { @@ -642,6 +646,7 @@ async fn a_restore_whose_pack_reads_fail_gives_a_retryable_storage_error() { #[cfg(unix)] #[test] +#[timeout("60s")] async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_denied() { use std::os::unix::fs::PermissionsExt; // SAFETY: `geteuid` has no preconditions. @@ -677,6 +682,7 @@ async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_d #[cfg(target_os = "linux")] #[test] +#[timeout("60s")] async fn a_restore_that_cannot_set_an_extended_attribute_gives_destination() { // A file on tmpfs takes a user attribute of 6,000 bytes. A file on ext4 with 4 KiB blocks // does not, so the restore cannot set it. The test checks nothing on a host where the source @@ -744,6 +750,7 @@ fn xattr_set(path: &Path, name: &str, value: &[u8]) -> std::io::Result<()> { } #[test] +#[timeout("60s")] async fn a_snapshot_file_that_fails_its_check_is_left_out_of_list_and_makes_an_unknown_name_corrupt() { let storage = Arc::new(InMemoryBlobStorage::new()); @@ -788,6 +795,7 @@ async fn a_snapshot_file_that_fails_its_check_is_left_out_of_list_and_makes_an_u } #[test] +#[timeout("60s")] async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_period() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( @@ -906,6 +914,7 @@ async fn put_ledger_entry( } #[test] +#[timeout("60s")] async fn a_delete_within_the_grace_period_does_not_list_the_packs() { // The ledger has no marked packs, so only the grace period keeps the first delete from a // listing. The second delete comes after the grace period and lists the packs one time. @@ -948,6 +957,7 @@ async fn a_delete_within_the_grace_period_does_not_list_the_packs() { } #[test] +#[timeout("60s")] async fn a_failed_listing_of_the_packs_gives_storage_and_records_no_prune() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { if op_label == "list_data" { @@ -1114,6 +1124,7 @@ async fn a_record_that_a_delete_adds_during_a_prune_stays_for_the_next_prune() { } #[test] +#[timeout("60s")] async fn a_late_older_ledger_entry_does_not_win() { // The newer entry is inside the grace period, so a delete does not prune. let storage = @@ -1866,6 +1877,7 @@ async fn a_backend_that_does_not_build_after_the_claim_releases_it() { } #[test] +#[timeout("60s")] async fn a_forget_that_fails_after_the_record_write_leaves_the_record() { // The forget deletes the snapshot file, and the storage refuses that call. let storage = @@ -2092,6 +2104,7 @@ async fn a_record_whose_snapshot_still_exists_counts_nothing_and_does_not_make_a } #[test] +#[timeout("60s")] async fn a_record_write_that_answers_already_exists_counts_as_written() { // A new try of a record write whose first answer was lost finds the record of the first try. let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { @@ -3030,6 +3043,7 @@ async fn a_prune_that_succeeds_deletes_the_claims_and_the_counted_records_of_fre } #[test] +#[timeout("60s")] async fn a_delete_below_the_threshold_does_not_prune() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( @@ -3063,6 +3077,7 @@ async fn a_delete_below_the_threshold_does_not_prune() { } #[test] +#[timeout("60s")] async fn no_second_prune_runs_within_the_grace_period() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( @@ -3101,6 +3116,7 @@ async fn no_second_prune_runs_within_the_grace_period() { } #[test] +#[timeout("60s")] async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { let refuse = Arc::new(AtomicBool::new(false)); let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { @@ -3159,6 +3175,7 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { } #[test] +#[timeout("60s")] async fn a_deleted_scope_holds_no_blob() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( @@ -3273,6 +3290,7 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam } #[test] +#[timeout("60s")] async fn a_blob_call_of_a_cancelled_operation_does_not_start() { // The in-memory storage answers at the first poll, so only the check before the call keeps // the call from the storage. @@ -3474,6 +3492,7 @@ async fn a_copy_fails_when_a_prune_removed_an_index_file_that_it_listed() { } #[test] +#[timeout("60s")] async fn a_copy_leaves_out_a_snapshot_file_that_a_delete_removed_after_the_listing() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { @@ -3690,6 +3709,7 @@ async fn a_save_that_reaches_its_publish_after_shut_down_publishes_nothing() { } #[test] +#[timeout("60s")] async fn delete_scope_and_copy_scope_after_shut_down_give_storage() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); @@ -3909,6 +3929,7 @@ fn id_of(files: &[rustic_core::repofile::SnapshotFile], name: &str) -> Option Vec<(Str #[cfg(target_os = "linux")] #[test] +#[timeout("60s")] async fn the_forget_of_a_delete_runs_its_storage_calls_in_a_rayon_pool_of_its_own() { let (storage, calls) = nice_recording_storage(); let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); @@ -4500,6 +4535,7 @@ async fn the_forget_of_a_delete_runs_its_storage_calls_in_a_rayon_pool_of_its_ow #[cfg(target_os = "linux")] #[test] +#[timeout("60s")] async fn the_index_load_of_a_restore_runs_its_storage_calls_in_a_rayon_pool_of_its_own() { let (storage, calls) = nice_recording_storage(); let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); @@ -4543,6 +4579,7 @@ fn labels_and_calls_not_at_nice_19( #[cfg(target_os = "linux")] #[test] +#[timeout("60s")] async fn the_storage_calls_of_a_save_run_at_nice_19() { // The publish of the snapshot file runs on the async runtime after the work, so it keeps // the normal priority. The second save reads the config, the index, the snapshot files and @@ -4580,6 +4617,7 @@ async fn the_storage_calls_of_a_save_run_at_nice_19() { #[cfg(target_os = "linux")] #[test] +#[timeout("60s")] async fn the_storage_calls_of_a_prune_run_at_nice_19() { // The forget of a delete runs before the ledger read at the normal priority. The ledger calls, // the calls of the freed records, the listing of the packs and the claim calls run on the @@ -4636,6 +4674,7 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { #[cfg(target_os = "linux")] #[test] +#[timeout("60s")] async fn the_storage_calls_of_a_restore_run_at_the_nice_value_of_the_process() { let process_nice = super::super::priority::own_nice(); let (storage, calls) = nice_recording_storage(); @@ -4718,6 +4757,7 @@ async fn after_saves_and_prunes_the_pools_keep_the_nice_value_of_the_process() { #[cfg(target_os = "linux")] #[test] +#[timeout("60s")] async fn the_storage_calls_of_the_rayon_workers_of_a_prune_that_repacks_run_at_nice_19() { // The deleted snapshot shares a pack with the kept one, so the prune repacks that pack. The // prune reads the index files and repacks with rayon, on the workers of the pool of the prune. @@ -4795,6 +4835,7 @@ fn no_pool( #[cfg(target_os = "linux")] #[test] +#[timeout("60s")] async fn the_global_rayon_pool_keeps_the_nice_value_of_the_process_after_saves_without_their_pool() { // Without its own pool, the rayon work of a save goes to the global pool from a thread at diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index f367be34f8..601ea33a76 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -238,6 +238,7 @@ async fn dropped_within_limit(dropped: oneshot::Receiver<()>) -> bool { } #[test] +#[timeout("60s")] async fn a_saved_tree_comes_back_the_same() { let storage = Arc::new(InMemoryBlobStorage::new()); let repository = repository(&storage, &new_scope()); @@ -257,6 +258,7 @@ async fn a_saved_tree_comes_back_the_same() { } #[test] +#[timeout("60s")] async fn a_restore_report_gives_each_phase_of_the_restore_in_order() { let storage = Arc::new(InMemoryBlobStorage::new()); let repository = repository(&storage, &new_scope()); @@ -287,6 +289,7 @@ async fn a_restore_report_gives_each_phase_of_the_restore_in_order() { } #[test] +#[timeout("60s")] async fn a_restore_reads_each_tree_pack_one_time_in_full_and_no_range_of_a_tree_pack() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -327,6 +330,7 @@ async fn a_restore_reads_each_tree_pack_one_time_in_full_and_no_range_of_a_tree_ } #[test] +#[timeout("60s")] async fn a_second_save_has_the_first_as_parent_and_reads_only_the_changed_file() { let storage = Arc::new(InMemoryBlobStorage::new()); let repository = repository(&storage, &new_scope()); @@ -368,6 +372,7 @@ async fn a_second_save_has_the_first_as_parent_and_reads_only_the_changed_file() } #[test] +#[timeout("60s")] async fn a_forgotten_name_does_not_restore_and_the_other_names_do() { let storage = Arc::new(InMemoryBlobStorage::new()); let repository = repository(&storage, &new_scope()); @@ -401,6 +406,7 @@ async fn a_forgotten_name_does_not_restore_and_the_other_names_do() { } #[test] +#[timeout("60s")] async fn the_first_save_creates_the_repository_with_no_key_file_and_later_saves_open_it() { let storage = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -446,6 +452,7 @@ async fn the_first_save_creates_the_repository_with_no_key_file_and_later_saves_ } #[test] +#[timeout("60s")] async fn each_scope_is_its_own_repository() { let storage = Arc::new(InMemoryBlobStorage::new()); let (one, other) = (new_scope(), new_scope()); @@ -901,6 +908,7 @@ async fn a_prune_whose_tree_pack_reads_get_no_answer_fails_and_stops_its_threads } #[test] +#[timeout("60s")] async fn a_prune_after_a_forget_deletes_the_packs_of_that_name_and_the_other_name_restores() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -1140,6 +1148,7 @@ fn each_setting_goes_into_its_rustic_option() { } #[test] +#[timeout("60s")] async fn a_repository_keeps_the_settings_of_its_first_save_and_inspect_gives_them() { let storage = Arc::new(InMemoryBlobStorage::new()); let (fixed_scope, default_scope) = (new_scope(), new_scope()); @@ -1187,6 +1196,7 @@ async fn a_repository_keeps_the_settings_of_its_first_save_and_inspect_gives_the } #[test] +#[timeout("60s")] async fn inspect_gives_an_error_for_fixed_chunks_that_no_setting_can_hold() { // A repository that rustic makes with fixed chunks of 4 GiB has a chunk size that does not fit // `Chunking::Fixed`. The inspection must not report it as a Rabin repository. @@ -1224,6 +1234,7 @@ async fn inspect_gives_an_error_for_fixed_chunks_that_no_setting_can_hold() { } #[test] +#[timeout("60s")] async fn inspect_gives_the_snapshots_the_name_and_the_phases_and_nothing_without_a_repository() { let storage = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -1261,6 +1272,7 @@ async fn inspect_gives_the_snapshots_the_name_and_the_phases_and_nothing_without } #[test] +#[timeout("60s")] async fn a_size_and_mtime_save_of_a_copied_tree_reads_no_file_and_a_ctime_save_reads_each() { // A copy gives each file a new inode and a new change time, and keeps its size and its // modification time. The size-and-mtime form must also not compare inodes: in rustic, @@ -1308,6 +1320,7 @@ async fn a_size_and_mtime_save_of_a_copied_tree_reads_no_file_and_a_ctime_save_r } #[test] +#[timeout("60s")] async fn a_size_and_mtime_save_misses_a_rewrite_of_the_same_size_with_the_old_mtime() { let storage = Arc::new(InMemoryBlobStorage::new()); let size_mtime = SaveSettings { @@ -1360,6 +1373,7 @@ async fn a_size_and_mtime_save_misses_a_rewrite_of_the_same_size_with_the_old_mt } #[test] +#[timeout("60s")] async fn two_prunes_without_a_grace_period_give_back_the_data_that_no_snapshot_uses() { // The first save puts both files in one pack. The second save rewrites one of them. After the // forget, that pack holds a used and an unused blob, so a prune without limits repacks it. @@ -1440,6 +1454,7 @@ async fn two_prunes_without_a_grace_period_give_back_the_data_that_no_snapshot_u } #[test] +#[timeout("60s")] async fn a_prune_of_a_scope_without_a_repository_gives_nothing() { let storage = Arc::new(InMemoryBlobStorage::new()); From 45a5f1d6ff55f538849d45e4099b7b976d87b1f4 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 21:11:35 -0700 Subject: [PATCH 098/126] Treat a final marker call as the start of the prune of a dropped delete in the sweep, and run that order as a fixed case --- .../filesystem_snapshot/rustic/store/tests.rs | 24 +++++++++++++++++ .../rustic/store/tests/sweep.rs | 27 ++++++++++++++----- 2 files changed, 44 insertions(+), 7 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index d4634c15c9..b3b0f190f2 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -4948,6 +4948,30 @@ async fn two_deletes_make_at_most_one_prune_when_one_is_dropped_at_its_first_ste assert!(cases.is_ok(), "{cases:?}"); } +#[test] +#[timeout("60s")] +async fn two_deletes_make_at_most_one_prune_when_one_is_dropped_as_its_prune_starts() { + // A drop right after the second read of the ledger can come just after the prune started. The + // listing of the packs then waits for a step until the cancel ends it, and the guard writes + // the final marker. The order ran into that race in a random case, so the test runs it many + // times. + let (shared, prepared) = prepared_scope().await; + let schedule = sweep::Schedule { + first: 1, + turns: vec![16, 10, 3, 9, 14], + fail: None, + late: None, + drop: Some((1, 10)), + }; + + let cases = futures::stream::iter(0..DROP_REPEATS) + .then(|_| sweep::run_case(&shared, &prepared, &schedule)) + .try_fold(0usize, |cases, _| async move { Ok(cases + 1) }) + .await; + + assert!(cases.is_ok(), "{cases:?}"); +} + #[test] #[timeout("60s")] async fn two_deletes_make_at_most_one_prune_in_random_orders_with_a_failed_call() { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs index 980ee7fc02..a5724991a3 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs @@ -161,11 +161,13 @@ async fn until(condition: impl Fn() -> bool) -> bool { /// Waits until the delete waits for a step or ended. While the prune waits for the listing of /// the packs, it also waits until the first new marker of the claim waits for a step, so each -/// prune writes a new marker right after that listing. +/// prune writes a new marker right after that listing. A dropped delete writes no new marker, +/// because the cancel stops the refresh, and the cancel also ends its listing of the packs. async fn settle(delete: &Delete) -> bool { until(|| { let waiting = delete.storage.waiting_steps(); - delete.finished() || waiting > usize::from(prune_waits(&delete.storage)) + let refreshing = !delete.dropped && prune_waits(&delete.storage); + delete.finished() || waiting > usize::from(refreshing) }) .await } @@ -468,9 +470,9 @@ fn forgotten_at(log: &[Step], who: usize) -> Option { } /// Gives each claim, or marker of a claim, that a delete wrote and that stays, when the first -/// failed call of that delete, or its drop, came after it took its claim and before it called for -/// the listing of the packs. A claim that the other delete wrote later at the same path is not the claim of the -/// delete. A blob whose own delete failed is left out, because no call can remove it then, and a +/// failed call of that delete, or its drop, came after it took its claim and before its prune +/// started: before it called for the listing of the packs, and when it made no final marker call. +/// A claim that the other delete wrote later at the same path is not the claim of the delete. A blob whose own delete failed is left out, because no call can remove it then, and a /// claim or a marker that stays only delays a prune. fn kept_claims(log: &[Step], claims: &[String]) -> Vec { [0, 1] @@ -484,7 +486,8 @@ fn kept_claims(log: &[Step], claims: &[String]) -> Vec { let pruning = log[..=failed_at] .iter() .any(|step| own(step) && is_prune_start(step.op_label, &step.path)); - (claimed_at < failed_at && !pruning).then_some((who, claimed_at)) + (claimed_at < failed_at && !pruning && !marked_final(log, who)) + .then_some((who, claimed_at)) }) .flat_map(|(who, claimed_at)| { let claim = &log[claimed_at].path; @@ -525,6 +528,15 @@ fn kept_claims(log: &[Step], claims: &[String]) -> Vec { .collect() } +/// Tells whether the delete called for the final marker of its claim. The guard writes it only +/// after the prune started. A drop can come right after the start, when the cancel ends the +/// listing of the packs before it takes a step, so the final marker is then the only step that +/// shows the start. +fn marked_final(log: &[Step], who: usize) -> bool { + log.iter() + .any(|step| step.delete == who && !step.landed && step.op_label == "final_marker") +} + /// Gives the claim of each delete whose prune started and that wrote no ledger entry, when that /// claim is gone. A prune that started keeps its claim, so the next prune waits for the hold. fn started_claims_gone(log: &[Step], claims: &[String]) -> Vec { @@ -539,7 +551,8 @@ fn started_claims_gone(log: &[Step], claims: &[String]) -> Vec { let started = log .iter() .filter(own) - .any(|step| is_prune_start(step.op_label, &step.path)); + .any(|step| is_prune_start(step.op_label, &step.path)) + || marked_final(log, who); let ledger_written = log .iter() .filter(own) From a2f1a1b77fb5ef0e1df5122bde23f32f49d80266 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 21:14:20 -0700 Subject: [PATCH 099/126] Leave the timeout off the three repository settings and inspect tests --- golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs | 3 --- 1 file changed, 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index 601ea33a76..d50d9814f5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -1148,7 +1148,6 @@ fn each_setting_goes_into_its_rustic_option() { } #[test] -#[timeout("60s")] async fn a_repository_keeps_the_settings_of_its_first_save_and_inspect_gives_them() { let storage = Arc::new(InMemoryBlobStorage::new()); let (fixed_scope, default_scope) = (new_scope(), new_scope()); @@ -1196,7 +1195,6 @@ async fn a_repository_keeps_the_settings_of_its_first_save_and_inspect_gives_the } #[test] -#[timeout("60s")] async fn inspect_gives_an_error_for_fixed_chunks_that_no_setting_can_hold() { // A repository that rustic makes with fixed chunks of 4 GiB has a chunk size that does not fit // `Chunking::Fixed`. The inspection must not report it as a Rabin repository. @@ -1234,7 +1232,6 @@ async fn inspect_gives_an_error_for_fixed_chunks_that_no_setting_can_hold() { } #[test] -#[timeout("60s")] async fn inspect_gives_the_snapshots_the_name_and_the_phases_and_nothing_without_a_repository() { let storage = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); From d9d36e86199b54e7d5180278dc8abc6377735907 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 21:53:02 -0700 Subject: [PATCH 100/126] Open the repository of the winner when a first save loses its creation at the second config check of the fork --- .../src/filesystem_snapshot/rustic/store.rs | 8 ++- .../filesystem_snapshot/rustic/store/tests.rs | 58 +++++++++++++++++++ 2 files changed, 65 insertions(+), 1 deletion(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index c58342f5b4..ae48bb3ab6 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -1097,7 +1097,13 @@ fn stage_save( tree: &Path, parent: Option<(SnapshotName, ChangeDetection)>, ) -> anyhow::Result> { - let (repository, _) = open_or_create(backend, key, &policy.repository)?; + // The init of the fork checks the config a second time before its write, so a save that loses + // the race to create the repository can fail at that check with an error that is not + // `ConfigExists`. A config that is there after the error is the repository of the winner. + let repository = match open_or_create(backend.clone(), key, &policy.repository) { + Ok((repository, _)) => repository, + Err(error) => open_existing(backend, key).ok().flatten().ok_or(error)?, + }; let before = scope_snapshots(&repository)?; if has_name(&before, name) { return Ok(None); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index b3b0f190f2..f6878bb526 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -3662,6 +3662,64 @@ async fn a_shut_down_between_the_claim_listing_and_the_claim_guard_makes_no_stor ); } +#[test] +#[timeout("60s")] +async fn a_save_that_loses_the_creation_of_the_repository_after_its_first_config_check_saves_into_the_winner() + { + // The gate holds the second check of the config by the losing save, which the init of rustic + // makes before its config write. The winning save creates the repository meanwhile, so that + // check finds the config. + let shared = Arc::new(InMemoryBlobStorage::new()); + let checks = Arc::new(AtomicUsize::new(0)); + let losing = ScriptedBlobStorage::new(shared.clone(), { + let checks = checks.clone(); + move |op_label, path| { + if op_label == "stat" + && path == Path::new("config") + && checks.fetch_add(1, Ordering::SeqCst) == 1 + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let winner = store(shared.clone(), policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let loser = store(losing.clone(), policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + let (won_tree, lost_tree) = (one_file_tree("winner"), one_file_tree("loser")); + let losing_save = tokio::spawn({ + let (loser, scope, path) = (loser.clone(), scope.clone(), lost_tree.path().to_path_buf()); + async move { loser.save(&scope, &name("p-loser"), &path, None).await } + }); + let held = eventually(|| checks.load(Ordering::SeqCst) >= 2).await; + + let won = winner + .save(&scope, &name("p-winner"), won_tree.path(), None) + .await; + losing.open_gate(); + let lost = tokio::time::timeout(LIMIT, losing_save).await; + + assert!(won.is_ok(), "{won:?}"); + assert!(matches!(lost, Ok(Ok(Ok(_)))), "{lost:?}"); + assert_eq!( + ( + held, + restored_listing(&winner, &scope, &name("p-loser")) + .await + .ok(), + restored_listing(&loser, &scope, &name("p-winner")) + .await + .ok(), + ), + ( + true, + Some(listing(lost_tree.path())), + Some(listing(won_tree.path())), + ) + ); +} + #[test] #[timeout("60s")] async fn a_save_that_reaches_its_publish_after_shut_down_publishes_nothing() { From 7392a2370b82c4a7222269107d04f9ef96a00a13 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 21:58:03 -0700 Subject: [PATCH 101/126] Shut the store down in the shut-down test only after the save holds its pack write --- .../filesystem_snapshot/rustic/store/tests.rs | 23 ++++++++++--------- 1 file changed, 12 insertions(+), 11 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index f6878bb526..cb964b447e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -3797,14 +3797,21 @@ async fn delete_scope_and_copy_scope_after_shut_down_give_storage() { #[test] #[timeout("60s")] async fn shut_down_ends_running_operations_before_it_returns() { - let storage = - ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + // The first pack write of the save waits at the gate, and it cancels `reached`, so the test + // shuts the store down only when the save holds a running storage call. The save runs at a low + // priority, so the test waits for that point without a bound of its own. + let reached = tokio_util::sync::CancellationToken::new(); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let reached = reached.clone(); + move |op_label, path| { if op_label == "write" && path.starts_with("data") { + reached.cancel(); Script::WaitForGate } else { Script::Pass } - }); + } + }); let store = store( storage.clone(), policy(LONG_DEADLINE, NEVER, Duration::ZERO), @@ -3817,13 +3824,7 @@ async fn shut_down_ends_running_operations_before_it_returns() { let path = tree.path().to_path_buf(); async move { store.save(&scope, &name("p-held"), &path, None).await } }); - let held = eventually(|| { - storage - .calls() - .iter() - .any(|(op_label, path)| *op_label == "write" && path.starts_with("data")) - }) - .await; + reached.cancelled().await; let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); let saved = tokio::time::timeout(LIMIT, saving).await; @@ -3838,7 +3839,7 @@ async fn shut_down_ends_running_operations_before_it_returns() { later.as_ref().is_err_and(|error| is_storage(error, false)), "{later:?}" ); - assert_eq!((held, stopped, store.work_in_flight()), (true, true, 0)); + assert_eq!((stopped, store.work_in_flight()), (true, 0)); } /// The operation of the store that a test drops. From 4c4c4ff4227c4996b3657eca4cd7abb4f022d44d Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 22:15:40 -0700 Subject: [PATCH 102/126] Wait a full hold of a claim after the last prune before the next one is due --- .../src/filesystem_snapshot/rustic/prune.rs | 94 +++++++++++++------ .../src/filesystem_snapshot/rustic/store.rs | 5 +- 2 files changed, 67 insertions(+), 32 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 99240ba86d..fdcbe0ea52 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -71,13 +71,14 @@ impl Percent { } } -/// Tells whether the grace period and the margin for clock skew passed at `now` since the last -/// prune. A time more than the margin after `now` counts as missing. -fn grace_passed(ledger: &PruneLedger, now: Timestamp, grace: Duration) -> bool { +/// Tells whether the hold of a claim passed at `now` since the last prune. The time of the last +/// prune is after its final marker, so the next prune waits a full hold from the newest claim +/// marker that the last prune wrote. A time more than the margin after `now` counts as missing. +fn hold_passed(ledger: &PruneLedger, now: Timestamp, grace: Duration, deadline: Duration) -> bool { ledger .last_prune .filter(|last| !beyond_margin(*last, now)) - .is_none_or(|last| passed_since(last, now, grace.saturating_add(CLOCK_SKEW_MARGIN))) + .is_none_or(|last| passed_since(last, now, claim_hold(grace, deadline))) } /// Tells whether the grace period passed at `now` since the time. @@ -89,19 +90,20 @@ fn passed_since(time: Timestamp, now: Timestamp, grace: Duration) -> bool { } /// Tells whether [`prune_due`] needs the size of the repository at `now`. Only freed bytes after -/// the grace period, without marked packs, need it. +/// the hold of a claim, without marked packs, need it. pub(super) fn needs_repository_size( ledger: &PruneLedger, freed_bytes: u64, now: Timestamp, grace: Duration, + deadline: Duration, ) -> bool { - grace_passed(ledger, now, grace) && freed_bytes > 0 && !ledger.awaiting_removal + hold_passed(ledger, now, grace, deadline) && freed_bytes > 0 && !ledger.awaiting_removal } /// Tells whether a prune is due at `now`. /// -/// A prune is due when the grace period passed since the last prune, and the freed bytes reach the +/// A prune is due when the hold of a claim passed since the last prune, and the freed bytes reach the /// threshold share of `repository_bytes`, rounded down to a whole byte, or the last prune marked /// packs. A threshold of zero bytes counts as one byte, so a prune never runs for a scope that /// freed nothing and marked nothing. @@ -112,9 +114,10 @@ pub(super) fn prune_due( repository_bytes: u64, threshold: Percent, grace: Duration, + deadline: Duration, ) -> bool { let work = freed_bytes >= threshold.of(repository_bytes).max(1) || ledger.awaiting_removal; - grace_passed(ledger, now, grace) && work + hold_passed(ledger, now, grace, deadline) && work } /// Gives the size of the repository of the scope: the sum of the sizes of its packs. @@ -811,6 +814,7 @@ mod tests { repository_bytes, TEN_PERCENT, GRACE, + DEADLINE, ) }; @@ -827,7 +831,9 @@ mod tests { } #[test] - fn no_second_prune_runs_within_the_grace_period() { + fn no_second_prune_runs_within_the_hold_of_a_claim_after_the_last_prune() { + // The grace period and the margin passed at `HELD_MILLIS`, and the full hold at + // `HOLD_MILLIS`. let last = 1_000_000; let full = |now| { prune_due( @@ -837,32 +843,35 @@ mod tests { 1000, TEN_PERCENT, GRACE, + DEADLINE, ) }; assert_eq!( [ full(last), - full(last + HELD_MILLIS - 1), full(last + HELD_MILLIS), - full(last + HELD_MILLIS + 1), + full(last + HOLD_MILLIS - 1), + full(last + HOLD_MILLIS), + full(last + HOLD_MILLIS + 1), ], - [false, false, true, true] + [false, false, false, true, true] ); } #[test] - fn marked_packs_make_a_prune_due_after_the_grace_period_without_freed_bytes() { + fn marked_packs_make_a_prune_due_after_the_hold_without_freed_bytes() { let last = 1_000_000; - let after_grace = at(last + HELD_MILLIS); + let after_hold = at(last + HOLD_MILLIS); let due = |awaiting_removal| { prune_due( &ledger(Some(last), awaiting_removal), 0, - after_grace, + after_hold, 1000, TEN_PERCENT, GRACE, + DEADLINE, ) }; @@ -872,42 +881,69 @@ mod tests { #[test] fn a_zero_threshold_prunes_after_each_delete_that_freed_bytes() { let now = at(10_000_000); - let due = |ledger, freed| prune_due(&ledger, freed, now, 1000, Percent(0), Duration::ZERO); + let due = |ledger, freed| { + prune_due( + &ledger, + freed, + now, + 1000, + Percent(0), + Duration::ZERO, + DEADLINE, + ) + }; + let hold = u64::try_from(claim_hold(Duration::ZERO, DEADLINE).as_millis()).unwrap(); assert_eq!( [ due(ledger(None, false), 0), due(ledger(None, false), 1), - due(ledger(Some(10_000_000 - 120_000), false), 1), + due(ledger(Some(10_000_000 - hold), false), 1), ], [false, true, true] ); } #[test] - fn only_freed_bytes_after_the_grace_period_without_marked_packs_need_the_repository_size() { + fn only_freed_bytes_after_the_hold_without_marked_packs_need_the_repository_size() { let last = 1_000_000; let needs = |freed, awaiting_removal, now| { - needs_repository_size(&ledger(Some(last), awaiting_removal), freed, at(now), GRACE) + needs_repository_size( + &ledger(Some(last), awaiting_removal), + freed, + at(now), + GRACE, + DEADLINE, + ) }; assert_eq!( [ - needs(1, false, last + HELD_MILLIS), - needs(1, false, last + HELD_MILLIS - 1), - needs(0, false, last + HELD_MILLIS), - needs(1, true, last + HELD_MILLIS), + needs(1, false, last + HOLD_MILLIS), + needs(1, false, last + HOLD_MILLIS - 1), + needs(0, false, last + HOLD_MILLIS), + needs(1, true, last + HOLD_MILLIS), ], [true, false, false, false] ); } #[test] - fn the_margin_extends_the_grace_period_and_a_time_more_than_the_margin_ahead_counts_as_missing() - { + fn the_hold_holds_the_ledger_and_the_claims_and_a_time_more_than_the_margin_ahead_counts_as_missing() + { let now = 10_000_000; let margin = u64::try_from(CLOCK_SKEW_MARGIN.as_millis()).unwrap(); - let due = |last| prune_due(&ledger(Some(last), false), 1, at(now), 0, Percent(0), GRACE); + let due = |last| { + prune_due( + &ledger(Some(last), false), + 1, + at(now), + 0, + Percent(0), + GRACE, + DEADLINE, + ) + }; let claim = |claimed_at| { next_claim( &[ClaimEntry::Claim(0), ClaimEntry::Marker(0, at(claimed_at))], @@ -916,13 +952,11 @@ mod tests { ) }; - let grace = u64::try_from(GRACE.as_millis()).unwrap(); - assert_eq!( ( [ - due(now - grace), - due(now - grace - margin), + due(now - HOLD_MILLIS + 1), + due(now - HOLD_MILLIS), due(now + margin), due(now + margin + 1) ], diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index ae48bb3ab6..5f868f9834 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -594,7 +594,8 @@ impl RusticSnapshotStore { // reading can put a marker that another host wrote within the margin beyond the margin. let now = self.now(); let grace = self.policy.prune.keep_delete; - let size = if needs_repository_size(&ledger, records.bytes, now, grace) { + let deadline = self.policy.deadline; + let size = if needs_repository_size(&ledger, records.bytes, now, grace, deadline) { repository_bytes(&files).await.map_err(storage_failure)? } else { 0 @@ -606,6 +607,7 @@ impl RusticSnapshotStore { size, self.policy.prune_threshold, grace, + deadline, ) { return Ok(()); } @@ -613,7 +615,6 @@ impl RusticSnapshotStore { let listed = list_claims(&files, &claims) .await .map_err(storage_failure)?; - let deadline = self.policy.deadline; let ClaimChoice::Claim(number) = next_claim(&listed, self.now(), claim_hold(grace, deadline)) else { From f62303c3bf4c13e4658e1d6da8745074c0a3056c Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 22:17:00 -0700 Subject: [PATCH 103/126] Say how long the next prune waits in the docs of the prune grace period and the prune settings --- .../src/filesystem_snapshot/rustic/store.rs | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 5f868f9834..3285ffd73c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -74,9 +74,10 @@ use tracing::warn; /// the scope. The threshold is 10% of the size, rounded down to a whole byte, so about 10%. const PRUNE_THRESHOLD: Percent = Percent(10); -/// How long a pack that a prune marks stays before a later prune deletes it. It is also the -/// shortest time between two prunes of one scope. It must be longer than the longest save and the -/// longest restore. +/// How long a pack that a prune marks stays before a later prune deletes it. It must be longer than +/// the longest save and the longest restore. Two prunes of one scope never run at once. The next +/// prune waits a full hold from the newest claim marker that was written, and the hold starts from +/// this time. When writes fail, only the gap between two prunes can be shorter. const PRUNE_GRACE: Duration = Duration::from_secs(15 * 60); /// The settings of the store: the rustic settings of each operation, and the prune threshold. @@ -91,7 +92,9 @@ pub(super) struct StorePolicy { pub(super) save_threads: Option, /// The number of threads that read packs in a restore. pub(super) restore_reader_threads: NonZeroUsize, - /// The settings of a prune. `keep_delete` is also the shortest time between two prunes. + /// The settings of a prune. Two prunes never run at once. The next prune waits a full hold, + /// which starts from `keep_delete`, from the newest claim marker that was written. When writes + /// fail, only the gap between two prunes can be shorter. pub(super) prune: PruneSettings, /// The share of the size of the repository that deleted snapshots must free before a delete /// prunes. From dfbc73bf6ca24c360114da79aa18b7365522bd74 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 22:18:48 -0700 Subject: [PATCH 104/126] Run low-priority work on its own thread in its own pool on each platform --- .../filesystem_snapshot/rustic/priority.rs | 9 +++----- .../rustic/priority/tests.rs | 23 +++++++++++++++++++ 2 files changed, 26 insertions(+), 6 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs index d7b9dbf966..5ce29dce45 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs @@ -50,17 +50,13 @@ impl LowPriority { /// Runs the work at nice 19 on a new thread with the name, inside a new rayon pool, and waits /// for it. The threads that the work starts get the same nice value. On a platform other than - /// Linux the work runs as it is. + /// Linux the priority stays as it is, and the work still runs on its own thread in its own pool. pub(super) fn run( self, name: &'static str, work: impl FnOnce() -> anyhow::Result + Send + 'static, ) -> anyhow::Result { - if cfg!(target_os = "linux") { - self.on_own_thread(name, work) - } else { - work() - } + self.on_own_thread(name, work) } /// Runs the work on a new thread that first lowers its priority and builds the pool. A failure @@ -166,6 +162,7 @@ fn lower_own_priority() -> std::io::Result<()> { } } +/// Keeps the priority of the calling thread on a platform other than Linux. #[cfg(not(target_os = "linux"))] fn lower_own_priority() -> std::io::Result<()> { Ok(()) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs index f9ea230354..0533a1e200 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs @@ -57,6 +57,29 @@ fn work_at_low_priority_runs_at_nice_19_in_a_pool_of_its_own_and_the_caller_keep ); } +#[test] +fn work_whose_priority_step_does_nothing_runs_on_its_own_thread_in_a_pool_of_its_own() { + // Off Linux, the step that lowers the priority does nothing. The rest of the path is the same + // on each platform. + let before = own_nice(); + let low_priority = LowPriority { + lower: || Ok(()), + ..LowPriority::new(NonZeroUsize::new(2)) + }; + + let inside = low_priority.run("fs-snap-test", seen); + + assert_eq!( + inside.ok().map(|(nice, name, in_pool, threads)| ( + nice, + name.is_some_and(|name| name.starts_with("fs-snap-test-")), + in_pool, + threads + )), + Some((before, true, true, 2)) + ); +} + #[test] fn work_whose_priority_cannot_be_lowered_still_runs_at_the_normal_priority() { let before = own_nice(); From 63f5301372f45ec3d9c9760c0779e1302823ad3e Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 22:29:12 -0700 Subject: [PATCH 105/126] Count the check of the local path of a save or a restore in the tracker --- .../src/filesystem_snapshot/rustic/store.rs | 106 +++++++++++------- .../filesystem_snapshot/rustic/store/tests.rs | 41 +++++++ 2 files changed, 104 insertions(+), 43 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 3285ffd73c..61a080ccc5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -167,9 +167,9 @@ pub(crate) struct RusticSnapshotStore { policy: StorePolicy, /// The parent of the token of each operation. root: CancellationToken, - /// Counts the blocking tasks, the backends, the blob calls of the store, the publishes, the - /// deletes of dropped publishes, and the claim guards with their release and final-marker - /// tasks. + /// Counts the blocking tasks, the checks of local paths, the backends, the blob calls of the + /// store, the publishes, the deletes of dropped publishes, and the claim guards with their + /// release and final-marker tasks. tracker: TaskTracker, /// Runs saves and prunes at a low priority. low_priority: LowPriority, @@ -180,6 +180,10 @@ pub(crate) struct RusticSnapshotStore { /// sets it. #[cfg(test)] pub(super) claim_gate: Option>, + /// Holds the check of the local path of a save or a restore on its blocking thread, when a + /// test sets it. + #[cfg(test)] + pub(super) path_check_gate: Option>, /// Makes each backend build fail while a test sets it. #[cfg(test)] pub(super) refuse_backends: Arc, @@ -467,6 +471,8 @@ impl RusticSnapshotStore { #[cfg(test)] claim_gate: None, #[cfg(test)] + path_check_gate: None, + #[cfg(test)] refuse_backends: Arc::default(), #[cfg(test)] clock_ahead: Arc::default(), @@ -478,9 +484,10 @@ impl RusticSnapshotStore { /// ends them: a publish that started before the cancel runs to its end, and a claim guard /// releases its claim, or writes the final marker of a prune that started. A save that /// reaches its publish after the cancel publishes nothing and gives `Storage`. The call waits - /// until no blocking task, backend, blob call of the store, publish, delete of a dropped - /// publish, claim guard, or release or final marker of a claim guard remains. A blob call that is not polled holds the wait - /// until it is polled again, and then it ends at once. The runtime must not drop before it + /// until no blocking task, check of a local path, backend, blob call of the store, publish, + /// delete of a dropped publish, claim guard, or release or final marker of a claim guard + /// remains. A blob call that is not polled holds the wait until it is polled again, and then + /// it ends at once. The runtime must not drop before it /// returns, because a storage call after its time driver stops aborts the process. pub(crate) async fn shut_down(&self) { self.root.cancel(); @@ -787,7 +794,7 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { parent: Option<(&SnapshotName, ChangeDetection)>, ) -> Result { let (token, _guard) = self.start()?; - check_tree(tree).await?; + self.check_tree(tree).await?; let stage = Arc::new(SnapshotStage::default()); let backend = Arc::new(self.backend(scope, &token)?.staging_in(stage.clone())); let key = self.key.clone(); @@ -830,7 +837,7 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { into: &Path, ) -> Result { let (token, _guard) = self.start()?; - check_destination(into).await?; + self.check_destination(into).await?; let backend = Arc::new(self.backend(scope, &token)?); let key = self.key.clone(); let options = store_restore_options(&self.policy); @@ -1002,31 +1009,55 @@ impl Lookup { } } -/// Runs the check of a local path on a blocking thread. A thread that fails gives the error of the -/// path. -async fn check_path( - operation: NativeOperation, - path: &Path, - error: fn(std::io::Error) -> SnapshotStoreError, - check: fn(&Path) -> Result<(), SnapshotStoreError>, -) -> Result<(), SnapshotStoreError> { - let path: Box = path.into(); - execute_native(NativeStorageProfile::Unknown, operation, move || { - check(&path) - }) - .await - .map_err(|failed| error(std::io::Error::other(failed)))? -} +impl RusticSnapshotStore { + /// Runs the check of a local path on a blocking thread. A thread that fails gives the error of + /// the path. The tracker counts the check until it ends, so `shut_down` waits for it, also when + /// the operation drops first. + async fn check_path( + &self, + operation: NativeOperation, + path: &Path, + error: fn(std::io::Error) -> SnapshotStoreError, + check: fn(&Path) -> Result<(), SnapshotStoreError>, + ) -> Result<(), SnapshotStoreError> { + let path: Box = path.into(); + let tracked = self.tracker.token(); + #[cfg(test)] + let gate = self.path_check_gate.clone(); + execute_native(NativeStorageProfile::Unknown, operation, move || { + let _tracked = tracked; + #[cfg(test)] + if let Some(gate) = gate { + gate.reached.notify_one(); + futures::executor::block_on(gate.open.notified()); + } + check(&path) + }) + .await + .map_err(|failed| error(std::io::Error::other(failed)))? + } -/// Checks that the tree of a save is a directory at an absolute path. -async fn check_tree(tree: &Path) -> Result<(), SnapshotStoreError> { - check_path( - NativeOperation::Metadata, - tree, - SnapshotStoreError::Source, - tree_is_valid, - ) - .await + /// Checks that the tree of a save is a directory at an absolute path. + async fn check_tree(&self, tree: &Path) -> Result<(), SnapshotStoreError> { + self.check_path( + NativeOperation::Metadata, + tree, + SnapshotStoreError::Source, + tree_is_valid, + ) + .await + } + + /// Checks that the directory of a restore is an empty directory with a UTF-8 path. + async fn check_destination(&self, into: &Path) -> Result<(), SnapshotStoreError> { + self.check_path( + NativeOperation::DirectoryEnumeration, + into, + SnapshotStoreError::Destination, + destination_is_valid, + ) + .await + } } fn tree_is_valid(tree: &Path) -> Result<(), SnapshotStoreError> { @@ -1043,17 +1074,6 @@ fn tree_is_valid(tree: &Path) -> Result<(), SnapshotStoreError> { Ok(()) } -/// Checks that the directory of a restore is an empty directory with a UTF-8 path. -async fn check_destination(into: &Path) -> Result<(), SnapshotStoreError> { - check_path( - NativeOperation::DirectoryEnumeration, - into, - SnapshotStoreError::Destination, - destination_is_valid, - ) - .await -} - fn destination_is_valid(into: &Path) -> Result<(), SnapshotStoreError> { let refused = |kind, reason: &str| { SnapshotStoreError::Destination(std::io::Error::new( diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index cb964b447e..f9ae5dcb49 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -3720,6 +3720,47 @@ async fn a_save_that_loses_the_creation_of_the_repository_after_its_first_config ); } +#[test] +#[timeout("60s")] +async fn shut_down_waits_for_the_check_of_the_tree_of_a_save() { + // The gate holds the check of the tree on its blocking thread. + let gate = Arc::new(StepGate::default()); + let store = Arc::new(RusticSnapshotStore { + path_check_gate: Some(gate.clone()), + ..RusticSnapshotStore::with_policy( + Arc::new(InMemoryBlobStorage::new()), + key(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ) + }); + let scope = new_scope(); + let tree = one_file_tree("checked"); + let saving = tokio::spawn({ + let (store, scope, path) = (store.clone(), scope.clone(), tree.path().to_path_buf()); + async move { store.save(&scope, &name("p-checked"), &path, None).await } + }); + let reached = tokio::time::timeout(LIMIT, gate.reached.notified()) + .await + .is_ok(); + + let shutting = store.shut_down(); + tokio::pin!(shutting); + let waited = tokio::time::timeout(Duration::from_millis(200), &mut shutting) + .await + .is_err(); + gate.open.notify_one(); + // A finished future must not be polled again, so the second wait runs only after a first wait + // that timed out. + let stopped = !waited || tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + let saved = tokio::time::timeout(LIMIT, saving).await; + + assert!( + matches!(&saved, Ok(Ok(Err(error))) if is_storage(error, false) || is_storage(error, true)), + "{saved:?}" + ); + assert_eq!((reached, waited, stopped), (true, true, true)); +} + #[test] #[timeout("60s")] async fn a_save_that_reaches_its_publish_after_shut_down_publishes_nothing() { From e2a377a4d114dade0374294ae87301aee16de384 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 22:32:20 -0700 Subject: [PATCH 106/126] Share the claim directory of a guard and its tasks as one Arc path --- .../src/filesystem_snapshot/rustic/store.rs | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 61a080ccc5..ee95d5979c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -239,7 +239,7 @@ struct ClaimGuard { /// The blobs of the scope, with a token that nothing cancels, so a release and a final marker /// also run after a cancel or a drop. files: SnapshotFiles, - directory: Box, + directory: Arc, number: u64, /// Whether this delete wrote the claim. A delete that did not write it deletes only its /// markers. @@ -256,7 +256,7 @@ impl ClaimGuard { /// write of that marker. fn new( files: &SnapshotFiles, - directory: &Path, + directory: &Arc, number: u64, marker: Box, tracked: TaskTrackerToken, @@ -266,7 +266,7 @@ impl ClaimGuard { cancel: CancellationToken::new(), ..files.clone() }, - directory: directory.into(), + directory: directory.clone(), number, claimed: AtomicBool::new(false), markers: Mutex::new(vec![marker]), @@ -621,7 +621,7 @@ impl RusticSnapshotStore { ) { return Ok(()); } - let claims = claims_directory(&ledger); + let claims: Arc = claims_directory(&ledger).into(); let listed = list_claims(&files, &claims) .await .map_err(storage_failure)?; From 19903c0480f174be5ca96e7b412d08f63ea5240b Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 22:54:19 -0700 Subject: [PATCH 107/126] Give retryable Storage for an index file that a prune deleted after the listing --- .../src/filesystem_snapshot/rustic/fault.rs | 56 +++++++- .../filesystem_snapshot/rustic/store/tests.rs | 126 ++++++++++++++++++ 2 files changed, 179 insertions(+), 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs index ad812d2364..4aafb945e7 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs @@ -21,6 +21,7 @@ use super::prune::SNAPSHOTS_PATH; use crate::filesystem_snapshot::SnapshotStoreError; use golem_service_base::storage::blob::BlobNameError; +use rustic_core::FileType; use std::error::Error; use std::fmt::{Display, Formatter}; use std::path::Path; @@ -126,6 +127,17 @@ pub(super) fn is_snapshot_missing(error: &(dyn Error + 'static)) -> bool { }) } +/// Tells whether an error in the chain is [`FileMissing`] for an index file. A prune writes its new +/// index files and then deletes the old ones at once, so an operation that listed an index file +/// before a prune can find it gone at its read. A later try lists the new index files. +pub(super) fn is_index_missing(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| { + error + .downcast_ref::() + .is_some_and(|missing| missing.path.starts_with(FileType::Index.dirname())) + }) +} + /// Tells whether an error in the chain is [`ConfigExists`]. pub(super) fn is_config_exists(error: &(dyn Error + 'static)) -> bool { chain(error).any(|error| error.is::()) @@ -144,11 +156,14 @@ pub(super) enum Operation { Prune, } -/// A failed storage call gives `Storage`, retryable unless a name error caused it. An I/O error +/// A failed storage call gives `Storage`, retryable unless a name error caused it. An index file +/// that a prune deleted after the listing gives retryable `Storage` in each operation. An I/O error /// gives `Source` in a save and `Destination` in a restore. Each other error gives `Storage` that /// is not retryable in a save or a prune, and `Corrupt` in a restore or a read of the repository. +/// So a pack that is gone stays `Corrupt`, because a prune deletes a pack only after the grace +/// period. pub(super) fn classify(operation: Operation, error: anyhow::Error) -> SnapshotStoreError { - let from_storage = is_storage_failure(error.as_ref()); + let from_storage = is_storage_failure(error.as_ref()) || is_index_missing(error.as_ref()); let io_kind = chain(error.as_ref()) .find_map(|error| error.downcast_ref::()) .map(std::io::Error::kind); @@ -201,7 +216,8 @@ fn chain<'a>(error: &'a (dyn Error + 'static)) -> impl Iterator, +) -> ( + Arc, + Arc, + Arc, + Arc, +) { + let (hold, held) = ( + Arc::new(AtomicBool::new(false)), + Arc::new(AtomicUsize::new(0)), + ); + let storage = ScriptedBlobStorage::new(shared.clone(), { + let (hold, held) = (hold.clone(), held.clone()); + move |op_label, path| { + if op_label == "read" && path.starts_with("index") && hold.load(Ordering::SeqCst) { + held.fetch_add(1, Ordering::SeqCst); + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + (store, storage, hold, held) +} + +/// Saves `p-1` and `p-2` of two trees in a new scope of the shared storage, and deletes `p-2` +/// through a store that prunes at once. The prune deletes the index file of `p-2`. +async fn scope_with_a_prune_to_come( + shared: &Arc, +) -> (SnapshotScope, Arc, Scratch) { + let pruning = store( + shared.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let (kept, deleted) = (fixture_tree(), one_file_tree("deleted")); + pruning + .save(&scope, &name("p-1"), kept.path(), None) + .await + .unwrap(); + pruning + .save(&scope, &name("p-2"), deleted.path(), None) + .await + .unwrap(); + (scope, pruning, kept) +} + +#[test] +#[timeout("60s")] +async fn a_restore_whose_index_file_a_prune_deleted_after_the_listing_gives_retryable_storage_and_a_retry_restores() + { + let shared = Arc::new(InMemoryBlobStorage::new()); + let (scope, pruning, kept) = scope_with_a_prune_to_come(&shared).await; + let (restoring, storage, hold, held) = holding_index_reads(&shared); + hold.store(true, Ordering::SeqCst); + let first = tokio::spawn({ + let (restoring, scope) = (restoring.clone(), scope.clone()); + async move { + let into = Scratch::new(); + restoring.restore(&scope, &name("p-1"), into.path()).await + } + }); + let reached = eventually(|| held.load(Ordering::SeqCst) > 0).await; + + let deleted = pruning.delete(&scope, &name("p-2")).await; + let pruned = ledger(&shared, &scope).await.last_prune.is_some(); + hold.store(false, Ordering::SeqCst); + storage.open_gate(); + let first = tokio::time::timeout(LIMIT, first).await; + let again = restored_listing(&restoring, &scope, &name("p-1")).await; + + assert!( + matches!(&first, Ok(Ok(Err(error))) if is_storage(error, true)), + "{first:?}" + ); + assert_eq!( + (reached, deleted.is_ok(), pruned, again.ok()), + (true, true, true, Some(listing(kept.path()))) + ); +} + +#[test] +#[timeout("60s")] +async fn a_save_whose_index_file_a_prune_deleted_after_the_listing_gives_retryable_storage_and_a_retry_saves() + { + let shared = Arc::new(InMemoryBlobStorage::new()); + let (scope, pruning, _) = scope_with_a_prune_to_come(&shared).await; + let (saving, storage, hold, held) = holding_index_reads(&shared); + let tree = one_file_tree("new"); + hold.store(true, Ordering::SeqCst); + let first = tokio::spawn({ + let (saving, scope, path) = (saving.clone(), scope.clone(), tree.path().to_path_buf()); + async move { saving.save(&scope, &name("p-3"), &path, None).await } + }); + let reached = eventually(|| held.load(Ordering::SeqCst) > 0).await; + + let deleted = pruning.delete(&scope, &name("p-2")).await; + let pruned = ledger(&shared, &scope).await.last_prune.is_some(); + hold.store(false, Ordering::SeqCst); + storage.open_gate(); + let first = tokio::time::timeout(LIMIT, first).await; + let again = saving.save(&scope, &name("p-3"), tree.path(), None).await; + + assert!( + matches!(&first, Ok(Ok(Err(error))) if is_storage(error, true)), + "{first:?}" + ); + assert!(again.is_ok(), "{again:?}"); + assert_eq!( + ( + reached, + deleted.is_ok(), + pruned, + restored_listing(&saving, &scope, &name("p-3")).await.ok(), + ), + (true, true, true, Some(listing(tree.path()))) + ); +} + #[test] #[timeout("60s")] async fn a_save_that_reaches_its_publish_after_shut_down_publishes_nothing() { From b04e4eb23967188e7e9a65387009d34581043e1c Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sat, 26 Sep 2026 22:57:05 -0700 Subject: [PATCH 108/126] Name two prune tests by what they prove, and fix the words and the layout of the store docs --- .../src/filesystem_snapshot/rustic/store.rs | 21 ++++++++++++------- .../filesystem_snapshot/rustic/store/tests.rs | 9 +++----- .../rustic/store/tests/sweep.rs | 5 +++-- 3 files changed, 19 insertions(+), 16 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index ee95d5979c..a9540d2adb 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -76,8 +76,10 @@ const PRUNE_THRESHOLD: Percent = Percent(10); /// How long a pack that a prune marks stays before a later prune deletes it. It must be longer than /// the longest save and the longest restore. Two prunes of one scope never run at once. The next -/// prune waits a full hold from the newest claim marker that was written, and the hold starts from -/// this time. When writes fail, only the gap between two prunes can be shorter. +/// prune waits a full hold from the newest claim marker that was written. The hold follows from +/// this time: this time less one storage call deadline, but at least one deadline, plus the margin +/// for clock skew, plus two deadlines. When writes fail, only the gap between two prunes can be +/// shorter. const PRUNE_GRACE: Duration = Duration::from_secs(15 * 60); /// The settings of the store: the rustic settings of each operation, and the prune threshold. @@ -92,9 +94,10 @@ pub(super) struct StorePolicy { pub(super) save_threads: Option, /// The number of threads that read packs in a restore. pub(super) restore_reader_threads: NonZeroUsize, - /// The settings of a prune. Two prunes never run at once. The next prune waits a full hold, - /// which starts from `keep_delete`, from the newest claim marker that was written. When writes - /// fail, only the gap between two prunes can be shorter. + /// The settings of a prune. Two prunes never run at once. The next prune waits a full hold from + /// the newest claim marker that was written. The hold follows from `keep_delete`: that time + /// less one storage call deadline, but at least one deadline, plus the margin for clock skew, + /// plus two deadlines. When writes fail, only the gap between two prunes can be shorter. pub(super) prune: PruneSettings, /// The share of the size of the repository that deleted snapshots must free before a delete /// prunes. @@ -487,15 +490,17 @@ impl RusticSnapshotStore { /// until no blocking task, check of a local path, backend, blob call of the store, publish, /// delete of a dropped publish, claim guard, or release or final marker of a claim guard /// remains. A blob call that is not polled holds the wait until it is polled again, and then - /// it ends at once. The runtime must not drop before it - /// returns, because a storage call after its time driver stops aborts the process. + /// it ends at once. The runtime must not drop before it returns, because a storage call after + /// its time driver stops aborts the process. pub(crate) async fn shut_down(&self) { self.root.cancel(); self.tracker.close(); self.tracker.wait().await; } - /// Gives the number of blocking tasks, backends and deletes of the store that have not ended. + /// Gives the number of blocking tasks, checks of local paths, backends, blob calls of the store, + /// publishes, deletes of dropped publishes, and claim guards with their release and + /// final-marker tasks that have not ended. #[cfg(test)] pub(super) fn work_in_flight(&self) -> usize { self.tracker.len() diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 1429192970..98a67ead47 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -2994,22 +2994,19 @@ async fn a_prune_that_fails_keeps_its_claim_so_no_second_prune_runs_within_the_h #[test] #[timeout("60s")] -async fn a_claim_older_than_the_grace_period_does_not_block_a_prune() { +async fn a_claim_without_a_marker_does_not_block_a_prune() { let grace = Duration::from_secs(3600); let storage = Arc::new(InMemoryBlobStorage::new()); let store = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); let scope = new_scope(); save_each(&store, &scope, &["p-1", "p-2"]).await; - let old = golem_common::model::Timestamp::now_utc() - .to_millis() - .saturating_sub(2 * 3_600_000); storage .put_raw( "test", "test", scope.0.clone(), Path::new("golem/prune-claims/none/0"), - old.to_string().as_bytes(), + &[], ) .await .unwrap(); @@ -3078,7 +3075,7 @@ async fn a_delete_below_the_threshold_does_not_prune() { #[test] #[timeout("60s")] -async fn no_second_prune_runs_within_the_grace_period() { +async fn no_second_prune_runs_within_the_hold_after_a_prune() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs index a5724991a3..513aaf3452 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs @@ -472,8 +472,9 @@ fn forgotten_at(log: &[Step], who: usize) -> Option { /// Gives each claim, or marker of a claim, that a delete wrote and that stays, when the first /// failed call of that delete, or its drop, came after it took its claim and before its prune /// started: before it called for the listing of the packs, and when it made no final marker call. -/// A claim that the other delete wrote later at the same path is not the claim of the delete. A blob whose own delete failed is left out, because no call can remove it then, and a -/// claim or a marker that stays only delays a prune. +/// A claim that the other delete wrote later at the same path is not the claim of the delete. A +/// blob whose own delete failed is left out, because no call can remove it then, and a claim or a +/// marker that stays only delays a prune. fn kept_claims(log: &[Step], claims: &[String]) -> Vec { [0, 1] .into_iter() From 0145e5952cb565b26f8db71d9249a50aeff7ac83 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 03:09:13 -0700 Subject: [PATCH 109/126] Test that a marker write that starts at the end of the lease does not move it --- .../rustic/backend/tests.rs | 19 ++++++++++++++++++- 1 file changed, 18 insertions(+), 1 deletion(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index d971620253..ef5c217038 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -21,7 +21,7 @@ use super::super::fault::{Operation, OperationCancelled, classify, is_config_exi use super::super::holding::{holding_storage, reached_deadline}; use super::super::publish::{SnapshotStage, StagedSnapshot}; use super::super::scripted::{Script, ScriptedBlobStorage}; -use super::{BlobBackend, file_size}; +use super::{BlobBackend, Lease, file_size}; use crate::filesystem_snapshot::SnapshotStoreError; use crate::services::golem_config::DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE as STORAGE_CALL_DEADLINE; use anyhow::anyhow; @@ -772,6 +772,23 @@ fn a_cancelled_backend_makes_no_storage_call() { assert_eq!((cancelled, storage.calls()), ([true; 6], Vec::new())); } +#[test] +fn a_marker_write_that_starts_at_the_end_of_the_lease_does_not_move_it_and_one_that_starts_before_does() + { + let end = Instant::now() + Duration::from_secs(60); + let span = Duration::from_secs(10); + let at_the_end = Lease::until(end); + let just_before = Lease::until(end); + + at_the_end.extend_from(end, span); + just_before.extend_from(end - Duration::from_nanos(1), span); + + assert_eq!( + (at_the_end.expiry(), just_before.expiry()), + (end, end - Duration::from_nanos(1) + span) + ); +} + #[test] fn a_cancel_ends_a_call_that_runs() { let runtime = Runtime::new().unwrap(); From ebdb4d39138a377e0af3a95df7254cc3fa9f2d35 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 03:11:16 -0700 Subject: [PATCH 110/126] Test that a record line parses only when it has 64 hex characters --- .../src/filesystem_snapshot/rustic/prune.rs | 41 +++++++++++++++++++ 1 file changed, 41 insertions(+) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index fdcbe0ea52..449f225d43 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -1472,6 +1472,47 @@ mod tests { ); } + #[test] + fn a_record_line_parses_only_when_it_has_64_characters_that_are_each_hex() { + // A record whose content does not parse names no snapshot file, so it counts as zero + // bytes and stays. + let hex = "0123456789abcdef".repeat(4); + let parsed = [ + parse_record("g".repeat(64).as_bytes()), + parse_record(b"abcdef0123"), + parse_record(hex.as_bytes()), + ]; + let records = parsed + .iter() + .enumerate() + .map(|(index, snapshots)| FreedRecord { + path: Path::new(FREED_PATH) + .join(format!("10-{index}")) + .into_boxed_path(), + bytes: 10, + snapshots: snapshots.clone(), + }) + .collect::>(); + + assert_eq!( + ( + parsed.clone(), + settle(&records, &std::collections::HashSet::new()), + ), + ( + [ + None, + None, + Some(Box::new([hex.clone().into_boxed_str()]) as Box<[Box]>) + ], + FreedRecords { + bytes: 10, + counted: Box::new([Path::new(FREED_PATH).join("10-2").into_boxed_path()]), + }, + ) + ); + } + #[test] #[timeout("60s")] async fn a_record_of_freed_bytes_is_written_and_listed() { From 470b831a657ceeabf64caa48359fe5ef00c7b208 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 03:14:57 -0700 Subject: [PATCH 111/126] Build the check for a lease that ran out only for tests --- golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs index 4aafb945e7..5f0ad956c6 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs @@ -76,7 +76,9 @@ impl Display for LeaseExpired { impl Error for LeaseExpired {} -/// Tells whether an error in the chain is [`LeaseExpired`]. +/// Tells whether an error in the chain is [`LeaseExpired`]. Only tests ask this, because the store +/// gives a lease that ran out as a retryable storage error, the same as a failed call. +#[cfg(test)] pub(super) fn is_lease_expired(error: &(dyn Error + 'static)) -> bool { chain(error).any(|error| error.is::()) } From d85b98800996b901dcd6f3b7f6fa884f6ade6eec Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 03:16:19 -0700 Subject: [PATCH 112/126] Test that a restore builds its pool with the restore reader threads --- .../filesystem_snapshot/rustic/store/tests.rs | 59 +++++++++++++++++++ 1 file changed, 59 insertions(+) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 98a67ead47..801a84675a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -5243,3 +5243,62 @@ async fn two_deletes_make_at_most_one_prune_in_random_orders_with_a_failed_call( println!("{SWEEP_CASES} random orders in {:?}", started.elapsed()); assert!(matches!(outcome, Ok(Ok(()))), "{outcome:?}"); } + +/// The name and the thread count of each pool that [`recording_pool`] built. +static BUILT_POOLS: std::sync::Mutex)>> = + std::sync::Mutex::new(Vec::new()); + +/// Builds a rayon pool with the name and the thread count, and records both. Only the store of +/// one test uses it, so the record holds only the pools of that store. +fn recording_pool( + name: &'static str, + threads: Option, +) -> Result { + BUILT_POOLS + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .push((name, threads)); + rayon::ThreadPoolBuilder::new() + .num_threads(threads.map_or(0, NonZeroUsize::get)) + .build() +} + +#[test] +#[timeout("60s")] +async fn a_restore_builds_its_pool_with_the_restore_reader_threads() { + // The save threads and the restore reader threads differ, so the count of the pool tells + // which setting the restore took. + let storage = Arc::new(InMemoryBlobStorage::new()); + let policy = StorePolicy { + save_threads: NonZeroUsize::new(2), + restore_reader_threads: NonZeroUsize::new(3).unwrap(), + ..policy(LONG_DEADLINE, NEVER, Duration::ZERO) + }; + let store = RusticSnapshotStore { + low_priority: super::super::priority::LowPriority { + build_pool: recording_pool, + ..super::super::priority::LowPriority::new(policy.save_threads) + }, + ..RusticSnapshotStore::with_policy(storage, key(), policy) + }; + let scope = new_scope(); + let tree = one_file_tree("restored"); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + + let restored = restored_listing(&store, &scope, &name("p-1")).await; + let restore_pools = BUILT_POOLS + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .iter() + .filter(|(name, _)| *name == "fs-snap-restore") + .map(|(_, threads)| *threads) + .collect::>(); + + assert_eq!( + (restored.ok(), restore_pools), + (Some(listing(tree.path())), vec![NonZeroUsize::new(3)]) + ); +} From 9fc809bcd50fe9489b2ad5cfc6c1d2e5b4556649 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 12:40:02 -0700 Subject: [PATCH 113/126] Keep the publish guard armed until the delete after a failed write ends --- .../src/filesystem_snapshot/rustic/publish.rs | 18 +++--- .../filesystem_snapshot/rustic/store/tests.rs | 61 +++++++++++++++++++ 2 files changed, 71 insertions(+), 8 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs index c2ea5c6e61..7c23d55d6e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs @@ -61,7 +61,9 @@ impl SnapshotStage { /// Writes the staged file only when its path has no blob, which makes the snapshot visible. The /// name is the hash of the content, so a blob at the path is this file. A failed write deletes the -/// path before the error returns, and a dropped write deletes it in a task of `tracker`. +/// path before the error returns, and a dropped write deletes it in a task of `tracker`. The guard +/// stays armed until that delete ends, so a publish that is dropped during the delete also deletes +/// the path in a task of `tracker`. pub(super) async fn publish( files: &SnapshotFiles, staged: &StagedSnapshot, @@ -76,14 +78,14 @@ pub(super) async fn publish( let written = files .put_if_absent("publish", &staged.path, &staged.content) .await; - retraction.armed = false; - match written { - Ok(_) => Ok(()), - Err(error) => { - retract_or_warn(files, &staged.path).await; - Err(error) - } + if let Err(error) = written { + // A write that lost its answer can have landed. + retract_or_warn(files, &staged.path).await; + retraction.armed = false; + return Err(error); } + retraction.armed = false; + Ok(()) } /// Deletes the snapshot file at the path. A path without a blob gives success. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 801a84675a..276d230ac2 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -3884,6 +3884,67 @@ async fn a_save_whose_index_file_a_prune_deleted_after_the_listing_gives_retryab ); } +#[test] +#[timeout("60s")] +async fn a_save_dropped_during_the_delete_after_a_failed_publish_still_deletes_the_snapshot_file() { + // The publish write lands and loses its answer, so the publish deletes the file. The gate + // holds that delete, and the test drops the save there. + let storage = + ScriptedBlobStorage::new( + Arc::new(InMemoryBlobStorage::new()), + |op_label, _| match op_label { + "publish" => Script::LoseTheAnswer, + "retract" => Script::WaitForGate, + _ => Script::Pass, + }, + ); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("dropped"); + let saving = tokio::spawn({ + let (store, scope, path) = (store.clone(), scope.clone(), tree.path().to_path_buf()); + async move { store.save(&scope, &name("p-dropped"), &path, None).await } + }); + let retracting = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "retract") + }) + .await; + let written = blobs(&*storage, &scope.0, "snapshots/").await.len(); + + saving.abort(); + let dropped = saving.await; + storage.open_gate(); + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + + assert!( + dropped.as_ref().is_err_and(|error| error.is_cancelled()), + "{dropped:?}" + ); + assert_eq!( + ( + retracting, + written, + stopped, + blobs(&*storage, &scope.0, "snapshots/").await, + RusticSnapshotStore::with_policy( + storage.clone(), + key(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ) + .stat(&scope, &name("p-dropped")) + .await + .ok(), + ), + (true, 1, true, Vec::::new(), Some(None)) + ); +} + #[test] #[timeout("60s")] async fn a_save_that_reaches_its_publish_after_shut_down_publishes_nothing() { From af609a23eee308c11f8f9d2a081b956c6093794e Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 12:45:42 -0700 Subject: [PATCH 114/126] Check the cancel of a save in the first poll of its publish --- .../src/filesystem_snapshot/rustic/store.rs | 30 ++++++++--- .../filesystem_snapshot/rustic/store/tests.rs | 51 +++++++++++++++++++ 2 files changed, 74 insertions(+), 7 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index a9540d2adb..59c1d31116 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -179,6 +179,10 @@ pub(crate) struct RusticSnapshotStore { /// Holds a save after its blocking work and before its publish, when a test sets it. #[cfg(test)] pub(super) publish_gate: Option>, + /// Holds a save after the tracker counts its publish and before the first poll of the publish, + /// when a test sets it. + #[cfg(test)] + pub(super) publish_poll_gate: Option>, /// Holds a delete after it chose its claim and before it builds its claim guard, when a test /// sets it. #[cfg(test)] @@ -472,6 +476,8 @@ impl RusticSnapshotStore { #[cfg(test)] publish_gate: None, #[cfg(test)] + publish_poll_gate: None, + #[cfg(test)] claim_gate: None, #[cfg(test)] path_check_gate: None, @@ -823,15 +829,25 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { } // The publish is the commit point, so no cancel ends it. The tracker counts it from here, // so a `shut_down` that has not cancelled yet waits for it, and the deadline limits that wait. + // The check of the cancel and the start of the write are in the first poll of the tracked + // future, so a publish never starts after the cancel. let files = self.files(scope, &CancellationToken::new()); - let publishing = self - .tracker - .track_future(publish(&files, &staged, &self.tracker)); - if self.root.is_cancelled() { - // No snapshot file is written. A later prune marks the packs of the save. - return Err(shut_down_error()); + let (root, tracker) = (self.root.clone(), self.tracker.clone()); + let publishing = self.tracker.track_future(async move { + if root.is_cancelled() { + // No snapshot file is written. A later prune marks the packs of the save. + return Err(shut_down_error()); + } + publish(&files, &staged, &tracker) + .await + .map_err(storage_failure) + }); + #[cfg(test)] + if let Some(gate) = &self.publish_poll_gate { + gate.reached.notify_one(); + gate.open.notified().await; } - publishing.await.map_err(storage_failure)?; + publishing.await?; Ok(info) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 276d230ac2..d58d60c3f1 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -3945,6 +3945,57 @@ async fn a_save_dropped_during_the_delete_after_a_failed_publish_still_deletes_t ); } +#[test] +#[timeout("60s")] +async fn a_save_whose_publish_is_counted_before_shut_down_and_polled_after_its_cancel_publishes_nothing() + { + // The gate holds the save after the tracker counts its publish and before the first poll of + // the publish. `shut_down` cancels and then waits for the publish, and the gate opens only + // after the cancel. + let storage = Arc::new(InMemoryBlobStorage::new()); + let gate = Arc::new(StepGate::default()); + let store = Arc::new(RusticSnapshotStore { + publish_poll_gate: Some(gate.clone()), + ..RusticSnapshotStore::with_policy( + storage.clone(), + key(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ) + }); + let scope = new_scope(); + let tree = one_file_tree("late"); + let saving = tokio::spawn({ + let (store, scope, path) = (store.clone(), scope.clone(), tree.path().to_path_buf()); + async move { store.save(&scope, &name("p-late"), &path, None).await } + }); + let reached = tokio::time::timeout(LIMIT, gate.reached.notified()) + .await + .is_ok(); + + let shutting_down = tokio::spawn({ + let store = store.clone(); + async move { store.shut_down().await } + }); + let cancelled = eventually(|| store.root.is_cancelled()).await; + gate.open.notify_one(); + let saved = tokio::time::timeout(LIMIT, saving).await; + let stopped = tokio::time::timeout(LIMIT, shutting_down).await; + + assert!( + matches!(&saved, Ok(Ok(Err(error))) if is_storage(error, false)), + "{saved:?}" + ); + assert!(matches!(stopped, Ok(Ok(()))), "{stopped:?}"); + assert_eq!( + ( + reached, + cancelled, + blobs(&*storage, &scope.0, "snapshots/").await + ), + (true, true, Vec::::new()) + ); +} + #[test] #[timeout("60s")] async fn a_save_that_reaches_its_publish_after_shut_down_publishes_nothing() { From c400d759c4df646c4da3dc992840c24c28fb7d99 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 12:47:08 -0700 Subject: [PATCH 115/126] Say what the save contract lets a store do with a parent, so that both stores meet it --- golem-worker-executor/src/filesystem_snapshot/mod.rs | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/mod.rs b/golem-worker-executor/src/filesystem_snapshot/mod.rs index a3b0ad53d5..aa2f8dc5ef 100644 --- a/golem-worker-executor/src/filesystem_snapshot/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/mod.rs @@ -220,10 +220,12 @@ pub(crate) trait FilesystemSnapshotStore: Send + Sync { /// snapshot. That time is later than the time of each snapshot that the scope held when the /// save started. /// - /// `parent` names the snapshot that the save compares with. With `SizeMtime`, a file whose - /// size and modification time equal those of the same path in the parent keeps the content of - /// the parent, and the save does not read it. With `Full`, the save reads every file. A parent - /// that the scope does not hold gives a save that reads every file. + /// `parent` names the snapshot that the save can compare with. With `SizeMtime`, the store can + /// keep the content of the parent for a file whose size and modification time equal those of + /// the same path in the parent, and then it does not read that file. So such a file that + /// changed can keep the content of the parent. A store can also read each file. With `Full`, + /// the save reads every file. A parent that the scope does not hold gives a save that reads + /// every file. async fn save( &self, scope: &SnapshotScope, From 7b261d2d44586c0ecc8255b350918cd6586d6690 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 12:49:24 -0700 Subject: [PATCH 116/126] Keep the directory and the number of a claim only in its guard, behind accessors --- .../src/filesystem_snapshot/rustic/store.rs | 51 +++++++++++++------ 1 file changed, 36 insertions(+), 15 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 59c1d31116..e1c8f7342b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -212,15 +212,13 @@ pub(super) struct StepGate { } /// The claim of a prune that a delete holds. -struct Claim<'a> { - directory: &'a Path, - number: u64, +struct Claim { /// The lease of the prune. Each marker write that succeeds moves its end. lease: Arc, /// The time that a marker write that succeeds adds to the lease. span: Duration, - /// Releases the claim when the delete stops before its prune starts, and writes the final - /// marker of a prune that started. + /// Holds the directory and the number of the claim. It releases the claim when the delete + /// stops before its prune starts, and writes the final marker of a prune that started. guard: ClaimGuard, } @@ -300,6 +298,31 @@ impl ClaimGuard { }) } + /// Gives the directory of the claims of the ledger of the claim. + fn directory(&self) -> &Path { + &self.directory + } + + /// Gives the number of the claim. + fn number(&self) -> u64 { + self.number + } + + /// Gives the markers of the claim that this delete wrote. + fn markers(&self) -> &Mutex>> { + &self.markers + } + + /// Gives the state of the claim, which the prune shares with the guard. + fn state(&self) -> Arc { + self.state.clone() + } + + /// Records that this delete wrote the claim, so a release deletes it. + fn mark_claimed(&self) { + self.claimed.store(true, Ordering::SeqCst); + } + /// Makes the release task, and moves the token of the tracker into it. fn spawn_release(&self) -> Option> { let files = self.files.clone(); @@ -666,10 +689,8 @@ impl RusticSnapshotStore { guard.disarm(); return Ok(()); }; - guard.claimed.store(true, Ordering::SeqCst); + guard.mark_claimed(); let claim = Claim { - directory: &claims, - number, lease: Arc::new(lease), span, guard, @@ -718,12 +739,12 @@ impl RusticSnapshotStore { scope: &SnapshotScope, token: &CancellationToken, files: &SnapshotFiles, - claim: &Claim<'_>, + claim: &Claim, ) -> Result>, SnapshotStoreError> { // A prune writes its ledger before it deletes the claims, so a delete that claims in a // directory that such a prune removed sees the new ledger here. let again = read_ledger(files).await.map_err(storage_failure)?; - if *claims_directory(&again) != *claim.directory { + if *claims_directory(&again) != *claim.guard.directory() { return Ok(None); } Ok(Some(Arc::new( @@ -737,13 +758,13 @@ impl RusticSnapshotStore { &self, backend: Arc, files: &SnapshotFiles, - claim: &Claim<'_>, + claim: &Claim, grace: Duration, ) -> Result, SnapshotStoreError> { let key = self.key.clone(); let settings = self.policy.prune; let low_priority = self.low_priority; - let state = claim.guard.state.clone(); + let state = claim.guard.state(); // The plan of a prune reads each snapshot file before the prune changes the repository. // A forget of another delete can remove a listed file before its read, so the prune // plans again from a new listing. @@ -769,10 +790,10 @@ impl RusticSnapshotStore { // cancelled. The tracker counts the whole step, so no timer of it runs after a shut down. let refreshing = keep_claim_fresh( files, - claim.directory, - claim.number, + claim.guard.directory(), + claim.guard.number(), refresh_period(grace, self.policy.deadline), - &claim.guard.markers, + claim.guard.markers(), &claim.lease, claim.span, ); From 130f62301754699422347620d8e852b60e1f07e4 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 12:50:22 -0700 Subject: [PATCH 117/126] Share one check for the forget of a delete between the store tests and the sweep --- .../src/filesystem_snapshot/rustic/store/tests/sweep.rs | 4 ---- 1 file changed, 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs index 513aaf3452..9e0753de34 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs @@ -76,10 +76,6 @@ fn is_refresh(op_label: &str) -> bool { /// quarter of it, so the markers come while the prune runs. const SWEEP_GRACE: Duration = Duration::from_millis(16); -fn is_forget(op_label: &str, path: &Path) -> bool { - op_label == "delete" && path.starts_with("snapshots") -} - fn is_prune_start(op_label: &str, path: &str) -> bool { op_label == "list" && path == "data" } From 39f77f4e59bcc61eb98be603949c98c8e564010e Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 13:42:25 -0700 Subject: [PATCH 118/126] Test that a prune refreshes its claim with its own number --- .../filesystem_snapshot/rustic/store/tests.rs | 95 +++++++++++++++++++ 1 file changed, 95 insertions(+) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index d58d60c3f1..80762e5c94 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1276,6 +1276,101 @@ async fn age_claims(storage: &Arc, scope: &Snapshot .await; } +#[test] +#[timeout("60s")] +async fn a_prune_refreshes_the_claim_with_its_own_number() { + // Old claims 0 and 1 with markers older than the hold stay in the claim directory, so the + // delete takes claim 2. The gate holds the prune at its listing of the packs while the claim + // gets new markers. + let grace = Duration::from_millis(400); + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let old = golem_common::model::Timestamp::now_utc() + .to_millis() + .saturating_sub(2 * 3_600_000); + futures::stream::iter([ + "golem/prune-claims/none/0".to_string(), + format!("golem/prune-claims/none/0@{old}-old"), + "golem/prune-claims/none/1".to_string(), + format!("golem/prune-claims/none/1@{old}-old"), + ]) + .for_each(|path| { + let (storage, scope) = (storage.clone(), scope.clone()); + async move { + storage + .put_raw("test", "test", scope.0.clone(), Path::new(&path), &[]) + .await + .unwrap(); + } + }) + .await; + let pruning = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let refreshes = || { + storage + .calls() + .iter() + .filter(|(op_label, _)| *op_label == "refresh_claim") + .count() + }; + let refreshed = eventually(|| held.load(Ordering::SeqCst) && refreshes() >= 2).await; + + let entries = claim_entries(&storage, &scope).await; + storage.open_gate(); + let pruned = tokio::time::timeout(LIMIT, pruning).await; + let young = entries + .iter() + .filter_map(|entry| match entry { + ClaimEntry::Marker(number, at) if at.to_millis() > old => Some(*number), + _ => None, + }) + .collect::>(); + let live_markers = entries + .iter() + .filter(|entry| matches!(entry, ClaimEntry::Marker(2, _))) + .cloned() + .collect::>(); + let hold = claim_hold(grace, LONG_DEADLINE); + + assert!(matches!(pruned, Ok(Ok(Ok(())))), "{pruned:?}"); + assert_eq!( + ( + refreshed, + entries.contains(&ClaimEntry::Claim(2)), + young.len() >= 3, + young.iter().all(|number| *number == 2), + next_claim( + &live_markers, + golem_common::model::Timestamp::now_utc(), + hold + ), + ), + (true, true, true, true, ClaimChoice::Held) + ); +} + #[test] #[timeout("60s")] async fn a_prune_slower_than_the_grace_period_keeps_its_claim_fresh() { From 1154ab57b7a151535a6f127d414c7f48316d27c6 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 13:42:43 -0700 Subject: [PATCH 119/126] Test that a prune whose attempts each find a snapshot file gone after refreshes deletes each marker of its claim --- .../filesystem_snapshot/rustic/store/tests.rs | 52 +++++++++++++++++++ 1 file changed, 52 insertions(+) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 80762e5c94..bd9daa4651 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1371,6 +1371,58 @@ async fn a_prune_refreshes_the_claim_with_its_own_number() { ); } +#[test] +#[timeout("60s")] +async fn a_prune_that_finds_a_snapshot_file_gone_at_each_attempt_after_refreshes_deletes_each_marker_of_its_claim() + { + // Each listing of the snapshot files by the prune takes 250 ms, and the grace period gives a + // new marker each 100 ms, so the claim gets new markers before the release. Each read of a + // snapshot file after the claim finds it gone. + let claimed = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let claimed = claimed.clone(); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + let after_claim = claimed.load(Ordering::SeqCst); + if after_claim && op_label == "list" && path == Path::new("snapshots") { + Script::Delay(Duration::from_millis(250)) + } else if after_claim && path.starts_with("snapshots") && !is_forget(op_label, path) { + Script::Vanish + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_millis(400)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let refreshes = storage + .calls() + .iter() + .filter(|(op_label, _)| *op_label == "refresh_claim") + .count(); + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert_eq!( + ( + refreshes > 0, + prunes(&storage.calls()), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, 0, Vec::::new()) + ); +} + #[test] #[timeout("60s")] async fn a_prune_slower_than_the_grace_period_keeps_its_claim_fresh() { From c212acb5368f5817388d06d9646eb5cbb0e31728 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 14:03:46 -0700 Subject: [PATCH 120/126] Say in the marker docs that a write which starts after the end of the lease does not move it --- .../src/filesystem_snapshot/rustic/prune.rs | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 449f225d43..1c6010f45c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -539,8 +539,9 @@ pub(super) async fn write_marker( Ok(path) } -/// Writes a marker of the claim with the number, and moves the end of the lease to `span` after -/// the start of the write when the write succeeds. +/// Writes a marker of the claim with the number. A write that succeeds and started before the end +/// of the lease moves the end to `span` after the start of the write, when that is later. A write +/// that started at or after the end does not move it. async fn write_leased_marker( files: &SnapshotFiles, op_label: &'static str, @@ -600,8 +601,9 @@ pub(super) fn refresh_period(grace: Duration, deadline: Duration) -> Duration { /// Writes a new marker of the claim with the number at each period, until the caller drops the /// future or the operation of the files is cancelled, and adds the path of each written marker to -/// `written`. Each write that succeeds moves the end of the lease to `span` after its start. A -/// failed write gives a warning, and the next period tries again. +/// `written`. A write that succeeds and started before the end of the lease moves the end to +/// `span` after its start, when that is later. A write that started at or after the end does not +/// move it. A failed write gives a warning, and the next period tries again. pub(super) async fn keep_claim_fresh( files: &SnapshotFiles, directory: &Path, From 4add917ab7fc2f5e1feba12c94dc1b670b6ece66 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 14:57:01 -0700 Subject: [PATCH 121/126] Read the records of freed bytes only when a prune can still be due --- .../src/filesystem_snapshot/rustic/prune.rs | 115 ++++++++++++++---- .../src/filesystem_snapshot/rustic/store.rs | 38 ++++-- .../filesystem_snapshot/rustic/store/tests.rs | 97 +++++++++++++++ 3 files changed, 217 insertions(+), 33 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 1c6010f45c..93fee1771e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -74,7 +74,12 @@ impl Percent { /// Tells whether the hold of a claim passed at `now` since the last prune. The time of the last /// prune is after its final marker, so the next prune waits a full hold from the newest claim /// marker that the last prune wrote. A time more than the margin after `now` counts as missing. -fn hold_passed(ledger: &PruneLedger, now: Timestamp, grace: Duration, deadline: Duration) -> bool { +pub(super) fn hold_passed( + ledger: &PruneLedger, + now: Timestamp, + grace: Duration, + deadline: Duration, +) -> bool { ledger .last_prune .filter(|last| !beyond_margin(*last, now)) @@ -120,6 +125,18 @@ pub(super) fn prune_due( hold_passed(ledger, now, grace, deadline) && work } +/// Tells whether the freed bytes in the names of the records, which are at least the settled +/// freed bytes, can make a prune due with the size of the repository. When this gives false, +/// [`prune_due`] gives false for the settled bytes too, so the records need no read. +pub(super) fn may_be_due( + ledger: &PruneLedger, + named_bytes: u64, + repository_bytes: u64, + threshold: Percent, +) -> bool { + named_bytes >= threshold.of(repository_bytes).max(1) || ledger.awaiting_removal +} + /// Gives the size of the repository of the scope: the sum of the sizes of its packs. pub(super) async fn repository_bytes(files: &SnapshotFiles) -> anyhow::Result { Ok(files @@ -324,29 +341,54 @@ pub(super) async fn record_freed( .map(|_| ()) } -/// Lists and reads the records of freed bytes, lists the snapshot files one time, and gives the -/// settled records. A name that does not parse counts as zero bytes and stays, and a record that -/// a prune deleted after the listing is left out. Without records, no snapshot file is listed. -pub(super) async fn list_freed(files: &SnapshotFiles) -> anyhow::Result { - let listed = files +/// A record of freed bytes that a listing found: its path and the bytes in its name. +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct ListedFreed { + pub(super) path: Box, + pub(super) bytes: u64, +} + +/// Lists the names of the records of freed bytes, and reads no content. A name that does not +/// parse is left out, so it counts as zero bytes and stays. +pub(super) async fn list_freed_names(files: &SnapshotFiles) -> anyhow::Result> { + Ok(files .list_below("list_freed", Path::new(FREED_PATH)) - .await?; - let named = listed + .await? .iter() .filter_map(|blob| { let bytes = parse_freed(blob.path.file_name()?.to_str()?)?; - Some((bytes, &blob.path)) + Some(ListedFreed { + path: blob.path.clone(), + bytes, + }) }) - .collect::>(); - if named.is_empty() { + .collect()) +} + +/// Gives the sum of the bytes in the names of the listed records. +pub(super) fn named_bytes(listed: &[ListedFreed]) -> u64 { + listed + .iter() + .map(|record| record.bytes) + .fold(0, u64::saturating_add) +} + +/// Reads the listed records of freed bytes, lists the snapshot files one time, and gives the +/// settled records. A record that a prune deleted after the listing is left out. Without records, +/// no snapshot file is listed. +pub(super) async fn settle_freed( + files: &SnapshotFiles, + listed: &[ListedFreed], +) -> anyhow::Result { + if listed.is_empty() { return Ok(FreedRecords::default()); } - let records = stream::iter(named.iter()) - .then(|(bytes, path)| async move { - let content = files.get("read_freed", path).await?; + let records = stream::iter(listed) + .then(|listed| async move { + let content = files.get("read_freed", &listed.path).await?; Ok::<_, anyhow::Error>(content.map(|content| FreedRecord { - path: (*path).clone(), - bytes: *bytes, + path: listed.path.clone(), + bytes: listed.bytes, snapshots: parse_record(&content), })) }) @@ -746,10 +788,11 @@ mod tests { use super::{ CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, ClaimEntry, FREED_PATH, FreedRecord, FreedRecords, LEDGERS_PATH, Lease, Percent, PruneLedger, claim_hold, claims_directory, - keep_claim_fresh, lease_span, list_claims, list_freed, marker_path, marker_time, - needs_repository_size, newest_ledger, next_claim, old_claim_directories, older_entries, - parse_claim_entry, parse_freed, parse_ledger_entry, parse_record, prune_due, read_ledger, - record_content, record_freed, refresh_period, settle, take_claim, write_ledger, + keep_claim_fresh, lease_span, list_claims, list_freed_names, marker_path, marker_time, + may_be_due, named_bytes, needs_repository_size, newest_ledger, next_claim, + old_claim_directories, older_entries, parse_claim_entry, parse_freed, parse_ledger_entry, + parse_record, prune_due, read_ledger, record_content, record_freed, refresh_period, settle, + settle_freed, take_claim, write_ledger, }; use futures::StreamExt; use golem_common::model::Timestamp; @@ -805,6 +848,30 @@ mod tests { } } + #[test] + fn the_bytes_in_the_names_of_the_records_can_make_a_prune_due_only_when_they_reach_the_threshold() + { + let may = |awaiting_removal, named, repository_bytes| { + may_be_due( + &ledger(None, awaiting_removal), + named, + repository_bytes, + TEN_PERCENT, + ) + }; + + assert_eq!( + [ + may(false, 99, 1000), + may(false, 100, 1000), + may(false, 0, 0), + may(false, 1, 0), + may(true, 0, 1000), + ], + [false, true, false, true, true] + ); + } + #[test] fn a_prune_is_due_when_the_freed_bytes_reach_ten_percent_of_the_repository_rounded_down() { let now = at(10_000_000); @@ -1523,9 +1590,13 @@ mod tests { let gone = ["0".repeat(64).into_boxed_str()]; record_freed(&files, 40, &gone).await.unwrap(); record_freed(&files, 2, &gone).await.unwrap(); - let listed = list_freed(&files).await.unwrap(); + let listed = list_freed_names(&files).await.unwrap(); + let settled = settle_freed(&files, &listed).await.unwrap(); - assert_eq!((listed.bytes, listed.counted.len()), (42, 2)); + assert_eq!( + (named_bytes(&listed), settled.bytes, settled.counted.len()), + (42, 42, 2) + ); } #[test] diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index e1c8f7342b..bd86467436 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -27,10 +27,11 @@ use super::fault::{ use super::files::SnapshotFiles; use super::priority::LowPriority; use super::prune::{ - ClaimChoice, Percent, claim_hold, claims_directory, keep_claim_fresh, lease_span, list_claims, - list_freed, marker_path, marker_time, needs_repository_size, next_claim, prune_due, - read_ledger, record_freed, refresh_period, release_claim, remove_freed, remove_old_claims, - remove_older_ledgers, repository_bytes, take_claim, write_ledger, write_marker, + ClaimChoice, Percent, claim_hold, claims_directory, hold_passed, keep_claim_fresh, lease_span, + list_claims, list_freed_names, marker_path, marker_time, may_be_due, named_bytes, + needs_repository_size, next_claim, prune_due, read_ledger, record_freed, refresh_period, + release_claim, remove_freed, remove_old_claims, remove_older_ledgers, repository_bytes, + settle_freed, take_claim, write_ledger, write_marker, }; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -631,25 +632,40 @@ impl RusticSnapshotStore { token: &CancellationToken, ) -> Result<(), SnapshotStoreError> { let files = self.files(scope, token); + let grace = self.policy.prune.keep_delete; + let deadline = self.policy.deadline; + let threshold = self.policy.prune_threshold; let ledger = read_ledger(&files).await.map_err(storage_failure)?; - let records = list_freed(&files).await.map_err(storage_failure)?; // Each comparison with a time from storage uses a clock reading from after the listing // that gave that time. A listing can take up to one storage call deadline, and a stale // reading can put a marker that another host wrote within the margin beyond the margin. - let now = self.now(); - let grace = self.policy.prune.keep_delete; - let deadline = self.policy.deadline; - let size = if needs_repository_size(&ledger, records.bytes, now, grace, deadline) { + // Each step below runs only when a prune can still be due, so a delete within the hold + // reads no record, and a delete whose records are too small reads no record content. + if !hold_passed(&ledger, self.now(), grace, deadline) { + return Ok(()); + } + let listed = list_freed_names(&files).await.map_err(storage_failure)?; + let upper = named_bytes(&listed); + if upper == 0 && !ledger.awaiting_removal { + return Ok(()); + } + let size = if needs_repository_size(&ledger, upper, self.now(), grace, deadline) { repository_bytes(&files).await.map_err(storage_failure)? } else { 0 }; + if !may_be_due(&ledger, upper, size, threshold) { + return Ok(()); + } + let records = settle_freed(&files, &listed) + .await + .map_err(storage_failure)?; if !prune_due( &ledger, records.bytes, - now, + self.now(), size, - self.policy.prune_threshold, + threshold, grace, deadline, ) { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index bd9daa4651..a57122c3e2 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1807,6 +1807,103 @@ async fn the_lease_of_a_prune_starts_at_its_first_marker_so_a_prune_without_a_re ); } +/// The operation labels of the calls of the prune decision that read the records of freed bytes +/// and what they name: the listing of the records, the read of a record, the listing of the +/// snapshot files, and the listing of the packs. +const DECISION_READS: [&str; 4] = ["list_freed", "read_freed", "list_snapshots", "list_data"]; + +/// Gives the labels of [`DECISION_READS`] that the calls from `from` on made. +fn decision_reads(storage: &ScriptedBlobStorage, from: usize) -> Vec<&'static str> { + let calls = storage.calls(); + DECISION_READS + .into_iter() + .filter(|label| { + calls + .iter() + .skip(from) + .any(|(op_label, _)| op_label == label) + }) + .collect() +} + +#[test] +#[timeout("60s")] +async fn a_delete_within_the_hold_reads_no_record_of_freed_bytes_and_lists_no_pack() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2", "p-3"]).await; + store.delete(&scope, &name("p-1")).await.unwrap(); + let pruned = ledger(&storage, &scope).await.last_prune.is_some(); + let from = storage.calls().len(); + + let deleted = store.delete(&scope, &name("p-2")).await; + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + (pruned, decision_reads(&storage, from)), + (true, Vec::<&str>::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_whose_named_bytes_are_below_the_threshold_reads_no_record_content() { + // A threshold of all the bytes of the repository is above the bytes that one delete frees. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, Percent(100), Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let from = storage.calls().len(); + + let deleted = store.delete(&scope, &name("p-1")).await; + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + decision_reads(&storage, from), + ledger(&storage, &scope).await.last_prune.is_some(), + freed(&storage, &scope).await > 0, + ), + (vec!["list_freed", "list_data"], false, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_whose_named_bytes_reach_the_threshold_reads_and_settles_the_records_and_prunes() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let from = storage.calls().len(); + + let deleted = store.delete(&scope, &name("p-1")).await; + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + decision_reads(&storage, from), + prunes(&storage.calls()), + ledger(&storage, &scope).await.last_prune.is_some(), + freed(&storage, &scope).await, + ), + (DECISION_READS.to_vec(), 1, true, 0) + ); +} + #[test] #[timeout("60s")] async fn a_prune_deletes_the_claims_of_old_ledgers() { From 93e43d15d2c5567f39ca964899162a7b7b6b6611 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 15:00:46 -0700 Subject: [PATCH 122/126] Read a range of a tree pack that is not kept once the kept packs fill their limit --- .../src/filesystem_snapshot/rustic/backend.rs | 10 ++-- .../rustic/backend/kept.rs | 21 +++++--- .../rustic/backend/tests.rs | 54 +++++++++++++++---- 3 files changed, 65 insertions(+), 20 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index bc1dee7dde..bb7703f16d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -368,9 +368,13 @@ impl ReadBackend for BlobBackend { return Ok(Bytes::new()); }; // rustic marks the reads of tree blobs as cacheable, and reads each tree blob on its own. - if cacheable && tpe == FileType::Pack { - let pack = self.kept.get_or_read(id, || self.read_full(tpe, id))?; - return range_of(&pack, &path, offset, last); + // A pack of tree blobs is read whole and kept, until the kept packs fill their limit. + // After that, a pack that is not kept is read by its range, as each other blob. + if cacheable + && tpe == FileType::Pack + && let Some(pack) = self.kept.get_or_read(id, || self.read_full(tpe, id)) + { + return range_of(&pack?, &path, offset, last); } let start = u64::from(offset); self.request( diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs index f5c0d36a8d..92a7a2cb2a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs @@ -15,7 +15,9 @@ //! The packs that one backend keeps in memory after a full read, up to a limit of bytes. //! //! rustic reads each tree blob with its own ranged read. A backend lives for one operation, so it -//! keeps each pack of tree blobs after its first read and gives the later ranges from memory. +//! keeps each pack of tree blobs after its first read and gives the later ranges from memory. Once +//! the kept packs fill the limit, a pack that is not kept is not read whole, because it cannot be +//! kept, and the caller reads only its range. use bytes::Bytes; use rustic_core::{Id, RusticResult}; @@ -23,7 +25,9 @@ use std::collections::{HashMap, HashSet}; use std::fmt::{Debug, Formatter}; use std::sync::{Condvar, Mutex, MutexGuard, PoisonError}; -/// The packs that one backend keeps, and the packs that a thread reads now. +/// The packs that one backend keeps, and the packs that a thread reads now. When the kept bytes +/// reach the limit, the set keeps no more packs, and a pack that it does not keep is not read +/// whole. pub(super) struct KeptPacks { limit: usize, state: Mutex, @@ -53,12 +57,14 @@ impl KeptPacks { /// Gives the kept pack, or reads it with `read`. While one thread reads a pack, the other /// threads that want it wait for that read, and then take the kept pack or read it again. A - /// pack is kept only when its read succeeds and it fits in the limit. + /// pack is kept only when its read succeeds and it fits in the limit. When the kept packs fill + /// the limit and the pack is not kept, it gives `None` and does not read, so the caller reads + /// only its range. pub(super) fn get_or_read( &self, id: &Id, read: impl FnOnce() -> RusticResult, - ) -> RusticResult { + ) -> Option> { #[cfg(test)] let mut counted = false; let mut state = self @@ -78,7 +84,10 @@ impl KeptPacks { state.waiters -= 1; } if let Some(pack) = state.packs.get(id) { - return Ok(pack.clone()); + return Some(Ok(pack.clone())); + } + if state.bytes >= self.limit { + return None; } state.reading.insert(*id); drop(state); @@ -90,7 +99,7 @@ impl KeptPacks { if let Ok(pack) = &read { reading.keep(pack); } - read + Some(read) } /// Gives the number of threads that wait for the read of a pack by another thread. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index ef5c217038..05c4333e36 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -490,8 +490,9 @@ fn pack_content() -> Vec { (0..100).collect() } -/// A backend over a storage that holds one pack at the path of the id `ab`, with the rule of -/// the storage and the limit of the kept packs. The storage records each call. +/// A backend over a storage that holds one pack at the path of the id `ab` and one at the path of +/// the id `cd`, with the rule of the storage and the limit of the kept packs. The storage records +/// each call. struct PackFixture { _runtime: Runtime, storage: Arc, @@ -503,15 +504,17 @@ impl PackFixture { let runtime = Runtime::new().unwrap(); let inner = Arc::new(InMemoryBlobStorage::new()); let namespace = new_namespace(); - runtime - .block_on(inner.put_raw( - "test", - "test", - namespace.clone(), - Path::new(&format!("data/ab/{}", "ab".repeat(32))), - &pack_content(), - )) - .unwrap(); + ["ab", "cd"].iter().for_each(|pack| { + runtime + .block_on(inner.put_raw( + "test", + "test", + namespace.clone(), + Path::new(&format!("data/{pack}/{}", pack.repeat(32))), + &pack_content(), + )) + .unwrap(); + }); let storage = ScriptedBlobStorage::new(inner, rule); let backend = BlobBackend::new( storage.clone(), @@ -665,6 +668,35 @@ fn a_pack_over_the_limit_is_read_again_at_its_next_range() { ); } +#[test] +fn once_the_kept_packs_fill_the_limit_a_range_of_another_pack_is_a_ranged_read() { + // The first pack fills the limit when it is kept. + let fixture = PackFixture::new(100, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + ( + tree_range(&backend, 0, 10).ok(), + backend + .read_partial(FileType::Pack, &id("cd"), true, 20, 10) + .ok(), + tree_range(&backend, 40, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)), + Some(Bytes::from_iter(40..50)) + )), + vec!["read", "read_range"] + ) + ); +} + #[test] fn a_failed_read_of_a_pack_is_not_kept_and_the_next_range_reads_again() { let refused = Arc::new(std::sync::atomic::AtomicBool::new(false)); From c437311533c80c8f0f1d069eb567e153af469bd2 Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 16:25:10 -0700 Subject: [PATCH 123/126] Say in the fixed drop tests why they repeat the order, not how the order was found --- .../src/filesystem_snapshot/rustic/store/tests.rs | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index a57122c3e2..595a746e14 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -5508,8 +5508,8 @@ async fn two_deletes_make_at_most_one_prune_in_each_order_with_up_to_two_switche #[timeout("60s")] async fn two_deletes_make_at_most_one_prune_when_one_is_dropped_at_its_first_step() { // A drop right after the first step cancels the forget of the delete while its storage call - // can already wait for a step. The order ran into that race in a random case, so the test - // runs it many times. + // can already wait for a step. The race depends on timing, so the test runs the order many + // times. let (shared, prepared) = prepared_scope().await; let schedule = sweep::Schedule { first: 0, @@ -5532,8 +5532,7 @@ async fn two_deletes_make_at_most_one_prune_when_one_is_dropped_at_its_first_ste async fn two_deletes_make_at_most_one_prune_when_one_is_dropped_as_its_prune_starts() { // A drop right after the second read of the ledger can come just after the prune started. The // listing of the packs then waits for a step until the cancel ends it, and the guard writes - // the final marker. The order ran into that race in a random case, so the test runs it many - // times. + // the final marker. The race depends on timing, so the test runs the order many times. let (shared, prepared) = prepared_scope().await; let schedule = sweep::Schedule { first: 1, From d5db89774c4bf698a1b21d6b2ed4ed528918fd5f Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 16:33:29 -0700 Subject: [PATCH 124/126] Close the kept packs when a pack that was read whole does not fit --- .../src/filesystem_snapshot/rustic/backend.rs | 5 ++- .../rustic/backend/kept.rs | 23 +++++++---- .../rustic/backend/tests.rs | 41 ++++++++++++++++++- 3 files changed, 56 insertions(+), 13 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index bb7703f16d..3035e47f27 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -368,8 +368,9 @@ impl ReadBackend for BlobBackend { return Ok(Bytes::new()); }; // rustic marks the reads of tree blobs as cacheable, and reads each tree blob on its own. - // A pack of tree blobs is read whole and kept, until the kept packs fill their limit. - // After that, a pack that is not kept is read by its range, as each other blob. + // A pack of tree blobs is read whole and kept, until the kept packs fill their limit or a + // pack that was read whole does not fit. After that, a pack that is not kept is read by its + // range, as each other blob. if cacheable && tpe == FileType::Pack && let Some(pack) = self.kept.get_or_read(id, || self.read_full(tpe, id)) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs index 92a7a2cb2a..e3d5bdb638 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs @@ -16,8 +16,9 @@ //! //! rustic reads each tree blob with its own ranged read. A backend lives for one operation, so it //! keeps each pack of tree blobs after its first read and gives the later ranges from memory. Once -//! the kept packs fill the limit, a pack that is not kept is not read whole, because it cannot be -//! kept, and the caller reads only its range. +//! the kept packs fill the limit, or a pack that was read whole did not fit, the set is closed: a +//! pack that is not kept is not read whole, because it cannot be kept, and the caller reads only +//! its range. use bytes::Bytes; use rustic_core::{Id, RusticResult}; @@ -26,8 +27,8 @@ use std::fmt::{Debug, Formatter}; use std::sync::{Condvar, Mutex, MutexGuard, PoisonError}; /// The packs that one backend keeps, and the packs that a thread reads now. When the kept bytes -/// reach the limit, the set keeps no more packs, and a pack that it does not keep is not read -/// whole. +/// reach the limit, or a pack that was read whole does not fit, the set is closed. A closed set +/// keeps no more packs, and a pack that it does not keep is not read whole. pub(super) struct KeptPacks { limit: usize, state: Mutex, @@ -39,6 +40,8 @@ pub(super) struct KeptPacks { struct State { packs: HashMap, bytes: usize, + /// A pack that was read whole did not fit in the limit. + closed: bool, reading: HashSet, /// The threads that wait for the read of a pack by another thread. #[cfg(test)] @@ -57,9 +60,9 @@ impl KeptPacks { /// Gives the kept pack, or reads it with `read`. While one thread reads a pack, the other /// threads that want it wait for that read, and then take the kept pack or read it again. A - /// pack is kept only when its read succeeds and it fits in the limit. When the kept packs fill - /// the limit and the pack is not kept, it gives `None` and does not read, so the caller reads - /// only its range. + /// pack is kept only when its read succeeds and it fits in the limit. When the set is closed and + /// the pack is not kept, it gives `None` and does not read, so the caller reads only its range. + /// A failed read keeps nothing and does not close the set. pub(super) fn get_or_read( &self, id: &Id, @@ -86,7 +89,7 @@ impl KeptPacks { if let Some(pack) = state.packs.get(id) { return Some(Ok(pack.clone())); } - if state.bytes >= self.limit { + if state.closed || state.bytes >= self.limit { return None; } state.reading.insert(*id); @@ -133,13 +136,15 @@ struct Reading<'a> { } impl Reading<'_> { - /// Keeps the pack when it fits in the limit. + /// Keeps the pack when it fits in the limit, and closes the set when it does not. fn keep(&self, pack: &Bytes) { let mut state = self.kept.state(); let bytes = state.bytes.saturating_add(pack.len()); if bytes <= self.kept.limit { state.bytes = bytes; state.packs.insert(self.id, pack.clone()); + } else { + state.closed = true; } } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index 05c4333e36..b8d48b123d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -645,7 +645,7 @@ fn two_threads_that_miss_one_pack_make_one_storage_read() { } #[test] -fn a_pack_over_the_limit_is_read_again_at_its_next_range() { +fn a_pack_over_the_limit_is_read_whole_one_time_and_its_next_range_is_a_ranged_read() { let fixture = PackFixture::new(99, |_, _| Script::Pass); let backend = fixture.backend.clone(); @@ -663,7 +663,44 @@ fn a_pack_over_the_limit_is_read_again_at_its_next_range() { Some(Bytes::from_iter(0..10)), Some(Bytes::from_iter(20..30)) )), - vec!["read", "read"] + vec!["read", "read_range"] + ) + ); +} + +#[test] +fn a_pack_that_does_not_fit_below_the_limit_is_read_whole_one_time_and_then_by_its_ranges() { + // The first pack is kept, and its 100 bytes stay below the limit of 150. The second pack does + // not fit, so it is read whole one time and not kept. + let fixture = PackFixture::new(150, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + let second = |offset| { + backend + .read_partial(FileType::Pack, &id("cd"), true, offset, 10) + .ok() + }; + ( + tree_range(&backend, 0, 10).ok(), + second(20), + second(40), + second(60), + tree_range(&backend, 80, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)), + Some(Bytes::from_iter(40..50)), + Some(Bytes::from_iter(60..70)), + Some(Bytes::from_iter(80..90)) + )), + vec!["read", "read", "read_range", "read_range"] ) ); } From afdfb70a35403429a8804442592afc56ec627a3e Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 17:22:19 -0700 Subject: [PATCH 125/126] Use a ledger entry below golem/prune-ledgers in the scope and cancelled-call tests --- .../src/filesystem_snapshot/rustic/scope/tests.rs | 2 +- .../src/filesystem_snapshot/rustic/store/tests.rs | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs index 1a4ba0191f..e40d33b055 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs @@ -31,7 +31,7 @@ const DEADLINE: Duration = Duration::from_secs(2); const REPOSITORY: [(&str, &str); 6] = [ ("config", "config"), ("data/ab/abab", "pack"), - ("golem/prune-ledger", "ledger"), + ("golem/prune-ledgers/1000-0-0f0f", "ledger"), ("index/cdcd", "index"), ("keys/efef", "key"), ("snapshots/0101", "snapshot"), diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 595a746e14..f97c5ca5c1 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -3548,7 +3548,7 @@ async fn a_blob_call_of_a_cancelled_operation_does_not_start() { }; let read = files - .get("read_ledger", Path::new("golem/prune-ledger")) + .get("read_ledger", Path::new("golem/prune-ledgers/1000-0-0f0f")) .await; assert!(read.is_err(), "{read:?}"); From 444af8e895e134c05504e8cc9cb2115ebd464a8d Mon Sep 17 00:00:00 2001 From: kmatasfp <33095685+kmatasfp@users.noreply.github.com> Date: Sun, 27 Sep 2026 17:51:08 -0700 Subject: [PATCH 126/126] Give each rustic module with tests or submodules a folder, and keep the test storages below the rustic tests --- .../{memory.rs => memory/mod.rs} | 0 .../rustic/{backend.rs => backend/mod.rs} | 0 .../rustic/backend/tests.rs | 4 ++-- .../src/filesystem_snapshot/rustic/mod.rs | 4 ---- .../rustic/{priority.rs => priority/mod.rs} | 0 .../src/filesystem_snapshot/rustic/prune.rs | 2 +- .../rustic/{publish.rs => publish/mod.rs} | 0 .../rustic/publish/tests.rs | 4 ++-- .../rustic/{scope.rs => scope/mod.rs} | 0 .../filesystem_snapshot/rustic/scope/tests.rs | 2 +- .../rustic/{store.rs => store/mod.rs} | 0 .../filesystem_snapshot/rustic/store/tests.rs | 2 +- .../rustic/{ => tests}/holding.rs | 8 +++---- .../rustic/{tests.rs => tests/mod.rs} | 7 ++++-- .../rustic/{ => tests}/scripted.rs | 24 +++++++++---------- 15 files changed, 28 insertions(+), 29 deletions(-) rename golem-worker-executor/src/filesystem_snapshot/{memory.rs => memory/mod.rs} (100%) rename golem-worker-executor/src/filesystem_snapshot/rustic/{backend.rs => backend/mod.rs} (100%) rename golem-worker-executor/src/filesystem_snapshot/rustic/{priority.rs => priority/mod.rs} (100%) rename golem-worker-executor/src/filesystem_snapshot/rustic/{publish.rs => publish/mod.rs} (100%) rename golem-worker-executor/src/filesystem_snapshot/rustic/{scope.rs => scope/mod.rs} (100%) rename golem-worker-executor/src/filesystem_snapshot/rustic/{store.rs => store/mod.rs} (100%) rename golem-worker-executor/src/filesystem_snapshot/rustic/{ => tests}/holding.rs (98%) rename golem-worker-executor/src/filesystem_snapshot/rustic/{tests.rs => tests/mod.rs} (99%) rename golem-worker-executor/src/filesystem_snapshot/rustic/{ => tests}/scripted.rs (97%) diff --git a/golem-worker-executor/src/filesystem_snapshot/memory.rs b/golem-worker-executor/src/filesystem_snapshot/memory/mod.rs similarity index 100% rename from golem-worker-executor/src/filesystem_snapshot/memory.rs rename to golem-worker-executor/src/filesystem_snapshot/memory/mod.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/mod.rs similarity index 100% rename from golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs rename to golem-worker-executor/src/filesystem_snapshot/rustic/backend/mod.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index b8d48b123d..8374bb269a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -18,9 +18,9 @@ //! runtime that the backend holds, as the threads of rustic are not. use super::super::fault::{Operation, OperationCancelled, classify, is_config_exists}; -use super::super::holding::{holding_storage, reached_deadline}; use super::super::publish::{SnapshotStage, StagedSnapshot}; -use super::super::scripted::{Script, ScriptedBlobStorage}; +use super::super::tests::holding::{holding_storage, reached_deadline}; +use super::super::tests::scripted::{Script, ScriptedBlobStorage}; use super::{BlobBackend, Lease, file_size}; use crate::filesystem_snapshot::SnapshotStoreError; use crate::services::golem_config::DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE as STORAGE_CALL_DEADLINE; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 594499f248..792181a44f 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -29,10 +29,6 @@ mod store; pub(crate) use store::RusticSnapshotStore; -#[cfg(test)] -mod holding; -#[cfg(test)] -mod scripted; #[cfg(test)] mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/mod.rs similarity index 100% rename from golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs rename to golem-worker-executor/src/filesystem_snapshot/rustic/priority/mod.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 93fee1771e..0541cc11f0 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -784,7 +784,7 @@ mod tests { use super::super::backend::BlobBackend; use super::super::fault::is_lease_expired; use super::super::files::SnapshotFiles; - use super::super::scripted::{Script, ScriptedBlobStorage}; + use super::super::tests::scripted::{Script, ScriptedBlobStorage}; use super::{ CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, ClaimEntry, FREED_PATH, FreedRecord, FreedRecords, LEDGERS_PATH, Lease, Percent, PruneLedger, claim_hold, claims_directory, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/mod.rs similarity index 100% rename from golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs rename to golem-worker-executor/src/filesystem_snapshot/rustic/publish/mod.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs index bff3f14afb..c787723cd2 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -13,8 +13,8 @@ // limitations under the License. use super::super::files::SnapshotFiles; -use super::super::holding::reached_deadline; -use super::super::scripted::{Script, ScriptedBlobStorage}; +use super::super::tests::holding::reached_deadline; +use super::super::tests::scripted::{Script, ScriptedBlobStorage}; use super::{SnapshotStage, StagedSnapshot, publish, retract}; use bytes::Bytes; use futures::FutureExt; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/mod.rs similarity index 100% rename from golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs rename to golem-worker-executor/src/filesystem_snapshot/rustic/scope/mod.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs index e40d33b055..6cd6dd06f3 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs @@ -13,7 +13,7 @@ // limitations under the License. use super::super::files::SnapshotFiles; -use super::super::scripted::{Script, ScriptedBlobStorage}; +use super::super::tests::scripted::{Script, ScriptedBlobStorage}; use super::{copy_scope, delete_scope}; use golem_common::model::environment::EnvironmentId; use golem_service_base::storage::blob::memory::InMemoryBlobStorage; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/mod.rs similarity index 100% rename from golem-worker-executor/src/filesystem_snapshot/rustic/store.rs rename to golem-worker-executor/src/filesystem_snapshot/rustic/store/mod.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index f97c5ca5c1..417b9108a2 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -23,7 +23,7 @@ use super::super::prune::{ CLOCK_SKEW_MARGIN, ClaimChoice, ClaimEntry, LEDGERS_PATH, Percent, PruneLedger, claim_hold, next_claim, parse_claim_entry, parse_freed, read_ledger, }; -use super::super::scripted::{Script, ScriptedBlobStorage}; +use super::super::tests::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; use super::{ diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/holding.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/holding.rs similarity index 98% rename from golem-worker-executor/src/filesystem_snapshot/rustic/holding.rs rename to golem-worker-executor/src/filesystem_snapshot/rustic/tests/holding.rs index b66549112f..daf660212a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/holding.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/holding.rs @@ -40,7 +40,7 @@ type Rule = Box bool + Send + Sync>; /// /// A held call waits until its gate opens, and then goes to the in-memory storage. Only the /// [`Gate`] of the storage opens the gate, so a held call waits until the test drops that value. -pub(super) struct HoldingBlobStorage { +pub(crate) struct HoldingBlobStorage { inner: Arc, rule: Rule, gate: CancellationToken, @@ -51,7 +51,7 @@ pub(super) struct HoldingBlobStorage { /// The gate of the held calls of a [`HoldingBlobStorage`]. /// /// The gate opens when the test drops this value, also when the test fails. -pub(super) struct Gate { +pub(crate) struct Gate { _open_on_drop: DropGuard, } @@ -60,7 +60,7 @@ pub(super) struct Gate { /// /// So the receiver resolves only when no value holds a copy of the storage, for example a backend /// on a thread of rustic. -pub(super) fn holding_storage( +pub(crate) fn holding_storage( inner: Arc, rule: impl Fn(&str, &Path) -> bool + Send + Sync + 'static, ) -> (Arc, Gate, oneshot::Receiver<()>) { @@ -82,7 +82,7 @@ pub(super) fn holding_storage( /// Tells whether the error or an error in its chain of sources is tokio's `Elapsed`, which is the /// root cause of the error of a call that got no answer within its deadline. -pub(super) fn reached_deadline(error: &(dyn std::error::Error + 'static)) -> bool { +pub(crate) fn reached_deadline(error: &(dyn std::error::Error + 'static)) -> bool { std::iter::successors(Some(error), |error| error.source()).any(|error| error.is::()) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/mod.rs similarity index 99% rename from golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs rename to golem-worker-executor/src/filesystem_snapshot/rustic/tests/mod.rs index d50d9814f5..33239e3407 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/mod.rs @@ -18,9 +18,12 @@ //! keeps the gate of the held calls closed until the storage is dropped. So the threads of rustic //! stop because of the deadline, and not because the gate opens. +pub(super) mod holding; +pub(super) mod scripted; + +use self::holding::{holding_storage, reached_deadline}; +use self::scripted::{Script, ScriptedBlobStorage}; use super::backend::BlobBackend; -use super::holding::{holding_storage, reached_deadline}; -use super::scripted::{Script, ScriptedBlobStorage}; use super::{ ChangeDetection, Chunking, Compression, OperationPhase, PruneSettings, RepackLimits, Repository, RepositoryKey, RepositorySettings, SaveSettings, backup_options, config_options, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/scripted.rs similarity index 97% rename from golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs rename to golem-worker-executor/src/filesystem_snapshot/rustic/tests/scripted.rs index 732fbc2790..b62e88d2ee 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/scripted.rs @@ -34,7 +34,7 @@ use tokio_util::sync::CancellationToken; /// What the storage does with one call. #[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) enum Script { +pub(crate) enum Script { /// Passes the call to the in-memory storage. Pass, /// Gives an error and does not pass the call. @@ -66,7 +66,7 @@ type Rule = Box Script + Send + Sync>; /// A blob storage that records the operation label and the path of each call, and does with each /// call what its rule gives. -pub(super) struct ScriptedBlobStorage { +pub(crate) struct ScriptedBlobStorage { inner: Arc, rule: Rule, calls: Mutex)>>, @@ -86,7 +86,7 @@ pub(super) struct ScriptedBlobStorage { } impl ScriptedBlobStorage { - pub(super) fn new( + pub(crate) fn new( inner: Arc, rule: impl Fn(&str, &Path) -> Script + Send + Sync + 'static, ) -> Arc { @@ -105,17 +105,17 @@ impl ScriptedBlobStorage { } /// Lets each call that waits for the gate, and each later such call, go on. - pub(super) fn open_gate(&self) { + pub(crate) fn open_gate(&self) { self.gate.cancel(); } /// Lets one call that waits for a step, now or later, go on. - pub(super) fn step(&self) { + pub(crate) fn step(&self) { self.steps.add_permits(1); } /// Takes back one step that no call took, and tells whether one was there. - pub(super) fn take_back_step(&self) -> bool { + pub(crate) fn take_back_step(&self) -> bool { self.steps .try_acquire() .map(tokio::sync::SemaphorePermit::forget) @@ -123,18 +123,18 @@ impl ScriptedBlobStorage { } /// Gives the number of calls that wait for a step. - pub(super) fn waiting_steps(&self) -> usize { + pub(crate) fn waiting_steps(&self) -> usize { self.waiting.load(Ordering::SeqCst) } /// Gives the number of calls that took a step and ended. - pub(super) fn stepped(&self) -> usize { + pub(crate) fn stepped(&self) -> usize { self.stepped.load(Ordering::SeqCst) } /// Gives the operation label and the path of each call that took a step, in the order of the /// steps. Two calls can wait for a step at one time, and the one that waited first takes it. - pub(super) fn took(&self) -> Vec<(&'static str, String)> { + pub(crate) fn took(&self) -> Vec<(&'static str, String)> { self.took .lock() .unwrap_or_else(PoisonError::into_inner) @@ -161,12 +161,12 @@ impl ScriptedBlobStorage { } /// Lets one late call, now or later, reach the storage. - pub(super) fn land_late(&self) { + pub(crate) fn land_late(&self) { self.landings.add_permits(1); } /// Gives the number of late calls that reached the storage. - pub(super) fn landed(&self) -> usize { + pub(crate) fn landed(&self) -> usize { self.landed.load(Ordering::SeqCst) } @@ -195,7 +195,7 @@ impl ScriptedBlobStorage { } /// Gives the operation label and the path of each call, in the order of the calls. - pub(super) fn calls(&self) -> Vec<(&'static str, String)> { + pub(crate) fn calls(&self) -> Vec<(&'static str, String)> { self.calls .lock() .unwrap_or_else(PoisonError::into_inner)