diff --git a/golem-debugging-service/src/config.rs b/golem-debugging-service/src/config.rs index 786571e83d..39432f2061 100644 --- a/golem-debugging-service/src/config.rs +++ b/golem-debugging-service/src/config.rs @@ -91,6 +91,7 @@ impl DebugConfig { public_worker_api: self.public_worker_api, memory: self.memory, filesystem_storage: Default::default(), + filesystem_snapshots: Default::default(), resource_usage_metering: Default::default(), rdbms: self.rdbms, resource_limits: self.resource_limits, diff --git a/golem-worker-executor/Cargo.toml b/golem-worker-executor/Cargo.toml index cd1e5ff1ec..d54d18649a 100644 --- a/golem-worker-executor/Cargo.toml +++ b/golem-worker-executor/Cargo.toml @@ -13,7 +13,7 @@ autotests = false [features] test-utils = [] -fs-snapshot-benchmark = ["dep:clap", "dep:rayon"] +fs-snapshot-benchmark = ["dep:clap"] [lib] path = "src/lib.rs" @@ -96,7 +96,7 @@ prometheus = { workspace = true } prost = { workspace = true } prost-types = { workspace = true } rand = { workspace = true } -rayon = { workspace = true, optional = true } +rayon = { workspace = true } regex = { workspace = true } ringbuf = { workspace = true } rustic_core = { workspace = true } @@ -169,7 +169,6 @@ goldenfile = { workspace = true } pretty_assertions = { workspace = true, features = [ "unstable" ] } proptest = { workspace = true } rand = { workspace = true } -rayon = { workspace = true } redis = { workspace = true } serde_json = { workspace = true } test-r = { workspace = true } diff --git a/golem-worker-executor/config/worker-executor.sample.env b/golem-worker-executor/config/worker-executor.sample.env index b3511b1cac..ef8278a3d0 100644 --- a/golem-worker-executor/config/worker-executor.sample.env +++ b/golem-worker-executor/config/worker-executor.sample.env @@ -38,6 +38,7 @@ GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_CAPACITY=1000 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_EVICTION_INTERVAL="1m" GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__NANOS=0 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__SECS=300 +GOLEM__FILESYSTEM_SNAPSHOTS__TYPE="Disabled" #GOLEM__FILESYSTEM_STORAGE__DETERMINISTIC_ROOT_DIR= #GOLEM__FILESYSTEM_STORAGE__MANAGED_XFS_ROOT_DIR= GOLEM__FILESYSTEM_STORAGE__CLEANUP_RETRY__MAX_ATTEMPTS=4 @@ -337,6 +338,7 @@ GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_CAPACITY=1000 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_EVICTION_INTERVAL="1m" GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__NANOS=0 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__SECS=300 +GOLEM__FILESYSTEM_SNAPSHOTS__TYPE="Disabled" #GOLEM__FILESYSTEM_STORAGE__DETERMINISTIC_ROOT_DIR= #GOLEM__FILESYSTEM_STORAGE__MANAGED_XFS_ROOT_DIR= GOLEM__FILESYSTEM_STORAGE__CLEANUP_RETRY__MAX_ATTEMPTS=4 @@ -614,6 +616,7 @@ GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_CAPACITY=1000 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_EVICTION_INTERVAL="1m" GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__NANOS=0 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__SECS=300 +GOLEM__FILESYSTEM_SNAPSHOTS__TYPE="Disabled" #GOLEM__FILESYSTEM_STORAGE__DETERMINISTIC_ROOT_DIR= #GOLEM__FILESYSTEM_STORAGE__MANAGED_XFS_ROOT_DIR= GOLEM__FILESYSTEM_STORAGE__CLEANUP_RETRY__MAX_ATTEMPTS=4 diff --git a/golem-worker-executor/config/worker-executor.toml b/golem-worker-executor/config/worker-executor.toml index 19ee85ad42..1d76c51c04 100644 --- a/golem-worker-executor/config/worker-executor.toml +++ b/golem-worker-executor/config/worker-executor.toml @@ -70,6 +70,11 @@ cache_eviction_interval = "1m" nanos = 0 secs = 300 +[filesystem_snapshots] +type = "Disabled" + +[filesystem_snapshots.config] + [filesystem_storage.cleanup_retry] max_attempts = 4 max_delay = "250ms" @@ -507,6 +512,11 @@ without_time = false # nanos = 0 # secs = 300 # +# [filesystem_snapshots] +# type = "Disabled" +# +# [filesystem_snapshots.config] +# # [filesystem_storage.cleanup_retry] # max_attempts = 4 # max_delay = "250ms" @@ -916,6 +926,11 @@ without_time = false # nanos = 0 # secs = 300 # +# [filesystem_snapshots] +# type = "Disabled" +# +# [filesystem_snapshots.config] +# # [filesystem_storage.cleanup_retry] # max_attempts = 4 # max_delay = "250ms" diff --git a/golem-worker-executor/docker/Dockerfile b/golem-worker-executor/docker/Dockerfile index 11fe1b2e8f..61eace6855 100644 --- a/golem-worker-executor/docker/Dockerfile +++ b/golem-worker-executor/docker/Dockerfile @@ -16,6 +16,10 @@ LABEL cloud.golem.cpu.features="neon,crc,lse,aes,sha2,dotprod" FROM platform-${TARGETARCH} AS final +# The image has no time zone database. jiff reads this POSIX rule without one, so the snapshot +# store writes no time zone warning for each timestamp. +ENV TZ=UTC0 + WORKDIR /app COPY /target/$RUST_TARGET/release/worker-executor ./ COPY /golem-worker-executor/config/worker-executor.toml ./config/worker-executor.toml diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs index 97ace67da1..7b05b1ad9e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs @@ -39,9 +39,10 @@ mod volume; use super::rustic::{ ChangeDetection, Chunking, Compression, InspectReport, PhaseTime, Repository, RepositoryKey, - RepositorySettings, STORAGE_CALL_DEADLINE, SaveSettings, + RepositorySettings, SaveSettings, }; use super::{SnapshotName, SnapshotScope}; +use crate::services::golem_config::DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE as STORAGE_CALL_DEADLINE; use agents::{AgentStorage, FIRST_AGENT}; use golem_common::model::environment::EnvironmentId; use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; diff --git a/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs b/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs index 1d57d21b91..bd4f6392bf 100644 --- a/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs @@ -40,7 +40,8 @@ pub(super) mod fixture; use super::{ - FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, + ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, + SnapshotStoreError, }; use fixture::{ Listed, Scratch, Spec, files_and_bytes, fixture, listing, one_file, pattern, write_tree, @@ -62,6 +63,10 @@ pub(crate) type OpenStore = Arc Arc + S type Case = fn(OpenStore) -> BoxFuture<'static, ()>; +/// The number of saves that the dropped-save case starts to find one that does not end in its +/// first poll. +const DROP_ATTEMPTS: usize = 20; + /// Each case of the contract, with its name. const CASES: &[(&str, Case)] = &[ ("a_saved_tree_comes_back_the_same", |open| { @@ -161,6 +166,12 @@ const CASES: &[(&str, Case)] = &[ "a_dropped_save_publishes_nothing_and_leaves_the_name_free", |open| a_dropped_save_publishes_nothing_and_leaves_the_name_free(open).boxed(), ), + ( + "a_save_with_a_truthful_parent_restores_the_same_tree_as_a_save_without_one", + |open| { + a_save_with_a_truthful_parent_restores_the_same_tree_as_a_save_without_one(open).boxed() + }, + ), ("no_method_blocks_the_runtime", |open| { no_method_blocks_the_runtime(open).boxed() }), @@ -246,7 +257,7 @@ async fn a_saved_tree_comes_back_the_same(open: OpenStore) { let tree = new_tree(&fixture()); let saved = store - .save(&scope, &name("p-fixture"), tree.path()) + .save(&scope, &name("p-fixture"), tree.path(), None) .await .unwrap(); let (restored, info) = restored(&*store, &scope, &name("p-fixture")).await.unwrap(); @@ -254,6 +265,67 @@ async fn a_saved_tree_comes_back_the_same(open: OpenStore) { assert_eq!((restored, info), (listing(tree.path()), saved)); } +async fn a_save_with_a_truthful_parent_restores_the_same_tree_as_a_save_without_one( + open: OpenStore, +) { + // Each changed file gets a new size, so the parent is truthful for both modes. + let store = open(); + let scope = new_scope(); + let file = |content: &str| Spec::File { + content: Box::from(content.as_bytes()), + mode: 0o644, + }; + let tree = new_tree(&[ + ("changed.txt", file("old")), + ("kept.txt", file("kept")), + ("removed.txt", file("removed")), + ]); + let parent = name("p-parent"); + store + .save(&scope, &parent, tree.path(), None) + .await + .unwrap(); + write_tree( + tree.path(), + &[ + ("changed.txt", file("new and longer")), + ("added.txt", file("added")), + ], + ); + std::fs::remove_file(tree.path().join("removed.txt")).unwrap(); + + store + .save(&scope, &name("p-none"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-size-mtime"), + tree.path(), + Some((&parent, ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + store + .save( + &scope, + &name("p-full"), + tree.path(), + Some((&parent, ChangeDetection::Full)), + ) + .await + .unwrap(); + let restored = [ + restored_listing(&*store, &scope, &name("p-none")).await, + restored_listing(&*store, &scope, &name("p-size-mtime")).await, + restored_listing(&*store, &scope, &name("p-full")).await, + ]; + + let expected = listing(tree.path()); + assert_eq!(restored, [expected.clone(), expected.clone(), expected]); +} + async fn each_name_of_a_hard_linked_file_comes_back_as_its_own_file(open: OpenStore) { let store = open(); let scope = new_scope(); @@ -266,7 +338,7 @@ async fn each_name_of_a_hard_linked_file_comes_back_as_its_own_file(open: OpenSt let into = Scratch::new(); store - .save(&scope, &name("p-linked"), tree.path()) + .save(&scope, &name("p-linked"), tree.path(), None) .await .unwrap(); store @@ -297,7 +369,7 @@ async fn save_stat_list_and_restore_give_the_same_info(open: OpenStore) { let before = Timestamp::now_utc(); let saved = store - .save(&scope, &name("p-info"), tree.path()) + .save(&scope, &name("p-info"), tree.path(), None) .await .unwrap(); let after = Timestamp::now_utc(); @@ -330,7 +402,7 @@ async fn a_save_leaves_the_tree_as_it_was(open: OpenStore) { let before = listing(tree.path()); store - .save(&scope, &name("p-source"), tree.path()) + .save(&scope, &name("p-source"), tree.path(), None) .await .unwrap(); @@ -343,11 +415,13 @@ async fn a_name_in_use_gives_already_exists_and_changes_nothing(open: OpenStore) let first = new_tree(&one_file("first")); let second = new_tree(&one_file("second tree")); let saved = store - .save(&scope, &name("p-taken"), first.path()) + .save(&scope, &name("p-taken"), first.path(), None) .await .unwrap(); - let again = store.save(&scope, &name("p-taken"), second.path()).await; + let again = store + .save(&scope, &name("p-taken"), second.path(), None) + .await; assert!( matches!(again, Err(SnapshotStoreError::AlreadyExists)), @@ -378,12 +452,14 @@ async fn a_tree_that_cannot_be_read_gives_source_and_publishes_nothing(open: Ope std::fs::write(&file, b"not a directory").unwrap(); let tree = new_tree(&one_file("real")); - let from_missing = store.save(&scope, &name("p-unread"), &missing).await; - let from_file = store.save(&scope, &name("p-unread"), &file).await; + let from_missing = store.save(&scope, &name("p-unread"), &missing, None).await; + let from_file = store.save(&scope, &name("p-unread"), &file, None).await; let stat = store.stat(&scope, &name("p-unread")).await.unwrap(); let names = listed_names(&*store, &scope).await; let restore = restored(&*store, &scope, &name("p-unread")).await; - let later = store.save(&scope, &name("p-unread"), tree.path()).await; + let later = store + .save(&scope, &name("p-unread"), tree.path(), None) + .await; assert!( matches!(from_missing, Err(SnapshotStoreError::Source(_))), @@ -417,7 +493,9 @@ async fn an_entry_that_cannot_be_read_gives_source_and_publishes_nothing(open: O std::fs::write(&locked, b"locked").unwrap(); std::fs::set_permissions(&locked, std::fs::Permissions::from_mode(0o000)).unwrap(); - let saved = store.save(&scope, &name("p-locked"), tree.path()).await; + let saved = store + .save(&scope, &name("p-locked"), tree.path(), None) + .await; let stat = store.stat(&scope, &name("p-locked")).await.unwrap(); let names = listed_names(&*store, &scope).await; @@ -444,7 +522,7 @@ async fn sequential_saves_get_later_times_and_list_newest_first(open: OpenStore) let store = store.clone(); let scope = scope.clone(); let tree = tree.path().to_path_buf(); - async move { store.save(&scope, &name(text), &tree).await.unwrap() } + async move { store.save(&scope, &name(text), &tree, None).await.unwrap() } }) .collect::>() .await; @@ -474,7 +552,7 @@ async fn an_unknown_name_gives_not_found_every_time(open: OpenStore) { let used = new_scope(); let tree = new_tree(&one_file("other")); store - .save(&used, &name("p-other"), tree.path()) + .save(&used, &name("p-other"), tree.path(), None) .await .unwrap(); @@ -501,7 +579,7 @@ async fn a_restore_into_a_directory_that_is_not_empty_writes_nothing(open: OpenS let scope = new_scope(); let tree = new_tree(&one_file("content")); store - .save(&scope, &name("p-busy"), tree.path()) + .save(&scope, &name("p-busy"), tree.path(), None) .await .unwrap(); let into = new_tree(&[( @@ -528,7 +606,7 @@ async fn a_restore_into_a_path_that_is_not_a_directory_writes_nothing(open: Open let scope = new_scope(); let tree = new_tree(&one_file("content")); store - .save(&scope, &name("p-nowhere"), tree.path()) + .save(&scope, &name("p-nowhere"), tree.path(), None) .await .unwrap(); let parent = Scratch::new(); @@ -558,11 +636,11 @@ async fn a_deleted_name_stops_resolving_at_once(open: OpenStore) { let scope = new_scope(); let tree = new_tree(&one_file("deleted")); store - .save(&scope, &name("p-deleted"), tree.path()) + .save(&scope, &name("p-deleted"), tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), tree.path()) + .save(&scope, &name("p-kept"), tree.path(), None) .await .unwrap(); @@ -583,7 +661,7 @@ async fn delete_is_idempotent(open: OpenStore) { let scope = new_scope(); let tree = new_tree(&one_file("twice")); store - .save(&scope, &name("p-twice"), tree.path()) + .save(&scope, &name("p-twice"), tree.path(), None) .await .unwrap(); @@ -607,15 +685,15 @@ async fn a_delete_keeps_every_other_snapshot(open: OpenStore) { let shared = new_tree(&fixture()); let other = new_tree(&one_file("other")); store - .save(&scope, &name("p-twin-1"), shared.path()) + .save(&scope, &name("p-twin-1"), shared.path(), None) .await .unwrap(); store - .save(&scope, &name("p-twin-2"), shared.path()) + .save(&scope, &name("p-twin-2"), shared.path(), None) .await .unwrap(); store - .save(&scope, &name("p-other"), other.path()) + .save(&scope, &name("p-other"), other.path(), None) .await .unwrap(); @@ -640,7 +718,7 @@ async fn a_restore_that_races_a_delete_of_its_name_gives_a_whole_tree_or_nothing let scope = new_scope(); let tree = new_tree(&fixture()); store - .save(&scope, &name("p-raced"), tree.path()) + .save(&scope, &name("p-raced"), tree.path(), None) .await .unwrap(); @@ -664,11 +742,11 @@ async fn a_restore_during_a_delete_of_another_name_gives_the_whole_tree(open: Op let tree = new_tree(&fixture()); let other = new_tree(&fixture()); store - .save(&scope, &name("p-restored"), tree.path()) + .save(&scope, &name("p-restored"), tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-deleted"), other.path()) + .save(&scope, &name("p-deleted"), other.path(), None) .await .unwrap(); @@ -691,14 +769,17 @@ async fn a_save_a_restore_and_a_delete_in_one_scope_run_at_the_same_time(open: O let second = new_tree(&one_file("second")); let third = new_tree(&one_file("third")); let (first_name, second_name, third_name) = (name("p-1"), name("p-2"), name("p-3")); - store.save(&scope, &first_name, first.path()).await.unwrap(); store - .save(&scope, &second_name, second.path()) + .save(&scope, &first_name, first.path(), None) + .await + .unwrap(); + store + .save(&scope, &second_name, second.path(), None) .await .unwrap(); let (saved, restore, deleted) = futures::join!( - store.save(&scope, &third_name, third.path()), + store.save(&scope, &third_name, third.path(), None), restored(&*store, &scope, &first_name), store.delete(&scope, &second_name) ); @@ -724,14 +805,23 @@ async fn a_deleted_scope_is_as_unused_as_before_its_first_save(open: OpenStore) let scope = new_scope(); let old = new_tree(&one_file("old")); let new = new_tree(&one_file("new tree")); - store.save(&scope, &name("p-1"), old.path()).await.unwrap(); - store.save(&scope, &name("p-2"), old.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), old.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), old.path(), None) + .await + .unwrap(); store.delete_scope(&scope).await.unwrap(); let names = listed_names(&*store, &scope).await; let stat = store.stat(&scope, &name("p-1")).await.unwrap(); let restore = restored(&*store, &scope, &name("p-2")).await; - store.save(&scope, &name("p-1"), new.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), new.path(), None) + .await + .unwrap(); assert!(is_not_found(&restore), "{restore:?}"); assert_eq!( @@ -750,10 +840,13 @@ async fn delete_scope_is_idempotent_and_keeps_other_scopes(open: OpenStore) { let kept = new_scope(); let tree = new_tree(&one_file("kept")); store - .save(&deleted, &name("p-1"), tree.path()) + .save(&deleted, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save(&kept, &name("p-1"), tree.path(), None) .await .unwrap(); - store.save(&kept, &name("p-1"), tree.path()).await.unwrap(); let results = [ store.delete_scope(&new_scope()).await.is_ok(), @@ -783,9 +876,12 @@ async fn a_copied_scope_has_the_same_names_infos_and_trees(open: OpenStore) { let to = new_scope(); let first = new_tree(&fixture()); let second = new_tree(&one_file("second")); - store.save(&from, &name("p-1"), first.path()).await.unwrap(); store - .save(&from, &name("u-2"), second.path()) + .save(&from, &name("p-1"), first.path(), None) + .await + .unwrap(); + store + .save(&from, &name("u-2"), second.path(), None) .await .unwrap(); let source = store.list(&from).await.unwrap(); @@ -818,12 +914,21 @@ async fn copied_scopes_are_independent(open: OpenStore) { let to = new_scope(); let tree = new_tree(&one_file("copied")); let later = new_tree(&one_file("later")); - store.save(&from, &name("p-1"), tree.path()).await.unwrap(); - store.save(&from, &name("p-2"), tree.path()).await.unwrap(); + store + .save(&from, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save(&from, &name("p-2"), tree.path(), None) + .await + .unwrap(); store.copy_scope(&from, &to).await.unwrap(); store.delete(&from, &name("p-1")).await.unwrap(); - store.save(&to, &name("p-3"), later.path()).await.unwrap(); + store + .save(&to, &name("p-3"), later.path(), None) + .await + .unwrap(); let target_after_source_delete = restored_listing(&*store, &to, &name("p-1")).await; let source_names = listed_names(&*store, &from).await; store.delete_scope(&to).await.unwrap(); @@ -851,7 +956,7 @@ async fn a_copy_of_an_unused_scope_leaves_the_target_unused(open: OpenStore) { let to = new_scope(); let tree = new_tree(&one_file("other scope")); store - .save(&other, &name("p-other"), tree.path()) + .save(&other, &name("p-other"), tree.path(), None) .await .unwrap(); @@ -867,11 +972,11 @@ async fn one_name_in_two_scopes_gives_two_snapshots(open: OpenStore) { let first = new_tree(&one_file("first scope")); let second = new_tree(&one_file("second scope")); store - .save(&first_scope, &name("p-same"), first.path()) + .save(&first_scope, &name("p-same"), first.path(), None) .await .unwrap(); store - .save(&second_scope, &name("p-same"), second.path()) + .save(&second_scope, &name("p-same"), second.path(), None) .await .unwrap(); let first_restored = restored_listing(&*store, &first_scope, &name("p-same")).await; @@ -900,7 +1005,7 @@ async fn a_save_through_one_store_resolves_through_another(open: OpenStore) { let tree = new_tree(&fixture()); let saved = writer - .save(&scope, &name("p-shared"), tree.path()) + .save(&scope, &name("p-shared"), tree.path(), None) .await .unwrap(); @@ -929,8 +1034,8 @@ async fn two_stores_save_into_a_new_scope_at_the_same_time(open: OpenStore) { let (first_name, second_name) = (name("p-first"), name("p-second")); let (first_saved, second_saved) = futures::join!( - first.save(&scope, &first_name, first_tree.path()), - second.save(&scope, &second_name, second_tree.path()) + first.save(&scope, &first_name, first_tree.path(), None), + second.save(&scope, &second_name, second_tree.path(), None) ); let mut names = listed_names(&*first, &scope).await; names.sort(); @@ -958,23 +1063,30 @@ async fn a_dropped_save_publishes_nothing_and_leaves_the_name_free(open: OpenSto // interrupted save, so it publishes nothing and leaves the name free. This case checks at // once after the drop, and it cannot see a publish that comes much later. So an adapter that // runs its save in the background also proves in its own tests that a dropped save stops. + // A save can end in its first poll when a busy host runs its read before that poll, so the + // case tries saves in new scopes until one does not end in its first poll. let store = open(); - let scope = new_scope(); let dropped_name = name("p-dropped"); let tree = new_tree(&fixture()); let other = new_tree(&one_file("other tree")); - let dropped = store - .save(&scope, &dropped_name, tree.path()) - .now_or_never(); - assert!( - dropped.is_none(), - "the save returned in one poll, so the case cannot drop it before it returns: {dropped:?}" - ); + let scope = futures::stream::iter(0..DROP_ATTEMPTS) + .filter_map(|_| { + let scope = new_scope(); + let dropped = store + .save(&scope, &dropped_name, tree.path(), None) + .now_or_never(); + std::future::ready(dropped.is_none().then_some(scope)) + }) + .next() + .await + .unwrap_or_else(|| { + panic!("each of {DROP_ATTEMPTS} saves returned in one poll, so the case cannot drop one before it returns") + }); let stat = store.stat(&scope, &dropped_name).await.unwrap(); let restore = restored(&*store, &scope, &dropped_name).await; let names_after_the_drop = listed_names(&*store, &scope).await; - let saved_again = store.save(&scope, &dropped_name, other.path()).await; + let saved_again = store.save(&scope, &dropped_name, other.path(), None).await; assert!(is_not_found(&restore), "{restore:?}"); assert!(saved_again.is_ok(), "{saved_again:?}"); @@ -1030,7 +1142,7 @@ async fn no_method_blocks_the_runtime(open: OpenStore) { let before_save = ticks.load(Ordering::SeqCst); store - .save(&scope, &name("p-large"), &tree_path) + .save(&scope, &name("p-large"), &tree_path, None) .await .unwrap(); let during_save = counted(before_save); diff --git a/golem-worker-executor/src/filesystem_snapshot/memory.rs b/golem-worker-executor/src/filesystem_snapshot/memory/mod.rs similarity index 97% rename from golem-worker-executor/src/filesystem_snapshot/memory.rs rename to golem-worker-executor/src/filesystem_snapshot/memory/mod.rs index b4fe682a18..3816efb695 100644 --- a/golem-worker-executor/src/filesystem_snapshot/memory.rs +++ b/golem-worker-executor/src/filesystem_snapshot/memory/mod.rs @@ -25,8 +25,8 @@ mod tree; mod tests; use super::{ - FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, - newest_first, snapshot_time, + ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, + SnapshotStoreError, newest_first, snapshot_time, }; use async_trait::async_trait; use golem_common::model::Timestamp; @@ -120,6 +120,7 @@ impl FilesystemSnapshotStore for InMemorySnapshotStore { scope: &SnapshotScope, name: &SnapshotName, tree: &Path, + _parent: Option<(&SnapshotName, ChangeDetection)>, ) -> Result { let snapshots = self.snapshots_of(scope); if found(&snapshots, name).is_some() { diff --git a/golem-worker-executor/src/filesystem_snapshot/memory/tests.rs b/golem-worker-executor/src/filesystem_snapshot/memory/tests.rs index af377f4d88..d71a58668b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/memory/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/memory/tests.rs @@ -57,7 +57,12 @@ async fn a_save_after_a_snapshot_from_a_clock_that_is_ahead_gets_a_later_time() let tree = tempfile::tempdir().unwrap(); let saved = store - .save(&scope, &SnapshotName::new("p-next").unwrap(), tree.path()) + .save( + &scope, + &SnapshotName::new("p-next").unwrap(), + tree.path(), + None, + ) .await .unwrap(); let listed = store.list(&scope).await.unwrap(); @@ -97,8 +102,8 @@ async fn of_two_saves_of_one_name_at_the_same_time_one_wins() { let second_tree = tree_with("second tree"); let (first_saved, second_saved) = futures::join!( - first.save(&scope, &name, first_tree.path()), - second.save(&scope, &name, second_tree.path()) + first.save(&scope, &name, first_tree.path(), None), + second.save(&scope, &name, second_tree.path(), None) ); let into = tempfile::tempdir().unwrap(); first.restore(&scope, &name, into.path()).await.unwrap(); @@ -145,7 +150,7 @@ mod unix { let scope = new_scope(); let name = SnapshotName::new("p-socket").unwrap(); - let saved = store.save(&scope, &name, tree.path()).await; + let saved = store.save(&scope, &name, tree.path(), None).await; let listed = store.list(&scope).await.unwrap(); assert!( diff --git a/golem-worker-executor/src/filesystem_snapshot/mod.rs b/golem-worker-executor/src/filesystem_snapshot/mod.rs index f9a6d904f3..aa2f8dc5ef 100644 --- a/golem-worker-executor/src/filesystem_snapshot/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/mod.rs @@ -31,9 +31,13 @@ pub(crate) mod benchmark; mod contract_tests; mod memory; mod rustic; +#[cfg(test)] +mod time_zone_tests; #[allow(unused_imports)] pub(crate) use memory::InMemorySnapshotStore; +#[allow(unused_imports)] +pub(crate) use rustic::RusticSnapshotStore; /// The place of the filesystem snapshots of one agent. /// @@ -117,6 +121,15 @@ pub(crate) struct SnapshotInfo { pub bytes: u64, } +/// How a save with a parent finds the files that did not change since the parent. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) enum ChangeDetection { + /// Compares each file with the parent by size and modification time. + SizeMtime, + /// Reads every file. + Full, +} + /// Why a call on a [`FilesystemSnapshotStore`] failed. #[derive(Debug)] pub(crate) enum SnapshotStoreError { @@ -206,11 +219,19 @@ pub(crate) trait FilesystemSnapshotStore: Send + Sync { /// The result gives the number of files, the size of the tree, and the time of the /// snapshot. That time is later than the time of each snapshot that the scope held when the /// save started. + /// + /// `parent` names the snapshot that the save can compare with. With `SizeMtime`, the store can + /// keep the content of the parent for a file whose size and modification time equal those of + /// the same path in the parent, and then it does not read that file. So such a file that + /// changed can keep the content of the parent. A store can also read each file. With `Full`, + /// the save reads every file. A parent that the scope does not hold gives a save that reads + /// every file. async fn save( &self, scope: &SnapshotScope, name: &SnapshotName, tree: &Path, + parent: Option<(&SnapshotName, ChangeDetection)>, ) -> Result; /// Rebuilds a saved tree in the empty directory `into`. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs new file mode 100644 index 0000000000..e3d5bdb638 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs @@ -0,0 +1,157 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The packs that one backend keeps in memory after a full read, up to a limit of bytes. +//! +//! rustic reads each tree blob with its own ranged read. A backend lives for one operation, so it +//! keeps each pack of tree blobs after its first read and gives the later ranges from memory. Once +//! the kept packs fill the limit, or a pack that was read whole did not fit, the set is closed: a +//! pack that is not kept is not read whole, because it cannot be kept, and the caller reads only +//! its range. + +use bytes::Bytes; +use rustic_core::{Id, RusticResult}; +use std::collections::{HashMap, HashSet}; +use std::fmt::{Debug, Formatter}; +use std::sync::{Condvar, Mutex, MutexGuard, PoisonError}; + +/// The packs that one backend keeps, and the packs that a thread reads now. When the kept bytes +/// reach the limit, or a pack that was read whole does not fit, the set is closed. A closed set +/// keeps no more packs, and a pack that it does not keep is not read whole. +pub(super) struct KeptPacks { + limit: usize, + state: Mutex, + /// Wakes the threads that wait for the read of a pack when that read ends. + read_ended: Condvar, +} + +#[derive(Default)] +struct State { + packs: HashMap, + bytes: usize, + /// A pack that was read whole did not fit in the limit. + closed: bool, + reading: HashSet, + /// The threads that wait for the read of a pack by another thread. + #[cfg(test)] + waiters: usize, +} + +impl KeptPacks { + /// Gives an empty set that keeps packs up to `limit` bytes in total. + pub(super) fn new(limit: usize) -> Self { + Self { + limit, + state: Mutex::default(), + read_ended: Condvar::new(), + } + } + + /// Gives the kept pack, or reads it with `read`. While one thread reads a pack, the other + /// threads that want it wait for that read, and then take the kept pack or read it again. A + /// pack is kept only when its read succeeds and it fits in the limit. When the set is closed and + /// the pack is not kept, it gives `None` and does not read, so the caller reads only its range. + /// A failed read keeps nothing and does not close the set. + pub(super) fn get_or_read( + &self, + id: &Id, + read: impl FnOnce() -> RusticResult, + ) -> Option> { + #[cfg(test)] + let mut counted = false; + let mut state = self + .read_ended + .wait_while(self.state(), |state| { + let waits = state.reading.contains(id); + #[cfg(test)] + if waits && !counted { + state.waiters += 1; + counted = true; + } + waits + }) + .unwrap_or_else(PoisonError::into_inner); + #[cfg(test)] + if counted { + state.waiters -= 1; + } + if let Some(pack) = state.packs.get(id) { + return Some(Ok(pack.clone())); + } + if state.closed || state.bytes >= self.limit { + return None; + } + state.reading.insert(*id); + drop(state); + let reading = Reading { + kept: self, + id: *id, + }; + let read = read(); + if let Ok(pack) = &read { + reading.keep(pack); + } + Some(read) + } + + /// Gives the number of threads that wait for the read of a pack by another thread. + #[cfg(test)] + pub(super) fn waiters(&self) -> usize { + self.state().waiters + } + + fn state(&self) -> MutexGuard<'_, State> { + self.state.lock().unwrap_or_else(PoisonError::into_inner) + } +} + +impl Debug for KeptPacks { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + let state = self.state(); + formatter + .debug_struct("KeptPacks") + .field("limit", &self.limit) + .field("packs", &state.packs.len()) + .field("bytes", &state.bytes) + .finish() + } +} + +/// The read of one pack by one thread. Its drop ends the read and wakes the waiting threads, also +/// when the read fails or panics. +struct Reading<'a> { + kept: &'a KeptPacks, + id: Id, +} + +impl Reading<'_> { + /// Keeps the pack when it fits in the limit, and closes the set when it does not. + fn keep(&self, pack: &Bytes) { + let mut state = self.kept.state(); + let bytes = state.bytes.saturating_add(pack.len()); + if bytes <= self.kept.limit { + state.bytes = bytes; + state.packs.insert(self.id, pack.clone()); + } else { + state.closed = true; + } + } +} + +impl Drop for Reading<'_> { + fn drop(&mut self) { + self.kept.state().reading.remove(&self.id); + self.kept.read_ended.notify_all(); + } +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/mod.rs similarity index 53% rename from golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs rename to golem-worker-executor/src/filesystem_snapshot/rustic/backend/mod.rs index 856eb0dcee..3035e47f27 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/mod.rs @@ -17,24 +17,67 @@ //! The backend keeps the files of one repository in one blob storage namespace, with the paths of //! the restic repository format. rustic calls the backend from threads outside the async runtime, //! and each call waits for the blob storage on the runtime that the backend holds. Each call waits -//! for at most a deadline. +//! for at most a deadline, and a cancelled operation makes no more calls. +use super::fault::{BlobCallFailed, ConfigExists, FileMissing, LeaseExpired, OperationCancelled}; +use super::files::TARGET_LABEL; +use super::publish::{SnapshotStage, StagedSnapshot}; use bytes::Bytes; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use golem_service_base::storage::blob::{ + BlobRangeError, BlobStorage, BlobStorageNamespace, PutIfAbsent, +}; +use kept::KeptPacks; use rustic_core::{ BytesList, ErrorKind, FileType, Id, ReadBackend, RusticError, RusticResult, WriteBackend, }; use std::future::Future; use std::path::{Path, PathBuf}; -use std::sync::Arc; -use std::time::Duration; +use std::sync::{Arc, Mutex, PoisonError}; +use std::time::{Duration, Instant}; use tokio::runtime::Handle; - -/// The target label of each blob storage call of the backend. -const TARGET_LABEL: &str = "filesystem_snapshot"; +use tokio_util::sync::CancellationToken; +use tokio_util::task::task_tracker::TaskTrackerToken; /// The path of the config file of a repository. -const CONFIG_PATH: &str = "config"; +pub(super) const CONFIG_PATH: &str = "config"; + +/// The largest number of bytes of tree packs that one backend keeps in memory. +const KEPT_PACKS_LIMIT: usize = 32 * 1024 * 1024; + +/// The time until which a prune may make storage calls. A prune holds its claim until the time in +/// its newest marker plus the hold, as the other deletes see it. The lease ends before that, so a +/// prune stops before another delete can take its claim over. +#[derive(Debug)] +pub(super) struct Lease { + expiry: Mutex, +} + +impl Lease { + /// Gives a lease that ends at the instant. + pub(super) fn until(expiry: Instant) -> Self { + Self { + expiry: Mutex::new(expiry), + } + } + + /// Moves the end of the lease to `span` after `started`, the start of a marker write that + /// succeeded, when that is later. A write that started at or after the end of the lease does + /// not move it, so a lease that ran out stays out: another delete can have taken the claim over + /// before the marker of that write was visible. A write that started before the end and ends + /// late can still move it, because no other delete can take the claim over before that marker + /// is visible. + pub(super) fn extend_from(&self, started: Instant, span: Duration) { + let mut current = self.expiry.lock().unwrap_or_else(PoisonError::into_inner); + if started < *current { + *current = (*current).max(started + span); + } + } + + /// Gives the end of the lease. + pub(super) fn expiry(&self) -> Instant { + *self.expiry.lock().unwrap_or_else(PoisonError::into_inner) + } +} /// A call that the backend makes on the blob storage. #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -77,12 +120,24 @@ impl StorageCall { /// A call that gets no answer from the blob storage within the deadline gives an error. The /// runtime must be a multi-thread runtime, because on a `current_thread` runtime `Handle::block_on` /// does not drive the timer of the deadline. +/// +/// The config file and the index files are written only when their path has no blob. A snapshot +/// file goes into the stage of the backend when it has one, and the backend does not write it. #[derive(Debug)] pub(super) struct BlobBackend { storage: Arc, namespace: BlobStorageNamespace, runtime: Handle, deadline: Duration, + cancel: CancellationToken, + stage: Option>, + /// Counts the backend as work of a tracker, until the last owner drops the backend. + _tracked: Option, + /// The packs of tree blobs that the operation of the backend read. + kept: KeptPacks, + /// The lease of the prune of the backend. A backend without a lease has no limit other than + /// the deadline of each call. + lease: Option>, } impl BlobBackend { @@ -99,23 +154,103 @@ impl BlobBackend { namespace, runtime, deadline, + cancel: CancellationToken::new(), + stage: None, + _tracked: None, + kept: KeptPacks::new(KEPT_PACKS_LIMIT), + lease: None, + } + } + + /// Gives the backend with another limit of the bytes of the tree packs that it keeps. + pub(super) fn keeping_packs_up_to(self, limit: usize) -> Self { + Self { + kept: KeptPacks::new(limit), + ..self + } + } + + /// Gives the backend with the token of its operation. When the token is cancelled, a call that + /// has not started gives an error at once, and a call that runs stops and gives an error. + pub(super) fn cancelled_by(self, cancel: CancellationToken) -> Self { + Self { cancel, ..self } + } + + /// Gives the backend with a stage for the snapshot file of a save. + pub(super) fn staging_in(self, stage: Arc) -> Self { + Self { + stage: Some(stage), + ..self + } + } + + /// Gives the backend with the lease of a prune. When the lease has run out, a call gives an + /// error at once, and a call that runs gives an error when the lease runs out. + pub(super) fn leased_by(self, lease: Arc) -> Self { + Self { + lease: Some(lease), + ..self + } + } + + /// Gives the backend with a token of a task tracker. The tracker counts the backend until the + /// last owner drops it, for example a thread of rustic. + pub(super) fn tracked_by(self, token: TaskTrackerToken) -> Self { + Self { + _tracked: Some(token), + ..self } } - /// Waits for one call on the blob storage, and gives its result as a rustic result. - /// - /// Each call of the backend on the blob storage goes through this function. A call that gives - /// no answer within the deadline gives an error, the same as a call that failed. + /// Waits for one call on the blob storage, which each call of the backend goes through. A call + /// without an answer within the deadline, of a cancelled operation, or of a backend whose lease + /// ran out, gives an error, the same as a call that failed. fn request( &self, call: StorageCall, path: &Path, future: impl Future>, ) -> RusticResult { + let answer = answer_or_cancel(self.deadline, &self.cancel, future); self.runtime - .block_on(answer_within(self.deadline, future)) + .block_on(async { + match &self.lease { + None => answer.await, + Some(lease) => within_lease(lease, answer).await, + } + }) .map_err(|error| storage_error(call, path, error)) } + + /// Writes the content at the path only when the path has no blob, and gives whether it wrote. + fn write_if_absent(&self, path: &Path, content: &[u8]) -> RusticResult { + self.request( + StorageCall::Write, + path, + self.storage.put_raw_if_absent( + TARGET_LABEL, + StorageCall::Write.label(), + self.namespace.clone(), + path, + content, + ), + ) + } +} + +/// Gives the output of the future, or [`LeaseExpired`] when the lease runs out first. A call does +/// not start when the lease has run out. +pub(super) async fn within_lease( + lease: &Lease, + future: impl Future>, +) -> anyhow::Result { + let left = lease.expiry().saturating_duration_since(Instant::now()); + if left.is_zero() { + return Err(anyhow::Error::new(LeaseExpired)); + } + tokio::time::timeout(left, future) + .await + .unwrap_or_else(|_| Err(anyhow::Error::new(LeaseExpired))) } /// Gives the output of the future, or an error when the future gives no output within the deadline. @@ -124,7 +259,7 @@ impl BlobBackend { /// thread without a runtime context can wait for the result with `Handle::block_on`. A future that /// is ready at its first poll always gives its output. At the deadline, the function drops the /// future and gives an error whose root cause is tokio's `Elapsed`. -async fn answer_within( +pub(super) async fn answer_within( deadline: Duration, future: impl Future>, ) -> anyhow::Result { @@ -137,6 +272,24 @@ async fn answer_within( }) } +/// Gives the output of the future within the deadline, or an error when the operation of the token +/// is cancelled. A call of a cancelled operation does not start, and a cancel ends a call that +/// runs. +pub(super) async fn answer_or_cancel( + deadline: Duration, + cancel: &CancellationToken, + future: impl Future>, +) -> anyhow::Result { + if cancel.is_cancelled() { + return Err(anyhow::Error::new(OperationCancelled)); + } + tokio::select! { + biased; + answer = answer_within(deadline, future) => answer, + () = cancel.cancelled() => Err(anyhow::Error::new(OperationCancelled)), + } +} + impl ReadBackend for BlobBackend { fn location(&self) -> String { format!("golem-blob-storage:{:?}", self.namespace) @@ -206,7 +359,7 @@ impl ReadBackend for BlobBackend { &self, tpe: FileType, id: &Id, - _cacheable: bool, + cacheable: bool, offset: u32, length: u32, ) -> RusticResult { @@ -214,6 +367,16 @@ impl ReadBackend for BlobBackend { let Some(last) = length.checked_sub(1) else { return Ok(Bytes::new()); }; + // rustic marks the reads of tree blobs as cacheable, and reads each tree blob on its own. + // A pack of tree blobs is read whole and kept, until the kept packs fill their limit or a + // pack that was read whole does not fit. After that, a pack that is not kept is read by its + // range, as each other blob. + if cacheable + && tpe == FileType::Pack + && let Some(pack) = self.kept.get_or_read(id, || self.read_full(tpe, id)) + { + return range_of(&pack?, &path, offset, last); + } let start = u64::from(offset); self.request( StorageCall::ReadRange, @@ -248,25 +411,36 @@ impl WriteBackend for BlobBackend { ) -> RusticResult<()> { let path = file_path(tpe, id)?; let parts = content.into_vec(); - let joined; - let data: &[u8] = match parts.as_slice() { - [part] => part, - parts => { - joined = join(parts); - &joined - } + let content = match parts.as_slice() { + [part] => part.clone(), + parts => Bytes::from(join(parts)), }; - self.request( - StorageCall::Write, - &path, - self.storage.put_raw( - TARGET_LABEL, - StorageCall::Write.label(), - self.namespace.clone(), + match (tpe, &self.stage) { + (FileType::Snapshot, Some(stage)) => stage + .keep(StagedSnapshot { + path: Arc::from(path), + content, + }) + .map_err(|staged| second_snapshot(&staged.path)), + (FileType::Config, _) => match self.write_if_absent(&path, &content)? { + PutIfAbsent::Written => Ok(()), + PutIfAbsent::AlreadyExists => Err(config_exists(&path)), + }, + // The name of an index file is the hash of its content, so a blob at the path holds + // the same content. + (FileType::Index, _) => self.write_if_absent(&path, &content).map(|_| ()), + _ => self.request( + StorageCall::Write, &path, - data, + self.storage.put_raw( + TARGET_LABEL, + StorageCall::Write.label(), + self.namespace.clone(), + &path, + &content, + ), ), - ) + } } fn remove(&self, tpe: FileType, id: &Id, _cacheable: bool) -> RusticResult<()> { @@ -333,11 +507,50 @@ fn join(parts: &[Bytes]) -> Box<[u8]> { .into_boxed_slice() } +/// Gives the bytes from `offset` to `last` of the pack as a slice of the pack. A range outside the +/// pack gives the error of a ranged read outside a blob. +fn range_of(pack: &Bytes, path: &Path, offset: u32, last: u32) -> RusticResult { + let start = u64::from(offset); + let end = start + u64::from(last); + usize::try_from(start) + .ok() + .zip(usize::try_from(end).ok()) + .filter(|(_, end)| *end < pack.len()) + .map(|(start, end)| pack.slice(start..=end)) + .ok_or_else(|| { + storage_error( + StorageCall::ReadRange, + path, + anyhow::Error::new(BlobRangeError { start, end }), + ) + }) +} + /// The error of a file that the blob storage does not hold. fn missing_file(path: &Path) -> Box { - RusticError::new( + RusticError::with_source( ErrorKind::Backend, "The blob storage holds no file at `{path}`.", + FileMissing { path: path.into() }, + ) + .attach_context("path", path.display().to_string()) +} + +/// The error of a config file that another writer made first. +fn config_exists(path: &Path) -> Box { + RusticError::with_source( + ErrorKind::Backend, + "The blob storage already holds the config file `{path}`.", + ConfigExists, + ) + .attach_context("path", path.display().to_string()) +} + +/// The error of a second snapshot file in one stage. +fn second_snapshot(path: &Path) -> Box { + RusticError::new( + ErrorKind::Internal, + "The stage already holds a snapshot file, so it cannot keep `{path}`.", ) .attach_context("path", path.display().to_string()) } @@ -347,11 +560,13 @@ fn storage_error(call: StorageCall, path: &Path, error: anyhow::Error) -> Box BytesList { /// Runs the calls on a new thread, which is not a thread of a runtime, and gives their result. /// `None` means that the calls did not end within the limit. fn within_limit(calls: impl FnOnce() -> T + Send + 'static) -> Option { + on_own_thread(calls).recv_timeout(LIMIT).ok() +} + +/// Starts the calls on a new thread, which is not a thread of a runtime. The receiver gets their +/// result, so a test can wait for it with a limit. +fn on_own_thread( + calls: impl FnOnce() -> T + Send + 'static, +) -> std::sync::mpsc::Receiver { let (sender, receiver) = std::sync::mpsc::channel(); std::thread::spawn(move || sender.send(calls())); - receiver.recv_timeout(LIMIT).ok() + receiver } #[test] @@ -459,18 +472,576 @@ fn a_thread_that_is_not_a_thread_of_the_runtime_can_call_the_backend() { let fixture = Fixture::new(); let backend = Arc::new(fixture.backend); - let read = std::thread::spawn({ + let read = within_limit({ let backend = backend.clone(); move || { backend .write_bytes(FileType::Index, &id("ab"), false, bytes("index")) .and_then(|()| backend.read_full(FileType::Index, &id("ab"))) + .ok() } - }) - .join() - .map(|read| read.ok()); + }); + + assert_eq!(read.flatten(), Some(Bytes::from_static(b"index"))); +} + +/// The content of the pack of the tests of the kept packs: 100 bytes, each its own offset. +fn pack_content() -> Vec { + (0..100).collect() +} + +/// A backend over a storage that holds one pack at the path of the id `ab` and one at the path of +/// the id `cd`, with the rule of the storage and the limit of the kept packs. The storage records +/// each call. +struct PackFixture { + _runtime: Runtime, + storage: Arc, + backend: Arc, +} + +impl PackFixture { + fn new(limit: usize, rule: impl Fn(&str, &Path) -> Script + Send + Sync + 'static) -> Self { + let runtime = Runtime::new().unwrap(); + let inner = Arc::new(InMemoryBlobStorage::new()); + let namespace = new_namespace(); + ["ab", "cd"].iter().for_each(|pack| { + runtime + .block_on(inner.put_raw( + "test", + "test", + namespace.clone(), + Path::new(&format!("data/{pack}/{}", pack.repeat(32))), + &pack_content(), + )) + .unwrap(); + }); + let storage = ScriptedBlobStorage::new(inner, rule); + let backend = BlobBackend::new( + storage.clone(), + namespace, + runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .keeping_packs_up_to(limit); + Self { + _runtime: runtime, + storage, + backend: Arc::new(backend), + } + } + + /// Gives the operation label of each call on the pack. + fn pack_calls(&self) -> Vec<&'static str> { + self.storage + .calls() + .into_iter() + .filter(|(_, path)| path.starts_with("data/")) + .map(|(op_label, _)| op_label) + .collect() + } +} + +/// Reads the range of the pack as a range of tree blobs, which rustic marks as cacheable. +fn tree_range(backend: &BlobBackend, offset: u32, length: u32) -> RusticResult { + backend.read_partial(FileType::Pack, &id("ab"), true, offset, length) +} + +#[test] +fn a_later_range_of_a_kept_pack_makes_no_storage_call() { + let fixture = PackFixture::new(1024, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + ( + tree_range(&backend, 0, 10).ok(), + tree_range(&backend, 20, 10).ok(), + tree_range(&backend, 90, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)), + Some(Bytes::from_iter(90..100)), + )), + vec!["read"] + ) + ); +} + +#[test] +fn a_range_that_is_not_cacheable_is_a_ranged_read_each_time() { + let fixture = PackFixture::new(1024, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + [(0, 10), (20, 10)].map(|(offset, length)| { + backend + .read_partial(FileType::Pack, &id("ab"), false, offset, length) + .ok() + }) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some([ + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)) + ]), + vec!["read_range", "read_range"] + ) + ); +} + +#[test] +fn two_threads_that_miss_one_pack_make_one_storage_read() { + // The first read waits at the gate. The gate opens only when the second thread waits for that + // read, or when the second thread reads the pack itself. + let fixture = PackFixture::new(1024, |op_label, _| { + if op_label == "read" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let first = on_own_thread({ + let backend = fixture.backend.clone(); + move || tree_range(&backend, 0, 10).ok() + }); + let first_read_started = (0..1000).any(|_| { + std::thread::sleep(Duration::from_millis(10)); + !fixture.pack_calls().is_empty() + }); + let second = on_own_thread({ + let backend = fixture.backend.clone(); + move || tree_range(&backend, 50, 10).ok() + }); + let second_asked = (0..1000).any(|_| { + std::thread::sleep(Duration::from_millis(10)); + fixture.backend.kept.waiters() == 1 || fixture.pack_calls().len() > 1 + }); + fixture.storage.open_gate(); + + assert_eq!( + ( + first_read_started, + second_asked, + first.recv_timeout(LIMIT).ok().flatten(), + second.recv_timeout(LIMIT).ok().flatten(), + fixture.pack_calls() + ), + ( + true, + true, + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(50..60)), + vec!["read"] + ) + ); +} + +#[test] +fn a_pack_over_the_limit_is_read_whole_one_time_and_its_next_range_is_a_ranged_read() { + let fixture = PackFixture::new(99, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + ( + tree_range(&backend, 0, 10).ok(), + tree_range(&backend, 20, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)) + )), + vec!["read", "read_range"] + ) + ); +} + +#[test] +fn a_pack_that_does_not_fit_below_the_limit_is_read_whole_one_time_and_then_by_its_ranges() { + // The first pack is kept, and its 100 bytes stay below the limit of 150. The second pack does + // not fit, so it is read whole one time and not kept. + let fixture = PackFixture::new(150, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + let second = |offset| { + backend + .read_partial(FileType::Pack, &id("cd"), true, offset, 10) + .ok() + }; + ( + tree_range(&backend, 0, 10).ok(), + second(20), + second(40), + second(60), + tree_range(&backend, 80, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)), + Some(Bytes::from_iter(40..50)), + Some(Bytes::from_iter(60..70)), + Some(Bytes::from_iter(80..90)) + )), + vec!["read", "read", "read_range", "read_range"] + ) + ); +} + +#[test] +fn once_the_kept_packs_fill_the_limit_a_range_of_another_pack_is_a_ranged_read() { + // The first pack fills the limit when it is kept. + let fixture = PackFixture::new(100, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + ( + tree_range(&backend, 0, 10).ok(), + backend + .read_partial(FileType::Pack, &id("cd"), true, 20, 10) + .ok(), + tree_range(&backend, 40, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)), + Some(Bytes::from_iter(40..50)) + )), + vec!["read", "read_range"] + ) + ); +} + +#[test] +fn a_failed_read_of_a_pack_is_not_kept_and_the_next_range_reads_again() { + let refused = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let fixture = PackFixture::new(1024, { + let refused = refused.clone(); + move |op_label, _| { + if op_label == "read" && !refused.swap(true, std::sync::atomic::Ordering::SeqCst) { + Script::Refuse + } else { + Script::Pass + } + } + }); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + ( + tree_range(&backend, 0, 10).is_err(), + tree_range(&backend, 20, 10).ok(), + tree_range(&backend, 40, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + true, + Some(Bytes::from_iter(20..30)), + Some(Bytes::from_iter(40..50)) + )), + vec!["read", "read"] + ) + ); +} + +#[test] +fn a_range_outside_a_kept_pack_gives_the_error_of_a_ranged_read_outside_a_blob() { + let fixture = PackFixture::new(1024, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let outside = within_limit(move || { + [(90, 11), (100, 1), (u32::MAX, 1)].map(|(offset, length)| { + tree_range(&backend, offset, length).err().map(|error| { + ( + text_of(&error).contains("is not in the blob"), + classify(Operation::Restore, anyhow::Error::new(error)) + .to_string() + .contains("storage"), + ) + }) + }) + }); + + assert_eq!(outside, Some([Some((true, true)); 3])); +} + +#[test] +fn a_tracked_backend_counts_in_its_tracker_until_it_drops() { + let fixture = Fixture::new(); + let tracker = tokio_util::task::TaskTracker::new(); + let backend = BlobBackend::new( + fixture.storage.clone(), + fixture.namespace.clone(), + fixture.runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .tracked_by(tracker.token()); + + let while_alive = tracker.len(); + drop(backend); + + assert_eq!((while_alive, tracker.len()), (1, 0)); +} + +#[test] +fn a_cancelled_backend_makes_no_storage_call() { + let runtime = Runtime::new().unwrap(); + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let cancel = CancellationToken::new(); + let backend = BlobBackend::new( + storage.clone(), + new_namespace(), + runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .cancelled_by(cancel.clone()); + cancel.cancel(); + + let cancelled = [ + backend.list_with_size(FileType::Config).err(), + backend.list_with_size(FileType::Pack).err(), + backend.read_full(FileType::Pack, &id("ab")).err(), + backend + .read_partial(FileType::Pack, &id("ab"), false, 0, 1) + .err(), + backend + .write_bytes(FileType::Pack, &id("ab"), false, bytes("pack")) + .err(), + backend.remove(FileType::Pack, &id("ab"), false).err(), + ] + .map(|error| error.is_some_and(|error| was_cancelled(&error))); + + assert_eq!((cancelled, storage.calls()), ([true; 6], Vec::new())); +} + +#[test] +fn a_marker_write_that_starts_at_the_end_of_the_lease_does_not_move_it_and_one_that_starts_before_does() + { + let end = Instant::now() + Duration::from_secs(60); + let span = Duration::from_secs(10); + let at_the_end = Lease::until(end); + let just_before = Lease::until(end); + + at_the_end.extend_from(end, span); + just_before.extend_from(end - Duration::from_nanos(1), span); + + assert_eq!( + (at_the_end.expiry(), just_before.expiry()), + (end, end - Duration::from_nanos(1) + span) + ); +} + +#[test] +fn a_cancel_ends_a_call_that_runs() { + let runtime = Runtime::new().unwrap(); + let held = CancellationToken::new(); + let (storage, _gate, _dropped) = holding_storage(Arc::new(InMemoryBlobStorage::new()), { + let held = held.clone(); + move |_, _| { + held.cancel(); + true + } + }); + let cancel = CancellationToken::new(); + let backend = BlobBackend::new( + storage, + new_namespace(), + runtime.handle().clone(), + Duration::from_secs(60), + ) + .cancelled_by(cancel.clone()); + runtime.spawn(async move { + held.cancelled().await; + cancel.cancel(); + }); + + let outcome = within_limit(move || { + backend + .read_full(FileType::Pack, &id("ab")) + .err() + .map(|error| (was_cancelled(&error), reached_deadline(&*error))) + }); + + assert_eq!(outcome, Some(Some((true, false)))); +} + +#[test] +fn a_backend_whose_token_is_not_cancelled_answers() { + let fixture = Fixture::new(); + let backend = BlobBackend::new( + fixture.storage.clone(), + fixture.namespace.clone(), + fixture.runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .cancelled_by(CancellationToken::new()); + + let read = backend + .write_bytes(FileType::Pack, &id("ab"), false, bytes("pack")) + .and_then(|()| backend.read_full(FileType::Pack, &id("ab"))); + + assert_eq!(read.ok(), Some(Bytes::from_static(b"pack"))); +} + +#[test] +fn the_config_file_is_written_only_when_the_repository_has_none() { + let fixture = Fixture::new(); + + let first = + fixture + .backend + .write_bytes(FileType::Config, &Id::default(), false, bytes("first")); + let second = + fixture + .backend + .write_bytes(FileType::Config, &Id::default(), false, bytes("second")); + + assert_eq!( + ( + first.is_ok(), + second + .as_ref() + .is_err_and(|error| is_config_exists(&**error)), + fixture.stored(), + ), + (true, true, vec![("config".to_string(), 5)]) + ); +} + +#[test] +fn an_index_file_that_is_there_is_kept_and_its_write_succeeds() { + let fixture = Fixture::new(); + let path = format!("index/{}", "ab".repeat(32)); + fixture.put(&path, b"kept"); + + let written = + fixture + .backend + .write_bytes(FileType::Index, &id("ab"), false, bytes("replacement")); + let read = fixture.backend.read_full(FileType::Index, &id("ab")); + + assert_eq!( + (written.is_ok(), read.ok()), + (true, Some(Bytes::from_static(b"kept"))) + ); +} + +#[test] +fn a_pack_file_that_is_there_is_written_again() { + let fixture = Fixture::new(); + fixture.put(&format!("data/ab/{}", "ab".repeat(32)), b"old"); + + let written = fixture + .backend + .write_bytes(FileType::Pack, &id("ab"), false, bytes("new")); + let read = fixture.backend.read_full(FileType::Pack, &id("ab")); - assert_eq!(read.ok().flatten(), Some(Bytes::from_static(b"index"))); + assert_eq!( + (written.is_ok(), read.ok()), + (true, Some(Bytes::from_static(b"new"))) + ); +} + +#[test] +fn a_backend_with_a_stage_keeps_the_snapshot_file_and_does_not_write_it() { + let fixture = Fixture::new(); + let stage = Arc::new(SnapshotStage::default()); + let backend = BlobBackend::new( + fixture.storage.clone(), + fixture.namespace.clone(), + fixture.runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .staging_in(stage.clone()); + let content = [Bytes::from_static(b"snap"), Bytes::from_static(b"shot")] + .into_iter() + .fold(BytesList::default(), |mut content, part| { + content.add(part); + content + }); + + let kept = backend.write_bytes(FileType::Snapshot, &id("cd"), false, content); + let second = backend.write_bytes(FileType::Snapshot, &id("ef"), false, bytes("other")); + let pack = backend.write_bytes(FileType::Pack, &id("ab"), false, bytes("pack")); + + assert_eq!( + ( + kept.is_ok(), + second.is_err(), + pack.is_ok(), + stage.take(), + fixture.stored(), + ), + ( + true, + true, + true, + Some(StagedSnapshot { + path: Arc::from(PathBuf::from(format!("snapshots/{}", "cd".repeat(32)))), + content: Bytes::from_static(b"snapshot"), + }), + vec![(format!("data/ab/{}", "ab".repeat(32)), 4)] + ) + ); +} + +#[test] +fn each_failed_call_is_a_storage_failure_to_the_classification() { + let runtime = Runtime::new().unwrap(); + let backend = BlobBackend::new( + Arc::new(FailingBlobStorage), + new_namespace(), + runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ); + + let classified = [ + backend.list_with_size(FileType::Pack).err(), + backend.read_full(FileType::Pack, &id("ab")).err(), + backend + .write_bytes(FileType::Pack, &id("ab"), false, bytes("pack")) + .err(), + ] + .map(|error| { + error.map(|error| { + matches!( + classify(Operation::Restore, anyhow::Error::new(error)), + SnapshotStoreError::Storage { + retryable: true, + .. + } + ) + }) + }); + + assert_eq!(classified, [Some(true); 3]); } #[test] @@ -494,14 +1065,18 @@ fn error_text(result: RusticResult) -> String { } } -/// Gives the text of the error, with the text of its source. +/// Gives the text of the error, with the text of each error in its chain of sources. fn text_of(error: &RusticError) -> String { - format!( - "{error} {}", - std::error::Error::source(error) - .map(ToString::to_string) - .unwrap_or_default() - ) + std::iter::successors(std::error::Error::source(error), |error| error.source()) + .fold(error.to_string(), |text, source| format!("{text} {source}")) +} + +/// Tells whether the error or an error in its chain of sources is [`OperationCancelled`]. +fn was_cancelled(error: &RusticError) -> bool { + std::iter::successors(Some(error as &(dyn std::error::Error + 'static)), |error| { + error.source() + }) + .any(|error| error.is::()) } /// A blob storage that fails every call. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs new file mode 100644 index 0000000000..5f0ad956c6 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs @@ -0,0 +1,484 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The errors that the backend puts into the chain of a rustic error, and the classification of a +//! failed operation as a [`SnapshotStoreError`]. +//! +//! rustic gives no public kind of an error. So the classification reads the chain of sources: the +//! markers of this module, the name errors of the blob storage, and the I/O errors. + +use super::prune::SNAPSHOTS_PATH; +use crate::filesystem_snapshot::SnapshotStoreError; +use golem_service_base::storage::blob::BlobNameError; +use rustic_core::FileType; +use std::error::Error; +use std::fmt::{Display, Formatter}; +use std::path::Path; + +/// A blob storage call of the backend that failed, got no answer within its deadline, or did not +/// start because its operation was cancelled. The source is the failure. +#[derive(Debug)] +pub(super) struct BlobCallFailed { + failure: anyhow::Error, +} + +impl BlobCallFailed { + pub(super) fn new(failure: anyhow::Error) -> Self { + Self { failure } + } +} + +impl Display for BlobCallFailed { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("the blob storage call gave an error") + } +} + +impl Error for BlobCallFailed { + fn source(&self) -> Option<&(dyn Error + 'static)> { + Some(self.failure.as_ref()) + } +} + +/// The operation of the backend was cancelled, so the backend made no more calls. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct OperationCancelled; + +impl Display for OperationCancelled { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("the filesystem snapshot operation was cancelled") + } +} + +impl Error for OperationCancelled {} + +/// The lease of the prune ran out, so the backend made no more calls. Another delete can then take +/// the claim of the prune, so the prune must not change the repository any more. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct LeaseExpired; + +impl Display for LeaseExpired { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("the lease of the prune claim ran out") + } +} + +impl Error for LeaseExpired {} + +/// Tells whether an error in the chain is [`LeaseExpired`]. Only tests ask this, because the store +/// gives a lease that ran out as a retryable storage error, the same as a failed call. +#[cfg(test)] +pub(super) fn is_lease_expired(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| error.is::()) +} + +/// Another writer made the config file of the repository first. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct ConfigExists; + +impl Display for ConfigExists { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("another writer made the config file of the repository first") + } +} + +impl Error for ConfigExists {} + +/// The blob storage holds no file at the path that rustic reads, for example because a delete +/// removed it after a listing. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(super) struct FileMissing { + /// The path of the file, relative to the root of the repository. + pub(super) path: Box, +} + +impl Display for FileMissing { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + write!( + formatter, + "the blob storage holds no file at {}", + self.path.display() + ) + } +} + +impl Error for FileMissing {} + +/// Tells whether an error in the chain is [`FileMissing`]. +pub(super) fn is_file_missing(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| error.is::()) +} + +/// Tells whether an error in the chain is [`FileMissing`] for a snapshot file. +pub(super) fn is_snapshot_missing(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| { + error + .downcast_ref::() + .is_some_and(|missing| missing.path.starts_with(SNAPSHOTS_PATH)) + }) +} + +/// Tells whether an error in the chain is [`FileMissing`] for an index file. A prune writes its new +/// index files and then deletes the old ones at once, so an operation that listed an index file +/// before a prune can find it gone at its read. A later try lists the new index files. +pub(super) fn is_index_missing(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| { + error + .downcast_ref::() + .is_some_and(|missing| missing.path.starts_with(FileType::Index.dirname())) + }) +} + +/// Tells whether an error in the chain is [`ConfigExists`]. +pub(super) fn is_config_exists(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| error.is::()) +} + +/// The kind of operation that failed. It tells where an I/O error came from. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum Operation { + /// Reads a tree from the local filesystem and writes it into the repository. + Save, + /// Reads the repository and writes a tree into the local filesystem. + Restore, + /// Reads or changes only the repository. + Repository, + /// Removes the data that no snapshot uses. + Prune, +} + +/// A failed storage call gives `Storage`, retryable unless a name error caused it. An index file +/// that a prune deleted after the listing gives retryable `Storage` in each operation. An I/O error +/// gives `Source` in a save and `Destination` in a restore. Each other error gives `Storage` that +/// is not retryable in a save or a prune, and `Corrupt` in a restore or a read of the repository. +/// So a pack that is gone stays `Corrupt`, because a prune deletes a pack only after the grace +/// period. +pub(super) fn classify(operation: Operation, error: anyhow::Error) -> SnapshotStoreError { + let from_storage = is_storage_failure(error.as_ref()) || is_index_missing(error.as_ref()); + let io_kind = chain(error.as_ref()) + .find_map(|error| error.downcast_ref::()) + .map(std::io::Error::kind); + let permanent = chain(error.as_ref()).any(|error| error.is::()); + match (from_storage, io_kind, operation) { + (true, _, _) => SnapshotStoreError::Storage { + retryable: !permanent, + source: error, + }, + (false, Some(kind), Operation::Save) => { + SnapshotStoreError::Source(std::io::Error::new(kind, error_text(&error))) + } + (false, Some(kind), Operation::Restore) => { + SnapshotStoreError::Destination(std::io::Error::new(kind, error_text(&error))) + } + (false, _, Operation::Save | Operation::Prune) => SnapshotStoreError::Storage { + retryable: false, + source: error, + }, + (false, _, Operation::Restore | Operation::Repository) => { + SnapshotStoreError::Corrupt(error) + } + } +} + +/// Tells whether an error in the chain is a failed blob storage call. +pub(super) fn is_storage_failure(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| error.is::()) +} + +/// Gives the error of the store for a blob storage call that the store made without rustic. It is +/// retryable unless a name error of the blob storage caused it. +pub(super) fn storage_failure(error: anyhow::Error) -> SnapshotStoreError { + classify( + Operation::Repository, + anyhow::Error::new(BlobCallFailed::new(error)), + ) +} + +/// Gives the text of the error with the text of each of its sources. +fn error_text(error: &anyhow::Error) -> String { + format!("{error:#}") +} + +/// Gives the error and each error in its chain of sources. +fn chain<'a>(error: &'a (dyn Error + 'static)) -> impl Iterator { + std::iter::successors(Some(error), |&error| error.source()) +} + +#[cfg(test)] +mod tests { + use super::{ + BlobCallFailed, ConfigExists, FileMissing, Operation, OperationCancelled, classify, + is_config_exists, + }; + use crate::filesystem_snapshot::SnapshotStoreError; + use golem_service_base::storage::blob::BlobNameError; + use pretty_assertions::assert_eq; + use rustic_core::{ErrorKind, RusticError}; + use std::io; + use test_r::test; + + /// Gives a rustic error whose source is the error, as rustic gives it to the store. + fn rustic(source: impl std::error::Error + Send + Sync + 'static) -> anyhow::Error { + anyhow::Error::new(RusticError::with_source( + ErrorKind::Backend, + "the operation failed", + source, + )) + } + + fn failed_call(failure: anyhow::Error) -> anyhow::Error { + rustic(BlobCallFailed::new(failure)) + } + + /// Gives the variant of the error, whether it is retryable, and the kind of its I/O error. + fn shape(error: &SnapshotStoreError) -> (&'static str, Option, Option) { + match error { + SnapshotStoreError::NotFound => ("NotFound", None, None), + SnapshotStoreError::AlreadyExists => ("AlreadyExists", None, None), + SnapshotStoreError::Source(error) => ("Source", None, Some(error.kind())), + SnapshotStoreError::Destination(error) => ("Destination", None, Some(error.kind())), + SnapshotStoreError::Storage { retryable, .. } => ("Storage", Some(*retryable), None), + SnapshotStoreError::Corrupt(_) => ("Corrupt", None, None), + } + } + + #[test] + fn a_failed_blob_storage_call_gives_a_retryable_storage_error_in_each_operation() { + let shapes = + [Operation::Save, Operation::Restore, Operation::Repository].map(|operation| { + shape(&classify( + operation, + failed_call(anyhow::anyhow!("the bucket is gone")), + )) + }); + + assert_eq!(shapes, [("Storage", Some(true), None); 3]); + } + + #[test] + fn a_blob_storage_call_that_holds_an_io_error_is_still_a_storage_error() { + let failure = anyhow::Error::new(io::Error::new(io::ErrorKind::StorageFull, "no space")); + + assert_eq!( + shape(&classify(Operation::Restore, failed_call(failure))), + ("Storage", Some(true), None) + ); + } + + #[test] + fn an_index_file_that_is_gone_gives_retryable_storage_and_a_pack_that_is_gone_does_not() { + // The backend gives a missing file as a rustic error of the kind `Backend` whose source is + // `FileMissing`. The index load of rustic passes that error on as it is, through the read + // of the file and the stream of all index files. + let missing = |path: &str| { + rustic(FileMissing { + path: std::path::Path::new(path).into(), + }) + }; + let operations = [ + Operation::Save, + Operation::Restore, + Operation::Repository, + Operation::Prune, + ]; + + assert_eq!( + ( + operations.map(|operation| shape(&classify(operation, missing("index/ab12")))), + operations.map(|operation| shape(&classify(operation, missing("data/ab/ab12")))), + ), + ( + [("Storage", Some(true), None); 4], + [ + ("Storage", Some(false), None), + ("Corrupt", None, None), + ("Corrupt", None, None), + ("Storage", Some(false), None), + ] + ) + ); + } + + #[test] + fn a_name_error_of_the_blob_storage_is_not_retryable() { + let failure = anyhow::Error::new(BlobNameError::NoName { + path: std::path::PathBuf::new(), + }); + + assert_eq!( + shape(&classify(Operation::Repository, failed_call(failure))), + ("Storage", Some(false), None) + ); + } + + #[test] + fn a_cancelled_operation_gives_a_retryable_storage_error_in_each_operation() { + let shapes = + [Operation::Save, Operation::Restore, Operation::Repository].map(|operation| { + shape(&classify( + operation, + failed_call(anyhow::Error::new(OperationCancelled)), + )) + }); + + assert_eq!(shapes, [("Storage", Some(true), None); 3]); + } + + #[test] + fn an_io_error_of_a_save_gives_source_with_its_kind() { + let error = rustic(io::Error::new(io::ErrorKind::PermissionDenied, "locked")); + + let classified = classify(Operation::Save, error); + + assert_eq!( + ( + shape(&classified), + classified.to_string().contains("locked") + ), + ( + ("Source", None, Some(io::ErrorKind::PermissionDenied)), + true + ) + ); + } + + #[test] + fn a_full_volume_during_a_restore_gives_destination() { + let error = rustic(io::Error::new(io::ErrorKind::StorageFull, "no space")); + + assert_eq!( + shape(&classify(Operation::Restore, error)), + ("Destination", None, Some(io::ErrorKind::StorageFull)) + ); + } + + #[test] + fn a_metadata_error_of_a_restore_gives_destination() { + let error = rustic(io::Error::new( + io::ErrorKind::PermissionDenied, + "setting extended attributes failed", + )); + + assert_eq!( + shape(&classify(Operation::Restore, error)), + ("Destination", None, Some(io::ErrorKind::PermissionDenied)) + ); + } + + #[test] + fn an_error_without_storage_or_io_is_corrupt_when_it_reads_and_not_retryable_in_a_save() { + let refused = || { + anyhow::Error::new(RusticError::new( + ErrorKind::Cryptography, + "the data failed its check", + )) + }; + + assert_eq!( + [ + shape(&classify(Operation::Restore, refused())), + shape(&classify(Operation::Repository, refused())), + shape(&classify(Operation::Save, refused())), + ], + [ + ("Corrupt", None, None), + ("Corrupt", None, None), + ("Storage", Some(false), None), + ] + ); + } + + /// The error below the rustic error of a restore that cannot set an extended attribute, as + /// the fork gives it on ext4 for a user attribute of 6,000 bytes. + #[derive(Debug)] + struct SettingXattrFailed(io::Error); + + impl std::fmt::Display for SettingXattrFailed { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + formatter, + "setting xattr `user.golem-test` on `\"/restore/file.txt\"` with `{:?}`", + self.0 + ) + } + } + + impl std::error::Error for SettingXattrFailed { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(&self.0) + } + } + + #[test] + fn the_real_chain_of_a_failed_extended_attribute_of_a_restore_gives_destination() { + let error = anyhow::Error::new(RusticError::with_source( + ErrorKind::InputOutput, + "The restore cannot set the extended attributes of `file.txt`.", + SettingXattrFailed(io::Error::from_raw_os_error(28)), + )); + + assert_eq!( + shape(&classify(Operation::Restore, error)), + ("Destination", None, Some(io::ErrorKind::StorageFull)) + ); + } + + #[test] + fn a_prune_error_keeps_the_retryable_flag_of_a_storage_failure_and_is_never_corrupt() { + let refused = anyhow::Error::new(RusticError::new( + ErrorKind::Internal, + "the pack has another size than the index says", + )); + + assert_eq!( + [ + shape(&classify(Operation::Prune, refused)), + shape(&classify( + Operation::Prune, + rustic(io::Error::new(io::ErrorKind::InvalidData, "bad pack")) + )), + shape(&classify( + Operation::Prune, + failed_call(anyhow::anyhow!("the bucket is gone")) + )), + shape(&classify( + Operation::Prune, + failed_call(anyhow::Error::new(BlobNameError::NoName { + path: std::path::PathBuf::new(), + })) + )), + ], + [ + ("Storage", Some(false), None), + ("Storage", Some(false), None), + ("Storage", Some(true), None), + ("Storage", Some(false), None), + ] + ); + } + + #[test] + fn the_config_marker_is_found_in_the_chain() { + let exists = failed_call(anyhow::Error::new(ConfigExists)); + let other = failed_call(anyhow::anyhow!("the bucket is gone")); + + assert_eq!( + ( + is_config_exists(exists.as_ref()), + is_config_exists(other.as_ref()) + ), + (true, false) + ); + } +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs new file mode 100644 index 0000000000..a30d63da75 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs @@ -0,0 +1,155 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The blobs of the repository of one scope, and the blob storage calls of the store on them. +//! +//! Each call waits for at most the deadline of the scope, and a cancel of its operation ends it. + +use super::backend::answer_or_cancel; +use golem_service_base::storage::blob::{ + BlobStorage, BlobStorageNamespace, ListedBlob, PutIfAbsent, +}; +use std::path::{Path, PathBuf}; +use std::sync::Arc; +use std::time::Duration; +use tokio_util::sync::CancellationToken; +use tokio_util::task::TaskTracker; + +/// The target label of each blob storage call of the rustic store. +pub(super) const TARGET_LABEL: &str = "filesystem_snapshot"; + +/// The blobs of one scope: the storage, the namespace of the scope, the deadline of each call, the +/// token of the operation, and the tracker of the store, which counts each call. +#[derive(Clone, Debug)] +pub(super) struct SnapshotFiles { + pub(super) storage: Arc, + pub(super) namespace: BlobStorageNamespace, + pub(super) deadline: Duration, + pub(super) cancel: CancellationToken, + pub(super) tracker: TaskTracker, +} + +impl SnapshotFiles { + /// Waits for one call within the deadline. A call of a cancelled operation does not start, and + /// a cancel ends a running call. Both give an error. The tracker counts the call before the + /// check of the cancel, so a shut down either stops the call or waits for it. + async fn answer( + &self, + future: impl Future>, + ) -> anyhow::Result { + self.tracker + .track_future(answer_or_cancel(self.deadline, &self.cancel, future)) + .await + } + + /// Gives the content of the blob at the path, or `None` when the path has no blob. + pub(super) async fn get( + &self, + op_label: &'static str, + path: &Path, + ) -> anyhow::Result>> { + self.answer( + self.storage + .get_raw(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await + } + + /// Writes the content as the blob at the path, over the blob that was there. + pub(super) async fn put( + &self, + op_label: &'static str, + path: &Path, + content: &[u8], + ) -> anyhow::Result<()> { + self.answer(self.storage.put_raw( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + content, + )) + .await + } + + /// Writes the content as the blob at the path only when the path has no blob. + pub(super) async fn put_if_absent( + &self, + op_label: &'static str, + path: &Path, + content: &[u8], + ) -> anyhow::Result { + self.answer(self.storage.put_raw_if_absent( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + content, + )) + .await + } + + /// Deletes the blob at the path. A path without a blob gives success. + pub(super) async fn delete(&self, op_label: &'static str, path: &Path) -> anyhow::Result<()> { + self.answer( + self.storage + .delete(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await + } + + /// Deletes the directory at the path and each blob below it. + pub(super) async fn delete_dir( + &self, + op_label: &'static str, + path: &Path, + ) -> anyhow::Result { + self.answer( + self.storage + .delete_dir(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await + } + + /// Gives each blob directly below the path, and each directory that the storage keeps an + /// entry for below the path. + pub(super) async fn list_dir( + &self, + op_label: &'static str, + path: &Path, + ) -> anyhow::Result]>> { + let listed = self + .answer( + self.storage + .list_dir(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await?; + Ok(listed.into_iter().map(PathBuf::into_boxed_path).collect()) + } + + /// Gives each blob below the path, at all depths, with its size. + pub(super) async fn list_below( + &self, + op_label: &'static str, + path: &Path, + ) -> anyhow::Result> { + self.answer(self.storage.list_blobs_below( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + )) + .await + } +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index bc9b4a00f6..792181a44f 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -19,9 +19,16 @@ //! module. mod backend; +mod fault; +mod files; +mod priority; +mod prune; +mod publish; +mod scope; +mod store; + +pub(crate) use store::RusticSnapshotStore; -#[cfg(test)] -mod holding; #[cfg(test)] mod tests; @@ -47,15 +54,6 @@ use std::sync::Arc; use std::time::{Duration, Instant}; use tokio::runtime::Handle; -/// The longest time that one call of a repository waits for the blob storage. -/// -/// A call that gets no answer within this time fails, and its operation fails with it. The value -/// stops a call that does not return. It is not a limit for a slow call. On S3, with the retries of -/// the S3 storage, a write of a pack took at most 1.7 s with eight saves at the same time. A ranged -/// read of a pack took at most 1.5 s under the CPU request of an executor. Keep the value at least -/// 10 times the longest measured call. -pub(super) const STORAGE_CALL_DEADLINE: Duration = Duration::from_secs(30); - /// The key that encrypts a repository. /// /// The key has 64 bytes: 32 bytes of the AES-256 key, then 16 bytes of the number `k` and 16 @@ -291,6 +289,8 @@ pub(super) struct PruneReport { pub(super) marked_bytes_deleted: u64, /// The packs that an earlier prune marked and that stay marked. pub(super) marked_packs_kept: u64, + /// The packs that no index lists. The prune marks each of them. + pub(super) packs_unindexed: u64, /// The bytes of the blobs that snapshots use. pub(super) bytes_used: u64, /// The bytes of the blobs that no snapshot uses. @@ -511,29 +511,51 @@ fn restore( let Some(snapshot) = snapshot else { return Ok(None); }; + let restored = restore_snapshot( + repository, + &snapshot, + into, + &RestoreOptions::default().reader_threads(reader_threads), + )?; + Ok(Some(RestoreReport { + phases: [open, lookup] + .into_iter() + .chain(restored.phases.iter().copied()) + .collect(), + ..restored + })) +} + +/// Writes the tree of the snapshot into the empty directory `into`. The phases of the result are +/// the index load, the plan and the writes. +fn restore_snapshot( + repository: RusticRepository, + snapshot: &SnapshotFile, + into: &Path, + options: &RestoreOptions, +) -> anyhow::Result { let (repository, index) = timed(OperationPhase::IndexLoad, || repository.to_indexed())?; let into = into .to_str() .context("the directory of a restore must have a UTF-8 path")?; let destination = LocalDestination::new(into, false, false)?; - let node = repository.node_from_snapshot_and_path(&snapshot, "")?; + let node = repository.node_from_snapshot_and_path(snapshot, "")?; let entries = repository.ls(&node, &LsOptions::default())?; - let options = RestoreOptions::default().reader_threads(reader_threads); let (plan, planning) = timed(OperationPhase::RestorePlan, || { - repository.prepare_restore(&options, entries.clone(), &destination, false) + repository.prepare_restore(options, entries.clone(), &destination, false) })?; let files = plan.stats.files.restore; let dirs = plan.stats.dirs.restore; let bytes = plan.restore_size; let ((), writing) = timed(OperationPhase::Restore, || { - repository.restore(plan, &options, entries, &destination) + repository.restore(plan, options, entries, &destination) })?; - Ok(Some(RestoreReport { + Ok(RestoreReport { files, dirs, bytes, - phases: Box::new([open, lookup, index, planning, writing]), - })) + phases: Box::new([index, planning, writing]), + }) } fn prune( @@ -603,24 +625,31 @@ fn forget( } /// Opens the repository, or makes it when the scope has none. +/// +/// When another writer makes the config file of the repository first, the call opens the +/// repository of that writer. fn open_or_create( backend: Arc, key: &RepositoryKey, settings: &RepositorySettings, ) -> RusticResult<(RusticRepository, OperationPhase)> { - let repository = unopened(backend)?; + let repository = unopened(backend.clone())?; let credentials = Credentials::Masterkey(key.master_key()); match repository.config_id()? { Some(_) => repository .open(&credentials) .map(|repository| (repository, OperationPhase::Open)), - None => repository - .init( - &credentials, - &KeyOptions::default(), - &config_options(settings), - ) - .map(|repository| (repository, OperationPhase::Create)), + None => match repository.init( + &credentials, + &KeyOptions::default(), + &config_options(settings), + ) { + Ok(repository) => Ok((repository, OperationPhase::Create)), + Err(error) if fault::is_config_exists(&*error) => unopened(backend)? + .open(&credentials) + .map(|repository| (repository, OperationPhase::Open)), + Err(error) => Err(error), + }, } } @@ -741,6 +770,7 @@ fn prune_report(stats: &PruneStats) -> PruneReport { marked_packs_deleted: stats.packs_to_delete.remove, marked_bytes_deleted: stats.size_to_delete.remove, marked_packs_kept: stats.packs_to_delete.keep, + packs_unindexed: stats.packs_unref, bytes_used: blobs.used, bytes_unused: blobs.unused, bytes_removed: blobs.remove, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/mod.rs new file mode 100644 index 0000000000..5ce29dce45 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/mod.rs @@ -0,0 +1,181 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Saves and prunes run at a low CPU priority, so the work of the agents comes first. +//! +//! A thread without privilege cannot raise its priority again after it lowers it. So the work runs +//! on a new thread, with a rayon pool of its own, and both end with the work. + +use rayon::{ThreadPool, ThreadPoolBuildError, ThreadPoolBuilder}; +use std::num::NonZeroUsize; +use std::sync::{Arc, Mutex, PoisonError}; +use tracing::warn; + +/// The nice value of the threads of a save or a prune. +pub(super) const LOW_PRIORITY: i32 = 19; + +/// How the store runs work at a low priority: the thread count of the rayon pool of the work, and +/// the two steps that a test can replace. +#[derive(Clone, Copy, Debug)] +pub(super) struct LowPriority { + /// The threads of the rayon pool of the work. `None` is the default count of rayon. + pub(super) threads: Option, + /// Gives the calling thread the low priority. + pub(super) lower: fn() -> std::io::Result<()>, + /// Builds the rayon pool of the work, with the name and the thread count. + pub(super) build_pool: + fn(&'static str, Option) -> Result, +} + +impl LowPriority { + /// Gives the steps of the platform, with a rayon pool of `threads` threads. + pub(super) fn new(threads: Option) -> Self { + Self { + threads, + lower: lower_own_priority, + build_pool, + } + } + + /// Runs the work at nice 19 on a new thread with the name, inside a new rayon pool, and waits + /// for it. The threads that the work starts get the same nice value. On a platform other than + /// Linux the priority stays as it is, and the work still runs on its own thread in its own pool. + pub(super) fn run( + self, + name: &'static str, + work: impl FnOnce() -> anyhow::Result + Send + 'static, + ) -> anyhow::Result { + self.on_own_thread(name, work) + } + + /// Runs the work on a new thread that first lowers its priority and builds the pool. A failure + /// of a step gives a warning, and the work runs without that step. + fn on_own_thread( + self, + name: &'static str, + work: impl FnOnce() -> anyhow::Result + Send + 'static, + ) -> anyhow::Result { + // The work waits in a slot, so the calling thread can still run it when no thread starts. + let slot = Arc::new(Mutex::new(Some(work))); + let spawned = std::thread::Builder::new().name(name.to_string()).spawn({ + let slot = slot.clone(); + move || { + if let Err(error) = (self.lower)() { + warn!( + error = %error, + "Failed to lower the CPU priority of filesystem snapshot work, so it runs at the normal priority" + ); + } + self.in_own_pool(name, || run_taken(&slot)) + } + }); + match spawned { + Ok(thread) => thread.join().map_err(|_| { + anyhow::anyhow!("the thread of the filesystem snapshot work panicked") + })?, + Err(error) => { + warn!( + error = %error, + "Failed to start a thread for filesystem snapshot work, so it runs at the normal priority" + ); + run_taken(&slot) + } + } + } +} + +impl LowPriority { + /// Runs the work at the normal priority on the calling thread, inside a new rayon pool with + /// the name, and gives its result. So the rayon work of one operation does not wait for the + /// rayon work of another operation on the global pool. + pub(super) fn run_at_normal_priority( + self, + name: &'static str, + work: impl FnOnce() -> anyhow::Result + Send, + ) -> anyhow::Result { + self.in_own_pool(name, work) + } + + /// Runs the work inside a new rayon pool. A pool that does not build gives a warning, and the + /// work runs without it. + fn in_own_pool( + self, + name: &'static str, + work: impl FnOnce() -> anyhow::Result + Send, + ) -> anyhow::Result { + match (self.build_pool)(name, self.threads) { + Ok(pool) => pool.install(work), + Err(error) => { + warn!( + error = %error, + "Failed to build the thread pool of filesystem snapshot work, so its parallel parts use the global pool" + ); + work() + } + } + } +} + +/// Takes the work out of the slot and runs it. +fn run_taken anyhow::Result>(slot: &Mutex>) -> anyhow::Result { + let work = slot.lock().unwrap_or_else(PoisonError::into_inner).take(); + work.map_or_else( + || Err(anyhow::anyhow!("the work already ran")), + |work| work(), + ) +} + +/// Builds a rayon pool whose threads get the nice value of the calling thread. +fn build_pool( + name: &'static str, + threads: Option, +) -> Result { + ThreadPoolBuilder::new() + .num_threads(threads.map_or(0, NonZeroUsize::get)) + .thread_name(move |index| format!("{name}-{index}")) + .build() +} + +/// Gives the calling thread the nice value 19. +#[cfg(target_os = "linux")] +fn lower_own_priority() -> std::io::Result<()> { + // SAFETY: `gettid` has no preconditions. + let thread = unsafe { libc::gettid() }; + let thread = libc::id_t::try_from(thread).map_err(std::io::Error::other)?; + // SAFETY: `setpriority` only reads its arguments. + let result = unsafe { libc::setpriority(libc::PRIO_PROCESS, thread, LOW_PRIORITY) }; + if result == 0 { + Ok(()) + } else { + Err(std::io::Error::last_os_error()) + } +} + +/// Keeps the priority of the calling thread on a platform other than Linux. +#[cfg(not(target_os = "linux"))] +fn lower_own_priority() -> std::io::Result<()> { + Ok(()) +} + +/// Gives the nice value of the calling thread. +#[cfg(all(test, target_os = "linux"))] +pub(super) fn own_nice() -> i32 { + // SAFETY: `gettid` has no preconditions. + let thread = libc::id_t::try_from(unsafe { libc::gettid() }).unwrap(); + // SAFETY: `getpriority` only reads its arguments. + unsafe { libc::getpriority(libc::PRIO_PROCESS, thread) } +} + +#[cfg(all(test, target_os = "linux"))] +mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs new file mode 100644 index 0000000000..0533a1e200 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs @@ -0,0 +1,136 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use super::{LOW_PRIORITY, LowPriority, lower_own_priority, own_nice}; +use pretty_assertions::assert_eq; +use rayon::{ThreadPool, ThreadPoolBuildError, ThreadPoolBuilder}; +use std::num::NonZeroUsize; +use test_r::test; + +/// What the work sees: its nice value, its thread name, whether it runs in a rayon pool, and +/// the thread count of the current rayon pool. +fn seen() -> anyhow::Result<(i32, Option, bool, usize)> { + Ok(( + own_nice(), + std::thread::current().name().map(str::to_string), + rayon::current_thread_index().is_some(), + rayon::current_num_threads(), + )) +} + +/// A pool builder that cannot start a thread. +fn no_pool(_: &str, _: Option) -> Result { + ThreadPoolBuilder::new() + .num_threads(1) + .spawn_handler(|_| Err(std::io::Error::other("no thread can start here"))) + .build() +} + +#[test] +fn work_at_low_priority_runs_at_nice_19_in_a_pool_of_its_own_and_the_caller_keeps_its_priority() { + let before = own_nice(); + + let inside = LowPriority::new(NonZeroUsize::new(3)).run("fs-snap-test", seen); + + assert_eq!( + ( + inside.ok().map(|(nice, name, in_pool, threads)| ( + nice, + name.is_some_and(|name| name.starts_with("fs-snap-test-")), + in_pool, + threads + )), + own_nice() + ), + (Some((LOW_PRIORITY, true, true, 3)), before) + ); +} + +#[test] +fn work_whose_priority_step_does_nothing_runs_on_its_own_thread_in_a_pool_of_its_own() { + // Off Linux, the step that lowers the priority does nothing. The rest of the path is the same + // on each platform. + let before = own_nice(); + let low_priority = LowPriority { + lower: || Ok(()), + ..LowPriority::new(NonZeroUsize::new(2)) + }; + + let inside = low_priority.run("fs-snap-test", seen); + + assert_eq!( + inside.ok().map(|(nice, name, in_pool, threads)| ( + nice, + name.is_some_and(|name| name.starts_with("fs-snap-test-")), + in_pool, + threads + )), + Some((before, true, true, 2)) + ); +} + +#[test] +fn work_whose_priority_cannot_be_lowered_still_runs_at_the_normal_priority() { + let before = own_nice(); + let low_priority = LowPriority { + lower: || Err(std::io::Error::other("the priority cannot change here")), + ..LowPriority::new(NonZeroUsize::new(2)) + }; + + let inside = low_priority.run("fs-snap-test", seen); + + assert_eq!( + inside.ok().map(|(nice, _, in_pool, _)| (nice, in_pool)), + Some((before, true)) + ); +} + +#[test] +fn work_without_its_pool_still_runs_at_nice_19_on_its_own_thread() { + let low_priority = LowPriority { + build_pool: no_pool, + ..LowPriority::new(NonZeroUsize::new(2)) + }; + + let inside = low_priority.run("fs-snap-test", seen); + + assert_eq!( + inside + .ok() + .map(|(nice, name, in_pool, _)| (nice, name, in_pool)), + Some((LOW_PRIORITY, Some("fs-snap-test".to_string()), false)) + ); +} + +#[test] +fn a_panic_of_the_work_gives_an_error() { + let done = LowPriority::new(NonZeroUsize::new(1)) + .run::<()>("fs-snap-test", || panic!("the work panics")); + + assert!(done.is_err()); +} + +#[test] +fn lowering_the_own_priority_gives_ok_and_nice_19() { + // The thread cannot raise its priority again, so the test lowers a thread of its own. + let lowered = std::thread::spawn(|| { + ( + lower_own_priority().map_err(|error| error.to_string()), + own_nice(), + ) + }) + .join(); + + assert_eq!(lowered.ok(), Some((Ok(()), LOW_PRIORITY))); +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs new file mode 100644 index 0000000000..0541cc11f0 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -0,0 +1,1623 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! When a delete of the store prunes the repository of its scope. +//! +//! The scope keeps a ledger next to the files of the repository: an entry for each prune, whose +//! name holds the time at which the prune ended and whether it marked packs that a later prune +//! removes. The newest entry is the ledger. Only the delete that holds the claim of a prune writes +//! an entry, and an entry is never written over. Each delete that freed bytes writes a record +//! of its own, with the count in the name, and a prune that succeeds deletes the records it counted. +//! +//! A delete whose prune is due takes a claim before it prunes, so two deletes that read the same +//! ledger make one prune. The claims of a ledger are in one directory, named by the time of the last +//! prune in that ledger. + +use super::backend::Lease; +use super::files::SnapshotFiles; +use futures::{StreamExt, TryStreamExt, stream}; +use golem_common::model::Timestamp; +use golem_service_base::storage::blob::{ListedBlob, PutIfAbsent}; +use std::collections::HashSet; +use std::path::Path; +use std::sync::{Mutex, PoisonError}; +use std::time::{Duration, Instant}; +use tracing::warn; + +/// The directory of the ledger entries, relative to the root of the namespace of the scope. +pub(super) const LEDGERS_PATH: &str = "golem/prune-ledgers"; + +/// How far the clock of another host can be ahead of the local clock. A time that is further +/// ahead counts as missing. +pub(super) const CLOCK_SKEW_MARGIN: Duration = Duration::from_secs(2 * 60); + +/// The directory of the prune claims, relative to the root of the namespace of the scope. +const CLAIMS_PATH: &str = "golem/prune-claims"; + +/// The directory of the records of freed bytes, relative to the root of the namespace of the scope. +const FREED_PATH: &str = "golem/prune-freed"; + +/// What the scope did since its last prune. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub(super) struct PruneLedger { + /// The time of the last prune. + pub(super) last_prune: Option, + /// Whether the last prune marked packs that a later prune removes. + pub(super) awaiting_removal: bool, +} + +/// The path of the packs of a repository, relative to the root of the namespace of the scope. +const DATA_PATH: &str = "data"; + +/// A share of the size of a repository, in percent. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) struct Percent(pub(super) u16); + +impl Percent { + /// Gives this share of the bytes, rounded down. + fn of(self, bytes: u64) -> u64 { + u64::try_from(u128::from(bytes) * u128::from(self.0) / 100).unwrap_or(u64::MAX) + } +} + +/// Tells whether the hold of a claim passed at `now` since the last prune. The time of the last +/// prune is after its final marker, so the next prune waits a full hold from the newest claim +/// marker that the last prune wrote. A time more than the margin after `now` counts as missing. +pub(super) fn hold_passed( + ledger: &PruneLedger, + now: Timestamp, + grace: Duration, + deadline: Duration, +) -> bool { + ledger + .last_prune + .filter(|last| !beyond_margin(*last, now)) + .is_none_or(|last| passed_since(last, now, claim_hold(grace, deadline))) +} + +/// Tells whether the grace period passed at `now` since the time. +fn passed_since(time: Timestamp, now: Timestamp, grace: Duration) -> bool { + now.to_millis() + >= time + .to_millis() + .saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) +} + +/// Tells whether [`prune_due`] needs the size of the repository at `now`. Only freed bytes after +/// the hold of a claim, without marked packs, need it. +pub(super) fn needs_repository_size( + ledger: &PruneLedger, + freed_bytes: u64, + now: Timestamp, + grace: Duration, + deadline: Duration, +) -> bool { + hold_passed(ledger, now, grace, deadline) && freed_bytes > 0 && !ledger.awaiting_removal +} + +/// Tells whether a prune is due at `now`. +/// +/// A prune is due when the hold of a claim passed since the last prune, and the freed bytes reach the +/// threshold share of `repository_bytes`, rounded down to a whole byte, or the last prune marked +/// packs. A threshold of zero bytes counts as one byte, so a prune never runs for a scope that +/// freed nothing and marked nothing. +pub(super) fn prune_due( + ledger: &PruneLedger, + freed_bytes: u64, + now: Timestamp, + repository_bytes: u64, + threshold: Percent, + grace: Duration, + deadline: Duration, +) -> bool { + let work = freed_bytes >= threshold.of(repository_bytes).max(1) || ledger.awaiting_removal; + hold_passed(ledger, now, grace, deadline) && work +} + +/// Tells whether the freed bytes in the names of the records, which are at least the settled +/// freed bytes, can make a prune due with the size of the repository. When this gives false, +/// [`prune_due`] gives false for the settled bytes too, so the records need no read. +pub(super) fn may_be_due( + ledger: &PruneLedger, + named_bytes: u64, + repository_bytes: u64, + threshold: Percent, +) -> bool { + named_bytes >= threshold.of(repository_bytes).max(1) || ledger.awaiting_removal +} + +/// Gives the size of the repository of the scope: the sum of the sizes of its packs. +pub(super) async fn repository_bytes(files: &SnapshotFiles) -> anyhow::Result { + Ok(files + .list_below("list_data", Path::new(DATA_PATH)) + .await? + .iter() + .map(|blob| blob.size) + .fold(0, u64::saturating_add)) +} + +/// Reads a ledger entry name, `-<0|1>-`, as the time of the prune and whether +/// it marked packs. +pub(super) fn parse_ledger_entry(name: &str) -> Option { + let mut parts = name.splitn(3, '-'); + let ended = parts.next()?.parse::().ok()?; + let awaiting_removal = match parts.next()? { + "0" => false, + "1" => true, + _ => return None, + }; + parts.next().filter(|unique| !unique.is_empty())?; + Some(PruneLedger { + last_prune: Some(Timestamp::from(ended)), + awaiting_removal, + }) +} + +/// Tells whether the time is more than [`CLOCK_SKEW_MARGIN`] after `now`. +fn beyond_margin(time: Timestamp, now: Timestamp) -> bool { + time.to_millis() + > now + .to_millis() + .saturating_add(u64::try_from(CLOCK_SKEW_MARGIN.as_millis()).unwrap_or(u64::MAX)) +} + +/// Gives the ledger from the listed entries: the entry with the greatest time. A name that does not +/// parse and a time more than the margin after `now` are left out. No entry gives the default. +pub(super) fn newest_ledger(listed: &[ListedBlob], now: Timestamp) -> PruneLedger { + listed + .iter() + .filter_map(|blob| parse_ledger_entry(blob.path.file_name()?.to_str()?)) + .filter(|entry| { + entry + .last_prune + .is_some_and(|ended| !beyond_margin(ended, now)) + }) + .max_by_key(|entry| (entry.last_prune, entry.awaiting_removal)) + .unwrap_or_default() +} + +/// Gives the paths of the listed entries whose time is before `ended`, in whole milliseconds as +/// an entry name holds it. +pub(super) fn older_entries(listed: &[ListedBlob], ended: Timestamp) -> Box<[Box]> { + listed + .iter() + .filter(|blob| { + blob.path + .file_name() + .and_then(|name| name.to_str()) + .and_then(parse_ledger_entry) + .and_then(|entry| entry.last_prune) + .is_some_and(|time| time.to_millis() < ended.to_millis()) + }) + .map(|blob| blob.path.clone()) + .collect() +} + +/// Reads the ledger of the scope from one listing of its entries. No read of content is needed. +pub(super) async fn read_ledger(files: &SnapshotFiles) -> anyhow::Result { + let listed = files + .list_below("read_ledger", Path::new(LEDGERS_PATH)) + .await?; + Ok(newest_ledger(&listed, Timestamp::now_utc())) +} + +/// Writes a new ledger entry for a prune that ended at `ended`. +pub(super) async fn write_ledger( + files: &SnapshotFiles, + ended: Timestamp, + awaiting_removal: bool, +) -> anyhow::Result<()> { + let name = format!( + "{}-{}-{}", + ended.to_millis(), + u8::from(awaiting_removal), + uuid::Uuid::new_v4() + ); + files + .put_if_absent("write_ledger", &Path::new(LEDGERS_PATH).join(name), &[]) + .await + .map(|_| ()) +} + +/// Deletes each ledger entry that is older than the entry of the prune that ended at `ended`. A +/// failure gives a warning, because an older entry is never the newest. +pub(super) async fn remove_older_ledgers(files: &SnapshotFiles, ended: Timestamp) { + let listed = match files + .list_below("list_ledgers", Path::new(LEDGERS_PATH)) + .await + { + Ok(listed) => listed, + Err(error) => { + warn!( + error = %format!("{error:#}"), + "Failed to list the prune ledger entries of a filesystem snapshot scope" + ); + return; + } + }; + stream::iter(older_entries(&listed, ended)) + .for_each(|path| async move { + if let Err(error) = files.delete("delete_ledger", &path).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete an old prune ledger entry of a filesystem snapshot scope" + ); + } + }) + .await; +} + +/// The freed bytes of the settled records, and the paths of those records. +#[derive(Clone, Debug, Default, PartialEq, Eq)] +pub(super) struct FreedRecords { + pub(super) bytes: u64, + pub(super) counted: Box<[Box]>, +} + +/// A record of freed bytes that a delete wrote: its path, the bytes in its name, and the ids of +/// the snapshot files of that delete, when its content parses. +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct FreedRecord { + pub(super) path: Box, + pub(super) bytes: u64, + pub(super) snapshots: Option]>>, +} + +/// The directory of the snapshot files of a repository. +pub(super) const SNAPSHOTS_PATH: &str = "snapshots"; + +/// Gives the content of a record: the id of each snapshot file of the delete, one on each line. +pub(super) fn record_content(snapshots: &[Box]) -> String { + snapshots.join("\n") +} + +/// Reads the snapshot ids from the content of a record. Each line must be an id of 64 hex +/// characters. A content without an id does not parse, because a reader can see a record that a +/// write has not filled yet. +pub(super) fn parse_record(content: &[u8]) -> Option]>> { + let text = std::str::from_utf8(content).ok()?; + text.lines() + .filter(|line| !line.is_empty()) + .map(|line| { + (line.len() == 64 && line.bytes().all(|byte| byte.is_ascii_hexdigit())) + .then(|| line.into()) + }) + .collect::]>>>() + .filter(|snapshots| !snapshots.is_empty()) +} + +/// Gives the settled records and the sum of their bytes. A record is settled when it names at +/// least one snapshot file and none of them exists. Any other record counts as zero bytes and +/// stays, and so does a record whose content does not parse. +pub(super) fn settle(records: &[FreedRecord], existing: &HashSet>) -> FreedRecords { + let settled = records + .iter() + .filter(|record| { + record.snapshots.as_ref().is_some_and(|snapshots| { + !snapshots.is_empty() && snapshots.iter().all(|id| !existing.contains(id)) + }) + }) + .collect::>(); + FreedRecords { + bytes: settled + .iter() + .map(|record| record.bytes) + .fold(0, u64::saturating_add), + counted: settled.iter().map(|record| record.path.clone()).collect(), + } +} + +/// Reads the freed bytes from the name of a record, `-`. +pub(super) fn parse_freed(name: &str) -> Option { + let (bytes, unique) = name.split_once('-')?; + if unique.is_empty() { + return None; + } + bytes.parse().ok() +} + +/// Writes a record of the freed bytes of one delete, with the ids of its snapshot files. +pub(super) async fn record_freed( + files: &SnapshotFiles, + bytes: u64, + snapshots: &[Box], +) -> anyhow::Result<()> { + let path = Path::new(FREED_PATH).join(format!("{bytes}-{}", uuid::Uuid::new_v4())); + // The name is unique, so `AlreadyExists` means that an earlier try of this call wrote it. + files + .put_if_absent("write_freed", &path, record_content(snapshots).as_bytes()) + .await + .map(|_| ()) +} + +/// A record of freed bytes that a listing found: its path and the bytes in its name. +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct ListedFreed { + pub(super) path: Box, + pub(super) bytes: u64, +} + +/// Lists the names of the records of freed bytes, and reads no content. A name that does not +/// parse is left out, so it counts as zero bytes and stays. +pub(super) async fn list_freed_names(files: &SnapshotFiles) -> anyhow::Result> { + Ok(files + .list_below("list_freed", Path::new(FREED_PATH)) + .await? + .iter() + .filter_map(|blob| { + let bytes = parse_freed(blob.path.file_name()?.to_str()?)?; + Some(ListedFreed { + path: blob.path.clone(), + bytes, + }) + }) + .collect()) +} + +/// Gives the sum of the bytes in the names of the listed records. +pub(super) fn named_bytes(listed: &[ListedFreed]) -> u64 { + listed + .iter() + .map(|record| record.bytes) + .fold(0, u64::saturating_add) +} + +/// Reads the listed records of freed bytes, lists the snapshot files one time, and gives the +/// settled records. A record that a prune deleted after the listing is left out. Without records, +/// no snapshot file is listed. +pub(super) async fn settle_freed( + files: &SnapshotFiles, + listed: &[ListedFreed], +) -> anyhow::Result { + if listed.is_empty() { + return Ok(FreedRecords::default()); + } + let records = stream::iter(listed) + .then(|listed| async move { + let content = files.get("read_freed", &listed.path).await?; + Ok::<_, anyhow::Error>(content.map(|content| FreedRecord { + path: listed.path.clone(), + bytes: listed.bytes, + snapshots: parse_record(&content), + })) + }) + .try_filter_map(|record| std::future::ready(Ok(record))) + .try_collect::>() + .await?; + let existing = files + .list_below("list_snapshots", Path::new(SNAPSHOTS_PATH)) + .await? + .iter() + .filter_map(|blob| Some(blob.path.file_name()?.to_str()?.into())) + .collect::>>(); + Ok(settle(&records, &existing)) +} + +/// Deletes the counted records after a prune. A failure gives a warning, because a record that +/// stays only makes the next prune come earlier. +pub(super) async fn remove_freed(files: &SnapshotFiles, records: &FreedRecords) { + stream::iter(&records.counted) + .for_each(|path| async move { + if let Err(error) = files.delete("delete_freed", path).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete a record of freed bytes of a filesystem snapshot scope" + ); + } + }) + .await; +} + +/// An entry of a claim directory that a listing found: a claim ``, or a marker +/// `@-` with the time at which a delete wrote it. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) enum ClaimEntry { + Claim(u64), + Marker(u64, Timestamp), +} + +impl ClaimEntry { + fn number(self) -> u64 { + match self { + Self::Claim(number) | Self::Marker(number, _) => number, + } + } +} + +/// Reads the name of an entry of a claim directory. +pub(super) fn parse_claim_entry(name: &str) -> Option { + match name.split_once('@') { + None => name.parse().ok().map(ClaimEntry::Claim), + Some((number, rest)) => { + let (millis, unique) = rest.split_once('-')?; + if unique.is_empty() { + return None; + } + Some(ClaimEntry::Marker( + number.parse().ok()?, + Timestamp::from(millis.parse::().ok()?), + )) + } + } +} + +/// What a delete whose prune is due does with the claims of its ledger. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) enum ClaimChoice { + /// Take the claim with the number, and prune when the write of the claim succeeds. + Claim(u64), + /// Another prune holds the claims of the ledger, so do not prune. + Held, +} + +/// The unit of the time in the name of a marker. +const MARKER_TIME_UNIT: Duration = Duration::from_millis(1); + +/// Gives the time from the start of a marker write to the end of the lease that the marker gives, +/// as the other deletes count it from the time in the name: the grace period less one storage call +/// deadline, but at least one deadline. +fn lease_bound(grace: Duration, deadline: Duration) -> Duration { + grace.saturating_sub(deadline).max(deadline) +} + +/// Gives how long a marker write that succeeds lets the prune of its claim go on: the lease bound +/// less one millisecond. The name of a marker keeps its time in whole milliseconds, so the time in +/// the name can be up to one millisecond before the time that the delete read. So the lease is above +/// zero for each deadline above one millisecond. +pub(super) fn lease_span(grace: Duration, deadline: Duration) -> Duration { + lease_bound(grace, deadline).saturating_sub(MARKER_TIME_UNIT) +} + +/// Gives how long a marker holds the claims of its ledger: the lease bound, then one margin for +/// clock skew, then two storage call deadlines. Another delete sees the marker time with up to one +/// margin of skew. A marker write that the storage received can still land up to one deadline after +/// the call gave up. A prune that its lease stopped writes its final marker after the lease ran +/// out, and that write can land up to one more deadline later. So a prune whose lease ran out stops +/// before another delete can take its claim, and the next prune waits a full hold from its final +/// marker. +pub(super) fn claim_hold(grace: Duration, deadline: Duration) -> Duration { + lease_bound(grace, deadline) + .saturating_add(CLOCK_SKEW_MARGIN) + .saturating_add(deadline.saturating_mul(2)) +} + +/// Chooses the claim of a delete from the entries of the claim directory of its ledger. Any marker +/// younger than the hold holds the ledger, whatever its number. A marker whose time is more than +/// the margin after `now` counts as missing, and a claim without a young marker is old. Otherwise +/// the delete takes the number after the largest one, or 0. +pub(super) fn next_claim(entries: &[ClaimEntry], now: Timestamp, hold: Duration) -> ClaimChoice { + let held = entries.iter().any(|entry| match entry { + ClaimEntry::Marker(_, at) => !beyond_margin(*at, now) && !passed_since(*at, now, hold), + ClaimEntry::Claim(_) => false, + }); + if held { + return ClaimChoice::Held; + } + entries + .iter() + .map(|entry| entry.number()) + .max() + .map_or(ClaimChoice::Claim(0), |largest| { + ClaimChoice::Claim(largest.saturating_add(1)) + }) +} + +/// Gives the directory of the claims of the ledger: the time of its last prune in milliseconds, or +/// `none`. +pub(super) fn claims_directory(ledger: &PruneLedger) -> Box { + let generation = ledger + .last_prune + .map_or_else(|| "none".to_string(), |last| last.to_millis().to_string()); + Path::new(CLAIMS_PATH).join(generation).into_boxed_path() +} + +/// Lists the claims and the markers in the directory, from their names. A name that does not +/// parse is left out. +pub(super) async fn list_claims( + files: &SnapshotFiles, + directory: &Path, +) -> anyhow::Result> { + Ok(files + .list_below("list_claims", directory) + .await? + .iter() + .filter_map(|blob| parse_claim_entry(blob.path.file_name()?.to_str()?)) + .collect()) +} + +/// Gives the time of a new marker: the instant, from which a write of the marker moves the lease, +/// and then the wall time in the name of the marker. The instant is read first, so the lease starts +/// no later than the time in the name, and it never ends later than the hold that other deletes +/// count from the name. +pub(super) fn marker_time() -> (Instant, Timestamp) { + let started = Instant::now(); + let time = Timestamp::now_utc(); + (started, time) +} + +/// Gives a new path of a marker of the claim with the number, with the time. The name is unique. +pub(super) fn marker_path(directory: &Path, number: u64, time: Timestamp) -> Box { + directory + .join(format!( + "{number}@{}-{}", + time.to_millis(), + uuid::Uuid::new_v4() + )) + .into_boxed_path() +} + +/// Writes the marker at the path. The name is unique, so `AlreadyExists` means that an earlier try +/// of this call wrote it. +async fn write_marker_at( + files: &SnapshotFiles, + op_label: &'static str, + path: &Path, +) -> anyhow::Result<()> { + let _: PutIfAbsent = files.put_if_absent(op_label, path, &[]).await?; + Ok(()) +} + +/// Writes a marker of the claim with the number, with the time, and gives its path. +pub(super) async fn write_marker( + files: &SnapshotFiles, + op_label: &'static str, + directory: &Path, + number: u64, + time: Timestamp, +) -> anyhow::Result> { + let path = marker_path(directory, number, time); + write_marker_at(files, op_label, &path).await?; + Ok(path) +} + +/// Writes a marker of the claim with the number. A write that succeeds and started before the end +/// of the lease moves the end to `span` after the start of the write, when that is later. A write +/// that started at or after the end does not move it. +async fn write_leased_marker( + files: &SnapshotFiles, + op_label: &'static str, + directory: &Path, + number: u64, + lease: &Lease, + span: Duration, +) -> anyhow::Result> { + let (started, time) = marker_time(); + let marker = write_marker(files, op_label, directory, number, time).await?; + lease.extend_from(started, span); + Ok(marker) +} + +/// Writes the first marker of the claim with the number at the path `marker`, then takes the +/// claim, and gives the lease of the prune when this delete holds the claim. The caller makes the +/// path before the write, so a guard can delete the marker when the delete stops during the write. +/// A delete that loses the claim deletes its marker. The lease starts with the marker write: it +/// ends `span` after `started`, the instant that [`marker_time`] gave with the time in the name of +/// the marker. +pub(super) async fn take_claim( + files: &SnapshotFiles, + directory: &Path, + number: u64, + marker: &Path, + started: Instant, + span: Duration, +) -> anyhow::Result> { + write_marker_at(files, "write_marker", marker).await?; + let lease = Lease::until(started + span); + let written = files + .put_if_absent("write_claim", &directory.join(number.to_string()), &[]) + .await?; + if written == PutIfAbsent::Written { + return Ok(Some(lease)); + } + if let Err(error) = files.delete("delete_marker", marker).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete the marker of a prune claim that a filesystem snapshot delete lost" + ); + } + Ok(None) +} + +/// Gives the time between two markers of a live claim: a fourth of the grace period, or a fourth +/// of the margin for clock skew when the grace period is zero, but at most a fourth of the lease, so +/// each lease has at least two refreshes. +pub(super) fn refresh_period(grace: Duration, deadline: Duration) -> Duration { + let base = if grace.is_zero() { + CLOCK_SKEW_MARGIN + } else { + grace + }; + (base / 4).min(lease_span(grace, deadline) / 4) +} + +/// Writes a new marker of the claim with the number at each period, until the caller drops the +/// future or the operation of the files is cancelled, and adds the path of each written marker to +/// `written`. A write that succeeds and started before the end of the lease moves the end to +/// `span` after its start, when that is later. A write that started at or after the end does not +/// move it. A failed write gives a warning, and the next period tries again. +pub(super) async fn keep_claim_fresh( + files: &SnapshotFiles, + directory: &Path, + number: u64, + period: Duration, + written: &Mutex>>, + lease: &Lease, + span: Duration, +) { + stream::repeat(()) + .then(|()| tokio::time::sleep(period)) + .take_until(files.cancel.cancelled()) + .for_each(|()| async move { + let marker = + write_leased_marker(files, "refresh_claim", directory, number, lease, span).await; + match marker { + Ok(path) => written + .lock() + .unwrap_or_else(PoisonError::into_inner) + .push(path), + Err(error) => warn!( + error = %format!("{error:#}"), + "Failed to write a new marker of the prune claim of a filesystem snapshot scope" + ), + } + }) + .await; +} + +/// Deletes the claim with the number when this delete took it, and then each of its markers by its +/// path. It tries each delete also when another one fails, and each failure gives a warning. A +/// claim that stays without its markers is old, and a marker that stays only delays a prune until +/// its hold passed. +pub(super) async fn release_claim( + files: &SnapshotFiles, + directory: &Path, + number: u64, + claimed: bool, + markers: &[Box], +) { + let claim = directory.join(number.to_string()); + stream::iter( + claimed + .then_some(("delete_claim", claim.as_path())) + .into_iter() + .chain( + markers + .iter() + .map(|marker| ("delete_marker", marker.as_ref())), + ), + ) + .for_each(|(op_label, path)| async move { + if let Err(error) = files.delete(op_label, path).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete the prune claim of a filesystem snapshot scope" + ); + } + }) + .await; +} + +/// Gives the claim directory of each listed path below the directory of all claims whose ledger is +/// older than the ledger of `ended`: the directory `none`, and each directory whose time is before +/// `ended`. A newer directory can hold a live claim of a later prune, and a directory whose name is +/// not a time is not a directory of claims, so both stay. +pub(super) fn old_claim_directories( + listed: impl IntoIterator>, + ended: Timestamp, +) -> Box<[Box]> { + let claims = Path::new(CLAIMS_PATH); + let mut directories = listed + .into_iter() + .filter_map(|path| { + let generation = path + .as_ref() + .strip_prefix(claims) + .ok()? + .components() + .next()? + .as_os_str() + .to_str()? + .to_owned(); + let old = generation == "none" + || generation + .parse::() + .is_ok_and(|millis| millis < ended.to_millis()); + old.then(|| claims.join(generation).into_boxed_path()) + }) + .collect::>(); + directories.sort(); + directories.dedup(); + directories.into_boxed_slice() +} + +/// Deletes each claim directory of a ledger older than the ledger of `ended`. A listing of the +/// directories finds an empty claim directory, and a listing of the blobs finds a claim directory +/// that the storage keeps no entry for. It never deletes the directory of all claims or a newer +/// directory, so a live claim of a later ledger stays. A failure gives a warning, because a claim +/// only delays a prune until its hold passed. +pub(super) async fn remove_old_claims(files: &SnapshotFiles, ended: Timestamp) { + let claims = Path::new(CLAIMS_PATH); + let listed = async { + let directories = files.list_dir("list_claim_directories", claims).await?; + let blobs = files.list_below("list_claim_blobs", claims).await?; + anyhow::Ok((directories, blobs)) + }; + let (directories, blobs) = match listed.await { + Ok(listed) => listed, + Err(error) => { + warn!( + error = %format!("{error:#}"), + "Failed to list the prune claims of a filesystem snapshot scope" + ); + return; + } + }; + let listed = directories + .iter() + .map(AsRef::as_ref) + .chain(blobs.iter().map(|blob| blob.path.as_ref())); + stream::iter(old_claim_directories(listed, ended)) + .for_each(|directory| async move { + if let Err(error) = files.delete_dir("delete_claims", &directory).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete the prune claims of a filesystem snapshot scope" + ); + } + }) + .await; +} + +#[cfg(test)] +mod tests { + use super::super::backend::BlobBackend; + use super::super::fault::is_lease_expired; + use super::super::files::SnapshotFiles; + use super::super::tests::scripted::{Script, ScriptedBlobStorage}; + use super::{ + CLAIMS_PATH, CLOCK_SKEW_MARGIN, ClaimChoice, ClaimEntry, FREED_PATH, FreedRecord, + FreedRecords, LEDGERS_PATH, Lease, Percent, PruneLedger, claim_hold, claims_directory, + keep_claim_fresh, lease_span, list_claims, list_freed_names, marker_path, marker_time, + may_be_due, named_bytes, needs_repository_size, newest_ledger, next_claim, + old_claim_directories, older_entries, parse_claim_entry, parse_freed, parse_ledger_entry, + parse_record, prune_due, read_ledger, record_content, record_freed, refresh_period, settle, + settle_freed, take_claim, write_ledger, + }; + use futures::StreamExt; + use golem_common::model::Timestamp; + use golem_common::model::environment::EnvironmentId; + use golem_service_base::storage::blob::BlobStorage; + use golem_service_base::storage::blob::BlobStorageNamespace; + use golem_service_base::storage::blob::ListedBlob; + use golem_service_base::storage::blob::memory::InMemoryBlobStorage; + use pretty_assertions::assert_eq; + use rustic_core::{FileType, ReadBackend}; + use std::path::{Path, PathBuf}; + use std::sync::Arc; + use std::sync::atomic::{AtomicBool, Ordering}; + use std::time::Duration; + use std::time::Instant; + use test_r::{test, timeout}; + use uuid::Uuid; + + const TEN_PERCENT: Percent = Percent(10); + const GRACE: Duration = Duration::from_secs(15 * 60); + /// The grace period and the margin for clock skew, in milliseconds. + const HELD_MILLIS: u64 = 15 * 60 * 1000 + 2 * 60 * 1000; + /// The hold of a claim with the grace period and the deadline, in milliseconds: the grace + /// period less one deadline, then the margin, then two deadlines. + const HOLD_MILLIS: u64 = HELD_MILLIS + 2 * 1000; + const DEADLINE: Duration = Duration::from_secs(2); + const MILLI: Duration = Duration::from_millis(1); + + fn at(millis: u64) -> Timestamp { + Timestamp::from(millis) + } + + fn ledger(last_prune_millis: Option, awaiting_removal: bool) -> PruneLedger { + PruneLedger { + last_prune: last_prune_millis.map(Timestamp::from), + awaiting_removal, + } + } + + fn new_files() -> SnapshotFiles { + files_over(Arc::new(InMemoryBlobStorage::new())) + } + + fn files_over(storage: Arc) -> SnapshotFiles { + SnapshotFiles { + storage, + namespace: BlobStorageNamespace::InitialAgentFiles { + environment_id: EnvironmentId(Uuid::new_v4()), + }, + deadline: DEADLINE, + cancel: tokio_util::sync::CancellationToken::new(), + tracker: tokio_util::task::TaskTracker::new(), + } + } + + #[test] + fn the_bytes_in_the_names_of_the_records_can_make_a_prune_due_only_when_they_reach_the_threshold() + { + let may = |awaiting_removal, named, repository_bytes| { + may_be_due( + &ledger(None, awaiting_removal), + named, + repository_bytes, + TEN_PERCENT, + ) + }; + + assert_eq!( + [ + may(false, 99, 1000), + may(false, 100, 1000), + may(false, 0, 0), + may(false, 1, 0), + may(true, 0, 1000), + ], + [false, true, false, true, true] + ); + } + + #[test] + fn a_prune_is_due_when_the_freed_bytes_reach_ten_percent_of_the_repository_rounded_down() { + let now = at(10_000_000); + let due = |freed, repository_bytes| { + prune_due( + &ledger(None, false), + freed, + now, + repository_bytes, + TEN_PERCENT, + GRACE, + DEADLINE, + ) + }; + + assert_eq!( + [ + due(99, 1000), + due(100, 1000), + due(101, 1000), + due(0, 0), + due(1, 0), + ], + [false, true, true, false, true] + ); + } + + #[test] + fn no_second_prune_runs_within_the_hold_of_a_claim_after_the_last_prune() { + // The grace period and the margin passed at `HELD_MILLIS`, and the full hold at + // `HOLD_MILLIS`. + let last = 1_000_000; + let full = |now| { + prune_due( + &ledger(Some(last), true), + 1000, + at(now), + 1000, + TEN_PERCENT, + GRACE, + DEADLINE, + ) + }; + + assert_eq!( + [ + full(last), + full(last + HELD_MILLIS), + full(last + HOLD_MILLIS - 1), + full(last + HOLD_MILLIS), + full(last + HOLD_MILLIS + 1), + ], + [false, false, false, true, true] + ); + } + + #[test] + fn marked_packs_make_a_prune_due_after_the_hold_without_freed_bytes() { + let last = 1_000_000; + let after_hold = at(last + HOLD_MILLIS); + let due = |awaiting_removal| { + prune_due( + &ledger(Some(last), awaiting_removal), + 0, + after_hold, + 1000, + TEN_PERCENT, + GRACE, + DEADLINE, + ) + }; + + assert_eq!([due(true), due(false)], [true, false]); + } + + #[test] + fn a_zero_threshold_prunes_after_each_delete_that_freed_bytes() { + let now = at(10_000_000); + let due = |ledger, freed| { + prune_due( + &ledger, + freed, + now, + 1000, + Percent(0), + Duration::ZERO, + DEADLINE, + ) + }; + let hold = u64::try_from(claim_hold(Duration::ZERO, DEADLINE).as_millis()).unwrap(); + + assert_eq!( + [ + due(ledger(None, false), 0), + due(ledger(None, false), 1), + due(ledger(Some(10_000_000 - hold), false), 1), + ], + [false, true, true] + ); + } + + #[test] + fn only_freed_bytes_after_the_hold_without_marked_packs_need_the_repository_size() { + let last = 1_000_000; + let needs = |freed, awaiting_removal, now| { + needs_repository_size( + &ledger(Some(last), awaiting_removal), + freed, + at(now), + GRACE, + DEADLINE, + ) + }; + + assert_eq!( + [ + needs(1, false, last + HOLD_MILLIS), + needs(1, false, last + HOLD_MILLIS - 1), + needs(0, false, last + HOLD_MILLIS), + needs(1, true, last + HOLD_MILLIS), + ], + [true, false, false, false] + ); + } + + #[test] + fn the_hold_holds_the_ledger_and_the_claims_and_a_time_more_than_the_margin_ahead_counts_as_missing() + { + let now = 10_000_000; + let margin = u64::try_from(CLOCK_SKEW_MARGIN.as_millis()).unwrap(); + let due = |last| { + prune_due( + &ledger(Some(last), false), + 1, + at(now), + 0, + Percent(0), + GRACE, + DEADLINE, + ) + }; + let claim = |claimed_at| { + next_claim( + &[ClaimEntry::Claim(0), ClaimEntry::Marker(0, at(claimed_at))], + at(now), + claim_hold(GRACE, DEADLINE), + ) + }; + + assert_eq!( + ( + [ + due(now - HOLD_MILLIS + 1), + due(now - HOLD_MILLIS), + due(now + margin), + due(now + margin + 1) + ], + [ + claim(now - HOLD_MILLIS + 1), + claim(now - HOLD_MILLIS), + claim(now + margin), + claim(now + margin + 1) + ] + ), + ( + [false, true, false, true], + [ + ClaimChoice::Held, + ClaimChoice::Claim(1), + ClaimChoice::Held, + ClaimChoice::Claim(1) + ] + ) + ); + } + + #[test] + fn any_young_marker_holds_a_ledger_and_a_claim_without_a_marker_is_old() { + let now = 10_000_000; + let claim = ClaimEntry::Claim; + let marker = |number, millis| ClaimEntry::Marker(number, at(millis)); + let choose = + |entries: &[ClaimEntry]| next_claim(entries, at(now), claim_hold(GRACE, DEADLINE)); + + assert_eq!( + [ + choose(&[]), + choose(&[claim(0), claim(1), marker(0, now - 1)]), + choose(&[claim(0), claim(1), marker(1, now - HOLD_MILLIS)]), + choose(&[claim(4)]), + choose(&[marker(2, now - 1)]), + choose(&[claim(0), claim(3), marker(0, now - HOLD_MILLIS)]), + ], + [ + ClaimChoice::Claim(0), + ClaimChoice::Held, + ClaimChoice::Claim(2), + ClaimChoice::Claim(5), + ClaimChoice::Held, + ClaimChoice::Claim(4), + ] + ); + } + + #[test] + fn the_lease_is_the_grace_period_less_one_deadline_and_at_least_one_deadline_less_one_millisecond() + { + let second = Duration::from_secs(1); + let minute = Duration::from_secs(60); + + assert_eq!( + [ + lease_span(GRACE, minute), + lease_span(2 * minute, minute), + lease_span(2 * minute - second, minute), + lease_span(Duration::ZERO, minute), + lease_span(minute, Duration::ZERO), + lease_span(Duration::ZERO, second), + ], + [GRACE - minute, minute, minute, minute, minute, second].map(|span| span - MILLI) + ); + } + + #[test] + fn the_hold_is_the_lease_then_one_margin_then_two_deadlines() { + let margin = CLOCK_SKEW_MARGIN; + let minute = Duration::from_secs(60); + + assert_eq!( + [ + claim_hold(GRACE, minute), + claim_hold(GRACE, DEADLINE), + claim_hold(Duration::ZERO, minute), + claim_hold(Duration::ZERO, Duration::from_millis(1)), + ], + [ + GRACE + margin + minute, + GRACE + margin + DEADLINE, + minute + margin + 2 * minute, + MILLI + margin + 2 * MILLI, + ] + ); + } + + #[test] + fn each_lease_has_at_least_two_refreshes() { + let minute = Duration::from_secs(60); + let cases = [ + (GRACE, minute), + (GRACE, DEADLINE), + (Duration::ZERO, minute), + (Duration::ZERO, 2 * MILLI), + (Duration::from_millis(16), minute), + (Duration::from_secs(2), Duration::from_secs(1)), + ]; + + assert_eq!( + cases.map(|(grace, deadline)| refresh_period(grace, deadline)), + [ + (GRACE - minute - MILLI) / 4, + (GRACE - DEADLINE - MILLI) / 4, + (minute - MILLI) / 4, + Duration::from_micros(250), + Duration::from_millis(4), + Duration::from_micros(249_750), + ] + ); + assert!(cases.iter().all(|(grace, deadline)| { + refresh_period(*grace, *deadline) * 2 <= lease_span(*grace, *deadline) + })); + } + + #[test] + fn a_claim_entry_name_is_a_number_or_a_number_with_a_time_and_a_unique_part() { + assert_eq!( + [ + parse_claim_entry("3"), + parse_claim_entry("3@42-a"), + parse_claim_entry("3@42-a-b"), + parse_claim_entry("3@42-"), + parse_claim_entry("3@x-a"), + parse_claim_entry("x"), + ], + [ + Some(ClaimEntry::Claim(3)), + Some(ClaimEntry::Marker(3, at(42))), + Some(ClaimEntry::Marker(3, at(42))), + None, + None, + None, + ] + ); + } + + #[test] + #[timeout("60s")] + async fn the_refresh_of_a_claim_ends_when_its_operation_is_cancelled() { + let files = new_files(); + files.cancel.cancel(); + + let ended = tokio::time::timeout( + Duration::from_secs(10), + keep_claim_fresh( + &files, + &claims_directory(&ledger(None, false)), + 0, + Duration::from_secs(3600), + &std::sync::Mutex::default(), + &Lease::until(Instant::now()), + GRACE, + ), + ) + .await + .is_ok(); + + assert!(ended); + } + + /// Refreshes the claim at each period until the first refresh write succeeds, and gives the + /// instant after that write. It gives `None` when the refresh ends before a write succeeded. + async fn refresh_until_written( + files: &SnapshotFiles, + period: Duration, + lease: &Lease, + span: Duration, + ) -> Option { + let written = std::sync::Mutex::>>::default(); + let directory = claims_directory(&ledger(None, false)); + tokio::select! { + () = keep_claim_fresh(files, &directory, 0, period, &written, lease, span) => None, + ended = async { + futures::stream::repeat(()) + .then(|()| tokio::time::sleep(Duration::from_millis(5))) + .take_while(|()| { + std::future::ready( + written.lock().unwrap_or_else(std::sync::PoisonError::into_inner).is_empty(), + ) + }) + .for_each(|()| std::future::ready(())) + .await; + Instant::now() + } => Some(ended), + } + } + + #[test] + #[timeout("60s")] + async fn a_refresh_that_starts_before_the_end_of_the_lease_and_ends_after_it_moves_the_lease_to_its_start_plus_the_span() + { + // The first refresh starts about 10 ms after the lease starts, before its end at 250 ms. + // Each refresh write takes the delay, so the write ends after the end of the lease. A lease + // from the end of the write would be later than a lease from its start by the delay. + let delay = Duration::from_millis(500); + let span = Duration::from_secs(1); + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), move |op_label, _| { + if op_label == "refresh_claim" { + Script::Delay(delay) + } else { + Script::Pass + } + }); + let files = files_over(storage); + let started = Instant::now(); + let first_expiry = started + Duration::from_millis(250); + let lease = Lease::until(first_expiry); + + let ended = refresh_until_written(&files, Duration::from_millis(10), &lease, span) + .await + .unwrap(); + + let expiry = lease.expiry(); + assert!( + ended > first_expiry, + "the write ended before the end of the lease" + ); + assert!(expiry >= started + span, "the lease did not move"); + assert!( + expiry + delay / 2 < ended + span, + "the lease moved to the end of the write" + ); + } + + #[test] + #[timeout("60s")] + async fn a_refresh_that_starts_after_the_lease_ran_out_does_not_move_it_and_the_next_call_is_refused() + { + // The lease ends when the test starts. The first refresh write fails, and a later one + // succeeds, but each refresh starts after the end of the lease. Nothing calls the backend + // while the lease is out, and the first call after the refresh finds the lease still out. + let refused = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refused = refused.clone(); + move |op_label, _| { + if op_label == "refresh_claim" && !refused.swap(true, Ordering::SeqCst) { + Script::Refuse + } else { + Script::Pass + } + } + }); + let files = files_over(storage.clone()); + let expiry = Instant::now(); + let lease = Arc::new(Lease::until(expiry)); + + let refreshed = refresh_until_written( + &files, + Duration::from_millis(10), + &lease, + Duration::from_secs(3600), + ) + .await + .is_some(); + let backend = BlobBackend::new( + storage, + files.namespace.clone(), + tokio::runtime::Handle::current(), + DEADLINE, + ) + .leased_by(lease.clone()); + let listed = tokio::task::spawn_blocking(move || backend.list(FileType::Snapshot)) + .await + .unwrap(); + + assert_eq!( + ( + refreshed, + refused.load(Ordering::SeqCst), + lease.expiry() == expiry, + listed + .as_ref() + .err() + .is_some_and(|error| is_lease_expired(error)), + ), + (true, true, true, true), + "{listed:?}" + ); + } + + #[test] + #[timeout("60s")] + async fn a_claim_is_taken_after_its_marker_and_a_loser_deletes_its_marker() { + let files = new_files(); + let directory = claims_directory(&ledger(Some(42), false)); + let marker = || { + let (at, time) = marker_time(); + (marker_path(&directory, 0, time), at) + }; + let (first_marker, first_at) = marker(); + let (second_marker, second_at) = marker(); + let first = take_claim(&files, &directory, 0, &first_marker, first_at, GRACE) + .await + .unwrap() + .map(|lease| lease.expiry()); + let again = take_claim(&files, &directory, 0, &second_marker, second_at, GRACE) + .await + .unwrap() + .map(|lease| lease.expiry()); + let listed = list_claims(&files, &directory).await.unwrap(); + + assert_eq!( + ( + directory.display().to_string(), + first, + again, + listed.len(), + listed.contains(&ClaimEntry::Claim(0)), + listed + .iter() + .any(|entry| matches!(entry, ClaimEntry::Marker(0, _))), + ), + ( + "golem/prune-claims/42".to_string(), + Some(first_at + GRACE), + None, + 2, + true, + true, + ) + ); + } + + #[test] + fn a_claim_directory_is_old_when_it_is_none_or_its_time_is_before_the_new_ledger() { + let claim = + |directory: &str, number: &str| Path::new(CLAIMS_PATH).join(directory).join(number); + let directory = |name: &str| Path::new(CLAIMS_PATH).join(name); + + assert_eq!( + old_claim_directories( + [ + claim("none", "0"), + claim("100", "0"), + claim("100", "1@5-a"), + claim("200", "3"), + directory("250"), + claim("260", "made/below"), + claim("300", "0"), + directory("300"), + claim("301", "0"), + directory("400"), + claim("x1", "0"), + Path::new(LEDGERS_PATH).join("5-0-a"), + PathBuf::from(CLAIMS_PATH), + ], + Timestamp::from(300) + ) + .to_vec(), + ["100", "200", "250", "260", "none"].map(|name| directory(name).into_boxed_path()) + ); + } + + #[test] + fn a_ledger_entry_name_gives_the_end_time_and_the_marked_packs() { + assert_eq!( + [ + parse_ledger_entry("42-1-a"), + parse_ledger_entry("43-0-a-b"), + parse_ledger_entry("42-2-a"), + parse_ledger_entry("42-1-"), + parse_ledger_entry("42-1"), + parse_ledger_entry("x-1-a"), + ], + [ + Some(ledger(Some(42), true)), + Some(ledger(Some(43), false)), + None, + None, + None, + None + ] + ); + } + + #[test] + fn the_newest_entry_within_the_margin_is_the_ledger() { + let now = 10_000_000; + let margin = u64::try_from(CLOCK_SKEW_MARGIN.as_millis()).unwrap(); + let entry = |name: &str| ListedBlob { + path: Path::new(LEDGERS_PATH).join(name).into(), + size: 0, + }; + let newest = |names: &[&str]| { + newest_ledger( + &names.iter().map(|name| entry(name)).collect::>(), + at(now), + ) + }; + + assert_eq!( + [ + newest(&[]), + newest(&["5-0-a", "9-1-b", "7-0-c"]), + newest(&["5-0-a", "not-an-entry"]), + newest(&["5-0-a", &format!("{}-1-b", now + margin)]), + newest(&["5-0-a", &format!("{}-1-b", now + margin + 1)]), + ], + [ + PruneLedger::default(), + ledger(Some(9), true), + ledger(Some(5), false), + ledger(Some(now + margin), true), + ledger(Some(5), false), + ] + ); + } + + #[test] + fn only_entries_before_the_end_of_a_prune_are_older() { + let entry = |name: &str| ListedBlob { + path: Path::new(LEDGERS_PATH).join(name).into(), + size: 0, + }; + + assert_eq!( + older_entries( + &[ + entry("5-0-a"), + entry("9-1-own"), + entry("9-0-same-time"), + entry("12-0-newer"), + entry("bad"), + ], + at(9) + ) + .to_vec(), + vec![Path::new(LEDGERS_PATH).join("5-0-a").into_boxed_path()] + ); + } + + #[test] + fn the_freed_bytes_of_a_record_are_the_number_before_the_first_dash() { + assert_eq!( + [ + parse_freed("123-0f4e"), + parse_freed("0-a-b"), + parse_freed("123"), + parse_freed("123-"), + parse_freed("x-0f4e"), + parse_freed("-0f4e"), + ], + [Some(123), Some(0), None, None, None, None] + ); + } + + #[test] + fn a_record_counts_only_when_each_of_its_snapshot_files_is_gone() { + let id = |digit: char| { + std::iter::repeat_n(digit, 64) + .collect::() + .into_boxed_str() + }; + let record = |name: &str, bytes: u64, snapshots: Option>>| FreedRecord { + path: Path::new(FREED_PATH).join(name).into(), + bytes, + snapshots: snapshots.map(Vec::into_boxed_slice), + }; + let existing = [id('b')].into_iter().collect(); + + let settled = settle( + &[ + record("5-gone", 5, Some(vec![id('a')])), + record("7-kept", 7, Some(vec![id('a'), id('b')])), + record("11-empty", 11, Some(Vec::new())), + record("13-bad", 13, None), + record(&format!("{}-max", u64::MAX), u64::MAX, Some(vec![id('c')])), + ], + &existing, + ); + + assert_eq!( + settled, + FreedRecords { + bytes: u64::MAX, + counted: Box::new([ + Path::new(FREED_PATH).join("5-gone").into(), + Path::new(FREED_PATH) + .join(format!("{}-max", u64::MAX)) + .into(), + ]), + } + ); + } + + #[test] + fn the_content_of_a_record_holds_one_snapshot_id_on_each_line() { + let id = |digit: char| { + std::iter::repeat_n(digit, 64) + .collect::() + .into_boxed_str() + }; + let ids = [id('a'), id('b')]; + + assert_eq!( + [ + parse_record(record_content(&ids).as_bytes()), + parse_record(b""), + parse_record(b"not an id"), + parse_record(&[0xff, 0xfe]), + ], + [ + Some(Box::new(ids.clone()) as Box<[Box]>), + None, + None, + None + ] + ); + } + + #[test] + fn a_record_line_parses_only_when_it_has_64_characters_that_are_each_hex() { + // A record whose content does not parse names no snapshot file, so it counts as zero + // bytes and stays. + let hex = "0123456789abcdef".repeat(4); + let parsed = [ + parse_record("g".repeat(64).as_bytes()), + parse_record(b"abcdef0123"), + parse_record(hex.as_bytes()), + ]; + let records = parsed + .iter() + .enumerate() + .map(|(index, snapshots)| FreedRecord { + path: Path::new(FREED_PATH) + .join(format!("10-{index}")) + .into_boxed_path(), + bytes: 10, + snapshots: snapshots.clone(), + }) + .collect::>(); + + assert_eq!( + ( + parsed.clone(), + settle(&records, &std::collections::HashSet::new()), + ), + ( + [ + None, + None, + Some(Box::new([hex.clone().into_boxed_str()]) as Box<[Box]>) + ], + FreedRecords { + bytes: 10, + counted: Box::new([Path::new(FREED_PATH).join("10-2").into_boxed_path()]), + }, + ) + ); + } + + #[test] + #[timeout("60s")] + async fn a_record_of_freed_bytes_is_written_and_listed() { + let files = new_files(); + + let gone = ["0".repeat(64).into_boxed_str()]; + record_freed(&files, 40, &gone).await.unwrap(); + record_freed(&files, 2, &gone).await.unwrap(); + let listed = list_freed_names(&files).await.unwrap(); + let settled = settle_freed(&files, &listed).await.unwrap(); + + assert_eq!( + (named_bytes(&listed), settled.bytes, settled.counted.len()), + (42, 42, 2) + ); + } + + #[test] + #[timeout("60s")] + async fn a_written_entry_is_the_ledger_that_a_read_gives() { + let files = new_files(); + let ended = Timestamp::from(Timestamp::now_utc().to_millis()); + + let before = read_ledger(&files).await.unwrap(); + write_ledger(&files, ended, true).await.unwrap(); + let after = read_ledger(&files).await.unwrap(); + + assert_eq!( + (before, after), + ( + PruneLedger::default(), + PruneLedger { + last_prune: Some(ended), + awaiting_removal: true + } + ) + ); + } +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/mod.rs new file mode 100644 index 0000000000..7c23d55d6e --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/mod.rs @@ -0,0 +1,137 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The publish of a snapshot file. +//! +//! In a save of the store, the backend keeps the snapshot file in a [`SnapshotStage`] and does not +//! write it. The save writes it later with [`publish`], after the blocking work returns. That +//! write is the step that makes the snapshot visible. A publish that fails, or that the caller +//! drops, deletes the file again, because a write that the storage received can still complete. + +use super::files::SnapshotFiles; +use bytes::Bytes; +use std::path::Path; +use std::sync::{Arc, Mutex, PoisonError}; +use tokio::runtime::Handle; +use tokio_util::task::TaskTracker; +use tracing::warn; + +/// A snapshot file that the backend kept and did not write. +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct StagedSnapshot { + /// The path of the file, relative to the root of the namespace. The drop guard of a publish + /// and its delete task share it. + pub(super) path: Arc, + pub(super) content: Bytes, +} + +/// The place where the backend of one save keeps its snapshot file. +#[derive(Debug, Default)] +pub(super) struct SnapshotStage(Mutex>); + +impl SnapshotStage { + /// Keeps the file. A stage holds one file, so a second file gives it back as the error. + pub(super) fn keep(&self, staged: StagedSnapshot) -> Result<(), StagedSnapshot> { + let mut slot = self.0.lock().unwrap_or_else(PoisonError::into_inner); + match *slot { + Some(_) => Err(staged), + None => { + *slot = Some(staged); + Ok(()) + } + } + } + + /// Takes the file out of the stage. + pub(super) fn take(&self) -> Option { + self.0.lock().unwrap_or_else(PoisonError::into_inner).take() + } +} + +/// Writes the staged file only when its path has no blob, which makes the snapshot visible. The +/// name is the hash of the content, so a blob at the path is this file. A failed write deletes the +/// path before the error returns, and a dropped write deletes it in a task of `tracker`. The guard +/// stays armed until that delete ends, so a publish that is dropped during the delete also deletes +/// the path in a task of `tracker`. +pub(super) async fn publish( + files: &SnapshotFiles, + staged: &StagedSnapshot, + tracker: &TaskTracker, +) -> anyhow::Result<()> { + let mut retraction = RetractOnDrop { + files: files.clone(), + path: staged.path.clone(), + tracker: tracker.clone(), + armed: true, + }; + let written = files + .put_if_absent("publish", &staged.path, &staged.content) + .await; + if let Err(error) = written { + // A write that lost its answer can have landed. + retract_or_warn(files, &staged.path).await; + retraction.armed = false; + return Err(error); + } + retraction.armed = false; + Ok(()) +} + +/// Deletes the snapshot file at the path. A path without a blob gives success. +pub(super) async fn retract(files: &SnapshotFiles, path: &Path) -> anyhow::Result<()> { + files.delete("retract", path).await +} + +async fn retract_or_warn(files: &SnapshotFiles, path: &Path) { + if let Err(error) = retract(files, path).await { + warn!( + path = %path.display(), + error = %format!("{error:#}"), + "Failed to delete a filesystem snapshot file whose publish did not finish" + ); + } +} + +/// Deletes the path in a task of the tracker when it is dropped while it is armed. +struct RetractOnDrop { + files: SnapshotFiles, + path: Arc, + tracker: TaskTracker, + armed: bool, +} + +impl Drop for RetractOnDrop { + fn drop(&mut self) { + if !self.armed { + return; + } + let files = self.files.clone(); + let path = self.path.clone(); + match Handle::try_current() { + Ok(runtime) => { + self.tracker.spawn_on( + async move { retract_or_warn(&files, &path).await }, + &runtime, + ); + } + Err(_) => warn!( + path = %path.display(), + "Failed to delete a dropped filesystem snapshot file, because no runtime runs" + ), + } + } +} + +#[cfg(test)] +mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs new file mode 100644 index 0000000000..c787723cd2 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -0,0 +1,239 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use super::super::files::SnapshotFiles; +use super::super::tests::holding::reached_deadline; +use super::super::tests::scripted::{Script, ScriptedBlobStorage}; +use super::{SnapshotStage, StagedSnapshot, publish, retract}; +use bytes::Bytes; +use futures::FutureExt; +use golem_common::model::environment::EnvironmentId; +use golem_service_base::storage::blob::memory::InMemoryBlobStorage; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use pretty_assertions::assert_eq; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; +use test_r::{test, timeout}; +use tokio_util::task::TaskTracker; +use uuid::Uuid; + +/// The longest time that a test waits for the tasks of a tracker. +const LIMIT: Duration = Duration::from_secs(10); + +const SNAPSHOT_PATH: &str = + "snapshots/cdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcd"; + +fn staged() -> StagedSnapshot { + StagedSnapshot { + path: Arc::from(Path::new(SNAPSHOT_PATH)), + content: Bytes::from_static(b"snapshot"), + } +} + +/// Gives the snapshot files of a new namespace over a storage whose script for the publish is +/// `publish` and for the delete is `retract`, and the in-memory storage below it. +fn files( + publish: Script, + retract: Script, + deadline: Duration, +) -> ( + SnapshotFiles, + Arc, + Arc, +) { + let inner = Arc::new(InMemoryBlobStorage::new()); + let storage = ScriptedBlobStorage::new(inner.clone(), move |op_label, _| match op_label { + "publish" => publish, + "retract" => retract, + _ => Script::Pass, + }); + ( + SnapshotFiles { + storage: storage.clone(), + namespace: BlobStorageNamespace::InitialAgentFiles { + environment_id: EnvironmentId(Uuid::new_v4()), + }, + deadline, + cancel: tokio_util::sync::CancellationToken::new(), + tracker: tokio_util::task::TaskTracker::new(), + }, + storage, + inner, + ) +} + +/// Gives the content of the snapshot file, when the storage holds it. +async fn stored(files: &SnapshotFiles, inner: &InMemoryBlobStorage) -> Option> { + inner + .get_raw( + "test", + "test", + files.namespace.clone(), + Path::new(SNAPSHOT_PATH), + ) + .await + .unwrap() +} + +#[test] +#[timeout("60s")] +async fn a_publish_writes_the_staged_file() { + let (files, _, inner) = files(Script::Pass, Script::Pass, Duration::from_secs(2)); + + let published = publish(&files, &staged(), &TaskTracker::new()).await; + + assert_eq!( + (published.is_ok(), stored(&files, &inner).await), + (true, Some(b"snapshot".to_vec())) + ); +} + +#[test] +#[timeout("60s")] +async fn a_publish_of_a_file_that_is_there_succeeds_and_keeps_the_file() { + let (files, _, inner) = files(Script::Pass, Script::Pass, Duration::from_secs(2)); + let tracker = TaskTracker::new(); + + let first = publish(&files, &staged(), &tracker).await; + let second = publish(&files, &staged(), &tracker).await; + + assert_eq!( + (first.is_ok(), second.is_ok(), stored(&files, &inner).await), + (true, true, Some(b"snapshot".to_vec())) + ); +} + +#[test] +#[timeout("60s")] +async fn a_publish_whose_answer_is_lost_deletes_the_file_and_gives_the_error() { + let (files, storage, inner) = + files(Script::LoseTheAnswer, Script::Pass, Duration::from_secs(2)); + + let published = publish(&files, &staged(), &TaskTracker::new()).await; + + assert_eq!( + ( + published.map_err(|error| error.to_string()), + stored(&files, &inner).await, + storage.calls(), + ), + ( + Err("the answer of the call was lost".to_string()), + None, + vec![ + ("publish", SNAPSHOT_PATH.to_string()), + ("retract", SNAPSHOT_PATH.to_string()), + ] + ) + ); +} + +#[test] +#[timeout("60s")] +async fn a_publish_that_reaches_the_deadline_deletes_the_file_that_the_storage_wrote() { + let (files, _, inner) = files( + Script::NeverAnswer, + Script::Pass, + Duration::from_millis(100), + ); + + let published = publish(&files, &staged(), &TaskTracker::new()).await; + + assert_eq!( + ( + published + .as_ref() + .is_err_and(|error| reached_deadline(error.as_ref())), + stored(&files, &inner).await + ), + (true, None) + ); +} + +#[test] +#[timeout("60s")] +async fn a_publish_that_the_caller_drops_deletes_the_file_in_a_task_of_the_tracker() { + // The delete waits for the gate, so the test reads the file that the dropped write left + // before the task of the tracker deletes it. + let (files, storage, inner) = files( + Script::NeverAnswer, + Script::WaitForGate, + Duration::from_secs(60), + ); + let tracker = TaskTracker::new(); + + let dropped = publish(&files, &staged(), &tracker).now_or_never(); + let written_before_the_drop = stored(&files, &inner).await; + storage.open_gate(); + tracker.close(); + let waited = tokio::time::timeout(LIMIT, tracker.wait()).await; + + assert_eq!( + ( + dropped.is_none(), + written_before_the_drop, + waited.is_ok(), + stored(&files, &inner).await + ), + (true, Some(b"snapshot".to_vec()), true, None) + ); +} + +#[test] +#[timeout("60s")] +async fn a_publish_that_returns_keeps_the_file_when_the_tasks_of_the_tracker_end() { + let (files, _, inner) = files(Script::Pass, Script::Pass, Duration::from_secs(2)); + let tracker = TaskTracker::new(); + + let published = publish(&files, &staged(), &tracker).await; + tracker.close(); + let waited = tokio::time::timeout(LIMIT, tracker.wait()).await; + + assert_eq!( + ( + published.is_ok(), + waited.is_ok(), + stored(&files, &inner).await + ), + (true, true, Some(b"snapshot".to_vec())) + ); +} + +#[test] +#[timeout("60s")] +async fn a_retract_of_a_path_without_a_file_succeeds() { + let (files, _, _) = files(Script::Pass, Script::Pass, Duration::from_secs(2)); + + assert!(retract(&files, Path::new(SNAPSHOT_PATH)).await.is_ok()); +} + +#[test] +fn a_stage_keeps_one_file_until_it_is_taken() { + let stage = SnapshotStage::default(); + let other = StagedSnapshot { + content: Bytes::from_static(b"other"), + ..staged() + }; + + let first = stage.keep(staged()); + let second = stage.keep(other.clone()); + let taken = stage.take(); + let taken_again = stage.take(); + + assert_eq!( + (first, second, taken, taken_again), + (Ok(()), Err(other), Some(staged()), None) + ); +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/mod.rs new file mode 100644 index 0000000000..17365f4996 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/mod.rs @@ -0,0 +1,94 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The copy and the delete of a whole scope, on the blobs of its repository. +//! +//! These operations do not read the repository format. They only know the directories of the +//! repository, its config file, and the ledger directory of the store. + +use super::backend::CONFIG_PATH; +use super::files::SnapshotFiles; +use super::prune::LEDGERS_PATH; +use futures::{StreamExt, TryStreamExt, stream}; +use golem_service_base::storage::blob::PutIfAbsent; +use std::path::Path; + +/// The directories of a repository in the order of a listing. A save writes them in the reverse +/// order, and so does a copy, so a snapshot file always has its data. +const LISTING_ORDER: [&str; 4] = [SNAPSHOTS_PATH, "index", "keys", "data"]; + +/// The directory of the snapshot files of a repository. +const SNAPSHOTS_PATH: &str = "snapshots"; + +/// Copies the repository of `from` into the empty scope `to`, with the config file last, so `to` +/// holds a repository only when all its blobs are there. It copies nothing without a config file, +/// and it does not copy the ledger or a blob that a delete removes after the listing. +pub(super) async fn copy_scope(from: &SnapshotFiles, to: &SnapshotFiles) -> anyhow::Result<()> { + let Some(config) = from.get("copy_read", Path::new(CONFIG_PATH)).await? else { + return Ok(()); + }; + let listed = stream::iter(LISTING_ORDER) + .then(|directory| from.list_below("copy_list", Path::new(directory))) + .try_collect::>() + .await? + .into_boxed_slice(); + let paths = listed + .iter() + .rev() + .flat_map(|blobs| blobs.iter().map(|blob| blob.path.clone())) + .collect::>(); + stream::iter(paths.iter().map(Ok)) + .try_for_each(|path| copy_blob(from, to, path)) + .await?; + to.put_if_absent("copy_write", Path::new(CONFIG_PATH), &config) + .await + .map(|_: PutIfAbsent| ()) +} + +async fn copy_blob(from: &SnapshotFiles, to: &SnapshotFiles, path: &Path) -> anyhow::Result<()> { + match from.get("copy_read", path).await? { + Some(content) => to.put("copy_write", path, &content).await, + // A delete removed the snapshot file after the listing, so the copy leaves it out. + None if path.starts_with(SNAPSHOTS_PATH) => Ok(()), + // A prune removed a file that the copy listed, so the copy can be incomplete. It fails + // before the config write, so the target holds no repository, and a new copy can succeed. + None => Err(anyhow::anyhow!( + "the blob {} that the copy listed is gone, because a prune removed it", + path.display() + )), + } +} + +/// Deletes the repository and the ledger of the scope. The config file goes first, so the scope +/// holds no repository from that step on. A scope that holds nothing gives success. +pub(super) async fn delete_scope(files: &SnapshotFiles) -> anyhow::Result<()> { + files.delete("delete_scope", Path::new(CONFIG_PATH)).await?; + stream::iter( + LISTING_ORDER + .iter() + .map(Path::new) + .chain(Path::new(LEDGERS_PATH).parent()) + .map(Ok), + ) + .try_for_each(|directory| async move { + files + .delete_dir("delete_scope", directory) + .await + .map(|_| ()) + }) + .await +} + +#[cfg(test)] +mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs new file mode 100644 index 0000000000..6cd6dd06f3 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs @@ -0,0 +1,302 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use super::super::files::SnapshotFiles; +use super::super::tests::scripted::{Script, ScriptedBlobStorage}; +use super::{copy_scope, delete_scope}; +use golem_common::model::environment::EnvironmentId; +use golem_service_base::storage::blob::memory::InMemoryBlobStorage; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use pretty_assertions::assert_eq; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; +use test_r::{test, timeout}; +use uuid::Uuid; + +const DEADLINE: Duration = Duration::from_secs(2); + +/// The blobs of a small repository, with the ledger of the store. +const REPOSITORY: [(&str, &str); 6] = [ + ("config", "config"), + ("data/ab/abab", "pack"), + ("golem/prune-ledgers/1000-0-0f0f", "ledger"), + ("index/cdcd", "index"), + ("keys/efef", "key"), + ("snapshots/0101", "snapshot"), +]; + +fn files( + storage: &Arc, + namespace: &BlobStorageNamespace, +) -> SnapshotFiles { + SnapshotFiles { + storage: storage.clone(), + namespace: namespace.clone(), + deadline: DEADLINE, + cancel: tokio_util::sync::CancellationToken::new(), + tracker: tokio_util::task::TaskTracker::new(), + } +} + +fn new_namespace() -> BlobStorageNamespace { + BlobStorageNamespace::InitialAgentFiles { + environment_id: EnvironmentId(Uuid::new_v4()), + } +} + +async fn put_all( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, + blobs: &[(&str, &str)], +) { + futures::future::join_all(blobs.iter().map(|(path, content)| { + storage.put_raw( + "test", + "test", + namespace.clone(), + Path::new(path), + content.as_bytes(), + ) + })) + .await + .into_iter() + .collect::>>() + .unwrap(); +} + +/// Gives the path and the content of each blob of the namespace, in the order of the paths. +async fn stored( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, +) -> Vec<(String, String)> { + let listed = storage + .list_blobs_below("test", "test", namespace.clone(), Path::new("")) + .await + .unwrap(); + let mut blobs = futures::future::join_all(listed.iter().map(|blob| async { + let content = storage + .get_raw("test", "test", namespace.clone(), &blob.path) + .await + .unwrap() + .unwrap(); + ( + blob.path.display().to_string(), + String::from_utf8(content).unwrap(), + ) + })) + .await; + blobs.sort(); + blobs +} + +fn owned(blobs: &[(&str, &str)]) -> Vec<(String, String)> { + blobs + .iter() + .map(|(path, content)| (path.to_string(), content.to_string())) + .collect() +} + +#[test] +#[timeout("60s")] +async fn a_copy_gives_the_target_each_blob_of_the_repository_and_not_the_ledger() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let (from, to) = (new_namespace(), new_namespace()); + put_all(&*storage, &from, &REPOSITORY).await; + + copy_scope(&files(&storage, &from), &files(&storage, &to)) + .await + .unwrap(); + + assert_eq!( + (stored(&*storage, &to).await, stored(&*storage, &from).await), + ( + owned( + &REPOSITORY + .into_iter() + .filter(|(path, _)| !path.starts_with("golem/")) + .collect::>() + ), + owned(&REPOSITORY) + ) + ); +} + +#[test] +#[timeout("60s")] +async fn a_copy_writes_the_packs_the_keys_the_index_files_the_snapshot_files_and_then_the_config() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let (from, to) = (new_namespace(), new_namespace()); + put_all(&*storage, &from, &REPOSITORY).await; + + copy_scope(&files(&storage, &from), &files(&storage, &to)) + .await + .unwrap(); + + assert_eq!( + storage + .calls() + .into_iter() + .filter(|(op_label, _)| *op_label == "copy_write") + .map(|(_, path)| path) + .collect::>(), + vec![ + "data/ab/abab", + "keys/efef", + "index/cdcd", + "snapshots/0101", + "config" + ] + ); +} + +#[test] +#[timeout("60s")] +async fn a_copy_lists_the_snapshot_files_before_the_index_files_the_keys_and_the_packs() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let (from, to) = (new_namespace(), new_namespace()); + put_all(&*storage, &from, &REPOSITORY).await; + + copy_scope(&files(&storage, &from), &files(&storage, &to)) + .await + .unwrap(); + + assert_eq!( + storage + .calls() + .into_iter() + .filter(|(op_label, _)| *op_label == "copy_list") + .map(|(_, path)| path) + .collect::>(), + vec!["snapshots", "index", "keys", "data"] + ); +} + +#[test] +#[timeout("60s")] +async fn a_copy_of_a_namespace_without_a_config_copies_nothing() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let (from, to) = (new_namespace(), new_namespace()); + put_all( + &*storage, + &from, + &REPOSITORY + .into_iter() + .filter(|(path, _)| *path != "config") + .collect::>(), + ) + .await; + + copy_scope(&files(&storage, &from), &files(&storage, &to)) + .await + .unwrap(); + + assert_eq!(stored(&*storage, &to).await, Vec::<(String, String)>::new()); +} + +#[test] +#[timeout("60s")] +async fn a_copy_that_fails_gives_the_error_and_the_target_has_no_config() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "copy_read" && path.starts_with("index") { + Script::Refuse + } else { + Script::Pass + } + }); + let (from, to) = (new_namespace(), new_namespace()); + put_all(&*storage, &from, &REPOSITORY).await; + + let copied = copy_scope(&files(&storage, &from), &files(&storage, &to)).await; + + assert_eq!( + ( + copied.is_err(), + stored(&*storage, &to) + .await + .into_iter() + .map(|(path, _)| path) + .collect::>() + ), + ( + true, + vec!["data/ab/abab".to_string(), "keys/efef".to_string()] + ) + ); +} + +#[test] +#[timeout("60s")] +async fn a_deleted_scope_holds_no_blob_and_another_scope_keeps_its_blobs() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let (deleted, kept) = (new_namespace(), new_namespace()); + put_all(&*storage, &deleted, &REPOSITORY).await; + put_all(&*storage, &kept, &REPOSITORY).await; + + delete_scope(&files(&storage, &deleted)).await.unwrap(); + + assert_eq!( + ( + stored(&*storage, &deleted).await, + stored(&*storage, &kept).await + ), + (Vec::new(), owned(&REPOSITORY)) + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_of_a_scope_deletes_the_config_first() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let namespace = new_namespace(); + put_all(&*storage, &namespace, &REPOSITORY).await; + + delete_scope(&files(&storage, &namespace)).await.unwrap(); + + assert_eq!( + storage + .calls() + .into_iter() + .filter(|(op_label, _)| *op_label == "delete_scope") + .map(|(_, path)| path) + .collect::>(), + vec!["config", "snapshots", "index", "keys", "data", "golem"] + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_of_an_unused_scope_succeeds_and_can_run_again() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let namespace = new_namespace(); + + let first = delete_scope(&files(&storage, &namespace)).await; + put_all(&*storage, &namespace, &REPOSITORY).await; + let second = delete_scope(&files(&storage, &namespace)).await; + let third = delete_scope(&files(&storage, &namespace)).await; + + assert_eq!( + ( + first.is_ok(), + second.is_ok(), + third.is_ok(), + stored(&*storage, &namespace).await + ), + (true, true, true, Vec::new()) + ); +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/mod.rs new file mode 100644 index 0000000000..bd86467436 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/mod.rs @@ -0,0 +1,1383 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The filesystem snapshot store over the rustic repositories of the scopes. +//! +//! A save makes the snapshot visible in one step: the backend keeps the snapshot file of the +//! backup, and the save writes that file after the blocking work returns. Each operation has a +//! cancellation token that its drop cancels, so the threads of a dropped operation stop at their +//! next storage call. The store counts each blocking task and each backend in a task tracker, and +//! [`RusticSnapshotStore::shut_down`] waits for them. + +use super::backend::{BlobBackend, Lease, within_lease}; +use super::fault::{ + Operation, classify, is_file_missing, is_snapshot_missing, is_storage_failure, storage_failure, +}; +use super::files::SnapshotFiles; +use super::priority::LowPriority; +use super::prune::{ + ClaimChoice, Percent, claim_hold, claims_directory, hold_passed, keep_claim_fresh, lease_span, + list_claims, list_freed_names, marker_path, marker_time, may_be_due, named_bytes, + needs_repository_size, next_claim, prune_due, read_ledger, record_freed, refresh_period, + release_claim, remove_freed, remove_old_claims, remove_older_ledgers, repository_bytes, + settle_freed, take_claim, write_ledger, write_marker, +}; +use super::publish::{SnapshotStage, StagedSnapshot, publish}; +use super::scope::{copy_scope, delete_scope}; +use super::{ + ChangeDetection as RusticChangeDetection, PruneReport, PruneSettings, RepackLimits, + RepositoryKey, RepositorySettings, SaveSettings, backup_options, open_existing, open_or_create, + prune, restore_snapshot, run_blocking, +}; +use crate::filesystem_snapshot::{ + ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, + SnapshotStoreError, newest_first, snapshot_time, +}; +use crate::sandbox_filesystem::{NativeOperation, NativeStorageProfile, execute_native}; +use crate::services::golem_config::FilesystemSnapshotStoreConfig; +use anyhow::Context; +use async_trait::async_trait; +use futures::future::{self, Either}; +use golem_common::model::Timestamp; +use golem_service_base::storage::blob::BlobStorage; +use rustic_core::jiff::tz::TimeZone; +use rustic_core::jiff::{Timestamp as SnapshotTime, Zoned}; +use rustic_core::repofile::{SnapshotFile, SnapshotId}; +use rustic_core::{ + BackupOptions, DevIdOption, LocalSourceSaveOptions, Open, PathList, + Repository as RusticRepository, RestoreOptions, SnapshotOptions, +}; +use serde::{Deserialize, Serialize}; +use std::num::NonZeroUsize; +use std::path::Path; +use std::pin::pin; +use std::sync::atomic::{AtomicBool, AtomicU8, Ordering}; +use std::sync::{Arc, Mutex, PoisonError}; +use std::time::Duration; +use tokio::runtime::{Handle, TryCurrentError}; +use tokio_util::sync::{CancellationToken, DropGuard}; +use tokio_util::task::TaskTracker; +use tokio_util::task::task_tracker::TaskTrackerToken; +use tracing::warn; + +/// The share of the size of the repository that deleted snapshots must free before a delete prunes +/// the scope. The threshold is 10% of the size, rounded down to a whole byte, so about 10%. +const PRUNE_THRESHOLD: Percent = Percent(10); + +/// How long a pack that a prune marks stays before a later prune deletes it. It must be longer than +/// the longest save and the longest restore. Two prunes of one scope never run at once. The next +/// prune waits a full hold from the newest claim marker that was written. The hold follows from +/// this time: this time less one storage call deadline, but at least one deadline, plus the margin +/// for clock skew, plus two deadlines. When writes fail, only the gap between two prunes can be +/// shorter. +const PRUNE_GRACE: Duration = Duration::from_secs(15 * 60); + +/// The settings of the store: the rustic settings of each operation, and the prune threshold. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) struct StorePolicy { + /// The longest time that one blob storage call waits for an answer. + pub(super) deadline: Duration, + /// The settings of a repository that a save makes. + pub(super) repository: RepositorySettings, + /// The number of threads of each parallel stage of a save. `None` is the number of CPUs that + /// the process can use. + pub(super) save_threads: Option, + /// The number of threads that read packs in a restore. + pub(super) restore_reader_threads: NonZeroUsize, + /// The settings of a prune. Two prunes never run at once. The next prune waits a full hold from + /// the newest claim marker that was written. The hold follows from `keep_delete`: that time + /// less one storage call deadline, but at least one deadline, plus the margin for clock skew, + /// plus two deadlines. When writes fail, only the gap between two prunes can be shorter. + pub(super) prune: PruneSettings, + /// The share of the size of the repository that deleted snapshots must free before a delete + /// prunes. + pub(super) prune_threshold: Percent, +} + +impl StorePolicy { + /// Gives the policy with the values of the configuration. + pub(super) fn from_config(config: &FilesystemSnapshotStoreConfig) -> Self { + Self { + deadline: config.storage_call_deadline(), + repository: RepositorySettings::DEFAULT, + save_threads: Some(config.save_threads()), + restore_reader_threads: config.restore_reader_threads(), + prune: PruneSettings { + fast_repack: true, + keep_delete: PRUNE_GRACE, + repack: RepackLimits::Rustic, + }, + prune_threshold: PRUNE_THRESHOLD, + } + } +} + +/// The options of a save of the store: a failed read of an entry fails the save, and no device id +/// is kept. `SizeMtime` compares with the parent that the id names, and each other case reads +/// every file. +fn store_backup_options( + policy: &StorePolicy, + parent: Option<(SnapshotId, ChangeDetection)>, +) -> BackupOptions { + let base = |detection| { + backup_options(&SaveSettings { + threads: policy.save_threads, + detection, + }) + .fail_on_read_error(true) + .ignore_save_opts(LocalSourceSaveOptions::default().set_devid(DevIdOption::No)) + }; + match parent { + Some((id, ChangeDetection::SizeMtime)) => { + let options = base(RusticChangeDetection::SizeMtime); + let parent_opts = options + .parent_opts + .clone() + .parents(vec![id.to_hex().to_string()]); + options.parent_opts(parent_opts) + } + None | Some((_, ChangeDetection::Full)) => { + let options = base(RusticChangeDetection::Ctime); + let parent_opts = options.parent_opts.clone().force(true); + options.parent_opts(parent_opts) + } + } +} + +/// The options of a restore of the store. A metadata error fails the restore. The restore does not +/// set the owner, because a snapshot does not keep it. +fn store_restore_options(policy: &StorePolicy) -> RestoreOptions { + RestoreOptions::default() + .reader_threads(Some(policy.restore_reader_threads)) + .fail_on_metadata_error(true) + .no_ownership(true) +} + +/// A filesystem snapshot store that keeps one rustic repository for each scope in blob storage. +pub(crate) struct RusticSnapshotStore { + storage: Arc, + key: RepositoryKey, + policy: StorePolicy, + /// The parent of the token of each operation. + root: CancellationToken, + /// Counts the blocking tasks, the checks of local paths, the backends, the blob calls of the + /// store, the publishes, the deletes of dropped publishes, and the claim guards with their + /// release and final-marker tasks. + tracker: TaskTracker, + /// Runs saves and prunes at a low priority. + low_priority: LowPriority, + /// Holds a save after its blocking work and before its publish, when a test sets it. + #[cfg(test)] + pub(super) publish_gate: Option>, + /// Holds a save after the tracker counts its publish and before the first poll of the publish, + /// when a test sets it. + #[cfg(test)] + pub(super) publish_poll_gate: Option>, + /// Holds a delete after it chose its claim and before it builds its claim guard, when a test + /// sets it. + #[cfg(test)] + pub(super) claim_gate: Option>, + /// Holds the check of the local path of a save or a restore on its blocking thread, when a + /// test sets it. + #[cfg(test)] + pub(super) path_check_gate: Option>, + /// Makes each backend build fail while a test sets it. + #[cfg(test)] + pub(super) refuse_backends: Arc, + /// The number of milliseconds that the clock of the prune decisions is ahead of the wall + /// clock. A test moves it to make time pass. + #[cfg(test)] + pub(super) clock_ahead: Arc, +} + +/// A gate that holds an operation at one point, for example a save after its blocking work and +/// before its publish. +#[cfg(test)] +#[derive(Debug, Default)] +pub(super) struct StepGate { + /// Notified when the operation reaches the gate. + pub(super) reached: tokio::sync::Notify, + /// Lets the operation go on. + pub(super) open: tokio::sync::Notify, +} + +/// The claim of a prune that a delete holds. +struct Claim { + /// The lease of the prune. Each marker write that succeeds moves its end. + lease: Arc, + /// The time that a marker write that succeeds adds to the lease. + span: Duration, + /// Holds the directory and the number of the claim. It releases the claim when the delete + /// stops before its prune starts, and writes the final marker of a prune that started. + guard: ClaimGuard, +} + +/// The claim is written, and its prune did not start. +const CLAIM_PENDING: u8 = 0; +/// The rustic prune of the claim started, so the claim stays, and it needs its final marker. +const CLAIM_STARTED: u8 = 1; +/// The claim was released, so its prune must not start. +const CLAIM_RELEASED: u8 = 2; +/// The final marker of the prune that started is written. +const CLAIM_FINISHED: u8 = 3; + +/// Holds a prune claim from the write of its first marker until the final marker of its prune. +/// When the delete drops before the prune starts, the guard releases the claim in a task. When the +/// delete drops after the prune started and before the final marker is written, the guard writes +/// the final marker in a task. The guard moves its token of the tracker into that task, so +/// `shut_down` waits for it. The prune and the guard change the state from pending with one atomic +/// step each, so only one of them wins: a released claim never starts a prune, and a drop after the +/// prune started does not release the claim. After the start, only the delete releases the claim, +/// when each attempt of the prune found a snapshot file gone, because such a prune changed nothing +/// and counts as a prune that did not run. +struct ClaimGuard { + /// The blobs of the scope, with a token that nothing cancels, so a release and a final marker + /// also run after a cancel or a drop. + files: SnapshotFiles, + directory: Arc, + number: u64, + /// Whether this delete wrote the claim. A delete that did not write it deletes only its + /// markers. + claimed: AtomicBool, + /// The markers of the claim that this delete wrote: the first marker, and each new marker while + /// the prune runs. + markers: Mutex>>, + state: Arc, + tracked: Mutex>, +} + +impl ClaimGuard { + /// Gives the guard of the claim with the number and the path of its first marker, before the + /// write of that marker. + fn new( + files: &SnapshotFiles, + directory: &Arc, + number: u64, + marker: Box, + tracked: TaskTrackerToken, + ) -> Self { + Self { + files: SnapshotFiles { + cancel: CancellationToken::new(), + ..files.clone() + }, + directory: directory.clone(), + number, + claimed: AtomicBool::new(false), + markers: Mutex::new(vec![marker]), + state: Arc::new(AtomicU8::new(CLAIM_PENDING)), + tracked: Mutex::new(Some(tracked)), + } + } + + /// Spawns the work on the runtime, and moves the token of the tracker into it. + fn spawn_tracked( + &self, + work: impl Future + Send + 'static, + ) -> Result, TryCurrentError> { + let tracked = self + .tracked + .lock() + .unwrap_or_else(PoisonError::into_inner) + .take(); + Handle::try_current().map(|runtime| { + runtime.spawn(async move { + let _tracked = tracked; + work.await; + }) + }) + } + + /// Gives the directory of the claims of the ledger of the claim. + fn directory(&self) -> &Path { + &self.directory + } + + /// Gives the number of the claim. + fn number(&self) -> u64 { + self.number + } + + /// Gives the markers of the claim that this delete wrote. + fn markers(&self) -> &Mutex>> { + &self.markers + } + + /// Gives the state of the claim, which the prune shares with the guard. + fn state(&self) -> Arc { + self.state.clone() + } + + /// Records that this delete wrote the claim, so a release deletes it. + fn mark_claimed(&self) { + self.claimed.store(true, Ordering::SeqCst); + } + + /// Makes the release task, and moves the token of the tracker into it. + fn spawn_release(&self) -> Option> { + let files = self.files.clone(); + let directory = self.directory.clone(); + let number = self.number; + let claimed = self.claimed.load(Ordering::SeqCst); + let markers = self + .markers + .lock() + .unwrap_or_else(PoisonError::into_inner) + .clone(); + self.spawn_tracked(async move { + release_claim(&files, &directory, number, claimed, &markers).await; + }) + .inspect_err(|error| { + warn!( + error = %error, + "The prune claim of a filesystem snapshot scope stays, because no runtime runs its release" + ); + }) + .ok() + } + + /// Makes the task that writes the final marker, and moves the token of the tracker into it. + fn spawn_final_marker(&self) { + let files = self.files.clone(); + let directory = self.directory.clone(); + let number = self.number; + if let Err(error) = self.spawn_tracked(async move { + write_final_marker(&files, &directory, number).await; + }) { + warn!( + error = %error, + "The prune claim of a filesystem snapshot scope gets no final marker, because no runtime runs its write" + ); + } + } + + /// Writes the final marker of a prune that started, and then marks the guard finished. A guard + /// whose prune did not start writes nothing. When the write fails, the drop of the guard tries + /// it again. + async fn finish(&self) { + if self.state.load(Ordering::SeqCst) == CLAIM_STARTED + && write_final_marker(&self.files, &self.directory, self.number).await + { + self.state.store(CLAIM_FINISHED, Ordering::SeqCst); + } + } + + /// Releases the claim and waits for the release. + async fn release(&self) { + self.state.store(CLAIM_RELEASED, Ordering::SeqCst); + if let Some(releasing) = self.spawn_release() + && let Err(error) = releasing.await + { + warn!( + error = %error, + "The release of the prune claim of a filesystem snapshot scope did not end" + ); + } + } + + /// Ends the guard without a release: another delete took the claim, and this delete already + /// deleted its marker. + fn disarm(&self) { + self.state.store(CLAIM_RELEASED, Ordering::SeqCst); + drop( + self.tracked + .lock() + .unwrap_or_else(PoisonError::into_inner) + .take(), + ); + } +} + +impl Drop for ClaimGuard { + fn drop(&mut self) { + let moved = |from, to| { + self.state + .compare_exchange(from, to, Ordering::SeqCst, Ordering::SeqCst) + .is_ok() + }; + if moved(CLAIM_PENDING, CLAIM_RELEASED) { + drop(self.spawn_release()); + } else if moved(CLAIM_STARTED, CLAIM_FINISHED) { + self.spawn_final_marker(); + } + } +} + +/// Writes the final marker of the claim with the number, and tells whether the write succeeded. A +/// failed write gives a warning. +async fn write_final_marker(files: &SnapshotFiles, directory: &Path, number: u64) -> bool { + write_marker( + files, + "final_marker", + directory, + number, + Timestamp::now_utc(), + ) + .await + .inspect_err(|error| { + warn!( + error = %format!("{error:#}"), + "Failed to write the final marker of the prune claim of a filesystem snapshot scope" + ); + }) + .is_ok() +} + +/// Moves the claim of the state from pending to started, and tells whether the prune may start. +fn start_prune(state: &AtomicU8) -> bool { + state + .compare_exchange( + CLAIM_PENDING, + CLAIM_STARTED, + Ordering::SeqCst, + Ordering::SeqCst, + ) + .is_ok() +} + +/// The number of times that a prune plans again when a snapshot file that it listed is gone at +/// its read. +const PRUNE_ATTEMPTS: usize = 3; + +/// The error of a delete whose prune found a snapshot file gone at each attempt. A concurrent +/// delete removed the files, so a retry of the delete can prune. +fn snapshots_changed_error() -> SnapshotStoreError { + SnapshotStoreError::Storage { + retryable: true, + source: anyhow::anyhow!( + "a concurrent delete removed a snapshot file at each attempt of the prune" + ), + } +} + +/// The error of an operation of a store that is shut down. +fn shut_down_error() -> SnapshotStoreError { + SnapshotStoreError::Storage { + retryable: false, + source: anyhow::anyhow!("the filesystem snapshot store is shut down"), + } +} + +impl RusticSnapshotStore { + /// Gives the store over the blob storage, with the key and the values of the configuration. + pub(crate) fn new( + storage: Arc, + config: &FilesystemSnapshotStoreConfig, + ) -> Self { + Self::with_policy( + storage, + RepositoryKey::new(*config.repository_key().bytes()), + StorePolicy::from_config(config), + ) + } + + pub(super) fn with_policy( + storage: Arc, + key: RepositoryKey, + policy: StorePolicy, + ) -> Self { + // The global rayon pool starts at its first use, and its threads keep the priority of the + // thread that starts it. It starts here, at the normal priority, before a save or a prune. + let _ = rayon::current_num_threads(); + Self { + storage, + key, + policy, + root: CancellationToken::new(), + tracker: TaskTracker::new(), + low_priority: LowPriority::new(policy.save_threads), + #[cfg(test)] + publish_gate: None, + #[cfg(test)] + publish_poll_gate: None, + #[cfg(test)] + claim_gate: None, + #[cfg(test)] + path_check_gate: None, + #[cfg(test)] + refuse_backends: Arc::default(), + #[cfg(test)] + clock_ahead: Arc::default(), + } + } + + /// Cancels each operation, so each running storage call ends and no new call starts, and later + /// operations give `Storage`. Some calls run after the cancel by design, because no cancel + /// ends them: a publish that started before the cancel runs to its end, and a claim guard + /// releases its claim, or writes the final marker of a prune that started. A save that + /// reaches its publish after the cancel publishes nothing and gives `Storage`. The call waits + /// until no blocking task, check of a local path, backend, blob call of the store, publish, + /// delete of a dropped publish, claim guard, or release or final marker of a claim guard + /// remains. A blob call that is not polled holds the wait until it is polled again, and then + /// it ends at once. The runtime must not drop before it returns, because a storage call after + /// its time driver stops aborts the process. + pub(crate) async fn shut_down(&self) { + self.root.cancel(); + self.tracker.close(); + self.tracker.wait().await; + } + + /// Gives the number of blocking tasks, checks of local paths, backends, blob calls of the store, + /// publishes, deletes of dropped publishes, and claim guards with their release and + /// final-marker tasks that have not ended. + #[cfg(test)] + pub(super) fn work_in_flight(&self) -> usize { + self.tracker.len() + } + + /// Gives the time now, for a comparison with a time from storage. + fn now(&self) -> Timestamp { + let now = Timestamp::now_utc(); + #[cfg(test)] + let now = Timestamp::from( + now.to_millis() + self.clock_ahead.load(std::sync::atomic::Ordering::SeqCst), + ); + now + } + + /// Starts an operation. The token of the operation is cancelled when the guard drops. + fn start(&self) -> Result<(CancellationToken, DropGuard), SnapshotStoreError> { + if self.root.is_cancelled() { + return Err(shut_down_error()); + } + let token = self.root.child_token(); + Ok((token.clone(), token.drop_guard())) + } + + /// Gives a backend over the repository of the scope for the operation with the token. + fn backend( + &self, + scope: &SnapshotScope, + token: &CancellationToken, + ) -> Result { + #[cfg(test)] + if self + .refuse_backends + .load(std::sync::atomic::Ordering::SeqCst) + { + return Err(SnapshotStoreError::Storage { + retryable: true, + source: anyhow::anyhow!("the test refuses to build a backend"), + }); + } + let runtime = Handle::try_current() + .context("a filesystem snapshot operation needs an async runtime") + .map_err(|source| SnapshotStoreError::Storage { + retryable: false, + source, + })?; + Ok(BlobBackend::new( + self.storage.clone(), + scope.0.clone(), + runtime, + self.policy.deadline, + ) + .cancelled_by(token.clone()) + .tracked_by(self.tracker.token())) + } + + /// Runs the task on a blocking thread that the tracker counts, and classifies its error. + async fn blocking( + &self, + operation: Operation, + task: impl FnOnce() -> anyhow::Result + Send + 'static, + ) -> Result { + let tracked = self.tracker.token(); + run_blocking(move || { + let _tracked = tracked; + task() + }) + .await + .map_err(|error| classify(operation, error)) + } + + /// Gives the blobs of the scope for the operation with the token. + fn files(&self, scope: &SnapshotScope, token: &CancellationToken) -> SnapshotFiles { + SnapshotFiles { + storage: self.storage.clone(), + namespace: scope.0.clone(), + deadline: self.policy.deadline, + cancel: token.clone(), + tracker: self.tracker.clone(), + } + } + + /// Prunes the repository when a prune is due. The records of freed bytes stay until a prune + /// succeeds, so a later prune counts them again. It lists the packs only when their size can + /// make a prune due. + /// A due prune runs only after the delete takes a claim of its ledger, and only when the ledger + /// did not change after the claim. After an error before the prune ran, the claim is deleted, + /// so a retry of the delete prunes again, and a claim write that the storage completes after + /// that delete can delay that prune by up to the hold of a claim. A prune that found a snapshot + /// file gone at each attempt changed nothing, so it counts as an error before the prune ran. + /// After a prune that started, the claim stays on each outcome, also when the prune or its + /// ledger write fails, or when the lease skips its ledger write. So the next prune waits a full + /// hold from the newest claim marker that was written, which is the end of the prune unless the + /// final marker write failed. Two prunes never run at once; when writes fail, only the gap + /// between them can be shorter. A prune that succeeds deletes each claim of its ledger. + async fn prune_when_due( + &self, + scope: &SnapshotScope, + token: &CancellationToken, + ) -> Result<(), SnapshotStoreError> { + let files = self.files(scope, token); + let grace = self.policy.prune.keep_delete; + let deadline = self.policy.deadline; + let threshold = self.policy.prune_threshold; + let ledger = read_ledger(&files).await.map_err(storage_failure)?; + // Each comparison with a time from storage uses a clock reading from after the listing + // that gave that time. A listing can take up to one storage call deadline, and a stale + // reading can put a marker that another host wrote within the margin beyond the margin. + // Each step below runs only when a prune can still be due, so a delete within the hold + // reads no record, and a delete whose records are too small reads no record content. + if !hold_passed(&ledger, self.now(), grace, deadline) { + return Ok(()); + } + let listed = list_freed_names(&files).await.map_err(storage_failure)?; + let upper = named_bytes(&listed); + if upper == 0 && !ledger.awaiting_removal { + return Ok(()); + } + let size = if needs_repository_size(&ledger, upper, self.now(), grace, deadline) { + repository_bytes(&files).await.map_err(storage_failure)? + } else { + 0 + }; + if !may_be_due(&ledger, upper, size, threshold) { + return Ok(()); + } + let records = settle_freed(&files, &listed) + .await + .map_err(storage_failure)?; + if !prune_due( + &ledger, + records.bytes, + self.now(), + size, + threshold, + grace, + deadline, + ) { + return Ok(()); + } + let claims: Arc = claims_directory(&ledger).into(); + let listed = list_claims(&files, &claims) + .await + .map_err(storage_failure)?; + let ClaimChoice::Claim(number) = + next_claim(&listed, self.now(), claim_hold(grace, deadline)) + else { + return Ok(()); + }; + #[cfg(test)] + if let Some(gate) = &self.claim_gate { + gate.reached.notify_one(); + gate.open.notified().await; + } + // The token of the tracker comes before the check of the cancel, so either the delete sees + // a shut down and makes no more storage calls, or `shut_down` waits for the guard and for + // each call that the guard makes. + let tracked = self.tracker.token(); + if self.root.is_cancelled() { + return Err(shut_down_error()); + } + let span = lease_span(grace, deadline); + let (started, time) = marker_time(); + let marker = marker_path(&claims, number, time); + // The guard exists before the marker write, so a drop of the delete from here until the + // prune starts releases the claim. + let guard = ClaimGuard::new(&files, &claims, number, marker.clone(), tracked); + let lease = take_claim(&files, &claims, number, &marker, started, span) + .await + .map_err(storage_failure)?; + let Some(lease) = lease else { + guard.disarm(); + return Ok(()); + }; + guard.mark_claimed(); + let claim = Claim { + lease: Arc::new(lease), + span, + guard, + }; + // Only an error before the prune starts releases the claim, so a retry of the delete + // prunes again. A prune that started can have marked packs, so its claim stays. + let backend = match self.prepare_prune(scope, token, &files, &claim).await { + Ok(Some(backend)) => backend, + other => { + claim.guard.release().await; + return other.map(|_| ()); + } + }; + // No attempt of a prune that found a snapshot file gone changed the repository, so no + // prune ran, and the claim goes. + let pruned = match self.run_prune(backend, &files, &claim, grace).await { + Ok(None) => { + claim.guard.release().await; + return Err(snapshots_changed_error()); + } + other => other.map(|marked| marked.unwrap_or_default()), + }; + // The final marker holds the claim for the hold from the end of this prune, also when the + // prune or its ledger write fails. It goes through the files of the guard, which no cancel + // ends, so a prune that a shut down stopped also gets it. + claim.guard.finish().await; + let marked_packs = pruned?; + let ended = Timestamp::now_utc(); + // The lease fences the ledger write as it fences the calls of rustic. A write that would + // start after the lease ran out is not sent, and a write that starts before is bounded by + // the time left, so the entry never lands after another delete can take the claim over. + // A prune whose ledger write the lease skipped keeps its claim and runs no cleanup. + within_lease(&claim.lease, write_ledger(&files, ended, marked_packs)) + .await + .map_err(storage_failure)?; + remove_older_ledgers(&files, ended).await; + remove_freed(&files, &records).await; + remove_old_claims(&files, ended).await; + Ok(()) + } + + /// Checks the ledger again and builds the backend of the prune. It gives `None` when another + /// prune ended after the claim. + async fn prepare_prune( + &self, + scope: &SnapshotScope, + token: &CancellationToken, + files: &SnapshotFiles, + claim: &Claim, + ) -> Result>, SnapshotStoreError> { + // A prune writes its ledger before it deletes the claims, so a delete that claims in a + // directory that such a prune removed sees the new ledger here. + let again = read_ledger(files).await.map_err(storage_failure)?; + if *claims_directory(&again) != *claim.guard.directory() { + return Ok(None); + } + Ok(Some(Arc::new( + self.backend(scope, token)?.leased_by(claim.lease.clone()), + ))) + } + + /// Runs the prune with a new marker of the claim at each refresh period, and tells whether the + /// prune leaves marked packs. It gives `None` when each attempt found a snapshot file gone. + async fn run_prune( + &self, + backend: Arc, + files: &SnapshotFiles, + claim: &Claim, + grace: Duration, + ) -> Result, SnapshotStoreError> { + let key = self.key.clone(); + let settings = self.policy.prune; + let low_priority = self.low_priority; + let state = claim.guard.state(); + // The plan of a prune reads each snapshot file before the prune changes the repository. + // A forget of another delete can remove a listed file before its read, so the prune + // plans again from a new listing. + let pruning = self.blocking(Operation::Prune, move || { + low_priority.run("fs-snap-prune", move || { + // A delete that dropped before this point released the claim, so no prune runs. + if !start_prune(&state) { + return Ok(None); + } + Ok( + std::iter::repeat_with(|| prune(backend.clone(), &key, &settings)) + .take(PRUNE_ATTEMPTS) + .find(|attempt| { + !attempt + .as_ref() + .is_err_and(|error| is_snapshot_missing(&**error)) + }), + ) + }) + }); + // The claim gets a new marker while the prune runs, so a prune slower than the grace + // period keeps its claim. The markers stop when the prune ends or the operation is + // cancelled. The tracker counts the whole step, so no timer of it runs after a shut down. + let refreshing = keep_claim_fresh( + files, + claim.guard.directory(), + claim.guard.number(), + refresh_period(grace, self.policy.deadline), + claim.guard.markers(), + &claim.lease, + claim.span, + ); + let attempts = self + .tracker + .track_future(async { + match future::select(pin!(pruning), pin!(refreshing)).await { + Either::Left((pruned, _)) => pruned, + Either::Right(((), pruning)) => pruning.await, + } + }) + .await?; + attempts + .map(|report| { + report + .map(|report| report.as_ref().is_some_and(leaves_marked_packs)) + .map_err(|error| classify(Operation::Prune, error)) + }) + .transpose() + } +} + +#[async_trait] +impl FilesystemSnapshotStore for RusticSnapshotStore { + async fn save( + &self, + scope: &SnapshotScope, + name: &SnapshotName, + tree: &Path, + parent: Option<(&SnapshotName, ChangeDetection)>, + ) -> Result { + let (token, _guard) = self.start()?; + self.check_tree(tree).await?; + let stage = Arc::new(SnapshotStage::default()); + let backend = Arc::new(self.backend(scope, &token)?.staging_in(stage.clone())); + let key = self.key.clone(); + let policy = self.policy; + let name = name.clone(); + let tree: Box = tree.into(); + let parent = parent.map(|(parent, detection)| (parent.clone(), detection)); + let low_priority = self.low_priority; + let staged = self + .blocking(Operation::Save, move || { + low_priority.run("fs-snap-save", move || { + stage_save(backend, &stage, &key, &policy, &name, &tree, parent) + }) + }) + .await?; + let (staged, info) = staged.ok_or(SnapshotStoreError::AlreadyExists)?; + #[cfg(test)] + if let Some(gate) = &self.publish_gate { + gate.reached.notify_one(); + gate.open.notified().await; + } + // The publish is the commit point, so no cancel ends it. The tracker counts it from here, + // so a `shut_down` that has not cancelled yet waits for it, and the deadline limits that wait. + // The check of the cancel and the start of the write are in the first poll of the tracked + // future, so a publish never starts after the cancel. + let files = self.files(scope, &CancellationToken::new()); + let (root, tracker) = (self.root.clone(), self.tracker.clone()); + let publishing = self.tracker.track_future(async move { + if root.is_cancelled() { + // No snapshot file is written. A later prune marks the packs of the save. + return Err(shut_down_error()); + } + publish(&files, &staged, &tracker) + .await + .map_err(storage_failure) + }); + #[cfg(test)] + if let Some(gate) = &self.publish_poll_gate { + gate.reached.notify_one(); + gate.open.notified().await; + } + publishing.await?; + Ok(info) + } + + async fn restore( + &self, + scope: &SnapshotScope, + name: &SnapshotName, + into: &Path, + ) -> Result { + let (token, _guard) = self.start()?; + self.check_destination(into).await?; + let backend = Arc::new(self.backend(scope, &token)?); + let key = self.key.clone(); + let options = store_restore_options(&self.policy); + let name = name.clone(); + let into: Box = into.into(); + // The index load of a restore uses rayon, so the restore runs in a pool of its own, with the + // reader threads of a restore. + let pool = LowPriority { + threads: Some(self.policy.restore_reader_threads), + ..self.low_priority + }; + self.blocking(Operation::Restore, move || { + pool.run_at_normal_priority("fs-snap-restore", move || { + let Some(repository) = open_existing(backend, &key)? else { + return Ok(Lookup::Missing); + }; + match lookup(scope_snapshots(&repository)?, &name) { + Lookup::Found(snapshot, info) => { + restore_snapshot(repository, &snapshot, &into, &options)?; + Ok(Lookup::Found(snapshot, info)) + } + other => Ok(other), + } + }) + }) + .await? + .into_info()? + .ok_or(SnapshotStoreError::NotFound) + } + + async fn stat( + &self, + scope: &SnapshotScope, + name: &SnapshotName, + ) -> Result, SnapshotStoreError> { + let (token, _guard) = self.start()?; + let backend = Arc::new(self.backend(scope, &token)?); + let key = self.key.clone(); + let name = name.clone(); + self.blocking(Operation::Repository, move || { + Ok(match open_existing(backend, &key)? { + Some(repository) => lookup(scope_snapshots(&repository)?, &name), + None => Lookup::Missing, + }) + }) + .await? + .into_info() + } + + async fn list( + &self, + scope: &SnapshotScope, + ) -> Result, SnapshotStoreError> { + let (token, _guard) = self.start()?; + let backend = Arc::new(self.backend(scope, &token)?); + let key = self.key.clone(); + self.blocking(Operation::Repository, move || { + Ok(match open_existing(backend, &key)? { + Some(repository) => newest_first(listed(scope_snapshots(&repository)?)), + None => Box::default(), + }) + }) + .await + } + + async fn delete( + &self, + scope: &SnapshotScope, + name: &SnapshotName, + ) -> Result<(), SnapshotStoreError> { + let (token, _guard) = self.start()?; + let backend = Arc::new(self.backend(scope, &token)?); + let key = self.key.clone(); + let name = name.clone(); + let found = self + .blocking(Operation::Repository, move || { + let Some(repository) = open_existing(backend, &key)? else { + return Ok(None); + }; + let named = scope_snapshots(&repository)? + .readable + .into_iter() + .filter(|snapshot| snapshot.label == name.as_str()) + .collect::>(); + let ids = named + .iter() + .map(|snapshot| snapshot.id) + .collect::>(); + let freed = named.iter().map(added_packed_bytes).sum::(); + Ok(Some((repository, ids, freed))) + }) + .await?; + let Some((repository, ids, freed)) = found else { + return Ok(()); + }; + // The record comes before the forget, so a stop between the two cannot lose the bytes. A + // forget that then fails gives its error, and the record stays, which only brings a prune + // earlier. + let files = self.files(scope, &token); + if freed > 0 { + let snapshots = ids + .iter() + .map(|id| id.to_hex().to_string().into_boxed_str()) + .collect::>(); + record_freed(&files, freed, &snapshots) + .await + .map_err(storage_failure)?; + } + // The forget deletes the snapshot files with rayon, so it runs in a pool of its own. + let pool = self.low_priority; + self.blocking(Operation::Repository, move || { + pool.run_at_normal_priority("fs-snap-delete", move || { + repository.delete_snapshots(&ids)?; + Ok(()) + }) + }) + .await?; + self.prune_when_due(scope, &token).await + } + + async fn delete_scope(&self, scope: &SnapshotScope) -> Result<(), SnapshotStoreError> { + let (token, _guard) = self.start()?; + delete_scope(&self.files(scope, &token)) + .await + .map_err(storage_failure) + } + + async fn copy_scope( + &self, + from: &SnapshotScope, + to: &SnapshotScope, + ) -> Result<(), SnapshotStoreError> { + let (token, _guard) = self.start()?; + copy_scope(&self.files(from, &token), &self.files(to, &token)) + .await + .map_err(storage_failure) + } +} + +/// What the tree of a snapshot holds. The store keeps it as the description of the snapshot, +/// because rustic counts a symlink as a file. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] +struct TreeContent { + files: u64, + bytes: u64, +} + +/// The snapshot files of a scope that rustic could read, and whether a file failed its integrity +/// check. A file that a delete removed after the listing is in neither. +struct ScopeSnapshots { + readable: Vec, + unreadable: bool, +} + +/// What a lookup of a name found. +enum Lookup { + Found(Box, SnapshotInfo), + Missing, + Corrupt(anyhow::Error), +} + +impl Lookup { + fn into_info(self) -> Result, SnapshotStoreError> { + match self { + Self::Found(_, info) => Ok(Some(info)), + Self::Missing => Ok(None), + Self::Corrupt(error) => Err(SnapshotStoreError::Corrupt(error)), + } + } +} + +impl RusticSnapshotStore { + /// Runs the check of a local path on a blocking thread. A thread that fails gives the error of + /// the path. The tracker counts the check until it ends, so `shut_down` waits for it, also when + /// the operation drops first. + async fn check_path( + &self, + operation: NativeOperation, + path: &Path, + error: fn(std::io::Error) -> SnapshotStoreError, + check: fn(&Path) -> Result<(), SnapshotStoreError>, + ) -> Result<(), SnapshotStoreError> { + let path: Box = path.into(); + let tracked = self.tracker.token(); + #[cfg(test)] + let gate = self.path_check_gate.clone(); + execute_native(NativeStorageProfile::Unknown, operation, move || { + let _tracked = tracked; + #[cfg(test)] + if let Some(gate) = gate { + gate.reached.notify_one(); + futures::executor::block_on(gate.open.notified()); + } + check(&path) + }) + .await + .map_err(|failed| error(std::io::Error::other(failed)))? + } + + /// Checks that the tree of a save is a directory at an absolute path. + async fn check_tree(&self, tree: &Path) -> Result<(), SnapshotStoreError> { + self.check_path( + NativeOperation::Metadata, + tree, + SnapshotStoreError::Source, + tree_is_valid, + ) + .await + } + + /// Checks that the directory of a restore is an empty directory with a UTF-8 path. + async fn check_destination(&self, into: &Path) -> Result<(), SnapshotStoreError> { + self.check_path( + NativeOperation::DirectoryEnumeration, + into, + SnapshotStoreError::Destination, + destination_is_valid, + ) + .await + } +} + +fn tree_is_valid(tree: &Path) -> Result<(), SnapshotStoreError> { + let metadata = std::fs::metadata(tree).map_err(SnapshotStoreError::Source)?; + if !tree.is_absolute() || !metadata.is_dir() { + return Err(SnapshotStoreError::Source(std::io::Error::new( + std::io::ErrorKind::NotADirectory, + format!( + "the tree {} is not a directory at an absolute path", + tree.display() + ), + ))); + } + Ok(()) +} + +fn destination_is_valid(into: &Path) -> Result<(), SnapshotStoreError> { + let refused = |kind, reason: &str| { + SnapshotStoreError::Destination(std::io::Error::new( + kind, + format!("the directory {} {reason}", into.display()), + )) + }; + let metadata = std::fs::metadata(into).map_err(SnapshotStoreError::Destination)?; + if !metadata.is_dir() { + return Err(refused( + std::io::ErrorKind::NotADirectory, + "is not a directory", + )); + } + if into.to_str().is_none() { + return Err(refused( + std::io::ErrorKind::InvalidInput, + "does not have a UTF-8 path", + )); + } + let first = std::fs::read_dir(into) + .map_err(SnapshotStoreError::Destination)? + .next() + .transpose() + .map_err(SnapshotStoreError::Destination)?; + match first { + Some(_) => Err(refused( + std::io::ErrorKind::DirectoryNotEmpty, + "is not empty", + )), + None => Ok(()), + } +} + +/// Backs up the tree with the snapshot file in the stage, and gives the staged file with the info +/// of the snapshot. The result is `None` when a snapshot of the scope already has the name. The +/// parent is the snapshot that [`lookup`] finds for its name, and a name without a snapshot gives +/// no parent. +fn stage_save( + backend: Arc, + stage: &SnapshotStage, + key: &RepositoryKey, + policy: &StorePolicy, + name: &SnapshotName, + tree: &Path, + parent: Option<(SnapshotName, ChangeDetection)>, +) -> anyhow::Result> { + // The init of the fork checks the config a second time before its write, so a save that loses + // the race to create the repository can fail at that check with an error that is not + // `ConfigExists`. A config that is there after the error is the repository of the winner. + let repository = match open_or_create(backend.clone(), key, &policy.repository) { + Ok((repository, _)) => repository, + Err(error) => open_existing(backend, key).ok().flatten().ok_or(error)?, + }; + let before = scope_snapshots(&repository)?; + if has_name(&before, name) { + return Ok(None); + } + let parent = parent.and_then(|(parent, detection)| { + named(&before.readable, &parent).map(|snapshot| (snapshot.id, detection)) + }); + let newest = before + .readable + .iter() + .filter_map(snapshot_info) + .map(|info| info.created_at) + .max(); + let created_at = snapshot_time(whole_millis_from(Timestamp::now_utc()), newest); + // rustic strips the root from the path of each entry. The canonical root is the path that the + // walk of rustic gives, also when the caller gives a path through a symlink such as + // `/proc/self/fd/N`. + let tree = std::fs::canonicalize(tree)?; + let content = tree_content(&tree)?; + let snapshot = SnapshotOptions::default() + .label(name.as_str().to_string()) + .time(snapshot_zoned(created_at)?) + .description(serde_json::to_string(&content)?) + .to_snapshot()?; + let repository = repository.to_indexed_ids()?; + repository.backup( + &store_backup_options(policy, parent), + &PathList::from_iter(Some(tree)), + snapshot, + )?; + let staged = stage + .take() + .context("the backup gave no snapshot file to the stage")?; + if has_name(&scope_snapshots(&repository)?, name) { + return Ok(None); + } + Ok(Some(( + staged, + SnapshotInfo { + created_at, + files: content.files, + bytes: content.bytes, + }, + ))) +} + +/// Reads each snapshot file of the repository. A failed storage call fails the read, a file that a +/// delete removed after the listing is left out, and each other failure counts as a failed check. +fn scope_snapshots(repository: &RusticRepository) -> anyhow::Result { + repository.list::()?.try_fold( + ScopeSnapshots { + readable: Vec::new(), + unreadable: false, + }, + |mut found, id| match repository.get_file::(&id) { + Ok(mut snapshot) => { + snapshot.id = id; + found.readable.push(snapshot); + Ok(found) + } + Err(error) if is_storage_failure(&*error) => Err(anyhow::Error::from(error)), + Err(error) if is_file_missing(&*error) => Ok(found), + Err(_) => Ok(ScopeSnapshots { + unreadable: true, + ..found + }), + }, + ) +} + +fn has_name(found: &ScopeSnapshots, name: &SnapshotName) -> bool { + found + .readable + .iter() + .any(|snapshot| snapshot.label == name.as_str()) +} + +/// Finds the snapshot with the name. Of the snapshot files with the name, the one with the least +/// time and id wins. When no file has the name and a file failed its integrity check, the result is +/// `Corrupt`, because that file can have the name. +fn lookup(found: ScopeSnapshots, name: &SnapshotName) -> Lookup { + match (named(&found.readable, name), found.unreadable) { + (Some(snapshot), _) => match snapshot_info(snapshot) { + Some(info) => Lookup::Found(Box::new(snapshot.clone()), info), + None => Lookup::Corrupt(anyhow::anyhow!( + "the snapshot {} does not describe its tree", + snapshot.id + )), + }, + (None, true) => Lookup::Corrupt(anyhow::anyhow!( + "a snapshot file of the scope failed its integrity check" + )), + (None, false) => Lookup::Missing, + } +} + +/// Of the snapshot files with the name, gives the one with the least time and id. +fn named<'a>(snapshots: &'a [SnapshotFile], name: &SnapshotName) -> Option<&'a SnapshotFile> { + snapshots + .iter() + .filter(|snapshot| snapshot.label == name.as_str()) + .min_by_key(|snapshot| (snapshot.time.timestamp(), snapshot.id)) +} + +/// Gives the name and the info of each snapshot whose label is a name and whose description +/// parses. Of the files with one name, only the one that [`lookup`] takes stays. +fn listed(found: ScopeSnapshots) -> Vec<(SnapshotName, SnapshotInfo)> { + let mut snapshots = found.readable; + snapshots.sort_by(|left, right| { + (&left.label, left.time.timestamp(), left.id).cmp(&( + &right.label, + right.time.timestamp(), + right.id, + )) + }); + snapshots.dedup_by(|later, first| later.label == first.label); + snapshots + .iter() + .filter_map(|snapshot| { + SnapshotName::new(&snapshot.label) + .ok() + .zip(snapshot_info(snapshot)) + }) + .collect() +} + +/// Gives the info of a snapshot from its time and its description. +fn snapshot_info(snapshot: &SnapshotFile) -> Option { + let content = serde_json::from_str::(snapshot.description.as_deref()?).ok()?; + let created_at = u64::try_from(snapshot.time.timestamp().as_millisecond()).ok()?; + Some(SnapshotInfo { + created_at: Timestamp::from(created_at), + files: content.files, + bytes: content.bytes, + }) +} + +/// Tells whether a later prune removes packs that this prune leaves marked: unused packs, repacked +/// packs, packs that no index lists, and packs of an earlier prune whose grace period is not over. +/// A marked pack is in an index after the prune, so a later prune does not count it as unindexed. +fn leaves_marked_packs(report: &PruneReport) -> bool { + report.packs_unused > 0 + || report.packs_repacked > 0 + || report.packs_unindexed > 0 + || report.marked_packs_kept > 0 +} + +/// Gives the packed bytes that the save of the snapshot added to the repository. +fn added_packed_bytes(snapshot: &SnapshotFile) -> u64 { + snapshot + .summary + .as_ref() + .map_or(0, |summary| summary.data_added_packed) +} + +/// Gives the first time in whole milliseconds that is not before the time. A snapshot keeps its +/// time in milliseconds, so the time of a save is not before the call. +fn whole_millis_from(time: Timestamp) -> Timestamp { + let truncated = Timestamp::from(time.to_millis()); + if truncated < time { + Timestamp::from(time.to_millis().saturating_add(1)) + } else { + truncated + } +} + +/// Gives the time as the time of a snapshot, in UTC. +fn snapshot_zoned(time: Timestamp) -> anyhow::Result { + Ok(SnapshotTime::from_millisecond(i64::try_from(time.to_millis())?)?.to_zoned(TimeZone::UTC)) +} + +/// Counts the names of the regular files below the root, and the sum of their sizes. The walk +/// reads the metadata of each entry and does not follow a symlink. +fn tree_content(root: &Path) -> std::io::Result { + fn walk(directory: &Path, content: TreeContent) -> std::io::Result { + std::fs::read_dir(directory)?.try_fold(content, |content, entry| { + let entry = entry?; + let kind = entry.file_type()?; + if kind.is_dir() { + walk(&entry.path(), content) + } else if kind.is_file() { + Ok(TreeContent { + files: content.files + 1, + bytes: content.bytes + entry.metadata()?.len(), + }) + } else { + Ok(content) + } + }) + } + walk(root, TreeContent { files: 0, bytes: 0 }) +} + +#[cfg(test)] +mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs new file mode 100644 index 0000000000..417b9108a2 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -0,0 +1,5659 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The rustic store through the interface of the store, on the in-memory blob storage. +//! +//! The contract suite runs on the store with the policy of the configuration. The other tests +//! give the store a short or a long deadline and a prune policy that the test controls. + +use super::super::fault::is_lease_expired; +use super::super::files::SnapshotFiles; +use super::super::prune::{ + CLOCK_SKEW_MARGIN, ClaimChoice, ClaimEntry, LEDGERS_PATH, Percent, PruneLedger, claim_hold, + next_claim, parse_claim_entry, parse_freed, read_ledger, +}; +use super::super::tests::scripted::{Script, ScriptedBlobStorage}; +use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; +use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; +use super::{ + RusticSnapshotStore, StepGate, StorePolicy, leaves_marked_packs, scope_snapshots, + store_backup_options, store_restore_options, whole_millis_from, +}; +use crate::filesystem_snapshot::contract_tests::fixture::{ + Listed, Scratch, Spec, fixture, listing, write_tree, +}; +use crate::filesystem_snapshot::contract_tests::{self, OpenStore, new_scope}; +use crate::filesystem_snapshot::{ + ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, + SnapshotStoreError, +}; +use crate::services::golem_config::FilesystemSnapshotStoreConfig; +use futures::{FutureExt, StreamExt, TryStreamExt}; +use golem_service_base::storage::blob::memory::InMemoryBlobStorage; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use pretty_assertions::assert_eq; +use rustic_core::repofile::SnapshotId; +use std::future::Future; +use std::num::NonZeroUsize; +use std::path::Path; +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; +use std::time::Duration; +use test_r::core::DynamicTestRegistration; +use test_r::{test, test_gen, timeout}; + +/// The id of a snapshot file that no scope holds, as the content of a record of freed bytes. +const GONE_SNAPSHOT: &str = "0000000000000000000000000000000000000000000000000000000000000000"; + +/// The longest time that a test waits for an operation or for the work of a store to end. +const LIMIT: Duration = Duration::from_secs(10); + +/// A prune threshold that a delete never reaches. +const NEVER: Percent = Percent(u16::MAX); + +/// A prune threshold of zero bytes, so each delete that frees bytes prunes. +const ALWAYS: Percent = Percent(0); + +/// A deadline that no call of these tests reaches, so a held call ends only by a cancel. +const LONG_DEADLINE: Duration = Duration::from_secs(60); + +const KEY: &str = "000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f\ + 202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f"; + +fn config() -> FilesystemSnapshotStoreConfig { + FilesystemSnapshotStoreConfig::new(KEY, Duration::from_secs(30), 4, 3).unwrap() +} + +fn key() -> RepositoryKey { + RepositoryKey::new(*config().repository_key().bytes()) +} + +/// The policy of the configuration, with the deadline, the prune threshold and the grace period +/// of the test. +fn policy(deadline: Duration, prune_threshold: Percent, grace: Duration) -> StorePolicy { + StorePolicy { + deadline, + prune: PruneSettings { + keep_delete: grace, + ..StorePolicy::from_config(&config()).prune + }, + prune_threshold, + ..StorePolicy::from_config(&config()) + } +} + +fn store(storage: Arc, policy: StorePolicy) -> Arc { + Arc::new(RusticSnapshotStore::with_policy(storage, key(), policy)) +} + +fn name(text: &str) -> SnapshotName { + SnapshotName::new(text).unwrap() +} + +/// Writes a tree of one file with the content into a new directory, and gives the directory. +fn one_file_tree(content: &str) -> Scratch { + let tree = Scratch::new(); + write_tree( + tree.path(), + &[( + "file.txt", + Spec::File { + content: Box::from(content.as_bytes()), + mode: 0o644, + }, + )], + ); + tree +} + +fn fixture_tree() -> Scratch { + let tree = Scratch::new(); + write_tree(tree.path(), &fixture()); + tree +} + +/// Restores the name into a new directory, and gives the listing of the directory. +async fn restored_listing( + store: &RusticSnapshotStore, + scope: &SnapshotScope, + name: &SnapshotName, +) -> Result, SnapshotStoreError> { + let into = Scratch::new(); + store.restore(scope, name, into.path()).await?; + Ok(listing(into.path())) +} + +async fn listed_names(store: &RusticSnapshotStore, scope: &SnapshotScope) -> Vec { + store + .list(scope) + .await + .unwrap() + .iter() + .map(|(name, _)| name.to_string()) + .collect() +} + +/// Gives the path of each blob of the namespace whose path starts with the prefix. +async fn blobs( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, + prefix: &str, +) -> Vec { + let mut paths = storage + .list_blobs_below("test", "test", namespace.clone(), Path::new("")) + .await + .unwrap() + .iter() + .map(|blob| blob.path.display().to_string()) + .filter(|path| path.starts_with(prefix)) + .collect::>(); + paths.sort(); + paths +} + +async fn ledger(storage: &Arc, scope: &SnapshotScope) -> PruneLedger { + read_ledger(&SnapshotFiles { + storage: storage.clone(), + namespace: scope.0.clone(), + deadline: Duration::from_secs(2), + cancel: tokio_util::sync::CancellationToken::new(), + tracker: tokio_util::task::TaskTracker::new(), + }) + .await + .unwrap() +} + +/// Gives the sum of the bytes in the names of the records of freed bytes of the scope. +async fn freed(storage: &Arc, scope: &SnapshotScope) -> u64 { + let listed = storage + .list_blobs_below( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-freed"), + ) + .await + .unwrap(); + listed + .iter() + .filter_map(|blob| parse_freed(blob.path.file_name()?.to_str()?)) + .fold(0, u64::saturating_add) +} + +/// Waits until the condition holds, or until the limit ends. Gives whether the condition holds. +async fn eventually(condition: impl Fn() -> bool) -> bool { + tokio::time::timeout(LIMIT, async { + futures::stream::repeat(()) + .then(|()| tokio::time::sleep(Duration::from_millis(5))) + .take_while(|()| std::future::ready(!condition())) + .for_each(|()| std::future::ready(())) + .await + }) + .await + .is_ok() +} + +/// Runs the operation until the calls of the storage match the condition, and then drops it. +/// Gives the output of the operation when it ends first. +async fn drop_when( + storage: &ScriptedBlobStorage, + condition: impl Fn(&[(&'static str, String)]) -> bool, + operation: impl Future, +) -> Option { + tokio::select! { + output = operation => Some(output), + _ = eventually(|| condition(&storage.calls())) => None, + } +} + +fn is_storage(error: &SnapshotStoreError, expected_retryable: bool) -> bool { + matches!(error, SnapshotStoreError::Storage { retryable, .. } if *retryable == expected_retryable) +} + +#[test_gen] +fn rustic_store_keeps_the_contract(r: &mut DynamicTestRegistration) { + contract_tests::register(r, || { + let storage: Arc = Arc::new(InMemoryBlobStorage::new()); + let open: OpenStore = Arc::new(move || { + Arc::new(RusticSnapshotStore::new(storage.clone(), &config())) + as Arc + }); + open + }); +} + +#[test] +fn the_policy_takes_the_configured_values_and_the_options_are_strict() { + let policy = StorePolicy::from_config(&config()); + let backup = store_backup_options(&policy, None); + let restore = store_restore_options(&policy); + + assert_eq!( + ( + policy.deadline, + policy.save_threads.map(NonZeroUsize::get), + policy.restore_reader_threads.get(), + policy.prune.keep_delete, + policy.prune.fast_repack, + policy.prune.repack, + policy.prune_threshold, + ), + ( + Duration::from_secs(30), + Some(3), + 4, + Duration::from_secs(15 * 60), + true, + RepackLimits::Rustic, + Percent(10), + ) + ); + assert_eq!( + ( + backup.fail_on_read_error, + backup.threads.map(NonZeroUsize::get), + backup.as_path.as_deref().map(Path::to_path_buf), + restore.fail_on_metadata_error, + restore.no_ownership, + restore.reader_threads.map(NonZeroUsize::get), + ), + ( + true, + Some(3), + Some(Path::new("/").to_path_buf()), + true, + true, + Some(4), + ) + ); +} + +#[test] +fn a_save_time_is_the_first_whole_millisecond_that_is_not_before_the_call() { + let now = golem_common::model::Timestamp::now_utc(); + let whole = golem_common::model::Timestamp::from(5_000); + let rounded = whole_millis_from(now); + + assert_eq!( + ( + whole_millis_from(whole), + rounded >= now, + rounded.to_millis() - now.to_millis() <= 1, + golem_common::model::Timestamp::from(rounded.to_millis()) == rounded, + ), + (whole, true, true, true) + ); +} + +#[cfg(target_os = "linux")] +#[test] +#[timeout("60s")] +async fn a_tree_saved_through_a_proc_self_fd_path_is_stored_below_the_root() { + use std::os::fd::AsRawFd; + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + let directory = std::fs::File::open(tree.path()).unwrap(); + let through_fd = format!("/proc/self/fd/{}", directory.as_raw_fd()); + + store + .save(&scope, &name("p-fd"), Path::new(&through_fd), None) + .await + .unwrap(); + let namespace = scope.0.clone(); + let paths = tokio::task::spawn_blocking(move || { + let backend = super::super::backend::BlobBackend::new( + storage, + namespace, + tokio::runtime::Handle::current(), + LONG_DEADLINE, + ); + let repository = open_existing(Arc::new(backend), &key()).unwrap().unwrap(); + scope_snapshots(&repository) + .unwrap() + .readable + .iter() + .map(|snapshot| snapshot.paths.to_string()) + .collect::>() + }) + .await + .unwrap(); + + assert_eq!( + ( + paths, + restored_listing(&store, &scope, &name("p-fd")) + .await + .unwrap() + ), + (vec!["/".to_string()], listing(tree.path())) + ); +} + +#[test] +#[timeout("60s")] +async fn a_save_whose_index_write_fails_publishes_nothing_and_leaves_the_name_free() { + let refuse = Arc::new(AtomicBool::new(true)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) && op_label == "write" && path.starts_with("index") { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + let tree = fixture_tree(); + + let failed = store.save(&scope, &name("p-1"), tree.path(), None).await; + let stat = store.stat(&scope, &name("p-1")).await.unwrap(); + let names = listed_names(&store, &scope).await; + let restore = restored_listing(&store, &scope, &name("p-1")).await; + refuse.store(false, Ordering::SeqCst); + let saved_again = store.save(&scope, &name("p-1"), tree.path(), None).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!( + matches!(restore, Err(SnapshotStoreError::NotFound)), + "{restore:?}" + ); + assert_eq!( + ( + stat, + names, + saved_again.is_ok(), + restored_listing(&store, &scope, &name("p-1")) + .await + .unwrap() + ), + (None, Vec::::new(), true, listing(tree.path())) + ); +} + +#[test] +#[timeout("60s")] +async fn a_publish_that_reaches_the_deadline_and_lands_late_publishes_nothing() { + let hang = Arc::new(AtomicBool::new(true)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let hang = hang.clone(); + move |op_label, _| { + if hang.load(Ordering::SeqCst) && op_label == "publish" { + Script::NeverAnswer + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(Duration::from_secs(1), NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("late"); + + let failed = store.save(&scope, &name("p-late"), tree.path(), None).await; + hang.store(false, Ordering::SeqCst); + let stat = store.stat(&scope, &name("p-late")).await.unwrap(); + let names = listed_names(&store, &scope).await; + let saved_again = store.save(&scope, &name("p-late"), tree.path(), None).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert_eq!( + ( + stat, + names, + saved_again.is_ok(), + storage + .calls() + .iter() + .filter(|(op_label, _)| *op_label == "retract") + .count() + ), + (None, Vec::::new(), true, 1) + ); +} + +#[test] +#[timeout("60s")] +async fn a_second_save_of_an_unchanged_tree_writes_no_pack() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + let first = storage.calls().len(); + store + .save(&scope, &name("p-2"), tree.path(), None) + .await + .unwrap(); + let writes = storage.calls()[first..] + .iter() + .filter(|(op_label, _)| *op_label == "write" || *op_label == "publish") + .map(|(op_label, path)| { + ( + *op_label, + path.split('/').next().unwrap_or_default().to_string(), + ) + }) + .collect::>(); + + let restored = restored_listing(&store, &scope, &name("p-2")).await.ok(); + + assert_eq!( + (writes, restored), + ( + vec![("publish", "snapshots".to_string())], + Some(listing(tree.path())) + ) + ); +} + +#[test] +#[timeout("60s")] +async fn two_stores_that_create_one_repository_at_the_same_time_both_save() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "write" && path == Path::new("config") { + Script::WaitForGate + } else { + Script::Pass + } + }); + let policy = policy(LONG_DEADLINE, NEVER, Duration::ZERO); + let (first, second) = ( + store(storage.clone(), policy), + store(storage.clone(), policy), + ); + let scope = new_scope(); + let (first_tree, second_tree) = (one_file_tree("first"), one_file_tree("second")); + let config_writes = || { + storage + .calls() + .iter() + .filter(|(op_label, path)| *op_label == "write" && path == "config") + .count() + }; + + let (first_name, second_name) = (name("p-first"), name("p-second")); + let (first_saved, second_saved, both_waited) = tokio::join!( + first.save(&scope, &first_name, first_tree.path(), None), + second.save(&scope, &second_name, second_tree.path(), None), + async { + let both = eventually(|| config_writes() == 2).await; + storage.open_gate(); + both + } + ); + let mut names = listed_names(&first, &scope).await; + names.sort(); + + assert_eq!( + ( + both_waited, + first_saved.map(|_| ()).map_err(|error| error.to_string()), + second_saved.map(|_| ()).map_err(|error| error.to_string()), + names, + restored_listing(&second, &scope, &name("p-first")) + .await + .unwrap(), + restored_listing(&first, &scope, &name("p-second")) + .await + .unwrap(), + ), + ( + true, + Ok(()), + Ok(()), + vec!["p-first".to_string(), "p-second".to_string()], + listing(first_tree.path()), + listing(second_tree.path()), + ) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { + // The first index write after the arm waits at the gate. That is the index write of the + // second save, so its packs are in no index while the delete prunes. + let hold_next_index = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let hold_next_index = hold_next_index.clone(); + move |op_label, path| { + if op_label == "write" + && path.starts_with("index") + && hold_next_index.swap(false, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let (old_tree, new_tree) = (one_file_tree("old"), fixture_tree()); + store + .save(&scope, &name("p-old"), old_tree.path(), None) + .await + .unwrap(); + hold_next_index.store(true, Ordering::SeqCst); + let index_writes = || { + storage + .calls() + .iter() + .filter(|(op_label, path)| *op_label == "write" && path.starts_with("index")) + .count() + }; + let before = index_writes(); + + let saving = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + let path = new_tree.path().to_path_buf(); + async move { store.save(&scope, &name("p-new"), &path, None).await } + }); + let held = eventually(|| index_writes() > before).await; + let deleted = store.delete(&scope, &name("p-old")).await; + let pruned_while_held = ledger(&storage, &scope).await.last_prune.is_some(); + storage.open_gate(); + let saved = saving.await.unwrap(); + let pruned_again = store.delete(&scope, &name("p-none")).await; + + assert_eq!( + ( + held, + deleted.is_ok(), + pruned_while_held, + saved.is_ok(), + pruned_again.is_ok(), + restored_listing(&store, &scope, &name("p-new")).await.ok(), + ), + (true, true, true, true, true, Some(listing(new_tree.path()))) + ); +} + +#[test] +#[timeout("60s")] +async fn a_restore_whose_pack_reads_fail_gives_a_retryable_storage_error() { + let refuse = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) + && matches!(op_label, "read" | "read_range") + && path.starts_with("data") + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + let tree = fixture_tree(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + refuse.store(true, Ordering::SeqCst); + + let restored = restored_listing(&store, &scope, &name("p-1")).await; + + assert!( + restored + .as_ref() + .is_err_and(|error| is_storage(error, true)), + "{restored:?}" + ); +} + +#[cfg(unix)] +#[test] +#[timeout("60s")] +async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_denied() { + use std::os::unix::fs::PermissionsExt; + // SAFETY: `geteuid` has no preconditions. + let uid = unsafe { libc::geteuid() }; + assert_ne!( + uid, 0, + "this test needs a user other than root, because permissions do not stop root from a read" + ); + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("readable"); + let locked = tree.path().join("locked.txt"); + std::fs::write(&locked, b"locked").unwrap(); + std::fs::set_permissions(&locked, std::fs::Permissions::from_mode(0o000)).unwrap(); + + let saved = store + .save(&scope, &name("p-locked"), tree.path(), None) + .await; + + assert!( + matches!(&saved, Err(SnapshotStoreError::Source(error)) if error.kind() == std::io::ErrorKind::PermissionDenied), + "{saved:?}" + ); + assert_eq!( + blobs(&*storage, &scope.0, "snapshots/").await, + Vec::::new() + ); +} + +#[cfg(target_os = "linux")] +#[test] +#[timeout("60s")] +async fn a_restore_that_cannot_set_an_extended_attribute_gives_destination() { + // A file on tmpfs takes a user attribute of 6,000 bytes. A file on ext4 with 4 KiB blocks + // does not, so the restore cannot set it. The test checks nothing on a host where the source + // does not take the attribute or where the destination takes it. + let value = vec![b'a'; 6000]; + let set = |path: &Path| xattr_set(path, "user.golem-test", &value); + let Ok(source) = tempfile::tempdir_in("/dev/shm") else { + println!("SKIPPED: /dev/shm has no directory for the source of the test"); + return; + }; + let file = source.path().join("file.txt"); + std::fs::write(&file, b"content").unwrap(); + let probe = Scratch::new(); + let probe_file = probe.path().join("probe"); + std::fs::write(&probe_file, b"probe").unwrap(); + if let Err(error) = set(&file) { + println!("SKIPPED: /dev/shm does not take a user attribute of 6,000 bytes: {error}"); + return; + } + if set(&probe_file).is_ok() { + println!( + "SKIPPED: the destination {} takes a user attribute of 6,000 bytes", + probe.path().display() + ); + return; + } + let store = store( + Arc::new(InMemoryBlobStorage::new()), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + store + .save(&scope, &name("p-xattr"), source.path(), None) + .await + .unwrap(); + + let restored = restored_listing(&store, &scope, &name("p-xattr")).await; + + assert!( + matches!(&restored, Err(SnapshotStoreError::Destination(_))), + "{restored:?}" + ); +} + +#[cfg(target_os = "linux")] +fn xattr_set(path: &Path, name: &str, value: &[u8]) -> std::io::Result<()> { + use std::os::unix::ffi::OsStrExt; + let path = std::ffi::CString::new(path.as_os_str().as_bytes())?; + let name = std::ffi::CString::new(name)?; + // SAFETY: the path and the name are NUL-terminated strings, and the value lives for the call. + let result = unsafe { + libc::setxattr( + path.as_ptr(), + name.as_ptr(), + value.as_ptr().cast(), + value.len(), + 0, + ) + }; + if result == 0 { + Ok(()) + } else { + Err(std::io::Error::last_os_error()) + } +} + +#[test] +#[timeout("60s")] +async fn a_snapshot_file_that_fails_its_check_is_left_out_of_list_and_makes_an_unknown_name_corrupt() + { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path(), None) + .await + .unwrap(); + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new(&format!("snapshots/{}", "ab".repeat(32))), + b"not a snapshot", + ) + .await + .unwrap(); + + let names = listed_names(&store, &scope).await; + let kept = store.stat(&scope, &name("p-kept")).await; + let unknown = store.stat(&scope, &name("p-unknown")).await; + let restore_unknown = restored_listing(&store, &scope, &name("p-unknown")).await; + + assert!( + matches!(unknown, Err(SnapshotStoreError::Corrupt(_))), + "{unknown:?}" + ); + assert!( + matches!(restore_unknown, Err(SnapshotStoreError::Corrupt(_))), + "{restore_unknown:?}" + ); + assert_eq!( + (names, kept.map(|info| info.is_some()).ok()), + (vec!["p-kept".to_string()], Some(true)) + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_period() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path(), None) + .await + .unwrap(); + let packs_before = blobs(&*storage, &scope.0, "data/").await; + + store.delete(&scope, &name("p-deleted")).await.unwrap(); + let after_first = ledger(&storage, &scope).await; + let after_first_freed = freed(&storage, &scope).await; + // The margin for clock skew keeps the next prune back, so the ledger moves back. + age_ledger(&storage, &scope, &after_first).await; + store.delete(&scope, &name("p-none")).await.unwrap(); + let packs_after = blobs(&*storage, &scope.0, "data/").await; + + assert_eq!( + ( + after_first_freed, + after_first.last_prune.is_some(), + after_first.awaiting_removal, + packs_after.len() < packs_before.len(), + packs_after.iter().all(|pack| packs_before.contains(pack)), + restored_listing(&store, &scope, &name("p-kept")).await.ok(), + ), + (0, true, true, true, true, Some(listing(kept_tree.path()))) + ); +} + +/// Counts the listings of the packs among the recorded calls. +fn data_listings(calls: &[(&'static str, String)]) -> usize { + calls + .iter() + .filter(|(op_label, _)| *op_label == "list_data") + .count() +} + +/// Makes the ledger one entry with no marked packs and a last prune at the time, and writes a +/// record of one freed byte. +async fn set_last_prune( + storage: &Arc, + scope: &SnapshotScope, + last_prune: golem_common::model::Timestamp, +) { + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-freed/1-test"), + GONE_SNAPSHOT.as_bytes(), + ) + .await + .unwrap(); + storage + .delete_dir("test", "test", scope.0.clone(), Path::new(LEDGERS_PATH)) + .await + .unwrap(); + put_ledger_entry( + storage, + scope, + &format!("{}-0-test", last_prune.to_millis()), + ) + .await; +} + +/// Makes the ledger one entry with the marked packs of `ledger` and a time before it by two hours, +/// which is more than the grace period of each test and the margin for clock skew. +async fn age_ledger( + storage: &Arc, + scope: &SnapshotScope, + ledger: &PruneLedger, +) { + let back = u64::try_from(CLOCK_SKEW_MARGIN.as_millis()).unwrap() + 2 * 3_600_000; + let aged = ledger + .last_prune + .map_or(0, |last| last.to_millis().saturating_sub(back)); + storage + .delete_dir("test", "test", scope.0.clone(), Path::new(LEDGERS_PATH)) + .await + .unwrap(); + put_ledger_entry( + storage, + scope, + &format!("{aged}-{}-aged", u8::from(ledger.awaiting_removal)), + ) + .await; +} + +/// Writes a ledger entry with the name. +async fn put_ledger_entry( + storage: &Arc, + scope: &SnapshotScope, + name: &str, +) { + storage + .put_raw( + "test", + "test", + scope.0.clone(), + &Path::new(LEDGERS_PATH).join(name), + b"", + ) + .await + .unwrap(); +} + +#[test] +#[timeout("60s")] +async fn a_delete_within_the_grace_period_does_not_list_the_packs() { + // The ledger has no marked packs, so only the grace period keeps the first delete from a + // listing. The second delete comes after the grace period and lists the packs one time. + let grace = Duration::from_secs(3600); + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store(storage.clone(), policy(LONG_DEADLINE, NEVER, grace)); + let scope = new_scope(); + let (first, second) = (one_file_tree("first"), one_file_tree("second")); + store + .save(&scope, &name("p-1"), first.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), second.path(), None) + .await + .unwrap(); + let now = golem_common::model::Timestamp::now_utc(); + set_last_prune(&storage, &scope, now).await; + let before_first = storage.calls().len(); + + store.delete(&scope, &name("p-1")).await.unwrap(); + let before_second = storage.calls().len(); + set_last_prune( + &storage, + &scope, + golem_common::model::Timestamp::from(now.to_millis().saturating_sub(2 * 3_600_000)), + ) + .await; + store.delete(&scope, &name("p-2")).await.unwrap(); + let calls = storage.calls(); + + assert_eq!( + ( + data_listings(&calls[before_first..before_second]), + data_listings(&calls[before_second..]), + ), + (0, 1) + ); +} + +#[test] +#[timeout("60s")] +async fn a_failed_listing_of_the_packs_gives_storage_and_records_no_prune() { + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "list_data" { + Script::Refuse + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), one_file_tree("kept")); + store + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path(), None) + .await + .unwrap(); + + let deleted = store.delete(&scope, &name("p-deleted")).await; + let after = ledger(&storage, &scope).await; + let after_freed = freed(&storage, &scope).await; + + assert!( + deleted.as_ref().is_err_and(|error| is_storage(error, true)), + "{deleted:?}" + ); + assert_eq!((after_freed > 0, after.last_prune), (true, None)); +} + +/// Counts the prunes among the recorded calls. A prune lists the packs, and no other step of a +/// delete makes that call. +fn prunes(calls: &[(&'static str, String)]) -> usize { + calls + .iter() + .filter(|(op_label, path)| *op_label == "list" && path == "data") + .count() +} + +/// Saves a tree of one file with the content under each name. +async fn save_each(store: &RusticSnapshotStore, scope: &SnapshotScope, names: &[&str]) { + futures::stream::iter(names) + .for_each(|text| async move { + let tree = one_file_tree(text); + store + .save(scope, &name(text), tree.path(), None) + .await + .unwrap(); + }) + .await; +} + +#[test] +#[timeout("60s")] +async fn a_delete_that_paused_after_its_ledger_read_does_not_put_back_the_old_ledger() { + // The gate holds the first delete after its record write and its ledger read. The second + // delete prunes to its end. The first delete then goes on with the ledger that it read. + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let held = held.clone(); + move |op_label, _| { + if op_label == "list_freed" && !held.swap(true, Ordering::SeqCst) { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let paused = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let paused_held = eventually(|| held.load(Ordering::SeqCst)).await; + + let pruned = store.delete(&scope, &name("p-2")).await; + let after_prune = ledger(&storage, &scope).await; + storage.open_gate(); + let paused = tokio::time::timeout(LIMIT, paused).await; + let after_all = ledger(&storage, &scope).await; + + assert!(pruned.is_ok(), "{pruned:?}"); + assert!(matches!(paused, Ok(Ok(Ok(())))), "{paused:?}"); + assert_eq!( + ( + paused_held, + prunes(&storage.calls()), + after_prune.last_prune.is_some(), + after_all, + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, 1, true, after_prune, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_record_that_a_delete_adds_during_a_prune_stays_for_the_next_prune() { + // The gate holds the prune at its listing of the packs, and a record comes in meanwhile. + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let pruning = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let prune_held = eventually(|| held.load(Ordering::SeqCst)).await; + + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-freed/7-late"), + GONE_SNAPSHOT.as_bytes(), + ) + .await + .unwrap(); + storage.open_gate(); + let pruned = tokio::time::timeout(LIMIT, pruning).await; + + assert!(matches!(pruned, Ok(Ok(Ok(())))), "{pruned:?}"); + assert_eq!( + ( + prune_held, + blobs(&*storage, &scope.0, "golem/prune-freed/").await, + freed(&storage, &scope).await, + ), + (true, vec!["golem/prune-freed/7-late".to_string()], 7) + ); +} + +#[test] +#[timeout("60s")] +async fn a_late_older_ledger_entry_does_not_win() { + // The newer entry is inside the grace period, so a delete does not prune. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let now = golem_common::model::Timestamp::now_utc().to_millis(); + let newer = now.saturating_sub(60_000); + put_ledger_entry(&storage, &scope, &format!("{newer}-0-newer")).await; + put_ledger_entry( + &storage, + &scope, + &format!("{}-1-older", now.saturating_sub(2 * 3_600_000)), + ) + .await; + + let read = ledger(&storage, &scope).await; + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert_eq!( + ( + read.last_prune.map(|last| last.to_millis()), + read.awaiting_removal, + prunes(&storage.calls()), + ), + (Some(newer), false, 0) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_deletes_the_older_ledger_entries_and_keeps_a_newer_one() { + // The gate holds the prune at its listing of the packs. An older and a newer entry come in + // meanwhile. + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let pruning = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let prune_held = eventually(|| held.load(Ordering::SeqCst)).await; + + let newer = format!( + "{}-0-newer", + golem_common::model::Timestamp::now_utc().to_millis() + 60_000 + ); + put_ledger_entry(&storage, &scope, "1000-0-older").await; + put_ledger_entry(&storage, &scope, &newer).await; + storage.open_gate(); + let pruned = tokio::time::timeout(LIMIT, pruning).await; + let entries = blobs(&*storage, &scope.0, "golem/prune-ledgers/").await; + + assert!(matches!(pruned, Ok(Ok(Ok(())))), "{pruned:?}"); + assert_eq!( + ( + prune_held, + entries.len(), + entries.contains(&format!("golem/prune-ledgers/{newer}")), + entries.contains(&"golem/prune-ledgers/1000-0-older".to_string()), + ), + (true, 2, true, false) + ); +} + +/// Gives the time of the newest marker of the claims of the scope. +async fn claim_time(storage: &ScriptedBlobStorage, scope: &SnapshotScope) -> Option { + blobs(storage, &scope.0, "golem/prune-claims/") + .await + .iter() + .filter_map( + |path| match parse_claim_entry(Path::new(path).file_name()?.to_str()?)? { + ClaimEntry::Marker(_, at) => Some(at.to_millis()), + ClaimEntry::Claim(_) => None, + }, + ) + .max() +} + +/// Moves each marker of the claims of the scope back by two hours and the margin, so no marker +/// holds its ledger. +async fn age_claims(storage: &Arc, scope: &SnapshotScope) { + let stale = golem_common::model::Timestamp::now_utc() + .to_millis() + .saturating_sub(2 * 3_600_000 + 2 * 60_000); + let markers = storage + .list_blobs_below( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-claims"), + ) + .await + .unwrap() + .iter() + .filter_map(|blob| { + let entry = parse_claim_entry(blob.path.file_name()?.to_str()?)?; + match entry { + ClaimEntry::Marker(number, _) => Some((blob.path.clone(), number)), + ClaimEntry::Claim(_) => None, + } + }) + .collect::>(); + futures::stream::iter(markers) + .for_each(|(path, number)| { + let (storage, scope) = (storage.clone(), scope.clone()); + async move { + storage + .delete("test", "test", scope.0.clone(), &path) + .await + .unwrap(); + let aged = path + .parent() + .unwrap_or(Path::new("")) + .join(format!("{number}@{stale}-aged")); + storage + .put_raw("test", "test", scope.0.clone(), &aged, b"") + .await + .unwrap(); + } + }) + .await; +} + +#[test] +#[timeout("60s")] +async fn a_prune_refreshes_the_claim_with_its_own_number() { + // Old claims 0 and 1 with markers older than the hold stay in the claim directory, so the + // delete takes claim 2. The gate holds the prune at its listing of the packs while the claim + // gets new markers. + let grace = Duration::from_millis(400); + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let old = golem_common::model::Timestamp::now_utc() + .to_millis() + .saturating_sub(2 * 3_600_000); + futures::stream::iter([ + "golem/prune-claims/none/0".to_string(), + format!("golem/prune-claims/none/0@{old}-old"), + "golem/prune-claims/none/1".to_string(), + format!("golem/prune-claims/none/1@{old}-old"), + ]) + .for_each(|path| { + let (storage, scope) = (storage.clone(), scope.clone()); + async move { + storage + .put_raw("test", "test", scope.0.clone(), Path::new(&path), &[]) + .await + .unwrap(); + } + }) + .await; + let pruning = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let refreshes = || { + storage + .calls() + .iter() + .filter(|(op_label, _)| *op_label == "refresh_claim") + .count() + }; + let refreshed = eventually(|| held.load(Ordering::SeqCst) && refreshes() >= 2).await; + + let entries = claim_entries(&storage, &scope).await; + storage.open_gate(); + let pruned = tokio::time::timeout(LIMIT, pruning).await; + let young = entries + .iter() + .filter_map(|entry| match entry { + ClaimEntry::Marker(number, at) if at.to_millis() > old => Some(*number), + _ => None, + }) + .collect::>(); + let live_markers = entries + .iter() + .filter(|entry| matches!(entry, ClaimEntry::Marker(2, _))) + .cloned() + .collect::>(); + let hold = claim_hold(grace, LONG_DEADLINE); + + assert!(matches!(pruned, Ok(Ok(Ok(())))), "{pruned:?}"); + assert_eq!( + ( + refreshed, + entries.contains(&ClaimEntry::Claim(2)), + young.len() >= 3, + young.iter().all(|number| *number == 2), + next_claim( + &live_markers, + golem_common::model::Timestamp::now_utc(), + hold + ), + ), + (true, true, true, true, ClaimChoice::Held) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_that_finds_a_snapshot_file_gone_at_each_attempt_after_refreshes_deletes_each_marker_of_its_claim() + { + // Each listing of the snapshot files by the prune takes 250 ms, and the grace period gives a + // new marker each 100 ms, so the claim gets new markers before the release. Each read of a + // snapshot file after the claim finds it gone. + let claimed = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let claimed = claimed.clone(); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + let after_claim = claimed.load(Ordering::SeqCst); + if after_claim && op_label == "list" && path == Path::new("snapshots") { + Script::Delay(Duration::from_millis(250)) + } else if after_claim && path.starts_with("snapshots") && !is_forget(op_label, path) { + Script::Vanish + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_millis(400)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let refreshes = storage + .calls() + .iter() + .filter(|(op_label, _)| *op_label == "refresh_claim") + .count(); + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert_eq!( + ( + refreshes > 0, + prunes(&storage.calls()), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, 0, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_slower_than_the_grace_period_keeps_its_claim_fresh() { + // The gate holds the prune at its listing of the packs for longer than the grace period. + let grace = Duration::from_millis(400); + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let pruning = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let prune_held = eventually(|| held.load(Ordering::SeqCst)).await; + let first = claim_time(&storage, &scope).await.unwrap_or(u64::MAX); + let wanted = first.saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)); + + let refreshed = tokio::time::timeout( + LIMIT, + futures::stream::repeat(()) + .then(|()| async { + tokio::time::sleep(Duration::from_millis(20)).await; + claim_time(&storage, &scope).await + }) + .filter(|time| std::future::ready(time.is_some_and(|time| time >= wanted))) + .boxed() + .next(), + ) + .await + .is_ok(); + let second = store.delete(&scope, &name("p-2")).await; + let prunes_while_held = prunes(&storage.calls()); + storage.open_gate(); + let pruned = tokio::time::timeout(LIMIT, pruning).await; + + assert!(second.is_ok(), "{second:?}"); + assert!(matches!(pruned, Ok(Ok(Ok(())))), "{pruned:?}"); + assert_eq!( + ( + prune_held, + refreshed, + prunes_while_held, + prunes(&storage.calls()), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, true, 1, 1, Vec::::new()) + ); +} + +/// Tells whether the operation label is a call of a rustic backend. +fn is_backend_call(op_label: &str) -> bool { + matches!( + op_label, + "stat" | "list" | "read" | "read_range" | "write" | "delete" + ) +} + +/// Gives the entries of the claims of the scope. +async fn claim_entries(storage: &ScriptedBlobStorage, scope: &SnapshotScope) -> Vec { + blobs(storage, &scope.0, "golem/prune-claims/") + .await + .iter() + .filter_map(|path| parse_claim_entry(Path::new(path).file_name()?.to_str()?)) + .collect() +} + +#[test] +#[timeout("60s")] +async fn a_prune_whose_refreshes_fail_stops_when_its_lease_runs_out_and_keeps_its_claim() { + // A zero grace period and a deadline of 200 ms give a lease of 199 ms. The claim write and the + // second read of the ledger each take 150 ms, so the lease has run out when the prune makes its + // first call. Each refresh fails. + let deadline = Duration::from_millis(200); + let slow = Duration::from_millis(150); + let claimed = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let claimed = claimed.clone(); + move |op_label, _| match op_label { + "write_claim" => { + claimed.store(true, Ordering::SeqCst); + Script::Delay(slow) + } + "read_ledger" if claimed.load(Ordering::SeqCst) => Script::Delay(slow), + "refresh_claim" => Script::Refuse, + _ => Script::Pass, + } + }); + let store = store(storage.clone(), policy(deadline, ALWAYS, Duration::ZERO)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + let calls = storage.calls(); + let backend_calls_after_claim = calls + .iter() + .position(|(op_label, _)| *op_label == "write_claim") + .map(|claimed_at| { + calls[claimed_at..] + .iter() + .filter(|(op_label, _)| is_backend_call(op_label)) + .count() + }); + let entries = claim_entries(&storage, &scope).await; + let hold = claim_hold(Duration::ZERO, deadline); + let now = golem_common::model::Timestamp::now_utc(); + let after_hold = golem_common::model::Timestamp::from( + now.to_millis() + u64::try_from((hold + Duration::from_secs(1)).as_millis()).unwrap(), + ); + + assert!( + matches!( + &deleted, + Err(error @ SnapshotStoreError::Storage { source, .. }) + if is_storage(error, true) && is_lease_expired(source.as_ref()) + ), + "{deleted:?}" + ); + assert_eq!( + ( + backend_calls_after_claim, + entries.contains(&ClaimEntry::Claim(0)), + next_claim(&entries, now, hold), + next_claim(&entries, after_hold, hold), + ), + (Some(0), true, ClaimChoice::Held, ClaimChoice::Claim(1)) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_goes_on_after_one_failed_refresh_when_the_later_refreshes_succeed() { + // A zero grace period and a deadline of 200 ms give a lease of 199 ms and a refresh about each + // 50 ms. Each call of the prune takes 40 ms, so the prune runs for more than one lease. The + // first refresh fails, and the later ones succeed. + let deadline = Duration::from_millis(200); + let claimed = Arc::new(AtomicBool::new(false)); + let refused = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, refused) = (claimed.clone(), refused.clone()); + move |op_label, _| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "refresh_claim" && !refused.swap(true, Ordering::SeqCst) { + Script::Refuse + } else if is_backend_call(op_label) && claimed.load(Ordering::SeqCst) { + Script::Delay(Duration::from_millis(40)) + } else { + Script::Pass + } + } + }); + let store = store(storage.clone(), policy(deadline, ALWAYS, Duration::ZERO)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + let calls = storage.calls(); + let backend_calls_after_claim = calls + .iter() + .position(|(op_label, _)| *op_label == "write_claim") + .map_or(0, |claimed_at| { + calls[claimed_at..] + .iter() + .filter(|(op_label, _)| is_backend_call(op_label)) + .count() + }); + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + refused.load(Ordering::SeqCst), + backend_calls_after_claim * 40 > 200, + prunes(&calls), + ledger(&storage, &scope).await.last_prune.is_some(), + ), + (true, true, 1, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_marker_ahead_within_the_margin_after_a_slow_claim_listing_holds_the_claim() { + // Another delete wrote a marker whose time is 130 s ahead of the clock at the start of the + // delete, beyond the margin of 120 s. The gate holds the claim listing while the clock goes + // 20 s on, so after the listing the marker is 110 s ahead, within the margin, and it holds. + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "list_claims" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let ahead = golem_common::model::Timestamp::now_utc().to_millis() + 130_000; + futures::stream::iter([ + "golem/prune-claims/none/0".to_string(), + format!("golem/prune-claims/none/0@{ahead}-{}", uuid::Uuid::new_v4()), + ]) + .for_each(|path| { + let (storage, scope) = (storage.clone(), scope.clone()); + async move { + storage + .put_raw("test", "test", scope.0.clone(), Path::new(&path), &[]) + .await + .unwrap(); + } + }) + .await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "list_claims") + }) + .await; + + store.clock_ahead.store(20_000, Ordering::SeqCst); + storage.open_gate(); + let deleted = tokio::time::timeout(LIMIT, deleting).await; + let calls = storage.calls(); + + assert!(matches!(deleted, Ok(Ok(Ok(())))), "{deleted:?}"); + assert_eq!( + ( + held, + prunes(&calls), + calls + .iter() + .filter(|(op_label, _)| *op_label == "write_claim") + .count(), + ), + (true, 0, 0) + ); +} + +#[test] +#[timeout("60s")] +async fn a_ledger_write_after_the_lease_ran_out_is_not_sent_and_the_claim_stays() { + // A zero grace period and a deadline of 1 s give a lease of 999 ms. Each refresh fails, and + // the final marker takes longer than its deadline, so it fails at 1 s, after the end of the + // lease. The prune itself ends well within the lease. + let deadline = Duration::from_secs(1); + let storage = + ScriptedBlobStorage::new( + Arc::new(InMemoryBlobStorage::new()), + |op_label, _| match op_label { + "refresh_claim" => Script::Refuse, + "final_marker" => Script::RefuseAfter(Duration::from_millis(1200)), + _ => Script::Pass, + }, + ); + let store = store(storage.clone(), policy(deadline, ALWAYS, Duration::ZERO)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + let calls = storage.calls(); + let entries = claim_entries(&storage, &scope).await; + + assert!( + matches!( + &deleted, + Err(SnapshotStoreError::Storage { source, .. }) if is_lease_expired(source.as_ref()) + ), + "{deleted:?}" + ); + assert_eq!( + ( + prunes(&calls), + calls + .iter() + .filter(|(op_label, _)| *op_label == "write_ledger") + .count(), + ledger(&storage, &scope).await.last_prune, + entries.contains(&ClaimEntry::Claim(0)), + ), + (1, 0, None, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_ledger_write_that_starts_within_the_lease_ends_at_the_end_of_the_lease() { + // A zero grace period and a deadline of 2 s give a lease of 1999 ms. The final marker takes + // 1.5 s, so the ledger write starts with about 0.5 s of the lease left. The ledger write takes + // 10 s, so the end of the lease ends it before its own deadline. + let deadline = Duration::from_secs(2); + let storage = + ScriptedBlobStorage::new( + Arc::new(InMemoryBlobStorage::new()), + |op_label, _| match op_label { + "final_marker" => Script::Delay(Duration::from_millis(1500)), + "write_ledger" => Script::Delay(Duration::from_secs(10)), + _ => Script::Pass, + }, + ); + let store = store(storage.clone(), policy(deadline, ALWAYS, Duration::ZERO)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + let calls = storage.calls(); + let entries = claim_entries(&storage, &scope).await; + + assert!( + matches!( + &deleted, + Err(SnapshotStoreError::Storage { source, .. }) if is_lease_expired(source.as_ref()) + ), + "{deleted:?}" + ); + assert_eq!( + ( + calls + .iter() + .filter(|(op_label, _)| *op_label == "write_ledger") + .count(), + ledger(&storage, &scope).await.last_prune, + entries.contains(&ClaimEntry::Claim(0)), + ), + (1, None, true) + ); +} + +#[test] +#[timeout("60s")] +async fn the_lease_of_a_prune_starts_at_its_first_marker_so_a_prune_without_a_refresh_prunes() { + // A grace period of one hour gives a refresh period of fifteen minutes, so the prune ends + // before its first refresh, and only the first marker gives the lease. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + let calls = storage.calls(); + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + prunes(&calls), + calls + .iter() + .filter(|(op_label, _)| *op_label == "refresh_claim") + .count(), + ledger(&storage, &scope).await.last_prune.is_some(), + ), + (1, 0, true) + ); +} + +/// The operation labels of the calls of the prune decision that read the records of freed bytes +/// and what they name: the listing of the records, the read of a record, the listing of the +/// snapshot files, and the listing of the packs. +const DECISION_READS: [&str; 4] = ["list_freed", "read_freed", "list_snapshots", "list_data"]; + +/// Gives the labels of [`DECISION_READS`] that the calls from `from` on made. +fn decision_reads(storage: &ScriptedBlobStorage, from: usize) -> Vec<&'static str> { + let calls = storage.calls(); + DECISION_READS + .into_iter() + .filter(|label| { + calls + .iter() + .skip(from) + .any(|(op_label, _)| op_label == label) + }) + .collect() +} + +#[test] +#[timeout("60s")] +async fn a_delete_within_the_hold_reads_no_record_of_freed_bytes_and_lists_no_pack() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2", "p-3"]).await; + store.delete(&scope, &name("p-1")).await.unwrap(); + let pruned = ledger(&storage, &scope).await.last_prune.is_some(); + let from = storage.calls().len(); + + let deleted = store.delete(&scope, &name("p-2")).await; + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + (pruned, decision_reads(&storage, from)), + (true, Vec::<&str>::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_whose_named_bytes_are_below_the_threshold_reads_no_record_content() { + // A threshold of all the bytes of the repository is above the bytes that one delete frees. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, Percent(100), Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let from = storage.calls().len(); + + let deleted = store.delete(&scope, &name("p-1")).await; + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + decision_reads(&storage, from), + ledger(&storage, &scope).await.last_prune.is_some(), + freed(&storage, &scope).await > 0, + ), + (vec!["list_freed", "list_data"], false, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_whose_named_bytes_reach_the_threshold_reads_and_settles_the_records_and_prunes() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let from = storage.calls().len(); + + let deleted = store.delete(&scope, &name("p-1")).await; + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + decision_reads(&storage, from), + prunes(&storage.calls()), + ledger(&storage, &scope).await.last_prune.is_some(), + freed(&storage, &scope).await, + ), + (DECISION_READS.to_vec(), 1, true, 0) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_deletes_the_claims_of_old_ledgers() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + futures::stream::iter(["golem/prune-claims/100/0", "golem/prune-claims/200/4"]) + .for_each(|path| { + let storage = storage.clone(); + let scope = scope.clone(); + async move { + storage + .put_raw("test", "test", scope.0.clone(), Path::new(path), b"100") + .await + .unwrap(); + } + }) + .await; + + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert_eq!( + ( + ledger(&storage, &scope).await.last_prune.is_some(), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_keeps_the_claims_of_a_newer_ledger() { + // A newer claim directory can hold a live claim of a later prune, so the cleanup of a prune + // leaves it. Its time is far after the end of this prune. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let newer = "golem/prune-claims/99999999999999/0"; + let not_a_time = "golem/prune-claims/later/0"; + futures::stream::iter(["golem/prune-claims/100/0", newer, not_a_time]) + .for_each(|path| { + let storage = storage.clone(); + let scope = scope.clone(); + async move { + storage + .put_raw("test", "test", scope.0.clone(), Path::new(path), b"") + .await + .unwrap(); + } + }) + .await; + + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert_eq!( + ( + ledger(&storage, &scope).await.last_prune.is_some(), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, vec![newer.to_string(), not_a_time.to_string()]) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_deletes_an_empty_claim_directory_of_an_old_ledger() { + // A listing of the blobs does not find an empty directory, so only a listing of the + // directories finds it. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + storage + .create_dir( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-claims/100"), + ) + .await + .unwrap(); + let list_claims = || { + storage.list_dir( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-claims"), + ) + }; + let before = list_claims().await.unwrap(); + + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert_eq!( + ( + before, + ledger(&storage, &scope).await.last_prune.is_some(), + list_claims().await.unwrap(), + ), + ( + vec![std::path::PathBuf::from("golem/prune-claims/100")], + true, + Vec::::new() + ) + ); +} + +#[test] +#[timeout("60s")] +async fn a_failed_ledger_write_after_a_prune_keeps_the_claim_so_no_second_prune_runs() { + // The first prune runs and its ledger write fails. A delete right after it finds the claim + // and does not prune. When the claim is older than the grace period and the margin, the next + // delete prunes. + let refused = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refused = refused.clone(); + move |op_label, _| { + if op_label == "write_ledger" && !refused.swap(true, Ordering::SeqCst) { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2", "p-3"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + let second = store.delete(&scope, &name("p-2")).await; + let prunes_after_second = prunes(&storage.calls()); + age_claims(&storage, &scope).await; + let third = store.delete(&scope, &name("p-3")).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!(second.is_ok(), "{second:?}"); + assert!(third.is_ok(), "{third:?}"); + assert_eq!( + ( + claims_after_failure + .iter() + .filter(|path| !path.contains('@')) + .count(), + prunes_after_second, + prunes(&storage.calls()), + ledger(&storage, &scope).await.last_prune.is_some(), + ), + (1, 1, 2, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_backend_that_does_not_build_after_the_claim_releases_it() { + // The first claim write makes the next backend build fail, and that build is the one of the + // prune. + let refuse_backends = Arc::new(AtomicBool::new(false)); + let armed = Arc::new(AtomicBool::new(true)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse_backends = refuse_backends.clone(); + move |op_label, _| { + if op_label == "write_claim" && armed.swap(false, Ordering::SeqCst) { + refuse_backends.store(true, Ordering::SeqCst); + } + Script::Pass + } + }); + let store = Arc::new(RusticSnapshotStore { + refuse_backends: refuse_backends.clone(), + ..RusticSnapshotStore::with_policy( + storage.clone(), + key(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ) + }); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + refuse_backends.store(false, Ordering::SeqCst); + let retried = store.delete(&scope, &name("p-2")).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!(retried.is_ok(), "{retried:?}"); + assert_eq!( + ( + claims_after_failure, + ledger(&storage, &scope).await.last_prune.is_some(), + ), + (Vec::::new(), true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_forget_that_fails_after_the_record_write_leaves_the_record() { + // The forget deletes the snapshot file, and the storage refuses that call. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "delete" && path.starts_with("snapshots") { + Script::Refuse + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + + assert!( + deleted.as_ref().is_err_and(|error| is_storage(error, true)), + "{deleted:?}" + ); + assert_eq!( + ( + freed(&storage, &scope).await > 0, + listed_names(&store, &scope).await + ), + (true, vec!["p-1".to_string()]) + ); +} + +/// Tells whether the call is the forget of a delete: the delete of a snapshot file. +fn is_forget(op_label: &str, path: &Path) -> bool { + op_label == "delete" && path.starts_with("snapshots") +} + +/// Gives the path of each record of freed bytes of the scope. +async fn records(storage: &InMemoryBlobStorage, scope: &SnapshotScope) -> Vec { + blobs(storage, &scope.0, "golem/prune-freed/").await +} + +#[test] +#[timeout("60s")] +async fn a_prune_keeps_the_record_of_a_delete_that_has_not_forgotten_its_snapshot() { + // The first delete writes its record and waits at its forget. The second delete prunes and + // must not count that record. After the forget, the next due prune counts it. + let inner = Arc::new(InMemoryBlobStorage::new()); + let plain = store( + inner.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&plain, &scope, &["p-1", "p-2"]).await; + let held = ScriptedBlobStorage::new(inner.clone(), |op_label, path| { + if is_forget(op_label, path) { + Script::WaitForGate + } else { + Script::Pass + } + }); + let pausing = store( + held.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let paused = tokio::spawn({ + let scope = scope.clone(); + async move { pausing.delete(&scope, &name("p-1")).await } + }); + let at_forget = eventually(|| { + held.calls() + .iter() + .any(|(op_label, path)| is_forget(op_label, Path::new(path))) + }) + .await; + let paused_record = records(&inner, &scope).await; + + plain.delete(&scope, &name("p-2")).await.unwrap(); + let after_prune = records(&inner, &scope).await; + let pruned = ledger(&inner, &scope).await; + age_ledger(&inner, &scope, &pruned).await; + held.open_gate(); + let resumed = tokio::time::timeout(LIMIT, paused).await; + + assert!(matches!(resumed, Ok(Ok(Ok(())))), "{resumed:?}"); + assert_eq!( + ( + at_forget, + paused_record.len(), + after_prune == paused_record, + pruned.last_prune.is_some(), + records(&inner, &scope).await, + ledger(&inner, &scope).await.last_prune > pruned.last_prune, + ), + (true, 1, true, true, Vec::::new(), true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_forget_that_lands_while_a_prune_runs_keeps_its_record_for_the_next_prune() { + // The prune of the second delete checks the records before the first delete forgets. The + // forget lands while that prune runs, so the record was not settled at the check, and it must + // stay for the next prune. + let inner = Arc::new(InMemoryBlobStorage::new()); + let plain = store( + inner.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&plain, &scope, &["p-1", "p-2", "p-3"]).await; + let forgetting = ScriptedBlobStorage::new(inner.clone(), |op_label, path| { + if is_forget(op_label, path) { + Script::Step { + refuse: false, + late: false, + } + } else { + Script::Pass + } + }); + let claimed = Arc::new(AtomicBool::new(false)); + let pruning = ScriptedBlobStorage::new(inner.clone(), { + let claimed = claimed.clone(); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" && path == Path::new("data") && claimed.load(Ordering::SeqCst) { + Script::Step { + refuse: false, + late: false, + } + } else { + Script::Pass + } + } + }); + let first = tokio::spawn({ + let deleting = store( + forgetting.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = scope.clone(); + async move { deleting.delete(&scope, &name("p-1")).await } + }); + let first_at_forget = eventually(|| forgetting.waiting_steps() > 0).await; + let first_record = records(&inner, &scope).await; + let second = tokio::spawn({ + let deleting = store( + pruning.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = scope.clone(); + async move { deleting.delete(&scope, &name("p-2")).await } + }); + let second_at_prune = eventually(|| pruning.waiting_steps() > 0).await; + + forgetting.step(); + let first = tokio::time::timeout(LIMIT, first).await; + pruning.step(); + let second = tokio::time::timeout(LIMIT, second).await; + let after_prune = records(&inner, &scope).await; + let pruned = ledger(&inner, &scope).await; + age_ledger(&inner, &scope, &pruned).await; + plain.delete(&scope, &name("p-3")).await.unwrap(); + + assert!(matches!(first, Ok(Ok(Ok(())))), "{first:?}"); + assert!(matches!(second, Ok(Ok(Ok(())))), "{second:?}"); + assert_eq!( + ( + first_at_forget, + second_at_prune, + first_record.len(), + after_prune == first_record, + pruned.last_prune.is_some(), + records(&inner, &scope).await, + ), + (true, true, 1, true, true, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_record_whose_snapshot_still_exists_counts_nothing_and_does_not_make_a_prune_due() { + let inner = Arc::new(InMemoryBlobStorage::new()); + let store = store( + inner.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1"]).await; + let snapshot = snapshot_files(inner.clone(), &scope) + .await + .first() + .map(|snapshot| snapshot.id.to_hex().to_string()) + .unwrap_or_default(); + inner + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-freed/1000000000-kept"), + snapshot.as_bytes(), + ) + .await + .unwrap(); + + store.delete(&scope, &name("p-unknown")).await.unwrap(); + store.delete(&scope, &name("p-unknown")).await.unwrap(); + + assert_eq!( + ( + snapshot.len(), + ledger(&inner, &scope).await.last_prune, + records(&inner, &scope).await, + ), + ( + 64, + None, + vec!["golem/prune-freed/1000000000-kept".to_string()] + ) + ); +} + +#[test] +#[timeout("60s")] +async fn a_record_write_that_answers_already_exists_counts_as_written() { + // A new try of a record write whose first answer was lost finds the record of the first try. + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "write_freed" { + Script::AnswerAlreadyExists + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + freed(&storage, &scope).await > 0, + listed_names(&store, &scope).await + ), + (true, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_sees_the_marker_of_a_claim_that_is_not_taken_yet_and_does_not_prune() { + // The gate holds the second of the two writes of the first delete: its marker and its claim. + // The second delete runs meanwhile. + let inner = Arc::new(InMemoryBlobStorage::new()); + let writes = Arc::new(AtomicUsize::new(0)); + let first = ScriptedBlobStorage::new(inner.clone(), { + let writes = writes.clone(); + move |op_label, _| { + if matches!(op_label, "write_marker" | "write_claim") + && writes.fetch_add(1, Ordering::SeqCst) == 1 + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let second = ScriptedBlobStorage::new(inner.clone(), |_, _| Script::Pass); + let grace = Duration::from_secs(3600); + let scope = new_scope(); + save_each( + &store(inner.clone(), policy(LONG_DEADLINE, NEVER, grace)), + &scope, + &["p-1", "p-2"], + ) + .await; + let holding = tokio::spawn({ + let deleting = store(first.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = scope.clone(); + async move { deleting.delete(&scope, &name("p-1")).await } + }); + let held = eventually(|| writes.load(Ordering::SeqCst) >= 2).await; + + let seen = store(second.clone(), policy(LONG_DEADLINE, ALWAYS, grace)) + .delete(&scope, &name("p-2")) + .await; + first.open_gate(); + let holding = tokio::time::timeout(LIMIT, holding).await; + + assert!(seen.is_ok(), "{seen:?}"); + assert!(matches!(holding, Ok(Ok(Ok(())))), "{holding:?}"); + assert_eq!( + (held, prunes(&second.calls()), prunes(&first.calls())), + (true, 0, 1) + ); +} + +#[test] +#[timeout("60s")] +async fn a_claim_without_a_marker_does_not_unblock_a_live_holder() { + // The first delete holds claim 0 and waits at the start of its prune. A claim 1 without a + // marker is in the same directory. + let inner = Arc::new(InMemoryBlobStorage::new()); + let claimed = Arc::new(AtomicBool::new(false)); + let first = ScriptedBlobStorage::new(inner.clone(), { + let claimed = claimed.clone(); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" && path == Path::new("data") && claimed.load(Ordering::SeqCst) { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let second = ScriptedBlobStorage::new(inner.clone(), |_, _| Script::Pass); + let grace = Duration::from_secs(3600); + let scope = new_scope(); + save_each( + &store(inner.clone(), policy(LONG_DEADLINE, NEVER, grace)), + &scope, + &["p-1", "p-2"], + ) + .await; + let holding = tokio::spawn({ + let deleting = store(first.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = scope.clone(); + async move { deleting.delete(&scope, &name("p-1")).await } + }); + let held = eventually(|| { + first + .calls() + .iter() + .any(|(op_label, path)| *op_label == "list" && path == "data") + }) + .await; + inner + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-claims/none/1"), + b"", + ) + .await + .unwrap(); + + let seen = store(second.clone(), policy(LONG_DEADLINE, ALWAYS, grace)) + .delete(&scope, &name("p-2")) + .await; + first.open_gate(); + let holding = tokio::time::timeout(LIMIT, holding).await; + + assert!(seen.is_ok(), "{seen:?}"); + assert!(matches!(holding, Ok(Ok(Ok(())))), "{holding:?}"); + assert_eq!((held, prunes(&second.calls())), (true, 0)); +} + +#[test] +#[timeout("60s")] +async fn two_deletes_that_read_the_same_ledger_make_one_prune() { + // The first read after the first claim is the start of the first prune. The gate holds it, + // so the second delete reads the ledger that the first delete read. + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, _| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "read" + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let first = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let first_held = eventually(|| held.load(Ordering::SeqCst)).await; + + let second = store.delete(&scope, &name("p-2")).await; + storage.open_gate(); + let first = tokio::time::timeout(LIMIT, first).await; + let calls = storage.calls(); + + assert!(matches!(first, Ok(Ok(Ok(())))), "{first:?}"); + assert!(second.is_ok(), "{second:?}"); + assert_eq!( + ( + first_held, + prunes(&calls), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, 1, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_that_claims_after_another_prune_removed_the_claims_does_not_prune() { + // The gate holds the first delete after its ledger read and before its listing of the claims. + // The second delete prunes to its end, so the first delete claims in a removed directory. + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let held = held.clone(); + move |op_label, _| { + if op_label == "list_claims" && !held.swap(true, Ordering::SeqCst) { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let late = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let late_held = eventually(|| held.load(Ordering::SeqCst)).await; + + let pruned = store.delete(&scope, &name("p-2")).await; + storage.open_gate(); + let late = tokio::time::timeout(LIMIT, late).await; + + assert!(pruned.is_ok(), "{pruned:?}"); + assert!(matches!(late, Ok(Ok(Ok(())))), "{late:?}"); + assert_eq!( + ( + late_held, + prunes(&storage.calls()), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, 1, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_failed_second_read_of_the_ledger_deletes_the_claim_and_a_retry_of_the_delete_prunes() { + // The first call after the first claim is the second read of the ledger, and it fails. So the + // first delete does not prune, and only the retry counts as a prune. + let claimed = Arc::new(AtomicBool::new(false)); + let refused = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, refused) = (claimed.clone(), refused.clone()); + move |op_label, _| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + Script::Pass + } else if claimed.load(Ordering::SeqCst) && !refused.swap(true, Ordering::SeqCst) { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + let retried = store.delete(&scope, &name("p-1")).await; + let after = ledger(&storage, &scope).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!(retried.is_ok(), "{retried:?}"); + assert_eq!( + ( + claims_after_failure, + after.last_prune.is_some(), + prunes(&storage.calls()) + ), + (Vec::::new(), true, 1) + ); +} + +/// A storage that refuses the first call after the first claim write, which is the second read of +/// the ledger. It holds each claim delete, and each ledger read after the claim when `hold_read` is +/// true, until the gate opens. +fn failing_after_the_claim(hold_read: bool) -> Arc { + let claimed = Arc::new(AtomicBool::new(false)); + let refused = Arc::new(AtomicBool::new(false)); + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), move |op_label, _| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + Script::Pass + } else if op_label == "delete_claim" + || (hold_read && op_label == "read_ledger" && claimed.load(Ordering::SeqCst)) + { + Script::WaitForGate + } else if !hold_read + && claimed.load(Ordering::SeqCst) + && !refused.swap(true, Ordering::SeqCst) + { + Script::Refuse + } else { + Script::Pass + } + }) +} + +#[test] +#[timeout("60s")] +async fn a_delete_dropped_after_a_failed_second_read_still_releases_its_claim() { + // The second read of the ledger fails, and the gate holds the delete of the claim. The test + // drops the delete there, and then opens the gate. + let storage = failing_after_the_claim(false); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "delete_claim") + }) + .await; + + deleting.abort(); + let dropped = deleting.await; + let claims_at_drop = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + storage.open_gate(); + let ended = eventually(|| store.work_in_flight() == 0).await; + + assert!( + dropped.as_ref().is_err_and(|error| error.is_cancelled()), + "{dropped:?}" + ); + assert_eq!( + ( + held, + claims_at_drop.is_empty(), + ended, + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, false, true, Vec::::new()) + ); +} + +/// A storage whose gate holds the second read of the ledger after the claim, and also each delete +/// of a claim or a marker when `hold_release` is true. +fn holding_after_the_claim(hold_release: bool) -> Arc { + let claimed = Arc::new(AtomicBool::new(false)); + ScriptedBlobStorage::new( + Arc::new(InMemoryBlobStorage::new()), + move |op_label, _| match op_label { + "write_claim" => { + claimed.store(true, Ordering::SeqCst); + Script::Pass + } + "read_ledger" if claimed.load(Ordering::SeqCst) => Script::WaitForGate, + "delete_claim" | "delete_marker" if hold_release => Script::WaitForGate, + _ => Script::Pass, + }, + ) +} + +#[test] +#[timeout("60s")] +async fn a_delete_dropped_after_its_claim_and_before_its_prune_releases_the_claim() { + // The gate holds the second read of the ledger. The test drops the delete there, before its + // prune starts. + let storage = holding_after_the_claim(false); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let claimed = |calls: &[(&'static str, String)]| { + calls + .iter() + .filter(|(op_label, _)| *op_label == "read_ledger") + .count() + >= 2 + }; + let held = eventually(|| claimed(&storage.calls())).await; + let claims_at_drop = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + + deleting.abort(); + let dropped = deleting.await; + let ended = eventually(|| store.work_in_flight() == 0).await; + let claims_after = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + storage.open_gate(); + + assert!( + dropped.as_ref().is_err_and(|error| error.is_cancelled()), + "{dropped:?}" + ); + assert_eq!( + ( + held, + claims_at_drop.len(), + ended, + prunes(&storage.calls()), + claims_after, + ), + (true, 2, true, 0, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn shut_down_waits_for_the_release_of_a_delete_dropped_after_its_claim() { + // The gate holds the second read of the ledger and each delete of the release. The test drops + // the delete at the read, so its guard starts the release, which waits at the gate. + let storage = holding_after_the_claim(true); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .filter(|(op_label, _)| *op_label == "read_ledger") + .count() + >= 2 + }) + .await; + deleting.abort(); + let _ = deleting.await; + let releasing = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "delete_claim") + }) + .await; + + let shutting = store.shut_down(); + tokio::pin!(shutting); + let waited = tokio::time::timeout(Duration::from_millis(200), &mut shutting) + .await + .is_err(); + storage.open_gate(); + // A finished future must not be polled again, so the second wait runs only after a first wait + // that timed out. + let stopped = !waited || tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + + assert_eq!( + ( + held, + releasing, + waited, + stopped, + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, true, true, true, Vec::::new()) + ); +} + +/// A storage whose gate holds the listing of the packs, which is the first call of a prune, and +/// whose final marker takes 200 ms. +fn holding_the_prune() -> Arc { + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "list" && path == Path::new("data") { + Script::WaitForGate + } else if op_label == "final_marker" { + Script::Delay(Duration::from_millis(200)) + } else { + Script::Pass + } + }) +} + +/// Gives the number of final marker writes in the calls. +fn final_markers(calls: &[(&'static str, String)]) -> usize { + calls + .iter() + .filter(|(op_label, _)| *op_label == "final_marker") + .count() +} + +/// Gives the number of markers of the claims of the scope. +async fn markers(storage: &ScriptedBlobStorage, scope: &SnapshotScope) -> usize { + claim_entries(storage, scope) + .await + .iter() + .filter(|entry| matches!(entry, ClaimEntry::Marker(..))) + .count() +} + +#[test] +#[timeout("60s")] +async fn a_shut_down_during_a_started_prune_still_writes_its_final_marker_and_waits_for_it() { + // The gate holds the first call of the prune, and the shut down cancels it. A grace period of + // one hour gives no refresh, so the claim has its first marker and its final marker. + let storage = holding_the_prune(); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let started = eventually(|| prunes(&storage.calls()) == 1).await; + + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + let markers_at_shut_down = markers(&storage, &scope).await; + storage.open_gate(); + let deleted = tokio::time::timeout(LIMIT, deleting).await; + + assert!(matches!(deleted, Ok(Ok(Err(_)))), "{deleted:?}"); + assert_eq!( + ( + started, + stopped, + markers_at_shut_down, + final_markers(&storage.calls()) + ), + (true, true, 2, 1) + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_dropped_during_a_started_prune_writes_its_final_marker() { + // The gate holds the first call of the prune, and the test drops the delete there. + let storage = holding_the_prune(); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let started = eventually(|| prunes(&storage.calls()) == 1).await; + + deleting.abort(); + let dropped = deleting.await; + let ended = eventually(|| store.work_in_flight() == 0).await; + let markers_after = markers(&storage, &scope).await; + storage.open_gate(); + + assert!( + dropped.as_ref().is_err_and(|error| error.is_cancelled()), + "{dropped:?}" + ); + assert_eq!( + ( + started, + ended, + markers_after, + final_markers(&storage.calls()) + ), + (true, true, 2, 1) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_writes_one_final_marker() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + let ended = eventually(|| store.work_in_flight() == 0).await; + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + ended, + prunes(&storage.calls()), + final_markers(&storage.calls()) + ), + (true, 1, 1) + ); +} + +#[test] +#[timeout("60s")] +async fn a_shut_down_after_the_claim_still_releases_it() { + // The gate holds the second read of the ledger, and the shut down cancels that read. The + // gate stays shut for the read, and it opens only for the delete of the claim. + let storage = failing_after_the_claim(true); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .skip_while(|(op_label, _)| *op_label != "write_claim") + .any(|(op_label, _)| *op_label == "read_ledger") + }) + .await; + + let shutting_down = tokio::spawn({ + let store = store.clone(); + async move { store.shut_down().await } + }); + let reached_release = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "delete_claim") + }) + .await; + let claims_at_release = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + storage.open_gate(); + let shut_down = tokio::time::timeout(LIMIT, shutting_down).await; + let deleted = tokio::time::timeout(LIMIT, deleting).await; + + assert!(matches!(shut_down, Ok(Ok(()))), "{shut_down:?}"); + assert!(matches!(deleted, Ok(Ok(Err(_)))), "{deleted:?}"); + assert_eq!( + ( + held, + reached_release, + claims_at_release.is_empty(), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, true, false, Vec::::new()) + ); +} + +/// A storage that refuses the first call after the first claim write, which is the second read of +/// the ledger, and each call of the release with the operation label. +fn refusing_the_release_call(release_label: &'static str) -> Arc { + let claimed = Arc::new(AtomicBool::new(false)); + let refused = Arc::new(AtomicBool::new(false)); + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), move |op_label, _| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + Script::Pass + } else if op_label == release_label + || (claimed.load(Ordering::SeqCst) && !refused.swap(true, Ordering::SeqCst)) + { + Script::Refuse + } else { + Script::Pass + } + }) +} + +#[test] +#[timeout("60s")] +async fn a_release_whose_claim_delete_is_refused_still_deletes_its_marker_and_the_next_delete_prunes() + { + let storage = refusing_the_release_call("delete_claim"); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + let retried = store.delete(&scope, &name("p-1")).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!(retried.is_ok(), "{retried:?}"); + assert_eq!( + ( + claims_after_failure, + prunes(&storage.calls()), + ledger(&storage, &scope).await.last_prune.is_some(), + ), + (vec!["golem/prune-claims/none/0".to_string()], 1, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_release_whose_marker_delete_is_refused_still_deletes_the_claim() { + let storage = refusing_the_release_call("delete_marker"); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert_eq!( + claims_after_failure + .iter() + .map(|path| path.starts_with("golem/prune-claims/none/0@")) + .collect::>(), + vec![true] + ); +} + +/// A storage that gives no blob to the first `vanishing` reads of a snapshot file after the first +/// claim write, as a forget of another delete between the listing and the read does. It counts +/// those reads. +fn vanishing_snapshots_after_the_claim( + vanishing: usize, +) -> (Arc, Arc) { + let claimed = Arc::new(AtomicBool::new(false)); + let vanished = Arc::new(AtomicUsize::new(0)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let vanished = vanished.clone(); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + let snapshot_read = claimed.load(Ordering::SeqCst) + && path.starts_with("snapshots") + && !is_forget(op_label, path) + && op_label != "list"; + if snapshot_read + && vanished + .fetch_update(Ordering::SeqCst, Ordering::SeqCst, |count| { + (count < vanishing).then_some(count + 1) + }) + .is_ok() + { + Script::Vanish + } else { + Script::Pass + } + } + }); + (storage, vanished) +} + +#[test] +#[timeout("60s")] +async fn a_prune_plans_again_when_a_snapshot_file_is_gone_at_its_read_and_prunes_once() { + let (storage, vanished) = vanishing_snapshots_after_the_claim(1); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let deleted = store.delete(&scope, &name("p-1")).await; + + assert!(deleted.is_ok(), "{deleted:?}"); + assert_eq!( + ( + vanished.load(Ordering::SeqCst), + prunes(&storage.calls()), + ledger(&storage, &scope).await.last_prune.is_some(), + ), + (1, 1, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_that_finds_a_snapshot_file_gone_at_each_attempt_releases_its_claim_and_gives_a_retryable_error() + { + let (storage, vanished) = vanishing_snapshots_after_the_claim(usize::MAX); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert_eq!( + ( + vanished.load(Ordering::SeqCst), + prunes(&storage.calls()), + ledger(&storage, &scope).await.last_prune.is_some(), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (3, 0, false, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_that_fails_keeps_its_claim_so_no_second_prune_runs_within_the_hold() { + // The first listing of the packs by a prune after the first claim fails. Only a prune lists + // the packs with that call, so the second read of the ledger passes and the prune fails after + // it started. Its claim stays, so a retry within the hold does not prune, and a retry after it + // does. + let claimed = Arc::new(AtomicBool::new(false)); + let refused = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, refused) = (claimed.clone(), refused.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !refused.swap(true, Ordering::SeqCst) + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + let retried = store.delete(&scope, &name("p-1")).await; + let after_retry = ledger(&storage, &scope).await; + age_claims(&storage, &scope).await; + let later = store.delete(&scope, &name("p-1")).await; + let after = ledger(&storage, &scope).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!(retried.is_ok(), "{retried:?}"); + assert!(later.is_ok(), "{later:?}"); + assert_eq!( + ( + claims_after_failure + .iter() + .filter(|path| !path.contains('@')) + .count(), + refused.load(Ordering::SeqCst), + after_retry.last_prune, + after.last_prune.is_some() + ), + (1, true, None, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_claim_without_a_marker_does_not_block_a_prune() { + let grace = Duration::from_secs(3600); + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-claims/none/0"), + &[], + ) + .await + .unwrap(); + + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert!(ledger(&storage, &scope).await.last_prune.is_some()); +} + +#[test] +#[timeout("60s")] +async fn a_prune_that_succeeds_deletes_the_claims_and_the_counted_records_of_freed_bytes() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert_eq!( + ( + ledger(&storage, &scope).await.last_prune.is_some(), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + blobs(&*storage, &scope.0, "golem/prune-freed/").await, + ), + (true, Vec::::new(), Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_below_the_threshold_does_not_prune() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), one_file_tree("kept")); + store + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path(), None) + .await + .unwrap(); + let packs_before = blobs(&*storage, &scope.0, "data/").await; + + store.delete(&scope, &name("p-deleted")).await.unwrap(); + let after = ledger(&storage, &scope).await; + let after_freed = freed(&storage, &scope).await; + + assert_eq!( + ( + after_freed > 0, + after.last_prune, + blobs(&*storage, &scope.0, "data/").await, + ), + (true, None, packs_before) + ); +} + +#[test] +#[timeout("60s")] +async fn no_second_prune_runs_within_the_hold_after_a_prune() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + let trees = [one_file_tree("a"), one_file_tree("b"), one_file_tree("c")]; + futures::stream::iter(["p-a", "p-b", "p-c"].into_iter().zip(&trees)) + .for_each(|(text, tree)| { + let store = store.clone(); + let scope = scope.clone(); + async move { + store + .save(&scope, &name(text), tree.path(), None) + .await + .unwrap(); + } + }) + .await; + + store.delete(&scope, &name("p-a")).await.unwrap(); + let after_first = ledger(&storage, &scope).await; + store.delete(&scope, &name("p-b")).await.unwrap(); + let after_second = ledger(&storage, &scope).await; + let after_second_freed = freed(&storage, &scope).await; + + assert_eq!( + ( + after_first.last_prune.is_some(), + after_second.last_prune == after_first.last_prune, + after_second_freed > 0, + ), + (true, true, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { + let refuse = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) + && matches!(op_label, "read" | "read_range") + && path.starts_with("data") + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path(), None) + .await + .unwrap(); + refuse.store(true, Ordering::SeqCst); + + let failed = store.delete(&scope, &name("p-deleted")).await; + let after_failure = ledger(&storage, &scope).await; + let after_failure_freed = freed(&storage, &scope).await; + refuse.store(false, Ordering::SeqCst); + // The prune started, so its claim stays until the hold passed. + age_claims(&storage, &scope).await; + let retried = store.delete(&scope, &name("p-deleted")).await; + let after_retry = ledger(&storage, &scope).await; + let after_retry_freed = freed(&storage, &scope).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert_eq!( + ( + after_failure_freed > 0, + after_failure.last_prune, + retried.is_ok(), + after_retry_freed, + after_retry.last_prune.is_some(), + ), + (true, None, true, 0, true) + ); +} + +#[test] +#[timeout("60s")] +async fn a_deleted_scope_holds_no_blob() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let (first, second) = (one_file_tree("first"), one_file_tree("second")); + store + .save(&scope, &name("p-1"), first.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), second.path(), None) + .await + .unwrap(); + store.delete(&scope, &name("p-1")).await.unwrap(); + let before = blobs(&*storage, &scope.0, "").await; + + store.delete_scope(&scope).await.unwrap(); + + assert_eq!( + ( + before + .iter() + .any(|path| path.starts_with("golem/prune-ledgers/")), + blobs(&*storage, &scope.0, "").await + ), + (true, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_name_free() { + // The first save counts the calls of a save. Each later round holds one of these calls: the + // call reaches the storage and never answers, as a write that S3 received and completes after + // the caller left. The round drops the save there, and waits until the store has no work. + let tree = one_file_tree("dropped"); + let other = one_file_tree("saved later"); + let counted = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + store( + counted.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ) + .save(&new_scope(), &name("p-dropped"), tree.path(), None) + .await + .unwrap(); + let calls = counted.calls().len(); + + let rounds = futures::stream::iter(1..=calls) + .then(|held| { + let (tree, other) = (tree.path().to_path_buf(), other.path().to_path_buf()); + async move { + let inner = Arc::new(InMemoryBlobStorage::new()); + let seen = Arc::new(AtomicUsize::new(0)); + let storage = ScriptedBlobStorage::new(inner.clone(), move |_, _| { + if seen.fetch_add(1, Ordering::SeqCst) + 1 == held { + Script::NeverAnswer + } else { + Script::Pass + } + }); + let policy = policy(LONG_DEADLINE, NEVER, Duration::ZERO); + let dropping = store(storage.clone(), policy); + let scope = new_scope(); + let ended = drop_when( + &storage, + |calls| calls.len() >= held, + dropping.save(&scope, &name("p-dropped"), &tree, None), + ) + .await; + let stopped = tokio::time::timeout(LIMIT, dropping.shut_down()) + .await + .is_ok(); + let later = store(inner, policy); + let stat = later.stat(&scope, &name("p-dropped")).await.ok().flatten(); + let names = listed_names(&later, &scope).await; + let saved_again = later.save(&scope, &name("p-dropped"), &other, None).await; + let restored = restored_listing(&later, &scope, &name("p-dropped")) + .await + .ok(); + ( + held, + ended.is_none(), + stopped, + stat, + names, + saved_again.is_ok(), + restored == Some(listing(&other)), + ) + } + }) + .collect::>() + .await; + + assert_eq!( + rounds, + (1..=calls) + .map(|held| (held, true, true, None, Vec::new(), true, true)) + .collect::, + Vec, + bool, + bool + )>>() + ); +} + +#[test] +#[timeout("60s")] +async fn a_blob_call_of_a_cancelled_operation_does_not_start() { + // The in-memory storage answers at the first poll, so only the check before the call keeps + // the call from the storage. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let cancel = tokio_util::sync::CancellationToken::new(); + cancel.cancel(); + let files = SnapshotFiles { + storage: storage.clone(), + namespace: new_scope().0, + deadline: LONG_DEADLINE, + cancel, + tracker: tokio_util::task::TaskTracker::new(), + }; + + let read = files + .get("read_ledger", Path::new("golem/prune-ledgers/1000-0-0f0f")) + .await; + + assert!(read.is_err(), "{read:?}"); + assert_eq!(storage.calls(), Vec::new()); +} + +#[test] +#[timeout("60s")] +async fn shut_down_waits_for_a_blob_call_of_the_store_that_is_not_polled() { + // The test polls the scope delete one time, so its first blob call waits at the gate, and + // then the test does not poll it again. Only the tracker makes the shut down wait for it. + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "delete_scope" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let deleting = store.delete_scope(&scope); + tokio::pin!(deleting); + let pending = futures::poll!(&mut deleting).is_pending(); + + let shutting = store.shut_down(); + tokio::pin!(shutting); + let waited = tokio::time::timeout(Duration::from_millis(200), &mut shutting) + .await + .is_err(); + let deleted = tokio::time::timeout(LIMIT, &mut deleting).await; + // A finished future must not be polled again, so the second wait runs only after a first wait + // that timed out. + let stopped = !waited || tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + storage.open_gate(); + + assert!( + matches!(&deleted, Ok(Err(error)) if is_storage(error, true)), + "{deleted:?}" + ); + assert_eq!( + (pending, waited, stopped, store.work_in_flight()), + (true, true, true, 0) + ); +} + +#[test] +#[timeout("60s")] +async fn shut_down_waits_for_the_step_of_a_prune_and_its_refresh_that_is_not_polled() { + // The test drives the delete until its prune waits at the gate and its refresh runs, and then + // does not poll it. Only the tracker of the step makes the shut down wait for it. + let grace = Duration::from_millis(400); + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleted_name = name("p-1"); + let deleting = store.delete(&scope, &deleted_name); + tokio::pin!(deleting); + let reached = tokio::select! { + biased; + () = async { + eventually(|| held.load(Ordering::SeqCst)).await; + } => true, + _ = &mut deleting => false, + }; + + let shutting = store.shut_down(); + tokio::pin!(shutting); + let waited = tokio::time::timeout(Duration::from_millis(200), &mut shutting) + .await + .is_err(); + let deleted = tokio::time::timeout(LIMIT, &mut deleting).await; + // A finished future must not be polled again, so the second wait runs only after a first wait + // that timed out. + let stopped = !waited || tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + let refreshes = |calls: &[(&'static str, String)]| { + calls + .iter() + .filter(|(op_label, _)| *op_label == "refresh_claim") + .count() + }; + let at_stop = refreshes(&storage.calls()); + tokio::time::sleep(grace).await; + storage.open_gate(); + + assert!( + matches!(&deleted, Ok(Err(error)) if is_storage(error, true)), + "{deleted:?}" + ); + assert_eq!( + ( + reached, + waited, + stopped, + store.work_in_flight(), + refreshes(&storage.calls()) - at_stop, + ), + (true, true, true, 0, 0) + ); +} + +#[test] +#[timeout("60s")] +async fn a_copy_fails_when_a_prune_removed_an_index_file_that_it_listed() { + // The gate holds the read of the listed index file, and the test deletes that file meanwhile, + // as a prune does. + let inner = Arc::new(InMemoryBlobStorage::new()); + let storage = ScriptedBlobStorage::new(inner.clone(), |op_label, path| { + if op_label == "copy_read" && path.starts_with("index") { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let (from, to) = (new_scope(), new_scope()); + save_each(&store, &from, &["p-1"]).await; + let copying = tokio::spawn({ + let store = store.clone(); + let (from, to) = (from.clone(), to.clone()); + async move { store.copy_scope(&from, &to).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, path)| *op_label == "copy_read" && path.starts_with("index")) + }) + .await; + let index_files = blobs(&*inner, &from.0, "index/").await; + + futures::stream::iter(&index_files) + .for_each(|path| { + let (inner, from) = (&inner, &from); + async move { + inner + .delete("test", "test", from.0.clone(), Path::new(path)) + .await + .unwrap(); + } + }) + .await; + storage.open_gate(); + let copied = tokio::time::timeout(LIMIT, copying).await; + + assert!( + matches!(&copied, Ok(Ok(Err(error))) if is_storage(error, true)), + "{copied:?}" + ); + assert_eq!( + ( + held, + index_files.is_empty(), + blobs(&*inner, &to.0, "config").await + ), + (true, false, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_copy_leaves_out_a_snapshot_file_that_a_delete_removed_after_the_listing() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "copy_read" && path.starts_with("snapshots") { + Script::Vanish + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let (from, to) = (new_scope(), new_scope()); + save_each(&store, &from, &["p-1"]).await; + + let copied = store.copy_scope(&from, &to).await; + + assert!(copied.is_ok(), "{copied:?}"); + assert_eq!( + ( + blobs(&*storage, &to.0, "config").await.len(), + blobs(&*storage, &to.0, "snapshots/").await + ), + (1, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_call() { + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "copy_list" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let (from, to) = (new_scope(), new_scope()); + let tree = one_file_tree("copied"); + store + .save(&from, &name("p-1"), tree.path(), None) + .await + .unwrap(); + let mut copying = tokio::spawn({ + let store = store.clone(); + let (from, to) = (from.clone(), to.clone()); + async move { store.copy_scope(&from, &to).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "copy_list") + }) + .await; + + // The gate stays closed until the end, so only the cancel can end the held call. + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + let calls_at_stop = storage.calls().len(); + let copied = tokio::time::timeout(LIMIT, &mut copying).await; + let calls_after_copy = storage.calls().len(); + storage.open_gate(); + + assert!( + matches!(&copied, Ok(Ok(Err(error))) if is_storage(error, true)), + "{copied:?}" + ); + assert_eq!( + (held, stopped, calls_after_copy), + (true, true, calls_at_stop) + ); +} + +#[test] +#[timeout("60s")] +async fn a_publish_held_at_its_storage_call_keeps_shut_down_waiting_until_it_ends() { + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "publish" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("published"); + let saving = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + let path = tree.path().to_path_buf(); + async move { store.save(&scope, &name("p-held"), &path, None).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "publish") + }) + .await; + + let shutting = store.shut_down(); + tokio::pin!(shutting); + let waited = tokio::time::timeout(Duration::from_millis(200), &mut shutting) + .await + .is_err(); + storage.open_gate(); + // A finished future must not be polled again, so the second wait runs only after a first wait + // that timed out. + let stopped = !waited || tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + let saved = tokio::time::timeout(LIMIT, saving).await; + + assert!(matches!(&saved, Ok(Ok(Ok(_)))), "{saved:?}"); + assert_eq!((held, waited, stopped), (true, true, true)); +} + +#[test] +#[timeout("60s")] +async fn a_shut_down_between_the_claim_listing_and_the_claim_guard_makes_no_storage_call_after_it_returns() + { + // The gate holds the delete after it listed the claims and before it builds its claim guard, + // so the tracker is empty and `shut_down` returns while the delete waits. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let gate = Arc::new(StepGate::default()); + let store = Arc::new(RusticSnapshotStore { + claim_gate: Some(gate.clone()), + ..RusticSnapshotStore::with_policy( + storage.clone(), + key(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ) + }); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let deleting = tokio::spawn({ + let (store, scope) = (store.clone(), scope.clone()); + async move { store.delete(&scope, &name("p-1")).await } + }); + let reached = tokio::time::timeout(LIMIT, gate.reached.notified()) + .await + .is_ok(); + + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + let calls_at_shut_down = storage.calls(); + gate.open.notify_one(); + let deleted = tokio::time::timeout(LIMIT, deleting).await; + let ended = eventually(|| store.work_in_flight() == 0).await; + + assert!( + matches!( + &deleted, + Ok(Ok(Err(error))) if is_storage(error, false) + ), + "{deleted:?}" + ); + assert_eq!( + (reached, stopped, ended, storage.calls()), + (true, true, true, calls_at_shut_down) + ); +} + +#[test] +#[timeout("60s")] +async fn a_save_that_loses_the_creation_of_the_repository_after_its_first_config_check_saves_into_the_winner() + { + // The gate holds the second check of the config by the losing save, which the init of rustic + // makes before its config write. The winning save creates the repository meanwhile, so that + // check finds the config. + let shared = Arc::new(InMemoryBlobStorage::new()); + let checks = Arc::new(AtomicUsize::new(0)); + let losing = ScriptedBlobStorage::new(shared.clone(), { + let checks = checks.clone(); + move |op_label, path| { + if op_label == "stat" + && path == Path::new("config") + && checks.fetch_add(1, Ordering::SeqCst) == 1 + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let winner = store(shared.clone(), policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let loser = store(losing.clone(), policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + let (won_tree, lost_tree) = (one_file_tree("winner"), one_file_tree("loser")); + let losing_save = tokio::spawn({ + let (loser, scope, path) = (loser.clone(), scope.clone(), lost_tree.path().to_path_buf()); + async move { loser.save(&scope, &name("p-loser"), &path, None).await } + }); + let held = eventually(|| checks.load(Ordering::SeqCst) >= 2).await; + + let won = winner + .save(&scope, &name("p-winner"), won_tree.path(), None) + .await; + losing.open_gate(); + let lost = tokio::time::timeout(LIMIT, losing_save).await; + + assert!(won.is_ok(), "{won:?}"); + assert!(matches!(lost, Ok(Ok(Ok(_)))), "{lost:?}"); + assert_eq!( + ( + held, + restored_listing(&winner, &scope, &name("p-loser")) + .await + .ok(), + restored_listing(&loser, &scope, &name("p-winner")) + .await + .ok(), + ), + ( + true, + Some(listing(lost_tree.path())), + Some(listing(won_tree.path())), + ) + ); +} + +#[test] +#[timeout("60s")] +async fn shut_down_waits_for_the_check_of_the_tree_of_a_save() { + // The gate holds the check of the tree on its blocking thread. + let gate = Arc::new(StepGate::default()); + let store = Arc::new(RusticSnapshotStore { + path_check_gate: Some(gate.clone()), + ..RusticSnapshotStore::with_policy( + Arc::new(InMemoryBlobStorage::new()), + key(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ) + }); + let scope = new_scope(); + let tree = one_file_tree("checked"); + let saving = tokio::spawn({ + let (store, scope, path) = (store.clone(), scope.clone(), tree.path().to_path_buf()); + async move { store.save(&scope, &name("p-checked"), &path, None).await } + }); + let reached = tokio::time::timeout(LIMIT, gate.reached.notified()) + .await + .is_ok(); + + let shutting = store.shut_down(); + tokio::pin!(shutting); + let waited = tokio::time::timeout(Duration::from_millis(200), &mut shutting) + .await + .is_err(); + gate.open.notify_one(); + // A finished future must not be polled again, so the second wait runs only after a first wait + // that timed out. + let stopped = !waited || tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + let saved = tokio::time::timeout(LIMIT, saving).await; + + assert!( + matches!(&saved, Ok(Ok(Err(error))) if is_storage(error, false) || is_storage(error, true)), + "{saved:?}" + ); + assert_eq!((reached, waited, stopped), (true, true, true)); +} + +/// A store over the shared storage whose index reads wait at the gate while `hold` is set, and the +/// number of index reads that reached the storage while it was set. +fn holding_index_reads( + shared: &Arc, +) -> ( + Arc, + Arc, + Arc, + Arc, +) { + let (hold, held) = ( + Arc::new(AtomicBool::new(false)), + Arc::new(AtomicUsize::new(0)), + ); + let storage = ScriptedBlobStorage::new(shared.clone(), { + let (hold, held) = (hold.clone(), held.clone()); + move |op_label, path| { + if op_label == "read" && path.starts_with("index") && hold.load(Ordering::SeqCst) { + held.fetch_add(1, Ordering::SeqCst); + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + (store, storage, hold, held) +} + +/// Saves `p-1` and `p-2` of two trees in a new scope of the shared storage, and deletes `p-2` +/// through a store that prunes at once. The prune deletes the index file of `p-2`. +async fn scope_with_a_prune_to_come( + shared: &Arc, +) -> (SnapshotScope, Arc, Scratch) { + let pruning = store( + shared.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let (kept, deleted) = (fixture_tree(), one_file_tree("deleted")); + pruning + .save(&scope, &name("p-1"), kept.path(), None) + .await + .unwrap(); + pruning + .save(&scope, &name("p-2"), deleted.path(), None) + .await + .unwrap(); + (scope, pruning, kept) +} + +#[test] +#[timeout("60s")] +async fn a_restore_whose_index_file_a_prune_deleted_after_the_listing_gives_retryable_storage_and_a_retry_restores() + { + let shared = Arc::new(InMemoryBlobStorage::new()); + let (scope, pruning, kept) = scope_with_a_prune_to_come(&shared).await; + let (restoring, storage, hold, held) = holding_index_reads(&shared); + hold.store(true, Ordering::SeqCst); + let first = tokio::spawn({ + let (restoring, scope) = (restoring.clone(), scope.clone()); + async move { + let into = Scratch::new(); + restoring.restore(&scope, &name("p-1"), into.path()).await + } + }); + let reached = eventually(|| held.load(Ordering::SeqCst) > 0).await; + + let deleted = pruning.delete(&scope, &name("p-2")).await; + let pruned = ledger(&shared, &scope).await.last_prune.is_some(); + hold.store(false, Ordering::SeqCst); + storage.open_gate(); + let first = tokio::time::timeout(LIMIT, first).await; + let again = restored_listing(&restoring, &scope, &name("p-1")).await; + + assert!( + matches!(&first, Ok(Ok(Err(error))) if is_storage(error, true)), + "{first:?}" + ); + assert_eq!( + (reached, deleted.is_ok(), pruned, again.ok()), + (true, true, true, Some(listing(kept.path()))) + ); +} + +#[test] +#[timeout("60s")] +async fn a_save_whose_index_file_a_prune_deleted_after_the_listing_gives_retryable_storage_and_a_retry_saves() + { + let shared = Arc::new(InMemoryBlobStorage::new()); + let (scope, pruning, _) = scope_with_a_prune_to_come(&shared).await; + let (saving, storage, hold, held) = holding_index_reads(&shared); + let tree = one_file_tree("new"); + hold.store(true, Ordering::SeqCst); + let first = tokio::spawn({ + let (saving, scope, path) = (saving.clone(), scope.clone(), tree.path().to_path_buf()); + async move { saving.save(&scope, &name("p-3"), &path, None).await } + }); + let reached = eventually(|| held.load(Ordering::SeqCst) > 0).await; + + let deleted = pruning.delete(&scope, &name("p-2")).await; + let pruned = ledger(&shared, &scope).await.last_prune.is_some(); + hold.store(false, Ordering::SeqCst); + storage.open_gate(); + let first = tokio::time::timeout(LIMIT, first).await; + let again = saving.save(&scope, &name("p-3"), tree.path(), None).await; + + assert!( + matches!(&first, Ok(Ok(Err(error))) if is_storage(error, true)), + "{first:?}" + ); + assert!(again.is_ok(), "{again:?}"); + assert_eq!( + ( + reached, + deleted.is_ok(), + pruned, + restored_listing(&saving, &scope, &name("p-3")).await.ok(), + ), + (true, true, true, Some(listing(tree.path()))) + ); +} + +#[test] +#[timeout("60s")] +async fn a_save_dropped_during_the_delete_after_a_failed_publish_still_deletes_the_snapshot_file() { + // The publish write lands and loses its answer, so the publish deletes the file. The gate + // holds that delete, and the test drops the save there. + let storage = + ScriptedBlobStorage::new( + Arc::new(InMemoryBlobStorage::new()), + |op_label, _| match op_label { + "publish" => Script::LoseTheAnswer, + "retract" => Script::WaitForGate, + _ => Script::Pass, + }, + ); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("dropped"); + let saving = tokio::spawn({ + let (store, scope, path) = (store.clone(), scope.clone(), tree.path().to_path_buf()); + async move { store.save(&scope, &name("p-dropped"), &path, None).await } + }); + let retracting = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "retract") + }) + .await; + let written = blobs(&*storage, &scope.0, "snapshots/").await.len(); + + saving.abort(); + let dropped = saving.await; + storage.open_gate(); + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + + assert!( + dropped.as_ref().is_err_and(|error| error.is_cancelled()), + "{dropped:?}" + ); + assert_eq!( + ( + retracting, + written, + stopped, + blobs(&*storage, &scope.0, "snapshots/").await, + RusticSnapshotStore::with_policy( + storage.clone(), + key(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ) + .stat(&scope, &name("p-dropped")) + .await + .ok(), + ), + (true, 1, true, Vec::::new(), Some(None)) + ); +} + +#[test] +#[timeout("60s")] +async fn a_save_whose_publish_is_counted_before_shut_down_and_polled_after_its_cancel_publishes_nothing() + { + // The gate holds the save after the tracker counts its publish and before the first poll of + // the publish. `shut_down` cancels and then waits for the publish, and the gate opens only + // after the cancel. + let storage = Arc::new(InMemoryBlobStorage::new()); + let gate = Arc::new(StepGate::default()); + let store = Arc::new(RusticSnapshotStore { + publish_poll_gate: Some(gate.clone()), + ..RusticSnapshotStore::with_policy( + storage.clone(), + key(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ) + }); + let scope = new_scope(); + let tree = one_file_tree("late"); + let saving = tokio::spawn({ + let (store, scope, path) = (store.clone(), scope.clone(), tree.path().to_path_buf()); + async move { store.save(&scope, &name("p-late"), &path, None).await } + }); + let reached = tokio::time::timeout(LIMIT, gate.reached.notified()) + .await + .is_ok(); + + let shutting_down = tokio::spawn({ + let store = store.clone(); + async move { store.shut_down().await } + }); + let cancelled = eventually(|| store.root.is_cancelled()).await; + gate.open.notify_one(); + let saved = tokio::time::timeout(LIMIT, saving).await; + let stopped = tokio::time::timeout(LIMIT, shutting_down).await; + + assert!( + matches!(&saved, Ok(Ok(Err(error))) if is_storage(error, false)), + "{saved:?}" + ); + assert!(matches!(stopped, Ok(Ok(()))), "{stopped:?}"); + assert_eq!( + ( + reached, + cancelled, + blobs(&*storage, &scope.0, "snapshots/").await + ), + (true, true, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_save_that_reaches_its_publish_after_shut_down_publishes_nothing() { + // The gate holds the save after its blocking work, so the tracker is empty and `shut_down` + // returns before the publish starts. + let storage = Arc::new(InMemoryBlobStorage::new()); + let gate = Arc::new(StepGate::default()); + let store = Arc::new(RusticSnapshotStore { + publish_gate: Some(gate.clone()), + ..RusticSnapshotStore::with_policy( + storage.clone(), + key(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ) + }); + let scope = new_scope(); + let tree = one_file_tree("late"); + let saving = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + let path = tree.path().to_path_buf(); + async move { store.save(&scope, &name("p-late"), &path, None).await } + }); + let reached = tokio::time::timeout(LIMIT, gate.reached.notified()) + .await + .is_ok(); + + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + gate.open.notify_one(); + let saved = tokio::time::timeout(LIMIT, saving).await; + + assert!( + matches!(&saved, Ok(Ok(Err(error))) if is_storage(error, false)), + "{saved:?}" + ); + assert_eq!( + ( + reached, + stopped, + blobs(&*storage, &scope.0, "snapshots/").await, + store.work_in_flight(), + ), + (true, true, Vec::::new(), 0) + ); +} + +#[test] +#[timeout("60s")] +async fn delete_scope_and_copy_scope_after_shut_down_give_storage() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let (scope, other) = (new_scope(), new_scope()); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store.shut_down().await; + + let deleted = store.delete_scope(&scope).await; + let copied = store.copy_scope(&scope, &other).await; + + assert!( + deleted + .as_ref() + .is_err_and(|error| is_storage(error, false)), + "{deleted:?}" + ); + assert!( + copied.as_ref().is_err_and(|error| is_storage(error, false)), + "{copied:?}" + ); +} + +#[test] +#[timeout("60s")] +async fn shut_down_ends_running_operations_before_it_returns() { + // The first pack write of the save waits at the gate, and it cancels `reached`, so the test + // shuts the store down only when the save holds a running storage call. The save runs at a low + // priority, so the test waits for that point without a bound of its own. + let reached = tokio_util::sync::CancellationToken::new(); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let reached = reached.clone(); + move |op_label, path| { + if op_label == "write" && path.starts_with("data") { + reached.cancel(); + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + let saving = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + let path = tree.path().to_path_buf(); + async move { store.save(&scope, &name("p-held"), &path, None).await } + }); + reached.cancelled().await; + + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + let saved = tokio::time::timeout(LIMIT, saving).await; + let later = store.stat(&scope, &name("p-held")).await; + storage.open_gate(); + + assert!( + matches!(&saved, Ok(Ok(Err(error))) if is_storage(error, true)), + "{saved:?}" + ); + assert!( + later.as_ref().is_err_and(|error| is_storage(error, false)), + "{later:?}" + ); + assert_eq!((stopped, store.work_in_flight()), (true, 0)); +} + +/// The operation of the store that a test drops. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum Dropped { + Restore, + Stat, + List, + Delete, +} + +#[test] +#[timeout("60s")] +async fn a_dropped_operation_stops_its_blocking_work() { + let outcomes = futures::stream::iter([ + Dropped::Restore, + Dropped::Stat, + Dropped::List, + Dropped::Delete, + ]) + .then(|dropped| async move { + let held_call = move |op_label: &str, path: &Path| match dropped { + Dropped::Restore => op_label == "read_range" && path.starts_with("data"), + Dropped::Stat | Dropped::List => op_label == "read" && path.starts_with("snapshots"), + Dropped::Delete => op_label == "delete" && path.starts_with("snapshots"), + }; + let hold = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let hold = hold.clone(); + move |op_label, path| { + if held_call(op_label, path) && hold.load(Ordering::SeqCst) { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + hold.store(true, Ordering::SeqCst); + let before = storage.calls().len(); + let reached = |calls: &[(&'static str, String)]| { + calls[before..] + .iter() + .any(|(op_label, path)| held_call(op_label, Path::new(path))) + }; + let into = Scratch::new(); + let ended = match dropped { + Dropped::Restore => { + drop_when( + &storage, + reached, + store.restore(&scope, &name("p-1"), into.path()).map(|_| ()), + ) + .await + } + Dropped::Stat => { + drop_when( + &storage, + reached, + store.stat(&scope, &name("p-1")).map(|_| ()), + ) + .await + } + Dropped::List => drop_when(&storage, reached, store.list(&scope).map(|_| ())).await, + Dropped::Delete => { + drop_when( + &storage, + reached, + store.delete(&scope, &name("p-1")).map(|_| ()), + ) + .await + } + }; + let held = reached(&storage.calls()); + let stopped = eventually(|| store.work_in_flight() == 0).await; + storage.open_gate(); + (dropped, held, ended.is_none(), stopped) + }) + .collect::>() + .await; + + assert_eq!( + outcomes, + [ + Dropped::Restore, + Dropped::Stat, + Dropped::List, + Dropped::Delete + ] + .map(|dropped| (dropped, true, true, true)) + ); +} + +/// Gives the snapshot files of the scope that the bridge reads, on a blocking thread. +async fn snapshot_files( + storage: Arc, + scope: &SnapshotScope, +) -> Vec { + let namespace = scope.0.clone(); + tokio::task::spawn_blocking(move || { + let backend = super::super::backend::BlobBackend::new( + storage, + namespace, + tokio::runtime::Handle::current(), + LONG_DEADLINE, + ); + let repository = open_existing(Arc::new(backend), &key()).unwrap().unwrap(); + scope_snapshots(&repository).unwrap().readable + }) + .await + .unwrap() +} + +/// Gives, for the snapshot with the name, the numbers of new, changed and unmodified files that its +/// save counted, and the id of its parent. +fn read_counts( + files: &[rustic_core::repofile::SnapshotFile], + name: &str, +) -> Option<((u64, u64, u64), Option)> { + let snapshot = files.iter().find(|snapshot| snapshot.label == name)?; + let summary = snapshot.summary.as_ref()?; + Some(( + ( + summary.files_new, + summary.files_changed, + summary.files_unmodified, + ), + snapshot.parent, + )) +} + +fn id_of(files: &[rustic_core::repofile::SnapshotFile], name: &str) -> Option { + files + .iter() + .find(|snapshot| snapshot.label == name) + .map(|snapshot| snapshot.id) +} + +#[test] +#[timeout("60s")] +async fn a_size_and_mtime_save_of_a_copied_tree_reads_no_unchanged_file() { + // A copy gives each file a new inode and a new change time, and keeps its size and its + // modification time, as a capture does. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = three_file_tree(); + let copy = Scratch::new(); + wait_past_change_times(&entries(tree.path())); + copy_flat_tree(tree.path(), copy.path()); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-2"), + copy.path(), + Some((&name("p-1"), ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + let files = snapshot_files(storage, &scope).await; + + assert_eq!( + read_counts(&files, "p-2"), + Some(((0, 0, 3), id_of(&files, "p-1"))) + ); +} + +#[test] +#[timeout("60s")] +async fn a_full_save_reads_each_file() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = three_file_tree(); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-2"), + tree.path(), + Some((&name("p-1"), ChangeDetection::Full)), + ) + .await + .unwrap(); + let files = snapshot_files(storage, &scope).await; + + assert_eq!(read_counts(&files, "p-2"), Some(((3, 0, 0), None))); +} + +#[test] +#[timeout("60s")] +async fn the_parent_of_a_save_is_the_named_snapshot_also_when_a_newer_snapshot_exists() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = three_file_tree(); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-3"), + tree.path(), + Some((&name("p-1"), ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + let files = snapshot_files(storage, &scope).await; + + assert_eq!( + ( + read_counts(&files, "p-3"), + id_of(&files, "p-1") == id_of(&files, "p-2") + ), + (Some(((0, 0, 3), id_of(&files, "p-1"))), false) + ); +} + +#[test] +#[timeout("60s")] +async fn a_save_without_a_parent_or_with_a_parent_that_the_scope_does_not_hold_reads_each_file() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = three_file_tree(); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-3"), + tree.path(), + Some((&name("p-missing"), ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + let files = snapshot_files(storage, &scope).await; + + assert_eq!( + (read_counts(&files, "p-2"), read_counts(&files, "p-3")), + (Some(((3, 0, 0), None)), Some(((3, 0, 0), None))) + ); +} + +#[test] +#[timeout("60s")] +async fn a_save_whose_read_of_the_snapshot_files_fails_while_it_finds_the_parent_gives_storage_and_publishes_nothing() + { + let refuse = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) && op_label == "read" && path.starts_with("snapshots") + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("parent"); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + refuse.store(true, Ordering::SeqCst); + let before = storage.calls().len(); + + let saved = store + .save( + &scope, + &name("p-2"), + tree.path(), + Some((&name("p-1"), ChangeDetection::SizeMtime)), + ) + .await; + let publishes = storage.calls()[before..] + .iter() + .filter(|(op_label, _)| *op_label == "publish") + .count(); + refuse.store(false, Ordering::SeqCst); + + assert!( + saved.as_ref().is_err_and(|error| is_storage(error, true)), + "{saved:?}" + ); + assert_eq!( + (publishes, listed_names(&store, &scope).await), + (0, vec!["p-1".to_string()]) + ); +} + +#[test] +#[timeout("60s")] +async fn a_failed_read_of_a_snapshot_file_fails_stat_and_list_with_a_retryable_storage_error() { + let refuse = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) && op_label == "read" && path.starts_with("snapshots") + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path(), None) + .await + .unwrap(); + refuse.store(true, Ordering::SeqCst); + + let stat = store.stat(&scope, &name("p-kept")).await; + let list = store.list(&scope).await; + + assert!( + stat.as_ref().is_err_and(|error| is_storage(error, true)), + "{stat:?}" + ); + assert!( + list.as_ref().is_err_and(|error| is_storage(error, true)), + "{list:?}" + ); +} + +#[test] +#[timeout("60s")] +async fn a_snapshot_file_that_is_gone_after_the_listing_is_left_out() { + let gone = format!("snapshots/{}", "cd".repeat(32)); + let inner = Arc::new(InMemoryBlobStorage::new()); + let storage = ScriptedBlobStorage::new(inner.clone(), { + let gone = gone.clone(); + move |op_label, path| { + if op_label == "read" && path == Path::new(&gone) { + Script::Vanish + } else { + Script::Pass + } + } + }); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path(), None) + .await + .unwrap(); + inner + .put_raw("test", "test", scope.0.clone(), Path::new(&gone), b"listed") + .await + .unwrap(); + + let unknown = store.stat(&scope, &name("p-unknown")).await; + let names = listed_names(&store, &scope).await; + + assert!(matches!(unknown, Ok(None)), "{unknown:?}"); + assert_eq!(names, vec!["p-kept".to_string()]); +} + +#[test] +#[timeout("60s")] +async fn a_delete_that_frees_nothing_writes_no_ledger() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path(), None) + .await + .unwrap(); + + store.delete(&scope, &name("p-unknown")).await.unwrap(); + + assert_eq!( + blobs(&*storage, &scope.0, "golem/").await, + Vec::::new() + ); +} + +#[test] +fn a_prune_that_marks_only_a_pack_that_no_index_lists_leaves_marked_packs() { + assert!(leaves_marked_packs(&PruneReport { + packs_used: 3, + packs_unindexed: 1, + ..PruneReport::default() + })); +} + +#[test] +#[timeout("60s")] +async fn a_due_prune_that_marks_a_pack_that_no_index_lists_records_the_marked_pack() { + // The pack of the kept snapshot stays in use, so the pack that no index lists is the only pack + // that the prune marks. The ledger holds freed bytes from an earlier delete, so a delete of an + // unknown name prunes. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path(), None) + .await + .unwrap(); + let unindexed = format!("data/ab/{}", "ab".repeat(32)); + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new(&unindexed), + b"a pack that no index lists", + ) + .await + .unwrap(); + set_last_prune(&storage, &scope, golem_common::model::Timestamp::from(0)).await; + + store.delete(&scope, &name("p-unknown")).await.unwrap(); + let after = ledger(&storage, &scope).await; + let after_freed = freed(&storage, &scope).await; + + assert_eq!( + ( + after_freed, + after.last_prune.is_some_and(|last| last.to_millis() > 0), + after.awaiting_removal, + blobs(&*storage, &scope.0, "data/") + .await + .contains(&unindexed), + restored_listing(&store, &scope, &name("p-kept")).await.ok(), + ), + (0, true, true, true, Some(listing(tree.path()))) + ); +} + +#[test] +fn a_prune_leaves_marked_packs_when_it_marks_repacks_or_keeps_marked_packs() { + let report = |packs_unused, packs_repacked, marked_packs_kept| PruneReport { + packs_unused, + packs_repacked, + marked_packs_kept, + ..PruneReport::default() + }; + + assert_eq!( + [ + leaves_marked_packs(&report(0, 0, 0)), + leaves_marked_packs(&report(1, 0, 0)), + leaves_marked_packs(&report(0, 1, 0)), + leaves_marked_packs(&report(0, 0, 1)), + leaves_marked_packs(&PruneReport { + packs_used: 3, + marked_packs_deleted: 2, + ..PruneReport::default() + }), + ], + [false, true, true, true, false] + ); +} + +#[test] +#[timeout("60s")] +async fn a_save_of_a_relative_directory_path_gives_source_and_writes_nothing() { + // Cargo runs the tests in the directory of the crate, so the path names a directory. + let relative = Path::new("src/filesystem_snapshot/contract_tests"); + assert!( + relative.is_dir(), + "the test runs in the directory of the crate" + ); + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + + let saved = store + .save(&scope, &name("p-relative"), relative, None) + .await; + + assert!( + matches!(saved, Err(SnapshotStoreError::Source(_))), + "{saved:?}" + ); + assert_eq!(blobs(&*storage, &scope.0, "").await, Vec::::new()); +} + +#[test] +#[timeout("60s")] +async fn a_save_of_a_regular_file_gives_source_and_writes_nothing() { + // The store refuses the tree before it makes a repository, so the scope stays unused. + let tree = one_file_tree("a file, not a tree"); + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + + let saved = store + .save(&scope, &name("p-file"), &tree.path().join("file.txt"), None) + .await; + + assert!( + matches!(saved, Err(SnapshotStoreError::Source(_))), + "{saved:?}" + ); + assert_eq!(blobs(&*storage, &scope.0, "").await, Vec::::new()); +} + +#[test] +#[timeout("60s")] +async fn the_ledger_counts_the_packed_bytes_that_the_deleted_snapshot_added() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + store + .save(&scope, &name("p-deleted"), tree.path(), None) + .await + .unwrap(); + let added = snapshot_files(storage.clone(), &scope) + .await + .iter() + .filter_map(|snapshot| snapshot.summary.as_ref()) + .map(|summary| summary.data_added_packed) + .sum::(); + + store.delete(&scope, &name("p-deleted")).await.unwrap(); + + assert_eq!((added > 1, freed(&storage, &scope).await), (true, added)); +} + +#[test] +#[timeout("60s")] +async fn a_config_write_that_fails_gives_a_storage_error_with_that_failure() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "write" && path == Path::new("config") { + Script::Refuse + } else { + Script::Pass + } + }); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + let tree = one_file_tree("never saved"); + + let saved = store.save(&scope, &name("p-1"), tree.path(), None).await; + + assert!( + matches!( + &saved, + Err(SnapshotStoreError::Storage { retryable: true, source }) + if format!("{source:#}").contains("the storage refused the call") + ), + "{saved:?}" + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_that_fails_without_a_storage_failure_gives_storage_that_is_not_retryable() { + // Packs of zeros with the sizes of the index give the prune a decryption error, not a failed + // storage call. The forget before the prune has succeeded, so the delete gives `Storage`. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path(), None) + .await + .unwrap(); + let packs = storage + .list_blobs_below("test", "test", scope.0.clone(), Path::new("data")) + .await + .unwrap(); + futures::future::join_all(packs.iter().map(|pack| { + let zeros = vec![0; usize::try_from(pack.size).unwrap()]; + let storage = storage.clone(); + let namespace = scope.0.clone(); + async move { + storage + .put_raw("test", "test", namespace, &pack.path, &zeros) + .await + } + })) + .await + .into_iter() + .collect::>>() + .unwrap(); + + let deleted = store.delete(&scope, &name("p-deleted")).await; + + assert!( + deleted + .as_ref() + .is_err_and(|error| is_storage(error, false)), + "{deleted:?}" + ); +} + +/// The operation label, the path, and the name and the nice value of the calling thread of each +/// storage call. +#[cfg(target_os = "linux")] +type NiceCalls = Arc>>; + +/// A storage that records the nice value of the thread of each call. +#[cfg(target_os = "linux")] +fn nice_recording_storage() -> (Arc, NiceCalls) { + let calls = NiceCalls::default(); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let calls = calls.clone(); + move |op_label, path| { + calls + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .push(( + op_label.to_string(), + path.display().to_string(), + std::thread::current() + .name() + .unwrap_or_default() + .to_string(), + super::super::priority::own_nice(), + )); + Script::Pass + } + }); + (storage, calls) +} + +/// Takes the recorded calls as the operation label, the path and the nice value. +#[cfg(target_os = "linux")] +fn taken_calls(calls: &NiceCalls) -> Vec<(String, String, i32)> { + std::mem::take( + &mut *calls + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner), + ) + .into_iter() + .map(|(op_label, path, _, nice)| (op_label, path, nice)) + .collect() +} + +/// Takes the recorded calls with the operation label and a path below the directory, as the path +/// and the name of the thread of each call. +#[cfg(target_os = "linux")] +fn taken_threads(calls: &NiceCalls, op_label: &str, directory: &str) -> Vec<(String, String)> { + std::mem::take( + &mut *calls + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner), + ) + .into_iter() + .filter(|(op, path, _, _)| op == op_label && path.starts_with(directory)) + .map(|(_, path, thread, _)| (path, thread)) + .collect() +} + +#[cfg(target_os = "linux")] +#[test] +#[timeout("60s")] +async fn the_forget_of_a_delete_runs_its_storage_calls_in_a_rayon_pool_of_its_own() { + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1"]).await; + taken_calls(&calls); + + store.delete(&scope, &name("p-1")).await.unwrap(); + let forgets = taken_threads(&calls, "delete", "snapshots/"); + + assert_eq!( + ( + forgets.len(), + forgets + .iter() + .filter(|(_, thread)| !thread.starts_with("fs-snap-delete-")) + .collect::>() + ), + (1, Vec::<&(String, String)>::new()) + ); +} + +#[cfg(target_os = "linux")] +#[test] +#[timeout("60s")] +async fn the_index_load_of_a_restore_runs_its_storage_calls_in_a_rayon_pool_of_its_own() { + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1"]).await; + taken_calls(&calls); + + let into = Scratch::new(); + store + .restore(&scope, &name("p-1"), into.path()) + .await + .unwrap(); + let index_reads = taken_threads(&calls, "read", "index/"); + + assert_eq!( + ( + index_reads.is_empty(), + index_reads + .iter() + .filter(|(_, thread)| !thread.starts_with("fs-snap-restore-")) + .collect::>() + ), + (false, Vec::<&(String, String)>::new()) + ); +} + +/// Gives the operation labels of the calls, and each call that does not run at nice 19. +#[cfg(target_os = "linux")] +fn labels_and_calls_not_at_nice_19( + calls: &[(String, String, i32)], +) -> (Vec<&str>, Vec<&(String, String, i32)>) { + let mut labels = calls + .iter() + .map(|(op_label, _, _)| op_label.as_str()) + .collect::>(); + labels.sort_unstable(); + labels.dedup(); + let not_at_nice_19 = calls.iter().filter(|(_, _, nice)| *nice != 19).collect(); + (labels, not_at_nice_19) +} + +#[cfg(target_os = "linux")] +#[test] +#[timeout("60s")] +async fn the_storage_calls_of_a_save_run_at_nice_19() { + // The publish of the snapshot file runs on the async runtime after the work, so it keeps + // the normal priority. The second save reads the config, the index, the snapshot files and + // the trees of its parent, and writes the added file. + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + let tree = fixture_tree(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + std::fs::write(tree.path().join("added.txt"), b"added").unwrap(); + taken_calls(&calls); + + store + .save( + &scope, + &name("p-2"), + tree.path(), + Some((&name("p-1"), ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + let work = taken_calls(&calls) + .into_iter() + .filter(|(op_label, _, _)| op_label != "publish") + .collect::>(); + + assert_eq!( + labels_and_calls_not_at_nice_19(&work), + (vec!["list", "read", "stat", "write"], Vec::new()) + ); +} + +#[cfg(target_os = "linux")] +#[test] +#[timeout("60s")] +async fn the_storage_calls_of_a_prune_run_at_nice_19() { + // The forget of a delete runs before the ledger read at the normal priority. The ledger calls, + // the calls of the freed records, the listing of the packs and the claim calls run on the + // async runtime. + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, ALWAYS, Duration::ZERO)); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path(), None) + .await + .unwrap(); + taken_calls(&calls); + + store.delete(&scope, &name("p-deleted")).await.unwrap(); + let prune = taken_calls(&calls) + .into_iter() + .skip_while(|(op_label, _, _)| op_label != "read_ledger") + .filter(|(op_label, _, _)| { + ![ + "read_ledger", + "write_ledger", + "list_data", + "list_claims", + "write_marker", + "write_claim", + "final_marker", + "delete_marker", + "delete_claims", + "write_freed", + "list_freed", + "delete_freed", + "list_ledgers", + "delete_ledger", + "refresh_claim", + "list_claim_directories", + "list_claim_blobs", + "read_freed", + "list_snapshots", + ] + .contains(&op_label.as_str()) + }) + .collect::>(); + + assert_eq!( + labels_and_calls_not_at_nice_19(&prune), + (vec!["delete", "list", "read", "stat", "write"], Vec::new()) + ); +} + +#[cfg(target_os = "linux")] +#[test] +#[timeout("60s")] +async fn the_storage_calls_of_a_restore_run_at_the_nice_value_of_the_process() { + let process_nice = super::super::priority::own_nice(); + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let scope = new_scope(); + let tree = fixture_tree(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + std::mem::take(&mut *calls.lock().unwrap()); + + let restored = restored_listing(&store, &scope, &name("p-1")).await; + let recorded = std::mem::take(&mut *calls.lock().unwrap()); + + assert_eq!( + ( + restored.ok(), + recorded.is_empty(), + recorded + .iter() + .filter(|(_, _, _, nice)| *nice != process_nice) + .collect::>() + ), + ( + Some(listing(tree.path())), + false, + Vec::<&(String, String, String, i32)>::new() + ) + ); +} + +#[cfg(target_os = "linux")] +#[test] +#[timeout("60s")] +async fn after_saves_and_prunes_the_pools_keep_the_nice_value_of_the_process() { + // All tasks wait for each other, so each runs on its own thread of the blocking pool, and the + // idle threads that ran the saves and the prune are among them. + const TASKS: usize = 16; + let process_nice = super::super::priority::own_nice(); + let store = store( + Arc::new(InMemoryBlobStorage::new()), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path(), None) + .await + .unwrap(); + store.delete(&scope, &name("p-deleted")).await.unwrap(); + + let barrier = Arc::new(std::sync::Barrier::new(TASKS)); + let blocking = futures::future::join_all((0..TASKS).map(|_| { + let barrier = barrier.clone(); + tokio::task::spawn_blocking(move || { + barrier.wait(); + super::super::priority::own_nice() + }) + })) + .await + .into_iter() + .map(Result::unwrap) + .collect::>(); + let rayon = rayon::broadcast(|_| super::super::priority::own_nice()); + + assert_eq!( + ( + blocking.iter().all(|nice| *nice == process_nice), + rayon.iter().all(|nice| *nice == process_nice), + ), + (true, true), + "{blocking:?} {rayon:?}" + ); +} + +#[cfg(target_os = "linux")] +#[test] +#[timeout("60s")] +async fn the_storage_calls_of_the_rayon_workers_of_a_prune_that_repacks_run_at_nice_19() { + // The deleted snapshot shares a pack with the kept one, so the prune repacks that pack. The + // prune reads the index files and repacks with rayon, on the workers of the pool of the prune. + let (storage, calls) = nice_recording_storage(); + let base = policy(LONG_DEADLINE, ALWAYS, Duration::ZERO); + let store = store( + storage, + StorePolicy { + prune: PruneSettings { + repack: RepackLimits::Unlimited, + ..base.prune + }, + ..base + }, + ); + let scope = new_scope(); + let file = |content: &[u8]| Spec::File { + content: Box::from(content), + mode: 0o644, + }; + let both = Scratch::new(); + write_tree( + both.path(), + &[ + ("kept.txt", file(b"kept content")), + ("deleted.txt", file(b"deleted content")), + ], + ); + let kept = Scratch::new(); + write_tree(kept.path(), &[("kept.txt", file(b"kept content"))]); + store + .save(&scope, &name("p-both"), both.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept.path(), None) + .await + .unwrap(); + std::mem::take(&mut *calls.lock().unwrap()); + + store.delete(&scope, &name("p-both")).await.unwrap(); + let recorded = std::mem::take(&mut *calls.lock().unwrap()); + let from_workers = recorded + .iter() + .filter(|(_, _, thread, _)| thread.starts_with("fs-snap-prune-")) + .collect::>(); + + assert_eq!( + ( + recorded + .iter() + .any(|(op, path, _, _)| op == "write" && path.starts_with("data/")), + from_workers.is_empty(), + from_workers + .iter() + .filter(|(_, _, _, nice)| *nice != 19) + .count(), + restored_listing(&store, &scope, &name("p-kept")).await.ok(), + ), + (true, false, 0, Some(listing(kept.path()))) + ); +} + +/// A pool builder that cannot start a thread. +#[cfg(target_os = "linux")] +fn no_pool( + _: &str, + _: Option, +) -> Result { + rayon::ThreadPoolBuilder::new() + .num_threads(1) + .spawn_handler(|_| Err(std::io::Error::other("no thread can start here"))) + .build() +} + +#[cfg(target_os = "linux")] +#[test] +#[timeout("60s")] +async fn the_global_rayon_pool_keeps_the_nice_value_of_the_process_after_saves_without_their_pool() +{ + // Without its own pool, the rayon work of a save goes to the global pool from a thread at + // nice 19. The store starts the global pool when it is made, so its threads keep the normal + // priority also when a save is the first rayon work of the process. + let process_nice = super::super::priority::own_nice(); + let store = RusticSnapshotStore { + low_priority: super::super::priority::LowPriority { + build_pool: no_pool, + ..super::super::priority::LowPriority::new(NonZeroUsize::new(2)) + }, + ..RusticSnapshotStore::new(Arc::new(InMemoryBlobStorage::new()), &config()) + }; + let scope = new_scope(); + let (first, second) = (one_file_tree("first"), fixture_tree()); + store + .save(&scope, &name("p-1"), first.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), second.path(), None) + .await + .unwrap(); + + let global = rayon::broadcast(|_| super::super::priority::own_nice()); + + assert!( + global.iter().all(|nice| *nice == process_nice), + "{global:?}" + ); +} + +mod sweep; + +/// The largest number of steps of one turn. A delete takes at most 21 steps of the protocol, and +/// its prune writes at least one new marker of its claim, so a turn of 22 steps runs a delete to +/// its end when its prune writes one new marker. +const SWEEP_TURN: usize = 22; + +/// The number of runs of the order with a drop at the first step. +const DROP_REPEATS: usize = 300; + +/// The number of random orders that the property test tries. +const SWEEP_CASES: u32 = 1000; + +/// Saves the two snapshots that each case deletes, in a scope that each case copies. +async fn prepared_scope() -> (Arc, SnapshotScope) { + let shared = Arc::new(InMemoryBlobStorage::new()); + let prepared = new_scope(); + save_each( + &store(shared.clone(), policy(LONG_DEADLINE, NEVER, Duration::ZERO)), + &prepared, + &["p-1", "p-2"], + ) + .await; + (shared, prepared) +} + +#[test] +#[timeout("60s")] +async fn two_deletes_make_at_most_one_prune_in_each_order_with_up_to_two_switches() { + // Each order where one delete runs some steps, the other runs some steps, and then each runs + // to its end. This holds each pause of one delete while the other runs. + let (shared, prepared) = prepared_scope().await; + let schedules = (0..2).flat_map(|first| { + (0..=SWEEP_TURN).flat_map(move |one| { + (0..=SWEEP_TURN).map(move |two| sweep::Schedule { + first, + turns: vec![one, two], + fail: None, + late: None, + drop: None, + }) + }) + }); + let started = std::time::Instant::now(); + + let cases = futures::stream::iter(schedules) + .then(|schedule| { + let (shared, prepared) = (&shared, &prepared); + async move { sweep::run_case(shared, prepared, &schedule).await } + }) + .try_fold(0usize, |cases, _| async move { Ok(cases + 1) }) + .await; + + println!("{cases:?} orders in {:?}", started.elapsed()); + assert!(cases.is_ok(), "{cases:?}"); +} + +#[test] +#[timeout("60s")] +async fn two_deletes_make_at_most_one_prune_when_one_is_dropped_at_its_first_step() { + // A drop right after the first step cancels the forget of the delete while its storage call + // can already wait for a step. The race depends on timing, so the test runs the order many + // times. + let (shared, prepared) = prepared_scope().await; + let schedule = sweep::Schedule { + first: 0, + turns: vec![6, 1, 17, 14, 16, 2, 1, 12], + fail: Some((0, 17)), + late: None, + drop: Some((0, 0)), + }; + + let cases = futures::stream::iter(0..DROP_REPEATS) + .then(|_| sweep::run_case(&shared, &prepared, &schedule)) + .try_fold(0usize, |cases, _| async move { Ok(cases + 1) }) + .await; + + assert!(cases.is_ok(), "{cases:?}"); +} + +#[test] +#[timeout("60s")] +async fn two_deletes_make_at_most_one_prune_when_one_is_dropped_as_its_prune_starts() { + // A drop right after the second read of the ledger can come just after the prune started. The + // listing of the packs then waits for a step until the cancel ends it, and the guard writes + // the final marker. The race depends on timing, so the test runs the order many times. + let (shared, prepared) = prepared_scope().await; + let schedule = sweep::Schedule { + first: 1, + turns: vec![16, 10, 3, 9, 14], + fail: None, + late: None, + drop: Some((1, 10)), + }; + + let cases = futures::stream::iter(0..DROP_REPEATS) + .then(|_| sweep::run_case(&shared, &prepared, &schedule)) + .try_fold(0usize, |cases, _| async move { Ok(cases + 1) }) + .await; + + assert!(cases.is_ok(), "{cases:?}"); +} + +#[test] +#[timeout("60s")] +async fn two_deletes_make_at_most_one_prune_in_random_orders_with_a_failed_call() { + // The test generates the order of the steps, a call that fails, and a call that gets no answer + // and reaches the storage later, and shrinks a failing case to the shortest order. The seed is fixed, and no file keeps a failing case. + use proptest::prelude::{Strategy, prop}; + use proptest::test_runner::{Config, RngAlgorithm, TestRng, TestRunner}; + let (shared, prepared) = prepared_scope().await; + let runtime = tokio::runtime::Handle::current(); + let strategy = ( + 0usize..2, + prop::collection::vec(0usize..=SWEEP_TURN, 0..=8), + prop::option::of((0usize..2, 0usize..SWEEP_TURN)), + prop::option::of((0usize..2, 0usize..SWEEP_TURN, 0usize..8)), + prop::option::of((0usize..2, 0usize..SWEEP_TURN)), + ) + .prop_map(|(first, turns, fail, late, drop)| sweep::Schedule { + first, + turns, + fail, + late, + drop, + }); + let started = std::time::Instant::now(); + + let outcome = tokio::task::spawn_blocking(move || { + let mut runner = TestRunner::new_with_rng( + Config { + cases: SWEEP_CASES, + failure_persistence: None, + ..Config::default() + }, + TestRng::deterministic_rng(RngAlgorithm::ChaCha), + ); + runner + .run(&strategy, |schedule| { + runtime + .block_on(sweep::run_case(&shared, &prepared, &schedule)) + .map(|_| ()) + .map_err(proptest::test_runner::TestCaseError::fail) + }) + .map_err(|error| error.to_string()) + }) + .await; + + println!("{SWEEP_CASES} random orders in {:?}", started.elapsed()); + assert!(matches!(outcome, Ok(Ok(()))), "{outcome:?}"); +} + +/// The name and the thread count of each pool that [`recording_pool`] built. +static BUILT_POOLS: std::sync::Mutex)>> = + std::sync::Mutex::new(Vec::new()); + +/// Builds a rayon pool with the name and the thread count, and records both. Only the store of +/// one test uses it, so the record holds only the pools of that store. +fn recording_pool( + name: &'static str, + threads: Option, +) -> Result { + BUILT_POOLS + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .push((name, threads)); + rayon::ThreadPoolBuilder::new() + .num_threads(threads.map_or(0, NonZeroUsize::get)) + .build() +} + +#[test] +#[timeout("60s")] +async fn a_restore_builds_its_pool_with_the_restore_reader_threads() { + // The save threads and the restore reader threads differ, so the count of the pool tells + // which setting the restore took. + let storage = Arc::new(InMemoryBlobStorage::new()); + let policy = StorePolicy { + save_threads: NonZeroUsize::new(2), + restore_reader_threads: NonZeroUsize::new(3).unwrap(), + ..policy(LONG_DEADLINE, NEVER, Duration::ZERO) + }; + let store = RusticSnapshotStore { + low_priority: super::super::priority::LowPriority { + build_pool: recording_pool, + ..super::super::priority::LowPriority::new(policy.save_threads) + }, + ..RusticSnapshotStore::with_policy(storage, key(), policy) + }; + let scope = new_scope(); + let tree = one_file_tree("restored"); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + + let restored = restored_listing(&store, &scope, &name("p-1")).await; + let restore_pools = BUILT_POOLS + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .iter() + .filter(|(name, _)| *name == "fs-snap-restore") + .map(|(_, threads)| *threads) + .collect::>(); + + assert_eq!( + (restored.ok(), restore_pools), + (Some(listing(tree.path())), vec![NonZeroUsize::new(3)]) + ); +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs new file mode 100644 index 0000000000..9e0753de34 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests/sweep.rs @@ -0,0 +1,739 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Two deletes of one scope whose prunes are both due, in each order of their blob calls. +//! +//! Each delete has its own store over its own scripted storage, and the two storages share one +//! in-memory storage, as two executors share a bucket. Each blob call of the prune protocol is a +//! step: it waits until the test gives its delete one step. So the test sets the order of the +//! calls of the two deletes. The grace period is short, so a prune writes new markers of its claim +//! while it runs. A timer starts each such write, so its place in the order can change from run to +//! run, and the log of a case gives the call that took each step. + +use super::*; +use futures::TryStreamExt; +use tokio::task::JoinHandle; + +/// The operation labels of the blob calls of the prune protocol. The listing of the packs by a +/// prune is a step too, and it is the start of the prune. The delete of a snapshot file is the +/// forget of a delete. +const STEP_LABELS: &[&str] = &[ + "write_freed", + "read_freed", + "list_snapshots", + "read_ledger", + "list_freed", + "list_data", + "list_claims", + "write_marker", + "write_claim", + "delete_marker", + "refresh_claim", + "final_marker", + "write_ledger", + "list_ledgers", + "delete_ledger", + "delete_freed", + "delete_claim", + "list_claim_directories", + "list_claim_blobs", + "delete_claims", +]; + +/// The most steps that one delete takes before the test gives up on it, without the writes of +/// new markers. Those writes end with the prune. +const MOST_STEPS: usize = 64; + +fn is_step(op_label: &str, path: &Path) -> bool { + STEP_LABELS.contains(&op_label) + || (op_label == "list" && path == Path::new("data")) + || is_forget(op_label, path) +} + +/// Tells whether the call writes or deletes, so that it can reach the storage late. +fn can_be_late(op_label: &str) -> bool { + op_label.starts_with("write_") || op_label.starts_with("delete") || op_label == "final_marker" +} + +/// Tells whether the call writes a new marker of a claim while its prune runs. A timer starts +/// such a call, so the schedule neither counts it nor makes it fail. +fn is_refresh(op_label: &str) -> bool { + op_label == "refresh_claim" +} + +/// The grace period of the deletes of a case. The claim of a prune gets a new marker at each +/// quarter of it, so the markers come while the prune runs. +const SWEEP_GRACE: Duration = Duration::from_millis(16); + +fn is_prune_start(op_label: &str, path: &str) -> bool { + op_label == "list" && path == "data" +} + +/// The order of the steps of one case: the delete that goes first, and the numbers of steps that +/// the deletes take in turn. A turn counts each step, also the write of a new marker. After the +/// listed turns, the delete whose turn is next runs to its end, and then the other one does. +/// `fail` refuses one step of one delete. `late` makes one write or delete of one delete give no +/// answer at its step, and reach the storage after the given number of further steps of the case. +/// `drop` drops one delete right after one of its steps, as a caller that stops waiting does. The +/// steps of `fail`, `late` and `drop` are numbered without the writes of new markers. +#[derive(Clone, Debug)] +pub(super) struct Schedule { + pub(super) first: usize, + pub(super) turns: Vec, + pub(super) fail: Option<(usize, usize)>, + pub(super) late: Option<(usize, usize, usize)>, + pub(super) drop: Option<(usize, usize)>, +} + +/// One step that a delete took, or a late call that reached the storage, in the order of all +/// steps of the case. +#[derive(Clone, Debug)] +struct Step { + delete: usize, + op_label: &'static str, + path: String, + /// The call gave an error to its caller: it was refused, or it was late. + failed: bool, + /// The call reached the storage: a step that passed, or the landing of a late call. + effect: bool, + /// The entry is the landing of a late call, not a call. + landed: bool, +} + +/// One delete of the case: its store, its scripted storage and its task. +struct Delete { + store: Arc, + storage: Arc, + task: JoinHandle>, + /// The steps that the delete took, other than the writes of new markers. + taken: usize, + /// Whether the case dropped the delete. A dropped delete makes a call only for the release of + /// its claim. + dropped: bool, +} + +impl Delete { + /// Tells whether the delete ended and its store has no work left, such as the release of the + /// claim of a dropped delete. + fn finished(&self) -> bool { + self.task.is_finished() && self.store.work_in_flight() == 0 + } +} + +/// The state of a case while it runs. +struct Case { + deletes: [Delete; 2], + log: Vec, + schedule: Schedule, + /// A late call that has not reached the storage: its delete, the log length at which it + /// lands, and its step. + pending: Option<(usize, usize, Step)>, +} + +/// Waits until the condition holds, and gives false when it does not hold within [`LIMIT`]. The +/// wait yields to the runtime between two checks, so a step ends as soon as it can. +async fn until(condition: impl Fn() -> bool) -> bool { + tokio::time::timeout(LIMIT, async { + futures::stream::repeat(()) + .then(|()| tokio::task::yield_now()) + .take_while(|()| std::future::ready(!condition())) + .for_each(|()| std::future::ready(())) + .await + }) + .await + .is_ok() +} + +/// Waits until the delete waits for a step or ended. While the prune waits for the listing of +/// the packs, it also waits until the first new marker of the claim waits for a step, so each +/// prune writes a new marker right after that listing. A dropped delete writes no new marker, +/// because the cancel stops the refresh, and the cancel also ends its listing of the packs. +async fn settle(delete: &Delete) -> bool { + until(|| { + let waiting = delete.storage.waiting_steps(); + let refreshing = !delete.dropped && prune_waits(&delete.storage); + delete.finished() || waiting > usize::from(refreshing) + }) + .await +} + +/// Tells whether the prune of the delete waits for its step: the storage has a call for the +/// listing of the packs that took no step yet. +fn prune_waits(storage: &ScriptedBlobStorage) -> bool { + let starts = |calls: Vec<(&'static str, String)>| { + calls + .iter() + .filter(|(op_label, path)| is_prune_start(op_label, path)) + .count() + }; + starts(storage.calls()) > starts(storage.took()) +} + +impl Case { + /// Lands the pending late call when its time came, or when `now` is true. + async fn land(&mut self, now: bool) -> Result<(), String> { + let due = self + .pending + .as_ref() + .is_some_and(|(_, at, _)| now || self.log.len() >= *at); + if !due { + return Ok(()); + } + let Some((who, _, step)) = self.pending.take() else { + return Ok(()); + }; + let storage = self.deletes[who].storage.clone(); + let before = storage.landed(); + storage.land_late(); + if !until(|| storage.landed() > before).await { + return Err(format!( + "the late call {} {} did not land", + step.op_label, step.path + )); + } + self.log.push(Step { + failed: false, + effect: true, + landed: true, + ..step + }); + Ok(()) + } + + /// Gives the delete one step and waits until the step ended and the delete waits again or + /// ended. + async fn take_step(&mut self, who: usize) -> Result<(), String> { + if self.deletes[who].finished() { + return Ok(()); + } + if self.deletes[who].dropped { + let delete = &self.deletes[who]; + if !until(|| delete.storage.waiting_steps() > 0 || delete.finished()).await { + return Err(format!("the dropped delete {who} did not end")); + } + if delete.finished() { + return Ok(()); + } + } + let storage = self.deletes[who].storage.clone(); + let (before, taker) = (storage.stepped(), storage.took().len()); + storage.step(); + // The drop cancels the operation of the delete, and a call that already waited for a step + // then ends without the step. So for a dropped delete, the step can stay with no call to + // take it, and the test takes it back. + let dropped = self.deletes[who].dropped; + let untaken = || { + dropped + && storage.stepped() == before + && storage.waiting_steps() == 0 + && storage.took().len() == taker + }; + if !until(|| storage.stepped() > before || untaken()).await { + return Err(format!( + "a step of the dropped delete {who} was neither taken nor left; its last calls {:?}", + storage.calls().iter().rev().take(6).collect::>() + )); + } + if untaken() && storage.take_back_step() { + return Ok(()); + } + if !until(|| storage.stepped() > before).await { + return Err(format!( + "a step of delete {who} did not end; its last calls {:?}", + storage.calls().iter().rev().take(6).collect::>() + )); + } + let took = storage.took(); + if took.len() != taker + 1 { + return Err(format!( + "one step of delete {who} was taken by {:?}", + took.get(taker..) + )); + } + let (op_label, path) = took + .get(taker) + .cloned() + .ok_or_else(|| format!("a step of delete {who} ended, and no call took it"))?; + let number = self.deletes[who].taken; + let counted = !is_refresh(op_label); + let refused = counted && self.schedule.fail == Some((who, number)); + let late = self.schedule.late.filter(|(late_who, late_step, _)| { + counted && (*late_who, *late_step) == (who, number) && can_be_late(op_label) + }); + let step = Step { + delete: who, + op_label, + path, + failed: refused || late.is_some(), + effect: !refused && late.is_none(), + landed: false, + }; + self.log.push(step.clone()); + if let Some((_, _, delay)) = late { + self.pending = Some((who, self.log.len() + delay, step)); + } + self.deletes[who].taken += usize::from(counted); + if self.deletes[who].taken > MOST_STEPS { + return Err(format!("delete {who} took more than {MOST_STEPS} steps")); + } + self.land(false).await?; + if counted && self.schedule.drop == Some((who, number)) { + self.deletes[who].task.abort(); + self.deletes[who].dropped = true; + // An abort only asks the task to end. Its waiting call counts as waiting until the + // task ends, so the case gives no step before that. + let task = &self.deletes[who].task; + if !until(|| task.is_finished()).await { + return Err(format!("the dropped delete {who} did not end")); + } + // The drop counts as a failed step of the delete, so the rules on a delete that + // stopped after its claim apply to it. + self.log.push(Step { + delete: who, + op_label: "drop", + path: String::new(), + failed: true, + effect: false, + landed: false, + }); + } + if settle(&self.deletes[who]).await { + Ok(()) + } else { + let calls = self.deletes[who].storage.calls(); + let last = calls.iter().rev().take(6).collect::>(); + Err(format!( + "delete {who} did not reach its next step; its last calls {last:?}; the other delete waits {}", + self.deletes[(who + 1) % 2].storage.waiting_steps() + )) + } + } + + /// Gives the delete the steps. + async fn take_steps(&mut self, who: usize, steps: usize) -> Result<(), String> { + futures::stream::iter(0..steps) + .map(Ok) + .try_fold(self, |case, _| async move { + case.take_step(who).await?; + Ok::<_, String>(case) + }) + .await + .map(|_| ()) + } + + /// Gives the delete steps until it ends. + async fn run_to_end(&mut self, who: usize) -> Result<(), String> { + futures::stream::unfold(self, |case| async move { + if case.deletes[who].finished() { + None + } else { + let stepped = case.take_step(who).await; + Some((stepped, case)) + } + }) + .try_for_each(|()| std::future::ready(Ok(()))) + .await + } +} + +/// What a case found. +#[derive(Debug)] +pub(super) struct Found { + pub(super) prunes: usize, + pub(super) failed: bool, + pub(super) steps: [usize; 2], +} + +/// Runs the two deletes of a copy of the prepared scope in the order of the schedule, and checks +/// the rules of the prune protocol. +pub(super) async fn run_case( + shared: &Arc, + prepared: &SnapshotScope, + schedule: &Schedule, +) -> Result { + let scope = new_scope(); + store(shared.clone(), policy(LONG_DEADLINE, NEVER, Duration::ZERO)) + .copy_scope(prepared, &scope) + .await + .map_err(|error| format!("the copy of the prepared scope failed: {error}"))?; + let deletes = [0, 1].map(|who| { + let counter = Arc::new(AtomicUsize::new(0)); + let (fail, late) = (schedule.fail, schedule.late); + let storage = ScriptedBlobStorage::new(shared.clone(), move |op_label, path| { + if is_refresh(op_label) { + Script::Step { + refuse: false, + late: false, + } + } else if is_step(op_label, path) { + let number = counter.fetch_add(1, Ordering::SeqCst); + Script::Step { + refuse: fail == Some((who, number)), + late: can_be_late(op_label) + && late.is_some_and(|(late_who, late_step, _)| { + (late_who, late_step) == (who, number) + }), + } + } else { + Script::Pass + } + }); + let deleting = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, SWEEP_GRACE)); + let scope = scope.clone(); + let task = tokio::spawn({ + let deleting = deleting.clone(); + async move { deleting.delete(&scope, &name(["p-1", "p-2"][who])).await } + }); + Delete { + store: deleting, + storage, + task, + taken: 0, + dropped: false, + } + }); + let mut case = Case { + deletes, + log: Vec::new(), + schedule: schedule.clone(), + pending: None, + }; + if !(settle(&case.deletes[0]).await && settle(&case.deletes[1]).await) { + return Err("a delete did not reach its first step".to_string()); + } + let turns = schedule.turns.clone(); + let ran = run_turns(&mut case, schedule, &turns).await; + if let Err(error) = ran { + return Err(format!( + "{error}; schedule {schedule:?}; steps {}", + order(&case.log) + )); + } + let results = + futures::future::join_all(case.deletes.iter_mut().map(|delete| &mut delete.task)).await; + check(shared, &scope, schedule, &case.log, results).await +} + +/// Runs the turns of the schedule, then each delete to its end, and then lands a late call that +/// did not land yet. +async fn run_turns(case: &mut Case, schedule: &Schedule, turns: &[usize]) -> Result<(), String> { + let next = futures::stream::iter(turns.iter().enumerate()) + .map(Ok) + .try_fold(&mut *case, |case, (turn, steps)| async move { + case.take_steps((schedule.first + turn) % 2, *steps).await?; + Ok::<_, String>(case) + }) + .await + .map(|_| (schedule.first + turns.len()) % 2)?; + case.run_to_end(next).await?; + case.run_to_end((next + 1) % 2).await?; + case.land(true).await +} + +/// Gives the steps of the log as text: the delete and the operation label, with `!` for a call +/// that gave an error. +fn order(log: &[Step]) -> String { + log.iter() + .map(|step| { + let mark = match (step.failed, step.effect) { + (true, _) => "!", + (false, true) => "", + (false, false) => "?", + }; + format!("{}:{}{}", step.delete, step.op_label, mark) + }) + .collect::>() + .join(" ") +} + +/// Gives the index in the log at which the delete forgot its snapshot, when its forget reached +/// the storage. +fn forgotten_at(log: &[Step], who: usize) -> Option { + log.iter().position(|step| { + step.delete == who && step.effect && is_forget(step.op_label, Path::new(&step.path)) + }) +} + +/// Gives each claim, or marker of a claim, that a delete wrote and that stays, when the first +/// failed call of that delete, or its drop, came after it took its claim and before its prune +/// started: before it called for the listing of the packs, and when it made no final marker call. +/// A claim that the other delete wrote later at the same path is not the claim of the delete. A +/// blob whose own delete failed is left out, because no call can remove it then, and a claim or a +/// marker that stays only delays a prune. +fn kept_claims(log: &[Step], claims: &[String]) -> Vec { + [0, 1] + .into_iter() + .filter_map(|who| { + let own = |step: &Step| step.delete == who; + let claimed_at = log.iter().position(|step| { + own(step) && step.op_label == "write_claim" && step.effect && !step.failed + })?; + let failed_at = log.iter().position(|step| own(step) && step.failed)?; + let pruning = log[..=failed_at] + .iter() + .any(|step| own(step) && is_prune_start(step.op_label, &step.path)); + (claimed_at < failed_at && !pruning && !marked_final(log, who)) + .then_some((who, claimed_at)) + }) + .flat_map(|(who, claimed_at)| { + let claim = &log[claimed_at].path; + let taken_again = log[claimed_at..].iter().any(|step| { + step.delete != who + && step.op_label == "write_claim" + && step.effect + && step.path == *claim + }); + let written = log + .iter() + .filter(|step| { + step.delete == who + && step.effect + && matches!( + step.op_label, + "write_marker" | "refresh_claim" | "final_marker" + ) + }) + .map(|step| step.path.clone()) + .chain((!taken_again).then(|| claim.clone())) + .collect::>(); + let refused = log + .iter() + .filter(|step| { + step.delete == who + && step.failed + && matches!(step.op_label, "delete_claim" | "delete_marker") + }) + .map(|step| step.path.clone()) + .collect::>(); + claims + .iter() + .filter(move |path| written.contains(path) && !refused.contains(path)) + .cloned() + .collect::>() + }) + .collect() +} + +/// Tells whether the delete called for the final marker of its claim. The guard writes it only +/// after the prune started. A drop can come right after the start, when the cancel ends the +/// listing of the packs before it takes a step, so the final marker is then the only step that +/// shows the start. +fn marked_final(log: &[Step], who: usize) -> bool { + log.iter() + .any(|step| step.delete == who && !step.landed && step.op_label == "final_marker") +} + +/// Gives the claim of each delete whose prune started and that wrote no ledger entry, when that +/// claim is gone. A prune that started keeps its claim, so the next prune waits for the hold. +fn started_claims_gone(log: &[Step], claims: &[String]) -> Vec { + [0, 1] + .into_iter() + .filter_map(|who| { + let own = |step: &&Step| step.delete == who; + let claim = log + .iter() + .filter(own) + .find(|step| step.op_label == "write_claim" && step.effect && !step.failed)?; + let started = log + .iter() + .filter(own) + .any(|step| is_prune_start(step.op_label, &step.path)) + || marked_final(log, who); + let ledger_written = log + .iter() + .filter(own) + .any(|step| step.op_label == "write_ledger" && step.effect); + (started && !ledger_written && !claims.contains(&claim.path)) + .then(|| claim.path.clone()) + }) + .collect() +} + +/// Gives each delete whose prune started and that made no call for its final marker. The prune +/// started when the delete called for the listing of the packs. A delete that called to delete its +/// claim released it, because each attempt of its prune found a snapshot file gone, so it writes +/// no final marker. +fn final_markers_missing(log: &[Step]) -> Vec { + [0, 1] + .into_iter() + .filter(|who| { + let own = log + .iter() + .filter(|step| step.delete == *who && !step.landed); + let (mut started, mut released, mut marked) = (false, false, false); + own.for_each(|step| { + started |= is_prune_start(step.op_label, &step.path); + released |= step.op_label == "delete_claim"; + marked |= step.op_label == "final_marker"; + }); + started && !released && !marked + }) + .collect() +} + +/// Gives each delete that called for a final marker after a final marker call of it that gave no +/// error. +fn final_markers_repeated(log: &[Step]) -> Vec { + [0, 1] + .into_iter() + .filter(|who| { + log.iter() + .filter(|step| { + step.delete == *who && !step.landed && step.op_label == "final_marker" + }) + .skip_while(|step| step.failed) + .nth(1) + .is_some() + }) + .collect() +} + +/// Checks the rules on the end state of a case. +async fn check( + shared: &Arc, + scope: &SnapshotScope, + schedule: &Schedule, + log: &[Step], + results: Vec, tokio::task::JoinError>>, +) -> Result { + let failed = log.iter().any(|step| step.failed); + let prune_starts = log + .iter() + .enumerate() + .filter(|(_, step)| step.effect && is_prune_start(step.op_label, &step.path)) + .collect::>(); + let prunes = prune_starts.len(); + let fail = |rule: &str| { + Err(format!( + "{rule}; schedule {schedule:?}; steps {}", + order(log) + )) + }; + if prunes > 1 { + return fail("more than one prune ran"); + } + if !failed && prunes != 1 { + return fail(&format!( + "no prune ran, although a prune was due and no call failed: {results:?}" + )); + } + if !failed && results.iter().any(|result| !matches!(result, Ok(Ok(())))) { + return fail(&format!("a delete failed with no failed call: {results:?}")); + } + let claims = blobs(&**shared, &scope.0, "golem/prune-claims/").await; + if !failed && !claims.is_empty() { + return fail(&format!("claims stay: {claims:?}")); + } + let released = kept_claims(log, &claims); + if !released.is_empty() { + return fail(&format!( + "a delete that failed after its claim and before its prune kept its claim: {released:?}" + )); + } + let dropped = started_claims_gone(log, &claims); + if !dropped.is_empty() { + return fail(&format!( + "a delete whose prune started and wrote no ledger lost its claim: {dropped:?}" + )); + } + let unmarked = final_markers_missing(log); + if !unmarked.is_empty() { + return fail(&format!( + "a delete whose prune started and kept its claim made no final marker call: {unmarked:?}" + )); + } + let repeated = final_markers_repeated(log); + if !repeated.is_empty() { + return fail(&format!( + "a delete made a final marker call after one that succeeded: {repeated:?}" + )); + } + let records = blobs(&**shared, &scope.0, "golem/prune-freed/").await; + let entries = blobs(&**shared, &scope.0, "golem/prune-ledgers/").await; + let ledger = ledger(shared, scope).await; + let written = |path: &str| { + log.iter() + .find(|step| step.op_label == "write_freed" && step.effect && step.path == path) + .map(|step| step.delete) + }; + let lost = log + .iter() + .enumerate() + .filter(|(_, step)| step.op_label == "delete_freed" && step.effect) + .filter_map(|(deleted_at, step)| { + let pruner = step.delete; + let decided_at = log[..deleted_at].iter().rposition(|earlier| { + earlier.delete == pruner + && matches!( + earlier.op_label, + "list_freed" | "read_freed" | "list_snapshots" + ) + })?; + let writer = written(&step.path)?; + let settled = forgotten_at(log, writer).is_some_and(|forgot| forgot < decided_at); + (!settled).then(|| step.path.clone()) + }) + .collect::>(); + if !lost.is_empty() { + return fail(&format!( + "a prune deleted a record whose snapshot was not gone at its count: {lost:?}" + )); + } + if let Some((start, pruner)) = prune_starts + .first() + .map(|(index, step)| (*index, step.delete)) + { + let wrote_ledger = log[start..] + .iter() + .find(|step| step.delete == pruner && step.op_label == "write_ledger" && step.effect); + if let Some(written) = wrote_ledger { + let written_ms = Path::new(&written.path) + .file_name() + .and_then(|name| name.to_str()) + .and_then(|name| name.split('-').next()) + .and_then(|ms| ms.parse::().ok()); + if ledger.last_prune.map(|last| last.to_millis()) != written_ms { + return fail(&format!( + "the ledger {ledger:?} is not the entry of the prune {written_ms:?}; entries {entries:?}" + )); + } + } + let counted_at = log[..start].iter().rposition(|step| { + step.delete == pruner + && matches!( + step.op_label, + "list_freed" | "read_freed" | "list_snapshots" + ) + }); + let gone = log + .iter() + .enumerate() + .filter(|(_, step)| step.op_label == "write_freed" && step.effect) + .filter(|(index, _)| counted_at.is_none_or(|counted| *index > counted)) + .filter(|(_, step)| !records.contains(&step.path)) + .map(|(_, step)| step.path.clone()) + .collect::>(); + if !gone.is_empty() { + return fail(&format!( + "records that the prune did not count are gone: {gone:?}" + )); + } + } + let steps = [0, 1].map(|who| log.iter().filter(|step| step.delete == who).count()); + Ok(Found { + prunes, + failed, + steps, + }) +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/holding.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/holding.rs similarity index 98% rename from golem-worker-executor/src/filesystem_snapshot/rustic/holding.rs rename to golem-worker-executor/src/filesystem_snapshot/rustic/tests/holding.rs index b66549112f..daf660212a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/holding.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/holding.rs @@ -40,7 +40,7 @@ type Rule = Box bool + Send + Sync>; /// /// A held call waits until its gate opens, and then goes to the in-memory storage. Only the /// [`Gate`] of the storage opens the gate, so a held call waits until the test drops that value. -pub(super) struct HoldingBlobStorage { +pub(crate) struct HoldingBlobStorage { inner: Arc, rule: Rule, gate: CancellationToken, @@ -51,7 +51,7 @@ pub(super) struct HoldingBlobStorage { /// The gate of the held calls of a [`HoldingBlobStorage`]. /// /// The gate opens when the test drops this value, also when the test fails. -pub(super) struct Gate { +pub(crate) struct Gate { _open_on_drop: DropGuard, } @@ -60,7 +60,7 @@ pub(super) struct Gate { /// /// So the receiver resolves only when no value holds a copy of the storage, for example a backend /// on a thread of rustic. -pub(super) fn holding_storage( +pub(crate) fn holding_storage( inner: Arc, rule: impl Fn(&str, &Path) -> bool + Send + Sync + 'static, ) -> (Arc, Gate, oneshot::Receiver<()>) { @@ -82,7 +82,7 @@ pub(super) fn holding_storage( /// Tells whether the error or an error in its chain of sources is tokio's `Elapsed`, which is the /// root cause of the error of a call that got no answer within its deadline. -pub(super) fn reached_deadline(error: &(dyn std::error::Error + 'static)) -> bool { +pub(crate) fn reached_deadline(error: &(dyn std::error::Error + 'static)) -> bool { std::iter::successors(Some(error), |error| error.source()).any(|error| error.is::()) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/mod.rs similarity index 89% rename from golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs rename to golem-worker-executor/src/filesystem_snapshot/rustic/tests/mod.rs index 4ed61a7a94..33239e3407 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/mod.rs @@ -18,19 +18,23 @@ //! keeps the gate of the held calls closed until the storage is dropped. So the threads of rustic //! stop because of the deadline, and not because the gate opens. +pub(super) mod holding; +pub(super) mod scripted; + +use self::holding::{holding_storage, reached_deadline}; +use self::scripted::{Script, ScriptedBlobStorage}; use super::backend::BlobBackend; -use super::holding::{holding_storage, reached_deadline}; use super::{ ChangeDetection, Chunking, Compression, OperationPhase, PruneSettings, RepackLimits, - Repository, RepositoryKey, RepositorySettings, STORAGE_CALL_DEADLINE, SaveSettings, - backup_options, config_options, open_existing, prune_options, repository_options, run_blocking, - unopened, + Repository, RepositoryKey, RepositorySettings, SaveSettings, backup_options, config_options, + open_existing, prune_options, repository_options, run_blocking, unopened, }; use crate::filesystem_snapshot::contract_tests::fixture::{ Scratch, Spec, fixture, listing, write_tree, }; use crate::filesystem_snapshot::contract_tests::new_scope; use crate::filesystem_snapshot::{SnapshotName, SnapshotScope}; +use crate::services::golem_config::DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE as STORAGE_CALL_DEADLINE; use anyhow::Context; use async_trait::async_trait; use bytes::Bytes; @@ -48,7 +52,7 @@ use std::path::{Path, PathBuf}; use std::sync::Arc; use std::sync::atomic::{AtomicUsize, Ordering}; use std::time::Duration; -use test_r::test; +use test_r::{test, timeout}; use tokio::runtime::Handle; use tokio::sync::{Notify, oneshot, watch}; use tokio::time::error::Elapsed; @@ -140,6 +144,52 @@ async fn data_packs(storage: &Arc, scope: &SnapshotScope) - .unwrap() } +/// Gives the id in hex of each pack of tree blobs in the repository of the scope. +async fn tree_packs(storage: &Arc, scope: &SnapshotScope) -> Box<[Box]> { + with_existing_repository( + storage.clone(), + scope, + STORAGE_CALL_DEADLINE, + |repository| { + let indexes = repository + .stream_files::()? + .collect::>>()?; + Ok(indexes + .into_iter() + .flat_map(|(_, index)| index.packs) + .filter(|pack| pack.blob_type() == BlobType::Tree) + .map(|pack| Box::from(pack.id.to_hex().as_str())) + .collect()) + }, + ) + .await + .unwrap() +} + +/// Writes a tree of `count` directories, each with one small file, into a new directory. +fn many_directories_tree(count: usize) -> Scratch { + let tree = Scratch::new(); + let names = (0..count) + .flat_map(|index| [format!("dir-{index}"), format!("dir-{index}/file.txt")]) + .collect::>(); + let entries = names + .iter() + .map(|name| { + let spec = if name.ends_with(".txt") { + Spec::File { + content: Box::from(name.as_bytes()), + mode: 0o644, + } + } else { + Spec::Directory { mode: 0o755 } + }; + (name.as_str(), spec) + }) + .collect::>(); + write_tree(tree.path(), &entries); + tree +} + /// Prunes the repository of the scope with the options, on a blocking thread. Each call on the /// storage waits for at most `deadline`. async fn prune( @@ -191,6 +241,7 @@ async fn dropped_within_limit(dropped: oneshot::Receiver<()>) -> bool { } #[test] +#[timeout("60s")] async fn a_saved_tree_comes_back_the_same() { let storage = Arc::new(InMemoryBlobStorage::new()); let repository = repository(&storage, &new_scope()); @@ -210,6 +261,79 @@ async fn a_saved_tree_comes_back_the_same() { } #[test] +#[timeout("60s")] +async fn a_restore_report_gives_each_phase_of_the_restore_in_order() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let repository = repository(&storage, &new_scope()); + let tree = fixture_tree(); + let into = Scratch::new(); + repository.save(&name("first"), tree.path()).await.unwrap(); + + let restored = repository + .restore(&name("first"), into.path(), None) + .await + .unwrap() + .unwrap(); + + assert_eq!( + restored + .phases + .iter() + .map(|phase| phase.phase) + .collect::>(), + vec![ + OperationPhase::Open, + OperationPhase::Lookup, + OperationPhase::IndexLoad, + OperationPhase::RestorePlan, + OperationPhase::Restore, + ] + ); +} + +#[test] +#[timeout("60s")] +async fn a_restore_reads_each_tree_pack_one_time_in_full_and_no_range_of_a_tree_pack() { + let inner = Arc::new(InMemoryBlobStorage::new()); + let scope = new_scope(); + let tree = many_directories_tree(60); + repository(&inner, &scope) + .save(&name("first"), tree.path()) + .await + .unwrap(); + let tree_packs = tree_packs(&inner, &scope).await; + let storage = ScriptedBlobStorage::new(inner.clone(), |_, _| Script::Pass); + let into = Scratch::new(); + + Repository::new(storage.clone(), scope.clone(), key(), STORAGE_CALL_DEADLINE) + .restore(&name("first"), into.path(), None) + .await + .unwrap(); + let calls_on_tree_packs = |op: &str| { + tree_packs + .iter() + .map(|pack| { + storage + .calls() + .iter() + .filter(|(op_label, path)| *op_label == op && path.ends_with(&**pack)) + .count() + }) + .collect::>() + }; + + assert_eq!( + ( + calls_on_tree_packs("read"), + calls_on_tree_packs("read_range").iter().sum::(), + listing(into.path()) + ), + (vec![1; tree_packs.len()], 0, listing(tree.path())) + ); +} + +#[test] +#[timeout("60s")] async fn a_second_save_has_the_first_as_parent_and_reads_only_the_changed_file() { let storage = Arc::new(InMemoryBlobStorage::new()); let repository = repository(&storage, &new_scope()); @@ -251,6 +375,7 @@ async fn a_second_save_has_the_first_as_parent_and_reads_only_the_changed_file() } #[test] +#[timeout("60s")] async fn a_forgotten_name_does_not_restore_and_the_other_names_do() { let storage = Arc::new(InMemoryBlobStorage::new()); let repository = repository(&storage, &new_scope()); @@ -284,6 +409,7 @@ async fn a_forgotten_name_does_not_restore_and_the_other_names_do() { } #[test] +#[timeout("60s")] async fn the_first_save_creates_the_repository_with_no_key_file_and_later_saves_open_it() { let storage = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -329,6 +455,7 @@ async fn the_first_save_creates_the_repository_with_no_key_file_and_later_saves_ } #[test] +#[timeout("60s")] async fn each_scope_is_its_own_repository() { let storage = Arc::new(InMemoryBlobStorage::new()); let (one, other) = (new_scope(), new_scope()); @@ -360,6 +487,7 @@ async fn each_scope_is_its_own_repository() { } #[test] +#[timeout("60s")] async fn a_restore_reads_data_on_at_most_its_reader_threads() { // Each save adds one pack with the data of its new file. The restore of the second snapshot // reads the data of both packs, one read for each pack. @@ -632,6 +760,7 @@ impl BlobStorage for OverlapCountingStorage { } #[test] +#[timeout("60s")] async fn a_save_whose_pack_write_gets_no_answer_fails_with_no_snapshot_and_its_threads_stop() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -657,6 +786,7 @@ async fn a_save_whose_pack_write_gets_no_answer_fails_with_no_snapshot_and_its_t } #[test] +#[timeout("60s")] async fn a_save_whose_pack_writes_answer_before_the_deadline_succeeds_and_restores() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -698,6 +828,7 @@ async fn a_save_whose_pack_writes_answer_before_the_deadline_succeeds_and_restor } #[test] +#[timeout("60s")] async fn a_restore_whose_data_pack_reads_get_no_answer_fails_and_stops_its_threads() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -755,7 +886,8 @@ async fn a_restore_whose_data_pack_reads_get_no_answer_fails_and_stops_its_threa } #[test] -async fn a_prune_whose_pack_reads_get_no_answer_fails_and_stops_its_threads() { +#[timeout("60s")] +async fn a_prune_whose_tree_pack_reads_get_no_answer_fails_and_stops_its_threads() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); let tree = fixture_tree(); @@ -763,8 +895,9 @@ async fn a_prune_whose_pack_reads_get_no_answer_fails_and_stops_its_threads() { .save(&name("first"), tree.path()) .await .unwrap(); + // A prune reads the trees of the snapshots, and each tree read is a full read of a pack. let (storage, _gate, dropped) = holding_storage(inner, |op_label, path| { - op_label == "read_range" && path.starts_with("data") + op_label == "read" && path.starts_with("data") }); let pruned = tokio::time::timeout( @@ -778,6 +911,7 @@ async fn a_prune_whose_pack_reads_get_no_answer_fails_and_stops_its_threads() { } #[test] +#[timeout("60s")] async fn a_prune_after_a_forget_deletes_the_packs_of_that_name_and_the_other_name_restores() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -849,7 +983,7 @@ fn set_modified(path: &Path, time: std::time::SystemTime) { /// Copies each file of the flat tree `from` into the new directory `to`, with its modification /// time. Each copy is a new inode with a new change time, as a capture gives. -fn copy_flat_tree(from: &Path, to: &Path) { +pub(super) fn copy_flat_tree(from: &Path, to: &Path) { std::fs::read_dir(from).unwrap().for_each(|entry| { let entry = entry.unwrap(); let target = to.join(entry.file_name()); @@ -874,7 +1008,7 @@ const CHANGE_TIME_WAIT: Duration = Duration::from_secs(10); /// changes within one tick get the same change time. The wait changes a probe file in its own /// directory until the change time of the probe is later than the latest change time of the /// files. It fails the test when that does not happen within [`CHANGE_TIME_WAIT`]. -fn wait_past_change_times(files: &[PathBuf]) { +pub(super) fn wait_past_change_times(files: &[PathBuf]) { let latest = files.iter().map(|file| changed_at(file)).max().unwrap(); let probe_directory = Scratch::new(); let probe = probe_directory.path().join("probe"); @@ -892,7 +1026,7 @@ fn wait_past_change_times(files: &[PathBuf]) { } /// Gives the path of each entry of the directory, in the order of the names. -fn entries(directory: &Path) -> Vec { +pub(super) fn entries(directory: &Path) -> Vec { let mut paths = std::fs::read_dir(directory) .unwrap() .map(|entry| entry.unwrap().path()) @@ -902,7 +1036,7 @@ fn entries(directory: &Path) -> Vec { } /// Writes a tree of three files into a new directory, and gives the directory. -fn three_file_tree() -> Scratch { +pub(super) fn three_file_tree() -> Scratch { let tree = Scratch::new(); ["a.txt", "b.txt", "c.txt"] .iter() @@ -1138,6 +1272,7 @@ async fn inspect_gives_the_snapshots_the_name_and_the_phases_and_nothing_without } #[test] +#[timeout("60s")] async fn a_size_and_mtime_save_of_a_copied_tree_reads_no_file_and_a_ctime_save_reads_each() { // A copy gives each file a new inode and a new change time, and keeps its size and its // modification time. The size-and-mtime form must also not compare inodes: in rustic, @@ -1185,6 +1320,7 @@ async fn a_size_and_mtime_save_of_a_copied_tree_reads_no_file_and_a_ctime_save_r } #[test] +#[timeout("60s")] async fn a_size_and_mtime_save_misses_a_rewrite_of_the_same_size_with_the_old_mtime() { let storage = Arc::new(InMemoryBlobStorage::new()); let size_mtime = SaveSettings { @@ -1237,6 +1373,7 @@ async fn a_size_and_mtime_save_misses_a_rewrite_of_the_same_size_with_the_old_mt } #[test] +#[timeout("60s")] async fn two_prunes_without_a_grace_period_give_back_the_data_that_no_snapshot_uses() { // The first save puts both files in one pack. The second save rewrites one of them. After the // forget, that pack holds a used and an unused blob, so a prune without limits repacks it. @@ -1317,6 +1454,7 @@ async fn two_prunes_without_a_grace_period_give_back_the_data_that_no_snapshot_u } #[test] +#[timeout("60s")] async fn a_prune_of_a_scope_without_a_repository_gives_nothing() { let storage = Arc::new(InMemoryBlobStorage::new()); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/scripted.rs new file mode 100644 index 0000000000..b62e88d2ee --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/scripted.rs @@ -0,0 +1,585 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! A blob storage for tests that records each call and follows a script for each call. +//! +//! This module is test code, and it compiles only for tests. + +use async_trait::async_trait; +use bytes::Bytes; +use futures::stream::BoxStream; +use golem_service_base::replayable_stream::ErasedReplayableStream; +use golem_service_base::storage::blob::memory::InMemoryBlobStorage; +use golem_service_base::storage::blob::{ + BlobMetadata, BlobStorage, BlobStorageNamespace, ExistsResult, ListedBlob, PutIfAbsent, +}; +use std::fmt::{Debug, Formatter}; +use std::future::Future; +use std::path::{Path, PathBuf}; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::{Arc, Mutex, PoisonError}; +use tokio::sync::Semaphore; +use tokio_util::sync::CancellationToken; + +/// What the storage does with one call. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) enum Script { + /// Passes the call to the in-memory storage. + Pass, + /// Gives an error and does not pass the call. + Refuse, + /// Passes the call, and then gives an error in place of its answer. + LoseTheAnswer, + /// Passes the call, and then never answers. + NeverAnswer, + /// Waits until the test opens the gate of the storage, and then passes the call. + WaitForGate, + /// Waits for the time, and then passes the call. + Delay(std::time::Duration), + /// Waits for the time, and then gives an error and does not pass the call. + RefuseAfter(std::time::Duration), + /// Gives no blob to a read of a whole blob, as a delete after a listing does. Each other call + /// passes. + Vanish, + /// Passes the call, and then answers a write if absent with `AlreadyExists`, as a new try of a + /// call whose first answer was lost does. Each other call passes. + AnswerAlreadyExists, + /// Waits until the test gives the storage one step, and then passes the call, or refuses it + /// when `refuse` is true. A `late` write or delete gives an error at its step, as a call that + /// got no answer within its deadline, and it reaches the storage when the test lands it. + Step { refuse: bool, late: bool }, +} + +/// A rule that gives the script of a call from its operation label and its path. +type Rule = Box Script + Send + Sync>; + +/// A blob storage that records the operation label and the path of each call, and does with each +/// call what its rule gives. +pub(crate) struct ScriptedBlobStorage { + inner: Arc, + rule: Rule, + calls: Mutex)>>, + gate: CancellationToken, + /// The steps that the test gave and that no call took yet. + steps: Semaphore, + /// The calls that wait for a step. + waiting: AtomicUsize, + /// The calls that took a step and ended, or that dropped after they took a step. + stepped: AtomicUsize, + /// The operation label and the path of each call that took a step, in the order of the steps. + took: Mutex)>>, + /// The landings that the test gave and that no late call took yet. + landings: Arc, + /// The late calls that reached the storage. + landed: Arc, +} + +impl ScriptedBlobStorage { + pub(crate) fn new( + inner: Arc, + rule: impl Fn(&str, &Path) -> Script + Send + Sync + 'static, + ) -> Arc { + Arc::new(Self { + inner, + rule: Box::new(rule), + calls: Mutex::new(Vec::new()), + gate: CancellationToken::new(), + steps: Semaphore::new(0), + waiting: AtomicUsize::new(0), + stepped: AtomicUsize::new(0), + took: Mutex::new(Vec::new()), + landings: Arc::new(Semaphore::new(0)), + landed: Arc::new(AtomicUsize::new(0)), + }) + } + + /// Lets each call that waits for the gate, and each later such call, go on. + pub(crate) fn open_gate(&self) { + self.gate.cancel(); + } + + /// Lets one call that waits for a step, now or later, go on. + pub(crate) fn step(&self) { + self.steps.add_permits(1); + } + + /// Takes back one step that no call took, and tells whether one was there. + pub(crate) fn take_back_step(&self) -> bool { + self.steps + .try_acquire() + .map(tokio::sync::SemaphorePermit::forget) + .is_ok() + } + + /// Gives the number of calls that wait for a step. + pub(crate) fn waiting_steps(&self) -> usize { + self.waiting.load(Ordering::SeqCst) + } + + /// Gives the number of calls that took a step and ended. + pub(crate) fn stepped(&self) -> usize { + self.stepped.load(Ordering::SeqCst) + } + + /// Gives the operation label and the path of each call that took a step, in the order of the + /// steps. Two calls can wait for a step at one time, and the one that waited first takes it. + pub(crate) fn took(&self) -> Vec<(&'static str, String)> { + self.took + .lock() + .unwrap_or_else(PoisonError::into_inner) + .iter() + .map(|(op_label, path)| (*op_label, path.display().to_string())) + .collect() + } + + /// Waits until the test gives the call a step. The call counts as waiting until it takes the + /// step or drops, and it counts as stepped when the returned guard drops. + async fn wait_for_step(&self, op_label: &'static str, path: &Path) -> Stepped<'_> { + let waiting = Waiting::new(&self.waiting); + if let Ok(permit) = self.steps.acquire().await { + permit.forget(); + } + // The call is in the steps that were taken before it stops waiting, so a test never sees + // a call that neither waits nor took its step. + self.took + .lock() + .unwrap_or_else(PoisonError::into_inner) + .push((op_label, path.into())); + drop(waiting); + Stepped(&self.stepped) + } + + /// Lets one late call, now or later, reach the storage. + pub(crate) fn land_late(&self) { + self.landings.add_permits(1); + } + + /// Gives the number of late calls that reached the storage. + pub(crate) fn landed(&self) -> usize { + self.landed.load(Ordering::SeqCst) + } + + /// Takes one step for a late call: the caller gets an error, and the call reaches the storage + /// in a task when the test lands it. + async fn late( + &self, + op_label: &'static str, + path: &Path, + landing: impl Future> + Send + 'static, + ) -> anyhow::Result { + self.record(op_label, path); + let stepped = self.wait_for_step(op_label, path).await; + let (landings, landed) = (self.landings.clone(), self.landed.clone()); + tokio::spawn(async move { + if let Ok(permit) = landings.acquire().await { + permit.forget(); + } + let _ = landing.await; + landed.fetch_add(1, Ordering::SeqCst); + }); + drop(stepped); + Err(anyhow::anyhow!( + "the call got no answer within its deadline" + )) + } + + /// Gives the operation label and the path of each call, in the order of the calls. + pub(crate) fn calls(&self) -> Vec<(&'static str, String)> { + self.calls + .lock() + .unwrap_or_else(PoisonError::into_inner) + .iter() + .map(|(op_label, path)| (*op_label, path.display().to_string())) + .collect() + } + + fn record(&self, op_label: &'static str, path: &Path) { + self.calls + .lock() + .unwrap_or_else(PoisonError::into_inner) + .push((op_label, path.into())); + } + + async fn answer( + &self, + op_label: &'static str, + path: &Path, + call: impl Future>, + ) -> anyhow::Result { + self.follow((self.rule)(op_label, path), op_label, path, call) + .await + } + + /// Records the call and does what the script says. The rule runs one time for each call. + async fn follow( + &self, + script: Script, + op_label: &'static str, + path: &Path, + call: impl Future>, + ) -> anyhow::Result { + self.record(op_label, path); + match script { + Script::Pass => call.await, + Script::Refuse => Err(anyhow::anyhow!("the storage refused the call")), + Script::LoseTheAnswer => { + call.await?; + Err(anyhow::anyhow!("the answer of the call was lost")) + } + Script::NeverAnswer => { + call.await?; + std::future::pending().await + } + Script::WaitForGate => { + self.gate.cancelled().await; + call.await + } + Script::Vanish | Script::AnswerAlreadyExists => call.await, + Script::Delay(time) => { + tokio::time::sleep(time).await; + call.await + } + Script::RefuseAfter(time) => { + tokio::time::sleep(time).await; + Err(anyhow::anyhow!("the storage refused the call")) + } + Script::Step { refuse, .. } => { + let _stepped = self.wait_for_step(op_label, path).await; + if refuse { + Err(anyhow::anyhow!("the storage refused the call")) + } else { + call.await + } + } + } + } +} + +/// Counts a call that waits for a step, until the guard drops. +struct Waiting<'a>(&'a AtomicUsize); + +impl<'a> Waiting<'a> { + fn new(waiting: &'a AtomicUsize) -> Self { + waiting.fetch_add(1, Ordering::SeqCst); + Self(waiting) + } +} + +impl Drop for Waiting<'_> { + fn drop(&mut self) { + self.0.fetch_sub(1, Ordering::SeqCst); + } +} + +/// Counts a call that took a step as stepped when the guard drops. +struct Stepped<'a>(&'a AtomicUsize); + +impl Drop for Stepped<'_> { + fn drop(&mut self) { + self.0.fetch_add(1, Ordering::SeqCst); + } +} + +impl Debug for ScriptedBlobStorage { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("ScriptedBlobStorage") + } +} + +#[async_trait] +impl BlobStorage for ScriptedBlobStorage { + async fn get_raw( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result>> { + match (self.rule)(op_label, path) { + Script::Vanish => { + self.record(op_label, path); + Ok(None) + } + script => { + self.follow( + script, + op_label, + path, + self.inner.get_raw(target_label, op_label, namespace, path), + ) + .await + } + } + } + + async fn get_stream( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result>>> { + self.answer( + op_label, + path, + self.inner + .get_stream(target_label, op_label, namespace, path), + ) + .await + } + + async fn get_raw_slice( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + start: u64, + end: u64, + ) -> anyhow::Result>> { + self.answer( + op_label, + path, + self.inner + .get_raw_slice(target_label, op_label, namespace, path, start, end), + ) + .await + } + + async fn get_metadata( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result> { + self.answer( + op_label, + path, + self.inner + .get_metadata(target_label, op_label, namespace, path), + ) + .await + } + + async fn put_raw( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + data: &[u8], + ) -> anyhow::Result<()> { + let script = (self.rule)(op_label, path); + if let Script::Step { late: true, .. } = script { + let (inner, owned_path, owned_data) = + (self.inner.clone(), path.to_path_buf(), data.to_vec()); + return self + .late(op_label, path, async move { + inner + .put_raw(target_label, op_label, namespace, &owned_path, &owned_data) + .await + }) + .await; + } + self.follow( + script, + op_label, + path, + self.inner + .put_raw(target_label, op_label, namespace, path, data), + ) + .await + } + + async fn put_raw_if_absent( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + data: &[u8], + ) -> anyhow::Result { + let script = (self.rule)(op_label, path); + if let Script::Step { late: true, .. } = script { + let (inner, owned_path, owned_data) = + (self.inner.clone(), path.to_path_buf(), data.to_vec()); + return self + .late(op_label, path, async move { + inner + .put_raw_if_absent( + target_label, + op_label, + namespace, + &owned_path, + &owned_data, + ) + .await + .map(|_| ()) + }) + .await; + } + if script == Script::AnswerAlreadyExists { + self.record(op_label, path); + let _: PutIfAbsent = self + .inner + .put_raw_if_absent(target_label, op_label, namespace, path, data) + .await?; + return Ok(PutIfAbsent::AlreadyExists); + } + self.follow( + script, + op_label, + path, + self.inner + .put_raw_if_absent(target_label, op_label, namespace, path, data), + ) + .await + } + + async fn put_stream( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + stream: &dyn ErasedReplayableStream>, Error = anyhow::Error>, + ) -> anyhow::Result<()> { + self.answer( + op_label, + path, + self.inner + .put_stream(target_label, op_label, namespace, path, stream), + ) + .await + } + + async fn delete( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result<()> { + let script = (self.rule)(op_label, path); + if let Script::Step { late: true, .. } = script { + let (inner, owned_path) = (self.inner.clone(), path.to_path_buf()); + return self + .late(op_label, path, async move { + inner + .delete(target_label, op_label, namespace, &owned_path) + .await + }) + .await; + } + self.follow( + script, + op_label, + path, + self.inner.delete(target_label, op_label, namespace, path), + ) + .await + } + + async fn create_dir( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result<()> { + self.answer( + op_label, + path, + self.inner + .create_dir(target_label, op_label, namespace, path), + ) + .await + } + + async fn list_dir( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result> { + self.answer( + op_label, + path, + self.inner.list_dir(target_label, op_label, namespace, path), + ) + .await + } + + async fn list_blobs_below( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result> { + self.answer( + op_label, + path, + self.inner + .list_blobs_below(target_label, op_label, namespace, path), + ) + .await + } + + async fn delete_dir( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result { + let script = (self.rule)(op_label, path); + if let Script::Step { late: true, .. } = script { + let (inner, owned_path) = (self.inner.clone(), path.to_path_buf()); + return self + .late(op_label, path, async move { + inner + .delete_dir(target_label, op_label, namespace, &owned_path) + .await + .map(|_| ()) + }) + .await; + } + self.follow( + script, + op_label, + path, + self.inner + .delete_dir(target_label, op_label, namespace, path), + ) + .await + } + + async fn exists( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result { + self.answer( + op_label, + path, + self.inner.exists(target_label, op_label, namespace, path), + ) + .await + } +} diff --git a/golem-worker-executor/src/filesystem_snapshot/time_zone_tests.rs b/golem-worker-executor/src/filesystem_snapshot/time_zone_tests.rs new file mode 100644 index 0000000000..76f13f62fc --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/time_zone_tests.rs @@ -0,0 +1,33 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The image of the executor must give rustic a time zone. rustic asks for the time zone of the +//! system for each timestamp that a save writes, and it logs a warning when it finds none. + +use test_r::test; + +const DOCKERFILE: &str = include_str!("../../docker/Dockerfile"); + +#[test] +fn the_executor_image_sets_a_posix_utc_time_zone() { + let final_stage = DOCKERFILE + .rsplit_once("\nFROM ") + .map(|(_, stage)| stage) + .unwrap_or_default(); + + assert!( + final_stage.lines().any(|line| line.trim() == "ENV TZ=UTC0"), + "the final stage of the executor image does not set `ENV TZ=UTC0`:\n{final_stage}" + ); +} diff --git a/golem-worker-executor/src/services/golem_config.rs b/golem-worker-executor/src/services/golem_config.rs index 661194cf5c..0b3a388b95 100644 --- a/golem-worker-executor/src/services/golem_config.rs +++ b/golem-worker-executor/src/services/golem_config.rs @@ -89,6 +89,8 @@ pub struct GolemConfig { pub memory: MemoryConfig, pub filesystem_storage: FilesystemStorageConfig, #[serde(default)] + pub filesystem_snapshots: FilesystemSnapshotsConfig, + #[serde(default)] pub resource_usage_metering: ResourceUsageMeteringConfig, pub rdbms: RdbmsConfig, pub resource_limits: ResourceLimitsConfig, @@ -267,6 +269,12 @@ impl SafeDisplay for GolemConfig { "{}", self.filesystem_storage.to_safe_string_indented() ); + let _ = writeln!(&mut result, "filesystem snapshots:"); + let _ = writeln!( + &mut result, + "{}", + self.filesystem_snapshots.to_safe_string_indented() + ); let _ = writeln!(&mut result, "resource usage metering:"); let _ = writeln!( &mut result, @@ -388,6 +396,7 @@ impl Default for GolemConfig { public_worker_api: WorkerServiceGrpcConfig::default(), memory: MemoryConfig::default(), filesystem_storage: FilesystemStorageConfig::default(), + filesystem_snapshots: FilesystemSnapshotsConfig::default(), resource_usage_metering: ResourceUsageMeteringConfig::default(), rdbms: RdbmsConfig::default(), resource_limits: ResourceLimitsConfig::default(), @@ -2330,6 +2339,201 @@ impl SafeDisplay for FilesystemPressureConfig { } } +/// The default of [`FilesystemSnapshotStoreConfig::storage_call_deadline`]. The slowest measured +/// call on S3 took 1.7 s, and the value stays at least 10 times the slowest measured call. +pub const DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE: Duration = Duration::from_secs(60); + +/// The default of [`FilesystemSnapshotStoreConfig::restore_reader_threads`]. +const DEFAULT_FILESYSTEM_SNAPSHOT_RESTORE_READER_THREADS: usize = 6; + +/// The default of [`FilesystemSnapshotStoreConfig::save_threads`]. +const DEFAULT_FILESYSTEM_SNAPSHOT_SAVE_THREADS: usize = 2; + +/// Tells whether the executor keeps filesystem snapshots, and gives the settings of the store. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(tag = "type", content = "config")] +pub enum FilesystemSnapshotsConfig { + Disabled(FilesystemSnapshotsDisabledConfig), + Managed(FilesystemSnapshotStoreConfig), +} + +impl Default for FilesystemSnapshotsConfig { + fn default() -> Self { + Self::Disabled(FilesystemSnapshotsDisabledConfig {}) + } +} + +impl SafeDisplay for FilesystemSnapshotsConfig { + fn to_safe_string(&self) -> String { + let mut result = String::new(); + match self { + Self::Disabled(_) => { + let _ = writeln!(&mut result, "disabled"); + } + Self::Managed(store) => { + let _ = writeln!(&mut result, "managed:"); + let _ = writeln!(&mut result, "{}", store.to_safe_string_indented()); + } + } + result + } +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct FilesystemSnapshotsDisabledConfig {} + +/// The settings of the store of filesystem snapshots. +#[derive(Clone, Debug, Serialize)] +pub struct FilesystemSnapshotStoreConfig { + /// The key that encrypts each repository, as 128 hex characters. The key is a secret. + repository_key: FilesystemSnapshotRepositoryKey, + /// The longest time that one blob storage call of the store waits for an answer. + #[serde(with = "humantime_serde")] + storage_call_deadline: Duration, + /// The number of threads that read packs in a restore. + restore_reader_threads: NonZeroUsize, + /// The number of threads of each parallel stage of a save. + save_threads: NonZeroUsize, +} + +#[derive(Deserialize)] +struct RawFilesystemSnapshotStoreConfig { + repository_key: String, + #[serde( + with = "humantime_serde", + default = "default_filesystem_snapshot_storage_call_deadline" + )] + storage_call_deadline: Duration, + #[serde(default = "default_filesystem_snapshot_restore_reader_threads")] + restore_reader_threads: usize, + #[serde(default = "default_filesystem_snapshot_save_threads")] + save_threads: usize, +} + +fn default_filesystem_snapshot_storage_call_deadline() -> Duration { + DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE +} + +fn default_filesystem_snapshot_restore_reader_threads() -> usize { + DEFAULT_FILESYSTEM_SNAPSHOT_RESTORE_READER_THREADS +} + +fn default_filesystem_snapshot_save_threads() -> usize { + DEFAULT_FILESYSTEM_SNAPSHOT_SAVE_THREADS +} + +impl FilesystemSnapshotStoreConfig { + pub fn new( + repository_key: &str, + storage_call_deadline: Duration, + restore_reader_threads: usize, + save_threads: usize, + ) -> Result { + let repository_key = FilesystemSnapshotRepositoryKey::parse(repository_key)?; + if storage_call_deadline.is_zero() { + return Err("storage_call_deadline must be greater than zero".to_string()); + } + let restore_reader_threads = NonZeroUsize::new(restore_reader_threads) + .ok_or_else(|| "restore_reader_threads must be greater than zero".to_string())?; + let save_threads = NonZeroUsize::new(save_threads) + .ok_or_else(|| "save_threads must be greater than zero".to_string())?; + Ok(Self { + repository_key, + storage_call_deadline, + restore_reader_threads, + save_threads, + }) + } + + pub fn repository_key(&self) -> &FilesystemSnapshotRepositoryKey { + &self.repository_key + } + + pub const fn storage_call_deadline(&self) -> Duration { + self.storage_call_deadline + } + + pub const fn restore_reader_threads(&self) -> NonZeroUsize { + self.restore_reader_threads + } + + pub const fn save_threads(&self) -> NonZeroUsize { + self.save_threads + } +} + +impl<'de> Deserialize<'de> for FilesystemSnapshotStoreConfig { + fn deserialize(deserializer: D) -> Result + where + D: Deserializer<'de>, + { + let raw = RawFilesystemSnapshotStoreConfig::deserialize(deserializer)?; + Self::new( + &raw.repository_key, + raw.storage_call_deadline, + raw.restore_reader_threads, + raw.save_threads, + ) + .map_err(D::Error::custom) + } +} + +impl SafeDisplay for FilesystemSnapshotStoreConfig { + fn to_safe_string(&self) -> String { + let mut result = String::new(); + let _ = writeln!(&mut result, "repository key: ****"); + let _ = writeln!( + &mut result, + "storage call deadline: {:?}", + self.storage_call_deadline + ); + let _ = writeln!( + &mut result, + "restore reader threads: {}", + self.restore_reader_threads + ); + let _ = writeln!(&mut result, "save threads: {}", self.save_threads); + result + } +} + +/// The key that encrypts the repositories of filesystem snapshots. It has 64 bytes. +#[derive(Clone, PartialEq, Eq)] +pub struct FilesystemSnapshotRepositoryKey([u8; 64]); + +impl FilesystemSnapshotRepositoryKey { + /// Gives the key from its 128 hex characters. + pub fn parse(text: &str) -> Result { + if text.is_empty() { + return Err("repository_key must not be empty".to_string()); + } + hex::decode(text) + .ok() + .and_then(|bytes| <[u8; 64]>::try_from(bytes).ok()) + .map(Self) + .ok_or_else(|| "repository_key must be 128 hex characters".to_string()) + } + + pub const fn bytes(&self) -> &[u8; 64] { + &self.0 + } +} + +impl std::fmt::Debug for FilesystemSnapshotRepositoryKey { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + formatter.write_str("FilesystemSnapshotRepositoryKey(****)") + } +} + +impl Serialize for FilesystemSnapshotRepositoryKey { + fn serialize(&self, serializer: S) -> Result + where + S: serde::Serializer, + { + serializer.serialize_str(&hex::encode(self.0)) + } +} + #[derive(Clone, Debug, Serialize)] pub struct FilesystemObjectLimitPolicyConfig { /// Number of filesystem objects granted per GiB of allocated storage. @@ -2663,9 +2867,13 @@ pub fn make_config_loader() -> ConfigLoader { #[cfg(test)] mod tests { - use super::{DurableStreamConfig, InvocationResultsConfig, Limits}; + use super::{ + DurableStreamConfig, FilesystemSnapshotStoreConfig, FilesystemSnapshotsConfig, GolemConfig, + InvocationResultsConfig, Limits, + }; use golem_common::SafeDisplay; - use serde_json::Value; + use serde_json::{Value, json}; + use std::time::Duration; use test_r::test; #[test] @@ -2741,4 +2949,137 @@ mod tests { assert!(serde_json::from_value::(serialized).is_err()); } + + const KEY: &str = "000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f\ + 202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f"; + + fn managed(config: Value) -> Result { + serde_json::from_value::(json!({ + "type": "Managed", + "config": config, + })) + .map_err(|error| error.to_string()) + .and_then(|parsed| match parsed { + FilesystemSnapshotsConfig::Managed(store) => Ok(store), + FilesystemSnapshotsConfig::Disabled(_) => Err("disabled".to_string()), + }) + } + + fn refusal(config: Value) -> String { + managed(config).unwrap_err() + } + + #[test] + fn filesystem_snapshots_are_disabled_by_default() { + let config = GolemConfig::default().filesystem_snapshots; + + assert!( + matches!(config, FilesystemSnapshotsConfig::Disabled(_)), + "{config:?}" + ); + } + + #[test] + fn filesystem_snapshots_managed_config_reads_the_key_the_deadline_and_the_thread_counts() { + let store = managed(json!({ + "repository_key": KEY, + "storage_call_deadline": "45s", + "restore_reader_threads": 7, + "save_threads": 3, + })) + .unwrap(); + + assert_eq!( + ( + store.repository_key().bytes().to_vec(), + store.storage_call_deadline(), + store.restore_reader_threads().get(), + store.save_threads().get(), + ), + ((0..64).collect::>(), Duration::from_secs(45), 7, 3) + ); + } + + #[test] + fn filesystem_snapshots_managed_config_gives_the_defaults_of_the_fields_it_does_not_set() { + let store = managed(json!({ "repository_key": KEY })).unwrap(); + + assert_eq!( + ( + store.storage_call_deadline(), + store.restore_reader_threads().get(), + store.save_threads().get(), + ), + (Duration::from_secs(60), 6, 2) + ); + } + + #[test] + fn filesystem_snapshots_managed_config_refuses_a_zero_deadline() { + assert!( + refusal(json!({ "repository_key": KEY, "storage_call_deadline": "0s" })) + .contains("storage_call_deadline must be greater than zero") + ); + } + + #[test] + fn filesystem_snapshots_managed_config_refuses_zero_restore_reader_threads() { + assert!( + refusal(json!({ "repository_key": KEY, "restore_reader_threads": 0 })) + .contains("restore_reader_threads must be greater than zero") + ); + } + + #[test] + fn filesystem_snapshots_managed_config_refuses_zero_save_threads() { + assert!( + refusal(json!({ "repository_key": KEY, "save_threads": 0 })) + .contains("save_threads must be greater than zero") + ); + } + + #[test] + fn filesystem_snapshots_managed_config_refuses_an_empty_short_or_non_hex_key() { + let non_hex = format!("{}g", &KEY[..127]); + let cases = [ + (json!({ "repository_key": "" }), "must not be empty"), + ( + json!({ "repository_key": &KEY[..126] }), + "must be 128 hex characters", + ), + ( + json!({ "repository_key": format!("{KEY}00") }), + "must be 128 hex characters", + ), + ( + json!({ "repository_key": non_hex }), + "must be 128 hex characters", + ), + (json!({}), "missing field `repository_key`"), + ]; + + let unexpected = cases + .into_iter() + .map(|(config, reason)| (refusal(config), reason)) + .filter(|(error, reason)| !error.contains(reason)) + .collect::>(); + + assert_eq!(unexpected, Vec::<(String, &str)>::new()); + } + + #[test] + fn filesystem_snapshots_managed_config_hides_the_key() { + let store = managed(json!({ "repository_key": KEY })).unwrap(); + let shown = FilesystemSnapshotsConfig::Managed(store.clone()).to_safe_string(); + let debugged = format!("{store:?}"); + + assert_eq!( + ( + shown.contains("0001020304"), + debugged.contains("0001020304"), + shown.contains("repository key: ****"), + ), + (false, false, true) + ); + } } diff --git a/golem-worker-service/src/gateway_server/tests.rs b/golem-worker-service/src/gateway_server/tests.rs index 0ed56f402b..ab432c8fb4 100644 --- a/golem-worker-service/src/gateway_server/tests.rs +++ b/golem-worker-service/src/gateway_server/tests.rs @@ -17,8 +17,6 @@ use tokio::sync::Notify; use super::run; -const CLEANUP_TIMEOUT: Duration = Duration::from_secs(2); - #[derive(Clone, Default)] struct LiveBodies { count: Arc, @@ -26,8 +24,10 @@ struct LiveBodies { } impl LiveBodies { + /// Counts one more live body, and wakes each wait for a count. fn guard(&self) -> BodyGuard { self.count.fetch_add(1, Ordering::SeqCst); + self.changed.notify_waiters(); BodyGuard(self.clone()) } @@ -35,20 +35,18 @@ impl LiveBodies { self.count.load(Ordering::SeqCst) } + /// Waits until the count is `expected`. The wait has no limit of its own; the timeout of each + /// test ends a wait that never ends. async fn wait_for(&self, expected: usize) { - tokio::time::timeout(CLEANUP_TIMEOUT, async { - loop { - let changed = self.changed.notified(); - tokio::pin!(changed); - changed.as_mut().enable(); - if self.count() == expected { - return; - } - changed.await; + loop { + let changed = self.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + if self.count() == expected { + return; } - }) - .await - .unwrap_or_else(|_| panic!("body count did not become {expected}; was {}", self.count())); + changed.await; + } } }