From 41e8eb87522c6c52894550151104bb467c45a1ef Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:10:14 -0700 Subject: [PATCH 01/55] Add the filesystem snapshots configuration block with its parse refusals --- golem-debugging-service/src/config.rs | 1 + .../config/worker-executor.sample.env | 3 + .../config/worker-executor.toml | 15 + .../src/services/golem_config.rs | 347 +++++++++++++++++- 4 files changed, 364 insertions(+), 2 deletions(-) diff --git a/golem-debugging-service/src/config.rs b/golem-debugging-service/src/config.rs index 786571e83d..39432f2061 100644 --- a/golem-debugging-service/src/config.rs +++ b/golem-debugging-service/src/config.rs @@ -91,6 +91,7 @@ impl DebugConfig { public_worker_api: self.public_worker_api, memory: self.memory, filesystem_storage: Default::default(), + filesystem_snapshots: Default::default(), resource_usage_metering: Default::default(), rdbms: self.rdbms, resource_limits: self.resource_limits, diff --git a/golem-worker-executor/config/worker-executor.sample.env b/golem-worker-executor/config/worker-executor.sample.env index b3511b1cac..ef8278a3d0 100644 --- a/golem-worker-executor/config/worker-executor.sample.env +++ b/golem-worker-executor/config/worker-executor.sample.env @@ -38,6 +38,7 @@ GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_CAPACITY=1000 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_EVICTION_INTERVAL="1m" GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__NANOS=0 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__SECS=300 +GOLEM__FILESYSTEM_SNAPSHOTS__TYPE="Disabled" #GOLEM__FILESYSTEM_STORAGE__DETERMINISTIC_ROOT_DIR= #GOLEM__FILESYSTEM_STORAGE__MANAGED_XFS_ROOT_DIR= GOLEM__FILESYSTEM_STORAGE__CLEANUP_RETRY__MAX_ATTEMPTS=4 @@ -337,6 +338,7 @@ GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_CAPACITY=1000 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_EVICTION_INTERVAL="1m" GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__NANOS=0 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__SECS=300 +GOLEM__FILESYSTEM_SNAPSHOTS__TYPE="Disabled" #GOLEM__FILESYSTEM_STORAGE__DETERMINISTIC_ROOT_DIR= #GOLEM__FILESYSTEM_STORAGE__MANAGED_XFS_ROOT_DIR= GOLEM__FILESYSTEM_STORAGE__CLEANUP_RETRY__MAX_ATTEMPTS=4 @@ -614,6 +616,7 @@ GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_CAPACITY=1000 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_EVICTION_INTERVAL="1m" GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__NANOS=0 GOLEM__ENVIRONMENT_STATE_SERVICE__CACHE_TTL__SECS=300 +GOLEM__FILESYSTEM_SNAPSHOTS__TYPE="Disabled" #GOLEM__FILESYSTEM_STORAGE__DETERMINISTIC_ROOT_DIR= #GOLEM__FILESYSTEM_STORAGE__MANAGED_XFS_ROOT_DIR= GOLEM__FILESYSTEM_STORAGE__CLEANUP_RETRY__MAX_ATTEMPTS=4 diff --git a/golem-worker-executor/config/worker-executor.toml b/golem-worker-executor/config/worker-executor.toml index 19ee85ad42..1d76c51c04 100644 --- a/golem-worker-executor/config/worker-executor.toml +++ b/golem-worker-executor/config/worker-executor.toml @@ -70,6 +70,11 @@ cache_eviction_interval = "1m" nanos = 0 secs = 300 +[filesystem_snapshots] +type = "Disabled" + +[filesystem_snapshots.config] + [filesystem_storage.cleanup_retry] max_attempts = 4 max_delay = "250ms" @@ -507,6 +512,11 @@ without_time = false # nanos = 0 # secs = 300 # +# [filesystem_snapshots] +# type = "Disabled" +# +# [filesystem_snapshots.config] +# # [filesystem_storage.cleanup_retry] # max_attempts = 4 # max_delay = "250ms" @@ -916,6 +926,11 @@ without_time = false # nanos = 0 # secs = 300 # +# [filesystem_snapshots] +# type = "Disabled" +# +# [filesystem_snapshots.config] +# # [filesystem_storage.cleanup_retry] # max_attempts = 4 # max_delay = "250ms" diff --git a/golem-worker-executor/src/services/golem_config.rs b/golem-worker-executor/src/services/golem_config.rs index 661194cf5c..0346bf94e2 100644 --- a/golem-worker-executor/src/services/golem_config.rs +++ b/golem-worker-executor/src/services/golem_config.rs @@ -89,6 +89,8 @@ pub struct GolemConfig { pub memory: MemoryConfig, pub filesystem_storage: FilesystemStorageConfig, #[serde(default)] + pub filesystem_snapshots: FilesystemSnapshotsConfig, + #[serde(default)] pub resource_usage_metering: ResourceUsageMeteringConfig, pub rdbms: RdbmsConfig, pub resource_limits: ResourceLimitsConfig, @@ -267,6 +269,12 @@ impl SafeDisplay for GolemConfig { "{}", self.filesystem_storage.to_safe_string_indented() ); + let _ = writeln!(&mut result, "filesystem snapshots:"); + let _ = writeln!( + &mut result, + "{}", + self.filesystem_snapshots.to_safe_string_indented() + ); let _ = writeln!(&mut result, "resource usage metering:"); let _ = writeln!( &mut result, @@ -388,6 +396,7 @@ impl Default for GolemConfig { public_worker_api: WorkerServiceGrpcConfig::default(), memory: MemoryConfig::default(), filesystem_storage: FilesystemStorageConfig::default(), + filesystem_snapshots: FilesystemSnapshotsConfig::default(), resource_usage_metering: ResourceUsageMeteringConfig::default(), rdbms: RdbmsConfig::default(), resource_limits: ResourceLimitsConfig::default(), @@ -2330,6 +2339,204 @@ impl SafeDisplay for FilesystemPressureConfig { } } +/// The default of [`FilesystemSnapshotStoreConfig::storage_call_deadline`]. +/// +/// On S3, with the retries of the S3 storage, a write of a pack took at most 1.7 s with eight saves +/// at the same time. A ranged read of a pack took at most 1.5 s under the CPU request of an +/// executor. Keep the value at least 10 times the longest measured call. +pub const DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE: Duration = Duration::from_secs(30); + +/// The default of [`FilesystemSnapshotStoreConfig::restore_reader_threads`]. +const DEFAULT_FILESYSTEM_SNAPSHOT_RESTORE_READER_THREADS: usize = 4; + +/// The default of [`FilesystemSnapshotStoreConfig::save_threads`]. +const DEFAULT_FILESYSTEM_SNAPSHOT_SAVE_THREADS: usize = 4; + +/// Tells whether the executor keeps filesystem snapshots, and gives the settings of the store. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(tag = "type", content = "config")] +pub enum FilesystemSnapshotsConfig { + Disabled(FilesystemSnapshotsDisabledConfig), + Managed(FilesystemSnapshotStoreConfig), +} + +impl Default for FilesystemSnapshotsConfig { + fn default() -> Self { + Self::Disabled(FilesystemSnapshotsDisabledConfig {}) + } +} + +impl SafeDisplay for FilesystemSnapshotsConfig { + fn to_safe_string(&self) -> String { + let mut result = String::new(); + match self { + Self::Disabled(_) => { + let _ = writeln!(&mut result, "disabled"); + } + Self::Managed(store) => { + let _ = writeln!(&mut result, "managed:"); + let _ = writeln!(&mut result, "{}", store.to_safe_string_indented()); + } + } + result + } +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct FilesystemSnapshotsDisabledConfig {} + +/// The settings of the store of filesystem snapshots. +#[derive(Clone, Debug, Serialize)] +pub struct FilesystemSnapshotStoreConfig { + /// The key that encrypts each repository, as 128 hex characters. The key is a secret. + repository_key: FilesystemSnapshotRepositoryKey, + /// The longest time that one blob storage call of the store waits for an answer. + #[serde(with = "humantime_serde")] + storage_call_deadline: Duration, + /// The number of threads that read packs in a restore. + restore_reader_threads: NonZeroUsize, + /// The number of threads of each parallel stage of a save. + save_threads: NonZeroUsize, +} + +#[derive(Deserialize)] +struct RawFilesystemSnapshotStoreConfig { + repository_key: String, + #[serde( + with = "humantime_serde", + default = "default_filesystem_snapshot_storage_call_deadline" + )] + storage_call_deadline: Duration, + #[serde(default = "default_filesystem_snapshot_restore_reader_threads")] + restore_reader_threads: usize, + #[serde(default = "default_filesystem_snapshot_save_threads")] + save_threads: usize, +} + +fn default_filesystem_snapshot_storage_call_deadline() -> Duration { + DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE +} + +fn default_filesystem_snapshot_restore_reader_threads() -> usize { + DEFAULT_FILESYSTEM_SNAPSHOT_RESTORE_READER_THREADS +} + +fn default_filesystem_snapshot_save_threads() -> usize { + DEFAULT_FILESYSTEM_SNAPSHOT_SAVE_THREADS +} + +impl FilesystemSnapshotStoreConfig { + pub fn new( + repository_key: &str, + storage_call_deadline: Duration, + restore_reader_threads: usize, + save_threads: usize, + ) -> Result { + let repository_key = FilesystemSnapshotRepositoryKey::parse(repository_key)?; + if storage_call_deadline.is_zero() { + return Err("storage_call_deadline must be greater than zero".to_string()); + } + let restore_reader_threads = NonZeroUsize::new(restore_reader_threads) + .ok_or_else(|| "restore_reader_threads must be greater than zero".to_string())?; + let save_threads = NonZeroUsize::new(save_threads) + .ok_or_else(|| "save_threads must be greater than zero".to_string())?; + Ok(Self { + repository_key, + storage_call_deadline, + restore_reader_threads, + save_threads, + }) + } + + pub fn repository_key(&self) -> &FilesystemSnapshotRepositoryKey { + &self.repository_key + } + + pub const fn storage_call_deadline(&self) -> Duration { + self.storage_call_deadline + } + + pub const fn restore_reader_threads(&self) -> NonZeroUsize { + self.restore_reader_threads + } + + pub const fn save_threads(&self) -> NonZeroUsize { + self.save_threads + } +} + +impl<'de> Deserialize<'de> for FilesystemSnapshotStoreConfig { + fn deserialize(deserializer: D) -> Result + where + D: Deserializer<'de>, + { + let raw = RawFilesystemSnapshotStoreConfig::deserialize(deserializer)?; + Self::new( + &raw.repository_key, + raw.storage_call_deadline, + raw.restore_reader_threads, + raw.save_threads, + ) + .map_err(D::Error::custom) + } +} + +impl SafeDisplay for FilesystemSnapshotStoreConfig { + fn to_safe_string(&self) -> String { + let mut result = String::new(); + let _ = writeln!(&mut result, "repository key: ****"); + let _ = writeln!( + &mut result, + "storage call deadline: {:?}", + self.storage_call_deadline + ); + let _ = writeln!( + &mut result, + "restore reader threads: {}", + self.restore_reader_threads + ); + let _ = writeln!(&mut result, "save threads: {}", self.save_threads); + result + } +} + +/// The key that encrypts the repositories of filesystem snapshots. It has 64 bytes. +#[derive(Clone, PartialEq, Eq)] +pub struct FilesystemSnapshotRepositoryKey([u8; 64]); + +impl FilesystemSnapshotRepositoryKey { + /// Gives the key from its 128 hex characters. + pub fn parse(text: &str) -> Result { + if text.is_empty() { + return Err("repository_key must not be empty".to_string()); + } + hex::decode(text) + .ok() + .and_then(|bytes| <[u8; 64]>::try_from(bytes).ok()) + .map(Self) + .ok_or_else(|| "repository_key must be 128 hex characters".to_string()) + } + + pub const fn bytes(&self) -> &[u8; 64] { + &self.0 + } +} + +impl std::fmt::Debug for FilesystemSnapshotRepositoryKey { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + formatter.write_str("FilesystemSnapshotRepositoryKey(****)") + } +} + +impl Serialize for FilesystemSnapshotRepositoryKey { + fn serialize(&self, serializer: S) -> Result + where + S: serde::Serializer, + { + serializer.serialize_str(&hex::encode(self.0)) + } +} + #[derive(Clone, Debug, Serialize)] pub struct FilesystemObjectLimitPolicyConfig { /// Number of filesystem objects granted per GiB of allocated storage. @@ -2663,9 +2870,13 @@ pub fn make_config_loader() -> ConfigLoader { #[cfg(test)] mod tests { - use super::{DurableStreamConfig, InvocationResultsConfig, Limits}; + use super::{ + DurableStreamConfig, FilesystemSnapshotStoreConfig, FilesystemSnapshotsConfig, GolemConfig, + InvocationResultsConfig, Limits, + }; use golem_common::SafeDisplay; - use serde_json::Value; + use serde_json::{Value, json}; + use std::time::Duration; use test_r::test; #[test] @@ -2741,4 +2952,136 @@ mod tests { assert!(serde_json::from_value::(serialized).is_err()); } + const KEY: &str = "000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f\ + 202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f"; + + fn managed(config: Value) -> Result { + serde_json::from_value::(json!({ + "type": "Managed", + "config": config, + })) + .map_err(|error| error.to_string()) + .and_then(|parsed| match parsed { + FilesystemSnapshotsConfig::Managed(store) => Ok(store), + FilesystemSnapshotsConfig::Disabled(_) => Err("disabled".to_string()), + }) + } + + fn refusal(config: Value) -> String { + managed(config).unwrap_err() + } + + #[test] + fn filesystem_snapshots_are_disabled_by_default() { + let config = GolemConfig::default().filesystem_snapshots; + + assert!( + matches!(config, FilesystemSnapshotsConfig::Disabled(_)), + "{config:?}" + ); + } + + #[test] + fn filesystem_snapshots_managed_config_reads_the_key_the_deadline_and_the_thread_counts() { + let store = managed(json!({ + "repository_key": KEY, + "storage_call_deadline": "45s", + "restore_reader_threads": 7, + "save_threads": 3, + })) + .unwrap(); + + assert_eq!( + ( + store.repository_key().bytes().to_vec(), + store.storage_call_deadline(), + store.restore_reader_threads().get(), + store.save_threads().get(), + ), + ((0..64).collect::>(), Duration::from_secs(45), 7, 3) + ); + } + + #[test] + fn filesystem_snapshots_managed_config_gives_the_defaults_of_the_fields_it_does_not_set() { + let store = managed(json!({ "repository_key": KEY })).unwrap(); + + assert_eq!( + ( + store.storage_call_deadline(), + store.restore_reader_threads().get(), + store.save_threads().get(), + ), + (Duration::from_secs(30), 4, 4) + ); + } + + #[test] + fn filesystem_snapshots_managed_config_refuses_a_zero_deadline() { + assert!( + refusal(json!({ "repository_key": KEY, "storage_call_deadline": "0s" })) + .contains("storage_call_deadline must be greater than zero") + ); + } + + #[test] + fn filesystem_snapshots_managed_config_refuses_zero_restore_reader_threads() { + assert!( + refusal(json!({ "repository_key": KEY, "restore_reader_threads": 0 })) + .contains("restore_reader_threads must be greater than zero") + ); + } + + #[test] + fn filesystem_snapshots_managed_config_refuses_zero_save_threads() { + assert!( + refusal(json!({ "repository_key": KEY, "save_threads": 0 })) + .contains("save_threads must be greater than zero") + ); + } + + #[test] + fn filesystem_snapshots_managed_config_refuses_an_empty_short_or_non_hex_key() { + let non_hex = format!("{}g", &KEY[..127]); + let cases = [ + (json!({ "repository_key": "" }), "must not be empty"), + ( + json!({ "repository_key": &KEY[..126] }), + "must be 128 hex characters", + ), + ( + json!({ "repository_key": format!("{KEY}00") }), + "must be 128 hex characters", + ), + ( + json!({ "repository_key": non_hex }), + "must be 128 hex characters", + ), + (json!({}), "missing field `repository_key`"), + ]; + + let unexpected = cases + .into_iter() + .map(|(config, reason)| (refusal(config), reason)) + .filter(|(error, reason)| !error.contains(reason)) + .collect::>(); + + assert_eq!(unexpected, Vec::<(String, &str)>::new()); + } + + #[test] + fn filesystem_snapshots_managed_config_hides_the_key() { + let store = managed(json!({ "repository_key": KEY })).unwrap(); + let shown = FilesystemSnapshotsConfig::Managed(store.clone()).to_safe_string(); + let debugged = format!("{store:?}"); + + assert_eq!( + ( + shown.contains("0001020304"), + debugged.contains("0001020304"), + shown.contains("repository key: ****"), + ), + (false, false, true) + ); + } } From e6cf655fb729d81ee35bfcf7039643dffb701b2e Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:10:14 -0700 Subject: [PATCH 02/55] Give the executor image the POSIX time zone UTC0 --- golem-worker-executor/docker/Dockerfile | 4 +++ .../src/filesystem_snapshot/mod.rs | 2 ++ .../filesystem_snapshot/time_zone_tests.rs | 33 +++++++++++++++++++ 3 files changed, 39 insertions(+) create mode 100644 golem-worker-executor/src/filesystem_snapshot/time_zone_tests.rs diff --git a/golem-worker-executor/docker/Dockerfile b/golem-worker-executor/docker/Dockerfile index 11fe1b2e8f..61eace6855 100644 --- a/golem-worker-executor/docker/Dockerfile +++ b/golem-worker-executor/docker/Dockerfile @@ -16,6 +16,10 @@ LABEL cloud.golem.cpu.features="neon,crc,lse,aes,sha2,dotprod" FROM platform-${TARGETARCH} AS final +# The image has no time zone database. jiff reads this POSIX rule without one, so the snapshot +# store writes no time zone warning for each timestamp. +ENV TZ=UTC0 + WORKDIR /app COPY /target/$RUST_TARGET/release/worker-executor ./ COPY /golem-worker-executor/config/worker-executor.toml ./config/worker-executor.toml diff --git a/golem-worker-executor/src/filesystem_snapshot/mod.rs b/golem-worker-executor/src/filesystem_snapshot/mod.rs index f9a6d904f3..9f1abd7681 100644 --- a/golem-worker-executor/src/filesystem_snapshot/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/mod.rs @@ -31,6 +31,8 @@ pub(crate) mod benchmark; mod contract_tests; mod memory; mod rustic; +#[cfg(test)] +mod time_zone_tests; #[allow(unused_imports)] pub(crate) use memory::InMemorySnapshotStore; diff --git a/golem-worker-executor/src/filesystem_snapshot/time_zone_tests.rs b/golem-worker-executor/src/filesystem_snapshot/time_zone_tests.rs new file mode 100644 index 0000000000..76f13f62fc --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/time_zone_tests.rs @@ -0,0 +1,33 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The image of the executor must give rustic a time zone. rustic asks for the time zone of the +//! system for each timestamp that a save writes, and it logs a warning when it finds none. + +use test_r::test; + +const DOCKERFILE: &str = include_str!("../../docker/Dockerfile"); + +#[test] +fn the_executor_image_sets_a_posix_utc_time_zone() { + let final_stage = DOCKERFILE + .rsplit_once("\nFROM ") + .map(|(_, stage)| stage) + .unwrap_or_default(); + + assert!( + final_stage.lines().any(|line| line.trim() == "ENV TZ=UTC0"), + "the final stage of the executor image does not set `ENV TZ=UTC0`:\n{final_stage}" + ); +} From 7845f9b7108cf13df80f4453055f904ff8521fc2 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:10:24 -0700 Subject: [PATCH 03/55] Classify a failed rustic operation as a snapshot store error --- .../src/filesystem_snapshot/rustic/fault.rs | 298 ++++++++++++++++++ .../src/filesystem_snapshot/rustic/mod.rs | 2 + 2 files changed, 300 insertions(+) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs new file mode 100644 index 0000000000..95a73d4718 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs @@ -0,0 +1,298 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The errors that the backend puts into the chain of a rustic error, and the classification of a +//! failed operation as a [`SnapshotStoreError`]. +//! +//! rustic gives no public kind of an error. So the classification reads the chain of sources: the +//! markers of this module, the name errors of the blob storage, and the I/O errors. + +use crate::filesystem_snapshot::SnapshotStoreError; +use golem_service_base::storage::blob::BlobNameError; +use std::error::Error; +use std::fmt::{Display, Formatter}; + +/// A blob storage call of the backend that failed, got no answer within its deadline, or did not +/// start because its operation was cancelled. The source is the failure. +#[derive(Debug)] +pub(super) struct BlobCallFailed { + failure: anyhow::Error, +} + +impl BlobCallFailed { + pub(super) fn new(failure: anyhow::Error) -> Self { + Self { failure } + } +} + +impl Display for BlobCallFailed { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("the blob storage call gave an error") + } +} + +impl Error for BlobCallFailed { + fn source(&self) -> Option<&(dyn Error + 'static)> { + Some(self.failure.as_ref()) + } +} + +/// The operation of the backend was cancelled, so the backend made no more calls. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct OperationCancelled; + +impl Display for OperationCancelled { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("the filesystem snapshot operation was cancelled") + } +} + +impl Error for OperationCancelled {} + +/// Another writer made the config file of the repository first. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct ConfigExists; + +impl Display for ConfigExists { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("another writer made the config file of the repository first") + } +} + +impl Error for ConfigExists {} + +/// Tells whether an error in the chain is [`ConfigExists`]. +pub(super) fn is_config_exists(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| error.is::()) +} + +/// The kind of operation that failed. It tells where an I/O error came from. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum Operation { + /// Reads a tree from the local filesystem and writes it into the repository. + Save, + /// Reads the repository and writes a tree into the local filesystem. + Restore, + /// Reads or changes only the repository. + Repository, +} + +/// Gives the error of the store for an operation that failed with the error. +/// +/// - A failed blob storage call gives `Storage`. It is retryable unless a name error of the blob +/// storage caused it, because a name error is the same for each new try. +/// - An I/O error gives `Source` in a save and `Destination` in a restore, with the kind of that +/// I/O error. +/// - Each other error gives `Storage` that is not retryable in a save, and `Corrupt` in the other +/// operations, because the repository gave data that rustic refused. +pub(super) fn classify(operation: Operation, error: anyhow::Error) -> SnapshotStoreError { + let from_storage = chain(error.as_ref()).any(|error| error.is::()); + let io_kind = chain(error.as_ref()) + .find_map(|error| error.downcast_ref::()) + .map(std::io::Error::kind); + let permanent = chain(error.as_ref()).any(|error| error.is::()); + match (from_storage, io_kind, operation) { + (true, _, _) => SnapshotStoreError::Storage { + retryable: !permanent, + source: error, + }, + (false, Some(kind), Operation::Save) => { + SnapshotStoreError::Source(std::io::Error::new(kind, error_text(&error))) + } + (false, Some(kind), Operation::Restore) => { + SnapshotStoreError::Destination(std::io::Error::new(kind, error_text(&error))) + } + (false, _, Operation::Save) => SnapshotStoreError::Storage { + retryable: false, + source: error, + }, + (false, _, Operation::Restore | Operation::Repository) => { + SnapshotStoreError::Corrupt(error) + } + } +} + +/// Gives the text of the error with the text of each of its sources. +fn error_text(error: &anyhow::Error) -> String { + format!("{error:#}") +} + +/// Gives the error and each error in its chain of sources. +fn chain<'a>(error: &'a (dyn Error + 'static)) -> impl Iterator { + std::iter::successors(Some(error), |&error| error.source()) +} + +#[cfg(test)] +mod tests { + use super::{ + BlobCallFailed, ConfigExists, Operation, OperationCancelled, classify, is_config_exists, + }; + use crate::filesystem_snapshot::SnapshotStoreError; + use golem_service_base::storage::blob::BlobNameError; + use pretty_assertions::assert_eq; + use rustic_core::{ErrorKind, RusticError}; + use std::io; + use test_r::test; + + /// Gives a rustic error whose source is the error, as rustic gives it to the store. + fn rustic(source: impl std::error::Error + Send + Sync + 'static) -> anyhow::Error { + anyhow::Error::new(RusticError::with_source( + ErrorKind::Backend, + "the operation failed", + source, + )) + } + + fn storage_failure(failure: anyhow::Error) -> anyhow::Error { + rustic(BlobCallFailed::new(failure)) + } + + /// Gives the variant of the error, whether it is retryable, and the kind of its I/O error. + fn shape(error: &SnapshotStoreError) -> (&'static str, Option, Option) { + match error { + SnapshotStoreError::NotFound => ("NotFound", None, None), + SnapshotStoreError::AlreadyExists => ("AlreadyExists", None, None), + SnapshotStoreError::Source(error) => ("Source", None, Some(error.kind())), + SnapshotStoreError::Destination(error) => ("Destination", None, Some(error.kind())), + SnapshotStoreError::Storage { retryable, .. } => ("Storage", Some(*retryable), None), + SnapshotStoreError::Corrupt(_) => ("Corrupt", None, None), + } + } + + #[test] + fn a_failed_blob_storage_call_gives_a_retryable_storage_error_in_each_operation() { + let shapes = + [Operation::Save, Operation::Restore, Operation::Repository].map(|operation| { + shape(&classify( + operation, + storage_failure(anyhow::anyhow!("the bucket is gone")), + )) + }); + + assert_eq!(shapes, [("Storage", Some(true), None); 3]); + } + + #[test] + fn a_blob_storage_call_that_holds_an_io_error_is_still_a_storage_error() { + let failure = anyhow::Error::new(io::Error::new(io::ErrorKind::StorageFull, "no space")); + + assert_eq!( + shape(&classify(Operation::Restore, storage_failure(failure))), + ("Storage", Some(true), None) + ); + } + + #[test] + fn a_name_error_of_the_blob_storage_is_not_retryable() { + let failure = anyhow::Error::new(BlobNameError::NoName { + path: std::path::PathBuf::new(), + }); + + assert_eq!( + shape(&classify(Operation::Repository, storage_failure(failure))), + ("Storage", Some(false), None) + ); + } + + #[test] + fn a_cancelled_operation_gives_a_retryable_storage_error_in_each_operation() { + let shapes = + [Operation::Save, Operation::Restore, Operation::Repository].map(|operation| { + shape(&classify( + operation, + storage_failure(anyhow::Error::new(OperationCancelled)), + )) + }); + + assert_eq!(shapes, [("Storage", Some(true), None); 3]); + } + + #[test] + fn an_io_error_of_a_save_gives_source_with_its_kind() { + let error = rustic(io::Error::new(io::ErrorKind::PermissionDenied, "locked")); + + let classified = classify(Operation::Save, error); + + assert_eq!( + ( + shape(&classified), + classified.to_string().contains("locked") + ), + ( + ("Source", None, Some(io::ErrorKind::PermissionDenied)), + true + ) + ); + } + + #[test] + fn a_full_volume_during_a_restore_gives_destination() { + let error = rustic(io::Error::new(io::ErrorKind::StorageFull, "no space")); + + assert_eq!( + shape(&classify(Operation::Restore, error)), + ("Destination", None, Some(io::ErrorKind::StorageFull)) + ); + } + + #[test] + fn a_metadata_error_of_a_restore_gives_destination() { + let error = rustic(io::Error::new( + io::ErrorKind::PermissionDenied, + "setting extended attributes failed", + )); + + assert_eq!( + shape(&classify(Operation::Restore, error)), + ("Destination", None, Some(io::ErrorKind::PermissionDenied)) + ); + } + + #[test] + fn an_error_without_storage_or_io_is_corrupt_when_it_reads_and_not_retryable_in_a_save() { + let refused = || { + anyhow::Error::new(RusticError::new( + ErrorKind::Cryptography, + "the data failed its check", + )) + }; + + assert_eq!( + [ + shape(&classify(Operation::Restore, refused())), + shape(&classify(Operation::Repository, refused())), + shape(&classify(Operation::Save, refused())), + ], + [ + ("Corrupt", None, None), + ("Corrupt", None, None), + ("Storage", Some(false), None), + ] + ); + } + + #[test] + fn the_config_marker_is_found_in_the_chain() { + let exists = storage_failure(anyhow::Error::new(ConfigExists)); + let other = storage_failure(anyhow::anyhow!("the bucket is gone")); + + assert_eq!( + ( + is_config_exists(exists.as_ref()), + is_config_exists(other.as_ref()) + ), + (true, false) + ); + } +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index bc9b4a00f6..9b30b446f8 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -19,6 +19,8 @@ //! module. mod backend; +#[cfg_attr(not(test), allow(dead_code))] +mod fault; #[cfg(test)] mod holding; From 13f0854f742e561c9bb0aea94ef4f5d6cc599515 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:10:24 -0700 Subject: [PATCH 04/55] Cancel, stage and write conditionally in the rustic backend, and publish a staged snapshot file --- .../src/filesystem_snapshot/rustic/backend.rs | 126 +++++-- .../rustic/backend/tests.rs | 235 ++++++++++++- .../src/filesystem_snapshot/rustic/mod.rs | 4 + .../src/filesystem_snapshot/rustic/publish.rs | 166 +++++++++ .../rustic/publish/tests.rs | 218 ++++++++++++ .../filesystem_snapshot/rustic/scripted.rs | 319 ++++++++++++++++++ 6 files changed, 1039 insertions(+), 29 deletions(-) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index 856eb0dcee..8d99fc90f5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -17,10 +17,12 @@ //! The backend keeps the files of one repository in one blob storage namespace, with the paths of //! the restic repository format. rustic calls the backend from threads outside the async runtime, //! and each call waits for the blob storage on the runtime that the backend holds. Each call waits -//! for at most a deadline. +//! for at most a deadline, and a cancelled operation makes no more calls. +use super::fault::{BlobCallFailed, ConfigExists, OperationCancelled}; +use super::publish::{SnapshotStage, StagedSnapshot}; use bytes::Bytes; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace, PutIfAbsent}; use rustic_core::{ BytesList, ErrorKind, FileType, Id, ReadBackend, RusticError, RusticResult, WriteBackend, }; @@ -29,6 +31,7 @@ use std::path::{Path, PathBuf}; use std::sync::Arc; use std::time::Duration; use tokio::runtime::Handle; +use tokio_util::sync::CancellationToken; /// The target label of each blob storage call of the backend. const TARGET_LABEL: &str = "filesystem_snapshot"; @@ -77,12 +80,17 @@ impl StorageCall { /// A call that gets no answer from the blob storage within the deadline gives an error. The /// runtime must be a multi-thread runtime, because on a `current_thread` runtime `Handle::block_on` /// does not drive the timer of the deadline. +/// +/// The config file and the index files are written only when their path has no blob. A snapshot +/// file goes into the stage of the backend when it has one, and the backend does not write it. #[derive(Debug)] pub(super) struct BlobBackend { storage: Arc, namespace: BlobStorageNamespace, runtime: Handle, deadline: Duration, + cancel: CancellationToken, + stage: Option>, } impl BlobBackend { @@ -99,23 +107,70 @@ impl BlobBackend { namespace, runtime, deadline, + cancel: CancellationToken::new(), + stage: None, + } + } + + /// Gives the backend with the token of its operation. When the token is cancelled, a call that + /// has not started gives an error at once, and a call that runs stops and gives an error. + #[cfg_attr(not(test), allow(dead_code))] + pub(super) fn cancelled_by(self, cancel: CancellationToken) -> Self { + Self { cancel, ..self } + } + + /// Gives the backend with a stage for the snapshot file of a save. + #[cfg_attr(not(test), allow(dead_code))] + pub(super) fn staging_in(self, stage: Arc) -> Self { + Self { + stage: Some(stage), + ..self } } /// Waits for one call on the blob storage, and gives its result as a rustic result. /// /// Each call of the backend on the blob storage goes through this function. A call that gives - /// no answer within the deadline gives an error, the same as a call that failed. + /// no answer within the deadline gives an error, the same as a call that failed. So does a + /// call of a cancelled operation. fn request( &self, call: StorageCall, path: &Path, future: impl Future>, ) -> RusticResult { + if self.cancel.is_cancelled() { + return Err(storage_error( + call, + path, + anyhow::Error::new(OperationCancelled), + )); + } self.runtime - .block_on(answer_within(self.deadline, future)) + .block_on(async { + tokio::select! { + biased; + answer = answer_within(self.deadline, future) => answer, + () = self.cancel.cancelled() => Err(anyhow::Error::new(OperationCancelled)), + } + }) .map_err(|error| storage_error(call, path, error)) } + + /// Writes the content at the path only when the path has no blob, and gives whether it wrote. + fn write_if_absent(&self, path: &Path, content: &[u8]) -> RusticResult { + self.request( + StorageCall::Write, + path, + self.storage.put_raw_if_absent( + TARGET_LABEL, + StorageCall::Write.label(), + self.namespace.clone(), + path, + content, + ), + ) + } } /// Gives the output of the future, or an error when the future gives no output within the deadline. @@ -124,7 +179,7 @@ impl BlobBackend { /// thread without a runtime context can wait for the result with `Handle::block_on`. A future that /// is ready at its first poll always gives its output. At the deadline, the function drops the /// future and gives an error whose root cause is tokio's `Elapsed`. -async fn answer_within( +pub(super) async fn answer_within( deadline: Duration, future: impl Future>, ) -> anyhow::Result { @@ -248,25 +303,33 @@ impl WriteBackend for BlobBackend { ) -> RusticResult<()> { let path = file_path(tpe, id)?; let parts = content.into_vec(); - let joined; - let data: &[u8] = match parts.as_slice() { - [part] => part, - parts => { - joined = join(parts); - &joined - } + let content = match parts.as_slice() { + [part] => part.clone(), + parts => Bytes::from(join(parts)), }; - self.request( - StorageCall::Write, - &path, - self.storage.put_raw( - TARGET_LABEL, - StorageCall::Write.label(), - self.namespace.clone(), + match (tpe, &self.stage) { + (FileType::Snapshot, Some(stage)) => stage + .keep(StagedSnapshot { path, content }) + .map_err(|staged| second_snapshot(&staged.path)), + (FileType::Config, _) => match self.write_if_absent(&path, &content)? { + PutIfAbsent::Written => Ok(()), + PutIfAbsent::AlreadyExists => Err(config_exists(&path)), + }, + // The name of an index file is the hash of its content, so a blob at the path holds + // the same content. + (FileType::Index, _) => self.write_if_absent(&path, &content).map(|_| ()), + _ => self.request( + StorageCall::Write, &path, - data, + self.storage.put_raw( + TARGET_LABEL, + StorageCall::Write.label(), + self.namespace.clone(), + &path, + &content, + ), ), - ) + } } fn remove(&self, tpe: FileType, id: &Id, _cacheable: bool) -> RusticResult<()> { @@ -342,12 +405,31 @@ fn missing_file(path: &Path) -> Box { .attach_context("path", path.display().to_string()) } +/// The error of a config file that another writer made first. +fn config_exists(path: &Path) -> Box { + RusticError::with_source( + ErrorKind::Backend, + "The blob storage already holds the config file `{path}`.", + ConfigExists, + ) + .attach_context("path", path.display().to_string()) +} + +/// The error of a second snapshot file in one stage. +fn second_snapshot(path: &Path) -> Box { + RusticError::new( + ErrorKind::Internal, + "The stage already holds a snapshot file, so it cannot keep `{path}`.", + ) + .attach_context("path", path.display().to_string()) +} + /// The error of a blob storage call that failed. fn storage_error(call: StorageCall, path: &Path, error: anyhow::Error) -> Box { RusticError::with_source( ErrorKind::Backend, "The blob storage call `{call}` failed at `{path}`.", - error, + BlobCallFailed::new(error), ) .attach_context("call", call.label()) .attach_context("path", path.display().to_string()) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index 95877bd875..66fa7afa3d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -18,8 +18,12 @@ //! runtime that the backend holds, as the threads of rustic are not. use super::super::STORAGE_CALL_DEADLINE; +use super::super::fault::{Operation, OperationCancelled, classify, is_config_exists}; use super::super::holding::{holding_storage, reached_deadline}; +use super::super::publish::{SnapshotStage, StagedSnapshot}; +use super::super::scripted::{Script, ScriptedBlobStorage}; use super::{BlobBackend, file_size}; +use crate::filesystem_snapshot::SnapshotStoreError; use anyhow::anyhow; use async_trait::async_trait; use bytes::Bytes; @@ -37,6 +41,7 @@ use std::sync::Arc; use std::time::{Duration, Instant}; use test_r::test; use tokio::runtime::Runtime; +use tokio_util::sync::CancellationToken; use uuid::Uuid; /// The longest time that a test waits for the calls on the backend. @@ -473,6 +478,218 @@ fn a_thread_that_is_not_a_thread_of_the_runtime_can_call_the_backend() { assert_eq!(read.ok().flatten(), Some(Bytes::from_static(b"index"))); } +#[test] +fn a_cancelled_backend_makes_no_storage_call() { + let runtime = Runtime::new().unwrap(); + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let cancel = CancellationToken::new(); + let backend = BlobBackend::new( + storage.clone(), + new_namespace(), + runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .cancelled_by(cancel.clone()); + cancel.cancel(); + + let cancelled = [ + backend.list_with_size(FileType::Config).err(), + backend.list_with_size(FileType::Pack).err(), + backend.read_full(FileType::Pack, &id("ab")).err(), + backend + .read_partial(FileType::Pack, &id("ab"), false, 0, 1) + .err(), + backend + .write_bytes(FileType::Pack, &id("ab"), false, bytes("pack")) + .err(), + backend.remove(FileType::Pack, &id("ab"), false).err(), + ] + .map(|error| error.is_some_and(|error| was_cancelled(&error))); + + assert_eq!((cancelled, storage.calls()), ([true; 6], Vec::new())); +} + +#[test] +fn a_cancel_ends_a_call_that_runs() { + let runtime = Runtime::new().unwrap(); + let (storage, _gate, _dropped) = + holding_storage(Arc::new(InMemoryBlobStorage::new()), |_, _| true); + let cancel = CancellationToken::new(); + let backend = BlobBackend::new( + storage, + new_namespace(), + runtime.handle().clone(), + Duration::from_secs(60), + ) + .cancelled_by(cancel.clone()); + runtime.spawn(async move { + tokio::time::sleep(Duration::from_millis(100)).await; + cancel.cancel(); + }); + + let outcome = within_limit(move || { + backend + .read_full(FileType::Pack, &id("ab")) + .err() + .map(|error| (was_cancelled(&error), reached_deadline(&*error))) + }); + + assert_eq!(outcome, Some(Some((true, false)))); +} + +#[test] +fn a_backend_whose_token_is_not_cancelled_answers() { + let fixture = Fixture::new(); + let backend = BlobBackend::new( + fixture.storage.clone(), + fixture.namespace.clone(), + fixture.runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .cancelled_by(CancellationToken::new()); + + let read = backend + .write_bytes(FileType::Pack, &id("ab"), false, bytes("pack")) + .and_then(|()| backend.read_full(FileType::Pack, &id("ab"))); + + assert_eq!(read.ok(), Some(Bytes::from_static(b"pack"))); +} + +#[test] +fn the_config_file_is_written_only_when_the_repository_has_none() { + let fixture = Fixture::new(); + + let first = + fixture + .backend + .write_bytes(FileType::Config, &Id::default(), false, bytes("first")); + let second = + fixture + .backend + .write_bytes(FileType::Config, &Id::default(), false, bytes("second")); + + assert_eq!( + ( + first.is_ok(), + second + .as_ref() + .is_err_and(|error| is_config_exists(&**error)), + fixture.stored(), + ), + (true, true, vec![("config".to_string(), 5)]) + ); +} + +#[test] +fn an_index_file_that_is_there_is_kept_and_its_write_succeeds() { + let fixture = Fixture::new(); + let path = format!("index/{}", "ab".repeat(32)); + fixture.put(&path, b"kept"); + + let written = + fixture + .backend + .write_bytes(FileType::Index, &id("ab"), false, bytes("replacement")); + let read = fixture.backend.read_full(FileType::Index, &id("ab")); + + assert_eq!( + (written.is_ok(), read.ok()), + (true, Some(Bytes::from_static(b"kept"))) + ); +} + +#[test] +fn a_pack_file_that_is_there_is_written_again() { + let fixture = Fixture::new(); + fixture.put(&format!("data/ab/{}", "ab".repeat(32)), b"old"); + + let written = fixture + .backend + .write_bytes(FileType::Pack, &id("ab"), false, bytes("new")); + let read = fixture.backend.read_full(FileType::Pack, &id("ab")); + + assert_eq!( + (written.is_ok(), read.ok()), + (true, Some(Bytes::from_static(b"new"))) + ); +} + +#[test] +fn a_backend_with_a_stage_keeps_the_snapshot_file_and_does_not_write_it() { + let fixture = Fixture::new(); + let stage = Arc::new(SnapshotStage::default()); + let backend = BlobBackend::new( + fixture.storage.clone(), + fixture.namespace.clone(), + fixture.runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .staging_in(stage.clone()); + let content = [Bytes::from_static(b"snap"), Bytes::from_static(b"shot")] + .into_iter() + .fold(BytesList::default(), |mut content, part| { + content.add(part); + content + }); + + let kept = backend.write_bytes(FileType::Snapshot, &id("cd"), false, content); + let second = backend.write_bytes(FileType::Snapshot, &id("ef"), false, bytes("other")); + let pack = backend.write_bytes(FileType::Pack, &id("ab"), false, bytes("pack")); + + assert_eq!( + ( + kept.is_ok(), + second.is_err(), + pack.is_ok(), + stage.take(), + fixture.stored(), + ), + ( + true, + true, + true, + Some(StagedSnapshot { + path: PathBuf::from(format!("snapshots/{}", "cd".repeat(32))).into_boxed_path(), + content: Bytes::from_static(b"snapshot"), + }), + vec![(format!("data/ab/{}", "ab".repeat(32)), 4)] + ) + ); +} + +#[test] +fn each_failed_call_is_a_storage_failure_to_the_classification() { + let runtime = Runtime::new().unwrap(); + let backend = BlobBackend::new( + Arc::new(FailingBlobStorage), + new_namespace(), + runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ); + + let classified = [ + backend.list_with_size(FileType::Pack).err(), + backend.read_full(FileType::Pack, &id("ab")).err(), + backend + .write_bytes(FileType::Pack, &id("ab"), false, bytes("pack")) + .err(), + ] + .map(|error| { + error.map(|error| { + matches!( + classify(Operation::Restore, anyhow::Error::new(error)), + SnapshotStoreError::Storage { + retryable: true, + .. + } + ) + }) + }); + + assert_eq!(classified, [Some(true); 3]); +} + #[test] fn a_size_that_does_not_fit_in_32_bits_is_an_error() { let path = Path::new("data/ab/file"); @@ -494,14 +711,18 @@ fn error_text(result: RusticResult) -> String { } } -/// Gives the text of the error, with the text of its source. +/// Gives the text of the error, with the text of each error in its chain of sources. fn text_of(error: &RusticError) -> String { - format!( - "{error} {}", - std::error::Error::source(error) - .map(ToString::to_string) - .unwrap_or_default() - ) + std::iter::successors(std::error::Error::source(error), |error| error.source()) + .fold(error.to_string(), |text, source| format!("{text} {source}")) +} + +/// Tells whether the error or an error in its chain of sources is [`OperationCancelled`]. +fn was_cancelled(error: &RusticError) -> bool { + std::iter::successors(Some(error as &(dyn std::error::Error + 'static)), |error| { + error.source() + }) + .any(|error| error.is::()) } /// A blob storage that fails every call. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 9b30b446f8..1e877f390d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -21,10 +21,14 @@ mod backend; #[cfg_attr(not(test), allow(dead_code))] mod fault; +#[cfg_attr(not(test), allow(dead_code))] +mod publish; #[cfg(test)] mod holding; #[cfg(test)] +mod scripted; +#[cfg(test)] mod tests; use super::{SnapshotName, SnapshotScope}; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs new file mode 100644 index 0000000000..e8d48455ca --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs @@ -0,0 +1,166 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The publish of a snapshot file. +//! +//! In a save of the store, the backend keeps the snapshot file in a [`SnapshotStage`] and does not +//! write it. The save writes it later with [`publish`], after the blocking work returns. That +//! write is the step that makes the snapshot visible. A publish that fails, or that the caller +//! drops, deletes the file again, because a write that the storage received can still complete. + +use super::backend::answer_within; +use bytes::Bytes; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use std::path::Path; +use std::sync::{Arc, Mutex, PoisonError}; +use std::time::Duration; +use tokio::runtime::Handle; +use tokio_util::task::TaskTracker; +use tracing::warn; + +/// The target label of each blob storage call of a publish. +const TARGET_LABEL: &str = "filesystem_snapshot"; + +/// A snapshot file that the backend kept and did not write. +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct StagedSnapshot { + /// The path of the file, relative to the root of the namespace. + pub(super) path: Box, + pub(super) content: Bytes, +} + +/// The place where the backend of one save keeps its snapshot file. +#[derive(Debug, Default)] +pub(super) struct SnapshotStage(Mutex>); + +impl SnapshotStage { + /// Keeps the file. A stage holds one file, so a second file gives it back as the error. + pub(super) fn keep(&self, staged: StagedSnapshot) -> Result<(), StagedSnapshot> { + let mut slot = self.0.lock().unwrap_or_else(PoisonError::into_inner); + match *slot { + Some(_) => Err(staged), + None => { + *slot = Some(staged); + Ok(()) + } + } + } + + /// Takes the file out of the stage. + pub(super) fn take(&self) -> Option { + self.0.lock().unwrap_or_else(PoisonError::into_inner).take() + } +} + +/// The snapshot files of one scope: the storage, the namespace of the scope, and the deadline of +/// each call. +#[derive(Clone, Debug)] +pub(super) struct SnapshotFiles { + pub(super) storage: Arc, + pub(super) namespace: BlobStorageNamespace, + pub(super) deadline: Duration, +} + +/// Writes the staged file, and so makes the snapshot visible. +/// +/// The file is written only when the path has no blob. The name of a snapshot file is the hash of +/// its content, so a blob at the path already holds this content, and the call succeeds. +/// +/// When the write fails, the call deletes the path before it gives the error. When the caller +/// drops the call during the write, a task of `tracker` deletes the path. +pub(super) async fn publish( + files: &SnapshotFiles, + staged: &StagedSnapshot, + tracker: &TaskTracker, +) -> anyhow::Result<()> { + let mut retraction = RetractOnDrop { + files: files.clone(), + path: staged.path.clone(), + tracker: tracker.clone(), + armed: true, + }; + let written = answer_within( + files.deadline, + files.storage.put_raw_if_absent( + TARGET_LABEL, + "publish", + files.namespace.clone(), + &staged.path, + &staged.content, + ), + ) + .await; + retraction.armed = false; + match written { + Ok(_) => Ok(()), + Err(error) => { + retract_or_warn(files, &staged.path).await; + Err(error) + } + } +} + +/// Deletes the snapshot file at the path. A path without a blob gives success. +pub(super) async fn retract(files: &SnapshotFiles, path: &Path) -> anyhow::Result<()> { + answer_within( + files.deadline, + files + .storage + .delete(TARGET_LABEL, "retract", files.namespace.clone(), path), + ) + .await +} + +async fn retract_or_warn(files: &SnapshotFiles, path: &Path) { + if let Err(error) = retract(files, path).await { + warn!( + path = %path.display(), + error = %format!("{error:#}"), + "Failed to delete a filesystem snapshot file whose publish did not finish" + ); + } +} + +/// Deletes the path in a task of the tracker when it is dropped while it is armed. +struct RetractOnDrop { + files: SnapshotFiles, + path: Box, + tracker: TaskTracker, + armed: bool, +} + +impl Drop for RetractOnDrop { + fn drop(&mut self) { + if !self.armed { + return; + } + let files = self.files.clone(); + let path = self.path.clone(); + match Handle::try_current() { + Ok(runtime) => { + self.tracker.spawn_on( + async move { retract_or_warn(&files, &path).await }, + &runtime, + ); + } + Err(_) => warn!( + path = %path.display(), + "Failed to delete a dropped filesystem snapshot file, because no runtime runs" + ), + } + } +} + +#[cfg(test)] +mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs new file mode 100644 index 0000000000..4042fee100 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -0,0 +1,218 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use super::super::holding::reached_deadline; +use super::super::scripted::{Script, ScriptedBlobStorage}; +use super::{SnapshotFiles, SnapshotStage, StagedSnapshot, publish, retract}; +use bytes::Bytes; +use futures::FutureExt; +use golem_common::model::environment::EnvironmentId; +use golem_service_base::storage::blob::memory::InMemoryBlobStorage; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use pretty_assertions::assert_eq; +use std::path::{Path, PathBuf}; +use std::sync::Arc; +use std::time::Duration; +use test_r::test; +use tokio_util::task::TaskTracker; +use uuid::Uuid; + +/// The longest time that a test waits for the tasks of a tracker. +const LIMIT: Duration = Duration::from_secs(10); + +const SNAPSHOT_PATH: &str = + "snapshots/cdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcd"; + +fn staged() -> StagedSnapshot { + StagedSnapshot { + path: PathBuf::from(SNAPSHOT_PATH).into_boxed_path(), + content: Bytes::from_static(b"snapshot"), + } +} + +/// Gives the snapshot files of a new namespace over a storage whose script for the publish is +/// `publish`, and the in-memory storage below it. +fn files( + publish: Script, + deadline: Duration, +) -> ( + SnapshotFiles, + Arc, + Arc, +) { + let inner = Arc::new(InMemoryBlobStorage::new()); + let storage = ScriptedBlobStorage::new(inner.clone(), move |op_label, _| { + if op_label == "publish" { + publish + } else { + Script::Pass + } + }); + ( + SnapshotFiles { + storage: storage.clone(), + namespace: BlobStorageNamespace::InitialAgentFiles { + environment_id: EnvironmentId(Uuid::new_v4()), + }, + deadline, + }, + storage, + inner, + ) +} + +/// Gives the content of the snapshot file, when the storage holds it. +async fn stored(files: &SnapshotFiles, inner: &InMemoryBlobStorage) -> Option> { + inner + .get_raw( + "test", + "test", + files.namespace.clone(), + Path::new(SNAPSHOT_PATH), + ) + .await + .unwrap() +} + +#[test] +async fn a_publish_writes_the_staged_file() { + let (files, _, inner) = files(Script::Pass, Duration::from_secs(2)); + + let published = publish(&files, &staged(), &TaskTracker::new()).await; + + assert_eq!( + (published.is_ok(), stored(&files, &inner).await), + (true, Some(b"snapshot".to_vec())) + ); +} + +#[test] +async fn a_publish_of_a_file_that_is_there_succeeds_and_keeps_the_file() { + let (files, _, inner) = files(Script::Pass, Duration::from_secs(2)); + let tracker = TaskTracker::new(); + + let first = publish(&files, &staged(), &tracker).await; + let second = publish(&files, &staged(), &tracker).await; + + assert_eq!( + (first.is_ok(), second.is_ok(), stored(&files, &inner).await), + (true, true, Some(b"snapshot".to_vec())) + ); +} + +#[test] +async fn a_publish_whose_answer_is_lost_deletes_the_file_and_gives_the_error() { + let (files, storage, inner) = files(Script::LoseTheAnswer, Duration::from_secs(2)); + + let published = publish(&files, &staged(), &TaskTracker::new()).await; + + assert_eq!( + ( + published.map_err(|error| error.to_string()), + stored(&files, &inner).await, + storage.calls(), + ), + ( + Err("the answer of the call was lost".to_string()), + None, + vec![ + ("publish", SNAPSHOT_PATH.to_string()), + ("retract", SNAPSHOT_PATH.to_string()), + ] + ) + ); +} + +#[test] +async fn a_publish_that_reaches_the_deadline_deletes_the_file_that_the_storage_wrote() { + let (files, _, inner) = files(Script::NeverAnswer, Duration::from_millis(100)); + + let published = publish(&files, &staged(), &TaskTracker::new()).await; + + assert_eq!( + ( + published + .as_ref() + .is_err_and(|error| reached_deadline(error.as_ref())), + stored(&files, &inner).await + ), + (true, None) + ); +} + +#[test] +async fn a_publish_that_the_caller_drops_deletes_the_file_in_a_task_of_the_tracker() { + let (files, _, inner) = files(Script::NeverAnswer, Duration::from_secs(60)); + let tracker = TaskTracker::new(); + + let dropped = publish(&files, &staged(), &tracker).now_or_never(); + let written_before_the_drop = stored(&files, &inner).await; + tracker.close(); + let waited = tokio::time::timeout(LIMIT, tracker.wait()).await; + + assert_eq!( + ( + dropped.is_none(), + written_before_the_drop, + waited.is_ok(), + stored(&files, &inner).await + ), + (true, Some(b"snapshot".to_vec()), true, None) + ); +} + +#[test] +async fn a_publish_that_returns_keeps_the_file_when_the_tasks_of_the_tracker_end() { + let (files, _, inner) = files(Script::Pass, Duration::from_secs(2)); + let tracker = TaskTracker::new(); + + let published = publish(&files, &staged(), &tracker).await; + tracker.close(); + let waited = tokio::time::timeout(LIMIT, tracker.wait()).await; + + assert_eq!( + ( + published.is_ok(), + waited.is_ok(), + stored(&files, &inner).await + ), + (true, true, Some(b"snapshot".to_vec())) + ); +} + +#[test] +async fn a_retract_of_a_path_without_a_file_succeeds() { + let (files, _, _) = files(Script::Pass, Duration::from_secs(2)); + + assert!(retract(&files, Path::new(SNAPSHOT_PATH)).await.is_ok()); +} + +#[test] +fn a_stage_keeps_one_file_until_it_is_taken() { + let stage = SnapshotStage::default(); + let other = StagedSnapshot { + content: Bytes::from_static(b"other"), + ..staged() + }; + + let first = stage.keep(staged()); + let second = stage.keep(other.clone()); + let taken = stage.take(); + let taken_again = stage.take(); + + assert_eq!( + (first, second, taken, taken_again), + (Ok(()), Err(other), Some(staged()), None) + ); +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs new file mode 100644 index 0000000000..ddb80b5e04 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -0,0 +1,319 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! A blob storage for tests that records each call and follows a script for each call. +//! +//! This module is test code, and it compiles only for tests. + +use async_trait::async_trait; +use bytes::Bytes; +use futures::stream::BoxStream; +use golem_service_base::replayable_stream::ErasedReplayableStream; +use golem_service_base::storage::blob::memory::InMemoryBlobStorage; +use golem_service_base::storage::blob::{ + BlobMetadata, BlobStorage, BlobStorageNamespace, ExistsResult, ListedBlob, PutIfAbsent, +}; +use std::fmt::{Debug, Formatter}; +use std::future::Future; +use std::path::{Path, PathBuf}; +use std::sync::{Arc, Mutex, PoisonError}; + +/// What the storage does with one call. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) enum Script { + /// Passes the call to the in-memory storage. + Pass, + /// Gives an error and does not pass the call. + Refuse, + /// Passes the call, and then gives an error in place of its answer. + LoseTheAnswer, + /// Passes the call, and then never answers. + NeverAnswer, +} + +/// A rule that gives the script of a call from its operation label and its path. +type Rule = Box Script + Send + Sync>; + +/// A blob storage that records the operation label and the path of each call, and does with each +/// call what its rule gives. +pub(super) struct ScriptedBlobStorage { + inner: Arc, + rule: Rule, + calls: Mutex)>>, +} + +impl ScriptedBlobStorage { + pub(super) fn new( + inner: Arc, + rule: impl Fn(&str, &Path) -> Script + Send + Sync + 'static, + ) -> Arc { + Arc::new(Self { + inner, + rule: Box::new(rule), + calls: Mutex::new(Vec::new()), + }) + } + + /// Gives the operation label and the path of each call, in the order of the calls. + pub(super) fn calls(&self) -> Vec<(&'static str, String)> { + self.calls + .lock() + .unwrap_or_else(PoisonError::into_inner) + .iter() + .map(|(op_label, path)| (*op_label, path.display().to_string())) + .collect() + } + + async fn answer( + &self, + op_label: &'static str, + path: &Path, + call: impl Future>, + ) -> anyhow::Result { + self.calls + .lock() + .unwrap_or_else(PoisonError::into_inner) + .push((op_label, path.into())); + match (self.rule)(op_label, path) { + Script::Pass => call.await, + Script::Refuse => Err(anyhow::anyhow!("the storage refused the call")), + Script::LoseTheAnswer => { + call.await?; + Err(anyhow::anyhow!("the answer of the call was lost")) + } + Script::NeverAnswer => { + call.await?; + std::future::pending().await + } + } + } +} + +impl Debug for ScriptedBlobStorage { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("ScriptedBlobStorage") + } +} + +#[async_trait] +impl BlobStorage for ScriptedBlobStorage { + async fn get_raw( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result>> { + self.answer( + op_label, + path, + self.inner.get_raw(target_label, op_label, namespace, path), + ) + .await + } + + async fn get_stream( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result>>> { + self.answer( + op_label, + path, + self.inner + .get_stream(target_label, op_label, namespace, path), + ) + .await + } + + async fn get_raw_slice( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + start: u64, + end: u64, + ) -> anyhow::Result>> { + self.answer( + op_label, + path, + self.inner + .get_raw_slice(target_label, op_label, namespace, path, start, end), + ) + .await + } + + async fn get_metadata( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result> { + self.answer( + op_label, + path, + self.inner + .get_metadata(target_label, op_label, namespace, path), + ) + .await + } + + async fn put_raw( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + data: &[u8], + ) -> anyhow::Result<()> { + self.answer( + op_label, + path, + self.inner + .put_raw(target_label, op_label, namespace, path, data), + ) + .await + } + + async fn put_raw_if_absent( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + data: &[u8], + ) -> anyhow::Result { + self.answer( + op_label, + path, + self.inner + .put_raw_if_absent(target_label, op_label, namespace, path, data), + ) + .await + } + + async fn put_stream( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + stream: &dyn ErasedReplayableStream>, Error = anyhow::Error>, + ) -> anyhow::Result<()> { + self.answer( + op_label, + path, + self.inner + .put_stream(target_label, op_label, namespace, path, stream), + ) + .await + } + + async fn delete( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result<()> { + self.answer( + op_label, + path, + self.inner.delete(target_label, op_label, namespace, path), + ) + .await + } + + async fn create_dir( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result<()> { + self.answer( + op_label, + path, + self.inner + .create_dir(target_label, op_label, namespace, path), + ) + .await + } + + async fn list_dir( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result> { + self.answer( + op_label, + path, + self.inner.list_dir(target_label, op_label, namespace, path), + ) + .await + } + + async fn list_blobs_below( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result> { + self.answer( + op_label, + path, + self.inner + .list_blobs_below(target_label, op_label, namespace, path), + ) + .await + } + + async fn delete_dir( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result { + self.answer( + op_label, + path, + self.inner + .delete_dir(target_label, op_label, namespace, path), + ) + .await + } + + async fn exists( + &self, + target_label: &'static str, + op_label: &'static str, + namespace: BlobStorageNamespace, + path: &Path, + ) -> anyhow::Result { + self.answer( + op_label, + path, + self.inner.exists(target_label, op_label, namespace, path), + ) + .await + } +} From 9d167c83aa06a2b691f6d3f5ccec135e5863722d Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:10:24 -0700 Subject: [PATCH 05/55] Add the prune ledger of a snapshot scope and the rule that makes a prune due --- .../src/filesystem_snapshot/rustic/mod.rs | 2 + .../src/filesystem_snapshot/rustic/prune.rs | 294 ++++++++++++++++++ 2 files changed, 296 insertions(+) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 1e877f390d..1ac0a6aa06 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -22,6 +22,8 @@ mod backend; #[cfg_attr(not(test), allow(dead_code))] mod fault; #[cfg_attr(not(test), allow(dead_code))] +mod prune; +#[cfg_attr(not(test), allow(dead_code))] mod publish; #[cfg(test)] diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs new file mode 100644 index 0000000000..f2510a8b93 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -0,0 +1,294 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! When a delete of the store prunes the repository of its scope. +//! +//! The scope keeps a small ledger blob next to the files of the repository. The ledger holds the +//! packed bytes that deleted snapshots added since the last prune, the time of the last prune, and +//! whether that prune marked packs that a later prune removes. The ledger is advice: two deletes at +//! the same time can lose a count, and that only makes a prune come later. + +use super::backend::answer_within; +use golem_common::model::Timestamp; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use serde::{Deserialize, Serialize}; +use std::path::Path; +use std::time::Duration; +use tracing::warn; + +/// The target label of each blob storage call on the ledger. +const TARGET_LABEL: &str = "filesystem_snapshot"; + +/// The path of the ledger blob, relative to the root of the namespace of the scope. +pub(super) const LEDGER_PATH: &str = "golem/prune-ledger"; + +/// What the scope did since its last prune. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] +pub(super) struct PruneLedger { + /// The packed bytes that the deleted snapshots added, since the last prune. + pub(super) freed_bytes: u64, + /// The time of the last prune, in milliseconds since the Unix epoch. + pub(super) last_prune_millis: Option, + /// Whether the last prune marked packs that a later prune removes. + pub(super) awaiting_removal: bool, +} + +impl PruneLedger { + /// Gives the ledger after a delete of snapshots that added `bytes` packed bytes. + pub(super) fn with_deleted(self, bytes: u64) -> Self { + Self { + freed_bytes: self.freed_bytes.saturating_add(bytes), + ..self + } + } + + /// Gives the ledger after a prune at `now` that marked packs or not. + pub(super) fn after_prune(now: Timestamp, marked_packs: bool) -> Self { + Self { + freed_bytes: 0, + last_prune_millis: Some(now.to_millis()), + awaiting_removal: marked_packs, + } + } +} + +/// Tells whether a prune is due at `now`. +/// +/// A prune is due when the grace period passed since the last prune, and the freed bytes reach the +/// threshold or the last prune marked packs. A threshold of zero counts as one byte, so a prune +/// never runs for a scope that freed nothing and marked nothing. +pub(super) fn prune_due( + ledger: &PruneLedger, + now: Timestamp, + threshold: u64, + grace: Duration, +) -> bool { + let grace_passed = ledger.last_prune_millis.is_none_or(|last| { + now.to_millis() >= last.saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) + }); + let work = ledger.freed_bytes >= threshold.max(1) || ledger.awaiting_removal; + grace_passed && work +} + +/// Reads the ledger of the scope. A scope without a ledger gives an empty ledger, and so does a +/// ledger that does not parse, because the ledger is advice. +pub(super) async fn read_ledger( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, + deadline: Duration, +) -> anyhow::Result { + let content = answer_within( + deadline, + storage.get_raw( + TARGET_LABEL, + "read_ledger", + namespace.clone(), + Path::new(LEDGER_PATH), + ), + ) + .await?; + Ok(content.map_or_else(PruneLedger::default, |content| { + serde_json::from_slice(&content).unwrap_or_else(|error| { + warn!( + error = %error, + "The prune ledger of a filesystem snapshot scope does not parse, so it starts again" + ); + PruneLedger::default() + }) + })) +} + +/// Writes the ledger of the scope over the ledger that was there. +pub(super) async fn write_ledger( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, + deadline: Duration, + ledger: &PruneLedger, +) -> anyhow::Result<()> { + let content = serde_json::to_vec(ledger)?; + answer_within( + deadline, + storage.put_raw( + TARGET_LABEL, + "write_ledger", + namespace.clone(), + Path::new(LEDGER_PATH), + &content, + ), + ) + .await +} + +#[cfg(test)] +mod tests { + use super::{LEDGER_PATH, PruneLedger, prune_due, read_ledger, write_ledger}; + use golem_common::model::Timestamp; + use golem_common::model::environment::EnvironmentId; + use golem_service_base::storage::blob::memory::InMemoryBlobStorage; + use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; + use pretty_assertions::assert_eq; + use std::path::Path; + use std::time::Duration; + use test_r::test; + use uuid::Uuid; + + const MIB: u64 = 1024 * 1024; + const THRESHOLD: u64 = 64 * MIB; + const GRACE: Duration = Duration::from_secs(3600); + const DEADLINE: Duration = Duration::from_secs(2); + + fn at(millis: u64) -> Timestamp { + Timestamp::from(millis) + } + + fn ledger( + freed_bytes: u64, + last_prune_millis: Option, + awaiting_removal: bool, + ) -> PruneLedger { + PruneLedger { + freed_bytes, + last_prune_millis, + awaiting_removal, + } + } + + fn new_namespace() -> BlobStorageNamespace { + BlobStorageNamespace::InitialAgentFiles { + environment_id: EnvironmentId(Uuid::new_v4()), + } + } + + #[test] + fn a_prune_is_due_when_the_freed_bytes_reach_the_threshold() { + let now = at(10_000_000); + + assert_eq!( + [ + prune_due(&ledger(THRESHOLD - 1, None, false), now, THRESHOLD, GRACE), + prune_due(&ledger(THRESHOLD, None, false), now, THRESHOLD, GRACE), + prune_due(&ledger(THRESHOLD + 1, None, false), now, THRESHOLD, GRACE), + ], + [false, true, true] + ); + } + + #[test] + fn no_second_prune_runs_within_the_grace_period() { + let last = 1_000_000; + let grace_millis = 3_600_000; + let full = |now| { + prune_due( + &ledger(THRESHOLD, Some(last), true), + at(now), + THRESHOLD, + GRACE, + ) + }; + + assert_eq!( + [ + full(last), + full(last + grace_millis - 1), + full(last + grace_millis), + full(last + grace_millis + 1), + ], + [false, false, true, true] + ); + } + + #[test] + fn marked_packs_make_a_prune_due_after_the_grace_period_without_freed_bytes() { + let last = 1_000_000; + let after_grace = at(last + 3_600_000); + + assert_eq!( + [ + prune_due(&ledger(0, Some(last), true), after_grace, THRESHOLD, GRACE), + prune_due(&ledger(0, Some(last), false), after_grace, THRESHOLD, GRACE), + ], + [true, false] + ); + } + + #[test] + fn a_zero_threshold_prunes_after_each_delete_that_freed_bytes() { + let now = at(10_000_000); + + assert_eq!( + [ + prune_due(&ledger(0, None, false), now, 0, Duration::ZERO), + prune_due(&ledger(1, None, false), now, 0, Duration::ZERO), + prune_due(&ledger(1, Some(10_000_000), false), now, 0, Duration::ZERO), + ], + [false, true, true] + ); + } + + #[test] + fn a_delete_adds_its_bytes_and_a_prune_starts_the_ledger_again() { + let deleted = ledger(5, Some(7), true) + .with_deleted(10) + .with_deleted(u64::MAX); + + assert_eq!( + ( + deleted, + PruneLedger::after_prune(at(42), true), + PruneLedger::after_prune(at(43), false) + ), + ( + ledger(u64::MAX, Some(7), true), + ledger(0, Some(42), true), + ledger(0, Some(43), false) + ) + ); + } + + #[test] + async fn the_ledger_is_written_and_read_back() { + let storage = InMemoryBlobStorage::new(); + let namespace = new_namespace(); + let written = ledger(123, Some(456), true); + + let before = read_ledger(&storage, &namespace, DEADLINE).await.unwrap(); + write_ledger(&storage, &namespace, DEADLINE, &written) + .await + .unwrap(); + let after = read_ledger(&storage, &namespace, DEADLINE).await.unwrap(); + + assert_eq!((before, after), (PruneLedger::default(), written)); + } + + #[test] + async fn a_ledger_that_does_not_parse_reads_as_an_empty_ledger() { + let storage = InMemoryBlobStorage::new(); + let namespace = new_namespace(); + storage + .put_raw( + "test", + "test", + namespace.clone(), + Path::new(LEDGER_PATH), + b"not json", + ) + .await + .unwrap(); + + assert_eq!( + read_ledger(&storage, &namespace, DEADLINE).await.unwrap(), + PruneLedger::default() + ); + } +} From a358148f3724a8eff610410a0635e891dbd6dd58 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:10:25 -0700 Subject: [PATCH 06/55] Copy and delete a snapshot scope on its blobs --- .../src/filesystem_snapshot/rustic/mod.rs | 2 + .../src/filesystem_snapshot/rustic/scope.rs | 173 +++++++++++ .../filesystem_snapshot/rustic/scope/tests.rs | 269 ++++++++++++++++++ 3 files changed, 444 insertions(+) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 1ac0a6aa06..67e2552bc5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -25,6 +25,8 @@ mod fault; mod prune; #[cfg_attr(not(test), allow(dead_code))] mod publish; +#[cfg_attr(not(test), allow(dead_code))] +mod scope; #[cfg(test)] mod holding; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs new file mode 100644 index 0000000000..30f125eca9 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs @@ -0,0 +1,173 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The copy and the delete of a whole scope, on the blobs of its repository. +//! +//! These operations do not read the repository format. They only know the directories of the +//! repository, its config file, and the ledger directory of the store. + +use super::backend::answer_within; +use super::prune::LEDGER_PATH; +use futures::{StreamExt, TryStreamExt, stream}; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace, PutIfAbsent}; +use std::path::{Path, PathBuf}; +use std::time::Duration; + +/// The target label of each blob storage call on a scope. +const TARGET_LABEL: &str = "filesystem_snapshot"; + +/// The path of the config file of a repository. +const CONFIG_PATH: &str = "config"; + +/// The directories of a repository in the order of a listing. A save writes them in the reverse +/// order, so each snapshot file in a listing has its index files and packs in the later listings. +const LISTING_ORDER: [&str; 4] = ["snapshots", "index", "keys", "data"]; + +/// The directories of a repository in the order of the writes of a copy: packs, index files, keys, +/// then snapshot files. So a snapshot file in the target always has its data. +const COPY_ORDER: [&str; 4] = ["data", "index", "keys", "snapshots"]; + +/// Copies the repository of the namespace `from` into the empty namespace `to`. +/// +/// A namespace without a config file holds no repository, so the call copies nothing. The call +/// writes the config file of `to` last, so `to` holds a repository only when all its blobs are +/// there. It does not copy the ledger. A blob that is gone when the call reads it was deleted +/// after the listing, and the call does not copy it. +pub(super) async fn copy_scope( + storage: &dyn BlobStorage, + from: &BlobStorageNamespace, + to: &BlobStorageNamespace, + deadline: Duration, +) -> anyhow::Result<()> { + let Some(config) = answer_within( + deadline, + storage.get_raw( + TARGET_LABEL, + "copy_read", + from.clone(), + Path::new(CONFIG_PATH), + ), + ) + .await? + else { + return Ok(()); + }; + let listed = stream::iter(LISTING_ORDER) + .then(|directory| async move { + answer_within( + deadline, + storage.list_blobs_below( + TARGET_LABEL, + "copy_list", + from.clone(), + Path::new(directory), + ), + ) + .await + .map(|blobs| (directory, blobs)) + }) + .try_collect::>() + .await?; + let paths = COPY_ORDER + .iter() + .flat_map(|directory| { + listed + .iter() + .filter(move |(listed_directory, _)| listed_directory == directory) + .flat_map(|(_, blobs)| blobs.iter().map(|blob| blob.path.clone())) + }) + .collect::>(); + stream::iter(paths.iter().map(Ok)) + .try_for_each(|path| copy_blob(storage, from, to, path, deadline)) + .await?; + answer_within( + deadline, + storage.put_raw_if_absent( + TARGET_LABEL, + "copy_write", + to.clone(), + Path::new(CONFIG_PATH), + &config, + ), + ) + .await + .map(|_: PutIfAbsent| ()) +} + +async fn copy_blob( + storage: &dyn BlobStorage, + from: &BlobStorageNamespace, + to: &BlobStorageNamespace, + path: &Path, + deadline: Duration, +) -> anyhow::Result<()> { + let content = answer_within( + deadline, + storage.get_raw(TARGET_LABEL, "copy_read", from.clone(), path), + ) + .await?; + match content { + Some(content) => { + answer_within( + deadline, + storage.put_raw(TARGET_LABEL, "copy_write", to.clone(), path, &content), + ) + .await + } + None => Ok(()), + } +} + +/// Deletes the repository of the namespace, and the ledger of the store. +/// +/// The call deletes the config file first, so the namespace holds no repository from that step on. +/// Then it deletes each directory. A namespace that holds nothing gives success. +pub(super) async fn delete_scope( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, + deadline: Duration, +) -> anyhow::Result<()> { + answer_within( + deadline, + storage.delete( + TARGET_LABEL, + "delete_scope", + namespace.clone(), + Path::new(CONFIG_PATH), + ), + ) + .await?; + let ledger_directory = Path::new(LEDGER_PATH) + .parent() + .map(Path::to_path_buf) + .unwrap_or_default(); + let directories = LISTING_ORDER + .iter() + .map(PathBuf::from) + .chain(std::iter::once(ledger_directory)) + .collect::>(); + stream::iter(directories.iter().map(Ok)) + .try_for_each(|directory| async move { + answer_within( + deadline, + storage.delete_dir(TARGET_LABEL, "delete_scope", namespace.clone(), directory), + ) + .await + .map(|_| ()) + }) + .await +} + +#[cfg(test)] +mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs new file mode 100644 index 0000000000..90b940d21c --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs @@ -0,0 +1,269 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use super::super::scripted::{Script, ScriptedBlobStorage}; +use super::{copy_scope, delete_scope}; +use golem_common::model::environment::EnvironmentId; +use golem_service_base::storage::blob::memory::InMemoryBlobStorage; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use pretty_assertions::assert_eq; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; +use test_r::test; +use uuid::Uuid; + +const DEADLINE: Duration = Duration::from_secs(2); + +/// The blobs of a small repository, with the ledger of the store. +const REPOSITORY: [(&str, &str); 6] = [ + ("config", "config"), + ("data/ab/abab", "pack"), + ("golem/prune-ledger", "ledger"), + ("index/cdcd", "index"), + ("keys/efef", "key"), + ("snapshots/0101", "snapshot"), +]; + +fn new_namespace() -> BlobStorageNamespace { + BlobStorageNamespace::InitialAgentFiles { + environment_id: EnvironmentId(Uuid::new_v4()), + } +} + +async fn put_all( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, + blobs: &[(&str, &str)], +) { + futures::future::join_all(blobs.iter().map(|(path, content)| { + storage.put_raw( + "test", + "test", + namespace.clone(), + Path::new(path), + content.as_bytes(), + ) + })) + .await + .into_iter() + .collect::>>() + .unwrap(); +} + +/// Gives the path and the content of each blob of the namespace, in the order of the paths. +async fn stored( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, +) -> Vec<(String, String)> { + let listed = storage + .list_blobs_below("test", "test", namespace.clone(), Path::new("")) + .await + .unwrap(); + let mut blobs = futures::future::join_all(listed.iter().map(|blob| async { + let content = storage + .get_raw("test", "test", namespace.clone(), &blob.path) + .await + .unwrap() + .unwrap(); + ( + blob.path.display().to_string(), + String::from_utf8(content).unwrap(), + ) + })) + .await; + blobs.sort(); + blobs +} + +fn owned(blobs: &[(&str, &str)]) -> Vec<(String, String)> { + blobs + .iter() + .map(|(path, content)| (path.to_string(), content.to_string())) + .collect() +} + +#[test] +async fn a_copy_gives_the_target_each_blob_of_the_repository_and_not_the_ledger() { + let storage = InMemoryBlobStorage::new(); + let (from, to) = (new_namespace(), new_namespace()); + put_all(&storage, &from, &REPOSITORY).await; + + copy_scope(&storage, &from, &to, DEADLINE).await.unwrap(); + + assert_eq!( + (stored(&storage, &to).await, stored(&storage, &from).await), + ( + owned( + &REPOSITORY + .into_iter() + .filter(|(path, _)| !path.starts_with("golem/")) + .collect::>() + ), + owned(&REPOSITORY) + ) + ); +} + +#[test] +async fn a_copy_writes_the_packs_the_index_files_the_keys_the_snapshot_files_and_then_the_config() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let (from, to) = (new_namespace(), new_namespace()); + put_all(&*storage, &from, &REPOSITORY).await; + + copy_scope(&*storage, &from, &to, DEADLINE).await.unwrap(); + + assert_eq!( + storage + .calls() + .into_iter() + .filter(|(op_label, _)| *op_label == "copy_write") + .map(|(_, path)| path) + .collect::>(), + vec![ + "data/ab/abab", + "index/cdcd", + "keys/efef", + "snapshots/0101", + "config" + ] + ); +} + +#[test] +async fn a_copy_lists_the_snapshot_files_before_the_index_files_the_keys_and_the_packs() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let (from, to) = (new_namespace(), new_namespace()); + put_all(&*storage, &from, &REPOSITORY).await; + + copy_scope(&*storage, &from, &to, DEADLINE).await.unwrap(); + + assert_eq!( + storage + .calls() + .into_iter() + .filter(|(op_label, _)| *op_label == "copy_list") + .map(|(_, path)| path) + .collect::>(), + vec!["snapshots", "index", "keys", "data"] + ); +} + +#[test] +async fn a_copy_of_a_namespace_without_a_config_copies_nothing() { + let storage = InMemoryBlobStorage::new(); + let (from, to) = (new_namespace(), new_namespace()); + put_all( + &storage, + &from, + &REPOSITORY + .into_iter() + .filter(|(path, _)| *path != "config") + .collect::>(), + ) + .await; + + copy_scope(&storage, &from, &to, DEADLINE).await.unwrap(); + + assert_eq!(stored(&storage, &to).await, Vec::<(String, String)>::new()); +} + +#[test] +async fn a_copy_that_fails_gives_the_error_and_the_target_has_no_config() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "copy_read" && path.starts_with("index") { + Script::Refuse + } else { + Script::Pass + } + }); + let (from, to) = (new_namespace(), new_namespace()); + put_all(&*storage, &from, &REPOSITORY).await; + + let copied = copy_scope(&*storage, &from, &to, DEADLINE).await; + + assert_eq!( + ( + copied.is_err(), + stored(&*storage, &to) + .await + .into_iter() + .map(|(path, _)| path) + .collect::>() + ), + (true, vec!["data/ab/abab".to_string()]) + ); +} + +#[test] +async fn a_deleted_scope_holds_no_blob_and_another_scope_keeps_its_blobs() { + let storage = InMemoryBlobStorage::new(); + let (deleted, kept) = (new_namespace(), new_namespace()); + put_all(&storage, &deleted, &REPOSITORY).await; + put_all(&storage, &kept, &REPOSITORY).await; + + delete_scope(&storage, &deleted, DEADLINE).await.unwrap(); + + assert_eq!( + ( + stored(&storage, &deleted).await, + stored(&storage, &kept).await + ), + (Vec::new(), owned(&REPOSITORY)) + ); +} + +#[test] +async fn a_delete_of_a_scope_deletes_the_config_first() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let namespace = new_namespace(); + put_all(&*storage, &namespace, &REPOSITORY).await; + + delete_scope(&*storage, &namespace, DEADLINE).await.unwrap(); + + assert_eq!( + storage + .calls() + .into_iter() + .filter(|(op_label, _)| *op_label == "delete_scope") + .map(|(_, path)| path) + .collect::>(), + vec!["config", "snapshots", "index", "keys", "data", "golem"] + ); +} + +#[test] +async fn a_delete_of_an_unused_scope_succeeds_and_can_run_again() { + let storage = InMemoryBlobStorage::new(); + let namespace = new_namespace(); + + let first = delete_scope(&storage, &namespace, DEADLINE).await; + put_all(&storage, &namespace, &REPOSITORY).await; + let second = delete_scope(&storage, &namespace, DEADLINE).await; + let third = delete_scope(&storage, &namespace, DEADLINE).await; + + assert_eq!( + ( + first.is_ok(), + second.is_ok(), + third.is_ok(), + stored(&storage, &namespace).await + ), + (true, true, true, Vec::new()) + ); +} From 96d550aeb75843ccd9c24ec32ce2b42239f792f7 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:34:48 -0700 Subject: [PATCH 07/55] Hold the delete of a dropped publish until the test reads the file --- .../rustic/publish/tests.rs | 39 ++++++++++++------- .../filesystem_snapshot/rustic/scripted.rs | 14 +++++++ 2 files changed, 39 insertions(+), 14 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs index 4042fee100..a61fc61724 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -42,9 +42,10 @@ fn staged() -> StagedSnapshot { } /// Gives the snapshot files of a new namespace over a storage whose script for the publish is -/// `publish`, and the in-memory storage below it. +/// `publish` and for the delete is `retract`, and the in-memory storage below it. fn files( publish: Script, + retract: Script, deadline: Duration, ) -> ( SnapshotFiles, @@ -52,12 +53,10 @@ fn files( Arc, ) { let inner = Arc::new(InMemoryBlobStorage::new()); - let storage = ScriptedBlobStorage::new(inner.clone(), move |op_label, _| { - if op_label == "publish" { - publish - } else { - Script::Pass - } + let storage = ScriptedBlobStorage::new(inner.clone(), move |op_label, _| match op_label { + "publish" => publish, + "retract" => retract, + _ => Script::Pass, }); ( SnapshotFiles { @@ -87,7 +86,7 @@ async fn stored(files: &SnapshotFiles, inner: &InMemoryBlobStorage) -> Option, rule: Rule, calls: Mutex)>>, + gate: CancellationToken, } impl ScriptedBlobStorage { @@ -62,9 +66,15 @@ impl ScriptedBlobStorage { inner, rule: Box::new(rule), calls: Mutex::new(Vec::new()), + gate: CancellationToken::new(), }) } + /// Lets each call that waits for the gate, and each later such call, go on. + pub(super) fn open_gate(&self) { + self.gate.cancel(); + } + /// Gives the operation label and the path of each call, in the order of the calls. pub(super) fn calls(&self) -> Vec<(&'static str, String)> { self.calls @@ -96,6 +106,10 @@ impl ScriptedBlobStorage { call.await?; std::future::pending().await } + Script::WaitForGate => { + self.gate.cancelled().await; + call.await + } } } } From ba222da8cc2a24c3003f8ed3d98eef3cb800ed12 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:34:49 -0700 Subject: [PATCH 08/55] Write the blobs of a scope copy in the reverse order of the listing --- .../src/filesystem_snapshot/rustic/scope.rs | 17 +++++------------ .../filesystem_snapshot/rustic/scope/tests.rs | 9 ++++++--- 2 files changed, 11 insertions(+), 15 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs index 30f125eca9..d80ebe91f6 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs @@ -32,12 +32,10 @@ const CONFIG_PATH: &str = "config"; /// The directories of a repository in the order of a listing. A save writes them in the reverse /// order, so each snapshot file in a listing has its index files and packs in the later listings. +/// A copy writes them in the reverse order too, so a snapshot file in the target always has its +/// data. const LISTING_ORDER: [&str; 4] = ["snapshots", "index", "keys", "data"]; -/// The directories of a repository in the order of the writes of a copy: packs, index files, keys, -/// then snapshot files. So a snapshot file in the target always has its data. -const COPY_ORDER: [&str; 4] = ["data", "index", "keys", "snapshots"]; - /// Copies the repository of the namespace `from` into the empty namespace `to`. /// /// A namespace without a config file holds no repository, so the call copies nothing. The call @@ -75,18 +73,13 @@ pub(super) async fn copy_scope( ), ) .await - .map(|blobs| (directory, blobs)) }) .try_collect::>() .await?; - let paths = COPY_ORDER + let paths = listed .iter() - .flat_map(|directory| { - listed - .iter() - .filter(move |(listed_directory, _)| listed_directory == directory) - .flat_map(|(_, blobs)| blobs.iter().map(|blob| blob.path.clone())) - }) + .rev() + .flat_map(|blobs| blobs.iter().map(|blob| blob.path.clone())) .collect::>(); stream::iter(paths.iter().map(Ok)) .try_for_each(|path| copy_blob(storage, from, to, path, deadline)) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs index 90b940d21c..aaff157543 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs @@ -117,7 +117,7 @@ async fn a_copy_gives_the_target_each_blob_of_the_repository_and_not_the_ledger( } #[test] -async fn a_copy_writes_the_packs_the_index_files_the_keys_the_snapshot_files_and_then_the_config() { +async fn a_copy_writes_the_packs_the_keys_the_index_files_the_snapshot_files_and_then_the_config() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); let (from, to) = (new_namespace(), new_namespace()); @@ -134,8 +134,8 @@ async fn a_copy_writes_the_packs_the_index_files_the_keys_the_snapshot_files_and .collect::>(), vec![ "data/ab/abab", - "index/cdcd", "keys/efef", + "index/cdcd", "snapshots/0101", "config" ] @@ -205,7 +205,10 @@ async fn a_copy_that_fails_gives_the_error_and_the_target_has_no_config() { .map(|(path, _)| path) .collect::>() ), - (true, vec!["data/ab/abab".to_string()]) + ( + true, + vec!["data/ab/abab".to_string(), "keys/efef".to_string()] + ) ); } From ac83e53f046aebbd0b499cd09428a36c17493e64 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:34:49 -0700 Subject: [PATCH 09/55] Separate the key constant of the configuration tests --- golem-worker-executor/src/services/golem_config.rs | 1 + 1 file changed, 1 insertion(+) diff --git a/golem-worker-executor/src/services/golem_config.rs b/golem-worker-executor/src/services/golem_config.rs index 0346bf94e2..82e667787f 100644 --- a/golem-worker-executor/src/services/golem_config.rs +++ b/golem-worker-executor/src/services/golem_config.rs @@ -2952,6 +2952,7 @@ mod tests { assert!(serde_json::from_value::(serialized).is_err()); } + const KEY: &str = "000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f\ 202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f"; From 6b3f43102bff4ca288b7ded436f304641b38ee59 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:16:14 -0700 Subject: [PATCH 10/55] Take the storage call deadline of the rustic bridge from the configuration default --- .../src/filesystem_snapshot/benchmark/mod.rs | 3 ++- .../src/filesystem_snapshot/rustic/backend/tests.rs | 2 +- .../src/filesystem_snapshot/rustic/mod.rs | 9 --------- .../src/filesystem_snapshot/rustic/tests.rs | 6 +++--- 4 files changed, 6 insertions(+), 14 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs index 97ace67da1..7b05b1ad9e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs @@ -39,9 +39,10 @@ mod volume; use super::rustic::{ ChangeDetection, Chunking, Compression, InspectReport, PhaseTime, Repository, RepositoryKey, - RepositorySettings, STORAGE_CALL_DEADLINE, SaveSettings, + RepositorySettings, SaveSettings, }; use super::{SnapshotName, SnapshotScope}; +use crate::services::golem_config::DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE as STORAGE_CALL_DEADLINE; use agents::{AgentStorage, FIRST_AGENT}; use golem_common::model::environment::EnvironmentId; use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index 66fa7afa3d..cb52483c36 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -17,13 +17,13 @@ //! Each test calls the backend from the thread of the test. That thread is not a thread of the //! runtime that the backend holds, as the threads of rustic are not. -use super::super::STORAGE_CALL_DEADLINE; use super::super::fault::{Operation, OperationCancelled, classify, is_config_exists}; use super::super::holding::{holding_storage, reached_deadline}; use super::super::publish::{SnapshotStage, StagedSnapshot}; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::{BlobBackend, file_size}; use crate::filesystem_snapshot::SnapshotStoreError; +use crate::services::golem_config::DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE as STORAGE_CALL_DEADLINE; use anyhow::anyhow; use async_trait::async_trait; use bytes::Bytes; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 67e2552bc5..5fe655745c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -57,15 +57,6 @@ use std::sync::Arc; use std::time::{Duration, Instant}; use tokio::runtime::Handle; -/// The longest time that one call of a repository waits for the blob storage. -/// -/// A call that gets no answer within this time fails, and its operation fails with it. The value -/// stops a call that does not return. It is not a limit for a slow call. On S3, with the retries of -/// the S3 storage, a write of a pack took at most 1.7 s with eight saves at the same time. A ranged -/// read of a pack took at most 1.5 s under the CPU request of an executor. Keep the value at least -/// 10 times the longest measured call. -pub(super) const STORAGE_CALL_DEADLINE: Duration = Duration::from_secs(30); - /// The key that encrypts a repository. /// /// The key has 64 bytes: 32 bytes of the AES-256 key, then 16 bytes of the number `k` and 16 diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index 4ed61a7a94..d94c238915 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -22,15 +22,15 @@ use super::backend::BlobBackend; use super::holding::{holding_storage, reached_deadline}; use super::{ ChangeDetection, Chunking, Compression, OperationPhase, PruneSettings, RepackLimits, - Repository, RepositoryKey, RepositorySettings, STORAGE_CALL_DEADLINE, SaveSettings, - backup_options, config_options, open_existing, prune_options, repository_options, run_blocking, - unopened, + Repository, RepositoryKey, RepositorySettings, SaveSettings, backup_options, config_options, + open_existing, prune_options, repository_options, run_blocking, unopened, }; use crate::filesystem_snapshot::contract_tests::fixture::{ Scratch, Spec, fixture, listing, write_tree, }; use crate::filesystem_snapshot::contract_tests::new_scope; use crate::filesystem_snapshot::{SnapshotName, SnapshotScope}; +use crate::services::golem_config::DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE as STORAGE_CALL_DEADLINE; use anyhow::Context; use async_trait::async_trait; use bytes::Bytes; From 56157e3fafa6cc1921bc05e3feb63cbc39e16461 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:16:20 -0700 Subject: [PATCH 11/55] Open the repository of the writer that makes the config file first --- .../src/filesystem_snapshot/rustic/mod.rs | 23 ++++++++++++------- 1 file changed, 15 insertions(+), 8 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 5fe655745c..f6bf801017 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -604,24 +604,31 @@ fn forget( } /// Opens the repository, or makes it when the scope has none. +/// +/// When another writer makes the config file of the repository first, the call opens the +/// repository of that writer. fn open_or_create( backend: Arc, key: &RepositoryKey, settings: &RepositorySettings, ) -> RusticResult<(RusticRepository, OperationPhase)> { - let repository = unopened(backend)?; + let repository = unopened(backend.clone())?; let credentials = Credentials::Masterkey(key.master_key()); match repository.config_id()? { Some(_) => repository .open(&credentials) .map(|repository| (repository, OperationPhase::Open)), - None => repository - .init( - &credentials, - &KeyOptions::default(), - &config_options(settings), - ) - .map(|repository| (repository, OperationPhase::Create)), + None => match repository.init( + &credentials, + &KeyOptions::default(), + &config_options(settings), + ) { + Ok(repository) => Ok((repository, OperationPhase::Create)), + Err(error) if fault::is_config_exists(&*error) => unopened(backend)? + .open(&credentials) + .map(|repository| (repository, OperationPhase::Open)), + Err(error) => Err(error), + }, } } From a9f8ea192a55fddcbc386fbd3a81c402d618b371 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:16:20 -0700 Subject: [PATCH 12/55] Add the rustic filesystem snapshot store --- .../src/filesystem_snapshot/mod.rs | 2 + .../src/filesystem_snapshot/rustic/backend.rs | 20 +- .../src/filesystem_snapshot/rustic/fault.rs | 32 + .../src/filesystem_snapshot/rustic/mod.rs | 43 +- .../src/filesystem_snapshot/rustic/store.rs | 706 +++++++++++ .../filesystem_snapshot/rustic/store/tests.rs | 1116 +++++++++++++++++ 6 files changed, 1904 insertions(+), 15 deletions(-) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/store.rs create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/mod.rs b/golem-worker-executor/src/filesystem_snapshot/mod.rs index 9f1abd7681..b4354f2a47 100644 --- a/golem-worker-executor/src/filesystem_snapshot/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/mod.rs @@ -36,6 +36,8 @@ mod time_zone_tests; #[allow(unused_imports)] pub(crate) use memory::InMemorySnapshotStore; +#[allow(unused_imports)] +pub(crate) use rustic::RusticSnapshotStore; /// The place of the filesystem snapshots of one agent. /// diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index 8d99fc90f5..0c64a04aeb 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -19,7 +19,7 @@ //! and each call waits for the blob storage on the runtime that the backend holds. Each call waits //! for at most a deadline, and a cancelled operation makes no more calls. -use super::fault::{BlobCallFailed, ConfigExists, OperationCancelled}; +use super::fault::{BlobCallFailed, ConfigExists, FileMissing, OperationCancelled}; use super::publish::{SnapshotStage, StagedSnapshot}; use bytes::Bytes; use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace, PutIfAbsent}; @@ -32,6 +32,7 @@ use std::sync::Arc; use std::time::Duration; use tokio::runtime::Handle; use tokio_util::sync::CancellationToken; +use tokio_util::task::task_tracker::TaskTrackerToken; /// The target label of each blob storage call of the backend. const TARGET_LABEL: &str = "filesystem_snapshot"; @@ -91,6 +92,8 @@ pub(super) struct BlobBackend { deadline: Duration, cancel: CancellationToken, stage: Option>, + /// Counts the backend as work of a tracker, until the last owner drops the backend. + _tracked: Option, } impl BlobBackend { @@ -109,18 +112,17 @@ impl BlobBackend { deadline, cancel: CancellationToken::new(), stage: None, + _tracked: None, } } /// Gives the backend with the token of its operation. When the token is cancelled, a call that /// has not started gives an error at once, and a call that runs stops and gives an error. - #[cfg_attr(not(test), allow(dead_code))] pub(super) fn cancelled_by(self, cancel: CancellationToken) -> Self { Self { cancel, ..self } } /// Gives the backend with a stage for the snapshot file of a save. - #[cfg_attr(not(test), allow(dead_code))] pub(super) fn staging_in(self, stage: Arc) -> Self { Self { stage: Some(stage), @@ -128,6 +130,15 @@ impl BlobBackend { } } + /// Gives the backend with a token of a task tracker. The tracker counts the backend until the + /// last owner drops it, for example a thread of rustic. + pub(super) fn tracked_by(self, token: TaskTrackerToken) -> Self { + Self { + _tracked: Some(token), + ..self + } + } + /// Waits for one call on the blob storage, and gives its result as a rustic result. /// /// Each call of the backend on the blob storage goes through this function. A call that gives @@ -398,9 +409,10 @@ fn join(parts: &[Bytes]) -> Box<[u8]> { /// The error of a file that the blob storage does not hold. fn missing_file(path: &Path) -> Box { - RusticError::new( + RusticError::with_source( ErrorKind::Backend, "The blob storage holds no file at `{path}`.", + FileMissing, ) .attach_context("path", path.display().to_string()) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs index 95a73d4718..0c273adf74 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs @@ -72,6 +72,24 @@ impl Display for ConfigExists { impl Error for ConfigExists {} +/// The blob storage holds no file at the path that rustic reads, for example because a delete +/// removed it after a listing. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct FileMissing; + +impl Display for FileMissing { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + formatter.write_str("the blob storage holds no file at the path") + } +} + +impl Error for FileMissing {} + +/// Tells whether an error in the chain is [`FileMissing`]. +pub(super) fn is_file_missing(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| error.is::()) +} + /// Tells whether an error in the chain is [`ConfigExists`]. pub(super) fn is_config_exists(error: &(dyn Error + 'static)) -> bool { chain(error).any(|error| error.is::()) @@ -123,6 +141,20 @@ pub(super) fn classify(operation: Operation, error: anyhow::Error) -> SnapshotSt } } +/// Tells whether an error in the chain is a failed blob storage call. +pub(super) fn is_storage_failure(error: &(dyn Error + 'static)) -> bool { + chain(error).any(|error| error.is::()) +} + +/// Gives the error of the store for a blob storage call that the store made without rustic. It is +/// retryable unless a name error of the blob storage caused it. +pub(super) fn storage_failure(error: anyhow::Error) -> SnapshotStoreError { + classify( + Operation::Repository, + anyhow::Error::new(BlobCallFailed::new(error)), + ) +} + /// Gives the text of the error with the text of each of its sources. fn error_text(error: &anyhow::Error) -> String { format!("{error:#}") diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index f6bf801017..6be31c2ff0 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -19,14 +19,13 @@ //! module. mod backend; -#[cfg_attr(not(test), allow(dead_code))] mod fault; -#[cfg_attr(not(test), allow(dead_code))] mod prune; -#[cfg_attr(not(test), allow(dead_code))] mod publish; -#[cfg_attr(not(test), allow(dead_code))] mod scope; +mod store; + +pub(crate) use store::RusticSnapshotStore; #[cfg(test)] mod holding; @@ -512,29 +511,51 @@ fn restore( let Some(snapshot) = snapshot else { return Ok(None); }; + let restored = restore_snapshot( + repository, + &snapshot, + into, + &RestoreOptions::default().reader_threads(reader_threads), + )?; + Ok(Some(RestoreReport { + phases: [open, lookup] + .into_iter() + .chain(restored.phases.iter().copied()) + .collect(), + ..restored + })) +} + +/// Writes the tree of the snapshot into the empty directory `into`. The phases of the result are +/// the index load, the plan and the writes. +fn restore_snapshot( + repository: RusticRepository, + snapshot: &SnapshotFile, + into: &Path, + options: &RestoreOptions, +) -> anyhow::Result { let (repository, index) = timed(OperationPhase::IndexLoad, || repository.to_indexed())?; let into = into .to_str() .context("the directory of a restore must have a UTF-8 path")?; let destination = LocalDestination::new(into, false, false)?; - let node = repository.node_from_snapshot_and_path(&snapshot, "")?; + let node = repository.node_from_snapshot_and_path(snapshot, "")?; let entries = repository.ls(&node, &LsOptions::default())?; - let options = RestoreOptions::default().reader_threads(reader_threads); let (plan, planning) = timed(OperationPhase::RestorePlan, || { - repository.prepare_restore(&options, entries.clone(), &destination, false) + repository.prepare_restore(options, entries.clone(), &destination, false) })?; let files = plan.stats.files.restore; let dirs = plan.stats.dirs.restore; let bytes = plan.restore_size; let ((), writing) = timed(OperationPhase::Restore, || { - repository.restore(plan, &options, entries, &destination) + repository.restore(plan, options, entries, &destination) })?; - Ok(Some(RestoreReport { + Ok(RestoreReport { files, dirs, bytes, - phases: Box::new([open, lookup, index, planning, writing]), - })) + phases: Box::new([index, planning, writing]), + }) } fn prune( diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs new file mode 100644 index 0000000000..705d2de252 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -0,0 +1,706 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The filesystem snapshot store over the rustic repositories of the scopes. +//! +//! A save makes the snapshot visible in one step: the backend keeps the snapshot file of the +//! backup, and the save writes that file after the blocking work returns. Each operation has a +//! cancellation token that its drop cancels, so the threads of a dropped operation stop at their +//! next storage call. The store counts each blocking task and each backend in a task tracker, and +//! [`RusticSnapshotStore::shut_down`] waits for them. + +use super::backend::BlobBackend; +use super::fault::{Operation, classify, is_file_missing, is_storage_failure, storage_failure}; +use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; +use super::publish::{SnapshotFiles, SnapshotStage, StagedSnapshot, publish}; +use super::scope::{copy_scope, delete_scope}; +use super::{ + ChangeDetection, PruneSettings, RepackLimits, RepositoryKey, RepositorySettings, SaveSettings, + backup_options, open_existing, open_or_create, prune, restore_snapshot, run_blocking, +}; +use crate::filesystem_snapshot::{ + FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, + newest_first, snapshot_time, +}; +use crate::services::golem_config::FilesystemSnapshotStoreConfig; +use anyhow::Context; +use async_trait::async_trait; +use golem_common::model::Timestamp; +use golem_service_base::storage::blob::BlobStorage; +use rustic_core::jiff::tz::TimeZone; +use rustic_core::jiff::{Timestamp as SnapshotTime, Zoned}; +use rustic_core::repofile::{SnapshotFile, SnapshotId}; +use rustic_core::{ + BackupOptions, DevIdOption, LocalSourceSaveOptions, Open, PathList, + Repository as RusticRepository, RestoreOptions, SnapshotOptions, +}; +use serde::{Deserialize, Serialize}; +use std::num::NonZeroUsize; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; +use tokio::runtime::Handle; +use tokio_util::sync::{CancellationToken, DropGuard}; +use tokio_util::task::TaskTracker; + +/// The packed bytes that deleted snapshots must free before a delete prunes the scope. +const PRUNE_THRESHOLD_BYTES: u64 = 64 * 1024 * 1024; + +/// How long a pack that a prune marks stays before a later prune deletes it. It is also the +/// shortest time between two prunes of one scope. It must be longer than the longest save and the +/// longest restore. +const PRUNE_GRACE: Duration = Duration::from_secs(3600); + +/// The settings of the store: the rustic settings of each operation, and the prune threshold. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) struct StorePolicy { + /// The longest time that one blob storage call waits for an answer. + pub(super) deadline: Duration, + /// The settings of a repository that a save makes. + pub(super) repository: RepositorySettings, + pub(super) save: SaveSettings, + /// The number of threads that read packs in a restore. + pub(super) restore_reader_threads: NonZeroUsize, + /// The settings of a prune. `keep_delete` is also the shortest time between two prunes. + pub(super) prune: PruneSettings, + /// The packed bytes that deleted snapshots must free before a delete prunes. + pub(super) prune_threshold: u64, +} + +impl StorePolicy { + /// Gives the policy with the values of the configuration. + pub(super) fn from_config(config: &FilesystemSnapshotStoreConfig) -> Self { + Self { + deadline: config.storage_call_deadline(), + repository: RepositorySettings::DEFAULT, + save: SaveSettings { + threads: Some(config.save_threads()), + detection: ChangeDetection::Ctime, + }, + restore_reader_threads: config.restore_reader_threads(), + prune: PruneSettings { + fast_repack: false, + keep_delete: PRUNE_GRACE, + repack: RepackLimits::Rustic, + }, + prune_threshold: PRUNE_THRESHOLD_BYTES, + } + } +} + +/// The options of a save of the store: the options of the bridge, and a save that cannot read an +/// entry fails before it writes the snapshot file. A save records no device id, so a restore gives +/// each name of a hard-linked file as its own file. +fn store_backup_options(policy: &StorePolicy) -> BackupOptions { + backup_options(&policy.save) + .fail_on_read_error(true) + .ignore_save_opts(LocalSourceSaveOptions::default().set_devid(DevIdOption::No)) +} + +/// The options of a restore of the store. A metadata error fails the restore. The restore does not +/// set the owner, because a snapshot does not keep it. +fn store_restore_options(policy: &StorePolicy) -> RestoreOptions { + RestoreOptions::default() + .reader_threads(Some(policy.restore_reader_threads)) + .fail_on_metadata_error(true) + .no_ownership(true) +} + +/// A filesystem snapshot store that keeps one rustic repository for each scope in blob storage. +pub(crate) struct RusticSnapshotStore { + storage: Arc, + key: RepositoryKey, + policy: StorePolicy, + /// The parent of the token of each operation. + root: CancellationToken, + /// Counts the blocking tasks, the backends and the deletes of dropped publishes. + tracker: TaskTracker, +} + +impl RusticSnapshotStore { + /// Gives the store over the blob storage, with the key and the values of the configuration. + pub(crate) fn new( + storage: Arc, + config: &FilesystemSnapshotStoreConfig, + ) -> Self { + Self::with_policy( + storage, + RepositoryKey::new(*config.repository_key().bytes()), + StorePolicy::from_config(config), + ) + } + + pub(super) fn with_policy( + storage: Arc, + key: RepositoryKey, + policy: StorePolicy, + ) -> Self { + Self { + storage, + key, + policy, + root: CancellationToken::new(), + tracker: TaskTracker::new(), + } + } + + /// Stops each operation at its next storage call, and waits until no blocking task and no + /// backend of the store remains. After the call, each operation gives `Storage`. + /// + /// The runtime must not drop before the call returns, because a storage call that waits on the + /// runtime after its time driver stops aborts the process. + pub(crate) async fn shut_down(&self) { + self.root.cancel(); + self.tracker.close(); + self.tracker.wait().await; + } + + /// Gives the number of blocking tasks, backends and deletes of the store that have not ended. + #[cfg(test)] + pub(super) fn work_in_flight(&self) -> usize { + self.tracker.len() + } + + /// Starts an operation. The token of the operation is cancelled when the guard drops. + fn start(&self) -> Result<(CancellationToken, DropGuard), SnapshotStoreError> { + if self.root.is_cancelled() { + return Err(SnapshotStoreError::Storage { + retryable: false, + source: anyhow::anyhow!("the filesystem snapshot store is shut down"), + }); + } + let token = self.root.child_token(); + Ok((token.clone(), token.drop_guard())) + } + + /// Gives a backend over the repository of the scope for the operation with the token. + fn backend( + &self, + scope: &SnapshotScope, + token: &CancellationToken, + ) -> Result { + let runtime = Handle::try_current() + .context("a filesystem snapshot operation needs an async runtime") + .map_err(|source| SnapshotStoreError::Storage { + retryable: false, + source, + })?; + Ok(BlobBackend::new( + self.storage.clone(), + scope.0.clone(), + runtime, + self.policy.deadline, + ) + .cancelled_by(token.clone()) + .tracked_by(self.tracker.token())) + } + + /// Runs the task on a blocking thread that the tracker counts, and classifies its error. + async fn blocking( + &self, + operation: Operation, + task: impl FnOnce() -> anyhow::Result + Send + 'static, + ) -> Result { + let tracked = self.tracker.token(); + run_blocking(move || { + let _tracked = tracked; + task() + }) + .await + .map_err(|error| classify(operation, error)) + } + + fn files(&self, scope: &SnapshotScope) -> SnapshotFiles { + SnapshotFiles { + storage: self.storage.clone(), + namespace: scope.0.clone(), + deadline: self.policy.deadline, + } + } + + /// Adds the freed bytes to the ledger of the scope, and prunes the repository when a prune is + /// due. The ledger keeps the freed bytes before the prune starts, so a delete that runs again + /// after a failed prune prunes again. + async fn prune_when_due( + &self, + scope: &SnapshotScope, + token: &CancellationToken, + freed: u64, + ) -> Result<(), SnapshotStoreError> { + let deadline = self.policy.deadline; + let ledger = read_ledger(&*self.storage, &scope.0, deadline) + .await + .map_err(storage_failure)? + .with_deleted(freed); + if freed > 0 { + write_ledger(&*self.storage, &scope.0, deadline, &ledger) + .await + .map_err(storage_failure)?; + } + let now = Timestamp::now_utc(); + if !prune_due( + &ledger, + now, + self.policy.prune_threshold, + self.policy.prune.keep_delete, + ) { + return Ok(()); + } + let backend = Arc::new(self.backend(scope, token)?); + let key = self.key.clone(); + let settings = self.policy.prune; + let report = self + .blocking(Operation::Repository, move || { + prune(backend, &key, &settings) + }) + .await?; + let marked_packs = report.is_some_and(|report| { + report.packs_unused + report.packs_repacked + report.marked_packs_kept > 0 + }); + write_ledger( + &*self.storage, + &scope.0, + deadline, + &PruneLedger::after_prune(now, marked_packs), + ) + .await + .map_err(storage_failure) + } +} + +#[async_trait] +impl FilesystemSnapshotStore for RusticSnapshotStore { + async fn save( + &self, + scope: &SnapshotScope, + name: &SnapshotName, + tree: &Path, + ) -> Result { + let (token, _guard) = self.start()?; + check_tree(tree).await?; + let stage = Arc::new(SnapshotStage::default()); + let backend = Arc::new(self.backend(scope, &token)?.staging_in(stage.clone())); + let key = self.key.clone(); + let policy = self.policy; + let name = name.clone(); + let tree: Box = tree.into(); + let staged = self + .blocking(Operation::Save, move || { + stage_save(backend, &stage, &key, &policy, &name, &tree) + }) + .await?; + let (staged, info) = staged.ok_or(SnapshotStoreError::AlreadyExists)?; + publish(&self.files(scope), &staged, &self.tracker) + .await + .map_err(storage_failure)?; + Ok(info) + } + + async fn restore( + &self, + scope: &SnapshotScope, + name: &SnapshotName, + into: &Path, + ) -> Result { + let (token, _guard) = self.start()?; + check_destination(into).await?; + let backend = Arc::new(self.backend(scope, &token)?); + let key = self.key.clone(); + let options = store_restore_options(&self.policy); + let name = name.clone(); + let into: Box = into.into(); + self.blocking(Operation::Restore, move || { + let Some(repository) = open_existing(backend, &key)? else { + return Ok(Lookup::Missing); + }; + match lookup(scope_snapshots(&repository)?, &name) { + Lookup::Found(snapshot, info) => { + restore_snapshot(repository, &snapshot, &into, &options)?; + Ok(Lookup::Found(snapshot, info)) + } + other => Ok(other), + } + }) + .await? + .into_info()? + .ok_or(SnapshotStoreError::NotFound) + } + + async fn stat( + &self, + scope: &SnapshotScope, + name: &SnapshotName, + ) -> Result, SnapshotStoreError> { + let (token, _guard) = self.start()?; + let backend = Arc::new(self.backend(scope, &token)?); + let key = self.key.clone(); + let name = name.clone(); + self.blocking(Operation::Repository, move || { + Ok(match open_existing(backend, &key)? { + Some(repository) => lookup(scope_snapshots(&repository)?, &name), + None => Lookup::Missing, + }) + }) + .await? + .into_info() + } + + async fn list( + &self, + scope: &SnapshotScope, + ) -> Result, SnapshotStoreError> { + let (token, _guard) = self.start()?; + let backend = Arc::new(self.backend(scope, &token)?); + let key = self.key.clone(); + self.blocking(Operation::Repository, move || { + Ok(match open_existing(backend, &key)? { + Some(repository) => newest_first(listed(scope_snapshots(&repository)?)), + None => Box::default(), + }) + }) + .await + } + + async fn delete( + &self, + scope: &SnapshotScope, + name: &SnapshotName, + ) -> Result<(), SnapshotStoreError> { + let (token, _guard) = self.start()?; + let backend = Arc::new(self.backend(scope, &token)?); + let key = self.key.clone(); + let name = name.clone(); + let freed = self + .blocking(Operation::Repository, move || { + let Some(repository) = open_existing(backend, &key)? else { + return Ok(None); + }; + let named = scope_snapshots(&repository)? + .readable + .into_iter() + .filter(|snapshot| snapshot.label == name.as_str()) + .collect::>(); + let ids = named + .iter() + .map(|snapshot| snapshot.id) + .collect::>(); + repository.delete_snapshots(&ids)?; + Ok(Some(named.iter().map(added_packed_bytes).sum::())) + }) + .await?; + match freed { + Some(freed) => self.prune_when_due(scope, &token, freed).await, + None => Ok(()), + } + } + + async fn delete_scope(&self, scope: &SnapshotScope) -> Result<(), SnapshotStoreError> { + let _operation = self.start()?; + delete_scope(&*self.storage, &scope.0, self.policy.deadline) + .await + .map_err(storage_failure) + } + + async fn copy_scope( + &self, + from: &SnapshotScope, + to: &SnapshotScope, + ) -> Result<(), SnapshotStoreError> { + let _operation = self.start()?; + copy_scope(&*self.storage, &from.0, &to.0, self.policy.deadline) + .await + .map_err(storage_failure) + } +} + +/// What the tree of a snapshot holds. The store keeps it as the description of the snapshot, +/// because rustic counts a symlink as a file. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] +struct TreeContent { + files: u64, + bytes: u64, +} + +/// The snapshot files of a scope that rustic could read, and whether a file failed its integrity +/// check. A file that a delete removed after the listing is in neither. +struct ScopeSnapshots { + readable: Vec, + unreadable: bool, +} + +/// What a lookup of a name found. +enum Lookup { + Found(Box, SnapshotInfo), + Missing, + Corrupt(anyhow::Error), +} + +impl Lookup { + fn into_info(self) -> Result, SnapshotStoreError> { + match self { + Self::Found(_, info) => Ok(Some(info)), + Self::Missing => Ok(None), + Self::Corrupt(error) => Err(SnapshotStoreError::Corrupt(error)), + } + } +} + +/// Checks that the tree of a save is a directory at an absolute path. +async fn check_tree(tree: &Path) -> Result<(), SnapshotStoreError> { + let metadata = tokio::fs::metadata(tree) + .await + .map_err(SnapshotStoreError::Source)?; + if !tree.is_absolute() || !metadata.is_dir() { + return Err(SnapshotStoreError::Source(std::io::Error::new( + std::io::ErrorKind::NotADirectory, + format!( + "the tree {} is not a directory at an absolute path", + tree.display() + ), + ))); + } + Ok(()) +} + +/// Checks that the directory of a restore is an empty directory with a UTF-8 path. +async fn check_destination(into: &Path) -> Result<(), SnapshotStoreError> { + let refused = |kind, reason: &str| { + SnapshotStoreError::Destination(std::io::Error::new( + kind, + format!("the directory {} {reason}", into.display()), + )) + }; + let metadata = tokio::fs::metadata(into) + .await + .map_err(SnapshotStoreError::Destination)?; + if !metadata.is_dir() { + return Err(refused( + std::io::ErrorKind::NotADirectory, + "is not a directory", + )); + } + if into.to_str().is_none() { + return Err(refused( + std::io::ErrorKind::InvalidInput, + "does not have a UTF-8 path", + )); + } + let first = tokio::fs::read_dir(into) + .await + .map_err(SnapshotStoreError::Destination)? + .next_entry() + .await + .map_err(SnapshotStoreError::Destination)?; + match first { + Some(_) => Err(refused( + std::io::ErrorKind::DirectoryNotEmpty, + "is not empty", + )), + None => Ok(()), + } +} + +/// Backs up the tree with the snapshot file in the stage, and gives the staged file with the info +/// of the snapshot. The result is `None` when a snapshot of the scope already has the name. +fn stage_save( + backend: Arc, + stage: &SnapshotStage, + key: &RepositoryKey, + policy: &StorePolicy, + name: &SnapshotName, + tree: &Path, +) -> anyhow::Result> { + let (repository, _) = open_or_create(backend, key, &policy.repository)?; + let before = scope_snapshots(&repository)?; + if has_name(&before, name) { + return Ok(None); + } + let newest = before + .readable + .iter() + .filter_map(snapshot_info) + .map(|info| info.created_at) + .max(); + let created_at = snapshot_time(whole_millis_from(Timestamp::now_utc()), newest); + // rustic strips the root from the path of each entry. The canonical root is the path that the + // walk of rustic gives, also when the caller gives a path through a symlink such as + // `/proc/self/fd/N`. + let tree = std::fs::canonicalize(tree)?; + let content = tree_content(&tree)?; + let snapshot = SnapshotOptions::default() + .label(name.as_str().to_string()) + .time(snapshot_zoned(created_at)?) + .description(serde_json::to_string(&content)?) + .to_snapshot()?; + let repository = repository.to_indexed_ids()?; + repository.backup( + &store_backup_options(policy), + &PathList::from_iter(Some(tree)), + snapshot, + )?; + let staged = stage + .take() + .context("the backup gave no snapshot file to the stage")?; + if has_name(&scope_snapshots(&repository)?, name) { + return Ok(None); + } + Ok(Some(( + staged, + SnapshotInfo { + created_at, + files: content.files, + bytes: content.bytes, + }, + ))) +} + +/// Reads each snapshot file of the repository. +/// +/// A failed storage call fails the read. A file that the storage no longer holds is left out, +/// because a delete removed it after the listing. Each other failure counts as a file that failed +/// its integrity check. +fn scope_snapshots(repository: &RusticRepository) -> anyhow::Result { + repository.list::()?.try_fold( + ScopeSnapshots { + readable: Vec::new(), + unreadable: false, + }, + |mut found, id| match repository.get_file::(&id) { + Ok(mut snapshot) => { + snapshot.id = id; + found.readable.push(snapshot); + Ok(found) + } + Err(error) if is_storage_failure(&*error) => Err(anyhow::Error::from(error)), + Err(error) if is_file_missing(&*error) => Ok(found), + Err(_) => Ok(ScopeSnapshots { + unreadable: true, + ..found + }), + }, + ) +} + +fn has_name(found: &ScopeSnapshots, name: &SnapshotName) -> bool { + found + .readable + .iter() + .any(|snapshot| snapshot.label == name.as_str()) +} + +/// Finds the snapshot with the name. Of the snapshot files with the name, the one with the least +/// time and id wins. When no file has the name and a file failed its integrity check, the result is +/// `Corrupt`, because that file can have the name. +fn lookup(found: ScopeSnapshots, name: &SnapshotName) -> Lookup { + let winner = found + .readable + .into_iter() + .filter(|snapshot| snapshot.label == name.as_str()) + .min_by_key(|snapshot| (snapshot.time.timestamp(), snapshot.id)); + match (winner, found.unreadable) { + (Some(snapshot), _) => match snapshot_info(&snapshot) { + Some(info) => Lookup::Found(Box::new(snapshot), info), + None => Lookup::Corrupt(anyhow::anyhow!( + "the snapshot {} does not describe its tree", + snapshot.id + )), + }, + (None, true) => Lookup::Corrupt(anyhow::anyhow!( + "a snapshot file of the scope failed its integrity check" + )), + (None, false) => Lookup::Missing, + } +} + +/// Gives the name and the info of each snapshot whose label is a name and whose description +/// parses. Of the files with one name, only the one that [`lookup`] takes stays. +fn listed(found: ScopeSnapshots) -> Vec<(SnapshotName, SnapshotInfo)> { + let mut snapshots = found.readable; + snapshots.sort_by_key(|snapshot| { + ( + snapshot.label.clone(), + snapshot.time.timestamp(), + snapshot.id, + ) + }); + snapshots.dedup_by(|later, first| later.label == first.label); + snapshots + .iter() + .filter_map(|snapshot| { + SnapshotName::new(&snapshot.label) + .ok() + .zip(snapshot_info(snapshot)) + }) + .collect() +} + +/// Gives the info of a snapshot from its time and its description. +fn snapshot_info(snapshot: &SnapshotFile) -> Option { + let content = serde_json::from_str::(snapshot.description.as_deref()?).ok()?; + let created_at = u64::try_from(snapshot.time.timestamp().as_millisecond()).ok()?; + Some(SnapshotInfo { + created_at: Timestamp::from(created_at), + files: content.files, + bytes: content.bytes, + }) +} + +/// Gives the packed bytes that the save of the snapshot added to the repository. +fn added_packed_bytes(snapshot: &SnapshotFile) -> u64 { + snapshot + .summary + .as_ref() + .map_or(0, |summary| summary.data_added_packed) +} + +/// Gives the first time in whole milliseconds that is not before the time. A snapshot keeps its +/// time in milliseconds, so the time of a save is not before the call. +fn whole_millis_from(time: Timestamp) -> Timestamp { + let truncated = Timestamp::from(time.to_millis()); + if truncated < time { + Timestamp::from(time.to_millis().saturating_add(1)) + } else { + truncated + } +} + +/// Gives the time as the time of a snapshot, in UTC. +fn snapshot_zoned(time: Timestamp) -> anyhow::Result { + Ok(SnapshotTime::from_millisecond(i64::try_from(time.to_millis())?)?.to_zoned(TimeZone::UTC)) +} + +/// Counts the names of the regular files below the root, and the sum of their sizes. The walk +/// reads the metadata of each entry and does not follow a symlink. +fn tree_content(root: &Path) -> std::io::Result { + fn walk(directory: &Path, content: TreeContent) -> std::io::Result { + std::fs::read_dir(directory)?.try_fold(content, |content, entry| { + let entry = entry?; + let kind = entry.file_type()?; + if kind.is_dir() { + walk(&entry.path(), content) + } else if kind.is_file() { + Ok(TreeContent { + files: content.files + 1, + bytes: content.bytes + entry.metadata()?.len(), + }) + } else { + Ok(content) + } + }) + } + walk(root, TreeContent { files: 0, bytes: 0 }) +} + +#[cfg(test)] +mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs new file mode 100644 index 0000000000..c6d7c85b16 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -0,0 +1,1116 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The rustic store through the interface of the store, on the in-memory blob storage. +//! +//! The contract suite runs on the store with the policy of the configuration. The other tests +//! give the store a short or a long deadline and a prune policy that the test controls. + +use super::super::prune::{PruneLedger, read_ledger}; +use super::super::scripted::{Script, ScriptedBlobStorage}; +use super::super::{PruneSettings, RepackLimits, RepositoryKey, open_existing}; +use super::{ + RusticSnapshotStore, StorePolicy, scope_snapshots, store_backup_options, store_restore_options, + whole_millis_from, +}; +use crate::filesystem_snapshot::contract_tests::fixture::{ + Listed, Scratch, Spec, fixture, listing, write_tree, +}; +use crate::filesystem_snapshot::contract_tests::{self, OpenStore, new_scope}; +use crate::filesystem_snapshot::{ + FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, +}; +use crate::services::golem_config::FilesystemSnapshotStoreConfig; +use futures::{FutureExt, StreamExt}; +use golem_service_base::storage::blob::memory::InMemoryBlobStorage; +use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; +use pretty_assertions::assert_eq; +use std::future::Future; +use std::num::NonZeroUsize; +use std::path::Path; +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; +use std::time::Duration; +use test_r::core::DynamicTestRegistration; +use test_r::{test, test_gen}; + +/// The longest time that a test waits for an operation or for the work of a store to end. +const LIMIT: Duration = Duration::from_secs(10); + +/// A deadline that no call of these tests reaches, so a held call ends only by a cancel. +const LONG_DEADLINE: Duration = Duration::from_secs(60); + +const KEY: &str = "000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f\ + 202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f"; + +fn config() -> FilesystemSnapshotStoreConfig { + FilesystemSnapshotStoreConfig::new(KEY, Duration::from_secs(30), 4, 3).unwrap() +} + +fn key() -> RepositoryKey { + RepositoryKey::new(*config().repository_key().bytes()) +} + +/// The policy of the configuration, with the deadline, the prune threshold and the grace period +/// of the test. +fn policy(deadline: Duration, prune_threshold: u64, grace: Duration) -> StorePolicy { + StorePolicy { + deadline, + prune: PruneSettings { + keep_delete: grace, + ..StorePolicy::from_config(&config()).prune + }, + prune_threshold, + ..StorePolicy::from_config(&config()) + } +} + +fn store(storage: Arc, policy: StorePolicy) -> Arc { + Arc::new(RusticSnapshotStore::with_policy(storage, key(), policy)) +} + +fn name(text: &str) -> SnapshotName { + SnapshotName::new(text).unwrap() +} + +/// Writes a tree of one file with the content into a new directory, and gives the directory. +fn one_file_tree(content: &str) -> Scratch { + let tree = Scratch::new(); + write_tree( + tree.path(), + &[( + "file.txt", + Spec::File { + content: Box::from(content.as_bytes()), + mode: 0o644, + }, + )], + ); + tree +} + +fn fixture_tree() -> Scratch { + let tree = Scratch::new(); + write_tree(tree.path(), &fixture()); + tree +} + +/// Restores the name into a new directory, and gives the listing of the directory. +async fn restored_listing( + store: &RusticSnapshotStore, + scope: &SnapshotScope, + name: &SnapshotName, +) -> Result, SnapshotStoreError> { + let into = Scratch::new(); + store.restore(scope, name, into.path()).await?; + Ok(listing(into.path())) +} + +async fn listed_names(store: &RusticSnapshotStore, scope: &SnapshotScope) -> Vec { + store + .list(scope) + .await + .unwrap() + .iter() + .map(|(name, _)| name.to_string()) + .collect() +} + +/// Gives the path of each blob of the namespace whose path starts with the prefix. +async fn blobs( + storage: &dyn BlobStorage, + namespace: &BlobStorageNamespace, + prefix: &str, +) -> Vec { + let mut paths = storage + .list_blobs_below("test", "test", namespace.clone(), Path::new("")) + .await + .unwrap() + .iter() + .map(|blob| blob.path.display().to_string()) + .filter(|path| path.starts_with(prefix)) + .collect::>(); + paths.sort(); + paths +} + +async fn ledger(storage: &dyn BlobStorage, scope: &SnapshotScope) -> PruneLedger { + read_ledger(storage, &scope.0, Duration::from_secs(2)) + .await + .unwrap() +} + +/// Waits until the condition holds, or until the limit ends. Gives whether the condition holds. +async fn eventually(condition: impl Fn() -> bool) -> bool { + tokio::time::timeout(LIMIT, async { + futures::stream::repeat(()) + .then(|()| tokio::time::sleep(Duration::from_millis(5))) + .take_while(|()| std::future::ready(!condition())) + .for_each(|()| std::future::ready(())) + .await + }) + .await + .is_ok() +} + +/// Runs the operation until the calls of the storage fulfil the condition, and then drops it. +/// Gives the output of the operation when it ends first. +async fn drop_when( + storage: &ScriptedBlobStorage, + condition: impl Fn(&[(&'static str, String)]) -> bool, + operation: impl Future, +) -> Option { + tokio::select! { + output = operation => Some(output), + _ = eventually(|| condition(&storage.calls())) => None, + } +} + +fn is_storage(error: &SnapshotStoreError, expected_retryable: bool) -> bool { + matches!(error, SnapshotStoreError::Storage { retryable, .. } if *retryable == expected_retryable) +} + +#[test_gen] +fn rustic_store_keeps_the_contract(r: &mut DynamicTestRegistration) { + contract_tests::register(r, || { + let storage: Arc = Arc::new(InMemoryBlobStorage::new()); + let open: OpenStore = Arc::new(move || { + Arc::new(RusticSnapshotStore::new(storage.clone(), &config())) + as Arc + }); + open + }); +} + +#[test] +fn the_policy_takes_the_configured_values_and_the_options_are_strict() { + let policy = StorePolicy::from_config(&config()); + let backup = store_backup_options(&policy); + let restore = store_restore_options(&policy); + + assert_eq!( + ( + policy.deadline, + policy.save.threads.map(NonZeroUsize::get), + policy.restore_reader_threads.get(), + policy.prune.keep_delete, + policy.prune.fast_repack, + policy.prune.repack, + policy.prune_threshold, + ), + ( + Duration::from_secs(30), + Some(3), + 4, + Duration::from_secs(3600), + false, + RepackLimits::Rustic, + 64 * 1024 * 1024, + ) + ); + assert_eq!( + ( + backup.fail_on_read_error, + backup.threads.map(NonZeroUsize::get), + backup.as_path.as_deref().map(Path::to_path_buf), + restore.fail_on_metadata_error, + restore.no_ownership, + restore.reader_threads.map(NonZeroUsize::get), + ), + ( + true, + Some(3), + Some(Path::new("/").to_path_buf()), + true, + true, + Some(4), + ) + ); +} + +#[test] +fn a_save_time_is_the_first_whole_millisecond_that_is_not_before_the_call() { + let now = golem_common::model::Timestamp::now_utc(); + let whole = golem_common::model::Timestamp::from(5_000); + let rounded = whole_millis_from(now); + + assert_eq!( + ( + whole_millis_from(whole), + rounded >= now, + rounded.to_millis() - now.to_millis() <= 1, + golem_common::model::Timestamp::from(rounded.to_millis()) == rounded, + ), + (whole, true, true, true) + ); +} + +#[cfg(target_os = "linux")] +#[test] +async fn a_tree_saved_through_a_proc_self_fd_path_is_stored_below_the_root() { + use std::os::fd::AsRawFd; + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + let directory = std::fs::File::open(tree.path()).unwrap(); + let through_fd = format!("/proc/self/fd/{}", directory.as_raw_fd()); + + store + .save(&scope, &name("p-fd"), Path::new(&through_fd)) + .await + .unwrap(); + let namespace = scope.0.clone(); + let paths = tokio::task::spawn_blocking(move || { + let backend = super::super::backend::BlobBackend::new( + storage, + namespace, + tokio::runtime::Handle::current(), + LONG_DEADLINE, + ); + let repository = open_existing(Arc::new(backend), &key()).unwrap().unwrap(); + scope_snapshots(&repository) + .unwrap() + .readable + .iter() + .map(|snapshot| snapshot.paths.to_string()) + .collect::>() + }) + .await + .unwrap(); + + assert_eq!( + ( + paths, + restored_listing(&store, &scope, &name("p-fd")) + .await + .unwrap() + ), + (vec!["/".to_string()], listing(tree.path())) + ); +} + +#[test] +async fn a_save_whose_index_write_fails_publishes_nothing_and_leaves_the_name_free() { + let refuse = Arc::new(AtomicBool::new(true)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) && op_label == "write" && path.starts_with("index") { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = fixture_tree(); + + let failed = store.save(&scope, &name("p-1"), tree.path()).await; + let stat = store.stat(&scope, &name("p-1")).await.unwrap(); + let names = listed_names(&store, &scope).await; + let restore = restored_listing(&store, &scope, &name("p-1")).await; + refuse.store(false, Ordering::SeqCst); + let saved_again = store.save(&scope, &name("p-1"), tree.path()).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!( + matches!(restore, Err(SnapshotStoreError::NotFound)), + "{restore:?}" + ); + assert_eq!( + ( + stat, + names, + saved_again.is_ok(), + restored_listing(&store, &scope, &name("p-1")) + .await + .unwrap() + ), + (None, Vec::::new(), true, listing(tree.path())) + ); +} + +#[test] +async fn a_publish_that_reaches_the_deadline_and_lands_late_publishes_nothing() { + let hang = Arc::new(AtomicBool::new(true)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let hang = hang.clone(); + move |op_label, _| { + if hang.load(Ordering::SeqCst) && op_label == "publish" { + Script::NeverAnswer + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(Duration::from_secs(1), u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("late"); + + let failed = store.save(&scope, &name("p-late"), tree.path()).await; + hang.store(false, Ordering::SeqCst); + let stat = store.stat(&scope, &name("p-late")).await.unwrap(); + let names = listed_names(&store, &scope).await; + let saved_again = store.save(&scope, &name("p-late"), tree.path()).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert_eq!( + ( + stat, + names, + saved_again.is_ok(), + storage + .calls() + .iter() + .filter(|(op_label, _)| *op_label == "retract") + .count() + ), + (None, Vec::::new(), true, 1) + ); +} + +#[test] +async fn a_second_save_of_an_unchanged_tree_writes_no_pack() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + + store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + let first = storage.calls().len(); + store.save(&scope, &name("p-2"), tree.path()).await.unwrap(); + let writes = storage.calls()[first..] + .iter() + .filter(|(op_label, _)| *op_label == "write" || *op_label == "publish") + .map(|(op_label, path)| { + ( + *op_label, + path.split('/').next().unwrap_or_default().to_string(), + ) + }) + .collect::>(); + + let restored = restored_listing(&store, &scope, &name("p-2")).await.ok(); + + assert_eq!( + (writes, restored), + ( + vec![("publish", "snapshots".to_string())], + Some(listing(tree.path())) + ) + ); +} + +#[test] +async fn two_stores_that_create_one_repository_at_the_same_time_both_save() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "write" && path == Path::new("config") { + Script::WaitForGate + } else { + Script::Pass + } + }); + let policy = policy(LONG_DEADLINE, u64::MAX, Duration::ZERO); + let (first, second) = ( + store(storage.clone(), policy), + store(storage.clone(), policy), + ); + let scope = new_scope(); + let (first_tree, second_tree) = (one_file_tree("first"), one_file_tree("second")); + let config_writes = || { + storage + .calls() + .iter() + .filter(|(op_label, path)| *op_label == "write" && path == "config") + .count() + }; + + let (first_name, second_name) = (name("p-first"), name("p-second")); + let (first_saved, second_saved, both_waited) = tokio::join!( + first.save(&scope, &first_name, first_tree.path()), + second.save(&scope, &second_name, second_tree.path()), + async { + let both = eventually(|| config_writes() == 2).await; + storage.open_gate(); + both + } + ); + let mut names = listed_names(&first, &scope).await; + names.sort(); + + assert_eq!( + ( + both_waited, + first_saved.map(|_| ()).map_err(|error| error.to_string()), + second_saved.map(|_| ()).map_err(|error| error.to_string()), + names, + restored_listing(&second, &scope, &name("p-first")) + .await + .unwrap(), + restored_listing(&first, &scope, &name("p-second")) + .await + .unwrap(), + ), + ( + true, + Ok(()), + Ok(()), + vec!["p-first".to_string(), "p-second".to_string()], + listing(first_tree.path()), + listing(second_tree.path()), + ) + ); +} + +#[test] +async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { + // The first index write after the arm waits at the gate. That is the index write of the + // second save, so its packs are in no index while the delete prunes. + let hold_next_index = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let hold_next_index = hold_next_index.clone(); + move |op_label, path| { + if op_label == "write" + && path.starts_with("index") + && hold_next_index.swap(false, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let scope = new_scope(); + let (old_tree, new_tree) = (one_file_tree("old"), fixture_tree()); + store + .save(&scope, &name("p-old"), old_tree.path()) + .await + .unwrap(); + hold_next_index.store(true, Ordering::SeqCst); + let index_writes = || { + storage + .calls() + .iter() + .filter(|(op_label, path)| *op_label == "write" && path.starts_with("index")) + .count() + }; + let before = index_writes(); + + let saving = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + let path = new_tree.path().to_path_buf(); + async move { store.save(&scope, &name("p-new"), &path).await } + }); + let held = eventually(|| index_writes() > before).await; + let deleted = store.delete(&scope, &name("p-old")).await; + let pruned_while_held = ledger(&*storage, &scope).await.last_prune_millis.is_some(); + storage.open_gate(); + let saved = saving.await.unwrap(); + let pruned_again = store.delete(&scope, &name("p-none")).await; + + assert_eq!( + ( + held, + deleted.is_ok(), + pruned_while_held, + saved.is_ok(), + pruned_again.is_ok(), + restored_listing(&store, &scope, &name("p-new")).await.ok(), + ), + (true, true, true, true, true, Some(listing(new_tree.path()))) + ); +} + +#[test] +async fn a_restore_whose_pack_reads_fail_gives_a_retryable_storage_error() { + let refuse = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) + && matches!(op_label, "read" | "read_range") + && path.starts_with("data") + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = fixture_tree(); + store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + refuse.store(true, Ordering::SeqCst); + + let restored = restored_listing(&store, &scope, &name("p-1")).await; + + assert!( + restored + .as_ref() + .is_err_and(|error| is_storage(error, true)), + "{restored:?}" + ); +} + +#[cfg(unix)] +#[test] +async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_denied() { + use std::os::unix::fs::PermissionsExt; + // SAFETY: `geteuid` has no preconditions. + if unsafe { libc::geteuid() } == 0 { + return; + } + let store = store( + Arc::new(InMemoryBlobStorage::new()), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("readable"); + let locked = tree.path().join("locked.txt"); + std::fs::write(&locked, b"locked").unwrap(); + std::fs::set_permissions(&locked, std::fs::Permissions::from_mode(0o000)).unwrap(); + + let saved = store.save(&scope, &name("p-locked"), tree.path()).await; + + assert!( + matches!(&saved, Err(SnapshotStoreError::Source(error)) if error.kind() == std::io::ErrorKind::PermissionDenied), + "{saved:?}" + ); +} + +#[cfg(target_os = "linux")] +#[test] +async fn a_restore_that_cannot_set_an_extended_attribute_gives_destination() { + // A file on tmpfs takes a user attribute of 6,000 bytes. A file on ext4 with 4 KiB blocks + // does not, so the restore cannot set it. The test checks nothing on a host where the source + // does not take the attribute or where the destination takes it. + let value = vec![b'a'; 6000]; + let set = |path: &Path| xattr_set(path, "user.golem-test", &value); + let Ok(source) = tempfile::tempdir_in("/dev/shm") else { + return; + }; + let file = source.path().join("file.txt"); + std::fs::write(&file, b"content").unwrap(); + let probe = Scratch::new(); + let probe_file = probe.path().join("probe"); + std::fs::write(&probe_file, b"probe").unwrap(); + if set(&file).is_err() || set(&probe_file).is_ok() { + return; + } + let store = store( + Arc::new(InMemoryBlobStorage::new()), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + store + .save(&scope, &name("p-xattr"), source.path()) + .await + .unwrap(); + + let restored = restored_listing(&store, &scope, &name("p-xattr")).await; + + assert!( + matches!(&restored, Err(SnapshotStoreError::Destination(_))), + "{restored:?}" + ); +} + +#[cfg(target_os = "linux")] +fn xattr_set(path: &Path, name: &str, value: &[u8]) -> std::io::Result<()> { + use std::os::unix::ffi::OsStrExt; + let path = std::ffi::CString::new(path.as_os_str().as_bytes())?; + let name = std::ffi::CString::new(name)?; + // SAFETY: the path and the name are NUL-terminated strings, and the value lives for the call. + let result = unsafe { + libc::setxattr( + path.as_ptr(), + name.as_ptr(), + value.as_ptr().cast(), + value.len(), + 0, + ) + }; + if result == 0 { + Ok(()) + } else { + Err(std::io::Error::last_os_error()) + } +} + +#[test] +async fn a_snapshot_file_that_fails_its_check_is_left_out_of_list_and_makes_an_unknown_name_corrupt() + { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path()) + .await + .unwrap(); + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new(&format!("snapshots/{}", "ab".repeat(32))), + b"not a snapshot", + ) + .await + .unwrap(); + + let names = listed_names(&store, &scope).await; + let kept = store.stat(&scope, &name("p-kept")).await; + let unknown = store.stat(&scope, &name("p-unknown")).await; + let restore_unknown = restored_listing(&store, &scope, &name("p-unknown")).await; + + assert!( + matches!(unknown, Err(SnapshotStoreError::Corrupt(_))), + "{unknown:?}" + ); + assert!( + matches!(restore_unknown, Err(SnapshotStoreError::Corrupt(_))), + "{restore_unknown:?}" + ); + assert_eq!( + (names, kept.map(|info| info.is_some()).ok()), + (vec!["p-kept".to_string()], Some(true)) + ); +} + +#[test] +async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_period() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path()) + .await + .unwrap(); + let packs_before = blobs(&*storage, &scope.0, "data/").await; + + store.delete(&scope, &name("p-deleted")).await.unwrap(); + let after_first = ledger(&*storage, &scope).await; + store.delete(&scope, &name("p-none")).await.unwrap(); + let packs_after = blobs(&*storage, &scope.0, "data/").await; + + assert_eq!( + ( + after_first.freed_bytes, + after_first.last_prune_millis.is_some(), + after_first.awaiting_removal, + packs_after.len() < packs_before.len(), + packs_after.iter().all(|pack| packs_before.contains(pack)), + restored_listing(&store, &scope, &name("p-kept")).await.ok(), + ), + (0, true, true, true, true, Some(listing(kept_tree.path()))) + ); +} + +#[test] +async fn a_delete_below_the_threshold_does_not_prune() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), one_file_tree("kept")); + store + .save(&scope, &name("p-deleted"), deleted_tree.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path()) + .await + .unwrap(); + let packs_before = blobs(&*storage, &scope.0, "data/").await; + + store.delete(&scope, &name("p-deleted")).await.unwrap(); + let after = ledger(&*storage, &scope).await; + + assert_eq!( + ( + after.freed_bytes > 0, + after.last_prune_millis, + blobs(&*storage, &scope.0, "data/").await, + ), + (true, None, packs_before) + ); +} + +#[test] +async fn no_second_prune_runs_within_the_grace_period() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, 1, Duration::from_secs(3600)), + ); + let scope = new_scope(); + let trees = [one_file_tree("a"), one_file_tree("b"), one_file_tree("c")]; + futures::stream::iter(["p-a", "p-b", "p-c"].into_iter().zip(&trees)) + .for_each(|(text, tree)| { + let store = store.clone(); + let scope = scope.clone(); + async move { + store.save(&scope, &name(text), tree.path()).await.unwrap(); + } + }) + .await; + + store.delete(&scope, &name("p-a")).await.unwrap(); + let after_first = ledger(&*storage, &scope).await; + store.delete(&scope, &name("p-b")).await.unwrap(); + let after_second = ledger(&*storage, &scope).await; + + assert_eq!( + ( + after_first.last_prune_millis.is_some(), + after_second.last_prune_millis == after_first.last_prune_millis, + after_second.freed_bytes > 0, + ), + (true, true, true) + ); +} + +#[test] +async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { + let refuse = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) + && matches!(op_label, "read" | "read_range") + && path.starts_with("data") + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path()) + .await + .unwrap(); + refuse.store(true, Ordering::SeqCst); + + let failed = store.delete(&scope, &name("p-deleted")).await; + let after_failure = ledger(&*storage, &scope).await; + refuse.store(false, Ordering::SeqCst); + let retried = store.delete(&scope, &name("p-deleted")).await; + let after_retry = ledger(&*storage, &scope).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert_eq!( + ( + after_failure.freed_bytes > 0, + after_failure.last_prune_millis, + retried.is_ok(), + after_retry.freed_bytes, + after_retry.last_prune_millis.is_some(), + ), + (true, None, true, 0, true) + ); +} + +#[test] +async fn a_deleted_scope_holds_no_blob() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let scope = new_scope(); + let (first, second) = (one_file_tree("first"), one_file_tree("second")); + store + .save(&scope, &name("p-1"), first.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), second.path()) + .await + .unwrap(); + store.delete(&scope, &name("p-1")).await.unwrap(); + let before = blobs(&*storage, &scope.0, "").await; + + store.delete_scope(&scope).await.unwrap(); + + assert_eq!( + ( + before.contains(&"golem/prune-ledger".to_string()), + blobs(&*storage, &scope.0, "").await + ), + (true, Vec::::new()) + ); +} + +#[test] +async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_name_free() { + // The first save counts the calls of a save. Each later round holds one of these calls: the + // call reaches the storage and never answers, as a write that S3 received and completes after + // the caller left. The round drops the save there, and waits until the store has no work. + let tree = one_file_tree("dropped"); + let other = one_file_tree("saved later"); + let counted = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + store( + counted.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ) + .save(&new_scope(), &name("p-dropped"), tree.path()) + .await + .unwrap(); + let calls = counted.calls().len(); + + let rounds = futures::stream::iter(1..=calls) + .then(|held| { + let (tree, other) = (tree.path().to_path_buf(), other.path().to_path_buf()); + async move { + let inner = Arc::new(InMemoryBlobStorage::new()); + let seen = Arc::new(AtomicUsize::new(0)); + let storage = ScriptedBlobStorage::new(inner.clone(), move |_, _| { + if seen.fetch_add(1, Ordering::SeqCst) + 1 == held { + Script::NeverAnswer + } else { + Script::Pass + } + }); + let policy = policy(LONG_DEADLINE, u64::MAX, Duration::ZERO); + let dropping = store(storage.clone(), policy); + let scope = new_scope(); + let ended = drop_when( + &storage, + |calls| calls.len() >= held, + dropping.save(&scope, &name("p-dropped"), &tree), + ) + .await; + let stopped = tokio::time::timeout(LIMIT, dropping.shut_down()) + .await + .is_ok(); + let later = store(inner, policy); + let stat = later.stat(&scope, &name("p-dropped")).await.ok().flatten(); + let names = listed_names(&later, &scope).await; + let saved_again = later.save(&scope, &name("p-dropped"), &other).await; + let restored = restored_listing(&later, &scope, &name("p-dropped")) + .await + .ok(); + ( + held, + ended.is_none(), + stopped, + stat, + names, + saved_again.is_ok(), + restored == Some(listing(&other)), + ) + } + }) + .collect::>() + .await; + + assert_eq!( + rounds, + (1..=calls) + .map(|held| (held, true, true, None, Vec::new(), true, true)) + .collect::, + Vec, + bool, + bool + )>>() + ); +} + +#[test] +async fn shut_down_ends_running_operations_before_it_returns() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "write" && path.starts_with("data") { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + let saving = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + let path = tree.path().to_path_buf(); + async move { store.save(&scope, &name("p-held"), &path).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, path)| *op_label == "write" && path.starts_with("data")) + }) + .await; + + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + let saved = tokio::time::timeout(LIMIT, saving).await; + let later = store.stat(&scope, &name("p-held")).await; + storage.open_gate(); + + assert!( + matches!(&saved, Ok(Ok(Err(error))) if is_storage(error, true)), + "{saved:?}" + ); + assert!( + later.as_ref().is_err_and(|error| is_storage(error, false)), + "{later:?}" + ); + assert_eq!((held, stopped, store.work_in_flight()), (true, true, 0)); +} + +/// The operation of the store that a test drops. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum Dropped { + Restore, + Stat, + List, + Delete, +} + +#[test] +async fn a_dropped_operation_stops_its_blocking_work() { + let outcomes = futures::stream::iter([ + Dropped::Restore, + Dropped::Stat, + Dropped::List, + Dropped::Delete, + ]) + .then(|dropped| async move { + let held_call = move |op_label: &str, path: &Path| match dropped { + Dropped::Restore => op_label == "read_range" && path.starts_with("data"), + Dropped::Stat | Dropped::List => op_label == "read" && path.starts_with("snapshots"), + Dropped::Delete => op_label == "delete" && path.starts_with("snapshots"), + }; + let hold = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let hold = hold.clone(); + move |op_label, path| { + if held_call(op_label, path) && hold.load(Ordering::SeqCst) { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + hold.store(true, Ordering::SeqCst); + let before = storage.calls().len(); + let reached = |calls: &[(&'static str, String)]| { + calls[before..] + .iter() + .any(|(op_label, path)| held_call(op_label, Path::new(path))) + }; + let into = Scratch::new(); + let ended = match dropped { + Dropped::Restore => { + drop_when( + &storage, + reached, + store.restore(&scope, &name("p-1"), into.path()).map(|_| ()), + ) + .await + } + Dropped::Stat => { + drop_when( + &storage, + reached, + store.stat(&scope, &name("p-1")).map(|_| ()), + ) + .await + } + Dropped::List => drop_when(&storage, reached, store.list(&scope).map(|_| ())).await, + Dropped::Delete => { + drop_when( + &storage, + reached, + store.delete(&scope, &name("p-1")).map(|_| ()), + ) + .await + } + }; + let held = reached(&storage.calls()); + let stopped = eventually(|| store.work_in_flight() == 0).await; + storage.open_gate(); + (dropped, held, ended.is_none(), stopped) + }) + .collect::>() + .await; + + assert_eq!( + outcomes, + [ + Dropped::Restore, + Dropped::Stat, + Dropped::List, + Dropped::Delete + ] + .map(|dropped| (dropped, true, true, true)) + ); +} From bba479a86f0aeeb49b4f1a279b35e03aca96cf1b Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 10:03:26 -0700 Subject: [PATCH 13/55] Decide from a pure function whether a prune leaves marked packs, and sort listed snapshots on borrowed labels --- .../filesystem_snapshot/rustic/scripted.rs | 20 ++++++++++--- .../src/filesystem_snapshot/rustic/store.rs | 30 ++++++++++++------- 2 files changed, 35 insertions(+), 15 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs index 348e5206da..8c02e95e1b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -43,6 +43,9 @@ pub(super) enum Script { NeverAnswer, /// Waits until the test opens the gate of the storage, and then passes the call. WaitForGate, + /// Gives no blob to a read of a whole blob, as a delete after a listing does. Each other call + /// passes. + Vanish, } /// A rule that gives the script of a call from its operation label and its path. @@ -85,16 +88,20 @@ impl ScriptedBlobStorage { .collect() } + fn record(&self, op_label: &'static str, path: &Path) { + self.calls + .lock() + .unwrap_or_else(PoisonError::into_inner) + .push((op_label, path.into())); + } + async fn answer( &self, op_label: &'static str, path: &Path, call: impl Future>, ) -> anyhow::Result { - self.calls - .lock() - .unwrap_or_else(PoisonError::into_inner) - .push((op_label, path.into())); + self.record(op_label, path); match (self.rule)(op_label, path) { Script::Pass => call.await, Script::Refuse => Err(anyhow::anyhow!("the storage refused the call")), @@ -110,6 +117,7 @@ impl ScriptedBlobStorage { self.gate.cancelled().await; call.await } + Script::Vanish => call.await, } } } @@ -129,6 +137,10 @@ impl BlobStorage for ScriptedBlobStorage { namespace: BlobStorageNamespace, path: &Path, ) -> anyhow::Result>> { + if (self.rule)(op_label, path) == Script::Vanish { + self.record(op_label, path); + return Ok(None); + } self.answer( op_label, path, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 705d2de252..59028be87c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -26,8 +26,9 @@ use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; use super::publish::{SnapshotFiles, SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; use super::{ - ChangeDetection, PruneSettings, RepackLimits, RepositoryKey, RepositorySettings, SaveSettings, - backup_options, open_existing, open_or_create, prune, restore_snapshot, run_blocking, + ChangeDetection, PruneReport, PruneSettings, RepackLimits, RepositoryKey, RepositorySettings, + SaveSettings, backup_options, open_existing, open_or_create, prune, restore_snapshot, + run_blocking, }; use crate::filesystem_snapshot::{ FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, @@ -265,9 +266,7 @@ impl RusticSnapshotStore { prune(backend, &key, &settings) }) .await?; - let marked_packs = report.is_some_and(|report| { - report.packs_unused + report.packs_repacked + report.marked_packs_kept > 0 - }); + let marked_packs = report.as_ref().is_some_and(leaves_marked_packs); write_ledger( &*self.storage, &scope.0, @@ -627,12 +626,12 @@ fn lookup(found: ScopeSnapshots, name: &SnapshotName) -> Lookup { /// parses. Of the files with one name, only the one that [`lookup`] takes stays. fn listed(found: ScopeSnapshots) -> Vec<(SnapshotName, SnapshotInfo)> { let mut snapshots = found.readable; - snapshots.sort_by_key(|snapshot| { - ( - snapshot.label.clone(), - snapshot.time.timestamp(), - snapshot.id, - ) + snapshots.sort_by(|left, right| { + (&left.label, left.time.timestamp(), left.id).cmp(&( + &right.label, + right.time.timestamp(), + right.id, + )) }); snapshots.dedup_by(|later, first| later.label == first.label); snapshots @@ -656,6 +655,15 @@ fn snapshot_info(snapshot: &SnapshotFile) -> Option { }) } +/// Tells whether a later prune removes packs that this prune leaves marked. +/// +/// A prune marks each pack that holds only unused blobs, and each pack that it repacks. A pack that +/// an earlier prune marked and whose time to stay is not over stays marked. A pack that no index +/// lists is also marked, but the report does not count it. A later due prune removes that pack. +fn leaves_marked_packs(report: &PruneReport) -> bool { + report.packs_unused > 0 || report.packs_repacked > 0 || report.marked_packs_kept > 0 +} + /// Gives the packed bytes that the save of the snapshot added to the repository. fn added_packed_bytes(snapshot: &SnapshotFile) -> u64 { snapshot From 4384f423e7df85489e4e2e2ca863615291a57415 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 10:03:26 -0700 Subject: [PATCH 14/55] Test the snapshot reads, the ledger, the tree check, the tracker and the restore phases of the rustic store --- .../rustic/backend/tests.rs | 18 ++ .../filesystem_snapshot/rustic/store/tests.rs | 241 +++++++++++++++++- .../src/filesystem_snapshot/rustic/tests.rs | 30 +++ 3 files changed, 286 insertions(+), 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index cb52483c36..e0ef93e606 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -478,6 +478,24 @@ fn a_thread_that_is_not_a_thread_of_the_runtime_can_call_the_backend() { assert_eq!(read.ok().flatten(), Some(Bytes::from_static(b"index"))); } +#[test] +fn a_tracked_backend_counts_in_its_tracker_until_it_drops() { + let fixture = Fixture::new(); + let tracker = tokio_util::task::TaskTracker::new(); + let backend = BlobBackend::new( + fixture.storage.clone(), + fixture.namespace.clone(), + fixture.runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .tracked_by(tracker.token()); + + let while_alive = tracker.len(); + drop(backend); + + assert_eq!((while_alive, tracker.len()), (1, 0)); +} + #[test] fn a_cancelled_backend_makes_no_storage_call() { let runtime = Runtime::new().unwrap(); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index c6d7c85b16..ed851d126b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -19,10 +19,10 @@ use super::super::prune::{PruneLedger, read_ledger}; use super::super::scripted::{Script, ScriptedBlobStorage}; -use super::super::{PruneSettings, RepackLimits, RepositoryKey, open_existing}; +use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; use super::{ - RusticSnapshotStore, StorePolicy, scope_snapshots, store_backup_options, store_restore_options, - whole_millis_from, + RusticSnapshotStore, StorePolicy, leaves_marked_packs, scope_snapshots, store_backup_options, + store_restore_options, whole_millis_from, }; use crate::filesystem_snapshot::contract_tests::fixture::{ Listed, Scratch, Spec, fixture, listing, write_tree, @@ -1114,3 +1114,238 @@ async fn a_dropped_operation_stops_its_blocking_work() { .map(|dropped| (dropped, true, true, true)) ); } + +/// Gives the snapshot files of the scope that the bridge reads, on a blocking thread. +async fn snapshot_files( + storage: Arc, + scope: &SnapshotScope, +) -> Vec { + let namespace = scope.0.clone(); + tokio::task::spawn_blocking(move || { + let backend = super::super::backend::BlobBackend::new( + storage, + namespace, + tokio::runtime::Handle::current(), + LONG_DEADLINE, + ); + let repository = open_existing(Arc::new(backend), &key()).unwrap().unwrap(); + scope_snapshots(&repository).unwrap().readable + }) + .await + .unwrap() +} + +#[test] +async fn a_failed_read_of_a_snapshot_file_fails_stat_and_list_with_a_retryable_storage_error() { + let refuse = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) && op_label == "read" && path.starts_with("snapshots") + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path()) + .await + .unwrap(); + refuse.store(true, Ordering::SeqCst); + + let stat = store.stat(&scope, &name("p-kept")).await; + let list = store.list(&scope).await; + + assert!( + stat.as_ref().is_err_and(|error| is_storage(error, true)), + "{stat:?}" + ); + assert!( + list.as_ref().is_err_and(|error| is_storage(error, true)), + "{list:?}" + ); +} + +#[test] +async fn a_snapshot_file_that_is_gone_after_the_listing_is_left_out() { + let gone = format!("snapshots/{}", "cd".repeat(32)); + let inner = Arc::new(InMemoryBlobStorage::new()); + let storage = ScriptedBlobStorage::new(inner.clone(), { + let gone = gone.clone(); + move |op_label, path| { + if op_label == "read" && path == Path::new(&gone) { + Script::Vanish + } else { + Script::Pass + } + } + }); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path()) + .await + .unwrap(); + inner + .put_raw("test", "test", scope.0.clone(), Path::new(&gone), b"listed") + .await + .unwrap(); + + let unknown = store.stat(&scope, &name("p-unknown")).await; + let names = listed_names(&store, &scope).await; + + assert!(matches!(unknown, Ok(None)), "{unknown:?}"); + assert_eq!(names, vec!["p-kept".to_string()]); +} + +#[test] +async fn a_delete_that_frees_nothing_writes_no_ledger() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path()) + .await + .unwrap(); + + store.delete(&scope, &name("p-unknown")).await.unwrap(); + + assert_eq!( + blobs(&*storage, &scope.0, "golem/").await, + Vec::::new() + ); +} + +#[test] +fn a_prune_leaves_marked_packs_when_it_marks_repacks_or_keeps_marked_packs() { + let report = |packs_unused, packs_repacked, marked_packs_kept| PruneReport { + packs_unused, + packs_repacked, + marked_packs_kept, + ..PruneReport::default() + }; + + assert_eq!( + [ + leaves_marked_packs(&report(0, 0, 0)), + leaves_marked_packs(&report(1, 0, 0)), + leaves_marked_packs(&report(0, 1, 0)), + leaves_marked_packs(&report(0, 0, 1)), + leaves_marked_packs(&PruneReport { + packs_used: 3, + marked_packs_deleted: 2, + ..PruneReport::default() + }), + ], + [false, true, true, true, false] + ); +} + +#[test] +async fn a_save_of_a_relative_directory_path_gives_source_and_publishes_nothing() { + // Cargo runs the tests in the directory of the crate, so the path names a directory. + let relative = Path::new("src/filesystem_snapshot/contract_tests"); + assert!( + relative.is_dir(), + "the test runs in the directory of the crate" + ); + let store = store( + Arc::new(InMemoryBlobStorage::new()), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + + let saved = store.save(&scope, &name("p-relative"), relative).await; + let names = listed_names(&store, &scope).await; + + assert!( + matches!(saved, Err(SnapshotStoreError::Source(_))), + "{saved:?}" + ); + assert_eq!(names, Vec::::new()); +} + +#[test] +async fn a_save_of_a_regular_file_gives_source_and_publishes_nothing() { + let tree = one_file_tree("a file, not a tree"); + let store = store( + Arc::new(InMemoryBlobStorage::new()), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + + let saved = store + .save(&scope, &name("p-file"), &tree.path().join("file.txt")) + .await; + let names = listed_names(&store, &scope).await; + + assert!( + matches!(saved, Err(SnapshotStoreError::Source(_))), + "{saved:?}" + ); + assert_eq!(names, Vec::::new()); +} + +#[test] +async fn the_ledger_counts_the_packed_bytes_that_the_deleted_snapshot_added() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = fixture_tree(); + store + .save(&scope, &name("p-deleted"), tree.path()) + .await + .unwrap(); + let added = snapshot_files(storage.clone(), &scope) + .await + .iter() + .filter_map(|snapshot| snapshot.summary.as_ref()) + .map(|summary| summary.data_added_packed) + .sum::(); + + store.delete(&scope, &name("p-deleted")).await.unwrap(); + + assert_eq!( + (added > 1, ledger(&*storage, &scope).await.freed_bytes), + (true, added) + ); +} + +#[test] +async fn a_config_write_that_fails_gives_a_storage_error_with_that_failure() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { + if op_label == "write" && path == Path::new("config") { + Script::Refuse + } else { + Script::Pass + } + }); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = one_file_tree("never saved"); + + let saved = store.save(&scope, &name("p-1"), tree.path()).await; + + assert!( + matches!( + &saved, + Err(SnapshotStoreError::Storage { retryable: true, source }) + if format!("{source:#}").contains("the storage refused the call") + ), + "{saved:?}" + ); +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index d94c238915..6d9fa71445 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -209,6 +209,36 @@ async fn a_saved_tree_comes_back_the_same() { ); } +#[test] +async fn a_restore_report_gives_each_phase_of_the_restore_in_order() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let repository = repository(&storage, &new_scope()); + let tree = fixture_tree(); + let into = Scratch::new(); + repository.save(&name("first"), tree.path()).await.unwrap(); + + let restored = repository + .restore(&name("first"), into.path(), None) + .await + .unwrap() + .unwrap(); + + assert_eq!( + restored + .phases + .iter() + .map(|phase| phase.phase) + .collect::>(), + vec![ + OperationPhase::Open, + OperationPhase::Lookup, + OperationPhase::IndexLoad, + OperationPhase::RestorePlan, + OperationPhase::Restore, + ] + ); +} + #[test] async fn a_second_save_has_the_first_as_parent_and_reads_only_the_changed_file() { let storage = Arc::new(InMemoryBlobStorage::new()); From 7c832e9d20f0c939c1877e628a2034181fa5cbee Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 10:20:34 -0700 Subject: [PATCH 15/55] Require that a refused tree leaves the scope without a blob --- .../filesystem_snapshot/rustic/store/tests.rs | 17 +++++++++-------- 1 file changed, 9 insertions(+), 8 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index ed851d126b..e429f0e3ae 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1252,34 +1252,36 @@ fn a_prune_leaves_marked_packs_when_it_marks_repacks_or_keeps_marked_packs() { } #[test] -async fn a_save_of_a_relative_directory_path_gives_source_and_publishes_nothing() { +async fn a_save_of_a_relative_directory_path_gives_source_and_writes_nothing() { // Cargo runs the tests in the directory of the crate, so the path names a directory. let relative = Path::new("src/filesystem_snapshot/contract_tests"); assert!( relative.is_dir(), "the test runs in the directory of the crate" ); + let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( - Arc::new(InMemoryBlobStorage::new()), + storage.clone(), policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), ); let scope = new_scope(); let saved = store.save(&scope, &name("p-relative"), relative).await; - let names = listed_names(&store, &scope).await; assert!( matches!(saved, Err(SnapshotStoreError::Source(_))), "{saved:?}" ); - assert_eq!(names, Vec::::new()); + assert_eq!(blobs(&*storage, &scope.0, "").await, Vec::::new()); } #[test] -async fn a_save_of_a_regular_file_gives_source_and_publishes_nothing() { +async fn a_save_of_a_regular_file_gives_source_and_writes_nothing() { + // The store refuses the tree before it makes a repository, so the scope stays unused. let tree = one_file_tree("a file, not a tree"); + let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( - Arc::new(InMemoryBlobStorage::new()), + storage.clone(), policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), ); let scope = new_scope(); @@ -1287,13 +1289,12 @@ async fn a_save_of_a_regular_file_gives_source_and_publishes_nothing() { let saved = store .save(&scope, &name("p-file"), &tree.path().join("file.txt")) .await; - let names = listed_names(&store, &scope).await; assert!( matches!(saved, Err(SnapshotStoreError::Source(_))), "{saved:?}" ); - assert_eq!(names, Vec::::new()); + assert_eq!(blobs(&*storage, &scope.0, "").await, Vec::::new()); } #[test] From 785cb0894eb5974da7c121f02a6a993c622f5c32 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 11:50:32 -0700 Subject: [PATCH 16/55] Wake the waits for gateway bodies when a body starts, and let the test timeout bound them --- .../src/gateway_server/tests.rs | 26 +++++++++---------- 1 file changed, 12 insertions(+), 14 deletions(-) diff --git a/golem-worker-service/src/gateway_server/tests.rs b/golem-worker-service/src/gateway_server/tests.rs index 0ed56f402b..ab432c8fb4 100644 --- a/golem-worker-service/src/gateway_server/tests.rs +++ b/golem-worker-service/src/gateway_server/tests.rs @@ -17,8 +17,6 @@ use tokio::sync::Notify; use super::run; -const CLEANUP_TIMEOUT: Duration = Duration::from_secs(2); - #[derive(Clone, Default)] struct LiveBodies { count: Arc, @@ -26,8 +24,10 @@ struct LiveBodies { } impl LiveBodies { + /// Counts one more live body, and wakes each wait for a count. fn guard(&self) -> BodyGuard { self.count.fetch_add(1, Ordering::SeqCst); + self.changed.notify_waiters(); BodyGuard(self.clone()) } @@ -35,20 +35,18 @@ impl LiveBodies { self.count.load(Ordering::SeqCst) } + /// Waits until the count is `expected`. The wait has no limit of its own; the timeout of each + /// test ends a wait that never ends. async fn wait_for(&self, expected: usize) { - tokio::time::timeout(CLEANUP_TIMEOUT, async { - loop { - let changed = self.changed.notified(); - tokio::pin!(changed); - changed.as_mut().enable(); - if self.count() == expected { - return; - } - changed.await; + loop { + let changed = self.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + if self.count() == expected { + return; } - }) - .await - .unwrap_or_else(|_| panic!("body count did not become {expected}; was {}", self.count())); + changed.await; + } } } From 94b2321a14da5d9ca4f359e4da26835a7f8a2edf Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:31:16 -0700 Subject: [PATCH 17/55] Try the dropped-save contract case in new scopes until a save does not end in its first poll --- .../filesystem_snapshot/contract_tests/mod.rs | 27 +++++++++++++------ 1 file changed, 19 insertions(+), 8 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs b/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs index 1d57d21b91..0b33ba5d99 100644 --- a/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs @@ -62,6 +62,10 @@ pub(crate) type OpenStore = Arc Arc + S type Case = fn(OpenStore) -> BoxFuture<'static, ()>; +/// The number of saves that the dropped-save case starts to find one that does not end in its +/// first poll. +const DROP_ATTEMPTS: usize = 20; + /// Each case of the contract, with its name. const CASES: &[(&str, Case)] = &[ ("a_saved_tree_comes_back_the_same", |open| { @@ -958,19 +962,26 @@ async fn a_dropped_save_publishes_nothing_and_leaves_the_name_free(open: OpenSto // interrupted save, so it publishes nothing and leaves the name free. This case checks at // once after the drop, and it cannot see a publish that comes much later. So an adapter that // runs its save in the background also proves in its own tests that a dropped save stops. + // A save can end in its first poll when a busy host runs its read before that poll, so the + // case tries saves in new scopes until one does not end in its first poll. let store = open(); - let scope = new_scope(); let dropped_name = name("p-dropped"); let tree = new_tree(&fixture()); let other = new_tree(&one_file("other tree")); - let dropped = store - .save(&scope, &dropped_name, tree.path()) - .now_or_never(); - assert!( - dropped.is_none(), - "the save returned in one poll, so the case cannot drop it before it returns: {dropped:?}" - ); + let scope = futures::stream::iter(0..DROP_ATTEMPTS) + .filter_map(|_| { + let scope = new_scope(); + let dropped = store + .save(&scope, &dropped_name, tree.path()) + .now_or_never(); + std::future::ready(dropped.is_none().then_some(scope)) + }) + .next() + .await + .unwrap_or_else(|| { + panic!("each of {DROP_ATTEMPTS} saves returned in one poll, so the case cannot drop one before it returns") + }); let stat = store.stat(&scope, &dropped_name).await.unwrap(); let restore = restored(&*store, &scope, &dropped_name).await; let names_after_the_drop = listed_names(&*store, &scope).await; From c73a5dcecca3ddaeb49b7324f220eb708e26d61a Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:55:48 -0700 Subject: [PATCH 18/55] Classify a prune error as a storage error, and test the real chain of a failed extended attribute --- .../src/filesystem_snapshot/rustic/fault.rs | 100 +++++++++++++++--- 1 file changed, 83 insertions(+), 17 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs index 0c273adf74..36c09adb04 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/fault.rs @@ -104,18 +104,15 @@ pub(super) enum Operation { Restore, /// Reads or changes only the repository. Repository, + /// Removes the data that no snapshot uses. + Prune, } -/// Gives the error of the store for an operation that failed with the error. -/// -/// - A failed blob storage call gives `Storage`. It is retryable unless a name error of the blob -/// storage caused it, because a name error is the same for each new try. -/// - An I/O error gives `Source` in a save and `Destination` in a restore, with the kind of that -/// I/O error. -/// - Each other error gives `Storage` that is not retryable in a save, and `Corrupt` in the other -/// operations, because the repository gave data that rustic refused. +/// A failed storage call gives `Storage`, retryable unless a name error caused it. An I/O error +/// gives `Source` in a save and `Destination` in a restore. Each other error gives `Storage` that +/// is not retryable in a save or a prune, and `Corrupt` in a restore or a read of the repository. pub(super) fn classify(operation: Operation, error: anyhow::Error) -> SnapshotStoreError { - let from_storage = chain(error.as_ref()).any(|error| error.is::()); + let from_storage = is_storage_failure(error.as_ref()); let io_kind = chain(error.as_ref()) .find_map(|error| error.downcast_ref::()) .map(std::io::Error::kind); @@ -131,7 +128,7 @@ pub(super) fn classify(operation: Operation, error: anyhow::Error) -> SnapshotSt (false, Some(kind), Operation::Restore) => { SnapshotStoreError::Destination(std::io::Error::new(kind, error_text(&error))) } - (false, _, Operation::Save) => SnapshotStoreError::Storage { + (false, _, Operation::Save | Operation::Prune) => SnapshotStoreError::Storage { retryable: false, source: error, }, @@ -186,7 +183,7 @@ mod tests { )) } - fn storage_failure(failure: anyhow::Error) -> anyhow::Error { + fn failed_call(failure: anyhow::Error) -> anyhow::Error { rustic(BlobCallFailed::new(failure)) } @@ -208,7 +205,7 @@ mod tests { [Operation::Save, Operation::Restore, Operation::Repository].map(|operation| { shape(&classify( operation, - storage_failure(anyhow::anyhow!("the bucket is gone")), + failed_call(anyhow::anyhow!("the bucket is gone")), )) }); @@ -220,7 +217,7 @@ mod tests { let failure = anyhow::Error::new(io::Error::new(io::ErrorKind::StorageFull, "no space")); assert_eq!( - shape(&classify(Operation::Restore, storage_failure(failure))), + shape(&classify(Operation::Restore, failed_call(failure))), ("Storage", Some(true), None) ); } @@ -232,7 +229,7 @@ mod tests { }); assert_eq!( - shape(&classify(Operation::Repository, storage_failure(failure))), + shape(&classify(Operation::Repository, failed_call(failure))), ("Storage", Some(false), None) ); } @@ -243,7 +240,7 @@ mod tests { [Operation::Save, Operation::Restore, Operation::Repository].map(|operation| { shape(&classify( operation, - storage_failure(anyhow::Error::new(OperationCancelled)), + failed_call(anyhow::Error::new(OperationCancelled)), )) }); @@ -314,10 +311,79 @@ mod tests { ); } + /// The error below the rustic error of a restore that cannot set an extended attribute, as + /// the fork gives it on ext4 for a user attribute of 6,000 bytes. + #[derive(Debug)] + struct SettingXattrFailed(io::Error); + + impl std::fmt::Display for SettingXattrFailed { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + formatter, + "setting xattr `user.golem-test` on `\"/restore/file.txt\"` with `{:?}`", + self.0 + ) + } + } + + impl std::error::Error for SettingXattrFailed { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(&self.0) + } + } + + #[test] + fn the_real_chain_of_a_failed_extended_attribute_of_a_restore_gives_destination() { + let error = anyhow::Error::new(RusticError::with_source( + ErrorKind::InputOutput, + "The restore cannot set the extended attributes of `file.txt`.", + SettingXattrFailed(io::Error::from_raw_os_error(28)), + )); + + assert_eq!( + shape(&classify(Operation::Restore, error)), + ("Destination", None, Some(io::ErrorKind::StorageFull)) + ); + } + + #[test] + fn a_prune_error_keeps_the_retryable_flag_of_a_storage_failure_and_is_never_corrupt() { + let refused = anyhow::Error::new(RusticError::new( + ErrorKind::Internal, + "the pack has another size than the index says", + )); + + assert_eq!( + [ + shape(&classify(Operation::Prune, refused)), + shape(&classify( + Operation::Prune, + rustic(io::Error::new(io::ErrorKind::InvalidData, "bad pack")) + )), + shape(&classify( + Operation::Prune, + failed_call(anyhow::anyhow!("the bucket is gone")) + )), + shape(&classify( + Operation::Prune, + failed_call(anyhow::Error::new(BlobNameError::NoName { + path: std::path::PathBuf::new(), + })) + )), + ], + [ + ("Storage", Some(false), None), + ("Storage", Some(false), None), + ("Storage", Some(true), None), + ("Storage", Some(false), None), + ] + ); + } + #[test] fn the_config_marker_is_found_in_the_chain() { - let exists = storage_failure(anyhow::Error::new(ConfigExists)); - let other = storage_failure(anyhow::anyhow!("the bucket is gone")); + let exists = failed_call(anyhow::Error::new(ConfigExists)); + let other = failed_call(anyhow::anyhow!("the bucket is gone")); assert_eq!( ( From c0e489dc07be2fdbf986b58936f4d089f0efaaf8 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:55:48 -0700 Subject: [PATCH 19/55] Shorten the doc of the default storage call deadline --- golem-worker-executor/src/services/golem_config.rs | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/golem-worker-executor/src/services/golem_config.rs b/golem-worker-executor/src/services/golem_config.rs index 82e667787f..3ab3c64114 100644 --- a/golem-worker-executor/src/services/golem_config.rs +++ b/golem-worker-executor/src/services/golem_config.rs @@ -2339,11 +2339,8 @@ impl SafeDisplay for FilesystemPressureConfig { } } -/// The default of [`FilesystemSnapshotStoreConfig::storage_call_deadline`]. -/// -/// On S3, with the retries of the S3 storage, a write of a pack took at most 1.7 s with eight saves -/// at the same time. A ranged read of a pack took at most 1.5 s under the CPU request of an -/// executor. Keep the value at least 10 times the longest measured call. +/// The default of [`FilesystemSnapshotStoreConfig::storage_call_deadline`]. The slowest measured +/// call on S3 took 1.7 s, and the value stays at least 10 times the slowest measured call. pub const DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE: Duration = Duration::from_secs(30); /// The default of [`FilesystemSnapshotStoreConfig::restore_reader_threads`]. From 855fd18aa0dfa48a858894ea575f01974aac91a8 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:55:49 -0700 Subject: [PATCH 20/55] Pass the blobs of a scope as one value with one target label, share staged paths, and tighten the store tests --- .../src/filesystem_snapshot/rustic/backend.rs | 17 +- .../rustic/backend/tests.rs | 2 +- .../src/filesystem_snapshot/rustic/files.rs | 130 ++++++++++++++++ .../src/filesystem_snapshot/rustic/mod.rs | 1 + .../src/filesystem_snapshot/rustic/prune.rs | 111 ++++++-------- .../src/filesystem_snapshot/rustic/publish.rs | 55 ++----- .../rustic/publish/tests.rs | 7 +- .../src/filesystem_snapshot/rustic/scope.rs | 145 ++++-------------- .../filesystem_snapshot/rustic/scope/tests.rs | 68 +++++--- .../src/filesystem_snapshot/rustic/store.rs | 49 +++--- .../filesystem_snapshot/rustic/store/tests.rs | 114 +++++++++++--- 11 files changed, 386 insertions(+), 313 deletions(-) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/files.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index 0c64a04aeb..c0afa5cd55 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -20,6 +20,7 @@ //! for at most a deadline, and a cancelled operation makes no more calls. use super::fault::{BlobCallFailed, ConfigExists, FileMissing, OperationCancelled}; +use super::files::TARGET_LABEL; use super::publish::{SnapshotStage, StagedSnapshot}; use bytes::Bytes; use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace, PutIfAbsent}; @@ -34,9 +35,6 @@ use tokio::runtime::Handle; use tokio_util::sync::CancellationToken; use tokio_util::task::task_tracker::TaskTrackerToken; -/// The target label of each blob storage call of the backend. -const TARGET_LABEL: &str = "filesystem_snapshot"; - /// The path of the config file of a repository. const CONFIG_PATH: &str = "config"; @@ -139,11 +137,9 @@ impl BlobBackend { } } - /// Waits for one call on the blob storage, and gives its result as a rustic result. - /// - /// Each call of the backend on the blob storage goes through this function. A call that gives - /// no answer within the deadline gives an error, the same as a call that failed. So does a - /// call of a cancelled operation. + /// Waits for one call on the blob storage, which each call of the backend goes through. A call + /// without an answer within the deadline, or of a cancelled operation, gives an error, the same + /// as a call that failed. fn request( &self, call: StorageCall, @@ -320,7 +316,10 @@ impl WriteBackend for BlobBackend { }; match (tpe, &self.stage) { (FileType::Snapshot, Some(stage)) => stage - .keep(StagedSnapshot { path, content }) + .keep(StagedSnapshot { + path: Arc::from(path), + content, + }) .map_err(|staged| second_snapshot(&staged.path)), (FileType::Config, _) => match self.write_if_absent(&path, &content)? { PutIfAbsent::Written => Ok(()), diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index e0ef93e606..413763cdc5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -668,7 +668,7 @@ fn a_backend_with_a_stage_keeps_the_snapshot_file_and_does_not_write_it() { true, true, Some(StagedSnapshot { - path: PathBuf::from(format!("snapshots/{}", "cd".repeat(32))).into_boxed_path(), + path: Arc::from(PathBuf::from(format!("snapshots/{}", "cd".repeat(32)))), content: Bytes::from_static(b"snapshot"), }), vec![(format!("data/ab/{}", "ab".repeat(32)), 4)] diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs new file mode 100644 index 0000000000..297f5c5959 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs @@ -0,0 +1,130 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! The blobs of the repository of one scope, and the blob storage calls of the store on them. +//! +//! Each call waits for at most the deadline of the scope. + +use super::backend::answer_within; +use golem_service_base::storage::blob::{ + BlobStorage, BlobStorageNamespace, ListedBlob, PutIfAbsent, +}; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; + +/// The target label of each blob storage call of the rustic store. +pub(super) const TARGET_LABEL: &str = "filesystem_snapshot"; + +/// The blobs of one scope: the storage, the namespace of the scope, and the deadline of each call. +#[derive(Clone, Debug)] +pub(super) struct SnapshotFiles { + pub(super) storage: Arc, + pub(super) namespace: BlobStorageNamespace, + pub(super) deadline: Duration, +} + +impl SnapshotFiles { + /// Gives the content of the blob at the path, or `None` when the path has no blob. + pub(super) async fn get( + &self, + op_label: &'static str, + path: &Path, + ) -> anyhow::Result>> { + answer_within( + self.deadline, + self.storage + .get_raw(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await + } + + /// Writes the content as the blob at the path, over the blob that was there. + pub(super) async fn put( + &self, + op_label: &'static str, + path: &Path, + content: &[u8], + ) -> anyhow::Result<()> { + answer_within( + self.deadline, + self.storage.put_raw( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + content, + ), + ) + .await + } + + /// Writes the content as the blob at the path only when the path has no blob. + pub(super) async fn put_if_absent( + &self, + op_label: &'static str, + path: &Path, + content: &[u8], + ) -> anyhow::Result { + answer_within( + self.deadline, + self.storage.put_raw_if_absent( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + content, + ), + ) + .await + } + + /// Deletes the blob at the path. A path without a blob gives success. + pub(super) async fn delete(&self, op_label: &'static str, path: &Path) -> anyhow::Result<()> { + answer_within( + self.deadline, + self.storage + .delete(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await + } + + /// Deletes the directory at the path and each blob below it. + pub(super) async fn delete_dir( + &self, + op_label: &'static str, + path: &Path, + ) -> anyhow::Result { + answer_within( + self.deadline, + self.storage + .delete_dir(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await + } + + /// Gives each blob below the path, at all depths, with its size. + pub(super) async fn list_below( + &self, + op_label: &'static str, + path: &Path, + ) -> anyhow::Result> { + answer_within( + self.deadline, + self.storage + .list_blobs_below(TARGET_LABEL, op_label, self.namespace.clone(), path), + ) + .await + } +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 6be31c2ff0..cd02c37a1d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -20,6 +20,7 @@ mod backend; mod fault; +mod files; mod prune; mod publish; mod scope; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index f2510a8b93..e412271a5d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -16,20 +16,16 @@ //! //! The scope keeps a small ledger blob next to the files of the repository. The ledger holds the //! packed bytes that deleted snapshots added since the last prune, the time of the last prune, and -//! whether that prune marked packs that a later prune removes. The ledger is advice: two deletes at -//! the same time can lose a count, and that only makes a prune come later. +//! whether that prune marked packs that a later prune removes. Two deletes at the same time can +//! lose a count. A lost count only delays a prune. -use super::backend::answer_within; +use super::files::SnapshotFiles; use golem_common::model::Timestamp; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; use serde::{Deserialize, Serialize}; use std::path::Path; use std::time::Duration; use tracing::warn; -/// The target label of each blob storage call on the ledger. -const TARGET_LABEL: &str = "filesystem_snapshot"; - /// The path of the ledger blob, relative to the root of the namespace of the scope. pub(super) const LEDGER_PATH: &str = "golem/prune-ledger"; @@ -38,8 +34,8 @@ pub(super) const LEDGER_PATH: &str = "golem/prune-ledger"; pub(super) struct PruneLedger { /// The packed bytes that the deleted snapshots added, since the last prune. pub(super) freed_bytes: u64, - /// The time of the last prune, in milliseconds since the Unix epoch. - pub(super) last_prune_millis: Option, + /// The time of the last prune. + pub(super) last_prune: Option, /// Whether the last prune marked packs that a later prune removes. pub(super) awaiting_removal: bool, } @@ -57,7 +53,7 @@ impl PruneLedger { pub(super) fn after_prune(now: Timestamp, marked_packs: bool) -> Self { Self { freed_bytes: 0, - last_prune_millis: Some(now.to_millis()), + last_prune: Some(now), awaiting_removal: marked_packs, } } @@ -74,30 +70,20 @@ pub(super) fn prune_due( threshold: u64, grace: Duration, ) -> bool { - let grace_passed = ledger.last_prune_millis.is_none_or(|last| { - now.to_millis() >= last.saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) + let grace_passed = ledger.last_prune.is_none_or(|last| { + now.to_millis() + >= last + .to_millis() + .saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) }); let work = ledger.freed_bytes >= threshold.max(1) || ledger.awaiting_removal; grace_passed && work } -/// Reads the ledger of the scope. A scope without a ledger gives an empty ledger, and so does a -/// ledger that does not parse, because the ledger is advice. -pub(super) async fn read_ledger( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - deadline: Duration, -) -> anyhow::Result { - let content = answer_within( - deadline, - storage.get_raw( - TARGET_LABEL, - "read_ledger", - namespace.clone(), - Path::new(LEDGER_PATH), - ), - ) - .await?; +/// Reads the ledger of the scope. A scope without a ledger, or with a ledger that does not parse, +/// gives an empty ledger, which only delays a prune. +pub(super) async fn read_ledger(files: &SnapshotFiles) -> anyhow::Result { + let content = files.get("read_ledger", Path::new(LEDGER_PATH)).await?; Ok(content.map_or_else(PruneLedger::default, |content| { serde_json::from_slice(&content).unwrap_or_else(|error| { warn!( @@ -111,34 +97,26 @@ pub(super) async fn read_ledger( /// Writes the ledger of the scope over the ledger that was there. pub(super) async fn write_ledger( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - deadline: Duration, + files: &SnapshotFiles, ledger: &PruneLedger, ) -> anyhow::Result<()> { let content = serde_json::to_vec(ledger)?; - answer_within( - deadline, - storage.put_raw( - TARGET_LABEL, - "write_ledger", - namespace.clone(), - Path::new(LEDGER_PATH), - &content, - ), - ) - .await + files + .put("write_ledger", Path::new(LEDGER_PATH), &content) + .await } #[cfg(test)] mod tests { + use super::super::files::SnapshotFiles; use super::{LEDGER_PATH, PruneLedger, prune_due, read_ledger, write_ledger}; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; + use golem_service_base::storage::blob::BlobStorageNamespace; use golem_service_base::storage::blob::memory::InMemoryBlobStorage; - use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; use pretty_assertions::assert_eq; use std::path::Path; + use std::sync::Arc; use std::time::Duration; use test_r::test; use uuid::Uuid; @@ -159,14 +137,18 @@ mod tests { ) -> PruneLedger { PruneLedger { freed_bytes, - last_prune_millis, + last_prune: last_prune_millis.map(Timestamp::from), awaiting_removal, } } - fn new_namespace() -> BlobStorageNamespace { - BlobStorageNamespace::InitialAgentFiles { - environment_id: EnvironmentId(Uuid::new_v4()), + fn new_files() -> SnapshotFiles { + SnapshotFiles { + storage: Arc::new(InMemoryBlobStorage::new()), + namespace: BlobStorageNamespace::InitialAgentFiles { + environment_id: EnvironmentId(Uuid::new_v4()), + }, + deadline: DEADLINE, } } @@ -257,38 +239,37 @@ mod tests { } #[test] - async fn the_ledger_is_written_and_read_back() { - let storage = InMemoryBlobStorage::new(); - let namespace = new_namespace(); - let written = ledger(123, Some(456), true); + async fn the_ledger_is_written_and_read_back_with_the_time_in_milliseconds() { + // The ledger keeps the time of the last prune as ISO 8601 text with milliseconds. + let files = new_files(); + let written = PruneLedger { + freed_bytes: 123, + last_prune: Some(Timestamp::from(Timestamp::now_utc().to_millis())), + awaiting_removal: true, + }; - let before = read_ledger(&storage, &namespace, DEADLINE).await.unwrap(); - write_ledger(&storage, &namespace, DEADLINE, &written) - .await - .unwrap(); - let after = read_ledger(&storage, &namespace, DEADLINE).await.unwrap(); + let before = read_ledger(&files).await.unwrap(); + write_ledger(&files, &written).await.unwrap(); + let after = read_ledger(&files).await.unwrap(); assert_eq!((before, after), (PruneLedger::default(), written)); } #[test] async fn a_ledger_that_does_not_parse_reads_as_an_empty_ledger() { - let storage = InMemoryBlobStorage::new(); - let namespace = new_namespace(); - storage + let files = new_files(); + files + .storage .put_raw( "test", "test", - namespace.clone(), + files.namespace.clone(), Path::new(LEDGER_PATH), b"not json", ) .await .unwrap(); - assert_eq!( - read_ledger(&storage, &namespace, DEADLINE).await.unwrap(), - PruneLedger::default() - ); + assert_eq!(read_ledger(&files).await.unwrap(), PruneLedger::default()); } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs index e8d48455ca..c2ea5c6e61 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish.rs @@ -19,24 +19,20 @@ //! write is the step that makes the snapshot visible. A publish that fails, or that the caller //! drops, deletes the file again, because a write that the storage received can still complete. -use super::backend::answer_within; +use super::files::SnapshotFiles; use bytes::Bytes; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; use std::path::Path; use std::sync::{Arc, Mutex, PoisonError}; -use std::time::Duration; use tokio::runtime::Handle; use tokio_util::task::TaskTracker; use tracing::warn; -/// The target label of each blob storage call of a publish. -const TARGET_LABEL: &str = "filesystem_snapshot"; - /// A snapshot file that the backend kept and did not write. #[derive(Clone, Debug, PartialEq, Eq)] pub(super) struct StagedSnapshot { - /// The path of the file, relative to the root of the namespace. - pub(super) path: Box, + /// The path of the file, relative to the root of the namespace. The drop guard of a publish + /// and its delete task share it. + pub(super) path: Arc, pub(super) content: Bytes, } @@ -63,22 +59,9 @@ impl SnapshotStage { } } -/// The snapshot files of one scope: the storage, the namespace of the scope, and the deadline of -/// each call. -#[derive(Clone, Debug)] -pub(super) struct SnapshotFiles { - pub(super) storage: Arc, - pub(super) namespace: BlobStorageNamespace, - pub(super) deadline: Duration, -} - -/// Writes the staged file, and so makes the snapshot visible. -/// -/// The file is written only when the path has no blob. The name of a snapshot file is the hash of -/// its content, so a blob at the path already holds this content, and the call succeeds. -/// -/// When the write fails, the call deletes the path before it gives the error. When the caller -/// drops the call during the write, a task of `tracker` deletes the path. +/// Writes the staged file only when its path has no blob, which makes the snapshot visible. The +/// name is the hash of the content, so a blob at the path is this file. A failed write deletes the +/// path before the error returns, and a dropped write deletes it in a task of `tracker`. pub(super) async fn publish( files: &SnapshotFiles, staged: &StagedSnapshot, @@ -90,17 +73,9 @@ pub(super) async fn publish( tracker: tracker.clone(), armed: true, }; - let written = answer_within( - files.deadline, - files.storage.put_raw_if_absent( - TARGET_LABEL, - "publish", - files.namespace.clone(), - &staged.path, - &staged.content, - ), - ) - .await; + let written = files + .put_if_absent("publish", &staged.path, &staged.content) + .await; retraction.armed = false; match written { Ok(_) => Ok(()), @@ -113,13 +88,7 @@ pub(super) async fn publish( /// Deletes the snapshot file at the path. A path without a blob gives success. pub(super) async fn retract(files: &SnapshotFiles, path: &Path) -> anyhow::Result<()> { - answer_within( - files.deadline, - files - .storage - .delete(TARGET_LABEL, "retract", files.namespace.clone(), path), - ) - .await + files.delete("retract", path).await } async fn retract_or_warn(files: &SnapshotFiles, path: &Path) { @@ -135,7 +104,7 @@ async fn retract_or_warn(files: &SnapshotFiles, path: &Path) { /// Deletes the path in a task of the tracker when it is dropped while it is armed. struct RetractOnDrop { files: SnapshotFiles, - path: Box, + path: Arc, tracker: TaskTracker, armed: bool, } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs index a61fc61724..ec5723fe13 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -12,16 +12,17 @@ // See the License for the specific language governing permissions and // limitations under the License. +use super::super::files::SnapshotFiles; use super::super::holding::reached_deadline; use super::super::scripted::{Script, ScriptedBlobStorage}; -use super::{SnapshotFiles, SnapshotStage, StagedSnapshot, publish, retract}; +use super::{SnapshotStage, StagedSnapshot, publish, retract}; use bytes::Bytes; use futures::FutureExt; use golem_common::model::environment::EnvironmentId; use golem_service_base::storage::blob::memory::InMemoryBlobStorage; use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; use pretty_assertions::assert_eq; -use std::path::{Path, PathBuf}; +use std::path::Path; use std::sync::Arc; use std::time::Duration; use test_r::test; @@ -36,7 +37,7 @@ const SNAPSHOT_PATH: &str = fn staged() -> StagedSnapshot { StagedSnapshot { - path: PathBuf::from(SNAPSHOT_PATH).into_boxed_path(), + path: Arc::from(Path::new(SNAPSHOT_PATH)), content: Bytes::from_static(b"snapshot"), } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs index d80ebe91f6..996e605de2 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs @@ -17,63 +17,28 @@ //! These operations do not read the repository format. They only know the directories of the //! repository, its config file, and the ledger directory of the store. -use super::backend::answer_within; +use super::files::SnapshotFiles; use super::prune::LEDGER_PATH; use futures::{StreamExt, TryStreamExt, stream}; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace, PutIfAbsent}; -use std::path::{Path, PathBuf}; -use std::time::Duration; - -/// The target label of each blob storage call on a scope. -const TARGET_LABEL: &str = "filesystem_snapshot"; +use golem_service_base::storage::blob::PutIfAbsent; +use std::path::Path; /// The path of the config file of a repository. const CONFIG_PATH: &str = "config"; /// The directories of a repository in the order of a listing. A save writes them in the reverse -/// order, so each snapshot file in a listing has its index files and packs in the later listings. -/// A copy writes them in the reverse order too, so a snapshot file in the target always has its -/// data. +/// order, and so does a copy, so a snapshot file always has its data. const LISTING_ORDER: [&str; 4] = ["snapshots", "index", "keys", "data"]; -/// Copies the repository of the namespace `from` into the empty namespace `to`. -/// -/// A namespace without a config file holds no repository, so the call copies nothing. The call -/// writes the config file of `to` last, so `to` holds a repository only when all its blobs are -/// there. It does not copy the ledger. A blob that is gone when the call reads it was deleted -/// after the listing, and the call does not copy it. -pub(super) async fn copy_scope( - storage: &dyn BlobStorage, - from: &BlobStorageNamespace, - to: &BlobStorageNamespace, - deadline: Duration, -) -> anyhow::Result<()> { - let Some(config) = answer_within( - deadline, - storage.get_raw( - TARGET_LABEL, - "copy_read", - from.clone(), - Path::new(CONFIG_PATH), - ), - ) - .await? - else { +/// Copies the repository of `from` into the empty scope `to`, with the config file last, so `to` +/// holds a repository only when all its blobs are there. It copies nothing without a config file, +/// and it does not copy the ledger or a blob that a delete removes after the listing. +pub(super) async fn copy_scope(from: &SnapshotFiles, to: &SnapshotFiles) -> anyhow::Result<()> { + let Some(config) = from.get("copy_read", Path::new(CONFIG_PATH)).await? else { return Ok(()); }; let listed = stream::iter(LISTING_ORDER) - .then(|directory| async move { - answer_within( - deadline, - storage.list_blobs_below( - TARGET_LABEL, - "copy_list", - from.clone(), - Path::new(directory), - ), - ) - .await - }) + .then(|directory| from.list_below("copy_list", Path::new(directory))) .try_collect::>() .await?; let paths = listed @@ -82,84 +47,38 @@ pub(super) async fn copy_scope( .flat_map(|blobs| blobs.iter().map(|blob| blob.path.clone())) .collect::>(); stream::iter(paths.iter().map(Ok)) - .try_for_each(|path| copy_blob(storage, from, to, path, deadline)) + .try_for_each(|path| copy_blob(from, to, path)) .await?; - answer_within( - deadline, - storage.put_raw_if_absent( - TARGET_LABEL, - "copy_write", - to.clone(), - Path::new(CONFIG_PATH), - &config, - ), - ) - .await - .map(|_: PutIfAbsent| ()) + to.put_if_absent("copy_write", Path::new(CONFIG_PATH), &config) + .await + .map(|_: PutIfAbsent| ()) } -async fn copy_blob( - storage: &dyn BlobStorage, - from: &BlobStorageNamespace, - to: &BlobStorageNamespace, - path: &Path, - deadline: Duration, -) -> anyhow::Result<()> { - let content = answer_within( - deadline, - storage.get_raw(TARGET_LABEL, "copy_read", from.clone(), path), - ) - .await?; - match content { - Some(content) => { - answer_within( - deadline, - storage.put_raw(TARGET_LABEL, "copy_write", to.clone(), path, &content), - ) - .await - } +async fn copy_blob(from: &SnapshotFiles, to: &SnapshotFiles, path: &Path) -> anyhow::Result<()> { + match from.get("copy_read", path).await? { + Some(content) => to.put("copy_write", path, &content).await, None => Ok(()), } } -/// Deletes the repository of the namespace, and the ledger of the store. -/// -/// The call deletes the config file first, so the namespace holds no repository from that step on. -/// Then it deletes each directory. A namespace that holds nothing gives success. -pub(super) async fn delete_scope( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - deadline: Duration, -) -> anyhow::Result<()> { - answer_within( - deadline, - storage.delete( - TARGET_LABEL, - "delete_scope", - namespace.clone(), - Path::new(CONFIG_PATH), - ), +/// Deletes the repository and the ledger of the scope. The config file goes first, so the scope +/// holds no repository from that step on. A scope that holds nothing gives success. +pub(super) async fn delete_scope(files: &SnapshotFiles) -> anyhow::Result<()> { + files.delete("delete_scope", Path::new(CONFIG_PATH)).await?; + stream::iter( + LISTING_ORDER + .iter() + .map(Path::new) + .chain(Path::new(LEDGER_PATH).parent()) + .map(Ok), ) - .await?; - let ledger_directory = Path::new(LEDGER_PATH) - .parent() - .map(Path::to_path_buf) - .unwrap_or_default(); - let directories = LISTING_ORDER - .iter() - .map(PathBuf::from) - .chain(std::iter::once(ledger_directory)) - .collect::>(); - stream::iter(directories.iter().map(Ok)) - .try_for_each(|directory| async move { - answer_within( - deadline, - storage.delete_dir(TARGET_LABEL, "delete_scope", namespace.clone(), directory), - ) + .try_for_each(|directory| async move { + files + .delete_dir("delete_scope", directory) .await .map(|_| ()) - }) - .await + }) + .await } #[cfg(test)] diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs index aaff157543..229d3380a9 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs @@ -12,6 +12,7 @@ // See the License for the specific language governing permissions and // limitations under the License. +use super::super::files::SnapshotFiles; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::{copy_scope, delete_scope}; use golem_common::model::environment::EnvironmentId; @@ -36,6 +37,17 @@ const REPOSITORY: [(&str, &str); 6] = [ ("snapshots/0101", "snapshot"), ]; +fn files( + storage: &Arc, + namespace: &BlobStorageNamespace, +) -> SnapshotFiles { + SnapshotFiles { + storage: storage.clone(), + namespace: namespace.clone(), + deadline: DEADLINE, + } +} + fn new_namespace() -> BlobStorageNamespace { BlobStorageNamespace::InitialAgentFiles { environment_id: EnvironmentId(Uuid::new_v4()), @@ -96,14 +108,16 @@ fn owned(blobs: &[(&str, &str)]) -> Vec<(String, String)> { #[test] async fn a_copy_gives_the_target_each_blob_of_the_repository_and_not_the_ledger() { - let storage = InMemoryBlobStorage::new(); + let storage = Arc::new(InMemoryBlobStorage::new()); let (from, to) = (new_namespace(), new_namespace()); - put_all(&storage, &from, &REPOSITORY).await; + put_all(&*storage, &from, &REPOSITORY).await; - copy_scope(&storage, &from, &to, DEADLINE).await.unwrap(); + copy_scope(&files(&storage, &from), &files(&storage, &to)) + .await + .unwrap(); assert_eq!( - (stored(&storage, &to).await, stored(&storage, &from).await), + (stored(&*storage, &to).await, stored(&*storage, &from).await), ( owned( &REPOSITORY @@ -123,7 +137,9 @@ async fn a_copy_writes_the_packs_the_keys_the_index_files_the_snapshot_files_and let (from, to) = (new_namespace(), new_namespace()); put_all(&*storage, &from, &REPOSITORY).await; - copy_scope(&*storage, &from, &to, DEADLINE).await.unwrap(); + copy_scope(&files(&storage, &from), &files(&storage, &to)) + .await + .unwrap(); assert_eq!( storage @@ -149,7 +165,9 @@ async fn a_copy_lists_the_snapshot_files_before_the_index_files_the_keys_and_the let (from, to) = (new_namespace(), new_namespace()); put_all(&*storage, &from, &REPOSITORY).await; - copy_scope(&*storage, &from, &to, DEADLINE).await.unwrap(); + copy_scope(&files(&storage, &from), &files(&storage, &to)) + .await + .unwrap(); assert_eq!( storage @@ -164,10 +182,10 @@ async fn a_copy_lists_the_snapshot_files_before_the_index_files_the_keys_and_the #[test] async fn a_copy_of_a_namespace_without_a_config_copies_nothing() { - let storage = InMemoryBlobStorage::new(); + let storage = Arc::new(InMemoryBlobStorage::new()); let (from, to) = (new_namespace(), new_namespace()); put_all( - &storage, + &*storage, &from, &REPOSITORY .into_iter() @@ -176,9 +194,11 @@ async fn a_copy_of_a_namespace_without_a_config_copies_nothing() { ) .await; - copy_scope(&storage, &from, &to, DEADLINE).await.unwrap(); + copy_scope(&files(&storage, &from), &files(&storage, &to)) + .await + .unwrap(); - assert_eq!(stored(&storage, &to).await, Vec::<(String, String)>::new()); + assert_eq!(stored(&*storage, &to).await, Vec::<(String, String)>::new()); } #[test] @@ -194,7 +214,7 @@ async fn a_copy_that_fails_gives_the_error_and_the_target_has_no_config() { let (from, to) = (new_namespace(), new_namespace()); put_all(&*storage, &from, &REPOSITORY).await; - let copied = copy_scope(&*storage, &from, &to, DEADLINE).await; + let copied = copy_scope(&files(&storage, &from), &files(&storage, &to)).await; assert_eq!( ( @@ -214,17 +234,17 @@ async fn a_copy_that_fails_gives_the_error_and_the_target_has_no_config() { #[test] async fn a_deleted_scope_holds_no_blob_and_another_scope_keeps_its_blobs() { - let storage = InMemoryBlobStorage::new(); + let storage = Arc::new(InMemoryBlobStorage::new()); let (deleted, kept) = (new_namespace(), new_namespace()); - put_all(&storage, &deleted, &REPOSITORY).await; - put_all(&storage, &kept, &REPOSITORY).await; + put_all(&*storage, &deleted, &REPOSITORY).await; + put_all(&*storage, &kept, &REPOSITORY).await; - delete_scope(&storage, &deleted, DEADLINE).await.unwrap(); + delete_scope(&files(&storage, &deleted)).await.unwrap(); assert_eq!( ( - stored(&storage, &deleted).await, - stored(&storage, &kept).await + stored(&*storage, &deleted).await, + stored(&*storage, &kept).await ), (Vec::new(), owned(&REPOSITORY)) ); @@ -237,7 +257,7 @@ async fn a_delete_of_a_scope_deletes_the_config_first() { let namespace = new_namespace(); put_all(&*storage, &namespace, &REPOSITORY).await; - delete_scope(&*storage, &namespace, DEADLINE).await.unwrap(); + delete_scope(&files(&storage, &namespace)).await.unwrap(); assert_eq!( storage @@ -252,20 +272,20 @@ async fn a_delete_of_a_scope_deletes_the_config_first() { #[test] async fn a_delete_of_an_unused_scope_succeeds_and_can_run_again() { - let storage = InMemoryBlobStorage::new(); + let storage = Arc::new(InMemoryBlobStorage::new()); let namespace = new_namespace(); - let first = delete_scope(&storage, &namespace, DEADLINE).await; - put_all(&storage, &namespace, &REPOSITORY).await; - let second = delete_scope(&storage, &namespace, DEADLINE).await; - let third = delete_scope(&storage, &namespace, DEADLINE).await; + let first = delete_scope(&files(&storage, &namespace)).await; + put_all(&*storage, &namespace, &REPOSITORY).await; + let second = delete_scope(&files(&storage, &namespace)).await; + let third = delete_scope(&files(&storage, &namespace)).await; assert_eq!( ( first.is_ok(), second.is_ok(), third.is_ok(), - stored(&storage, &namespace).await + stored(&*storage, &namespace).await ), (true, true, true, Vec::new()) ); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 59028be87c..63849aa514 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -22,8 +22,9 @@ use super::backend::BlobBackend; use super::fault::{Operation, classify, is_file_missing, is_storage_failure, storage_failure}; +use super::files::SnapshotFiles; use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; -use super::publish::{SnapshotFiles, SnapshotStage, StagedSnapshot, publish}; +use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; use super::{ ChangeDetection, PruneReport, PruneSettings, RepackLimits, RepositoryKey, RepositorySettings, @@ -157,10 +158,8 @@ impl RusticSnapshotStore { } /// Stops each operation at its next storage call, and waits until no blocking task and no - /// backend of the store remains. After the call, each operation gives `Storage`. - /// - /// The runtime must not drop before the call returns, because a storage call that waits on the - /// runtime after its time driver stops aborts the process. + /// backend of the store remains; later operations give `Storage`. The runtime must not drop + /// before it returns, because a storage call after its time driver stops aborts the process. pub(crate) async fn shut_down(&self) { self.root.cancel(); self.tracker.close(); @@ -239,13 +238,13 @@ impl RusticSnapshotStore { token: &CancellationToken, freed: u64, ) -> Result<(), SnapshotStoreError> { - let deadline = self.policy.deadline; - let ledger = read_ledger(&*self.storage, &scope.0, deadline) + let files = self.files(scope); + let ledger = read_ledger(&files) .await .map_err(storage_failure)? .with_deleted(freed); if freed > 0 { - write_ledger(&*self.storage, &scope.0, deadline, &ledger) + write_ledger(&files, &ledger) .await .map_err(storage_failure)?; } @@ -262,19 +261,12 @@ impl RusticSnapshotStore { let key = self.key.clone(); let settings = self.policy.prune; let report = self - .blocking(Operation::Repository, move || { - prune(backend, &key, &settings) - }) + .blocking(Operation::Prune, move || prune(backend, &key, &settings)) .await?; let marked_packs = report.as_ref().is_some_and(leaves_marked_packs); - write_ledger( - &*self.storage, - &scope.0, - deadline, - &PruneLedger::after_prune(now, marked_packs), - ) - .await - .map_err(storage_failure) + write_ledger(&files, &PruneLedger::after_prune(now, marked_packs)) + .await + .map_err(storage_failure) } } @@ -406,7 +398,7 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { async fn delete_scope(&self, scope: &SnapshotScope) -> Result<(), SnapshotStoreError> { let _operation = self.start()?; - delete_scope(&*self.storage, &scope.0, self.policy.deadline) + delete_scope(&self.files(scope)) .await .map_err(storage_failure) } @@ -417,7 +409,7 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { to: &SnapshotScope, ) -> Result<(), SnapshotStoreError> { let _operation = self.start()?; - copy_scope(&*self.storage, &from.0, &to.0, self.policy.deadline) + copy_scope(&self.files(from), &self.files(to)) .await .map_err(storage_failure) } @@ -564,11 +556,8 @@ fn stage_save( ))) } -/// Reads each snapshot file of the repository. -/// -/// A failed storage call fails the read. A file that the storage no longer holds is left out, -/// because a delete removed it after the listing. Each other failure counts as a file that failed -/// its integrity check. +/// Reads each snapshot file of the repository. A failed storage call fails the read, a file that a +/// delete removed after the listing is left out, and each other failure counts as a failed check. fn scope_snapshots(repository: &RusticRepository) -> anyhow::Result { repository.list::()?.try_fold( ScopeSnapshots { @@ -655,11 +644,9 @@ fn snapshot_info(snapshot: &SnapshotFile) -> Option { }) } -/// Tells whether a later prune removes packs that this prune leaves marked. -/// -/// A prune marks each pack that holds only unused blobs, and each pack that it repacks. A pack that -/// an earlier prune marked and whose time to stay is not over stays marked. A pack that no index -/// lists is also marked, but the report does not count it. A later due prune removes that pack. +/// Tells whether a later prune removes packs that this prune leaves marked: unused packs, repacked +/// packs, and packs of an earlier prune whose grace period is not over. The report does not count a +/// marked pack that no index lists, so the next due prune removes it. fn leaves_marked_packs(report: &PruneReport) -> bool { report.packs_unused > 0 || report.packs_repacked > 0 || report.marked_packs_kept > 0 } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index e429f0e3ae..1dc3d1deb4 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -17,6 +17,7 @@ //! The contract suite runs on the store with the policy of the configuration. The other tests //! give the store a short or a long deadline and a prune policy that the test controls. +use super::super::files::SnapshotFiles; use super::super::prune::{PruneLedger, read_ledger}; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; @@ -145,10 +146,14 @@ async fn blobs( paths } -async fn ledger(storage: &dyn BlobStorage, scope: &SnapshotScope) -> PruneLedger { - read_ledger(storage, &scope.0, Duration::from_secs(2)) - .await - .unwrap() +async fn ledger(storage: &Arc, scope: &SnapshotScope) -> PruneLedger { + read_ledger(&SnapshotFiles { + storage: storage.clone(), + namespace: scope.0.clone(), + deadline: Duration::from_secs(2), + }) + .await + .unwrap() } /// Waits until the condition holds, or until the limit ends. Gives whether the condition holds. @@ -164,7 +169,7 @@ async fn eventually(condition: impl Fn() -> bool) -> bool { .is_ok() } -/// Runs the operation until the calls of the storage fulfil the condition, and then drops it. +/// Runs the operation until the calls of the storage match the condition, and then drops it. /// Gives the output of the operation when it ends first. async fn drop_when( storage: &ScriptedBlobStorage, @@ -535,7 +540,7 @@ async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { }); let held = eventually(|| index_writes() > before).await; let deleted = store.delete(&scope, &name("p-old")).await; - let pruned_while_held = ledger(&*storage, &scope).await.last_prune_millis.is_some(); + let pruned_while_held = ledger(&storage, &scope).await.last_prune.is_some(); storage.open_gate(); let saved = saving.await.unwrap(); let pruned_again = store.delete(&scope, &name("p-none")).await; @@ -590,11 +595,14 @@ async fn a_restore_whose_pack_reads_fail_gives_a_retryable_storage_error() { async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_denied() { use std::os::unix::fs::PermissionsExt; // SAFETY: `geteuid` has no preconditions. - if unsafe { libc::geteuid() } == 0 { - return; - } + let uid = unsafe { libc::geteuid() }; + assert_ne!( + uid, 0, + "this test needs a user other than root, because permissions do not stop root from a read" + ); + let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( - Arc::new(InMemoryBlobStorage::new()), + storage.clone(), policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), ); let scope = new_scope(); @@ -609,6 +617,10 @@ async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_d matches!(&saved, Err(SnapshotStoreError::Source(error)) if error.kind() == std::io::ErrorKind::PermissionDenied), "{saved:?}" ); + assert_eq!( + blobs(&*storage, &scope.0, "snapshots/").await, + Vec::::new() + ); } #[cfg(target_os = "linux")] @@ -620,6 +632,7 @@ async fn a_restore_that_cannot_set_an_extended_attribute_gives_destination() { let value = vec![b'a'; 6000]; let set = |path: &Path| xattr_set(path, "user.golem-test", &value); let Ok(source) = tempfile::tempdir_in("/dev/shm") else { + println!("SKIPPED: /dev/shm has no directory for the source of the test"); return; }; let file = source.path().join("file.txt"); @@ -627,7 +640,15 @@ async fn a_restore_that_cannot_set_an_extended_attribute_gives_destination() { let probe = Scratch::new(); let probe_file = probe.path().join("probe"); std::fs::write(&probe_file, b"probe").unwrap(); - if set(&file).is_err() || set(&probe_file).is_ok() { + if let Err(error) = set(&file) { + println!("SKIPPED: /dev/shm does not take a user attribute of 6,000 bytes: {error}"); + return; + } + if set(&probe_file).is_ok() { + println!( + "SKIPPED: the destination {} takes a user attribute of 6,000 bytes", + probe.path().display() + ); return; } let store = store( @@ -731,14 +752,14 @@ async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_per let packs_before = blobs(&*storage, &scope.0, "data/").await; store.delete(&scope, &name("p-deleted")).await.unwrap(); - let after_first = ledger(&*storage, &scope).await; + let after_first = ledger(&storage, &scope).await; store.delete(&scope, &name("p-none")).await.unwrap(); let packs_after = blobs(&*storage, &scope.0, "data/").await; assert_eq!( ( after_first.freed_bytes, - after_first.last_prune_millis.is_some(), + after_first.last_prune.is_some(), after_first.awaiting_removal, packs_after.len() < packs_before.len(), packs_after.iter().all(|pack| packs_before.contains(pack)), @@ -768,12 +789,12 @@ async fn a_delete_below_the_threshold_does_not_prune() { let packs_before = blobs(&*storage, &scope.0, "data/").await; store.delete(&scope, &name("p-deleted")).await.unwrap(); - let after = ledger(&*storage, &scope).await; + let after = ledger(&storage, &scope).await; assert_eq!( ( after.freed_bytes > 0, - after.last_prune_millis, + after.last_prune, blobs(&*storage, &scope.0, "data/").await, ), (true, None, packs_before) @@ -800,14 +821,14 @@ async fn no_second_prune_runs_within_the_grace_period() { .await; store.delete(&scope, &name("p-a")).await.unwrap(); - let after_first = ledger(&*storage, &scope).await; + let after_first = ledger(&storage, &scope).await; store.delete(&scope, &name("p-b")).await.unwrap(); - let after_second = ledger(&*storage, &scope).await; + let after_second = ledger(&storage, &scope).await; assert_eq!( ( - after_first.last_prune_millis.is_some(), - after_second.last_prune_millis == after_first.last_prune_millis, + after_first.last_prune.is_some(), + after_second.last_prune == after_first.last_prune, after_second.freed_bytes > 0, ), (true, true, true) @@ -844,10 +865,10 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { refuse.store(true, Ordering::SeqCst); let failed = store.delete(&scope, &name("p-deleted")).await; - let after_failure = ledger(&*storage, &scope).await; + let after_failure = ledger(&storage, &scope).await; refuse.store(false, Ordering::SeqCst); let retried = store.delete(&scope, &name("p-deleted")).await; - let after_retry = ledger(&*storage, &scope).await; + let after_retry = ledger(&storage, &scope).await; assert!( failed.as_ref().is_err_and(|error| is_storage(error, true)), @@ -856,10 +877,10 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { assert_eq!( ( after_failure.freed_bytes > 0, - after_failure.last_prune_millis, + after_failure.last_prune, retried.is_ok(), after_retry.freed_bytes, - after_retry.last_prune_millis.is_some(), + after_retry.last_prune.is_some(), ), (true, None, true, 0, true) ); @@ -1320,7 +1341,7 @@ async fn the_ledger_counts_the_packed_bytes_that_the_deleted_snapshot_added() { store.delete(&scope, &name("p-deleted")).await.unwrap(); assert_eq!( - (added > 1, ledger(&*storage, &scope).await.freed_bytes), + (added > 1, ledger(&storage, &scope).await.freed_bytes), (true, added) ); } @@ -1350,3 +1371,48 @@ async fn a_config_write_that_fails_gives_a_storage_error_with_that_failure() { "{saved:?}" ); } + +#[test] +async fn a_prune_that_fails_without_a_storage_failure_gives_storage_that_is_not_retryable() { + // Packs of zeros with the sizes of the index give the prune a decryption error, not a failed + // storage call. The forget before the prune has succeeded, so the delete gives `Storage`. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path()) + .await + .unwrap(); + let packs = storage + .list_blobs_below("test", "test", scope.0.clone(), Path::new("data")) + .await + .unwrap(); + futures::future::join_all(packs.iter().map(|pack| { + let zeros = vec![0; usize::try_from(pack.size).unwrap()]; + let storage = storage.clone(); + let namespace = scope.0.clone(); + async move { + storage + .put_raw("test", "test", namespace, &pack.path, &zeros) + .await + } + })) + .await + .into_iter() + .collect::>>() + .unwrap(); + + let deleted = store.delete(&scope, &name("p-deleted")).await; + + assert!( + deleted + .as_ref() + .is_err_and(|error| is_storage(error, false)), + "{deleted:?}" + ); +} From 224e1a4baa6802effa3bf1eec75be0137eae3c8f Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 21:30:06 -0700 Subject: [PATCH 21/55] Run the rule of the scripted test storage one time for each call --- .../filesystem_snapshot/rustic/scripted.rs | 37 ++++++++++++++----- 1 file changed, 27 insertions(+), 10 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs index 8c02e95e1b..0c4946aeee 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scripted.rs @@ -100,9 +100,21 @@ impl ScriptedBlobStorage { op_label: &'static str, path: &Path, call: impl Future>, + ) -> anyhow::Result { + self.follow((self.rule)(op_label, path), op_label, path, call) + .await + } + + /// Records the call and does what the script says. The rule runs one time for each call. + async fn follow( + &self, + script: Script, + op_label: &'static str, + path: &Path, + call: impl Future>, ) -> anyhow::Result { self.record(op_label, path); - match (self.rule)(op_label, path) { + match script { Script::Pass => call.await, Script::Refuse => Err(anyhow::anyhow!("the storage refused the call")), Script::LoseTheAnswer => { @@ -137,16 +149,21 @@ impl BlobStorage for ScriptedBlobStorage { namespace: BlobStorageNamespace, path: &Path, ) -> anyhow::Result>> { - if (self.rule)(op_label, path) == Script::Vanish { - self.record(op_label, path); - return Ok(None); + match (self.rule)(op_label, path) { + Script::Vanish => { + self.record(op_label, path); + Ok(None) + } + script => { + self.follow( + script, + op_label, + path, + self.inner.get_raw(target_label, op_label, namespace, path), + ) + .await + } } - self.answer( - op_label, - path, - self.inner.get_raw(target_label, op_label, namespace, path), - ) - .await } async fn get_stream( From a49c636043d1ca21b6dd472ffe09546daa398652 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 21:30:06 -0700 Subject: [PATCH 22/55] Keep the tree packs of one operation in memory, with one read for each pack --- .../src/filesystem_snapshot/rustic/backend.rs | 47 +++- .../rustic/backend/kept.rs | 120 +++++++++ .../rustic/backend/tests.rs | 232 ++++++++++++++++++ .../src/filesystem_snapshot/rustic/tests.rs | 92 ++++++- 4 files changed, 487 insertions(+), 4 deletions(-) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/backend/kept.rs diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index c0afa5cd55..8e66cd1214 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -23,7 +23,10 @@ use super::fault::{BlobCallFailed, ConfigExists, FileMissing, OperationCancelled use super::files::TARGET_LABEL; use super::publish::{SnapshotStage, StagedSnapshot}; use bytes::Bytes; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace, PutIfAbsent}; +use golem_service_base::storage::blob::{ + BlobRangeError, BlobStorage, BlobStorageNamespace, PutIfAbsent, +}; +use kept::KeptPacks; use rustic_core::{ BytesList, ErrorKind, FileType, Id, ReadBackend, RusticError, RusticResult, WriteBackend, }; @@ -38,6 +41,9 @@ use tokio_util::task::task_tracker::TaskTrackerToken; /// The path of the config file of a repository. const CONFIG_PATH: &str = "config"; +/// The largest number of bytes of tree packs that one backend keeps in memory. +const KEPT_PACKS_LIMIT: usize = 32 * 1024 * 1024; + /// A call that the backend makes on the blob storage. #[derive(Clone, Copy, Debug, PartialEq, Eq)] enum StorageCall { @@ -92,6 +98,8 @@ pub(super) struct BlobBackend { stage: Option>, /// Counts the backend as work of a tracker, until the last owner drops the backend. _tracked: Option, + /// The packs of tree blobs that the operation of the backend read. + kept: KeptPacks, } impl BlobBackend { @@ -111,6 +119,15 @@ impl BlobBackend { cancel: CancellationToken::new(), stage: None, _tracked: None, + kept: KeptPacks::new(KEPT_PACKS_LIMIT), + } + } + + /// Gives the backend with another limit of the bytes of the tree packs that it keeps. + pub(super) fn keeping_packs_up_to(self, limit: usize) -> Self { + Self { + kept: KeptPacks::new(limit), + ..self } } @@ -268,7 +285,7 @@ impl ReadBackend for BlobBackend { &self, tpe: FileType, id: &Id, - _cacheable: bool, + cacheable: bool, offset: u32, length: u32, ) -> RusticResult { @@ -276,6 +293,11 @@ impl ReadBackend for BlobBackend { let Some(last) = length.checked_sub(1) else { return Ok(Bytes::new()); }; + // rustic marks the reads of tree blobs as cacheable, and reads each tree blob on its own. + if cacheable && tpe == FileType::Pack { + let pack = self.kept.get_or_read(id, || self.read_full(tpe, id))?; + return range_of(&pack, &path, offset, last); + } let start = u64::from(offset); self.request( StorageCall::ReadRange, @@ -406,6 +428,25 @@ fn join(parts: &[Bytes]) -> Box<[u8]> { .into_boxed_slice() } +/// Gives the bytes from `offset` to `last` of the pack as a slice of the pack. A range outside the +/// pack gives the error of a ranged read outside a blob. +fn range_of(pack: &Bytes, path: &Path, offset: u32, last: u32) -> RusticResult { + let start = u64::from(offset); + let end = start + u64::from(last); + usize::try_from(start) + .ok() + .zip(usize::try_from(end).ok()) + .filter(|(_, end)| *end < pack.len()) + .map(|(start, end)| pack.slice(start..=end)) + .ok_or_else(|| { + storage_error( + StorageCall::ReadRange, + path, + anyhow::Error::new(BlobRangeError { start, end }), + ) + }) +} + /// The error of a file that the blob storage does not hold. fn missing_file(path: &Path) -> Box { RusticError::with_source( @@ -446,5 +487,7 @@ fn storage_error(call: StorageCall, path: &Path, error: anyhow::Error) -> Box, + /// Wakes the threads that wait for the read of a pack when that read ends. + read_ended: Condvar, +} + +#[derive(Default)] +struct State { + packs: HashMap, + bytes: usize, + reading: HashSet, +} + +impl KeptPacks { + /// Gives an empty set that keeps packs up to `limit` bytes in total. + pub(super) fn new(limit: usize) -> Self { + Self { + limit, + state: Mutex::default(), + read_ended: Condvar::new(), + } + } + + /// Gives the kept pack, or reads it with `read`. While one thread reads a pack, the other + /// threads that want it wait for that read, and then take the kept pack or read it again. A + /// pack is kept only when its read succeeds and it fits in the limit. + pub(super) fn get_or_read( + &self, + id: &Id, + read: impl FnOnce() -> RusticResult, + ) -> RusticResult { + let mut state = self + .read_ended + .wait_while(self.state(), |state| state.reading.contains(id)) + .unwrap_or_else(PoisonError::into_inner); + if let Some(pack) = state.packs.get(id) { + return Ok(pack.clone()); + } + state.reading.insert(*id); + drop(state); + let reading = Reading { + kept: self, + id: *id, + }; + let read = read(); + if let Ok(pack) = &read { + reading.keep(pack); + } + read + } + + fn state(&self) -> MutexGuard<'_, State> { + self.state.lock().unwrap_or_else(PoisonError::into_inner) + } +} + +impl Debug for KeptPacks { + fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + let state = self.state(); + formatter + .debug_struct("KeptPacks") + .field("limit", &self.limit) + .field("packs", &state.packs.len()) + .field("bytes", &state.bytes) + .finish() + } +} + +/// The read of one pack by one thread. Its drop ends the read and wakes the waiting threads, also +/// when the read fails or panics. +struct Reading<'a> { + kept: &'a KeptPacks, + id: Id, +} + +impl Reading<'_> { + /// Keeps the pack when it fits in the limit. + fn keep(&self, pack: &Bytes) { + let mut state = self.kept.state(); + let bytes = state.bytes.saturating_add(pack.len()); + if bytes <= self.kept.limit { + state.bytes = bytes; + state.packs.insert(self.id, pack.clone()); + } + } +} + +impl Drop for Reading<'_> { + fn drop(&mut self) { + self.kept.state().reading.remove(&self.id); + self.kept.read_ended.notify_all(); + } +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index 413763cdc5..994c7b88ea 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -478,6 +478,238 @@ fn a_thread_that_is_not_a_thread_of_the_runtime_can_call_the_backend() { assert_eq!(read.ok().flatten(), Some(Bytes::from_static(b"index"))); } +/// The content of the pack of the tests of the kept packs: 100 bytes, each its own offset. +fn pack_content() -> Vec { + (0..100).collect() +} + +/// A backend over a storage that holds one pack at the path of the id `ab`, with the rule of +/// the storage and the limit of the kept packs. The storage records each call. +struct PackFixture { + _runtime: Runtime, + storage: Arc, + backend: Arc, +} + +impl PackFixture { + fn new(limit: usize, rule: impl Fn(&str, &Path) -> Script + Send + Sync + 'static) -> Self { + let runtime = Runtime::new().unwrap(); + let inner = Arc::new(InMemoryBlobStorage::new()); + let namespace = new_namespace(); + runtime + .block_on(inner.put_raw( + "test", + "test", + namespace.clone(), + Path::new(&format!("data/ab/{}", "ab".repeat(32))), + &pack_content(), + )) + .unwrap(); + let storage = ScriptedBlobStorage::new(inner, rule); + let backend = BlobBackend::new( + storage.clone(), + namespace, + runtime.handle().clone(), + STORAGE_CALL_DEADLINE, + ) + .keeping_packs_up_to(limit); + Self { + _runtime: runtime, + storage, + backend: Arc::new(backend), + } + } + + /// Gives the operation label of each call on the pack. + fn pack_calls(&self) -> Vec<&'static str> { + self.storage + .calls() + .into_iter() + .filter(|(_, path)| path.starts_with("data/")) + .map(|(op_label, _)| op_label) + .collect() + } +} + +/// Reads the range of the pack as a range of tree blobs, which rustic marks as cacheable. +fn tree_range(backend: &BlobBackend, offset: u32, length: u32) -> RusticResult { + backend.read_partial(FileType::Pack, &id("ab"), true, offset, length) +} + +#[test] +fn a_later_range_of_a_kept_pack_makes_no_storage_call() { + let fixture = PackFixture::new(1024, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + ( + tree_range(&backend, 0, 10).ok(), + tree_range(&backend, 20, 10).ok(), + tree_range(&backend, 90, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)), + Some(Bytes::from_iter(90..100)), + )), + vec!["read"] + ) + ); +} + +#[test] +fn a_range_that_is_not_cacheable_is_a_ranged_read_each_time() { + let fixture = PackFixture::new(1024, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + [(0, 10), (20, 10)].map(|(offset, length)| { + backend + .read_partial(FileType::Pack, &id("ab"), false, offset, length) + .ok() + }) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some([ + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)) + ]), + vec!["read_range", "read_range"] + ) + ); +} + +#[test] +fn two_threads_that_miss_one_pack_make_one_storage_read() { + // The first read waits at the gate. The second thread starts while it waits, and the gate + // opens only after the second thread had time to ask for the pack. + let fixture = PackFixture::new(1024, |op_label, _| { + if op_label == "read" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let first = std::thread::spawn({ + let backend = fixture.backend.clone(); + move || tree_range(&backend, 0, 10).ok() + }); + let first_read_started = (0..1000).any(|_| { + std::thread::sleep(Duration::from_millis(10)); + !fixture.pack_calls().is_empty() + }); + let second = std::thread::spawn({ + let backend = fixture.backend.clone(); + move || tree_range(&backend, 50, 10).ok() + }); + std::thread::sleep(Duration::from_millis(200)); + fixture.storage.open_gate(); + + assert_eq!( + ( + first_read_started, + first.join().ok().flatten(), + second.join().ok().flatten(), + fixture.pack_calls() + ), + ( + true, + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(50..60)), + vec!["read"] + ) + ); +} + +#[test] +fn a_pack_over_the_limit_is_read_again_at_its_next_range() { + let fixture = PackFixture::new(99, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + ( + tree_range(&backend, 0, 10).ok(), + tree_range(&backend, 20, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + Some(Bytes::from_iter(0..10)), + Some(Bytes::from_iter(20..30)) + )), + vec!["read", "read"] + ) + ); +} + +#[test] +fn a_failed_read_of_a_pack_is_not_kept_and_the_next_range_reads_again() { + let refused = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let fixture = PackFixture::new(1024, { + let refused = refused.clone(); + move |op_label, _| { + if op_label == "read" && !refused.swap(true, std::sync::atomic::Ordering::SeqCst) { + Script::Refuse + } else { + Script::Pass + } + } + }); + let backend = fixture.backend.clone(); + + let ranges = within_limit(move || { + ( + tree_range(&backend, 0, 10).is_err(), + tree_range(&backend, 20, 10).ok(), + tree_range(&backend, 40, 10).ok(), + ) + }); + + assert_eq!( + (ranges, fixture.pack_calls()), + ( + Some(( + true, + Some(Bytes::from_iter(20..30)), + Some(Bytes::from_iter(40..50)) + )), + vec!["read", "read"] + ) + ); +} + +#[test] +fn a_range_outside_a_kept_pack_gives_the_error_of_a_ranged_read_outside_a_blob() { + let fixture = PackFixture::new(1024, |_, _| Script::Pass); + let backend = fixture.backend.clone(); + + let outside = within_limit(move || { + [(90, 11), (100, 1), (u32::MAX, 1)].map(|(offset, length)| { + tree_range(&backend, offset, length).err().map(|error| { + ( + text_of(&error).contains("is not in the blob"), + classify(Operation::Restore, anyhow::Error::new(error)) + .to_string() + .contains("storage"), + ) + }) + }) + }); + + assert_eq!(outside, Some([Some((true, true)); 3])); +} + #[test] fn a_tracked_backend_counts_in_its_tracker_until_it_drops() { let fixture = Fixture::new(); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index 6d9fa71445..a50db65f08 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -20,6 +20,7 @@ use super::backend::BlobBackend; use super::holding::{holding_storage, reached_deadline}; +use super::scripted::{Script, ScriptedBlobStorage}; use super::{ ChangeDetection, Chunking, Compression, OperationPhase, PruneSettings, RepackLimits, Repository, RepositoryKey, RepositorySettings, SaveSettings, backup_options, config_options, @@ -140,6 +141,52 @@ async fn data_packs(storage: &Arc, scope: &SnapshotScope) - .unwrap() } +/// Gives the id in hex of each pack of tree blobs in the repository of the scope. +async fn tree_packs(storage: &Arc, scope: &SnapshotScope) -> Box<[Box]> { + with_existing_repository( + storage.clone(), + scope, + STORAGE_CALL_DEADLINE, + |repository| { + let indexes = repository + .stream_files::()? + .collect::>>()?; + Ok(indexes + .into_iter() + .flat_map(|(_, index)| index.packs) + .filter(|pack| pack.blob_type() == BlobType::Tree) + .map(|pack| Box::from(pack.id.to_hex().as_str())) + .collect()) + }, + ) + .await + .unwrap() +} + +/// Writes a tree of `count` directories, each with one small file, into a new directory. +fn many_directories_tree(count: usize) -> Scratch { + let tree = Scratch::new(); + let names = (0..count) + .flat_map(|index| [format!("dir-{index}"), format!("dir-{index}/file.txt")]) + .collect::>(); + let entries = names + .iter() + .map(|name| { + let spec = if name.ends_with(".txt") { + Spec::File { + content: Box::from(name.as_bytes()), + mode: 0o644, + } + } else { + Spec::Directory { mode: 0o755 } + }; + (name.as_str(), spec) + }) + .collect::>(); + write_tree(tree.path(), &entries); + tree +} + /// Prunes the repository of the scope with the options, on a blocking thread. Each call on the /// storage waits for at most `deadline`. async fn prune( @@ -239,6 +286,46 @@ async fn a_restore_report_gives_each_phase_of_the_restore_in_order() { ); } +#[test] +async fn a_restore_reads_each_tree_pack_one_time_in_full_and_no_range_of_a_tree_pack() { + let inner = Arc::new(InMemoryBlobStorage::new()); + let scope = new_scope(); + let tree = many_directories_tree(60); + repository(&inner, &scope) + .save(&name("first"), tree.path()) + .await + .unwrap(); + let tree_packs = tree_packs(&inner, &scope).await; + let storage = ScriptedBlobStorage::new(inner.clone(), |_, _| Script::Pass); + let into = Scratch::new(); + + Repository::new(storage.clone(), scope.clone(), key(), STORAGE_CALL_DEADLINE) + .restore(&name("first"), into.path(), None) + .await + .unwrap(); + let calls_on_tree_packs = |op: &str| { + tree_packs + .iter() + .map(|pack| { + storage + .calls() + .iter() + .filter(|(op_label, path)| *op_label == op && path.ends_with(&**pack)) + .count() + }) + .collect::>() + }; + + assert_eq!( + ( + calls_on_tree_packs("read"), + calls_on_tree_packs("read_range").iter().sum::(), + listing(into.path()) + ), + (vec![1; tree_packs.len()], 0, listing(tree.path())) + ); +} + #[test] async fn a_second_save_has_the_first_as_parent_and_reads_only_the_changed_file() { let storage = Arc::new(InMemoryBlobStorage::new()); @@ -785,7 +872,7 @@ async fn a_restore_whose_data_pack_reads_get_no_answer_fails_and_stops_its_threa } #[test] -async fn a_prune_whose_pack_reads_get_no_answer_fails_and_stops_its_threads() { +async fn a_prune_whose_tree_pack_reads_get_no_answer_fails_and_stops_its_threads() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); let tree = fixture_tree(); @@ -793,8 +880,9 @@ async fn a_prune_whose_pack_reads_get_no_answer_fails_and_stops_its_threads() { .save(&name("first"), tree.path()) .await .unwrap(); + // A prune reads the trees of the snapshots, and each tree read is a full read of a pack. let (storage, _gate, dropped) = holding_storage(inner, |op_label, path| { - op_label == "read_range" && path.starts_with("data") + op_label == "read" && path.starts_with("data") }); let pruned = tokio::time::timeout( From ccd253e260ab8122bd82aae946271018ed0f9a32 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:38:09 -0700 Subject: [PATCH 23/55] Run saves and prunes at nice 19 on their own threads --- golem-worker-executor/Cargo.toml | 5 +- .../src/filesystem_snapshot/rustic/mod.rs | 1 + .../filesystem_snapshot/rustic/priority.rs | 109 +++++++++++ .../rustic/priority/tests.rs | 57 ++++++ .../src/filesystem_snapshot/rustic/store.rs | 9 +- .../filesystem_snapshot/rustic/store/tests.rs | 172 ++++++++++++++++++ 6 files changed, 348 insertions(+), 5 deletions(-) create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs create mode 100644 golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs diff --git a/golem-worker-executor/Cargo.toml b/golem-worker-executor/Cargo.toml index cd1e5ff1ec..d54d18649a 100644 --- a/golem-worker-executor/Cargo.toml +++ b/golem-worker-executor/Cargo.toml @@ -13,7 +13,7 @@ autotests = false [features] test-utils = [] -fs-snapshot-benchmark = ["dep:clap", "dep:rayon"] +fs-snapshot-benchmark = ["dep:clap"] [lib] path = "src/lib.rs" @@ -96,7 +96,7 @@ prometheus = { workspace = true } prost = { workspace = true } prost-types = { workspace = true } rand = { workspace = true } -rayon = { workspace = true, optional = true } +rayon = { workspace = true } regex = { workspace = true } ringbuf = { workspace = true } rustic_core = { workspace = true } @@ -169,7 +169,6 @@ goldenfile = { workspace = true } pretty_assertions = { workspace = true, features = [ "unstable" ] } proptest = { workspace = true } rand = { workspace = true } -rayon = { workspace = true } redis = { workspace = true } serde_json = { workspace = true } test-r = { workspace = true } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index cd02c37a1d..ddcebdba84 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -21,6 +21,7 @@ mod backend; mod fault; mod files; +mod priority; mod prune; mod publish; mod scope; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs new file mode 100644 index 0000000000..31b746eaa4 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs @@ -0,0 +1,109 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Saves and prunes run at a low CPU priority, so the work of the agents comes first. +//! +//! A thread without privilege cannot raise its priority again after it lowers it. So the work runs +//! on a new thread that ends with the work, and no thread of a pool gets the low priority. + +use std::sync::{Arc, Mutex, PoisonError}; +use tracing::warn; + +/// The nice value of the threads of a save or a prune. +#[cfg(target_os = "linux")] +pub(super) const LOW_PRIORITY: i32 = 19; + +/// Runs the work at nice 19 on a new thread with the name, and waits for it. The threads that the +/// work starts get the same nice value. On a platform other than Linux the work runs as it is. +pub(super) fn at_low_priority( + name: &str, + work: impl FnOnce() -> anyhow::Result + Send + 'static, +) -> anyhow::Result { + #[cfg(target_os = "linux")] + { + // The global rayon pool of rustic starts at its first use. It starts here, so its threads + // keep the normal priority and a restore that uses them does not run at nice 19. + let _ = rayon::current_num_threads(); + on_own_thread(name, work, lower_own_priority) + } + #[cfg(not(target_os = "linux"))] + { + let _ = name; + work() + } +} + +/// Runs the work on a new thread that first calls `lower`, and waits for it. A failed `lower` or a +/// failed start of the thread gives a warning, and the work runs at the normal priority. +fn on_own_thread( + name: &str, + work: impl FnOnce() -> anyhow::Result + Send + 'static, + lower: fn() -> std::io::Result<()>, +) -> anyhow::Result { + // The work waits in a slot, so the calling thread can still run it when no thread starts. + let slot = Arc::new(Mutex::new(Some(work))); + let spawned = std::thread::Builder::new().name(name.to_string()).spawn({ + let slot = slot.clone(); + move || { + if let Err(error) = lower() { + warn!( + error = %error, + "Failed to lower the CPU priority of filesystem snapshot work, so it runs at the normal priority" + ); + } + take(&slot).map_or_else(|| Err(anyhow::anyhow!("the work was taken")), |work| work()) + } + }); + match spawned { + Ok(thread) => thread + .join() + .map_err(|_| anyhow::anyhow!("the thread of the filesystem snapshot work panicked"))?, + Err(error) => { + warn!( + error = %error, + "Failed to start a thread for filesystem snapshot work, so it runs at the normal priority" + ); + take(&slot).map_or_else(|| Err(anyhow::anyhow!("the work was taken")), |work| work()) + } + } +} + +fn take(slot: &Mutex>) -> Option { + slot.lock().unwrap_or_else(PoisonError::into_inner).take() +} + +/// Gives the calling thread the nice value 19. +#[cfg(target_os = "linux")] +fn lower_own_priority() -> std::io::Result<()> { + // SAFETY: `gettid` has no preconditions. + let thread = unsafe { libc::gettid() }; + let thread = libc::id_t::try_from(thread).map_err(std::io::Error::other)?; + // SAFETY: `setpriority` only reads its arguments. + let result = unsafe { libc::setpriority(libc::PRIO_PROCESS, thread, LOW_PRIORITY) }; + if result == 0 { + Ok(()) + } else { + Err(std::io::Error::last_os_error()) + } +} + +/// Gives the nice value of the calling thread. +#[cfg(all(test, target_os = "linux"))] +pub(super) fn own_nice() -> i32 { + // SAFETY: `gettid` has no preconditions, and `getpriority` only reads its arguments. + unsafe { libc::getpriority(libc::PRIO_PROCESS, libc::gettid() as libc::id_t) } +} + +#[cfg(all(test, target_os = "linux"))] +mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs new file mode 100644 index 0000000000..0a08a433a0 --- /dev/null +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs @@ -0,0 +1,57 @@ +// Copyright 2024-2026 Golem Cloud +// +// Licensed under the Golem Source License v1.1 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://license.golem.cloud/LICENSE +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use super::{LOW_PRIORITY, at_low_priority, on_own_thread, own_nice}; +use pretty_assertions::assert_eq; +use test_r::test; + +#[test] +fn work_at_low_priority_runs_at_nice_19_on_its_own_thread_and_the_caller_keeps_its_priority() { + let before = own_nice(); + + let inside = at_low_priority("fs-snap-test", || { + Ok(( + own_nice(), + std::thread::current().name().map(str::to_string), + )) + }); + + assert_eq!( + (inside.ok(), own_nice()), + ( + Some((LOW_PRIORITY, Some("fs-snap-test".to_string()))), + before + ) + ); +} + +#[test] +fn work_whose_priority_cannot_be_lowered_still_runs_at_the_normal_priority() { + let before = own_nice(); + + let done = on_own_thread( + "fs-snap-test", + || Ok(own_nice()), + || Err(std::io::Error::other("the priority cannot change here")), + ); + + assert_eq!(done.ok(), Some(before)); +} + +#[test] +fn a_panic_of_the_work_gives_an_error() { + let done = on_own_thread::<()>("fs-snap-test", || panic!("the work panics"), || Ok(())); + + assert!(done.is_err()); +} diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 63849aa514..5c1a30a753 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -23,6 +23,7 @@ use super::backend::BlobBackend; use super::fault::{Operation, classify, is_file_missing, is_storage_failure, storage_failure}; use super::files::SnapshotFiles; +use super::priority::at_low_priority; use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -261,7 +262,9 @@ impl RusticSnapshotStore { let key = self.key.clone(); let settings = self.policy.prune; let report = self - .blocking(Operation::Prune, move || prune(backend, &key, &settings)) + .blocking(Operation::Prune, move || { + at_low_priority("fs-snap-prune", move || prune(backend, &key, &settings)) + }) .await?; let marked_packs = report.as_ref().is_some_and(leaves_marked_packs); write_ledger(&files, &PruneLedger::after_prune(now, marked_packs)) @@ -288,7 +291,9 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { let tree: Box = tree.into(); let staged = self .blocking(Operation::Save, move || { - stage_save(backend, &stage, &key, &policy, &name, &tree) + at_low_priority("fs-snap-save", move || { + stage_save(backend, &stage, &key, &policy, &name, &tree) + }) }) .await?; let (staged, info) = staged.ok_or(SnapshotStoreError::AlreadyExists)?; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 1dc3d1deb4..288a66fd16 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1416,3 +1416,175 @@ async fn a_prune_that_fails_without_a_storage_failure_gives_storage_that_is_not_ "{deleted:?}" ); } + +/// The operation label, the path and the nice value of the calling thread of each storage call. +#[cfg(target_os = "linux")] +type NiceCalls = Arc>>; + +/// A storage that records the nice value of the thread of each call. +#[cfg(target_os = "linux")] +fn nice_recording_storage() -> (Arc, NiceCalls) { + let calls = NiceCalls::default(); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let calls = calls.clone(); + move |op_label, path| { + calls + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .push(( + op_label.to_string(), + path.display().to_string(), + super::super::priority::own_nice(), + )); + Script::Pass + } + }); + (storage, calls) +} + +/// Takes the recorded calls with the operation label. +#[cfg(target_os = "linux")] +fn taken_calls(calls: &NiceCalls, op_label: &str) -> Vec<(String, i32)> { + std::mem::take( + &mut *calls + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner), + ) + .into_iter() + .filter(|(op, _, _)| op == op_label) + .map(|(_, path, nice)| (path, nice)) + .collect() +} + +#[cfg(target_os = "linux")] +#[test] +async fn the_writes_of_a_save_run_at_nice_19() { + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = fixture_tree(); + + store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + let writes = taken_calls(&calls, "write"); + + assert_eq!( + ( + writes.is_empty(), + writes + .iter() + .filter(|(_, nice)| *nice != 19) + .collect::>() + ), + (false, Vec::<&(String, i32)>::new()) + ); +} + +#[cfg(target_os = "linux")] +#[test] +async fn the_writes_of_a_prune_run_at_nice_19() { + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, 1, Duration::ZERO)); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path()) + .await + .unwrap(); + taken_calls(&calls, "write"); + + store.delete(&scope, &name("p-deleted")).await.unwrap(); + let writes = taken_calls(&calls, "write"); + + assert_eq!( + ( + writes.is_empty(), + writes + .iter() + .filter(|(_, nice)| *nice != 19) + .collect::>() + ), + (false, Vec::<&(String, i32)>::new()) + ); +} + +#[cfg(target_os = "linux")] +#[test] +async fn the_storage_calls_of_a_restore_run_at_the_nice_value_of_the_process() { + let process_nice = super::super::priority::own_nice(); + let (storage, calls) = nice_recording_storage(); + let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let scope = new_scope(); + let tree = fixture_tree(); + store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + std::mem::take(&mut *calls.lock().unwrap()); + + let restored = restored_listing(&store, &scope, &name("p-1")).await; + let recorded = std::mem::take(&mut *calls.lock().unwrap()); + + assert_eq!( + ( + restored.ok(), + recorded.is_empty(), + recorded + .iter() + .filter(|(_, _, nice)| *nice != process_nice) + .collect::>() + ), + ( + Some(listing(tree.path())), + false, + Vec::<&(String, String, i32)>::new() + ) + ); +} + +#[cfg(target_os = "linux")] +#[test] +async fn after_saves_and_prunes_the_pools_keep_the_nice_value_of_the_process() { + // All tasks wait for each other, so each runs on its own thread of the blocking pool, and the + // idle threads that ran the saves and the prune are among them. + const TASKS: usize = 16; + let process_nice = super::super::priority::own_nice(); + let store = store( + Arc::new(InMemoryBlobStorage::new()), + policy(LONG_DEADLINE, 1, Duration::ZERO), + ); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); + store + .save(&scope, &name("p-deleted"), deleted_tree.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path()) + .await + .unwrap(); + store.delete(&scope, &name("p-deleted")).await.unwrap(); + + let barrier = Arc::new(std::sync::Barrier::new(TASKS)); + let blocking = futures::future::join_all((0..TASKS).map(|_| { + let barrier = barrier.clone(); + tokio::task::spawn_blocking(move || { + barrier.wait(); + super::super::priority::own_nice() + }) + })) + .await + .into_iter() + .map(Result::unwrap) + .collect::>(); + let rayon = rayon::broadcast(|_| super::super::priority::own_nice()); + + assert_eq!( + ( + blocking.iter().all(|nice| *nice == process_nice), + rayon.iter().all(|nice| *nice == process_nice), + ), + (true, true), + "{blocking:?} {rayon:?}" + ); +} From a359a2bab6d2fd97b879bdb8180f9eb29044e06f Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:02:53 -0700 Subject: [PATCH 24/55] Give each save and prune a rayon pool of its own, and start the global rayon pool when the store is made --- .../filesystem_snapshot/rustic/priority.rs | 148 ++++++++++++------ .../rustic/priority/tests.rs | 78 ++++++--- .../src/filesystem_snapshot/rustic/store.rs | 14 +- .../filesystem_snapshot/rustic/store/tests.rs | 128 ++++++++++++++- 4 files changed, 293 insertions(+), 75 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs index 31b746eaa4..e03448bef9 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs @@ -15,72 +15,119 @@ //! Saves and prunes run at a low CPU priority, so the work of the agents comes first. //! //! A thread without privilege cannot raise its priority again after it lowers it. So the work runs -//! on a new thread that ends with the work, and no thread of a pool gets the low priority. +//! on a new thread, with a rayon pool of its own, and both end with the work. +use rayon::{ThreadPool, ThreadPoolBuildError, ThreadPoolBuilder}; +use std::num::NonZeroUsize; use std::sync::{Arc, Mutex, PoisonError}; use tracing::warn; /// The nice value of the threads of a save or a prune. -#[cfg(target_os = "linux")] pub(super) const LOW_PRIORITY: i32 = 19; -/// Runs the work at nice 19 on a new thread with the name, and waits for it. The threads that the -/// work starts get the same nice value. On a platform other than Linux the work runs as it is. -pub(super) fn at_low_priority( - name: &str, - work: impl FnOnce() -> anyhow::Result + Send + 'static, -) -> anyhow::Result { - #[cfg(target_os = "linux")] - { - // The global rayon pool of rustic starts at its first use. It starts here, so its threads - // keep the normal priority and a restore that uses them does not run at nice 19. - let _ = rayon::current_num_threads(); - on_own_thread(name, work, lower_own_priority) +/// How the store runs work at a low priority: the thread count of the rayon pool of the work, and +/// the two steps that a test can replace. +#[derive(Clone, Copy, Debug)] +pub(super) struct LowPriority { + /// The threads of the rayon pool of the work. `None` is the default count of rayon. + pub(super) threads: Option, + /// Gives the calling thread the low priority. + pub(super) lower: fn() -> std::io::Result<()>, + /// Builds the rayon pool of the work, with the name and the thread count. + pub(super) build_pool: + fn(&str, Option) -> Result, +} + +impl LowPriority { + /// Gives the steps of the platform, with a rayon pool of `threads` threads. + pub(super) fn new(threads: Option) -> Self { + Self { + threads, + lower: lower_own_priority, + build_pool, + } } - #[cfg(not(target_os = "linux"))] - { - let _ = name; - work() + + /// Runs the work at nice 19 on a new thread with the name, inside a new rayon pool, and waits + /// for it. The threads that the work starts get the same nice value. On a platform other than + /// Linux the work runs as it is. + pub(super) fn run( + self, + name: &str, + work: impl FnOnce() -> anyhow::Result + Send + 'static, + ) -> anyhow::Result { + if cfg!(target_os = "linux") { + self.on_own_thread(name, work) + } else { + work() + } } -} -/// Runs the work on a new thread that first calls `lower`, and waits for it. A failed `lower` or a -/// failed start of the thread gives a warning, and the work runs at the normal priority. -fn on_own_thread( - name: &str, - work: impl FnOnce() -> anyhow::Result + Send + 'static, - lower: fn() -> std::io::Result<()>, -) -> anyhow::Result { - // The work waits in a slot, so the calling thread can still run it when no thread starts. - let slot = Arc::new(Mutex::new(Some(work))); - let spawned = std::thread::Builder::new().name(name.to_string()).spawn({ - let slot = slot.clone(); - move || { - if let Err(error) = lower() { + /// Runs the work on a new thread that first lowers its priority and builds the pool. A failure + /// of a step gives a warning, and the work runs without that step. + fn on_own_thread( + self, + name: &str, + work: impl FnOnce() -> anyhow::Result + Send + 'static, + ) -> anyhow::Result { + // The work waits in a slot, so the calling thread can still run it when no thread starts. + let slot = Arc::new(Mutex::new(Some(work))); + let pool_name = name.to_string(); + let spawned = std::thread::Builder::new().name(name.to_string()).spawn({ + let slot = slot.clone(); + move || { + if let Err(error) = (self.lower)() { + warn!( + error = %error, + "Failed to lower the CPU priority of filesystem snapshot work, so it runs at the normal priority" + ); + } + match (self.build_pool)(&pool_name, self.threads) { + Ok(pool) => pool.install(|| run_taken(&slot)), + Err(error) => { + warn!( + error = %error, + "Failed to build the thread pool of filesystem snapshot work, so its parallel parts use the global pool" + ); + run_taken(&slot) + } + } + } + }); + match spawned { + Ok(thread) => thread.join().map_err(|_| { + anyhow::anyhow!("the thread of the filesystem snapshot work panicked") + })?, + Err(error) => { warn!( error = %error, - "Failed to lower the CPU priority of filesystem snapshot work, so it runs at the normal priority" + "Failed to start a thread for filesystem snapshot work, so it runs at the normal priority" ); + run_taken(&slot) } - take(&slot).map_or_else(|| Err(anyhow::anyhow!("the work was taken")), |work| work()) - } - }); - match spawned { - Ok(thread) => thread - .join() - .map_err(|_| anyhow::anyhow!("the thread of the filesystem snapshot work panicked"))?, - Err(error) => { - warn!( - error = %error, - "Failed to start a thread for filesystem snapshot work, so it runs at the normal priority" - ); - take(&slot).map_or_else(|| Err(anyhow::anyhow!("the work was taken")), |work| work()) } } } -fn take(slot: &Mutex>) -> Option { - slot.lock().unwrap_or_else(PoisonError::into_inner).take() +/// Takes the work out of the slot and runs it. +fn run_taken anyhow::Result>(slot: &Mutex>) -> anyhow::Result { + let work = slot.lock().unwrap_or_else(PoisonError::into_inner).take(); + work.map_or_else( + || Err(anyhow::anyhow!("the work already ran")), + |work| work(), + ) +} + +/// Builds a rayon pool whose threads get the nice value of the calling thread. +fn build_pool( + name: &str, + threads: Option, +) -> Result { + let name = name.to_string(); + ThreadPoolBuilder::new() + .num_threads(threads.map_or(0, NonZeroUsize::get)) + .thread_name(move |index| format!("{name}-{index}")) + .build() } /// Gives the calling thread the nice value 19. @@ -98,6 +145,11 @@ fn lower_own_priority() -> std::io::Result<()> { } } +#[cfg(not(target_os = "linux"))] +fn lower_own_priority() -> std::io::Result<()> { + Ok(()) +} + /// Gives the nice value of the calling thread. #[cfg(all(test, target_os = "linux"))] pub(super) fn own_nice() -> i32 { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs index 0a08a433a0..6610032d75 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs @@ -12,46 +12,88 @@ // See the License for the specific language governing permissions and // limitations under the License. -use super::{LOW_PRIORITY, at_low_priority, on_own_thread, own_nice}; +use super::{LOW_PRIORITY, LowPriority, own_nice}; use pretty_assertions::assert_eq; +use rayon::{ThreadPool, ThreadPoolBuildError, ThreadPoolBuilder}; +use std::num::NonZeroUsize; use test_r::test; +/// What the work sees: its nice value, its thread name, whether it runs in a rayon pool, and +/// the thread count of the current rayon pool. +fn seen() -> anyhow::Result<(i32, Option, bool, usize)> { + Ok(( + own_nice(), + std::thread::current().name().map(str::to_string), + rayon::current_thread_index().is_some(), + rayon::current_num_threads(), + )) +} + +/// A pool builder that cannot start a thread. +fn no_pool(_: &str, _: Option) -> Result { + ThreadPoolBuilder::new() + .num_threads(1) + .spawn_handler(|_| Err(std::io::Error::other("no thread can start here"))) + .build() +} + #[test] -fn work_at_low_priority_runs_at_nice_19_on_its_own_thread_and_the_caller_keeps_its_priority() { +fn work_at_low_priority_runs_at_nice_19_in_a_pool_of_its_own_and_the_caller_keeps_its_priority() { let before = own_nice(); - let inside = at_low_priority("fs-snap-test", || { - Ok(( - own_nice(), - std::thread::current().name().map(str::to_string), - )) - }); + let inside = LowPriority::new(NonZeroUsize::new(3)).run("fs-snap-test", seen); assert_eq!( - (inside.ok(), own_nice()), ( - Some((LOW_PRIORITY, Some("fs-snap-test".to_string()))), - before - ) + inside.ok().map(|(nice, name, in_pool, threads)| ( + nice, + name.is_some_and(|name| name.starts_with("fs-snap-test-")), + in_pool, + threads + )), + own_nice() + ), + (Some((LOW_PRIORITY, true, true, 3)), before) ); } #[test] fn work_whose_priority_cannot_be_lowered_still_runs_at_the_normal_priority() { let before = own_nice(); + let low_priority = LowPriority { + lower: || Err(std::io::Error::other("the priority cannot change here")), + ..LowPriority::new(NonZeroUsize::new(2)) + }; + + let inside = low_priority.run("fs-snap-test", seen); - let done = on_own_thread( - "fs-snap-test", - || Ok(own_nice()), - || Err(std::io::Error::other("the priority cannot change here")), + assert_eq!( + inside.ok().map(|(nice, _, in_pool, _)| (nice, in_pool)), + Some((before, true)) ); +} + +#[test] +fn work_without_its_pool_still_runs_at_nice_19_on_its_own_thread() { + let low_priority = LowPriority { + build_pool: no_pool, + ..LowPriority::new(NonZeroUsize::new(2)) + }; - assert_eq!(done.ok(), Some(before)); + let inside = low_priority.run("fs-snap-test", seen); + + assert_eq!( + inside + .ok() + .map(|(nice, name, in_pool, _)| (nice, name, in_pool)), + Some((LOW_PRIORITY, Some("fs-snap-test".to_string()), false)) + ); } #[test] fn a_panic_of_the_work_gives_an_error() { - let done = on_own_thread::<()>("fs-snap-test", || panic!("the work panics"), || Ok(())); + let done = LowPriority::new(NonZeroUsize::new(1)) + .run::<()>("fs-snap-test", || panic!("the work panics")); assert!(done.is_err()); } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 5c1a30a753..a95f140817 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -23,7 +23,7 @@ use super::backend::BlobBackend; use super::fault::{Operation, classify, is_file_missing, is_storage_failure, storage_failure}; use super::files::SnapshotFiles; -use super::priority::at_low_priority; +use super::priority::LowPriority; use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -129,6 +129,8 @@ pub(crate) struct RusticSnapshotStore { root: CancellationToken, /// Counts the blocking tasks, the backends and the deletes of dropped publishes. tracker: TaskTracker, + /// Runs saves and prunes at a low priority. + low_priority: LowPriority, } impl RusticSnapshotStore { @@ -149,12 +151,16 @@ impl RusticSnapshotStore { key: RepositoryKey, policy: StorePolicy, ) -> Self { + // The global rayon pool starts at its first use, and its threads keep the priority of the + // thread that starts it. It starts here, at the normal priority, before a save or a prune. + let _ = rayon::current_num_threads(); Self { storage, key, policy, root: CancellationToken::new(), tracker: TaskTracker::new(), + low_priority: LowPriority::new(policy.save.threads), } } @@ -261,9 +267,10 @@ impl RusticSnapshotStore { let backend = Arc::new(self.backend(scope, token)?); let key = self.key.clone(); let settings = self.policy.prune; + let low_priority = self.low_priority; let report = self .blocking(Operation::Prune, move || { - at_low_priority("fs-snap-prune", move || prune(backend, &key, &settings)) + low_priority.run("fs-snap-prune", move || prune(backend, &key, &settings)) }) .await?; let marked_packs = report.as_ref().is_some_and(leaves_marked_packs); @@ -289,9 +296,10 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { let policy = self.policy; let name = name.clone(); let tree: Box = tree.into(); + let low_priority = self.low_priority; let staged = self .blocking(Operation::Save, move || { - at_low_priority("fs-snap-save", move || { + low_priority.run("fs-snap-save", move || { stage_save(backend, &stage, &key, &policy, &name, &tree) }) }) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 288a66fd16..1cfa5704d6 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1417,9 +1417,10 @@ async fn a_prune_that_fails_without_a_storage_failure_gives_storage_that_is_not_ ); } -/// The operation label, the path and the nice value of the calling thread of each storage call. +/// The operation label, the path, and the name and the nice value of the calling thread of each +/// storage call. #[cfg(target_os = "linux")] -type NiceCalls = Arc>>; +type NiceCalls = Arc>>; /// A storage that records the nice value of the thread of each call. #[cfg(target_os = "linux")] @@ -1434,6 +1435,10 @@ fn nice_recording_storage() -> (Arc, NiceCalls) { .push(( op_label.to_string(), path.display().to_string(), + std::thread::current() + .name() + .unwrap_or_default() + .to_string(), super::super::priority::own_nice(), )); Script::Pass @@ -1451,8 +1456,8 @@ fn taken_calls(calls: &NiceCalls, op_label: &str) -> Vec<(String, i32)> { .unwrap_or_else(std::sync::PoisonError::into_inner), ) .into_iter() - .filter(|(op, _, _)| op == op_label) - .map(|(_, path, nice)| (path, nice)) + .filter(|(op, _, _, _)| op == op_label) + .map(|(_, path, _, nice)| (path, nice)) .collect() } @@ -1531,13 +1536,13 @@ async fn the_storage_calls_of_a_restore_run_at_the_nice_value_of_the_process() { recorded.is_empty(), recorded .iter() - .filter(|(_, _, nice)| *nice != process_nice) + .filter(|(_, _, _, nice)| *nice != process_nice) .collect::>() ), ( Some(listing(tree.path())), false, - Vec::<&(String, String, i32)>::new() + Vec::<&(String, String, String, i32)>::new() ) ); } @@ -1588,3 +1593,114 @@ async fn after_saves_and_prunes_the_pools_keep_the_nice_value_of_the_process() { "{blocking:?} {rayon:?}" ); } + +#[cfg(target_os = "linux")] +#[test] +async fn the_storage_calls_of_the_rayon_workers_of_a_prune_that_repacks_run_at_nice_19() { + // The deleted snapshot shares a pack with the kept one, so the prune repacks that pack. The + // prune reads the index files and repacks with rayon, on the workers of the pool of the prune. + let (storage, calls) = nice_recording_storage(); + let base = policy(LONG_DEADLINE, 1, Duration::ZERO); + let store = store( + storage, + StorePolicy { + prune: PruneSettings { + repack: RepackLimits::Unlimited, + ..base.prune + }, + ..base + }, + ); + let scope = new_scope(); + let file = |content: &[u8]| Spec::File { + content: Box::from(content), + mode: 0o644, + }; + let both = Scratch::new(); + write_tree( + both.path(), + &[ + ("kept.txt", file(b"kept content")), + ("deleted.txt", file(b"deleted content")), + ], + ); + let kept = Scratch::new(); + write_tree(kept.path(), &[("kept.txt", file(b"kept content"))]); + store + .save(&scope, &name("p-both"), both.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept.path()) + .await + .unwrap(); + std::mem::take(&mut *calls.lock().unwrap()); + + store.delete(&scope, &name("p-both")).await.unwrap(); + let recorded = std::mem::take(&mut *calls.lock().unwrap()); + let from_workers = recorded + .iter() + .filter(|(_, _, thread, _)| thread.starts_with("fs-snap-prune-")) + .collect::>(); + + assert_eq!( + ( + recorded + .iter() + .any(|(op, path, _, _)| op == "write" && path.starts_with("data/")), + from_workers.is_empty(), + from_workers + .iter() + .filter(|(_, _, _, nice)| *nice != 19) + .count(), + restored_listing(&store, &scope, &name("p-kept")).await.ok(), + ), + (true, false, 0, Some(listing(kept.path()))) + ); +} + +/// A pool builder that cannot start a thread. +#[cfg(target_os = "linux")] +fn no_pool( + _: &str, + _: Option, +) -> Result { + rayon::ThreadPoolBuilder::new() + .num_threads(1) + .spawn_handler(|_| Err(std::io::Error::other("no thread can start here"))) + .build() +} + +#[cfg(target_os = "linux")] +#[test] +async fn the_global_rayon_pool_keeps_the_nice_value_of_the_process_after_saves_without_their_pool() +{ + // Without its own pool, the rayon work of a save goes to the global pool from a thread at + // nice 19. The store starts the global pool when it is made, so its threads keep the normal + // priority also when a save is the first rayon work of the process. + let process_nice = super::super::priority::own_nice(); + let store = RusticSnapshotStore { + low_priority: super::super::priority::LowPriority { + build_pool: no_pool, + ..super::super::priority::LowPriority::new(NonZeroUsize::new(2)) + }, + ..RusticSnapshotStore::new(Arc::new(InMemoryBlobStorage::new()), &config()) + }; + let scope = new_scope(); + let (first, second) = (one_file_tree("first"), fixture_tree()); + store + .save(&scope, &name("p-1"), first.path()) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), second.path()) + .await + .unwrap(); + + let global = rayon::broadcast(|_| super::super::priority::own_nice()); + + assert!( + global.iter().all(|nice| *nice == process_nice), + "{global:?}" + ); +} From f18d52a3f00d0268534b9f9a5bca12b91a5633ae Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:53:45 -0700 Subject: [PATCH 25/55] Check that lowering the own priority gives Ok and nice 19 --- .../filesystem_snapshot/rustic/priority/tests.rs | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs index 6610032d75..f9ea230354 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority/tests.rs @@ -12,7 +12,7 @@ // See the License for the specific language governing permissions and // limitations under the License. -use super::{LOW_PRIORITY, LowPriority, own_nice}; +use super::{LOW_PRIORITY, LowPriority, lower_own_priority, own_nice}; use pretty_assertions::assert_eq; use rayon::{ThreadPool, ThreadPoolBuildError, ThreadPoolBuilder}; use std::num::NonZeroUsize; @@ -97,3 +97,17 @@ fn a_panic_of_the_work_gives_an_error() { assert!(done.is_err()); } + +#[test] +fn lowering_the_own_priority_gives_ok_and_nice_19() { + // The thread cannot raise its priority again, so the test lowers a thread of its own. + let lowered = std::thread::spawn(|| { + ( + lower_own_priority().map_err(|error| error.to_string()), + own_nice(), + ) + }) + .join(); + + assert_eq!(lowered.ok(), Some((Ok(()), LOW_PRIORITY))); +} From fb620d13c5ad9d7b776906127735b777343090f7 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:14:02 -0700 Subject: [PATCH 26/55] Let the caller of a save name its parent and choose the change detection --- .../filesystem_snapshot/contract_tests/mod.rs | 193 +++++++++--- .../src/filesystem_snapshot/memory.rs | 5 +- .../src/filesystem_snapshot/memory/tests.rs | 13 +- .../src/filesystem_snapshot/mod.rs | 16 + .../src/filesystem_snapshot/rustic/store.rs | 84 +++-- .../filesystem_snapshot/rustic/store/tests.rs | 292 +++++++++++++++--- .../src/filesystem_snapshot/rustic/tests.rs | 8 +- 7 files changed, 480 insertions(+), 131 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs b/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs index 0b33ba5d99..bd4f6392bf 100644 --- a/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/contract_tests/mod.rs @@ -40,7 +40,8 @@ pub(super) mod fixture; use super::{ - FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, + ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, + SnapshotStoreError, }; use fixture::{ Listed, Scratch, Spec, files_and_bytes, fixture, listing, one_file, pattern, write_tree, @@ -165,6 +166,12 @@ const CASES: &[(&str, Case)] = &[ "a_dropped_save_publishes_nothing_and_leaves_the_name_free", |open| a_dropped_save_publishes_nothing_and_leaves_the_name_free(open).boxed(), ), + ( + "a_save_with_a_truthful_parent_restores_the_same_tree_as_a_save_without_one", + |open| { + a_save_with_a_truthful_parent_restores_the_same_tree_as_a_save_without_one(open).boxed() + }, + ), ("no_method_blocks_the_runtime", |open| { no_method_blocks_the_runtime(open).boxed() }), @@ -250,7 +257,7 @@ async fn a_saved_tree_comes_back_the_same(open: OpenStore) { let tree = new_tree(&fixture()); let saved = store - .save(&scope, &name("p-fixture"), tree.path()) + .save(&scope, &name("p-fixture"), tree.path(), None) .await .unwrap(); let (restored, info) = restored(&*store, &scope, &name("p-fixture")).await.unwrap(); @@ -258,6 +265,67 @@ async fn a_saved_tree_comes_back_the_same(open: OpenStore) { assert_eq!((restored, info), (listing(tree.path()), saved)); } +async fn a_save_with_a_truthful_parent_restores_the_same_tree_as_a_save_without_one( + open: OpenStore, +) { + // Each changed file gets a new size, so the parent is truthful for both modes. + let store = open(); + let scope = new_scope(); + let file = |content: &str| Spec::File { + content: Box::from(content.as_bytes()), + mode: 0o644, + }; + let tree = new_tree(&[ + ("changed.txt", file("old")), + ("kept.txt", file("kept")), + ("removed.txt", file("removed")), + ]); + let parent = name("p-parent"); + store + .save(&scope, &parent, tree.path(), None) + .await + .unwrap(); + write_tree( + tree.path(), + &[ + ("changed.txt", file("new and longer")), + ("added.txt", file("added")), + ], + ); + std::fs::remove_file(tree.path().join("removed.txt")).unwrap(); + + store + .save(&scope, &name("p-none"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-size-mtime"), + tree.path(), + Some((&parent, ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + store + .save( + &scope, + &name("p-full"), + tree.path(), + Some((&parent, ChangeDetection::Full)), + ) + .await + .unwrap(); + let restored = [ + restored_listing(&*store, &scope, &name("p-none")).await, + restored_listing(&*store, &scope, &name("p-size-mtime")).await, + restored_listing(&*store, &scope, &name("p-full")).await, + ]; + + let expected = listing(tree.path()); + assert_eq!(restored, [expected.clone(), expected.clone(), expected]); +} + async fn each_name_of_a_hard_linked_file_comes_back_as_its_own_file(open: OpenStore) { let store = open(); let scope = new_scope(); @@ -270,7 +338,7 @@ async fn each_name_of_a_hard_linked_file_comes_back_as_its_own_file(open: OpenSt let into = Scratch::new(); store - .save(&scope, &name("p-linked"), tree.path()) + .save(&scope, &name("p-linked"), tree.path(), None) .await .unwrap(); store @@ -301,7 +369,7 @@ async fn save_stat_list_and_restore_give_the_same_info(open: OpenStore) { let before = Timestamp::now_utc(); let saved = store - .save(&scope, &name("p-info"), tree.path()) + .save(&scope, &name("p-info"), tree.path(), None) .await .unwrap(); let after = Timestamp::now_utc(); @@ -334,7 +402,7 @@ async fn a_save_leaves_the_tree_as_it_was(open: OpenStore) { let before = listing(tree.path()); store - .save(&scope, &name("p-source"), tree.path()) + .save(&scope, &name("p-source"), tree.path(), None) .await .unwrap(); @@ -347,11 +415,13 @@ async fn a_name_in_use_gives_already_exists_and_changes_nothing(open: OpenStore) let first = new_tree(&one_file("first")); let second = new_tree(&one_file("second tree")); let saved = store - .save(&scope, &name("p-taken"), first.path()) + .save(&scope, &name("p-taken"), first.path(), None) .await .unwrap(); - let again = store.save(&scope, &name("p-taken"), second.path()).await; + let again = store + .save(&scope, &name("p-taken"), second.path(), None) + .await; assert!( matches!(again, Err(SnapshotStoreError::AlreadyExists)), @@ -382,12 +452,14 @@ async fn a_tree_that_cannot_be_read_gives_source_and_publishes_nothing(open: Ope std::fs::write(&file, b"not a directory").unwrap(); let tree = new_tree(&one_file("real")); - let from_missing = store.save(&scope, &name("p-unread"), &missing).await; - let from_file = store.save(&scope, &name("p-unread"), &file).await; + let from_missing = store.save(&scope, &name("p-unread"), &missing, None).await; + let from_file = store.save(&scope, &name("p-unread"), &file, None).await; let stat = store.stat(&scope, &name("p-unread")).await.unwrap(); let names = listed_names(&*store, &scope).await; let restore = restored(&*store, &scope, &name("p-unread")).await; - let later = store.save(&scope, &name("p-unread"), tree.path()).await; + let later = store + .save(&scope, &name("p-unread"), tree.path(), None) + .await; assert!( matches!(from_missing, Err(SnapshotStoreError::Source(_))), @@ -421,7 +493,9 @@ async fn an_entry_that_cannot_be_read_gives_source_and_publishes_nothing(open: O std::fs::write(&locked, b"locked").unwrap(); std::fs::set_permissions(&locked, std::fs::Permissions::from_mode(0o000)).unwrap(); - let saved = store.save(&scope, &name("p-locked"), tree.path()).await; + let saved = store + .save(&scope, &name("p-locked"), tree.path(), None) + .await; let stat = store.stat(&scope, &name("p-locked")).await.unwrap(); let names = listed_names(&*store, &scope).await; @@ -448,7 +522,7 @@ async fn sequential_saves_get_later_times_and_list_newest_first(open: OpenStore) let store = store.clone(); let scope = scope.clone(); let tree = tree.path().to_path_buf(); - async move { store.save(&scope, &name(text), &tree).await.unwrap() } + async move { store.save(&scope, &name(text), &tree, None).await.unwrap() } }) .collect::>() .await; @@ -478,7 +552,7 @@ async fn an_unknown_name_gives_not_found_every_time(open: OpenStore) { let used = new_scope(); let tree = new_tree(&one_file("other")); store - .save(&used, &name("p-other"), tree.path()) + .save(&used, &name("p-other"), tree.path(), None) .await .unwrap(); @@ -505,7 +579,7 @@ async fn a_restore_into_a_directory_that_is_not_empty_writes_nothing(open: OpenS let scope = new_scope(); let tree = new_tree(&one_file("content")); store - .save(&scope, &name("p-busy"), tree.path()) + .save(&scope, &name("p-busy"), tree.path(), None) .await .unwrap(); let into = new_tree(&[( @@ -532,7 +606,7 @@ async fn a_restore_into_a_path_that_is_not_a_directory_writes_nothing(open: Open let scope = new_scope(); let tree = new_tree(&one_file("content")); store - .save(&scope, &name("p-nowhere"), tree.path()) + .save(&scope, &name("p-nowhere"), tree.path(), None) .await .unwrap(); let parent = Scratch::new(); @@ -562,11 +636,11 @@ async fn a_deleted_name_stops_resolving_at_once(open: OpenStore) { let scope = new_scope(); let tree = new_tree(&one_file("deleted")); store - .save(&scope, &name("p-deleted"), tree.path()) + .save(&scope, &name("p-deleted"), tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), tree.path()) + .save(&scope, &name("p-kept"), tree.path(), None) .await .unwrap(); @@ -587,7 +661,7 @@ async fn delete_is_idempotent(open: OpenStore) { let scope = new_scope(); let tree = new_tree(&one_file("twice")); store - .save(&scope, &name("p-twice"), tree.path()) + .save(&scope, &name("p-twice"), tree.path(), None) .await .unwrap(); @@ -611,15 +685,15 @@ async fn a_delete_keeps_every_other_snapshot(open: OpenStore) { let shared = new_tree(&fixture()); let other = new_tree(&one_file("other")); store - .save(&scope, &name("p-twin-1"), shared.path()) + .save(&scope, &name("p-twin-1"), shared.path(), None) .await .unwrap(); store - .save(&scope, &name("p-twin-2"), shared.path()) + .save(&scope, &name("p-twin-2"), shared.path(), None) .await .unwrap(); store - .save(&scope, &name("p-other"), other.path()) + .save(&scope, &name("p-other"), other.path(), None) .await .unwrap(); @@ -644,7 +718,7 @@ async fn a_restore_that_races_a_delete_of_its_name_gives_a_whole_tree_or_nothing let scope = new_scope(); let tree = new_tree(&fixture()); store - .save(&scope, &name("p-raced"), tree.path()) + .save(&scope, &name("p-raced"), tree.path(), None) .await .unwrap(); @@ -668,11 +742,11 @@ async fn a_restore_during_a_delete_of_another_name_gives_the_whole_tree(open: Op let tree = new_tree(&fixture()); let other = new_tree(&fixture()); store - .save(&scope, &name("p-restored"), tree.path()) + .save(&scope, &name("p-restored"), tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-deleted"), other.path()) + .save(&scope, &name("p-deleted"), other.path(), None) .await .unwrap(); @@ -695,14 +769,17 @@ async fn a_save_a_restore_and_a_delete_in_one_scope_run_at_the_same_time(open: O let second = new_tree(&one_file("second")); let third = new_tree(&one_file("third")); let (first_name, second_name, third_name) = (name("p-1"), name("p-2"), name("p-3")); - store.save(&scope, &first_name, first.path()).await.unwrap(); store - .save(&scope, &second_name, second.path()) + .save(&scope, &first_name, first.path(), None) + .await + .unwrap(); + store + .save(&scope, &second_name, second.path(), None) .await .unwrap(); let (saved, restore, deleted) = futures::join!( - store.save(&scope, &third_name, third.path()), + store.save(&scope, &third_name, third.path(), None), restored(&*store, &scope, &first_name), store.delete(&scope, &second_name) ); @@ -728,14 +805,23 @@ async fn a_deleted_scope_is_as_unused_as_before_its_first_save(open: OpenStore) let scope = new_scope(); let old = new_tree(&one_file("old")); let new = new_tree(&one_file("new tree")); - store.save(&scope, &name("p-1"), old.path()).await.unwrap(); - store.save(&scope, &name("p-2"), old.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), old.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), old.path(), None) + .await + .unwrap(); store.delete_scope(&scope).await.unwrap(); let names = listed_names(&*store, &scope).await; let stat = store.stat(&scope, &name("p-1")).await.unwrap(); let restore = restored(&*store, &scope, &name("p-2")).await; - store.save(&scope, &name("p-1"), new.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), new.path(), None) + .await + .unwrap(); assert!(is_not_found(&restore), "{restore:?}"); assert_eq!( @@ -754,10 +840,13 @@ async fn delete_scope_is_idempotent_and_keeps_other_scopes(open: OpenStore) { let kept = new_scope(); let tree = new_tree(&one_file("kept")); store - .save(&deleted, &name("p-1"), tree.path()) + .save(&deleted, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save(&kept, &name("p-1"), tree.path(), None) .await .unwrap(); - store.save(&kept, &name("p-1"), tree.path()).await.unwrap(); let results = [ store.delete_scope(&new_scope()).await.is_ok(), @@ -787,9 +876,12 @@ async fn a_copied_scope_has_the_same_names_infos_and_trees(open: OpenStore) { let to = new_scope(); let first = new_tree(&fixture()); let second = new_tree(&one_file("second")); - store.save(&from, &name("p-1"), first.path()).await.unwrap(); store - .save(&from, &name("u-2"), second.path()) + .save(&from, &name("p-1"), first.path(), None) + .await + .unwrap(); + store + .save(&from, &name("u-2"), second.path(), None) .await .unwrap(); let source = store.list(&from).await.unwrap(); @@ -822,12 +914,21 @@ async fn copied_scopes_are_independent(open: OpenStore) { let to = new_scope(); let tree = new_tree(&one_file("copied")); let later = new_tree(&one_file("later")); - store.save(&from, &name("p-1"), tree.path()).await.unwrap(); - store.save(&from, &name("p-2"), tree.path()).await.unwrap(); + store + .save(&from, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save(&from, &name("p-2"), tree.path(), None) + .await + .unwrap(); store.copy_scope(&from, &to).await.unwrap(); store.delete(&from, &name("p-1")).await.unwrap(); - store.save(&to, &name("p-3"), later.path()).await.unwrap(); + store + .save(&to, &name("p-3"), later.path(), None) + .await + .unwrap(); let target_after_source_delete = restored_listing(&*store, &to, &name("p-1")).await; let source_names = listed_names(&*store, &from).await; store.delete_scope(&to).await.unwrap(); @@ -855,7 +956,7 @@ async fn a_copy_of_an_unused_scope_leaves_the_target_unused(open: OpenStore) { let to = new_scope(); let tree = new_tree(&one_file("other scope")); store - .save(&other, &name("p-other"), tree.path()) + .save(&other, &name("p-other"), tree.path(), None) .await .unwrap(); @@ -871,11 +972,11 @@ async fn one_name_in_two_scopes_gives_two_snapshots(open: OpenStore) { let first = new_tree(&one_file("first scope")); let second = new_tree(&one_file("second scope")); store - .save(&first_scope, &name("p-same"), first.path()) + .save(&first_scope, &name("p-same"), first.path(), None) .await .unwrap(); store - .save(&second_scope, &name("p-same"), second.path()) + .save(&second_scope, &name("p-same"), second.path(), None) .await .unwrap(); let first_restored = restored_listing(&*store, &first_scope, &name("p-same")).await; @@ -904,7 +1005,7 @@ async fn a_save_through_one_store_resolves_through_another(open: OpenStore) { let tree = new_tree(&fixture()); let saved = writer - .save(&scope, &name("p-shared"), tree.path()) + .save(&scope, &name("p-shared"), tree.path(), None) .await .unwrap(); @@ -933,8 +1034,8 @@ async fn two_stores_save_into_a_new_scope_at_the_same_time(open: OpenStore) { let (first_name, second_name) = (name("p-first"), name("p-second")); let (first_saved, second_saved) = futures::join!( - first.save(&scope, &first_name, first_tree.path()), - second.save(&scope, &second_name, second_tree.path()) + first.save(&scope, &first_name, first_tree.path(), None), + second.save(&scope, &second_name, second_tree.path(), None) ); let mut names = listed_names(&*first, &scope).await; names.sort(); @@ -973,7 +1074,7 @@ async fn a_dropped_save_publishes_nothing_and_leaves_the_name_free(open: OpenSto .filter_map(|_| { let scope = new_scope(); let dropped = store - .save(&scope, &dropped_name, tree.path()) + .save(&scope, &dropped_name, tree.path(), None) .now_or_never(); std::future::ready(dropped.is_none().then_some(scope)) }) @@ -985,7 +1086,7 @@ async fn a_dropped_save_publishes_nothing_and_leaves_the_name_free(open: OpenSto let stat = store.stat(&scope, &dropped_name).await.unwrap(); let restore = restored(&*store, &scope, &dropped_name).await; let names_after_the_drop = listed_names(&*store, &scope).await; - let saved_again = store.save(&scope, &dropped_name, other.path()).await; + let saved_again = store.save(&scope, &dropped_name, other.path(), None).await; assert!(is_not_found(&restore), "{restore:?}"); assert!(saved_again.is_ok(), "{saved_again:?}"); @@ -1041,7 +1142,7 @@ async fn no_method_blocks_the_runtime(open: OpenStore) { let before_save = ticks.load(Ordering::SeqCst); store - .save(&scope, &name("p-large"), &tree_path) + .save(&scope, &name("p-large"), &tree_path, None) .await .unwrap(); let during_save = counted(before_save); diff --git a/golem-worker-executor/src/filesystem_snapshot/memory.rs b/golem-worker-executor/src/filesystem_snapshot/memory.rs index b4fe682a18..3816efb695 100644 --- a/golem-worker-executor/src/filesystem_snapshot/memory.rs +++ b/golem-worker-executor/src/filesystem_snapshot/memory.rs @@ -25,8 +25,8 @@ mod tree; mod tests; use super::{ - FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, - newest_first, snapshot_time, + ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, + SnapshotStoreError, newest_first, snapshot_time, }; use async_trait::async_trait; use golem_common::model::Timestamp; @@ -120,6 +120,7 @@ impl FilesystemSnapshotStore for InMemorySnapshotStore { scope: &SnapshotScope, name: &SnapshotName, tree: &Path, + _parent: Option<(&SnapshotName, ChangeDetection)>, ) -> Result { let snapshots = self.snapshots_of(scope); if found(&snapshots, name).is_some() { diff --git a/golem-worker-executor/src/filesystem_snapshot/memory/tests.rs b/golem-worker-executor/src/filesystem_snapshot/memory/tests.rs index af377f4d88..d71a58668b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/memory/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/memory/tests.rs @@ -57,7 +57,12 @@ async fn a_save_after_a_snapshot_from_a_clock_that_is_ahead_gets_a_later_time() let tree = tempfile::tempdir().unwrap(); let saved = store - .save(&scope, &SnapshotName::new("p-next").unwrap(), tree.path()) + .save( + &scope, + &SnapshotName::new("p-next").unwrap(), + tree.path(), + None, + ) .await .unwrap(); let listed = store.list(&scope).await.unwrap(); @@ -97,8 +102,8 @@ async fn of_two_saves_of_one_name_at_the_same_time_one_wins() { let second_tree = tree_with("second tree"); let (first_saved, second_saved) = futures::join!( - first.save(&scope, &name, first_tree.path()), - second.save(&scope, &name, second_tree.path()) + first.save(&scope, &name, first_tree.path(), None), + second.save(&scope, &name, second_tree.path(), None) ); let into = tempfile::tempdir().unwrap(); first.restore(&scope, &name, into.path()).await.unwrap(); @@ -145,7 +150,7 @@ mod unix { let scope = new_scope(); let name = SnapshotName::new("p-socket").unwrap(); - let saved = store.save(&scope, &name, tree.path()).await; + let saved = store.save(&scope, &name, tree.path(), None).await; let listed = store.list(&scope).await.unwrap(); assert!( diff --git a/golem-worker-executor/src/filesystem_snapshot/mod.rs b/golem-worker-executor/src/filesystem_snapshot/mod.rs index b4354f2a47..5cbfe09faf 100644 --- a/golem-worker-executor/src/filesystem_snapshot/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/mod.rs @@ -121,6 +121,16 @@ pub(crate) struct SnapshotInfo { pub bytes: u64, } +/// How a save with a parent finds the files that did not change since the parent. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) enum ChangeDetection { + /// A file whose size and modification time equal those of the same path in the parent keeps + /// the content of the parent, and the save does not read it. + SizeMtime, + /// The save reads every file. + Full, +} + /// Why a call on a [`FilesystemSnapshotStore`] failed. #[derive(Debug)] pub(crate) enum SnapshotStoreError { @@ -210,11 +220,17 @@ pub(crate) trait FilesystemSnapshotStore: Send + Sync { /// The result gives the number of files, the size of the tree, and the time of the /// snapshot. That time is later than the time of each snapshot that the scope held when the /// save started. + /// + /// `parent` names the snapshot that the save compares with. With `SizeMtime`, a file whose + /// size and modification time equal those of the same path in the parent keeps the content of + /// the parent, and the save does not read it. With `Full`, the save reads every file. A parent + /// that the scope does not hold gives a save that reads every file. async fn save( &self, scope: &SnapshotScope, name: &SnapshotName, tree: &Path, + parent: Option<(&SnapshotName, ChangeDetection)>, ) -> Result; /// Rebuilds a saved tree in the empty directory `into`. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index a95f140817..b90dbb8262 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -28,13 +28,13 @@ use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; use super::{ - ChangeDetection, PruneReport, PruneSettings, RepackLimits, RepositoryKey, RepositorySettings, - SaveSettings, backup_options, open_existing, open_or_create, prune, restore_snapshot, - run_blocking, + ChangeDetection as RusticChangeDetection, PruneReport, PruneSettings, RepackLimits, + RepositoryKey, RepositorySettings, SaveSettings, backup_options, open_existing, open_or_create, + prune, restore_snapshot, run_blocking, }; use crate::filesystem_snapshot::{ - FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, - newest_first, snapshot_time, + ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, + SnapshotStoreError, newest_first, snapshot_time, }; use crate::services::golem_config::FilesystemSnapshotStoreConfig; use anyhow::Context; @@ -72,7 +72,9 @@ pub(super) struct StorePolicy { pub(super) deadline: Duration, /// The settings of a repository that a save makes. pub(super) repository: RepositorySettings, - pub(super) save: SaveSettings, + /// The number of threads of each parallel stage of a save. `None` is the number of CPUs that + /// the process can use. + pub(super) save_threads: Option, /// The number of threads that read packs in a restore. pub(super) restore_reader_threads: NonZeroUsize, /// The settings of a prune. `keep_delete` is also the shortest time between two prunes. @@ -87,10 +89,7 @@ impl StorePolicy { Self { deadline: config.storage_call_deadline(), repository: RepositorySettings::DEFAULT, - save: SaveSettings { - threads: Some(config.save_threads()), - detection: ChangeDetection::Ctime, - }, + save_threads: Some(config.save_threads()), restore_reader_threads: config.restore_reader_threads(), prune: PruneSettings { fast_repack: false, @@ -105,8 +104,34 @@ impl StorePolicy { /// The options of a save of the store: the options of the bridge, and a save that cannot read an /// entry fails before it writes the snapshot file. A save records no device id, so a restore gives /// each name of a hard-linked file as its own file. -fn store_backup_options(policy: &StorePolicy) -> BackupOptions { - backup_options(&policy.save) +/// +/// With a parent and `SizeMtime`, rustic compares each file with the parent that the id names, by +/// size and modification time. Without a parent, or with `Full`, rustic uses no parent and reads +/// every file. +fn store_backup_options( + policy: &StorePolicy, + parent: Option<(SnapshotId, ChangeDetection)>, +) -> BackupOptions { + let settings = |detection| SaveSettings { + threads: policy.save_threads, + detection, + }; + let options = match parent { + Some((id, ChangeDetection::SizeMtime)) => { + let options = backup_options(&settings(RusticChangeDetection::SizeMtime)); + let parent_opts = options + .parent_opts + .clone() + .parents(vec![id.to_hex().to_string()]); + options.parent_opts(parent_opts) + } + None | Some((_, ChangeDetection::Full)) => { + let options = backup_options(&settings(RusticChangeDetection::Ctime)); + let parent_opts = options.parent_opts.clone().force(true); + options.parent_opts(parent_opts) + } + }; + options .fail_on_read_error(true) .ignore_save_opts(LocalSourceSaveOptions::default().set_devid(DevIdOption::No)) } @@ -160,7 +185,7 @@ impl RusticSnapshotStore { policy, root: CancellationToken::new(), tracker: TaskTracker::new(), - low_priority: LowPriority::new(policy.save.threads), + low_priority: LowPriority::new(policy.save_threads), } } @@ -287,6 +312,7 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { scope: &SnapshotScope, name: &SnapshotName, tree: &Path, + parent: Option<(&SnapshotName, ChangeDetection)>, ) -> Result { let (token, _guard) = self.start()?; check_tree(tree).await?; @@ -296,11 +322,12 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { let policy = self.policy; let name = name.clone(); let tree: Box = tree.into(); + let parent = parent.map(|(parent, detection)| (parent.clone(), detection)); let low_priority = self.low_priority; let staged = self .blocking(Operation::Save, move || { low_priority.run("fs-snap-save", move || { - stage_save(backend, &stage, &key, &policy, &name, &tree) + stage_save(backend, &stage, &key, &policy, &name, &tree, parent) }) }) .await?; @@ -516,7 +543,9 @@ async fn check_destination(into: &Path) -> Result<(), SnapshotStoreError> { } /// Backs up the tree with the snapshot file in the stage, and gives the staged file with the info -/// of the snapshot. The result is `None` when a snapshot of the scope already has the name. +/// of the snapshot. The result is `None` when a snapshot of the scope already has the name. The +/// parent is the snapshot that [`lookup`] finds for its name, and a name without a snapshot gives +/// no parent. fn stage_save( backend: Arc, stage: &SnapshotStage, @@ -524,12 +553,16 @@ fn stage_save( policy: &StorePolicy, name: &SnapshotName, tree: &Path, + parent: Option<(SnapshotName, ChangeDetection)>, ) -> anyhow::Result> { let (repository, _) = open_or_create(backend, key, &policy.repository)?; let before = scope_snapshots(&repository)?; if has_name(&before, name) { return Ok(None); } + let parent = parent.and_then(|(parent, detection)| { + named(&before.readable, &parent).map(|snapshot| (snapshot.id, detection)) + }); let newest = before .readable .iter() @@ -549,7 +582,7 @@ fn stage_save( .to_snapshot()?; let repository = repository.to_indexed_ids()?; repository.backup( - &store_backup_options(policy), + &store_backup_options(policy, parent), &PathList::from_iter(Some(tree)), snapshot, )?; @@ -604,14 +637,9 @@ fn has_name(found: &ScopeSnapshots, name: &SnapshotName) -> bool { /// time and id wins. When no file has the name and a file failed its integrity check, the result is /// `Corrupt`, because that file can have the name. fn lookup(found: ScopeSnapshots, name: &SnapshotName) -> Lookup { - let winner = found - .readable - .into_iter() - .filter(|snapshot| snapshot.label == name.as_str()) - .min_by_key(|snapshot| (snapshot.time.timestamp(), snapshot.id)); - match (winner, found.unreadable) { - (Some(snapshot), _) => match snapshot_info(&snapshot) { - Some(info) => Lookup::Found(Box::new(snapshot), info), + match (named(&found.readable, name), found.unreadable) { + (Some(snapshot), _) => match snapshot_info(snapshot) { + Some(info) => Lookup::Found(Box::new(snapshot.clone()), info), None => Lookup::Corrupt(anyhow::anyhow!( "the snapshot {} does not describe its tree", snapshot.id @@ -624,6 +652,14 @@ fn lookup(found: ScopeSnapshots, name: &SnapshotName) -> Lookup { } } +/// Of the snapshot files with the name, gives the one with the least time and id. +fn named<'a>(snapshots: &'a [SnapshotFile], name: &SnapshotName) -> Option<&'a SnapshotFile> { + snapshots + .iter() + .filter(|snapshot| snapshot.label == name.as_str()) + .min_by_key(|snapshot| (snapshot.time.timestamp(), snapshot.id)) +} + /// Gives the name and the info of each snapshot whose label is a name and whose description /// parses. Of the files with one name, only the one that [`lookup`] takes stays. fn listed(found: ScopeSnapshots) -> Vec<(SnapshotName, SnapshotInfo)> { diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 1cfa5704d6..16b7c1e7ae 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -20,6 +20,7 @@ use super::super::files::SnapshotFiles; use super::super::prune::{PruneLedger, read_ledger}; use super::super::scripted::{Script, ScriptedBlobStorage}; +use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; use super::{ RusticSnapshotStore, StorePolicy, leaves_marked_packs, scope_snapshots, store_backup_options, @@ -30,13 +31,15 @@ use crate::filesystem_snapshot::contract_tests::fixture::{ }; use crate::filesystem_snapshot::contract_tests::{self, OpenStore, new_scope}; use crate::filesystem_snapshot::{ - FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, SnapshotStoreError, + ChangeDetection, FilesystemSnapshotStore, SnapshotInfo, SnapshotName, SnapshotScope, + SnapshotStoreError, }; use crate::services::golem_config::FilesystemSnapshotStoreConfig; use futures::{FutureExt, StreamExt}; use golem_service_base::storage::blob::memory::InMemoryBlobStorage; use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; use pretty_assertions::assert_eq; +use rustic_core::repofile::SnapshotId; use std::future::Future; use std::num::NonZeroUsize; use std::path::Path; @@ -201,13 +204,13 @@ fn rustic_store_keeps_the_contract(r: &mut DynamicTestRegistration) { #[test] fn the_policy_takes_the_configured_values_and_the_options_are_strict() { let policy = StorePolicy::from_config(&config()); - let backup = store_backup_options(&policy); + let backup = store_backup_options(&policy, None); let restore = store_restore_options(&policy); assert_eq!( ( policy.deadline, - policy.save.threads.map(NonZeroUsize::get), + policy.save_threads.map(NonZeroUsize::get), policy.restore_reader_threads.get(), policy.prune.keep_delete, policy.prune.fast_repack, @@ -276,7 +279,7 @@ async fn a_tree_saved_through_a_proc_self_fd_path_is_stored_below_the_root() { let through_fd = format!("/proc/self/fd/{}", directory.as_raw_fd()); store - .save(&scope, &name("p-fd"), Path::new(&through_fd)) + .save(&scope, &name("p-fd"), Path::new(&through_fd), None) .await .unwrap(); let namespace = scope.0.clone(); @@ -326,12 +329,12 @@ async fn a_save_whose_index_write_fails_publishes_nothing_and_leaves_the_name_fr let scope = new_scope(); let tree = fixture_tree(); - let failed = store.save(&scope, &name("p-1"), tree.path()).await; + let failed = store.save(&scope, &name("p-1"), tree.path(), None).await; let stat = store.stat(&scope, &name("p-1")).await.unwrap(); let names = listed_names(&store, &scope).await; let restore = restored_listing(&store, &scope, &name("p-1")).await; refuse.store(false, Ordering::SeqCst); - let saved_again = store.save(&scope, &name("p-1"), tree.path()).await; + let saved_again = store.save(&scope, &name("p-1"), tree.path(), None).await; assert!( failed.as_ref().is_err_and(|error| is_storage(error, true)), @@ -374,11 +377,11 @@ async fn a_publish_that_reaches_the_deadline_and_lands_late_publishes_nothing() let scope = new_scope(); let tree = one_file_tree("late"); - let failed = store.save(&scope, &name("p-late"), tree.path()).await; + let failed = store.save(&scope, &name("p-late"), tree.path(), None).await; hang.store(false, Ordering::SeqCst); let stat = store.stat(&scope, &name("p-late")).await.unwrap(); let names = listed_names(&store, &scope).await; - let saved_again = store.save(&scope, &name("p-late"), tree.path()).await; + let saved_again = store.save(&scope, &name("p-late"), tree.path(), None).await; assert!( failed.as_ref().is_err_and(|error| is_storage(error, true)), @@ -410,9 +413,15 @@ async fn a_second_save_of_an_unchanged_tree_writes_no_pack() { let scope = new_scope(); let tree = fixture_tree(); - store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); let first = storage.calls().len(); - store.save(&scope, &name("p-2"), tree.path()).await.unwrap(); + store + .save(&scope, &name("p-2"), tree.path(), None) + .await + .unwrap(); let writes = storage.calls()[first..] .iter() .filter(|(op_label, _)| *op_label == "write" || *op_label == "publish") @@ -462,8 +471,8 @@ async fn two_stores_that_create_one_repository_at_the_same_time_both_save() { let (first_name, second_name) = (name("p-first"), name("p-second")); let (first_saved, second_saved, both_waited) = tokio::join!( - first.save(&scope, &first_name, first_tree.path()), - second.save(&scope, &second_name, second_tree.path()), + first.save(&scope, &first_name, first_tree.path(), None), + second.save(&scope, &second_name, second_tree.path(), None), async { let both = eventually(|| config_writes() == 2).await; storage.open_gate(); @@ -519,7 +528,7 @@ async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { let scope = new_scope(); let (old_tree, new_tree) = (one_file_tree("old"), fixture_tree()); store - .save(&scope, &name("p-old"), old_tree.path()) + .save(&scope, &name("p-old"), old_tree.path(), None) .await .unwrap(); hold_next_index.store(true, Ordering::SeqCst); @@ -536,7 +545,7 @@ async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { let store = store.clone(); let scope = scope.clone(); let path = new_tree.path().to_path_buf(); - async move { store.save(&scope, &name("p-new"), &path).await } + async move { store.save(&scope, &name("p-new"), &path, None).await } }); let held = eventually(|| index_writes() > before).await; let deleted = store.delete(&scope, &name("p-old")).await; @@ -577,7 +586,10 @@ async fn a_restore_whose_pack_reads_fail_gives_a_retryable_storage_error() { let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); - store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); refuse.store(true, Ordering::SeqCst); let restored = restored_listing(&store, &scope, &name("p-1")).await; @@ -611,7 +623,9 @@ async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_d std::fs::write(&locked, b"locked").unwrap(); std::fs::set_permissions(&locked, std::fs::Permissions::from_mode(0o000)).unwrap(); - let saved = store.save(&scope, &name("p-locked"), tree.path()).await; + let saved = store + .save(&scope, &name("p-locked"), tree.path(), None) + .await; assert!( matches!(&saved, Err(SnapshotStoreError::Source(error)) if error.kind() == std::io::ErrorKind::PermissionDenied), @@ -657,7 +671,7 @@ async fn a_restore_that_cannot_set_an_extended_attribute_gives_destination() { ); let scope = new_scope(); store - .save(&scope, &name("p-xattr"), source.path()) + .save(&scope, &name("p-xattr"), source.path(), None) .await .unwrap(); @@ -702,7 +716,7 @@ async fn a_snapshot_file_that_fails_its_check_is_left_out_of_list_and_makes_an_u let scope = new_scope(); let tree = one_file_tree("kept"); store - .save(&scope, &name("p-kept"), tree.path()) + .save(&scope, &name("p-kept"), tree.path(), None) .await .unwrap(); storage @@ -742,11 +756,11 @@ async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_per let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store - .save(&scope, &name("p-deleted"), deleted_tree.path()) + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept_tree.path()) + .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); let packs_before = blobs(&*storage, &scope.0, "data/").await; @@ -779,11 +793,11 @@ async fn a_delete_below_the_threshold_does_not_prune() { let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), one_file_tree("kept")); store - .save(&scope, &name("p-deleted"), deleted_tree.path()) + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept_tree.path()) + .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); let packs_before = blobs(&*storage, &scope.0, "data/").await; @@ -815,7 +829,10 @@ async fn no_second_prune_runs_within_the_grace_period() { let store = store.clone(); let scope = scope.clone(); async move { - store.save(&scope, &name(text), tree.path()).await.unwrap(); + store + .save(&scope, &name(text), tree.path(), None) + .await + .unwrap(); } }) .await; @@ -855,11 +872,11 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store - .save(&scope, &name("p-deleted"), deleted_tree.path()) + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept_tree.path()) + .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); refuse.store(true, Ordering::SeqCst); @@ -893,11 +910,11 @@ async fn a_deleted_scope_holds_no_blob() { let scope = new_scope(); let (first, second) = (one_file_tree("first"), one_file_tree("second")); store - .save(&scope, &name("p-1"), first.path()) + .save(&scope, &name("p-1"), first.path(), None) .await .unwrap(); store - .save(&scope, &name("p-2"), second.path()) + .save(&scope, &name("p-2"), second.path(), None) .await .unwrap(); store.delete(&scope, &name("p-1")).await.unwrap(); @@ -927,7 +944,7 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam counted.clone(), policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), ) - .save(&new_scope(), &name("p-dropped"), tree.path()) + .save(&new_scope(), &name("p-dropped"), tree.path(), None) .await .unwrap(); let calls = counted.calls().len(); @@ -951,7 +968,7 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam let ended = drop_when( &storage, |calls| calls.len() >= held, - dropping.save(&scope, &name("p-dropped"), &tree), + dropping.save(&scope, &name("p-dropped"), &tree, None), ) .await; let stopped = tokio::time::timeout(LIMIT, dropping.shut_down()) @@ -960,7 +977,7 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam let later = store(inner, policy); let stat = later.stat(&scope, &name("p-dropped")).await.ok().flatten(); let names = listed_names(&later, &scope).await; - let saved_again = later.save(&scope, &name("p-dropped"), &other).await; + let saved_again = later.save(&scope, &name("p-dropped"), &other, None).await; let restored = restored_listing(&later, &scope, &name("p-dropped")) .await .ok(); @@ -1014,7 +1031,7 @@ async fn shut_down_ends_running_operations_before_it_returns() { let store = store.clone(); let scope = scope.clone(); let path = tree.path().to_path_buf(); - async move { store.save(&scope, &name("p-held"), &path).await } + async move { store.save(&scope, &name("p-held"), &path, None).await } }); let held = eventually(|| { storage @@ -1080,7 +1097,10 @@ async fn a_dropped_operation_stops_its_blocking_work() { ); let scope = new_scope(); let tree = fixture_tree(); - store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); hold.store(true, Ordering::SeqCst); let before = storage.calls().len(); let reached = |calls: &[(&'static str, String)]| { @@ -1156,6 +1176,168 @@ async fn snapshot_files( .unwrap() } +/// Gives, for the snapshot with the name, the numbers of new, changed and unmodified files that its +/// save counted, and the id of its parent. +fn read_counts( + files: &[rustic_core::repofile::SnapshotFile], + name: &str, +) -> Option<((u64, u64, u64), Option)> { + let snapshot = files.iter().find(|snapshot| snapshot.label == name)?; + let summary = snapshot.summary.as_ref()?; + Some(( + ( + summary.files_new, + summary.files_changed, + summary.files_unmodified, + ), + snapshot.parent, + )) +} + +fn id_of(files: &[rustic_core::repofile::SnapshotFile], name: &str) -> Option { + files + .iter() + .find(|snapshot| snapshot.label == name) + .map(|snapshot| snapshot.id) +} + +#[test] +async fn a_size_and_mtime_save_of_a_copied_tree_reads_no_unchanged_file() { + // A copy gives each file a new inode and a new change time, and keeps its size and its + // modification time, as a capture does. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = three_file_tree(); + let copy = Scratch::new(); + wait_past_change_times(&entries(tree.path())); + copy_flat_tree(tree.path(), copy.path()); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-2"), + copy.path(), + Some((&name("p-1"), ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + let files = snapshot_files(storage, &scope).await; + + assert_eq!( + read_counts(&files, "p-2"), + Some(((0, 0, 3), id_of(&files, "p-1"))) + ); +} + +#[test] +async fn a_full_save_reads_each_file() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = three_file_tree(); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-2"), + tree.path(), + Some((&name("p-1"), ChangeDetection::Full)), + ) + .await + .unwrap(); + let files = snapshot_files(storage, &scope).await; + + assert_eq!(read_counts(&files, "p-2"), Some(((3, 0, 0), None))); +} + +#[test] +async fn the_parent_of_a_save_is_the_named_snapshot_also_when_a_newer_snapshot_exists() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = three_file_tree(); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-3"), + tree.path(), + Some((&name("p-1"), ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + let files = snapshot_files(storage, &scope).await; + + assert_eq!( + ( + read_counts(&files, "p-3"), + id_of(&files, "p-1") == id_of(&files, "p-2") + ), + (Some(((0, 0, 3), id_of(&files, "p-1"))), false) + ); +} + +#[test] +async fn a_save_without_a_parent_or_with_a_parent_that_the_scope_does_not_hold_reads_each_file() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = three_file_tree(); + + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), tree.path(), None) + .await + .unwrap(); + store + .save( + &scope, + &name("p-3"), + tree.path(), + Some((&name("p-missing"), ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + let files = snapshot_files(storage, &scope).await; + + assert_eq!( + (read_counts(&files, "p-2"), read_counts(&files, "p-3")), + (Some(((3, 0, 0), None)), Some(((3, 0, 0), None))) + ); +} + #[test] async fn a_failed_read_of_a_snapshot_file_fails_stat_and_list_with_a_retryable_storage_error() { let refuse = Arc::new(AtomicBool::new(false)); @@ -1174,7 +1356,7 @@ async fn a_failed_read_of_a_snapshot_file_fails_stat_and_list_with_a_retryable_s let scope = new_scope(); let tree = one_file_tree("kept"); store - .save(&scope, &name("p-kept"), tree.path()) + .save(&scope, &name("p-kept"), tree.path(), None) .await .unwrap(); refuse.store(true, Ordering::SeqCst); @@ -1210,7 +1392,7 @@ async fn a_snapshot_file_that_is_gone_after_the_listing_is_left_out() { let scope = new_scope(); let tree = one_file_tree("kept"); store - .save(&scope, &name("p-kept"), tree.path()) + .save(&scope, &name("p-kept"), tree.path(), None) .await .unwrap(); inner @@ -1235,7 +1417,7 @@ async fn a_delete_that_frees_nothing_writes_no_ledger() { let scope = new_scope(); let tree = one_file_tree("kept"); store - .save(&scope, &name("p-kept"), tree.path()) + .save(&scope, &name("p-kept"), tree.path(), None) .await .unwrap(); @@ -1287,7 +1469,9 @@ async fn a_save_of_a_relative_directory_path_gives_source_and_writes_nothing() { ); let scope = new_scope(); - let saved = store.save(&scope, &name("p-relative"), relative).await; + let saved = store + .save(&scope, &name("p-relative"), relative, None) + .await; assert!( matches!(saved, Err(SnapshotStoreError::Source(_))), @@ -1308,7 +1492,7 @@ async fn a_save_of_a_regular_file_gives_source_and_writes_nothing() { let scope = new_scope(); let saved = store - .save(&scope, &name("p-file"), &tree.path().join("file.txt")) + .save(&scope, &name("p-file"), &tree.path().join("file.txt"), None) .await; assert!( @@ -1328,7 +1512,7 @@ async fn the_ledger_counts_the_packed_bytes_that_the_deleted_snapshot_added() { let scope = new_scope(); let tree = fixture_tree(); store - .save(&scope, &name("p-deleted"), tree.path()) + .save(&scope, &name("p-deleted"), tree.path(), None) .await .unwrap(); let added = snapshot_files(storage.clone(), &scope) @@ -1360,7 +1544,7 @@ async fn a_config_write_that_fails_gives_a_storage_error_with_that_failure() { let scope = new_scope(); let tree = one_file_tree("never saved"); - let saved = store.save(&scope, &name("p-1"), tree.path()).await; + let saved = store.save(&scope, &name("p-1"), tree.path(), None).await; assert!( matches!( @@ -1381,11 +1565,11 @@ async fn a_prune_that_fails_without_a_storage_failure_gives_storage_that_is_not_ let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store - .save(&scope, &name("p-deleted"), deleted_tree.path()) + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept_tree.path()) + .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); let packs = storage @@ -1469,7 +1653,10 @@ async fn the_writes_of_a_save_run_at_nice_19() { let scope = new_scope(); let tree = fixture_tree(); - store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); let writes = taken_calls(&calls, "write"); assert_eq!( @@ -1492,11 +1679,11 @@ async fn the_writes_of_a_prune_run_at_nice_19() { let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store - .save(&scope, &name("p-deleted"), deleted_tree.path()) + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept_tree.path()) + .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); taken_calls(&calls, "write"); @@ -1524,7 +1711,10 @@ async fn the_storage_calls_of_a_restore_run_at_the_nice_value_of_the_process() { let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); - store.save(&scope, &name("p-1"), tree.path()).await.unwrap(); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); std::mem::take(&mut *calls.lock().unwrap()); let restored = restored_listing(&store, &scope, &name("p-1")).await; @@ -1561,11 +1751,11 @@ async fn after_saves_and_prunes_the_pools_keep_the_nice_value_of_the_process() { let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store - .save(&scope, &name("p-deleted"), deleted_tree.path()) + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept_tree.path()) + .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); store.delete(&scope, &name("p-deleted")).await.unwrap(); @@ -1627,11 +1817,11 @@ async fn the_storage_calls_of_the_rayon_workers_of_a_prune_that_repacks_run_at_n let kept = Scratch::new(); write_tree(kept.path(), &[("kept.txt", file(b"kept content"))]); store - .save(&scope, &name("p-both"), both.path()) + .save(&scope, &name("p-both"), both.path(), None) .await .unwrap(); store - .save(&scope, &name("p-kept"), kept.path()) + .save(&scope, &name("p-kept"), kept.path(), None) .await .unwrap(); std::mem::take(&mut *calls.lock().unwrap()); @@ -1689,11 +1879,11 @@ async fn the_global_rayon_pool_keeps_the_nice_value_of_the_process_after_saves_w let scope = new_scope(); let (first, second) = (one_file_tree("first"), fixture_tree()); store - .save(&scope, &name("p-1"), first.path()) + .save(&scope, &name("p-1"), first.path(), None) .await .unwrap(); store - .save(&scope, &name("p-2"), second.path()) + .save(&scope, &name("p-2"), second.path(), None) .await .unwrap(); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index a50db65f08..d10f6697e4 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -967,7 +967,7 @@ fn set_modified(path: &Path, time: std::time::SystemTime) { /// Copies each file of the flat tree `from` into the new directory `to`, with its modification /// time. Each copy is a new inode with a new change time, as a capture gives. -fn copy_flat_tree(from: &Path, to: &Path) { +pub(super) fn copy_flat_tree(from: &Path, to: &Path) { std::fs::read_dir(from).unwrap().for_each(|entry| { let entry = entry.unwrap(); let target = to.join(entry.file_name()); @@ -992,7 +992,7 @@ const CHANGE_TIME_WAIT: Duration = Duration::from_secs(10); /// changes within one tick get the same change time. The wait changes a probe file in its own /// directory until the change time of the probe is later than the latest change time of the /// files. It fails the test when that does not happen within [`CHANGE_TIME_WAIT`]. -fn wait_past_change_times(files: &[PathBuf]) { +pub(super) fn wait_past_change_times(files: &[PathBuf]) { let latest = files.iter().map(|file| changed_at(file)).max().unwrap(); let probe_directory = Scratch::new(); let probe = probe_directory.path().join("probe"); @@ -1010,7 +1010,7 @@ fn wait_past_change_times(files: &[PathBuf]) { } /// Gives the path of each entry of the directory, in the order of the names. -fn entries(directory: &Path) -> Vec { +pub(super) fn entries(directory: &Path) -> Vec { let mut paths = std::fs::read_dir(directory) .unwrap() .map(|entry| entry.unwrap().path()) @@ -1020,7 +1020,7 @@ fn entries(directory: &Path) -> Vec { } /// Writes a tree of three files into a new directory, and gives the directory. -fn three_file_tree() -> Scratch { +pub(super) fn three_file_tree() -> Scratch { let tree = Scratch::new(); ["a.txt", "b.txt", "c.txt"] .iter() From 8be6236c78bd74555d31496670037017d88f4996 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 10:01:18 -0700 Subject: [PATCH 27/55] Shorten the docs of the change detection and the save options, and return the options from the match --- .../src/filesystem_snapshot/mod.rs | 5 ++-- .../src/filesystem_snapshot/rustic/store.rs | 29 +++++++++---------- 2 files changed, 15 insertions(+), 19 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/mod.rs b/golem-worker-executor/src/filesystem_snapshot/mod.rs index 5cbfe09faf..a3b0ad53d5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/mod.rs @@ -124,10 +124,9 @@ pub(crate) struct SnapshotInfo { /// How a save with a parent finds the files that did not change since the parent. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub(crate) enum ChangeDetection { - /// A file whose size and modification time equal those of the same path in the parent keeps - /// the content of the parent, and the save does not read it. + /// Compares each file with the parent by size and modification time. SizeMtime, - /// The save reads every file. + /// Reads every file. Full, } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index b90dbb8262..2351129eee 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -101,24 +101,24 @@ impl StorePolicy { } } -/// The options of a save of the store: the options of the bridge, and a save that cannot read an -/// entry fails before it writes the snapshot file. A save records no device id, so a restore gives -/// each name of a hard-linked file as its own file. -/// -/// With a parent and `SizeMtime`, rustic compares each file with the parent that the id names, by -/// size and modification time. Without a parent, or with `Full`, rustic uses no parent and reads +/// The options of a save of the store: a failed read of an entry fails the save, and no device id +/// is kept. `SizeMtime` compares with the parent that the id names, and each other case reads /// every file. fn store_backup_options( policy: &StorePolicy, parent: Option<(SnapshotId, ChangeDetection)>, ) -> BackupOptions { - let settings = |detection| SaveSettings { - threads: policy.save_threads, - detection, + let base = |detection| { + backup_options(&SaveSettings { + threads: policy.save_threads, + detection, + }) + .fail_on_read_error(true) + .ignore_save_opts(LocalSourceSaveOptions::default().set_devid(DevIdOption::No)) }; - let options = match parent { + match parent { Some((id, ChangeDetection::SizeMtime)) => { - let options = backup_options(&settings(RusticChangeDetection::SizeMtime)); + let options = base(RusticChangeDetection::SizeMtime); let parent_opts = options .parent_opts .clone() @@ -126,14 +126,11 @@ fn store_backup_options( options.parent_opts(parent_opts) } None | Some((_, ChangeDetection::Full)) => { - let options = backup_options(&settings(RusticChangeDetection::Ctime)); + let options = base(RusticChangeDetection::Ctime); let parent_opts = options.parent_opts.clone().force(true); options.parent_opts(parent_opts) } - }; - options - .fail_on_read_error(true) - .ignore_save_opts(LocalSourceSaveOptions::default().set_devid(DevIdOption::No)) + } } /// The options of a restore of the store. A metadata error fails the restore. The restore does not From b5c2b090e606d6e780694ccfddc634cf2110638a Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 10:01:18 -0700 Subject: [PATCH 28/55] Pass the name of low-priority work as a static str, and read the thread id of the test nice with try_from --- .../src/filesystem_snapshot/rustic/priority.rs | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs index e03448bef9..71be890c12 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/priority.rs @@ -35,7 +35,7 @@ pub(super) struct LowPriority { pub(super) lower: fn() -> std::io::Result<()>, /// Builds the rayon pool of the work, with the name and the thread count. pub(super) build_pool: - fn(&str, Option) -> Result, + fn(&'static str, Option) -> Result, } impl LowPriority { @@ -53,7 +53,7 @@ impl LowPriority { /// Linux the work runs as it is. pub(super) fn run( self, - name: &str, + name: &'static str, work: impl FnOnce() -> anyhow::Result + Send + 'static, ) -> anyhow::Result { if cfg!(target_os = "linux") { @@ -67,12 +67,11 @@ impl LowPriority { /// of a step gives a warning, and the work runs without that step. fn on_own_thread( self, - name: &str, + name: &'static str, work: impl FnOnce() -> anyhow::Result + Send + 'static, ) -> anyhow::Result { // The work waits in a slot, so the calling thread can still run it when no thread starts. let slot = Arc::new(Mutex::new(Some(work))); - let pool_name = name.to_string(); let spawned = std::thread::Builder::new().name(name.to_string()).spawn({ let slot = slot.clone(); move || { @@ -82,7 +81,7 @@ impl LowPriority { "Failed to lower the CPU priority of filesystem snapshot work, so it runs at the normal priority" ); } - match (self.build_pool)(&pool_name, self.threads) { + match (self.build_pool)(name, self.threads) { Ok(pool) => pool.install(|| run_taken(&slot)), Err(error) => { warn!( @@ -120,10 +119,9 @@ fn run_taken anyhow::Result>(slot: &Mutex>) -> an /// Builds a rayon pool whose threads get the nice value of the calling thread. fn build_pool( - name: &str, + name: &'static str, threads: Option, ) -> Result { - let name = name.to_string(); ThreadPoolBuilder::new() .num_threads(threads.map_or(0, NonZeroUsize::get)) .thread_name(move |index| format!("{name}-{index}")) @@ -153,8 +151,10 @@ fn lower_own_priority() -> std::io::Result<()> { /// Gives the nice value of the calling thread. #[cfg(all(test, target_os = "linux"))] pub(super) fn own_nice() -> i32 { - // SAFETY: `gettid` has no preconditions, and `getpriority` only reads its arguments. - unsafe { libc::getpriority(libc::PRIO_PROCESS, libc::gettid() as libc::id_t) } + // SAFETY: `gettid` has no preconditions. + let thread = libc::id_t::try_from(unsafe { libc::gettid() }).unwrap(); + // SAFETY: `getpriority` only reads its arguments. + unsafe { libc::getpriority(libc::PRIO_PROCESS, thread) } } #[cfg(all(test, target_os = "linux"))] From f7af20e758e0072b2b5957eb5e1d7f0a93b7ebe7 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 10:03:12 -0700 Subject: [PATCH 29/55] Check that a failed read of the snapshot files while a save finds its parent gives storage and publishes nothing --- .../filesystem_snapshot/rustic/store/tests.rs | 52 +++++++++++++++++++ 1 file changed, 52 insertions(+) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 16b7c1e7ae..4f7b3d2422 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1338,6 +1338,58 @@ async fn a_save_without_a_parent_or_with_a_parent_that_the_scope_does_not_hold_r ); } +#[test] +async fn a_save_whose_read_of_the_snapshot_files_fails_while_it_finds_the_parent_gives_storage_and_publishes_nothing() + { + let refuse = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let refuse = refuse.clone(); + move |op_label, path| { + if refuse.load(Ordering::SeqCst) && op_label == "read" && path.starts_with("snapshots") + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("parent"); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + refuse.store(true, Ordering::SeqCst); + let before = storage.calls().len(); + + let saved = store + .save( + &scope, + &name("p-2"), + tree.path(), + Some((&name("p-1"), ChangeDetection::SizeMtime)), + ) + .await; + let publishes = storage.calls()[before..] + .iter() + .filter(|(op_label, _)| *op_label == "publish") + .count(); + refuse.store(false, Ordering::SeqCst); + + assert!( + saved.as_ref().is_err_and(|error| is_storage(error, true)), + "{saved:?}" + ); + assert_eq!( + (publishes, listed_names(&store, &scope).await), + (0, vec!["p-1".to_string()]) + ); +} + #[test] async fn a_failed_read_of_a_snapshot_file_fails_stat_and_list_with_a_retryable_storage_error() { let refuse = Arc::new(AtomicBool::new(false)); From ff8047efced28e6745eb5d1d530d50c5df4ebee7 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 10:07:22 -0700 Subject: [PATCH 30/55] Check every storage call of the work of a save and of a prune at nice 19, reads included --- .../filesystem_snapshot/rustic/store/tests.rs | 77 ++++++++++++------- 1 file changed, 51 insertions(+), 26 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 4f7b3d2422..6b51ff640a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1683,49 +1683,76 @@ fn nice_recording_storage() -> (Arc, NiceCalls) { (storage, calls) } -/// Takes the recorded calls with the operation label. +/// Takes the recorded calls as the operation label, the path and the nice value. #[cfg(target_os = "linux")] -fn taken_calls(calls: &NiceCalls, op_label: &str) -> Vec<(String, i32)> { +fn taken_calls(calls: &NiceCalls) -> Vec<(String, String, i32)> { std::mem::take( &mut *calls .lock() .unwrap_or_else(std::sync::PoisonError::into_inner), ) .into_iter() - .filter(|(op, _, _, _)| op == op_label) - .map(|(_, path, _, nice)| (path, nice)) + .map(|(op_label, path, _, nice)| (op_label, path, nice)) .collect() } +/// Gives the operation labels of the calls, and each call that does not run at nice 19. +#[cfg(target_os = "linux")] +fn labels_and_calls_not_at_nice_19( + calls: &[(String, String, i32)], +) -> (Vec<&str>, Vec<&(String, String, i32)>) { + let mut labels = calls + .iter() + .map(|(op_label, _, _)| op_label.as_str()) + .collect::>(); + labels.sort_unstable(); + labels.dedup(); + let not_at_nice_19 = calls.iter().filter(|(_, _, nice)| *nice != 19).collect(); + (labels, not_at_nice_19) +} + #[cfg(target_os = "linux")] #[test] -async fn the_writes_of_a_save_run_at_nice_19() { +async fn the_storage_calls_of_a_save_run_at_nice_19() { + // The publish of the snapshot file runs on the async runtime after the work, so it keeps + // the normal priority. The second save reads the config, the index, the snapshot files and + // the trees of its parent, and writes the added file. let (storage, calls) = nice_recording_storage(); let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); - store .save(&scope, &name("p-1"), tree.path(), None) .await .unwrap(); - let writes = taken_calls(&calls, "write"); + std::fs::write(tree.path().join("added.txt"), b"added").unwrap(); + taken_calls(&calls); + + store + .save( + &scope, + &name("p-2"), + tree.path(), + Some((&name("p-1"), ChangeDetection::SizeMtime)), + ) + .await + .unwrap(); + let work = taken_calls(&calls) + .into_iter() + .filter(|(op_label, _, _)| op_label != "publish") + .collect::>(); assert_eq!( - ( - writes.is_empty(), - writes - .iter() - .filter(|(_, nice)| *nice != 19) - .collect::>() - ), - (false, Vec::<&(String, i32)>::new()) + labels_and_calls_not_at_nice_19(&work), + (vec!["list", "read", "stat", "write"], Vec::new()) ); } #[cfg(target_os = "linux")] #[test] -async fn the_writes_of_a_prune_run_at_nice_19() { +async fn the_storage_calls_of_a_prune_run_at_nice_19() { + // The forget of a delete runs before the ledger read at the normal priority, and the ledger + // read and writes run on the async runtime. The prune runs between them. let (storage, calls) = nice_recording_storage(); let store = store(storage, policy(LONG_DEADLINE, 1, Duration::ZERO)); let scope = new_scope(); @@ -1738,20 +1765,18 @@ async fn the_writes_of_a_prune_run_at_nice_19() { .save(&scope, &name("p-kept"), kept_tree.path(), None) .await .unwrap(); - taken_calls(&calls, "write"); + taken_calls(&calls); store.delete(&scope, &name("p-deleted")).await.unwrap(); - let writes = taken_calls(&calls, "write"); + let prune = taken_calls(&calls) + .into_iter() + .skip_while(|(op_label, _, _)| op_label != "read_ledger") + .filter(|(op_label, _, _)| op_label != "read_ledger" && op_label != "write_ledger") + .collect::>(); assert_eq!( - ( - writes.is_empty(), - writes - .iter() - .filter(|(_, nice)| *nice != 19) - .collect::>() - ), - (false, Vec::<&(String, i32)>::new()) + labels_and_calls_not_at_nice_19(&prune), + (vec!["delete", "list", "read", "stat", "write"], Vec::new()) ); } From 361a1d2cfd690116b20293e6597253399b0826bd Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 22:21:52 -0700 Subject: [PATCH 31/55] Set the approved defaults of the filesystem snapshot store: a 60 s deadline, 6 restore readers and 2 save threads --- golem-worker-executor/src/services/golem_config.rs | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/services/golem_config.rs b/golem-worker-executor/src/services/golem_config.rs index 3ab3c64114..0b3a388b95 100644 --- a/golem-worker-executor/src/services/golem_config.rs +++ b/golem-worker-executor/src/services/golem_config.rs @@ -2341,13 +2341,13 @@ impl SafeDisplay for FilesystemPressureConfig { /// The default of [`FilesystemSnapshotStoreConfig::storage_call_deadline`]. The slowest measured /// call on S3 took 1.7 s, and the value stays at least 10 times the slowest measured call. -pub const DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE: Duration = Duration::from_secs(30); +pub const DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE: Duration = Duration::from_secs(60); /// The default of [`FilesystemSnapshotStoreConfig::restore_reader_threads`]. -const DEFAULT_FILESYSTEM_SNAPSHOT_RESTORE_READER_THREADS: usize = 4; +const DEFAULT_FILESYSTEM_SNAPSHOT_RESTORE_READER_THREADS: usize = 6; /// The default of [`FilesystemSnapshotStoreConfig::save_threads`]. -const DEFAULT_FILESYSTEM_SNAPSHOT_SAVE_THREADS: usize = 4; +const DEFAULT_FILESYSTEM_SNAPSHOT_SAVE_THREADS: usize = 2; /// Tells whether the executor keeps filesystem snapshots, and gives the settings of the store. #[derive(Clone, Debug, Serialize, Deserialize)] @@ -3010,7 +3010,7 @@ mod tests { store.restore_reader_threads().get(), store.save_threads().get(), ), - (Duration::from_secs(30), 4, 4) + (Duration::from_secs(60), 6, 2) ); } From 3f8546400b0a0f6ba043280159b870f57850402b Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 22:21:52 -0700 Subject: [PATCH 32/55] Keep marked packs for 15 minutes and repack fast in a prune of the store --- golem-worker-executor/src/filesystem_snapshot/rustic/store.rs | 4 ++-- .../src/filesystem_snapshot/rustic/store/tests.rs | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 2351129eee..524ae4c539 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -63,7 +63,7 @@ const PRUNE_THRESHOLD_BYTES: u64 = 64 * 1024 * 1024; /// How long a pack that a prune marks stays before a later prune deletes it. It is also the /// shortest time between two prunes of one scope. It must be longer than the longest save and the /// longest restore. -const PRUNE_GRACE: Duration = Duration::from_secs(3600); +const PRUNE_GRACE: Duration = Duration::from_secs(15 * 60); /// The settings of the store: the rustic settings of each operation, and the prune threshold. #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -92,7 +92,7 @@ impl StorePolicy { save_threads: Some(config.save_threads()), restore_reader_threads: config.restore_reader_threads(), prune: PruneSettings { - fast_repack: false, + fast_repack: true, keep_delete: PRUNE_GRACE, repack: RepackLimits::Rustic, }, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 6b51ff640a..6c6a9e85e7 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -221,8 +221,8 @@ fn the_policy_takes_the_configured_values_and_the_options_are_strict() { Duration::from_secs(30), Some(3), 4, - Duration::from_secs(3600), - false, + Duration::from_secs(15 * 60), + true, RepackLimits::Rustic, 64 * 1024 * 1024, ) From 1fdc959be3d051f9df71977acfec8e5acaf7a400 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 23:04:33 -0700 Subject: [PATCH 33/55] Prune a scope when its deleted snapshots free 10 percent of the size of its packs --- .../src/filesystem_snapshot/rustic/prune.rs | 144 ++++++++++---- .../src/filesystem_snapshot/rustic/store.rs | 32 +-- .../filesystem_snapshot/rustic/store/tests.rs | 186 ++++++++++++++---- 3 files changed, 270 insertions(+), 92 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index e412271a5d..8b8cd5f8ce 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -59,25 +59,61 @@ impl PruneLedger { } } +/// The path of the packs of a repository, relative to the root of the namespace of the scope. +const DATA_PATH: &str = "data"; + +/// A share of the size of a repository, in percent. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) struct Percent(pub(super) u16); + +impl Percent { + /// Gives this share of the bytes, rounded down. + fn of(self, bytes: u64) -> u64 { + u64::try_from(u128::from(bytes) * u128::from(self.0) / 100).unwrap_or(u64::MAX) + } +} + +/// Tells whether the grace period passed at `now` since the last prune. +fn grace_passed(ledger: &PruneLedger, now: Timestamp, grace: Duration) -> bool { + ledger.last_prune.is_none_or(|last| { + now.to_millis() + >= last + .to_millis() + .saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) + }) +} + +/// Tells whether [`prune_due`] needs the size of the repository at `now`. Only freed bytes after +/// the grace period, without marked packs, need it. +pub(super) fn needs_repository_size(ledger: &PruneLedger, now: Timestamp, grace: Duration) -> bool { + grace_passed(ledger, now, grace) && ledger.freed_bytes > 0 && !ledger.awaiting_removal +} + /// Tells whether a prune is due at `now`. /// /// A prune is due when the grace period passed since the last prune, and the freed bytes reach the -/// threshold or the last prune marked packs. A threshold of zero counts as one byte, so a prune -/// never runs for a scope that freed nothing and marked nothing. +/// threshold share of `repository_bytes` or the last prune marked packs. A threshold of zero bytes +/// counts as one byte, so a prune never runs for a scope that freed nothing and marked nothing. pub(super) fn prune_due( ledger: &PruneLedger, now: Timestamp, - threshold: u64, + repository_bytes: u64, + threshold: Percent, grace: Duration, ) -> bool { - let grace_passed = ledger.last_prune.is_none_or(|last| { - now.to_millis() - >= last - .to_millis() - .saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) - }); - let work = ledger.freed_bytes >= threshold.max(1) || ledger.awaiting_removal; - grace_passed && work + let work = + ledger.freed_bytes >= threshold.of(repository_bytes).max(1) || ledger.awaiting_removal; + grace_passed(ledger, now, grace) && work +} + +/// Gives the size of the repository of the scope: the sum of the sizes of its packs. +pub(super) async fn repository_bytes(files: &SnapshotFiles) -> anyhow::Result { + Ok(files + .list_below("list_data", Path::new(DATA_PATH)) + .await? + .iter() + .map(|blob| blob.size) + .fold(0, u64::saturating_add)) } /// Reads the ledger of the scope. A scope without a ledger, or with a ledger that does not parse, @@ -109,7 +145,10 @@ pub(super) async fn write_ledger( #[cfg(test)] mod tests { use super::super::files::SnapshotFiles; - use super::{LEDGER_PATH, PruneLedger, prune_due, read_ledger, write_ledger}; + use super::{ + LEDGER_PATH, Percent, PruneLedger, needs_repository_size, prune_due, read_ledger, + write_ledger, + }; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; use golem_service_base::storage::blob::BlobStorageNamespace; @@ -121,9 +160,9 @@ mod tests { use test_r::test; use uuid::Uuid; - const MIB: u64 = 1024 * 1024; - const THRESHOLD: u64 = 64 * MIB; - const GRACE: Duration = Duration::from_secs(3600); + const TEN_PERCENT: Percent = Percent(10); + const GRACE: Duration = Duration::from_secs(15 * 60); + const GRACE_MILLIS: u64 = 15 * 60 * 1000; const DEADLINE: Duration = Duration::from_secs(2); fn at(millis: u64) -> Timestamp { @@ -153,28 +192,39 @@ mod tests { } #[test] - fn a_prune_is_due_when_the_freed_bytes_reach_the_threshold() { + fn a_prune_is_due_when_the_freed_bytes_reach_ten_percent_of_the_repository() { let now = at(10_000_000); + let due = |freed, repository_bytes| { + prune_due( + &ledger(freed, None, false), + now, + repository_bytes, + TEN_PERCENT, + GRACE, + ) + }; assert_eq!( [ - prune_due(&ledger(THRESHOLD - 1, None, false), now, THRESHOLD, GRACE), - prune_due(&ledger(THRESHOLD, None, false), now, THRESHOLD, GRACE), - prune_due(&ledger(THRESHOLD + 1, None, false), now, THRESHOLD, GRACE), + due(99, 1000), + due(100, 1000), + due(101, 1000), + due(0, 0), + due(1, 0), ], - [false, true, true] + [false, true, true, false, true] ); } #[test] fn no_second_prune_runs_within_the_grace_period() { let last = 1_000_000; - let grace_millis = 3_600_000; let full = |now| { prune_due( - &ledger(THRESHOLD, Some(last), true), + &ledger(1000, Some(last), true), at(now), - THRESHOLD, + 1000, + TEN_PERCENT, GRACE, ) }; @@ -182,9 +232,9 @@ mod tests { assert_eq!( [ full(last), - full(last + grace_millis - 1), - full(last + grace_millis), - full(last + grace_millis + 1), + full(last + GRACE_MILLIS - 1), + full(last + GRACE_MILLIS), + full(last + GRACE_MILLIS + 1), ], [false, false, true, true] ); @@ -193,28 +243,50 @@ mod tests { #[test] fn marked_packs_make_a_prune_due_after_the_grace_period_without_freed_bytes() { let last = 1_000_000; - let after_grace = at(last + 3_600_000); + let after_grace = at(last + GRACE_MILLIS); + let due = |awaiting_removal| { + prune_due( + &ledger(0, Some(last), awaiting_removal), + after_grace, + 1000, + TEN_PERCENT, + GRACE, + ) + }; + + assert_eq!([due(true), due(false)], [true, false]); + } + + #[test] + fn a_zero_threshold_prunes_after_each_delete_that_freed_bytes() { + let now = at(10_000_000); + let due = |ledger| prune_due(&ledger, now, 1000, Percent(0), Duration::ZERO); assert_eq!( [ - prune_due(&ledger(0, Some(last), true), after_grace, THRESHOLD, GRACE), - prune_due(&ledger(0, Some(last), false), after_grace, THRESHOLD, GRACE), + due(ledger(0, None, false)), + due(ledger(1, None, false)), + due(ledger(1, Some(10_000_000), false)), ], - [true, false] + [false, true, true] ); } #[test] - fn a_zero_threshold_prunes_after_each_delete_that_freed_bytes() { - let now = at(10_000_000); + fn only_freed_bytes_after_the_grace_period_without_marked_packs_need_the_repository_size() { + let last = 1_000_000; + let needs = |freed, awaiting_removal, now| { + needs_repository_size(&ledger(freed, Some(last), awaiting_removal), at(now), GRACE) + }; assert_eq!( [ - prune_due(&ledger(0, None, false), now, 0, Duration::ZERO), - prune_due(&ledger(1, None, false), now, 0, Duration::ZERO), - prune_due(&ledger(1, Some(10_000_000), false), now, 0, Duration::ZERO), + needs(1, false, last + GRACE_MILLIS), + needs(1, false, last + GRACE_MILLIS - 1), + needs(0, false, last + GRACE_MILLIS), + needs(1, true, last + GRACE_MILLIS), ], - [false, true, true] + [true, false, false, false] ); } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 524ae4c539..4490ec1c0a 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -24,7 +24,10 @@ use super::backend::BlobBackend; use super::fault::{Operation, classify, is_file_missing, is_storage_failure, storage_failure}; use super::files::SnapshotFiles; use super::priority::LowPriority; -use super::prune::{PruneLedger, prune_due, read_ledger, write_ledger}; +use super::prune::{ + Percent, PruneLedger, needs_repository_size, prune_due, read_ledger, repository_bytes, + write_ledger, +}; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; use super::{ @@ -57,8 +60,9 @@ use tokio::runtime::Handle; use tokio_util::sync::{CancellationToken, DropGuard}; use tokio_util::task::TaskTracker; -/// The packed bytes that deleted snapshots must free before a delete prunes the scope. -const PRUNE_THRESHOLD_BYTES: u64 = 64 * 1024 * 1024; +/// The share of the size of the repository that deleted snapshots must free before a delete prunes +/// the scope. +const PRUNE_THRESHOLD: Percent = Percent(10); /// How long a pack that a prune marks stays before a later prune deletes it. It is also the /// shortest time between two prunes of one scope. It must be longer than the longest save and the @@ -79,8 +83,9 @@ pub(super) struct StorePolicy { pub(super) restore_reader_threads: NonZeroUsize, /// The settings of a prune. `keep_delete` is also the shortest time between two prunes. pub(super) prune: PruneSettings, - /// The packed bytes that deleted snapshots must free before a delete prunes. - pub(super) prune_threshold: u64, + /// The share of the size of the repository that deleted snapshots must free before a delete + /// prunes. + pub(super) prune_threshold: Percent, } impl StorePolicy { @@ -96,7 +101,7 @@ impl StorePolicy { keep_delete: PRUNE_GRACE, repack: RepackLimits::Rustic, }, - prune_threshold: PRUNE_THRESHOLD_BYTES, + prune_threshold: PRUNE_THRESHOLD, } } } @@ -260,7 +265,7 @@ impl RusticSnapshotStore { /// Adds the freed bytes to the ledger of the scope, and prunes the repository when a prune is /// due. The ledger keeps the freed bytes before the prune starts, so a delete that runs again - /// after a failed prune prunes again. + /// after a failed prune prunes again. It lists the packs only when their size can make a prune due. async fn prune_when_due( &self, scope: &SnapshotScope, @@ -278,12 +283,13 @@ impl RusticSnapshotStore { .map_err(storage_failure)?; } let now = Timestamp::now_utc(); - if !prune_due( - &ledger, - now, - self.policy.prune_threshold, - self.policy.prune.keep_delete, - ) { + let grace = self.policy.prune.keep_delete; + let size = if needs_repository_size(&ledger, now, grace) { + repository_bytes(&files).await.map_err(storage_failure)? + } else { + 0 + }; + if !prune_due(&ledger, now, size, self.policy.prune_threshold, grace) { return Ok(()); } let backend = Arc::new(self.backend(scope, token)?); diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 6c6a9e85e7..0144dae408 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -18,7 +18,7 @@ //! give the store a short or a long deadline and a prune policy that the test controls. use super::super::files::SnapshotFiles; -use super::super::prune::{PruneLedger, read_ledger}; +use super::super::prune::{Percent, PruneLedger, read_ledger}; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; @@ -52,6 +52,12 @@ use test_r::{test, test_gen}; /// The longest time that a test waits for an operation or for the work of a store to end. const LIMIT: Duration = Duration::from_secs(10); +/// A prune threshold that a delete never reaches. +const NEVER: Percent = Percent(u16::MAX); + +/// A prune threshold of zero bytes, so each delete that frees bytes prunes. +const ALWAYS: Percent = Percent(0); + /// A deadline that no call of these tests reaches, so a held call ends only by a cancel. const LONG_DEADLINE: Duration = Duration::from_secs(60); @@ -68,7 +74,7 @@ fn key() -> RepositoryKey { /// The policy of the configuration, with the deadline, the prune threshold and the grace period /// of the test. -fn policy(deadline: Duration, prune_threshold: u64, grace: Duration) -> StorePolicy { +fn policy(deadline: Duration, prune_threshold: Percent, grace: Duration) -> StorePolicy { StorePolicy { deadline, prune: PruneSettings { @@ -224,7 +230,7 @@ fn the_policy_takes_the_configured_values_and_the_options_are_strict() { Duration::from_secs(15 * 60), true, RepackLimits::Rustic, - 64 * 1024 * 1024, + Percent(10), ) ); assert_eq!( @@ -271,7 +277,7 @@ async fn a_tree_saved_through_a_proc_self_fd_path_is_stored_below_the_root() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = fixture_tree(); @@ -325,7 +331,7 @@ async fn a_save_whose_index_write_fails_publishes_nothing_and_leaves_the_name_fr } } }); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); @@ -372,7 +378,7 @@ async fn a_publish_that_reaches_the_deadline_and_lands_late_publishes_nothing() }); let store = store( storage.clone(), - policy(Duration::from_secs(1), u64::MAX, Duration::ZERO), + policy(Duration::from_secs(1), NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = one_file_tree("late"); @@ -408,7 +414,7 @@ async fn a_second_save_of_an_unchanged_tree_writes_no_pack() { ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = fixture_tree(); @@ -454,7 +460,7 @@ async fn two_stores_that_create_one_repository_at_the_same_time_both_save() { Script::Pass } }); - let policy = policy(LONG_DEADLINE, u64::MAX, Duration::ZERO); + let policy = policy(LONG_DEADLINE, NEVER, Duration::ZERO); let (first, second) = ( store(storage.clone(), policy), store(storage.clone(), policy), @@ -524,7 +530,10 @@ async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { } } }); - let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); let scope = new_scope(); let (old_tree, new_tree) = (one_file_tree("old"), fixture_tree()); store @@ -583,7 +592,7 @@ async fn a_restore_whose_pack_reads_fail_gives_a_retryable_storage_error() { } } }); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); store @@ -615,7 +624,7 @@ async fn a_save_of_a_file_without_read_permission_gives_source_with_permission_d let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = one_file_tree("readable"); @@ -667,7 +676,7 @@ async fn a_restore_that_cannot_set_an_extended_attribute_gives_destination() { } let store = store( Arc::new(InMemoryBlobStorage::new()), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); store @@ -711,7 +720,7 @@ async fn a_snapshot_file_that_fails_its_check_is_left_out_of_list_and_makes_an_u let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = one_file_tree("kept"); @@ -752,7 +761,10 @@ async fn a_snapshot_file_that_fails_its_check_is_left_out_of_list_and_makes_an_u #[test] async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_period() { let storage = Arc::new(InMemoryBlobStorage::new()); - let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store @@ -783,12 +795,89 @@ async fn a_delete_past_the_threshold_prunes_and_the_packs_go_after_the_grace_per ); } +/// Counts the listings of the packs among the recorded calls. +fn data_listings(calls: &[(&'static str, String)]) -> usize { + calls + .iter() + .filter(|(op_label, _)| *op_label == "list_data") + .count() +} + +#[test] +async fn a_delete_within_the_grace_period_does_not_list_the_packs() { + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + let (first, second) = (one_file_tree("first"), one_file_tree("second")); + store + .save(&scope, &name("p-1"), first.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-2"), second.path(), None) + .await + .unwrap(); + let before_first = storage.calls().len(); + + store.delete(&scope, &name("p-1")).await.unwrap(); + let before_second = storage.calls().len(); + store.delete(&scope, &name("p-2")).await.unwrap(); + let calls = storage.calls(); + + assert_eq!( + ( + data_listings(&calls[before_first..before_second]), + data_listings(&calls[before_second..]), + ledger(&storage, &scope).await.freed_bytes > 0, + ), + (1, 0, true) + ); +} + +#[test] +async fn a_failed_listing_of_the_packs_gives_storage_and_records_no_prune() { + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "list_data" { + Script::Refuse + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), one_file_tree("kept")); + store + .save(&scope, &name("p-deleted"), deleted_tree.path(), None) + .await + .unwrap(); + store + .save(&scope, &name("p-kept"), kept_tree.path(), None) + .await + .unwrap(); + + let deleted = store.delete(&scope, &name("p-deleted")).await; + let after = ledger(&storage, &scope).await; + + assert!( + deleted.as_ref().is_err_and(|error| is_storage(error, true)), + "{deleted:?}" + ); + assert_eq!((after.freed_bytes > 0, after.last_prune), (true, None)); +} + #[test] async fn a_delete_below_the_threshold_does_not_prune() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), one_file_tree("kept")); @@ -820,7 +909,7 @@ async fn no_second_prune_runs_within_the_grace_period() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, 1, Duration::from_secs(3600)), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), ); let scope = new_scope(); let trees = [one_file_tree("a"), one_file_tree("b"), one_file_tree("c")]; @@ -868,7 +957,10 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { } } }); - let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store @@ -906,7 +998,10 @@ async fn a_delete_whose_prune_fails_gives_storage_and_a_retry_prunes() { #[test] async fn a_deleted_scope_holds_no_blob() { let storage = Arc::new(InMemoryBlobStorage::new()); - let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); let scope = new_scope(); let (first, second) = (one_file_tree("first"), one_file_tree("second")); store @@ -942,7 +1037,7 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); store( counted.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ) .save(&new_scope(), &name("p-dropped"), tree.path(), None) .await @@ -962,7 +1057,7 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam Script::Pass } }); - let policy = policy(LONG_DEADLINE, u64::MAX, Duration::ZERO); + let policy = policy(LONG_DEADLINE, NEVER, Duration::ZERO); let dropping = store(storage.clone(), policy); let scope = new_scope(); let ended = drop_when( @@ -1023,7 +1118,7 @@ async fn shut_down_ends_running_operations_before_it_returns() { }); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = fixture_tree(); @@ -1093,7 +1188,7 @@ async fn a_dropped_operation_stops_its_blocking_work() { }); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = fixture_tree(); @@ -1208,7 +1303,7 @@ async fn a_size_and_mtime_save_of_a_copied_tree_reads_no_unchanged_file() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = three_file_tree(); @@ -1242,7 +1337,7 @@ async fn a_full_save_reads_each_file() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = three_file_tree(); @@ -1270,7 +1365,7 @@ async fn the_parent_of_a_save_is_the_named_snapshot_also_when_a_newer_snapshot_e let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = three_file_tree(); @@ -1308,7 +1403,7 @@ async fn a_save_without_a_parent_or_with_a_parent_that_the_scope_does_not_hold_r let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = three_file_tree(); @@ -1355,7 +1450,7 @@ async fn a_save_whose_read_of_the_snapshot_files_fails_while_it_finds_the_parent }); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = one_file_tree("parent"); @@ -1404,7 +1499,7 @@ async fn a_failed_read_of_a_snapshot_file_fails_stat_and_list_with_a_retryable_s } } }); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = one_file_tree("kept"); store @@ -1440,7 +1535,7 @@ async fn a_snapshot_file_that_is_gone_after_the_listing_is_left_out() { } } }); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = one_file_tree("kept"); store @@ -1464,7 +1559,7 @@ async fn a_delete_that_frees_nothing_writes_no_ledger() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = one_file_tree("kept"); @@ -1517,7 +1612,7 @@ async fn a_save_of_a_relative_directory_path_gives_source_and_writes_nothing() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); @@ -1539,7 +1634,7 @@ async fn a_save_of_a_regular_file_gives_source_and_writes_nothing() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); @@ -1559,7 +1654,7 @@ async fn the_ledger_counts_the_packed_bytes_that_the_deleted_snapshot_added() { let storage = Arc::new(InMemoryBlobStorage::new()); let store = store( storage.clone(), - policy(LONG_DEADLINE, u64::MAX, Duration::ZERO), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), ); let scope = new_scope(); let tree = fixture_tree(); @@ -1592,7 +1687,7 @@ async fn a_config_write_that_fails_gives_a_storage_error_with_that_failure() { Script::Pass } }); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = one_file_tree("never saved"); @@ -1613,7 +1708,10 @@ async fn a_prune_that_fails_without_a_storage_failure_gives_storage_that_is_not_ // Packs of zeros with the sizes of the index give the prune a decryption error, not a failed // storage call. The forget before the prune has succeeded, so the delete gives `Storage`. let storage = Arc::new(InMemoryBlobStorage::new()); - let store = store(storage.clone(), policy(LONG_DEADLINE, 1, Duration::ZERO)); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store @@ -1718,7 +1816,7 @@ async fn the_storage_calls_of_a_save_run_at_nice_19() { // the normal priority. The second save reads the config, the index, the snapshot files and // the trees of its parent, and writes the added file. let (storage, calls) = nice_recording_storage(); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); store @@ -1751,10 +1849,10 @@ async fn the_storage_calls_of_a_save_run_at_nice_19() { #[cfg(target_os = "linux")] #[test] async fn the_storage_calls_of_a_prune_run_at_nice_19() { - // The forget of a delete runs before the ledger read at the normal priority, and the ledger - // read and writes run on the async runtime. The prune runs between them. + // The forget of a delete runs before the ledger read at the normal priority. The ledger calls + // and the listing of the packs run on the async runtime, and the prune runs after them. let (storage, calls) = nice_recording_storage(); - let store = store(storage, policy(LONG_DEADLINE, 1, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, ALWAYS, Duration::ZERO)); let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); store @@ -1771,7 +1869,9 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { let prune = taken_calls(&calls) .into_iter() .skip_while(|(op_label, _, _)| op_label != "read_ledger") - .filter(|(op_label, _, _)| op_label != "read_ledger" && op_label != "write_ledger") + .filter(|(op_label, _, _)| { + !["read_ledger", "write_ledger", "list_data"].contains(&op_label.as_str()) + }) .collect::>(); assert_eq!( @@ -1785,7 +1885,7 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { async fn the_storage_calls_of_a_restore_run_at_the_nice_value_of_the_process() { let process_nice = super::super::priority::own_nice(); let (storage, calls) = nice_recording_storage(); - let store = store(storage, policy(LONG_DEADLINE, u64::MAX, Duration::ZERO)); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); let scope = new_scope(); let tree = fixture_tree(); store @@ -1823,7 +1923,7 @@ async fn after_saves_and_prunes_the_pools_keep_the_nice_value_of_the_process() { let process_nice = super::super::priority::own_nice(); let store = store( Arc::new(InMemoryBlobStorage::new()), - policy(LONG_DEADLINE, 1, Duration::ZERO), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), ); let scope = new_scope(); let (deleted_tree, kept_tree) = (one_file_tree("deleted content"), fixture_tree()); @@ -1867,7 +1967,7 @@ async fn the_storage_calls_of_the_rayon_workers_of_a_prune_that_repacks_run_at_n // The deleted snapshot shares a pack with the kept one, so the prune repacks that pack. The // prune reads the index files and repacks with rayon, on the workers of the pool of the prune. let (storage, calls) = nice_recording_storage(); - let base = policy(LONG_DEADLINE, 1, Duration::ZERO); + let base = policy(LONG_DEADLINE, ALWAYS, Duration::ZERO); let store = store( storage, StorePolicy { From 2eb1a1668a960beaedca3988c6fb282970383503 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Thu, 24 Sep 2026 23:11:30 -0700 Subject: [PATCH 34/55] Check the listing of the packs against the grace period alone, with a ledger that holds no marked packs --- .../filesystem_snapshot/rustic/store/tests.rs | 44 ++++++++++++++++--- 1 file changed, 37 insertions(+), 7 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 0144dae408..e08862fb16 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -18,7 +18,7 @@ //! give the store a short or a long deadline and a prune policy that the test controls. use super::super::files::SnapshotFiles; -use super::super::prune::{Percent, PruneLedger, read_ledger}; +use super::super::prune::{Percent, PruneLedger, read_ledger, write_ledger}; use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; @@ -803,14 +803,37 @@ fn data_listings(calls: &[(&'static str, String)]) -> usize { .count() } +/// Writes a ledger with freed bytes, no marked packs, and a last prune at the time. +async fn set_last_prune( + storage: &Arc, + scope: &SnapshotScope, + last_prune: golem_common::model::Timestamp, +) { + let ledger = PruneLedger { + freed_bytes: 1, + last_prune: Some(last_prune), + awaiting_removal: false, + }; + write_ledger( + &SnapshotFiles { + storage: storage.clone(), + namespace: scope.0.clone(), + deadline: Duration::from_secs(2), + }, + &ledger, + ) + .await + .unwrap(); +} + #[test] async fn a_delete_within_the_grace_period_does_not_list_the_packs() { + // The ledger has no marked packs, so only the grace period keeps the first delete from a + // listing. The second delete comes after the grace period and lists the packs one time. + let grace = Duration::from_secs(3600); let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); - let store = store( - storage.clone(), - policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), - ); + let store = store(storage.clone(), policy(LONG_DEADLINE, NEVER, grace)); let scope = new_scope(); let (first, second) = (one_file_tree("first"), one_file_tree("second")); store @@ -821,10 +844,18 @@ async fn a_delete_within_the_grace_period_does_not_list_the_packs() { .save(&scope, &name("p-2"), second.path(), None) .await .unwrap(); + let now = golem_common::model::Timestamp::now_utc(); + set_last_prune(&storage, &scope, now).await; let before_first = storage.calls().len(); store.delete(&scope, &name("p-1")).await.unwrap(); let before_second = storage.calls().len(); + set_last_prune( + &storage, + &scope, + golem_common::model::Timestamp::from(now.to_millis().saturating_sub(2 * 3_600_000)), + ) + .await; store.delete(&scope, &name("p-2")).await.unwrap(); let calls = storage.calls(); @@ -832,9 +863,8 @@ async fn a_delete_within_the_grace_period_does_not_list_the_packs() { ( data_listings(&calls[before_first..before_second]), data_listings(&calls[before_second..]), - ledger(&storage, &scope).await.freed_bytes > 0, ), - (1, 0, true) + (0, 1) ); } From 73bcc7a51eaa49d042ae1664d5badabd1fc6146e Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 03:22:28 -0700 Subject: [PATCH 35/55] Record the packs that no index lists as marked packs of a prune --- .../src/filesystem_snapshot/rustic/mod.rs | 3 + .../src/filesystem_snapshot/rustic/store.rs | 9 ++- .../filesystem_snapshot/rustic/store/tests.rs | 55 +++++++++++++++++++ 3 files changed, 64 insertions(+), 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index ddcebdba84..594499f248 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -293,6 +293,8 @@ pub(super) struct PruneReport { pub(super) marked_bytes_deleted: u64, /// The packs that an earlier prune marked and that stay marked. pub(super) marked_packs_kept: u64, + /// The packs that no index lists. The prune marks each of them. + pub(super) packs_unindexed: u64, /// The bytes of the blobs that snapshots use. pub(super) bytes_used: u64, /// The bytes of the blobs that no snapshot uses. @@ -772,6 +774,7 @@ fn prune_report(stats: &PruneStats) -> PruneReport { marked_packs_deleted: stats.packs_to_delete.remove, marked_bytes_deleted: stats.size_to_delete.remove, marked_packs_kept: stats.packs_to_delete.keep, + packs_unindexed: stats.packs_unref, bytes_used: blobs.used, bytes_unused: blobs.unused, bytes_removed: blobs.remove, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 4490ec1c0a..85ebbda7df 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -697,10 +697,13 @@ fn snapshot_info(snapshot: &SnapshotFile) -> Option { } /// Tells whether a later prune removes packs that this prune leaves marked: unused packs, repacked -/// packs, and packs of an earlier prune whose grace period is not over. The report does not count a -/// marked pack that no index lists, so the next due prune removes it. +/// packs, packs that no index lists, and packs of an earlier prune whose grace period is not over. +/// A marked pack is in an index after the prune, so a later prune does not count it as unindexed. fn leaves_marked_packs(report: &PruneReport) -> bool { - report.packs_unused > 0 || report.packs_repacked > 0 || report.marked_packs_kept > 0 + report.packs_unused > 0 + || report.packs_repacked > 0 + || report.packs_unindexed > 0 + || report.marked_packs_kept > 0 } /// Gives the packed bytes that the save of the snapshot added to the repository. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index e08862fb16..a066273d19 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1606,6 +1606,61 @@ async fn a_delete_that_frees_nothing_writes_no_ledger() { ); } +#[test] +fn a_prune_that_marks_only_a_pack_that_no_index_lists_leaves_marked_packs() { + assert!(leaves_marked_packs(&PruneReport { + packs_used: 3, + packs_unindexed: 1, + ..PruneReport::default() + })); +} + +#[test] +async fn a_due_prune_that_marks_a_pack_that_no_index_lists_records_the_marked_pack() { + // The pack of the kept snapshot stays in use, so the pack that no index lists is the only pack + // that the prune marks. The ledger holds freed bytes from an earlier delete, so a delete of an + // unknown name prunes. + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-kept"), tree.path(), None) + .await + .unwrap(); + let unindexed = format!("data/ab/{}", "ab".repeat(32)); + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new(&unindexed), + b"a pack that no index lists", + ) + .await + .unwrap(); + set_last_prune(&storage, &scope, golem_common::model::Timestamp::from(0)).await; + + store.delete(&scope, &name("p-unknown")).await.unwrap(); + let after = ledger(&storage, &scope).await; + + assert_eq!( + ( + after.freed_bytes, + after.last_prune.is_some_and(|last| last.to_millis() > 0), + after.awaiting_removal, + blobs(&*storage, &scope.0, "data/") + .await + .contains(&unindexed), + restored_listing(&store, &scope, &name("p-kept")).await.ok(), + ), + (0, true, true, true, Some(listing(tree.path()))) + ); +} + #[test] fn a_prune_leaves_marked_packs_when_it_marks_repacks_or_keeps_marked_packs() { let report = |packs_unused, packs_repacked, marked_packs_kept| PruneReport { From 1e8c0c6d57440365ad18ec783edf7b99966e2bf2 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 03:27:30 -0700 Subject: [PATCH 36/55] Stop the blob calls of an operation at shut down, and let shut down wait for a publish --- .../src/filesystem_snapshot/rustic/files.rs | 78 +++++++----- .../src/filesystem_snapshot/rustic/prune.rs | 1 + .../rustic/publish/tests.rs | 1 + .../filesystem_snapshot/rustic/scope/tests.rs | 1 + .../src/filesystem_snapshot/rustic/store.rs | 27 ++-- .../filesystem_snapshot/rustic/store/tests.rs | 118 ++++++++++++++++++ 6 files changed, 183 insertions(+), 43 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs index 297f5c5959..8d8119ba69 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs @@ -14,36 +14,55 @@ //! The blobs of the repository of one scope, and the blob storage calls of the store on them. //! -//! Each call waits for at most the deadline of the scope. +//! Each call waits for at most the deadline of the scope, and a cancel of its operation ends it. use super::backend::answer_within; +use super::fault::OperationCancelled; use golem_service_base::storage::blob::{ BlobStorage, BlobStorageNamespace, ListedBlob, PutIfAbsent, }; use std::path::Path; use std::sync::Arc; use std::time::Duration; +use tokio_util::sync::CancellationToken; /// The target label of each blob storage call of the rustic store. pub(super) const TARGET_LABEL: &str = "filesystem_snapshot"; -/// The blobs of one scope: the storage, the namespace of the scope, and the deadline of each call. +/// The blobs of one scope: the storage, the namespace of the scope, the deadline of each call, and +/// the token of the operation. #[derive(Clone, Debug)] pub(super) struct SnapshotFiles { pub(super) storage: Arc, pub(super) namespace: BlobStorageNamespace, pub(super) deadline: Duration, + pub(super) cancel: CancellationToken, } impl SnapshotFiles { + /// Waits for one call within the deadline. A call of a cancelled operation does not start, and + /// a cancel ends a running call. Both give an error. + async fn answer( + &self, + future: impl Future>, + ) -> anyhow::Result { + if self.cancel.is_cancelled() { + return Err(anyhow::Error::new(OperationCancelled)); + } + tokio::select! { + biased; + answer = answer_within(self.deadline, future) => answer, + () = self.cancel.cancelled() => Err(anyhow::Error::new(OperationCancelled)), + } + } + /// Gives the content of the blob at the path, or `None` when the path has no blob. pub(super) async fn get( &self, op_label: &'static str, path: &Path, ) -> anyhow::Result>> { - answer_within( - self.deadline, + self.answer( self.storage .get_raw(TARGET_LABEL, op_label, self.namespace.clone(), path), ) @@ -57,16 +76,13 @@ impl SnapshotFiles { path: &Path, content: &[u8], ) -> anyhow::Result<()> { - answer_within( - self.deadline, - self.storage.put_raw( - TARGET_LABEL, - op_label, - self.namespace.clone(), - path, - content, - ), - ) + self.answer(self.storage.put_raw( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + content, + )) .await } @@ -77,23 +93,19 @@ impl SnapshotFiles { path: &Path, content: &[u8], ) -> anyhow::Result { - answer_within( - self.deadline, - self.storage.put_raw_if_absent( - TARGET_LABEL, - op_label, - self.namespace.clone(), - path, - content, - ), - ) + self.answer(self.storage.put_raw_if_absent( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + content, + )) .await } /// Deletes the blob at the path. A path without a blob gives success. pub(super) async fn delete(&self, op_label: &'static str, path: &Path) -> anyhow::Result<()> { - answer_within( - self.deadline, + self.answer( self.storage .delete(TARGET_LABEL, op_label, self.namespace.clone(), path), ) @@ -106,8 +118,7 @@ impl SnapshotFiles { op_label: &'static str, path: &Path, ) -> anyhow::Result { - answer_within( - self.deadline, + self.answer( self.storage .delete_dir(TARGET_LABEL, op_label, self.namespace.clone(), path), ) @@ -120,11 +131,12 @@ impl SnapshotFiles { op_label: &'static str, path: &Path, ) -> anyhow::Result> { - answer_within( - self.deadline, - self.storage - .list_blobs_below(TARGET_LABEL, op_label, self.namespace.clone(), path), - ) + self.answer(self.storage.list_blobs_below( + TARGET_LABEL, + op_label, + self.namespace.clone(), + path, + )) .await } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 8b8cd5f8ce..90ec11fd06 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -188,6 +188,7 @@ mod tests { environment_id: EnvironmentId(Uuid::new_v4()), }, deadline: DEADLINE, + cancel: tokio_util::sync::CancellationToken::new(), } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs index ec5723fe13..7acfc8d002 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -66,6 +66,7 @@ fn files( environment_id: EnvironmentId(Uuid::new_v4()), }, deadline, + cancel: tokio_util::sync::CancellationToken::new(), }, storage, inner, diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs index 229d3380a9..692a5fc642 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope/tests.rs @@ -45,6 +45,7 @@ fn files( storage: storage.clone(), namespace: namespace.clone(), deadline: DEADLINE, + cancel: tokio_util::sync::CancellationToken::new(), } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 85ebbda7df..32e9e3c3d5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -191,9 +191,10 @@ impl RusticSnapshotStore { } } - /// Stops each operation at its next storage call, and waits until no blocking task and no - /// backend of the store remains; later operations give `Storage`. The runtime must not drop - /// before it returns, because a storage call after its time driver stops aborts the process. + /// Cancels each operation, so each running storage call ends and no new call starts, and later + /// operations give `Storage`. A publish is not cancelled. The call waits until no blocking task, + /// backend, publish or delete of a dropped publish remains. The runtime must not drop before it + /// returns, because a storage call after its time driver stops aborts the process. pub(crate) async fn shut_down(&self) { self.root.cancel(); self.tracker.close(); @@ -255,11 +256,13 @@ impl RusticSnapshotStore { .map_err(|error| classify(operation, error)) } - fn files(&self, scope: &SnapshotScope) -> SnapshotFiles { + /// Gives the blobs of the scope for the operation with the token. + fn files(&self, scope: &SnapshotScope, token: &CancellationToken) -> SnapshotFiles { SnapshotFiles { storage: self.storage.clone(), namespace: scope.0.clone(), deadline: self.policy.deadline, + cancel: token.clone(), } } @@ -272,7 +275,7 @@ impl RusticSnapshotStore { token: &CancellationToken, freed: u64, ) -> Result<(), SnapshotStoreError> { - let files = self.files(scope); + let files = self.files(scope, token); let ledger = read_ledger(&files) .await .map_err(storage_failure)? @@ -335,7 +338,11 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { }) .await?; let (staged, info) = staged.ok_or(SnapshotStoreError::AlreadyExists)?; - publish(&self.files(scope), &staged, &self.tracker) + // The publish is the commit point, so no cancel ends it. The tracker counts it, so + // `shut_down` waits for it, and the deadline limits that wait. + let files = self.files(scope, &CancellationToken::new()); + self.tracker + .track_future(publish(&files, &staged, &self.tracker)) .await .map_err(storage_failure)?; Ok(info) @@ -440,8 +447,8 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { } async fn delete_scope(&self, scope: &SnapshotScope) -> Result<(), SnapshotStoreError> { - let _operation = self.start()?; - delete_scope(&self.files(scope)) + let (token, _guard) = self.start()?; + delete_scope(&self.files(scope, &token)) .await .map_err(storage_failure) } @@ -451,8 +458,8 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { from: &SnapshotScope, to: &SnapshotScope, ) -> Result<(), SnapshotStoreError> { - let _operation = self.start()?; - copy_scope(&self.files(from), &self.files(to)) + let (token, _guard) = self.start()?; + copy_scope(&self.files(from, &token), &self.files(to, &token)) .await .map_err(storage_failure) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index a066273d19..ab82f96dee 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -160,6 +160,7 @@ async fn ledger(storage: &Arc, scope: &SnapshotScop storage: storage.clone(), namespace: scope.0.clone(), deadline: Duration::from_secs(2), + cancel: tokio_util::sync::CancellationToken::new(), }) .await .unwrap() @@ -819,6 +820,7 @@ async fn set_last_prune( storage: storage.clone(), namespace: scope.0.clone(), deadline: Duration::from_secs(2), + cancel: tokio_util::sync::CancellationToken::new(), }, &ledger, ) @@ -1136,6 +1138,122 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam ); } +#[test] +async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_call() { + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "copy_list" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let (from, to) = (new_scope(), new_scope()); + let tree = one_file_tree("copied"); + store + .save(&from, &name("p-1"), tree.path(), None) + .await + .unwrap(); + let copying = tokio::spawn({ + let store = store.clone(); + let (from, to) = (from.clone(), to.clone()); + async move { store.copy_scope(&from, &to).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "copy_list") + }) + .await; + + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + let calls_at_stop = storage.calls().len(); + storage.open_gate(); + let copied = tokio::time::timeout(LIMIT, copying).await; + + assert!( + matches!(&copied, Ok(Ok(Err(error))) if is_storage(error, true)), + "{copied:?}" + ); + assert_eq!( + (held, stopped, storage.calls().len()), + (true, true, calls_at_stop) + ); +} + +#[test] +async fn a_publish_held_at_its_storage_call_keeps_shut_down_waiting_until_it_ends() { + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { + if op_label == "publish" { + Script::WaitForGate + } else { + Script::Pass + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ); + let scope = new_scope(); + let tree = one_file_tree("published"); + let saving = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + let path = tree.path().to_path_buf(); + async move { store.save(&scope, &name("p-held"), &path, None).await } + }); + let held = eventually(|| { + storage + .calls() + .iter() + .any(|(op_label, _)| *op_label == "publish") + }) + .await; + + let shutting = store.shut_down(); + tokio::pin!(shutting); + let waited = tokio::time::timeout(Duration::from_millis(200), &mut shutting) + .await + .is_err(); + storage.open_gate(); + let stopped = tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + let saved = tokio::time::timeout(LIMIT, saving).await; + + assert!(matches!(&saved, Ok(Ok(Ok(_)))), "{saved:?}"); + assert_eq!((held, waited, stopped), (true, true, true)); +} + +#[test] +async fn delete_scope_and_copy_scope_after_shut_down_give_storage() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store(storage, policy(LONG_DEADLINE, NEVER, Duration::ZERO)); + let (scope, other) = (new_scope(), new_scope()); + let tree = one_file_tree("kept"); + store + .save(&scope, &name("p-1"), tree.path(), None) + .await + .unwrap(); + store.shut_down().await; + + let deleted = store.delete_scope(&scope).await; + let copied = store.copy_scope(&scope, &other).await; + + assert!( + deleted + .as_ref() + .is_err_and(|error| is_storage(error, false)), + "{deleted:?}" + ); + assert!( + copied.as_ref().is_err_and(|error| is_storage(error, false)), + "{copied:?}" + ); +} + #[test] async fn shut_down_ends_running_operations_before_it_returns() { let storage = From 79fdad4ca79bce526a17b6ee0809fd7a61e500b8 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 03:28:13 -0700 Subject: [PATCH 37/55] Keep one constant for the path of the config file of a repository --- .../src/filesystem_snapshot/rustic/backend.rs | 2 +- golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs | 4 +--- 2 files changed, 2 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index 8e66cd1214..cd994935bc 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -39,7 +39,7 @@ use tokio_util::sync::CancellationToken; use tokio_util::task::task_tracker::TaskTrackerToken; /// The path of the config file of a repository. -const CONFIG_PATH: &str = "config"; +pub(super) const CONFIG_PATH: &str = "config"; /// The largest number of bytes of tree packs that one backend keeps in memory. const KEPT_PACKS_LIMIT: usize = 32 * 1024 * 1024; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs index 996e605de2..ba992a8c6b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/scope.rs @@ -17,15 +17,13 @@ //! These operations do not read the repository format. They only know the directories of the //! repository, its config file, and the ledger directory of the store. +use super::backend::CONFIG_PATH; use super::files::SnapshotFiles; use super::prune::LEDGER_PATH; use futures::{StreamExt, TryStreamExt, stream}; use golem_service_base::storage::blob::PutIfAbsent; use std::path::Path; -/// The path of the config file of a repository. -const CONFIG_PATH: &str = "config"; - /// The directories of a repository in the order of a listing. A save writes them in the reverse /// order, and so does a copy, so a snapshot file always has its data. const LISTING_ORDER: [&str; 4] = ["snapshots", "index", "keys", "data"]; From ad5036612561cfd3b3749f07fdbaa989762e7c30 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 03:36:27 -0700 Subject: [PATCH 38/55] Poll the shut down of the publish test again only when its first wait timed out --- .../src/filesystem_snapshot/rustic/store/tests.rs | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index ab82f96dee..0af6fdef00 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1220,7 +1220,9 @@ async fn a_publish_held_at_its_storage_call_keeps_shut_down_waiting_until_it_end .await .is_err(); storage.open_gate(); - let stopped = tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); + // A finished future must not be polled again, so the second wait runs only after a first wait + // that timed out. + let stopped = !waited || tokio::time::timeout(LIMIT, &mut shutting).await.is_ok(); let saved = tokio::time::timeout(LIMIT, saving).await; assert!(matches!(&saved, Ok(Ok(Ok(_)))), "{saved:?}"); From 66c9ee7b3309f03ecb79efa0660d2074a5d06507 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 03:53:07 -0700 Subject: [PATCH 39/55] Publish nothing when a save reaches its publish after shut down --- .../src/filesystem_snapshot/rustic/store.rs | 56 ++++++++++++++----- .../filesystem_snapshot/rustic/store/tests.rs | 49 +++++++++++++++- 2 files changed, 90 insertions(+), 15 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 32e9e3c3d5..5b2dea6451 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -154,10 +154,31 @@ pub(crate) struct RusticSnapshotStore { policy: StorePolicy, /// The parent of the token of each operation. root: CancellationToken, - /// Counts the blocking tasks, the backends and the deletes of dropped publishes. + /// Counts the blocking tasks, the backends, the publishes and the deletes of dropped publishes. tracker: TaskTracker, /// Runs saves and prunes at a low priority. low_priority: LowPriority, + /// Holds a save after its blocking work and before its publish, when a test sets it. + #[cfg(test)] + pub(super) publish_gate: Option>, +} + +/// A gate that holds a save after its blocking work and before its publish. +#[cfg(test)] +#[derive(Debug, Default)] +pub(super) struct PublishGate { + /// Notified when a save reaches the gate. + pub(super) reached: tokio::sync::Notify, + /// Lets the save go on. + pub(super) open: tokio::sync::Notify, +} + +/// The error of an operation of a store that is shut down. +fn shut_down_error() -> SnapshotStoreError { + SnapshotStoreError::Storage { + retryable: false, + source: anyhow::anyhow!("the filesystem snapshot store is shut down"), + } } impl RusticSnapshotStore { @@ -188,12 +209,15 @@ impl RusticSnapshotStore { root: CancellationToken::new(), tracker: TaskTracker::new(), low_priority: LowPriority::new(policy.save_threads), + #[cfg(test)] + publish_gate: None, } } /// Cancels each operation, so each running storage call ends and no new call starts, and later - /// operations give `Storage`. A publish is not cancelled. The call waits until no blocking task, - /// backend, publish or delete of a dropped publish remains. The runtime must not drop before it + /// operations give `Storage`. A publish that starts before the cancel runs to its end. A save + /// that reaches its publish after the cancel publishes nothing and gives `Storage`. The call + /// waits until no blocking task, backend, publish or delete of a dropped publish remains. The runtime must not drop before it /// returns, because a storage call after its time driver stops aborts the process. pub(crate) async fn shut_down(&self) { self.root.cancel(); @@ -210,10 +234,7 @@ impl RusticSnapshotStore { /// Starts an operation. The token of the operation is cancelled when the guard drops. fn start(&self) -> Result<(CancellationToken, DropGuard), SnapshotStoreError> { if self.root.is_cancelled() { - return Err(SnapshotStoreError::Storage { - retryable: false, - source: anyhow::anyhow!("the filesystem snapshot store is shut down"), - }); + return Err(shut_down_error()); } let token = self.root.child_token(); Ok((token.clone(), token.drop_guard())) @@ -338,13 +359,22 @@ impl FilesystemSnapshotStore for RusticSnapshotStore { }) .await?; let (staged, info) = staged.ok_or(SnapshotStoreError::AlreadyExists)?; - // The publish is the commit point, so no cancel ends it. The tracker counts it, so - // `shut_down` waits for it, and the deadline limits that wait. + #[cfg(test)] + if let Some(gate) = &self.publish_gate { + gate.reached.notify_one(); + gate.open.notified().await; + } + // The publish is the commit point, so no cancel ends it. The tracker counts it from here, + // so a `shut_down` that has not cancelled yet waits for it, and the deadline limits that wait. let files = self.files(scope, &CancellationToken::new()); - self.tracker - .track_future(publish(&files, &staged, &self.tracker)) - .await - .map_err(storage_failure)?; + let publishing = self + .tracker + .track_future(publish(&files, &staged, &self.tracker)); + if self.root.is_cancelled() { + // No snapshot file is written. A later prune marks the packs of the save. + return Err(shut_down_error()); + } + publishing.await.map_err(storage_failure)?; Ok(info) } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 0af6fdef00..07ea2c6510 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -23,8 +23,8 @@ use super::super::scripted::{Script, ScriptedBlobStorage}; use super::super::tests::{copy_flat_tree, entries, three_file_tree, wait_past_change_times}; use super::super::{PruneReport, PruneSettings, RepackLimits, RepositoryKey, open_existing}; use super::{ - RusticSnapshotStore, StorePolicy, leaves_marked_packs, scope_snapshots, store_backup_options, - store_restore_options, whole_millis_from, + PublishGate, RusticSnapshotStore, StorePolicy, leaves_marked_packs, scope_snapshots, + store_backup_options, store_restore_options, whole_millis_from, }; use crate::filesystem_snapshot::contract_tests::fixture::{ Listed, Scratch, Spec, fixture, listing, write_tree, @@ -1229,6 +1229,51 @@ async fn a_publish_held_at_its_storage_call_keeps_shut_down_waiting_until_it_end assert_eq!((held, waited, stopped), (true, true, true)); } +#[test] +async fn a_save_that_reaches_its_publish_after_shut_down_publishes_nothing() { + // The gate holds the save after its blocking work, so the tracker is empty and `shut_down` + // returns before the publish starts. + let storage = Arc::new(InMemoryBlobStorage::new()); + let gate = Arc::new(PublishGate::default()); + let store = Arc::new(RusticSnapshotStore { + publish_gate: Some(gate.clone()), + ..RusticSnapshotStore::with_policy( + storage.clone(), + key(), + policy(LONG_DEADLINE, NEVER, Duration::ZERO), + ) + }); + let scope = new_scope(); + let tree = one_file_tree("late"); + let saving = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + let path = tree.path().to_path_buf(); + async move { store.save(&scope, &name("p-late"), &path, None).await } + }); + let reached = tokio::time::timeout(LIMIT, gate.reached.notified()) + .await + .is_ok(); + + let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); + gate.open.notify_one(); + let saved = tokio::time::timeout(LIMIT, saving).await; + + assert!( + matches!(&saved, Ok(Ok(Err(error))) if is_storage(error, false)), + "{saved:?}" + ); + assert_eq!( + ( + reached, + stopped, + blobs(&*storage, &scope.0, "snapshots/").await, + store.work_in_flight(), + ), + (true, true, Vec::::new(), 0) + ); +} + #[test] async fn delete_scope_and_copy_scope_after_shut_down_give_storage() { let storage = Arc::new(InMemoryBlobStorage::new()); From 2a11af91f977341d3b4ebf28c3061a95b5b8df4c Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 03:55:16 -0700 Subject: [PATCH 40/55] Wrap the doc of the shut down of the store --- .../src/filesystem_snapshot/rustic/store.rs | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 5b2dea6451..697ba735d7 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -217,8 +217,9 @@ impl RusticSnapshotStore { /// Cancels each operation, so each running storage call ends and no new call starts, and later /// operations give `Storage`. A publish that starts before the cancel runs to its end. A save /// that reaches its publish after the cancel publishes nothing and gives `Storage`. The call - /// waits until no blocking task, backend, publish or delete of a dropped publish remains. The runtime must not drop before it - /// returns, because a storage call after its time driver stops aborts the process. + /// waits until no blocking task, backend, publish or delete of a dropped publish remains. The + /// runtime must not drop before it returns, because a storage call after its time driver stops + /// aborts the process. pub(crate) async fn shut_down(&self) { self.root.cancel(); self.tracker.close(); From a5a20474402f09aaf6b465bc06c38243d4c9f6c1 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 04:57:25 -0700 Subject: [PATCH 41/55] Check that a cancel ends a running blob call of a copy --- .../src/filesystem_snapshot/rustic/store/tests.rs | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 07ea2c6510..49b4d8b119 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1157,7 +1157,7 @@ async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_cal .save(&from, &name("p-1"), tree.path(), None) .await .unwrap(); - let copying = tokio::spawn({ + let mut copying = tokio::spawn({ let store = store.clone(); let (from, to) = (from.clone(), to.clone()); async move { store.copy_scope(&from, &to).await } @@ -1170,17 +1170,19 @@ async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_cal }) .await; + // The gate stays closed until the end, so only the cancel can end the held call. let stopped = tokio::time::timeout(LIMIT, store.shut_down()).await.is_ok(); let calls_at_stop = storage.calls().len(); + let copied = tokio::time::timeout(LIMIT, &mut copying).await; + let calls_after_copy = storage.calls().len(); storage.open_gate(); - let copied = tokio::time::timeout(LIMIT, copying).await; assert!( matches!(&copied, Ok(Ok(Err(error))) if is_storage(error, true)), "{copied:?}" ); assert_eq!( - (held, stopped, storage.calls().len()), + (held, stopped, calls_after_copy), (true, true, calls_at_stop) ); } From 754f1f9495f338cc32b24514d120e30e35982bfb Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 05:04:57 -0700 Subject: [PATCH 42/55] Check that a blob call of a cancelled operation does not start --- .../filesystem_snapshot/rustic/store/tests.rs | 23 +++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 49b4d8b119..b935b01f98 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1138,6 +1138,29 @@ async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_nam ); } +#[test] +async fn a_blob_call_of_a_cancelled_operation_does_not_start() { + // The in-memory storage answers at the first poll, so only the check before the call keeps + // the call from the storage. + let storage = + ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |_, _| Script::Pass); + let cancel = tokio_util::sync::CancellationToken::new(); + cancel.cancel(); + let files = SnapshotFiles { + storage: storage.clone(), + namespace: new_scope().0, + deadline: LONG_DEADLINE, + cancel, + }; + + let read = files + .get("read_ledger", Path::new("golem/prune-ledger")) + .await; + + assert!(read.is_err(), "{read:?}"); + assert_eq!(storage.calls(), Vec::new()); +} + #[test] async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_call() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { From f8e94e6bd628eadc308942768e75e894b473e8e5 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 10:31:06 -0700 Subject: [PATCH 43/55] Give each waiting test of the rustic store a time limit --- .../rustic/backend/tests.rs | 27 ++++++++++++------- .../rustic/publish/tests.rs | 4 ++- .../filesystem_snapshot/rustic/store/tests.rs | 12 ++++++++- .../src/filesystem_snapshot/rustic/tests.rs | 7 ++++- 4 files changed, 37 insertions(+), 13 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs index 994c7b88ea..18402a6557 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend/tests.rs @@ -124,9 +124,17 @@ fn bytes(text: &str) -> BytesList { /// Runs the calls on a new thread, which is not a thread of a runtime, and gives their result. /// `None` means that the calls did not end within the limit. fn within_limit(calls: impl FnOnce() -> T + Send + 'static) -> Option { + on_own_thread(calls).recv_timeout(LIMIT).ok() +} + +/// Starts the calls on a new thread, which is not a thread of a runtime. The receiver gets their +/// result, so a test can wait for it with a limit. +fn on_own_thread( + calls: impl FnOnce() -> T + Send + 'static, +) -> std::sync::mpsc::Receiver { let (sender, receiver) = std::sync::mpsc::channel(); std::thread::spawn(move || sender.send(calls())); - receiver.recv_timeout(LIMIT).ok() + receiver } #[test] @@ -464,18 +472,17 @@ fn a_thread_that_is_not_a_thread_of_the_runtime_can_call_the_backend() { let fixture = Fixture::new(); let backend = Arc::new(fixture.backend); - let read = std::thread::spawn({ + let read = within_limit({ let backend = backend.clone(); move || { backend .write_bytes(FileType::Index, &id("ab"), false, bytes("index")) .and_then(|()| backend.read_full(FileType::Index, &id("ab"))) + .ok() } - }) - .join() - .map(|read| read.ok()); + }); - assert_eq!(read.ok().flatten(), Some(Bytes::from_static(b"index"))); + assert_eq!(read.flatten(), Some(Bytes::from_static(b"index"))); } /// The content of the pack of the tests of the kept packs: 100 bytes, each its own offset. @@ -598,7 +605,7 @@ fn two_threads_that_miss_one_pack_make_one_storage_read() { Script::Pass } }); - let first = std::thread::spawn({ + let first = on_own_thread({ let backend = fixture.backend.clone(); move || tree_range(&backend, 0, 10).ok() }); @@ -606,7 +613,7 @@ fn two_threads_that_miss_one_pack_make_one_storage_read() { std::thread::sleep(Duration::from_millis(10)); !fixture.pack_calls().is_empty() }); - let second = std::thread::spawn({ + let second = on_own_thread({ let backend = fixture.backend.clone(); move || tree_range(&backend, 50, 10).ok() }); @@ -616,8 +623,8 @@ fn two_threads_that_miss_one_pack_make_one_storage_read() { assert_eq!( ( first_read_started, - first.join().ok().flatten(), - second.join().ok().flatten(), + first.recv_timeout(LIMIT).ok().flatten(), + second.recv_timeout(LIMIT).ok().flatten(), fixture.pack_calls() ), ( diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs index 7acfc8d002..b6aeac819b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/publish/tests.rs @@ -25,7 +25,7 @@ use pretty_assertions::assert_eq; use std::path::Path; use std::sync::Arc; use std::time::Duration; -use test_r::test; +use test_r::{test, timeout}; use tokio_util::task::TaskTracker; use uuid::Uuid; @@ -137,6 +137,7 @@ async fn a_publish_whose_answer_is_lost_deletes_the_file_and_gives_the_error() { } #[test] +#[timeout("60s")] async fn a_publish_that_reaches_the_deadline_deletes_the_file_that_the_storage_wrote() { let (files, _, inner) = files( Script::NeverAnswer, @@ -158,6 +159,7 @@ async fn a_publish_that_reaches_the_deadline_deletes_the_file_that_the_storage_w } #[test] +#[timeout("60s")] async fn a_publish_that_the_caller_drops_deletes_the_file_in_a_task_of_the_tracker() { // The delete waits for the gate, so the test reads the file that the dropped write left // before the task of the tracker deletes it. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index b935b01f98..fa7df64f6e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -47,7 +47,7 @@ use std::sync::Arc; use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; use std::time::Duration; use test_r::core::DynamicTestRegistration; -use test_r::{test, test_gen}; +use test_r::{test, test_gen, timeout}; /// The longest time that a test waits for an operation or for the work of a store to end. const LIMIT: Duration = Duration::from_secs(10); @@ -365,6 +365,7 @@ async fn a_save_whose_index_write_fails_publishes_nothing_and_leaves_the_name_fr } #[test] +#[timeout("60s")] async fn a_publish_that_reaches_the_deadline_and_lands_late_publishes_nothing() { let hang = Arc::new(AtomicBool::new(true)); let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { @@ -452,6 +453,7 @@ async fn a_second_save_of_an_unchanged_tree_writes_no_pack() { } #[test] +#[timeout("60s")] async fn two_stores_that_create_one_repository_at_the_same_time_both_save() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { @@ -514,6 +516,7 @@ async fn two_stores_that_create_one_repository_at_the_same_time_both_save() { } #[test] +#[timeout("60s")] async fn a_prune_during_a_save_keeps_the_packs_of_the_save() { // The first index write after the arm waits at the gate. That is the index write of the // second save, so its packs are in no index while the delete prunes. @@ -1059,6 +1062,7 @@ async fn a_deleted_scope_holds_no_blob() { } #[test] +#[timeout("60s")] async fn a_save_dropped_at_any_storage_call_publishes_nothing_and_leaves_the_name_free() { // The first save counts the calls of a save. Each later round holds one of these calls: the // call reaches the storage and never answers, as a write that S3 received and completes after @@ -1162,6 +1166,7 @@ async fn a_blob_call_of_a_cancelled_operation_does_not_start() { } #[test] +#[timeout("60s")] async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_call() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { if op_label == "copy_list" { @@ -1211,6 +1216,7 @@ async fn a_copy_held_at_a_storage_call_stops_at_shut_down_and_makes_no_later_cal } #[test] +#[timeout("60s")] async fn a_publish_held_at_its_storage_call_keeps_shut_down_waiting_until_it_ends() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, _| { if op_label == "publish" { @@ -1255,6 +1261,7 @@ async fn a_publish_held_at_its_storage_call_keeps_shut_down_waiting_until_it_end } #[test] +#[timeout("60s")] async fn a_save_that_reaches_its_publish_after_shut_down_publishes_nothing() { // The gate holds the save after its blocking work, so the tracker is empty and `shut_down` // returns before the publish starts. @@ -1327,6 +1334,7 @@ async fn delete_scope_and_copy_scope_after_shut_down_give_storage() { } #[test] +#[timeout("60s")] async fn shut_down_ends_running_operations_before_it_returns() { let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), |op_label, path| { @@ -1382,6 +1390,7 @@ enum Dropped { } #[test] +#[timeout("60s")] async fn a_dropped_operation_stops_its_blocking_work() { let outcomes = futures::stream::iter([ Dropped::Restore, @@ -2191,6 +2200,7 @@ async fn the_storage_calls_of_a_restore_run_at_the_nice_value_of_the_process() { #[cfg(target_os = "linux")] #[test] +#[timeout("60s")] async fn after_saves_and_prunes_the_pools_keep_the_nice_value_of_the_process() { // All tasks wait for each other, so each runs on its own thread of the blocking pool, and the // idle threads that ran the saves and the prune are among them. diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index d10f6697e4..f367be34f8 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -49,7 +49,7 @@ use std::path::{Path, PathBuf}; use std::sync::Arc; use std::sync::atomic::{AtomicUsize, Ordering}; use std::time::Duration; -use test_r::test; +use test_r::{test, timeout}; use tokio::runtime::Handle; use tokio::sync::{Notify, oneshot, watch}; use tokio::time::error::Elapsed; @@ -477,6 +477,7 @@ async fn each_scope_is_its_own_repository() { } #[test] +#[timeout("60s")] async fn a_restore_reads_data_on_at_most_its_reader_threads() { // Each save adds one pack with the data of its new file. The restore of the second snapshot // reads the data of both packs, one read for each pack. @@ -749,6 +750,7 @@ impl BlobStorage for OverlapCountingStorage { } #[test] +#[timeout("60s")] async fn a_save_whose_pack_write_gets_no_answer_fails_with_no_snapshot_and_its_threads_stop() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -774,6 +776,7 @@ async fn a_save_whose_pack_write_gets_no_answer_fails_with_no_snapshot_and_its_t } #[test] +#[timeout("60s")] async fn a_save_whose_pack_writes_answer_before_the_deadline_succeeds_and_restores() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -815,6 +818,7 @@ async fn a_save_whose_pack_writes_answer_before_the_deadline_succeeds_and_restor } #[test] +#[timeout("60s")] async fn a_restore_whose_data_pack_reads_get_no_answer_fails_and_stops_its_threads() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); @@ -872,6 +876,7 @@ async fn a_restore_whose_data_pack_reads_get_no_answer_fails_and_stops_its_threa } #[test] +#[timeout("60s")] async fn a_prune_whose_tree_pack_reads_get_no_answer_fails_and_stops_its_threads() { let inner = Arc::new(InMemoryBlobStorage::new()); let scope = new_scope(); From c6214409d15bf4f5b5133b8e2138a28f187bf377 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 10:33:47 -0700 Subject: [PATCH 44/55] Wait for a blob call of the store and of the backend in one function --- .../src/filesystem_snapshot/rustic/backend.rs | 33 +++++++++++-------- .../src/filesystem_snapshot/rustic/files.rs | 12 ++----- 2 files changed, 21 insertions(+), 24 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs index cd994935bc..406bd18f91 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/backend.rs @@ -163,21 +163,8 @@ impl BlobBackend { path: &Path, future: impl Future>, ) -> RusticResult { - if self.cancel.is_cancelled() { - return Err(storage_error( - call, - path, - anyhow::Error::new(OperationCancelled), - )); - } self.runtime - .block_on(async { - tokio::select! { - biased; - answer = answer_within(self.deadline, future) => answer, - () = self.cancel.cancelled() => Err(anyhow::Error::new(OperationCancelled)), - } - }) + .block_on(answer_or_cancel(self.deadline, &self.cancel, future)) .map_err(|error| storage_error(call, path, error)) } @@ -216,6 +203,24 @@ pub(super) async fn answer_within( }) } +/// Gives the output of the future within the deadline, or an error when the operation of the token +/// is cancelled. A call of a cancelled operation does not start, and a cancel ends a call that +/// runs. +pub(super) async fn answer_or_cancel( + deadline: Duration, + cancel: &CancellationToken, + future: impl Future>, +) -> anyhow::Result { + if cancel.is_cancelled() { + return Err(anyhow::Error::new(OperationCancelled)); + } + tokio::select! { + biased; + answer = answer_within(deadline, future) => answer, + () = cancel.cancelled() => Err(anyhow::Error::new(OperationCancelled)), + } +} + impl ReadBackend for BlobBackend { fn location(&self) -> String { format!("golem-blob-storage:{:?}", self.namespace) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs index 8d8119ba69..e60817137c 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/files.rs @@ -16,8 +16,7 @@ //! //! Each call waits for at most the deadline of the scope, and a cancel of its operation ends it. -use super::backend::answer_within; -use super::fault::OperationCancelled; +use super::backend::answer_or_cancel; use golem_service_base::storage::blob::{ BlobStorage, BlobStorageNamespace, ListedBlob, PutIfAbsent, }; @@ -46,14 +45,7 @@ impl SnapshotFiles { &self, future: impl Future>, ) -> anyhow::Result { - if self.cancel.is_cancelled() { - return Err(anyhow::Error::new(OperationCancelled)); - } - tokio::select! { - biased; - answer = answer_within(self.deadline, future) => answer, - () = self.cancel.cancelled() => Err(anyhow::Error::new(OperationCancelled)), - } + answer_or_cancel(self.deadline, &self.cancel, future).await } /// Gives the content of the blob at the path, or `None` when the path has no blob. From c3347804d5f08209469de107c41d38bd6f786324 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 10:48:58 -0700 Subject: [PATCH 45/55] Take a claim before a prune, so concurrent deletes make one prune --- .../src/filesystem_snapshot/rustic/prune.rs | 179 ++++++++++++++++- .../src/filesystem_snapshot/rustic/store.rs | 35 +++- .../filesystem_snapshot/rustic/store/tests.rs | 186 +++++++++++++++++- 3 files changed, 383 insertions(+), 17 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 90ec11fd06..f84b25977e 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -18,17 +18,26 @@ //! packed bytes that deleted snapshots added since the last prune, the time of the last prune, and //! whether that prune marked packs that a later prune removes. Two deletes at the same time can //! lose a count. A lost count only delays a prune. +//! +//! A delete whose prune is due takes a claim before it prunes, so two deletes that read the same +//! ledger make one prune. The claims of a ledger are in one directory, named by the time of the last +//! prune in that ledger. use super::files::SnapshotFiles; +use futures::{StreamExt, TryStreamExt, stream}; use golem_common::model::Timestamp; +use golem_service_base::storage::blob::PutIfAbsent; use serde::{Deserialize, Serialize}; -use std::path::Path; +use std::path::{Path, PathBuf}; use std::time::Duration; use tracing::warn; /// The path of the ledger blob, relative to the root of the namespace of the scope. pub(super) const LEDGER_PATH: &str = "golem/prune-ledger"; +/// The directory of the prune claims, relative to the root of the namespace of the scope. +const CLAIMS_PATH: &str = "golem/prune-claims"; + /// What the scope did since its last prune. #[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] pub(super) struct PruneLedger { @@ -75,12 +84,17 @@ impl Percent { /// Tells whether the grace period passed at `now` since the last prune. fn grace_passed(ledger: &PruneLedger, now: Timestamp, grace: Duration) -> bool { - ledger.last_prune.is_none_or(|last| { - now.to_millis() - >= last - .to_millis() - .saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) - }) + ledger + .last_prune + .is_none_or(|last| passed_since(last, now, grace)) +} + +/// Tells whether the grace period passed at `now` since the time. +fn passed_since(time: Timestamp, now: Timestamp, grace: Duration) -> bool { + now.to_millis() + >= time + .to_millis() + .saturating_add(u64::try_from(grace.as_millis()).unwrap_or(u64::MAX)) } /// Tells whether [`prune_due`] needs the size of the repository at `now`. Only freed bytes after @@ -142,12 +156,132 @@ pub(super) async fn write_ledger( .await } +/// A claim of a prune that a listing found: its number, and its time when its content parses. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) struct ListedClaim { + pub(super) number: u64, + pub(super) claimed_at: Option, +} + +/// What a delete whose prune is due does with the claims of its ledger. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) enum ClaimChoice { + /// Take the claim with the number, and prune when the write of the claim succeeds. + Claim(u64), + /// Another prune holds the claims of the ledger, so do not prune. + Held, +} + +/// Chooses the claim of a delete from the claims of its ledger. The newest claim holds the ledger +/// until the grace period passed since its time. A claim whose content does not parse is old. +pub(super) fn next_claim(claims: &[ListedClaim], now: Timestamp, grace: Duration) -> ClaimChoice { + match claims.iter().max_by_key(|claim| claim.number) { + None => ClaimChoice::Claim(0), + Some(newest) + if newest + .claimed_at + .is_some_and(|at| !passed_since(at, now, grace)) => + { + ClaimChoice::Held + } + Some(newest) => ClaimChoice::Claim(newest.number.saturating_add(1)), + } +} + +/// Gives the directory of the claims of the ledger: the time of its last prune in milliseconds, or +/// `none`. +pub(super) fn claims_directory(ledger: &PruneLedger) -> PathBuf { + let generation = ledger + .last_prune + .map_or_else(|| "none".to_string(), |last| last.to_millis().to_string()); + Path::new(CLAIMS_PATH).join(generation) +} + +/// Lists the claims in the directory. A name that is not a number is not a claim, and a claim +/// that a delete removed after the listing counts as old. +pub(super) async fn list_claims( + files: &SnapshotFiles, + directory: &Path, +) -> anyhow::Result> { + let listed = files.list_below("list_claims", directory).await?; + let numbered = listed + .iter() + .filter_map(|blob| { + let number = blob.path.file_name()?.to_str()?.parse::().ok()?; + Some((number, blob.path.clone())) + }) + .collect::>(); + stream::iter(numbered) + .then(|(number, path)| async move { + let content = files.get("read_claim", &path).await?; + Ok::<_, anyhow::Error>(ListedClaim { + number, + claimed_at: content.as_deref().and_then(parse_claim), + }) + }) + .try_collect() + .await +} + +/// Writes the claim with the number, and tells whether this call wrote it. +pub(super) async fn take_claim( + files: &SnapshotFiles, + directory: &Path, + number: u64, + now: Timestamp, +) -> anyhow::Result { + let content = now.to_millis().to_string(); + let written = files + .put_if_absent( + "write_claim", + &directory.join(number.to_string()), + content.as_bytes(), + ) + .await?; + Ok(written == PutIfAbsent::Written) +} + +/// Deletes the claim with the number. A failure gives a warning, because a claim only delays a +/// prune until its grace period passed. +pub(super) async fn release_claim(files: &SnapshotFiles, directory: &Path, number: u64) { + if let Err(error) = files + .delete("delete_claim", &directory.join(number.to_string())) + .await + { + warn!( + error = %format!("{error:#}"), + "Failed to delete the prune claim of a filesystem snapshot scope" + ); + } +} + +/// Deletes the claims of a ledger after its prune. A failure gives a warning, because a claim only +/// delays a prune until its grace period passed. +pub(super) async fn end_claims(files: &SnapshotFiles, directory: &Path) { + if let Err(error) = files.delete_dir("delete_claims", directory).await { + warn!( + error = %format!("{error:#}"), + "Failed to delete the prune claims of a filesystem snapshot scope" + ); + } +} + +/// Reads the time of a claim, in milliseconds. +fn parse_claim(content: &[u8]) -> Option { + std::str::from_utf8(content) + .ok()? + .trim() + .parse::() + .ok() + .map(Timestamp::from) +} + #[cfg(test)] mod tests { use super::super::files::SnapshotFiles; use super::{ - LEDGER_PATH, Percent, PruneLedger, needs_repository_size, prune_due, read_ledger, - write_ledger, + ClaimChoice, LEDGER_PATH, ListedClaim, Percent, PruneLedger, needs_repository_size, + next_claim, prune_due, read_ledger, write_ledger, }; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; @@ -291,6 +425,33 @@ mod tests { ); } + #[test] + fn the_newest_claim_holds_a_ledger_until_the_grace_period_passed_since_its_time() { + let now = 10_000_000; + let claim = |number, claimed_at: Option| ListedClaim { + number, + claimed_at: claimed_at.map(Timestamp::from), + }; + let choose = |claims: &[ListedClaim]| next_claim(claims, at(now), GRACE); + + assert_eq!( + [ + choose(&[]), + choose(&[claim(0, Some(now - GRACE_MILLIS)), claim(1, Some(now - 1))]), + choose(&[claim(1, Some(now)), claim(2, Some(now - GRACE_MILLIS))]), + choose(&[claim(4, None)]), + choose(&[claim(0, Some(now - 1)), claim(3, None)]), + ], + [ + ClaimChoice::Claim(0), + ClaimChoice::Held, + ClaimChoice::Claim(3), + ClaimChoice::Claim(5), + ClaimChoice::Claim(4), + ] + ); + } + #[test] fn a_delete_adds_its_bytes_and_a_prune_starts_the_ledger_again() { let deleted = ledger(5, Some(7), true) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index 697ba735d7..e8637c7802 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -25,8 +25,9 @@ use super::fault::{Operation, classify, is_file_missing, is_storage_failure, sto use super::files::SnapshotFiles; use super::priority::LowPriority; use super::prune::{ - Percent, PruneLedger, needs_repository_size, prune_due, read_ledger, repository_bytes, - write_ledger, + ClaimChoice, Percent, PruneLedger, claims_directory, end_claims, list_claims, + needs_repository_size, next_claim, prune_due, read_ledger, release_claim, repository_bytes, + take_claim, write_ledger, }; use super::publish::{SnapshotStage, StagedSnapshot, publish}; use super::scope::{copy_scope, delete_scope}; @@ -291,6 +292,8 @@ impl RusticSnapshotStore { /// Adds the freed bytes to the ledger of the scope, and prunes the repository when a prune is /// due. The ledger keeps the freed bytes before the prune starts, so a delete that runs again /// after a failed prune prunes again. It lists the packs only when their size can make a prune due. + /// A due prune runs only after the delete takes a claim of its ledger. A failed prune deletes + /// the claim, and a prune that succeeds deletes each claim of its ledger. async fn prune_when_due( &self, scope: &SnapshotScope, @@ -317,19 +320,41 @@ impl RusticSnapshotStore { if !prune_due(&ledger, now, size, self.policy.prune_threshold, grace) { return Ok(()); } + let claims = claims_directory(&ledger); + let listed = list_claims(&files, &claims) + .await + .map_err(storage_failure)?; + let ClaimChoice::Claim(number) = next_claim(&listed, now, grace) else { + return Ok(()); + }; + if !take_claim(&files, &claims, number, now) + .await + .map_err(storage_failure)? + { + return Ok(()); + } let backend = Arc::new(self.backend(scope, token)?); let key = self.key.clone(); let settings = self.policy.prune; let low_priority = self.low_priority; - let report = self + let pruned = self .blocking(Operation::Prune, move || { low_priority.run("fs-snap-prune", move || prune(backend, &key, &settings)) }) - .await?; + .await; + let report = match pruned { + Ok(report) => report, + Err(error) => { + release_claim(&files, &claims, number).await; + return Err(error); + } + }; let marked_packs = report.as_ref().is_some_and(leaves_marked_packs); write_ledger(&files, &PruneLedger::after_prune(now, marked_packs)) .await - .map_err(storage_failure) + .map_err(storage_failure)?; + end_claims(&files, &claims).await; + Ok(()) } } diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index fa7df64f6e..83ebd818d0 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -907,6 +907,177 @@ async fn a_failed_listing_of_the_packs_gives_storage_and_records_no_prune() { assert_eq!((after.freed_bytes > 0, after.last_prune), (true, None)); } +/// Counts the prunes among the recorded calls. A prune lists the packs, and no other step of a +/// delete makes that call. +fn prunes(calls: &[(&'static str, String)]) -> usize { + calls + .iter() + .filter(|(op_label, path)| *op_label == "list" && path == "data") + .count() +} + +/// Saves a tree of one file with the content under each name. +async fn save_each(store: &RusticSnapshotStore, scope: &SnapshotScope, names: &[&str]) { + futures::stream::iter(names) + .for_each(|text| async move { + let tree = one_file_tree(text); + store + .save(scope, &name(text), tree.path(), None) + .await + .unwrap(); + }) + .await; +} + +#[test] +#[timeout("60s")] +async fn two_deletes_that_read_the_same_ledger_make_one_prune() { + // The first read after the first claim is the start of the first prune. The gate holds it, + // so the second delete reads the ledger that the first delete read. + let claimed = Arc::new(AtomicBool::new(false)); + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, held) = (claimed.clone(), held.clone()); + move |op_label, _| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "read" + && claimed.load(Ordering::SeqCst) + && !held.swap(true, Ordering::SeqCst) + { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let first = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let first_held = eventually(|| held.load(Ordering::SeqCst)).await; + + let second = store.delete(&scope, &name("p-2")).await; + storage.open_gate(); + let first = tokio::time::timeout(LIMIT, first).await; + let calls = storage.calls(); + + assert!(matches!(first, Ok(Ok(Ok(())))), "{first:?}"); + assert!(second.is_ok(), "{second:?}"); + assert_eq!( + ( + first_held, + prunes(&calls), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, 1, Vec::::new()) + ); +} + +#[test] +#[timeout("60s")] +async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() { + // The first call after the first claim is the start of the first prune, and it fails before + // the prune lists the packs. So only the retry counts as a prune. + let claimed = Arc::new(AtomicBool::new(false)); + let refused = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, refused) = (claimed.clone(), refused.clone()); + move |op_label, _| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + Script::Pass + } else if claimed.load(Ordering::SeqCst) && !refused.swap(true, Ordering::SeqCst) { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + let retried = store.delete(&scope, &name("p-1")).await; + let after = ledger(&storage, &scope).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!(retried.is_ok(), "{retried:?}"); + assert_eq!( + ( + claims_after_failure, + after.last_prune.is_some(), + prunes(&storage.calls()) + ), + (Vec::::new(), true, 1) + ); +} + +#[test] +#[timeout("60s")] +async fn a_claim_older_than_the_grace_period_does_not_block_a_prune() { + let grace = Duration::from_secs(3600); + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store(storage.clone(), policy(LONG_DEADLINE, ALWAYS, grace)); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let old = golem_common::model::Timestamp::now_utc() + .to_millis() + .saturating_sub(2 * 3_600_000); + storage + .put_raw( + "test", + "test", + scope.0.clone(), + Path::new("golem/prune-claims/none/0"), + old.to_string().as_bytes(), + ) + .await + .unwrap(); + + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert!(ledger(&storage, &scope).await.last_prune.is_some()); +} + +#[test] +#[timeout("60s")] +async fn a_prune_that_succeeds_deletes_the_claims_of_its_ledger() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::ZERO), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + store.delete(&scope, &name("p-1")).await.unwrap(); + + assert_eq!( + ( + ledger(&storage, &scope).await.last_prune.is_some(), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, Vec::::new()) + ); +} + #[test] async fn a_delete_below_the_threshold_does_not_prune() { let storage = Arc::new(InMemoryBlobStorage::new()); @@ -2133,8 +2304,8 @@ async fn the_storage_calls_of_a_save_run_at_nice_19() { #[cfg(target_os = "linux")] #[test] async fn the_storage_calls_of_a_prune_run_at_nice_19() { - // The forget of a delete runs before the ledger read at the normal priority. The ledger calls - // and the listing of the packs run on the async runtime, and the prune runs after them. + // The forget of a delete runs before the ledger read at the normal priority. The ledger calls, + // the listing of the packs and the claim calls run on the async runtime. let (storage, calls) = nice_recording_storage(); let store = store(storage, policy(LONG_DEADLINE, ALWAYS, Duration::ZERO)); let scope = new_scope(); @@ -2154,7 +2325,16 @@ async fn the_storage_calls_of_a_prune_run_at_nice_19() { .into_iter() .skip_while(|(op_label, _, _)| op_label != "read_ledger") .filter(|(op_label, _, _)| { - !["read_ledger", "write_ledger", "list_data"].contains(&op_label.as_str()) + ![ + "read_ledger", + "write_ledger", + "list_data", + "list_claims", + "read_claim", + "write_claim", + "delete_claims", + ] + .contains(&op_label.as_str()) }) .collect::>(); From 9c940503c37264f943db8d7c4bae926a687c7430 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 10:49:20 -0700 Subject: [PATCH 46/55] Say that the prune threshold is 10 percent of the size, rounded down --- .../src/filesystem_snapshot/rustic/prune.rs | 7 ++++--- .../src/filesystem_snapshot/rustic/store.rs | 2 +- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index f84b25977e..88ac3bc1e1 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -106,8 +106,9 @@ pub(super) fn needs_repository_size(ledger: &PruneLedger, now: Timestamp, grace: /// Tells whether a prune is due at `now`. /// /// A prune is due when the grace period passed since the last prune, and the freed bytes reach the -/// threshold share of `repository_bytes` or the last prune marked packs. A threshold of zero bytes -/// counts as one byte, so a prune never runs for a scope that freed nothing and marked nothing. +/// threshold share of `repository_bytes`, rounded down to a whole byte, or the last prune marked +/// packs. A threshold of zero bytes counts as one byte, so a prune never runs for a scope that +/// freed nothing and marked nothing. pub(super) fn prune_due( ledger: &PruneLedger, now: Timestamp, @@ -327,7 +328,7 @@ mod tests { } #[test] - fn a_prune_is_due_when_the_freed_bytes_reach_ten_percent_of_the_repository() { + fn a_prune_is_due_when_the_freed_bytes_reach_ten_percent_of_the_repository_rounded_down() { let now = at(10_000_000); let due = |freed, repository_bytes| { prune_due( diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index e8637c7802..acfc3a3684 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -62,7 +62,7 @@ use tokio_util::sync::{CancellationToken, DropGuard}; use tokio_util::task::TaskTracker; /// The share of the size of the repository that deleted snapshots must free before a delete prunes -/// the scope. +/// the scope. The threshold is 10% of the size, rounded down to a whole byte, so about 10%. const PRUNE_THRESHOLD: Percent = Percent(10); /// How long a pack that a prune marks stays before a later prune deletes it. It is also the From d83cd13269ec1faa0611d1ca4504993bb4aa96aa Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 10:51:46 -0700 Subject: [PATCH 47/55] Check that a claim is taken one time and listed with its time --- .../src/filesystem_snapshot/rustic/prune.rs | 30 +++++++++++++++++-- 1 file changed, 28 insertions(+), 2 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs index 88ac3bc1e1..fb8795a02b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/prune.rs @@ -281,8 +281,8 @@ fn parse_claim(content: &[u8]) -> Option { mod tests { use super::super::files::SnapshotFiles; use super::{ - ClaimChoice, LEDGER_PATH, ListedClaim, Percent, PruneLedger, needs_repository_size, - next_claim, prune_due, read_ledger, write_ledger, + ClaimChoice, LEDGER_PATH, ListedClaim, Percent, PruneLedger, claims_directory, list_claims, + needs_repository_size, next_claim, prune_due, read_ledger, take_claim, write_ledger, }; use golem_common::model::Timestamp; use golem_common::model::environment::EnvironmentId; @@ -453,6 +453,32 @@ mod tests { ); } + #[test] + async fn a_claim_is_taken_one_time_and_listed_with_its_time() { + let files = new_files(); + let directory = claims_directory(&ledger(1, Some(42), false)); + let now = at(10_000_000); + + let first = take_claim(&files, &directory, 0, now).await.unwrap(); + let again = take_claim(&files, &directory, 0, at(20_000_000)) + .await + .unwrap(); + let listed = list_claims(&files, &directory).await.unwrap(); + + assert_eq!( + (directory.display().to_string(), first, again, listed), + ( + "golem/prune-claims/42".to_string(), + true, + false, + vec![ListedClaim { + number: 0, + claimed_at: Some(now) + }] + ) + ); + } + #[test] fn a_delete_adds_its_bytes_and_a_prune_starts_the_ledger_again() { let deleted = ledger(5, Some(7), true) From 5e67f23f2a9d0fdc83f51037d19e70f9ad52e58d Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 11:10:48 -0700 Subject: [PATCH 48/55] Prune only when the ledger did not change after the claim --- .../src/filesystem_snapshot/rustic/store.rs | 18 +++++++- .../filesystem_snapshot/rustic/store/tests.rs | 45 +++++++++++++++++++ 2 files changed, 61 insertions(+), 2 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs index acfc3a3684..289eb584c8 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store.rs @@ -292,8 +292,9 @@ impl RusticSnapshotStore { /// Adds the freed bytes to the ledger of the scope, and prunes the repository when a prune is /// due. The ledger keeps the freed bytes before the prune starts, so a delete that runs again /// after a failed prune prunes again. It lists the packs only when their size can make a prune due. - /// A due prune runs only after the delete takes a claim of its ledger. A failed prune deletes - /// the claim, and a prune that succeeds deletes each claim of its ledger. + /// A due prune runs only after the delete takes a claim of its ledger, and only when the ledger + /// did not change after the claim. A failed prune deletes the claim, and a prune that succeeds + /// deletes each claim of its ledger. async fn prune_when_due( &self, scope: &SnapshotScope, @@ -333,6 +334,19 @@ impl RusticSnapshotStore { { return Ok(()); } + // A prune writes its ledger before it deletes the claims, so a delete that claims in a + // directory that such a prune removed sees the new ledger here. + match read_ledger(&files).await { + Ok(again) if claims_directory(&again) == claims => {} + Ok(_) => { + release_claim(&files, &claims, number).await; + return Ok(()); + } + Err(error) => { + release_claim(&files, &claims, number).await; + return Err(storage_failure(error)); + } + } let backend = Arc::new(self.backend(scope, token)?); let key = self.key.clone(); let settings = self.policy.prune; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 83ebd818d0..3dd5a17846 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -982,6 +982,51 @@ async fn two_deletes_that_read_the_same_ledger_make_one_prune() { ); } +#[test] +#[timeout("60s")] +async fn a_delete_that_claims_after_another_prune_removed_the_claims_does_not_prune() { + // The gate holds the first delete after its ledger read and before its listing of the claims. + // The second delete prunes to its end, so the first delete claims in a removed directory. + let held = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let held = held.clone(); + move |op_label, _| { + if op_label == "list_claims" && !held.swap(true, Ordering::SeqCst) { + Script::WaitForGate + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + let late = tokio::spawn({ + let store = store.clone(); + let scope = scope.clone(); + async move { store.delete(&scope, &name("p-1")).await } + }); + let late_held = eventually(|| held.load(Ordering::SeqCst)).await; + + let pruned = store.delete(&scope, &name("p-2")).await; + storage.open_gate(); + let late = tokio::time::timeout(LIMIT, late).await; + + assert!(pruned.is_ok(), "{pruned:?}"); + assert!(matches!(late, Ok(Ok(Ok(())))), "{late:?}"); + assert_eq!( + ( + late_held, + prunes(&storage.calls()), + blobs(&*storage, &scope.0, "golem/prune-claims/").await, + ), + (true, 1, Vec::::new()) + ); +} + #[test] #[timeout("60s")] async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() { From bc746e2cd1412cea721fab534b31f53b29ece467 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 12:16:22 -0700 Subject: [PATCH 49/55] Check that a failed prune and a failed second read each delete the claim --- .../filesystem_snapshot/rustic/store/tests.rs | 57 ++++++++++++++++++- 1 file changed, 54 insertions(+), 3 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs index 3dd5a17846..7fd5332540 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/store/tests.rs @@ -1029,9 +1029,9 @@ async fn a_delete_that_claims_after_another_prune_removed_the_claims_does_not_pr #[test] #[timeout("60s")] -async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() { - // The first call after the first claim is the start of the first prune, and it fails before - // the prune lists the packs. So only the retry counts as a prune. +async fn a_failed_second_read_of_the_ledger_deletes_the_claim_and_a_retry_of_the_delete_prunes() { + // The first call after the first claim is the second read of the ledger, and it fails. So the + // first delete does not prune, and only the retry counts as a prune. let claimed = Arc::new(AtomicBool::new(false)); let refused = Arc::new(AtomicBool::new(false)); let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { @@ -1074,6 +1074,57 @@ async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() ); } +#[test] +#[timeout("60s")] +async fn a_prune_that_fails_deletes_its_claim_and_a_retry_of_the_delete_prunes() { + // The first listing of the packs by a prune after the first claim fails. Only a prune lists + // the packs with that call, so the second read of the ledger passes and the prune fails. + let claimed = Arc::new(AtomicBool::new(false)); + let refused = Arc::new(AtomicBool::new(false)); + let storage = ScriptedBlobStorage::new(Arc::new(InMemoryBlobStorage::new()), { + let (claimed, refused) = (claimed.clone(), refused.clone()); + move |op_label, path| { + if op_label == "write_claim" { + claimed.store(true, Ordering::SeqCst); + } + if op_label == "list" + && path == Path::new("data") + && claimed.load(Ordering::SeqCst) + && !refused.swap(true, Ordering::SeqCst) + { + Script::Refuse + } else { + Script::Pass + } + } + }); + let store = store( + storage.clone(), + policy(LONG_DEADLINE, ALWAYS, Duration::from_secs(3600)), + ); + let scope = new_scope(); + save_each(&store, &scope, &["p-1", "p-2"]).await; + + let failed = store.delete(&scope, &name("p-1")).await; + let claims_after_failure = blobs(&*storage, &scope.0, "golem/prune-claims/").await; + let retried = store.delete(&scope, &name("p-1")).await; + let after = ledger(&storage, &scope).await; + + assert!( + failed.as_ref().is_err_and(|error| is_storage(error, true)), + "{failed:?}" + ); + assert!(retried.is_ok(), "{retried:?}"); + assert_eq!( + ( + claims_after_failure, + refused.load(Ordering::SeqCst), + after.last_prune.is_some() + ), + (Vec::::new(), true, true) + ); +} + #[test] #[timeout("60s")] async fn a_claim_older_than_the_grace_period_does_not_block_a_prune() { From 581ce654c667f93035c5291535c9f583da0c1704 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 21:38:59 -0700 Subject: [PATCH 50/55] Remove the filesystem snapshot benchmark, its binary, its feature and clap --- Cargo.lock | 1 - golem-worker-executor/Cargo.toml | 9 - .../filesystem_snapshot/benchmark/agents.rs | 460 ----- .../filesystem_snapshot/benchmark/capture.rs | 250 --- .../src/filesystem_snapshot/benchmark/cli.rs | 357 ---- .../benchmark/concurrent.rs | 621 ------ .../benchmark/golden/phase_result.json | 108 - .../filesystem_snapshot/benchmark/history.rs | 392 ---- .../filesystem_snapshot/benchmark/measure.rs | 435 ----- .../src/filesystem_snapshot/benchmark/mod.rs | 1320 ------------- .../filesystem_snapshot/benchmark/report.rs | 301 --- .../filesystem_snapshot/benchmark/requests.rs | 659 ------- .../filesystem_snapshot/benchmark/scopes.rs | 97 - .../filesystem_snapshot/benchmark/sqlite.rs | 141 -- .../filesystem_snapshot/benchmark/tests.rs | 1736 ----------------- .../filesystem_snapshot/benchmark/trees.rs | 1324 ------------- .../filesystem_snapshot/benchmark/volume.rs | 555 ------ .../src/filesystem_snapshot/mod.rs | 2 - .../src/fs_snapshot_benchmark.rs | 30 - golem-worker-executor/src/lib.rs | 4 - 20 files changed, 8802 deletions(-) delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/agents.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/capture.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/cli.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/concurrent.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/golden/phase_result.json delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/history.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/measure.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/report.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/requests.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/scopes.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/sqlite.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/tests.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/trees.rs delete mode 100644 golem-worker-executor/src/filesystem_snapshot/benchmark/volume.rs delete mode 100644 golem-worker-executor/src/fs_snapshot_benchmark.rs diff --git a/Cargo.lock b/Cargo.lock index 943cc74d43..1e340eeb3a 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4606,7 +4606,6 @@ dependencies = [ "cap-std", "cap-time-ext", "chrono", - "clap", "criterion", "dashmap", "desert_rust", diff --git a/golem-worker-executor/Cargo.toml b/golem-worker-executor/Cargo.toml index d54d18649a..70039382c5 100644 --- a/golem-worker-executor/Cargo.toml +++ b/golem-worker-executor/Cargo.toml @@ -13,7 +13,6 @@ autotests = false [features] test-utils = [] -fs-snapshot-benchmark = ["dep:clap"] [lib] path = "src/lib.rs" @@ -25,12 +24,6 @@ path = "src/server.rs" harness = false test = false -[[bin]] -name = "fs-snapshot-benchmark" -path = "src/fs_snapshot_benchmark.rs" -required-features = ["fs-snapshot-benchmark"] -test = false - [[test]] name = "integration" path = "tests/lib.rs" @@ -61,7 +54,6 @@ cap-fs-ext = { workspace = true } cap-std = { workspace = true } cap-time-ext = { workspace = true } # keep in sync with wasmtime chrono = { workspace = true } -clap = { workspace = true, optional = true } dashmap = { workspace = true } desert_rust = { workspace = true } drop-stream = { workspace = true } @@ -160,7 +152,6 @@ golem-worker-executor-test-utils = { workspace = true } assert2 = { workspace = true } aws-config = { workspace = true } aws-sdk-s3 = { workspace = true } -clap = { workspace = true } criterion = { workspace = true } axum = { workspace = true } figment = { workspace = true, features = ["test"] } diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/agents.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/agents.rs deleted file mode 100644 index ac885922b3..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/agents.rs +++ /dev/null @@ -1,460 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The repositories of the agents of a scenario, a CPU setting and a tree. -//! -//! Each agent has its own repository, as in production. The repository of an agent is below the -//! path `agents/` of the namespace of its scenario, CPU setting and tree. Thus a copy of -//! the repository of one agent into another agent is a copy of blobs in one namespace, which the -//! S3 backend does in the bucket with `CopyObject`. - -use super::TARGET_LABEL; -use async_trait::async_trait; -use bytes::Bytes; -use futures::stream::BoxStream; -use futures::{StreamExt, TryStreamExt}; -use golem_service_base::replayable_stream::ErasedReplayableStream; -use golem_service_base::storage::blob::{ - BlobMetadata, BlobStorage, BlobStorageNamespace, ExistsResult, ListedBlob, PutIfAbsent, -}; -use serde_json::{Value, json}; -use std::path::{Path, PathBuf}; -use std::sync::Arc; - -/// The first name of the path of the repository of each agent. -pub(super) const AGENTS: &str = "agents"; - -/// The agent whose repository the phases copy to the other agents. -pub(super) const FIRST_AGENT: &str = "0"; - -/// The number of blob copies that are in progress at the same time. -const COPY_CONCURRENCY: usize = 64; - -/// The name of the file that a repository writes when it is made. -const CONFIG: &str = "config"; - -/// Gives the path of the repository of the agent in its namespace. -fn agent_root(agent: &str) -> PathBuf { - Path::new(AGENTS).join(agent) -} - -/// A blob storage that keeps the repository of one agent below its path in the namespace. -/// -/// Each call goes to `inner`, with the path below the path of the agent. Each path that `inner` -/// gives is relative to the path of the agent again. -#[derive(Debug)] -pub(super) struct AgentStorage { - inner: Arc, - root: Box, -} - -impl AgentStorage { - pub(super) fn new(inner: Arc, agent: &str) -> Self { - Self { - inner, - root: agent_root(agent).into_boxed_path(), - } - } - - fn path(&self, path: &Path) -> PathBuf { - self.root.join(path) - } - - /// Gives the path that `inner` gave, relative to the path of the agent. - fn relative(&self, path: &Path) -> anyhow::Result { - path.strip_prefix(&self.root) - .map(Path::to_path_buf) - .map_err(|_| { - anyhow::anyhow!( - "the storage gave the path {}, which is not below {}", - path.display(), - self.root.display() - ) - }) - } -} - -#[async_trait] -impl BlobStorage for AgentStorage { - async fn get_raw( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result>> { - self.inner - .get_raw(target_label, op_label, namespace, &self.path(path)) - .await - } - - async fn get_stream( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result>>> { - self.inner - .get_stream(target_label, op_label, namespace, &self.path(path)) - .await - } - - async fn get_raw_slice( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - start: u64, - end: u64, - ) -> anyhow::Result>> { - self.inner - .get_raw_slice( - target_label, - op_label, - namespace, - &self.path(path), - start, - end, - ) - .await - } - - async fn get_metadata( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.inner - .get_metadata(target_label, op_label, namespace, &self.path(path)) - .await - } - - async fn put_raw( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - data: &[u8], - ) -> anyhow::Result<()> { - self.inner - .put_raw(target_label, op_label, namespace, &self.path(path), data) - .await - } - - async fn put_raw_if_absent( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - data: &[u8], - ) -> anyhow::Result { - self.inner - .put_raw_if_absent(target_label, op_label, namespace, &self.path(path), data) - .await - } - - async fn put_stream( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - stream: &dyn ErasedReplayableStream>, Error = anyhow::Error>, - ) -> anyhow::Result<()> { - self.inner - .put_stream(target_label, op_label, namespace, &self.path(path), stream) - .await - } - - async fn delete( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result<()> { - self.inner - .delete(target_label, op_label, namespace, &self.path(path)) - .await - } - - async fn create_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result<()> { - self.inner - .create_dir(target_label, op_label, namespace, &self.path(path)) - .await - } - - async fn list_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.inner - .list_dir(target_label, op_label, namespace, &self.path(path)) - .await? - .iter() - .map(|path| self.relative(path)) - .collect() - } - - async fn list_blobs_below( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.inner - .list_blobs_below(target_label, op_label, namespace, &self.path(path)) - .await? - .iter() - .map(|blob| { - self.relative(&blob.path).map(|path| ListedBlob { - path: path.into_boxed_path(), - size: blob.size, - }) - }) - .collect() - } - - async fn delete_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result { - self.inner - .delete_dir(target_label, op_label, namespace, &self.path(path)) - .await - } - - async fn exists( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result { - self.inner - .exists(target_label, op_label, namespace, &self.path(path)) - .await - } -} - -/// Copies the repository of [`FIRST_AGENT`] to each agent from `1` to `agents - 1` that has no -/// repository, and gives the numbers of the copied repositories and blobs. -/// -/// The config of a repository is its last copied blob, so an agent with a config has each blob of -/// the repository, also after a copy that stopped. The copies of the other blobs of all agents -/// are in progress at the same time, up to [`COPY_CONCURRENCY`]. -pub(super) async fn copy_first_agent( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - agents: usize, -) -> anyhow::Result { - let source = &agent_root(FIRST_AGENT); - let blobs = storage - .list_blobs_below(TARGET_LABEL, "list_agent", namespace.clone(), source) - .await? - .iter() - .map(|blob| blob.path.strip_prefix(source).map(Path::to_path_buf)) - .collect::, _>>()?; - anyhow::ensure!( - blobs.iter().any(|path| path == Path::new(CONFIG)), - "the repository of the agent {FIRST_AGENT} has no {CONFIG}" - ); - let missing = futures::stream::iter(1..agents) - .map(|agent| async move { - let config = agent_root(&agent.to_string()).join(CONFIG); - storage - .get_metadata(TARGET_LABEL, "find_agent", namespace.clone(), &config) - .await - .map(|metadata| metadata.is_none().then_some(agent)) - }) - .buffered(COPY_CONCURRENCY) - .try_filter_map(|agent| async move { Ok(agent) }) - .try_collect::>() - .await?; - let (configs, others): (Vec<_>, Vec<_>) = missing - .iter() - .flat_map(|agent| blobs.iter().map(move |blob| (*agent, blob))) - .partition(|(_, blob)| *blob == Path::new(CONFIG)); - futures::stream::iter(others.iter().copied()) - .map(|(agent, blob)| copy_blob(storage, namespace, source, agent, blob)) - .buffer_unordered(COPY_CONCURRENCY) - .try_collect::<()>() - .await?; - futures::stream::iter(configs.iter().copied()) - .map(|(agent, blob)| copy_blob(storage, namespace, source, agent, blob)) - .buffer_unordered(COPY_CONCURRENCY) - .try_collect::<()>() - .await?; - Ok(json!({ - "agents_copied": missing.len(), - "blobs_copied": configs.len() + others.len(), - })) -} - -/// Copies the blob of the repository at `source` to the repository of the agent. -async fn copy_blob( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - source: &Path, - agent: usize, - blob: &Path, -) -> anyhow::Result<()> { - storage - .copy( - TARGET_LABEL, - "copy_agent", - namespace.clone(), - &source.join(blob), - &agent_root(&agent.to_string()).join(blob), - ) - .await -} - -/// The directories of a repository in the order in which a copy of the repository lists and -/// copies them. A save writes them in the opposite order, so a snapshot that the copy has also has -/// its index and its packs. The config goes last. -const COPY_ORDER: [&str; 3] = ["snapshots", "index", "data"]; - -/// Gives each blob of the repository of the agent, with the path relative to the repository, in -/// the order of the paths. -pub(super) async fn agent_blobs( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - agent: &str, -) -> anyhow::Result> { - let root = agent_root(agent); - let mut blobs = storage - .list_blobs_below(TARGET_LABEL, "list_agent", namespace.clone(), &root) - .await? - .iter() - .map(|blob| { - Ok(ListedBlob { - path: blob.path.strip_prefix(&root)?.into(), - size: blob.size, - }) - }) - .collect::>>()?; - blobs.sort_by(|left, right| left.path.cmp(&right.path)); - Ok(blobs.into_boxed_slice()) -} - -/// Gives the sum of the sizes of the blobs. -pub(super) fn total_bytes(blobs: &[ListedBlob]) -> u64 { - blobs.iter().map(|blob| blob.size).sum() -} - -/// Copies the repository of the agent `from` to the agent `to` on the server, as the copy of a -/// scope does it, and gives the numbers of the copied blobs and bytes. -/// -/// The copy lists the snapshot files, the index files and the packs, in this order, and then -/// copies them in the same order, each group after the one before it. The config goes last, so an -/// agent with a config has each blob of the copy. -pub(super) async fn copy_agent( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - from: &str, - to: &str, -) -> anyhow::Result { - let source = &agent_root(from); - let groups = futures::stream::iter(COPY_ORDER) - .then(|directory| async move { - storage - .list_blobs_below( - TARGET_LABEL, - "list_agent", - namespace.clone(), - &source.join(directory), - ) - .await - }) - .try_collect::>() - .await?; - let config = storage - .get_metadata( - TARGET_LABEL, - "find_agent", - namespace.clone(), - &source.join(CONFIG), - ) - .await? - .ok_or_else(|| anyhow::anyhow!("the repository of the agent {from} has no {CONFIG}"))?; - let groups = groups - .into_iter() - .chain(std::iter::once(Box::from([ListedBlob { - path: source.join(CONFIG).into_boxed_path(), - size: config.size, - }]))) - .collect::>(); - let target = &agent_root(to); - futures::stream::iter(groups.iter()) - .map(Ok) - .try_for_each(|group| async move { - futures::stream::iter(group.iter()) - .map(|blob| async move { - let relative = blob.path.strip_prefix(source)?; - storage - .copy( - TARGET_LABEL, - "copy_agent", - namespace.clone(), - &blob.path, - &target.join(relative), - ) - .await - }) - .buffer_unordered(COPY_CONCURRENCY) - .try_collect::<()>() - .await - }) - .await?; - Ok(json!({ - "blobs_copied": groups.iter().map(|group| group.len()).sum::(), - "bytes_copied": groups.iter().map(|group| total_bytes(group)).sum::(), - })) -} - -/// Deletes the repository of the agent, and tells whether it had one. -pub(super) async fn delete_agent( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - agent: &str, -) -> anyhow::Result { - storage - .delete_dir( - TARGET_LABEL, - "delete_agent", - namespace.clone(), - &agent_root(agent), - ) - .await -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/capture.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/capture.rs deleted file mode 100644 index b0a8bbfb36..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/capture.rs +++ /dev/null @@ -1,250 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The phase of the capture scenario: the time of a capture, and the warm saves of a capture with -//! each change detection. -//! -//! A capture is a reflink copy of the tree of an agent that keeps the permissions and the -//! modification times, as the executor makes it before an upload. Each captured file is a new -//! inode with a new change time. So a save that compares change times reads every file of a -//! capture, and a save that compares only sizes and modification times reads the changed files. - -use super::agents::{FIRST_AGENT, copy_agent}; -use super::measure::measure; -use super::report::{Outcome, StepRecord, TreeFacts}; -use super::trees::{self, CopyCounts, Times}; -use super::{ - COLD_SAVE, PhaseContext, PhaseOutcome, WARM_SAVE, change_detection_name, failed, save_record, - snapshot_name, -}; -use crate::filesystem_snapshot::rustic::{ChangeDetection, SaveSettings}; -use futures::{StreamExt, TryStreamExt}; -use serde_json::{Value, json}; -use std::path::{Path, PathBuf}; - -/// The agent whose repository gets the warm save with change detection by size and modification -/// time. The first agent gets the warm save that compares change times. -const SIZE_MTIME_AGENT: &str = "1"; - -/// The capture phase: a new tree, a capture of it, and a cold save of the capture into the -/// repository of the first agent, which then goes to a second agent on the server. Then a small -/// change of the tree, and for each change detection a new capture and a warm save of it into the -/// repository of one of the agents. So both warm saves have the same parent, and each reads a -/// capture that no step read before. -/// -/// The time of a capture step is the time of the capture: the copy does not sync the volume. -pub(super) async fn capture(context: &PhaseContext) -> PhaseOutcome { - let spec = context.selection.tree; - let storage = &context.storage; - let tree = context.work_dir.join("tree"); - let facts = TreeFacts { - name: spec.name, - content: Some(spec.content.label()), - page_cache: Some("dropped"), - ..TreeFacts::default() - }; - let later = [ - "capture", - "cold_save", - "copy_scopes", - "small_change", - FORM_STEPS[0], - FORM_STEPS[1], - FORM_STEPS[2], - FORM_STEPS[3], - ]; - - let (record, generated) = measure("generate_tree", storage, trees::generate(spec, &tree)).await; - let mut steps = vec![record]; - let Ok(counts) = generated else { - return failed(facts, steps, "generate_tree", &later); - }; - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - ..facts - }; - - let first = context.work_dir.join("capture-cold"); - let (record, captured) = measure("capture", storage, capture_tree(&tree, &first)).await; - steps.push(capture_record(record, &captured)); - if captured.is_err() { - return failed(facts, steps, "capture", &later[1..]); - } - - let (record, cold) = measure("cold_save", storage, async { - context - .repository() - .save_with(&snapshot_name(COLD_SAVE)?, &first, SaveSettings::DEFAULT) - .await - }) - .await; - steps.push(save_record(record, &cold)); - if cold.is_err() { - return failed(facts, steps, "cold_save", &later[2..]); - } - - let (record, copied) = measure( - "copy_scopes", - storage, - copy_agent( - storage.as_ref(), - &context.scope().0, - FIRST_AGENT, - SIZE_MTIME_AGENT, - ), - ) - .await; - steps.push(record.with_details( - copied.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )); - if copied.is_err() { - return failed(facts, steps, "copy_scopes", &later[3..]); - } - - let (record, changed) = measure("small_change", storage, trees::change(spec, &tree)).await; - steps.push(record.with_details( - changed.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )); - let Ok(change) = changed else { - return failed(facts, steps, "small_change", &later[4..]); - }; - let facts = TreeFacts { - change: Some(change), - ..facts - }; - - match warm_forms(context, &tree, steps).await { - Ok(steps) => PhaseOutcome { - tree_facts: facts, - steps, - outcome: Outcome::Ok, - }, - Err((steps, step, skipped)) => failed(facts, steps, step, skipped), - } -} - -/// One form of a warm save. -#[derive(Clone, Copy, Debug)] -struct Form { - /// The name of the step that captures the tree for the save. - capture_step: &'static str, - /// The name of the step of the save. - save_step: &'static str, - /// The agent whose repository gets the save. - agent: &'static str, - detection: ChangeDetection, -} - -/// The forms of the warm saves, in the order in which they run. -const FORMS: [Form; 2] = [ - Form { - capture_step: "capture_full_read", - save_step: "warm_save_full_read", - agent: FIRST_AGENT, - detection: ChangeDetection::Ctime, - }, - Form { - capture_step: "capture_size_mtime", - save_step: "warm_save_size_mtime", - agent: SIZE_MTIME_AGENT, - detection: ChangeDetection::SizeMtime, - }, -]; - -/// The names of the steps of the forms, in the order in which they run. -static FORM_STEPS: [&str; 4] = [ - FORMS[0].capture_step, - FORMS[0].save_step, - FORMS[1].capture_step, - FORMS[1].save_step, -]; - -/// Runs the capture and the warm save of each form, one form after the other. A failed step gives -/// the records so far, the name of the step and the steps that did not run. -async fn warm_forms( - context: &PhaseContext, - tree: &Path, - steps: Vec, -) -> Result, (Vec, &'static str, &'static [&'static str])> { - let storage = &context.storage; - futures::stream::iter(FORMS.iter().copied().enumerate()) - .map(Ok) - .try_fold( - steps, - |mut steps, - ( - index, - Form { - capture_step, - save_step, - agent, - detection, - }, - )| async move { - let target: PathBuf = context.work_dir.join(capture_step); - let (record, captured) = - measure(capture_step, storage, capture_tree(tree, &target)).await; - steps.push(capture_record(record, &captured)); - if captured.is_err() { - return Err((steps, capture_step, &FORM_STEPS[2 * index + 1..])); - } - let settings = SaveSettings { - detection, - ..SaveSettings::DEFAULT - }; - let (record, warm) = measure(save_step, storage, async { - context - .agent_repository(agent) - .save_with(&snapshot_name(WARM_SAVE)?, &target, settings) - .await - }) - .await; - steps.push(save_record(record, &warm).with_parameters(json!({ - "change_detection": change_detection_name(detection), - "agent": agent, - }))); - if warm.is_err() { - return Err((steps, save_step, &FORM_STEPS[2 * index + 2..])); - } - Ok(steps) - }, - ) - .await -} - -/// Captures the tree `from` into the new directory `to` on a blocking thread. -async fn capture_tree(from: &Path, to: &Path) -> anyhow::Result { - let (from, to): (Box, Box) = (from.into(), to.into()); - tokio::task::spawn_blocking(move || trees::copy_tree(&from, &to, Times::Keep)).await? -} - -/// Gives the record of a capture with the number of files that are reflinks and copies. -fn capture_record(record: StepRecord, captured: &anyhow::Result) -> StepRecord { - record.with_details( - captured - .as_ref() - .map(|counts| { - json!({ - "files_reflinked": counts.reflinked, - "files_copied": counts.copied, - }) - }) - .unwrap_or(Value::Null), - Box::default(), - ) -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/cli.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/cli.rs deleted file mode 100644 index 203051a105..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/cli.rs +++ /dev/null @@ -1,357 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The command line of the filesystem snapshot benchmark. -//! -//! `plan` prints the phases of each tree of the scenarios, one JSON line for each tree. `run` runs -//! one phase and prints its result as the last line of the standard output. The blob storage -//! comes from the `GOLEM__BLOB_STORAGE__*` environment variables of the executor, with the object -//! prefix of the run. The exit code is 0 when each step succeeded, 1 when a step failed, the -//! restored tree differs or the result could not be written, and 2 for an error of the arguments -//! or of the configuration, or for an error of the environment that each later phase finds again, -//! such as an async runtime that does not start. - -use super::{Selection, is_key_segment, plan, run_phase}; -use clap::Parser; -use figment::Figment; -use figment::providers::{Env, Serialized}; -use golem_common::tracing::{OutputConfig, TracingConfig, init_tracing_with_default_env_filter}; -use golem_service_base::config::{BlobStorageConfig, S3BlobStorageConfig}; -use golem_service_base::storage::blob::BlobStorage; -use golem_service_base::storage::blob::fs::FileSystemBlobStorage; -use golem_service_base::storage::blob::s3::S3BlobStorage; -use serde_json::{Value, json}; -use std::path::{Path, PathBuf}; -use std::process::ExitCode; -use std::sync::Arc; - -/// The start of each object prefix that the benchmark accepts. -const OBJECT_PREFIX_START: &str = "fs-snapshot-bench/"; - -const ENV_PREFIX: &str = "GOLEM__BLOB_STORAGE__"; - -#[derive(Debug, PartialEq, Eq, Parser)] -#[command(name = "fs-snapshot-benchmark")] -enum Command { - /// Prints the phases of each tree of the scenarios, one JSON line for each tree. - Plan { - /// The names of the scenarios, separated by commas. - #[arg(long, value_delimiter = ',', required = true)] - scenarios: Vec, - }, - /// Runs one phase of one tree of a scenario. - Run(RunArguments), -} - -#[derive(Debug, PartialEq, Eq, clap::Args)] -struct RunArguments { - /// The id of the run: 1 to 64 ASCII letters, digits, `-` or `_`. - #[arg(long)] - run_id: String, - #[arg(long)] - scenario: String, - #[arg(long)] - tree: String, - #[arg(long)] - phase: String, - /// The label of the CPU setting of the pod, which the result records. - #[arg(long)] - cpu_setting: String, - /// The directory on the benchmark volume. - #[arg(long)] - work_dir: PathBuf, - /// The object prefix of the run: `fs-snapshot-bench/`, with or without one `/` at the - /// end. - #[arg(long)] - object_prefix: String, -} - -/// Runs the command of the arguments of the process. -pub fn main() -> ExitCode { - match Command::try_parse() { - Ok(Command::Plan { scenarios }) => print_plan(&scenarios), - Ok(Command::Run(arguments)) => run(arguments), - Err(error) => { - let _ = error.print(); - ExitCode::from(u8::try_from(error.exit_code()).unwrap_or(2)) - } - } -} - -fn print_plan(scenarios: &[String]) -> ExitCode { - match plan(scenarios) { - Ok(entries) => { - entries - .iter() - .for_each(|entry| println!("{}", serde_json::to_string(entry).unwrap_or_default())); - ExitCode::SUCCESS - } - Err(name) => usage_error(&format!("no scenario has the name {name:?}")), - } -} - -fn usage_error(message: &str) -> ExitCode { - eprintln!("error: {message}"); - ExitCode::from(2) -} - -fn run(arguments: RunArguments) -> ExitCode { - let selection = match validate(&arguments) { - Ok(selection) => selection, - Err(message) => return usage_error(&message), - }; - let config = match storage_config(&arguments.object_prefix) { - Ok(config) => config, - Err(message) => return usage_error(&message), - }; - let _ = rustls::crypto::ring::default_provider().install_default(); - let _ = init_tracing_with_default_env_filter(&TracingConfig { - stdout: OutputConfig::disabled(), - stderr: OutputConfig::text(), - ..TracingConfig::local_dev("fs-snapshot-benchmark") - }); - let runtime = match crate::bootstrap::create_runtime() { - Ok(runtime) => runtime, - Err(error) => { - eprintln!("error: failed to start the async runtime: {error}"); - return ExitCode::from(2); - } - }; - runtime.block_on(async { - let storage = match storage(&config).await { - Ok(storage) => storage, - Err(error) => { - eprintln!("error: failed to open the blob storage: {error:#}"); - return ExitCode::from(2); - } - }; - let (result, written) = run_phase( - &arguments.run_id, - &arguments.cpu_setting, - selection, - &arguments.work_dir, - storage, - environment(&config), - ) - .await; - println!("{}", serde_json::to_string(&result).unwrap_or_default()); - if let Err(error) = &written { - eprintln!("error: failed to write the result into the blob storage: {error:#}"); - } - if written.is_ok() && result.outcome == super::report::Outcome::Ok { - ExitCode::SUCCESS - } else { - ExitCode::from(1) - } - }) -} - -/// Checks the arguments of `run`, and gives the phase that they name. -fn validate(arguments: &RunArguments) -> Result { - if !is_key_segment(&arguments.run_id) { - return Err(format!( - "the run id {:?} does not have 1 to 64 ASCII letters, digits, `-` or `_`", - arguments.run_id - )); - } - if !is_key_segment(&arguments.cpu_setting) { - return Err(format!( - "the CPU setting {:?} does not have 1 to 64 ASCII letters, digits, `-` or `_`", - arguments.cpu_setting - )); - } - let expected = format!("{OBJECT_PREFIX_START}{}", arguments.run_id); - if arguments.object_prefix != expected && arguments.object_prefix != format!("{expected}/") { - return Err(format!( - "the object prefix {:?} is not {expected:?}, the prefix of the run", - arguments.object_prefix - )); - } - Selection::find(&arguments.scenario, &arguments.tree, &arguments.phase) -} - -/// Reads the blob storage configuration of the executor from the environment, with the object -/// prefix of the run. -fn storage_config(object_prefix: &str) -> Result { - let config = Figment::from(Serialized::defaults(BlobStorageConfig::default_s3())) - .merge(Env::prefixed(ENV_PREFIX).split("__")) - .extract::() - .map_err(|error| format!("the blob storage configuration is not valid: {error}"))?; - match config { - BlobStorageConfig::S3(config) => Ok(BlobStorageConfig::S3(S3BlobStorageConfig { - object_prefix: object_prefix.trim_end_matches('/').to_string(), - ..config - })), - BlobStorageConfig::LocalFileSystem(config) => { - Ok(BlobStorageConfig::LocalFileSystem(config)) - } - _ => { - Err("the benchmark supports the blob storage types S3 and LocalFileSystem".to_string()) - } - } -} - -async fn storage(config: &BlobStorageConfig) -> anyhow::Result> { - match config { - BlobStorageConfig::S3(config) => Ok(Arc::new(S3BlobStorage::new(config.clone()).await)), - BlobStorageConfig::LocalFileSystem(config) => { - Ok(Arc::new(FileSystemBlobStorage::new(&config.root).await?)) - } - _ => anyhow::bail!("the benchmark supports the blob storage types S3 and LocalFileSystem"), - } -} - -/// Records the pod, the host and the storage of the run. -fn environment(config: &BlobStorageConfig) -> Value { - let read = |path: &str| { - std::fs::read_to_string(Path::new(path)) - .ok() - .map(|text| text.trim().to_string()) - }; - json!({ - "pod": std::env::var("POD_NAME").ok(), - "node": std::env::var("NODE_NAME").ok(), - "kernel": read("/proc/sys/kernel/osrelease"), - "available_parallelism": std::thread::available_parallelism().map(|count| count.get()).ok(), - "tokio_workers": tokio::runtime::Handle::try_current().ok().map(|handle| handle.metrics().num_workers()), - "rayon_threads": rayon::current_num_threads(), - "cgroup": { - "cpu_max": read("/sys/fs/cgroup/cpu.max"), - "cpuset": read("/sys/fs/cgroup/cpuset.cpus.effective"), - "memory_max": read("/sys/fs/cgroup/memory.max").and_then(|text| text.parse::().ok()), - }, - "storage": match config { - BlobStorageConfig::S3(config) => json!({ - "type": "S3", - "bucket": config.initial_agent_files_bucket, - "region": config.region, - "object_prefix": config.object_prefix, - "retries": { - "max_attempts": config.retries.max_attempts, - "min_delay_ms": config.retries.min_delay.as_millis() as u64, - "max_delay_ms": config.retries.max_delay.as_millis() as u64, - "multiplier": config.retries.multiplier, - }, - }), - BlobStorageConfig::LocalFileSystem(config) => json!({ - "type": "LocalFileSystem", - "root": config.root.display().to_string(), - }), - _ => Value::Null, - }, - }) -} - -#[cfg(test)] -mod tests { - use super::{Command, RunArguments, storage_config, validate}; - use clap::Parser; - use golem_service_base::config::BlobStorageConfig; - use pretty_assertions::assert_eq; - use std::path::PathBuf; - use test_r::test; - - fn run_arguments() -> RunArguments { - RunArguments { - run_id: "123-1".to_string(), - scenario: "base".to_string(), - tree: "files-1g".to_string(), - phase: "save".to_string(), - cpu_setting: "limit-3".to_string(), - work_dir: PathBuf::from("/data"), - object_prefix: "fs-snapshot-bench/123-1".to_string(), - } - } - - #[test] - fn the_plan_and_run_command_lines_of_the_workflow_parse() { - let plan = - Command::try_parse_from(["fs-snapshot-benchmark", "plan", "--scenarios", "base,smoke"]); - let run = Command::try_parse_from([ - "fs-snapshot-benchmark", - "run", - "--run-id", - "123-1", - "--scenario", - "base", - "--tree", - "files-1g", - "--phase", - "save", - "--cpu-setting", - "limit-3", - "--work-dir", - "/data", - "--object-prefix", - "fs-snapshot-bench/123-1", - ]); - let missing = Command::try_parse_from(["fs-snapshot-benchmark", "plan"]) - .map_err(|error| error.exit_code()); - - assert_eq!( - (plan.ok(), run.ok(), missing), - ( - Some(Command::Plan { - scenarios: vec!["base".to_string(), "smoke".to_string()], - }), - Some(Command::Run(run_arguments())), - Err(2), - ) - ); - } - - #[test] - fn a_run_needs_a_valid_run_id_cpu_setting_prefix_and_selection() { - let with = |change: fn(&mut RunArguments)| { - let mut arguments = run_arguments(); - change(&mut arguments); - validate(&arguments).is_ok() - }; - - assert_eq!( - [ - with(|_| {}), - with(|arguments| arguments.run_id = "a/b".to_string()), - with(|arguments| arguments.cpu_setting = String::new()), - with(|arguments| arguments.object_prefix = "release".to_string()), - with(|arguments| arguments.object_prefix = "fs-snapshot-bench/".to_string()), - with( - |arguments| arguments.object_prefix = "fs-snapshot-bench/other-run".to_string() - ), - with(|arguments| { - arguments.object_prefix = "fs-snapshot-bench/123-1/extra".to_string() - }), - with(|arguments| arguments.object_prefix = "fs-snapshot-bench/123-1//".to_string()), - with(|arguments| arguments.object_prefix = "fs-snapshot-bench/123-1/".to_string()), - with(|arguments| arguments.tree = "files-tiny".to_string()), - ], - [ - true, false, false, false, false, false, false, false, true, false - ] - ); - } - - #[test] - fn the_object_prefix_of_the_run_replaces_the_prefix_of_the_configuration() { - let config = storage_config("fs-snapshot-bench/123-1/").unwrap(); - - assert_eq!( - match config { - BlobStorageConfig::S3(config) => Some(config.object_prefix), - _ => None, - }, - Some("fs-snapshot-bench/123-1".to_string()) - ); - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/concurrent.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/concurrent.rs deleted file mode 100644 index f9c9c451f9..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/concurrent.rs +++ /dev/null @@ -1,621 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The phases that run the operations of several agents at the same time, in one pod. -//! -//! Each agent has its own repository (see [`super::agents`]). A measured step starts the -//! operations of all agents at the same time on the async runtime. Each operation runs on a -//! blocking thread of the runtime, as the executor runs it. The requests of all agents go into the -//! record of the step. - -use super::agents::copy_first_agent; -use super::measure::measure; -use super::report::{Outcome, StepRecord, StepStatus, TreeFacts}; -use super::requests::times; -use super::trees::{self, CopyCounts}; -use super::{ - COLD_SAVE, PhaseContext, PhaseOutcome, WARM_SAVE, failed, saved_hash, snapshot_name, - without_save, -}; -use crate::filesystem_snapshot::rustic::SaveReport; -use futures::{StreamExt, TryStreamExt}; -use serde_json::{Map, Value, json}; -use std::future::Future; -use std::num::NonZeroUsize; -use std::path::{Path, PathBuf}; -use std::time::{Duration, Instant}; - -/// The result of the operation of one agent in a batch. -struct AgentRun { - /// The time from the start of the operation to its end. - time: Duration, - /// The time from the start of the batch to the end of the operation. - finished: Duration, - result: anyhow::Result, -} - -/// Starts the operations at the same time, and gives the result of each, in the order of the -/// operations. -async fn batch( - operations: impl IntoIterator>>, -) -> Box<[AgentRun]> { - let started = Instant::now(); - futures::future::join_all(operations.into_iter().map(|operation| async move { - let begin = Instant::now(); - let result = operation.await; - AgentRun { - time: begin.elapsed(), - finished: started.elapsed(), - result, - } - })) - .await - .into_boxed_slice() -} - -/// Gives the details of a batch: the number of agents, the times of their operations, the number -/// of operations that failed, and the error of the operation that failed first. -fn batch_details(runs: &[AgentRun]) -> Map { - let mut sorted = runs.iter().map(|run| run.time).collect::>(); - sorted.sort(); - let first_error = runs - .iter() - .filter_map(|run| run.result.as_ref().err().map(|error| (run.finished, error))) - .min_by_key(|(finished, _)| *finished) - .map(|(_, error)| format!("{error:#}")); - let details = json!({ - "agents": runs.len(), - "time_ms": times(&sorted), - "failed": runs.iter().filter(|run| run.result.is_err()).count(), - "first_error": first_error, - }); - match details { - Value::Object(fields) => fields, - _ => Map::new(), - } -} - -/// Gives the record of the step of a batch with the parameters and the details. A batch with a -/// failed operation is a failed step, with the error of the operation that failed first. -fn batch_record(record: StepRecord, parameters: Value, details: Map) -> StepRecord { - let status = match details.get("first_error").and_then(Value::as_str) { - Some(error) => StepStatus::Error(error.into()), - None => record.status.clone(), - }; - StepRecord { - status, - ..record - .with_parameters(parameters) - .with_details(Value::Object(details), Box::default()) - } -} - -/// Tells whether each operation of the batch succeeded. -fn all_ok(runs: &anyhow::Result]>>) -> bool { - runs.as_ref() - .is_ok_and(|runs| runs.iter().all(|run| run.result.is_ok())) -} - -/// A restore phase of several agents: the repository of the first agent goes to each agent that -/// has none, the agents restore the warm save at the same time, and the hash of each restored -/// tree is compared with the hash that the save phase recorded. -/// -/// The phase deletes the restored trees at its end, so the next phase has the space of the -/// volume. -pub(super) async fn concurrent_restore( - context: &PhaseContext, - agents: usize, - reader_threads: Option, -) -> PhaseOutcome { - let into = context.work_dir.join("restore"); - let outcome = restore_agents(context, agents, reader_threads, &into).await; - let _ = tokio::fs::remove_dir_all(&into).await; - outcome -} - -async fn restore_agents( - context: &PhaseContext, - agents: usize, - reader_threads: Option, - into: &Path, -) -> PhaseOutcome { - let storage = &context.storage; - let facts = TreeFacts { - name: context.selection.tree.name, - ..TreeFacts::default() - }; - let parameters = json!({ "agents": agents, "reader_threads": reader_threads }); - let expected = match saved_hash(context).await { - Ok(expected) => expected, - Err(error) => { - return without_save( - facts, - &error, - &["copy_scopes", "concurrent_restore", "hash_trees"], - ); - } - }; - - let (record, copied) = measure( - "copy_scopes", - storage, - copy_first_agent(storage.as_ref(), &context.scope().0, agents), - ) - .await; - let mut steps = vec![record.with_details( - copied.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )]; - if copied.is_err() { - return failed( - facts, - steps, - "copy_scopes", - &["concurrent_restore", "hash_trees"], - ); - } - - let targets = (0..agents) - .map(|agent| into.join(agent.to_string())) - .collect::>(); - let (record, restored) = measure("concurrent_restore", storage, async { - targets.iter().try_for_each(std::fs::create_dir_all)?; - let name = snapshot_name(WARM_SAVE)?; - let name = &name; - Ok(batch( - targets - .iter() - .enumerate() - .map(|(agent, target)| async move { - context - .agent_repository(&agent.to_string()) - .restore(name, target, reader_threads) - .await? - .ok_or_else(|| anyhow::anyhow!("no snapshot has the name {WARM_SAVE}")) - }), - ) - .await) - }) - .await; - steps.push(match &restored { - Ok(runs) => batch_record(record, parameters, batch_details(runs)), - Err(_) => record.with_parameters(parameters), - }); - if !all_ok(&restored) { - return failed(facts, steps, "concurrent_restore", &["hash_trees"]); - } - - compare_hashes(context, facts, steps, &targets, &expected).await -} - -/// Measures the hash of each restored tree, compares each hash with the expected hash, and gives -/// the outcome of the phase with the record of that step. -async fn compare_hashes( - context: &PhaseContext, - facts: TreeFacts, - mut steps: Vec, - targets: &[PathBuf], - expected: &str, -) -> PhaseOutcome { - let (record, hashed) = measure("hash_trees", &context.storage, hash_trees(targets)).await; - let Ok(hashes) = hashed else { - steps.push(record); - return failed(facts, steps, "hash_trees", &[]); - }; - let matches = hashes - .iter() - .filter(|(hash, _)| **hash == *expected) - .count(); - steps.push(record.with_details( - json!({ "trees": hashes.len(), "matches": matches, "expected": expected }), - Box::default(), - )); - let facts = match hashes.first() { - Some((hash, counts)) => TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - hash: Some(hash.clone()), - ..facts - }, - None => facts, - }; - PhaseOutcome { - tree_facts: facts, - steps, - outcome: if matches == hashes.len() { - Outcome::Ok - } else { - Outcome::Failed { - reason: format!( - "{} of {} restored trees differ from the saved tree", - hashes.len() - matches, - hashes.len() - ) - .into(), - } - }, - } -} - -/// Gives the hash of each tree, in the order of the trees. The number of trees that are read at -/// the same time is the number of CPUs that the process can use. -async fn hash_trees(roots: &[PathBuf]) -> anyhow::Result, trees::TreeCounts)]>> { - let parallel = std::thread::available_parallelism().map_or(1, NonZeroUsize::get); - Ok(futures::stream::iter(roots) - .map(|root| trees::hash(root)) - .buffered(parallel) - .try_collect::>() - .await? - .into_boxed_slice()) -} - -/// A save phase of several agents: a new tree and a copy of it for each other agent, the cold -/// saves of all agents at the same time, the small change of each tree, and the warm saves of all -/// agents at the same time. -/// -/// Each agent of the phase has a repository of its own, whose name is the number of the agent -/// after `prefix`. The other phases of the scenario do not use these repositories, so each cold -/// save makes a repository. The copies of the tree are reflinks where the volume has them. After -/// the copies, no page of a tree is in the page cache, so each save reads its tree from the -/// volume. -pub(super) async fn concurrent_save( - context: &PhaseContext, - agents: usize, - prefix: &str, -) -> PhaseOutcome { - let spec = context.selection.tree; - let storage = &context.storage; - let roots = tree_roots(context, agents); - let names = (0..agents) - .map(|agent| format!("{prefix}-{agent}")) - .collect::>(); - let parameters = json!({ "agents": agents }); - let facts = TreeFacts { - name: spec.name, - content: Some(spec.content.label()), - page_cache: Some("dropped"), - ..TreeFacts::default() - }; - let later = [ - "copy_trees", - "concurrent_cold_save", - "small_change", - "concurrent_warm_save", - ]; - - let (record, generated) = - measure("generate_tree", storage, generate_first(context, &roots)).await; - let mut steps = vec![record]; - let Ok(counts) = generated else { - return failed(facts, steps, "generate_tree", &later); - }; - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - ..facts - }; - - let (record, copied) = measure("copy_trees", storage, copy_trees(&roots)).await; - steps.push( - record.with_details( - copied - .as_ref() - .map(|counts| { - json!({ - "copies": agents.saturating_sub(1), - "files_reflinked": counts.reflinked, - "files_copied": counts.copied, - }) - }) - .unwrap_or(Value::Null), - Box::default(), - ), - ); - if copied.is_err() { - return failed(facts, steps, "copy_trees", &later[1..]); - } - - let (record, cold) = measure( - "concurrent_cold_save", - storage, - save_agents(context, &names, &roots, COLD_SAVE), - ) - .await; - steps.push(save_batch_record(record, parameters.clone(), &cold)); - if !all_ok(&cold) { - return failed(facts, steps, "concurrent_cold_save", &later[2..]); - } - - let (record, changed) = measure("small_change", storage, async { - futures::stream::iter(roots.iter()) - .then(|root| trees::change(spec, root)) - .try_collect::>() - .await - }) - .await; - let change = changed - .as_ref() - .ok() - .and_then(|changes| changes.first().cloned()); - steps.push(record.with_details(json!({ "trees": agents, "change": change }), Box::default())); - if changed.is_err() { - return failed(facts, steps, "small_change", &later[3..]); - } - let facts = TreeFacts { change, ..facts }; - - let (record, warm) = measure( - "concurrent_warm_save", - storage, - save_agents(context, &names, &roots, WARM_SAVE), - ) - .await; - steps.push(save_batch_record(record, parameters, &warm)); - if !all_ok(&warm) { - return failed(facts, steps, "concurrent_warm_save", &[]); - } - PhaseOutcome { - tree_facts: facts, - steps, - outcome: Outcome::Ok, - } -} - -/// Gives the root of the tree of each of the agents. -fn tree_roots(context: &PhaseContext, agents: usize) -> Box<[PathBuf]> { - (0..agents) - .map(|agent| context.work_dir.join("trees").join(agent.to_string())) - .collect() -} - -/// Makes the tree of the phase at the first root. -async fn generate_first( - context: &PhaseContext, - roots: &[PathBuf], -) -> anyhow::Result { - let first = roots - .first() - .ok_or_else(|| anyhow::anyhow!("a phase of agents needs one agent or more"))?; - std::fs::create_dir_all(context.work_dir.join("trees"))?; - trees::generate(context.selection.tree, first).await -} - -/// Copies the first tree to each other root, and then removes the pages of each tree from the -/// page cache. -async fn copy_trees(roots: &[PathBuf]) -> anyhow::Result { - let roots = roots.to_vec(); - tokio::task::spawn_blocking(move || { - let counts = match roots.split_first() { - Some((first, others)) => { - others - .iter() - .try_fold(CopyCounts::default(), |counts, root| { - trees::copy_tree(first, root, trees::Times::Drop) - .map(|copied| counts.with(copied)) - })? - } - None => CopyCounts::default(), - }; - roots.iter().try_for_each(|root| trees::settle(root))?; - Ok(counts) - }) - .await? -} - -/// Saves the tree of each agent into the repository of the agent with the snapshot name, all at -/// the same time. -async fn save_agents( - context: &PhaseContext, - names: &[String], - roots: &[PathBuf], - snapshot: &str, -) -> anyhow::Result]>> { - let name = snapshot_name(snapshot)?; - let name = &name; - let settings = context.save_settings(); - Ok( - batch(names.iter().zip(roots).map(|(agent, root)| async move { - context - .agent_repository(agent) - .save_with(name, root, settings) - .await - })) - .await, - ) -} - -/// Gives the record of a step of saves: the details of the batch, and the sums of the data that -/// the saves added. -fn save_batch_record( - record: StepRecord, - parameters: Value, - runs: &anyhow::Result]>>, -) -> StepRecord { - match runs { - Ok(runs) => batch_record(record, parameters, save_batch_details(runs)), - Err(_) => record.with_parameters(parameters), - } -} - -/// Gives the details of a batch of saves, and the sums of the data that the saves added. -fn save_batch_details(runs: &[AgentRun]) -> Map { - let reports = runs - .iter() - .filter_map(|run| run.result.as_ref().ok()) - .collect::>(); - let mut details = batch_details(runs); - details.insert( - "data_added".to_string(), - json!(reports.iter().map(|report| report.data_added).sum::()), - ); - details.insert( - "data_added_packed".to_string(), - json!( - reports - .iter() - .map(|report| report.data_added_packed) - .sum::() - ), - ); - details -} - -/// A mixed phase: the repository of the first agent goes to each of `restores` agents that has -/// none, then `saves` agents save a new tree cold and the `restores` agents restore the warm save -/// cold, all at the same time, and the hash of each restored tree is compared with the hash that -/// the save phase recorded. -/// -/// The saves use the save threads of the variant of the phase, and the restores use -/// `reader_threads`. So the step gives the memory of one choice of both values. The time of a -/// restore is the time until its files are readable, not until they are durable, because nothing -/// syncs the volume. The phase deletes the restored trees at its end. -pub(super) async fn mixed( - context: &PhaseContext, - saves: usize, - restores: usize, - reader_threads: Option, -) -> PhaseOutcome { - let into = context.work_dir.join("restore"); - let outcome = mix(context, saves, restores, reader_threads, &into).await; - let _ = tokio::fs::remove_dir_all(&into).await; - outcome -} - -async fn mix( - context: &PhaseContext, - saves: usize, - restores: usize, - reader_threads: Option, - into: &Path, -) -> PhaseOutcome { - let storage = &context.storage; - let spec = context.selection.tree; - let facts = TreeFacts { - name: spec.name, - content: Some(spec.content.label()), - page_cache: Some("dropped"), - ..TreeFacts::default() - }; - let parameters = json!({ - "saves": saves, - "restores": restores, - "reader_threads": reader_threads, - }); - let later = ["generate_tree", "copy_trees", "mixed", "hash_trees"]; - let expected = match saved_hash(context).await { - Ok(expected) => expected, - Err(error) => { - return without_save( - facts, - &error, - &[ - "copy_scopes", - "generate_tree", - "copy_trees", - "mixed", - "hash_trees", - ], - ); - } - }; - - let (record, copied) = measure( - "copy_scopes", - storage, - copy_first_agent(storage.as_ref(), &context.scope().0, restores + 1), - ) - .await; - let mut steps = vec![record.with_details( - copied.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )]; - if copied.is_err() { - return failed(facts, steps, "copy_scopes", &later); - } - - let roots = tree_roots(context, saves); - let (record, generated) = - measure("generate_tree", storage, generate_first(context, &roots)).await; - steps.push(record); - let Ok(counts) = generated else { - return failed(facts, steps, "generate_tree", &later[1..]); - }; - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - ..facts - }; - let (record, copied) = measure("copy_trees", storage, copy_trees(&roots)).await; - steps.push(record); - if copied.is_err() { - return failed(facts, steps, "copy_trees", &later[2..]); - } - - let names = (0..saves) - .map(|agent| format!("{}-{agent}", context.selection.phase.name)) - .collect::>(); - let targets = (1..=restores) - .map(|agent| into.join(agent.to_string())) - .collect::>(); - let (record, ran) = measure("mixed", storage, async { - targets.iter().try_for_each(std::fs::create_dir_all)?; - let name = snapshot_name(WARM_SAVE)?; - let name = &name; - let restoring = batch( - targets - .iter() - .enumerate() - .map(|(index, target)| async move { - context - .agent_repository(&(index + 1).to_string()) - .restore(name, target, reader_threads) - .await? - .ok_or_else(|| anyhow::anyhow!("no snapshot has the name {WARM_SAVE}")) - }), - ); - let (saved, restored) = - futures::join!(save_agents(context, &names, &roots, COLD_SAVE), restoring); - Ok((saved?, restored)) - }) - .await; - let failed_mix = match &ran { - Ok((saved, restored)) => { - let save_details = save_batch_details(saved); - let restore_details = batch_details(restored); - let first_error = save_details - .get("first_error") - .filter(|error| !error.is_null()) - .or_else(|| restore_details.get("first_error")) - .cloned() - .unwrap_or(Value::Null); - let mut details = Map::new(); - details.insert("first_error".to_string(), first_error); - details.insert("saves".to_string(), Value::Object(save_details)); - details.insert("restores".to_string(), Value::Object(restore_details)); - steps.push(batch_record(record, parameters, details)); - !(saved.iter().all(|run| run.result.is_ok()) - && restored.iter().all(|run| run.result.is_ok())) - } - Err(_) => { - steps.push(record.with_parameters(parameters)); - true - } - }; - if failed_mix { - return failed(facts, steps, "mixed", &later[3..]); - } - compare_hashes(context, facts, steps, &targets, &expected).await -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/golden/phase_result.json b/golem-worker-executor/src/filesystem_snapshot/benchmark/golden/phase_result.json deleted file mode 100644 index 3ced4cd170..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/golden/phase_result.json +++ /dev/null @@ -1,108 +0,0 @@ -{ - "format": "golem-fs-snapshot-benchmark/1", - "run_id": "123-1", - "scenario": "base", - "phase": "save", - "tree": "files-128m", - "cpu_setting": "limit-3", - "environment": { - "pod": "p" - }, - "volume": { - "check": { - "status": "ok" - } - }, - "tree_facts": { - "name": "files-128m", - "files": 10000, - "directories": 100, - "bytes": 134217728, - "content": "incompressible", - "page_cache": "dropped", - "hash": "h1", - "hash_after_change": "h2", - "change": { - "files_rewritten": 10 - } - }, - "steps": [ - { - "name": "cold_save", - "status": "ok", - "parameters": {}, - "wall_ms": 1500.25, - "cpu": { - "user_ms": 900.5, - "system_ms": 100.0, - "cgroup": { - "usage_ms": 1000.0, - "user_ms": 900.0, - "system_ms": 100.0, - "nr_periods": 15, - "nr_throttled": 2, - "throttled_ms": 30.5 - } - }, - "memory": { - "rss_start_bytes": 1000, - "rss_peak_bytes": 5000, - "rss_peak_source": "VmHWM", - "cgroup_current_peak_bytes": 9000, - "cgroup_anon_peak_bytes": 6000, - "cgroup_file_peak_bytes": null - }, - "threads": { - "start": 10, - "peak": 40, - "sample_interval_ms": 10 - }, - "requests": [ - { - "call": "put_raw", - "file_type": "pack", - "count": 3, - "errors": 0, - "bytes": 3000, - "time_ms": { - "min": 1.0, - "p50": 2.0, - "p90": 3.0, - "p99": 3.0, - "max": 3.0, - "total": 6.0 - } - } - ], - "bytes_written": 3000, - "bytes_read": 0, - "phases": [ - { - "name": "backup", - "wall_ms": 1400.0 - } - ], - "details": { - "snapshot": "abc" - } - }, - { - "name": "warm_save", - "status": "skipped", - "parameters": {}, - "wall_ms": null, - "cpu": null, - "memory": null, - "threads": null, - "requests": [], - "bytes_written": 0, - "bytes_read": 0, - "phases": [], - "details": {} - } - ], - "outcome": { - "status": "failed", - "reason": "the step warm_save failed" - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/history.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/history.rs deleted file mode 100644 index b4b163cac3..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/history.rs +++ /dev/null @@ -1,392 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The phases of the prune and repository open scenarios. -//! -//! The history phase makes a repository with a history of saves in the repository of the first -//! agent. Each prune phase copies that repository on the server into the repository of its own -//! agent, and prunes the copy. So each prune starts from the same repository. - -use super::agents::{FIRST_AGENT, agent_blobs, copy_agent, total_bytes}; -use super::measure::measure; -use super::report::{Outcome, StepRecord, TreeFacts}; -use super::{ - COLD_SAVE, PhaseContext, PhaseOutcome, failed, inspect_record, phase_walls, save_record, - saved_hash, snapshot_name, trees, without_save, -}; -use crate::filesystem_snapshot::rustic::{PruneReport, PruneSettings, RepackLimits, Repository}; -use futures::{StreamExt, TryStreamExt}; -use serde_json::{Value, json}; -use std::time::Duration; - -/// The number of saves after the cold save: the warm save and 10 more. -const ROUNDS: u8 = 11; - -/// The saves after which the history phase opens the repository. -const OPEN_AFTER: [u8; 3] = [1, 11, 12]; - -/// The number of newest snapshots that the forget keeps. -const KEEP: u8 = 2; - -/// The name of the snapshot of the save of the round, from 1 to [`ROUNDS`]. -fn snapshot_of_round(round: u8) -> Box { - format!("round-{round:02}").into() -} - -/// The name of the newest snapshot of the history. -fn newest() -> Box { - snapshot_of_round(ROUNDS) -} - -/// The settings of both prunes of a prune phase: no grace period, and no limit that leaves unused -/// data or stops a repack. -const fn prune_settings(fast_repack: bool) -> PruneSettings { - PruneSettings { - fast_repack, - keep_delete: Duration::ZERO, - repack: RepackLimits::Unlimited, - } -} - -/// The history phase: a cold save of a new tree, then a small change and a save in each of -/// [`ROUNDS`] rounds, each round with other content, a forget of every snapshot but the -/// [`KEEP`] newest, and the hash of the tree. -/// -/// The phase opens the repository after the saves in [`OPEN_AFTER`]. Each open finds the newest -/// snapshot by its name and loads the index, as a restore does. -pub(super) async fn history(context: &PhaseContext) -> PhaseOutcome { - let spec = context.selection.tree; - let tree = context.work_dir.join("tree"); - let repository = context.repository(); - let storage = &context.storage; - let settings = context.save_settings(); - let facts = TreeFacts { - name: spec.name, - content: Some(spec.content.label()), - page_cache: Some("dropped"), - ..TreeFacts::default() - }; - - let (record, generated) = measure("generate_tree", storage, trees::generate(spec, &tree)).await; - let mut steps = vec![record]; - let Ok(counts) = generated else { - return failed(facts, steps, "generate_tree", &["cold_save"]); - }; - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - ..facts - }; - - let (record, cold) = measure("cold_save", storage, async { - repository - .save_with(&snapshot_name(COLD_SAVE)?, &tree, settings) - .await - }) - .await; - steps.push(save_record(record, &cold)); - if cold.is_err() { - return failed(facts, steps, "cold_save", &[]); - } - let steps = open_step(context, &repository, COLD_SAVE, 1, steps).await; - - let (repository_ref, tree_ref) = (&repository, tree.as_path()); - let saved = futures::stream::iter(1..=ROUNDS) - .map(Ok::<_, Vec>) - .try_fold(steps, |steps, round| async move { - save_round(context, repository_ref, tree_ref, round, steps).await - }) - .await; - let steps = match saved { - Ok(steps) => steps, - Err(steps) => { - let step = steps.last().map_or("round", |step| step.name); - return failed(facts, steps, step, &[]); - } - }; - - let forgotten = (0..=ROUNDS - KEEP) - .map(|round| match round { - 0 => Box::from(COLD_SAVE), - round => snapshot_of_round(round), - }) - .collect::>(); - let (record, forgot) = measure("forget", storage, async { - futures::stream::iter(forgotten.iter()) - .then(|name| async { repository.forget(&snapshot_name(name)?).await }) - .try_fold(0_u64, |total, count| async move { Ok(total + count) }) - .await - }) - .await; - let mut steps = steps; - steps.push(record.with_details( - json!({ "snapshots_forgotten": forgot.as_ref().ok(), "snapshots_kept": KEEP }), - Box::default(), - )); - if forgot.is_err() { - return failed(facts, steps, "forget", &["hash_tree"]); - } - - let (record, hashed) = measure("hash_tree", storage, trees::hash(&tree)).await; - steps.push(record); - let Ok((hash, _)) = hashed else { - return failed(facts, steps, "hash_tree", &[]); - }; - PhaseOutcome { - tree_facts: TreeFacts { - hash_after_change: Some(hash), - ..facts - }, - steps, - outcome: Outcome::Ok, - } -} - -/// Changes the tree for the round and saves it, and opens the repository after the save when -/// [`OPEN_AFTER`] names it. A failed step gives the records so far as the error. -async fn save_round( - context: &PhaseContext, - repository: &Repository, - tree: &std::path::Path, - round: u8, - mut steps: Vec, -) -> Result, Vec> { - let storage = &context.storage; - let spec = context.selection.tree; - let parameters = json!({ "round": round }); - let (record, changed) = measure( - "small_change", - storage, - trees::change_round(spec, tree, round), - ) - .await; - steps.push(record.with_parameters(parameters.clone()).with_details( - changed.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )); - if changed.is_err() { - return Err(steps); - } - let name = snapshot_of_round(round); - let (record, saved) = measure("save", storage, async { - repository - .save_with(&snapshot_name(&name)?, tree, context.save_settings()) - .await - }) - .await; - steps.push(save_record(record, &saved).with_parameters(parameters)); - if saved.is_err() { - return Err(steps); - } - Ok(open_step(context, repository, &name, round + 1, steps).await) -} - -/// Opens the repository and finds the snapshot with the name, when the number of saves is in -/// [`OPEN_AFTER`], and gives the steps with the record of the open. A failed open is recorded and -/// does not stop the phase. -async fn open_step( - context: &PhaseContext, - repository: &Repository, - name: &str, - saves: u8, - mut steps: Vec, -) -> Vec { - if OPEN_AFTER.contains(&saves) { - let (record, inspected) = measure("open", &context.storage, async { - repository.inspect(&snapshot_name(name)?).await - }) - .await; - steps.push(inspect_record(record, &inspected).with_parameters(json!({ "saves": saves }))); - } - steps -} - -/// A prune phase: the repository of the history phase goes on the server to the agent of the -/// variant of the phase, two prunes with no grace period run on it, the first to mark the packs -/// that no snapshot uses and the second to delete them, then an open, and a restore of the newest -/// snapshot whose hash is compared with the hash of the history phase. -/// -/// The details of the second prune give the listed bytes of the repository before the first prune -/// and after the second, and the difference, which is what the prunes gave back. The restore time -/// is the time until the files are readable, not until they are durable. -pub(super) async fn prune(context: &PhaseContext, fast_repack: bool) -> PhaseOutcome { - let into = context.work_dir.join("restore"); - let storage = &context.storage; - let namespace = context.scope().0; - let agent = context.variant().agent; - let repository = context.repository(); - let settings = prune_settings(fast_repack); - let parameters = json!({ - "fast_repack": fast_repack, - "keep_delete_s": settings.keep_delete.as_secs(), - "max_unused": "0%", - "max_repack": "unlimited", - }); - let facts = TreeFacts { - name: context.selection.tree.name, - ..TreeFacts::default() - }; - let later = [ - "copy_scopes", - "prune_mark", - "prune_delete", - "open", - "cold_restore", - "hash_tree", - ]; - let expected = match saved_hash(context).await { - Ok(expected) => expected, - Err(error) => return without_save(facts, &error, &later), - }; - - let (record, copied) = measure( - "copy_scopes", - storage, - copy_agent(storage.as_ref(), &namespace, FIRST_AGENT, agent), - ) - .await; - let mut steps = vec![record.with_details( - copied.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )]; - if copied.is_err() { - return failed(facts, steps, "copy_scopes", &later[1..]); - } - let before = agent_blobs(storage.as_ref(), &namespace, agent).await; - - let (record, marked) = measure("prune_mark", storage, repository.prune(settings)).await; - steps.push(prune_record(record, &marked, parameters.clone())); - if marked.is_err() { - return failed(facts, steps, "prune_mark", &later[2..]); - } - let (record, deleted) = measure("prune_delete", storage, repository.prune(settings)).await; - let after = agent_blobs(storage.as_ref(), &namespace, agent).await; - let record = prune_record(record, &deleted, parameters); - let given_back = match (&before, &after) { - (Ok(before), Ok(after)) => json!({ - "bytes_before": total_bytes(before), - "bytes_after": total_bytes(after), - "bytes_given_back": total_bytes(before).saturating_sub(total_bytes(after)), - "blobs_before": before.len(), - "blobs_after": after.len(), - }), - _ => Value::Null, - }; - steps.push(with_detail(record, "repository", given_back)); - if deleted.is_err() { - return failed(facts, steps, "prune_delete", &later[3..]); - } - - let newest = newest(); - let (record, inspected) = measure("open", storage, async { - repository.inspect(&snapshot_name(&newest)?).await - }) - .await; - steps.push(inspect_record(record, &inspected).with_parameters(json!({ "saves": "pruned" }))); - - let (record, restored) = measure("cold_restore", storage, async { - std::fs::create_dir(&into)?; - repository - .restore(&snapshot_name(&newest)?, &into, None) - .await? - .ok_or_else(|| anyhow::anyhow!("no snapshot has the name {newest}")) - }) - .await; - steps.push(match &restored { - Ok(report) => record.with_details( - json!({ "files": report.files, "dirs": report.dirs, "bytes": report.bytes }), - phase_walls(&report.phases), - ), - Err(_) => record, - }); - if restored.is_err() { - return failed(facts, steps, "cold_restore", &later[5..]); - } - let (record, hashed) = measure("hash_tree", storage, trees::hash(&into)).await; - let _ = tokio::fs::remove_dir_all(&into).await; - let Ok((hash, counts)) = hashed else { - steps.push(record); - return failed(facts, steps, "hash_tree", &[]); - }; - let matches = *hash == *expected; - steps.push(record.with_details( - json!({ "hash": hash, "expected": expected, "matches": matches }), - Box::default(), - )); - PhaseOutcome { - tree_facts: TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - hash: Some(hash), - ..facts - }, - steps, - outcome: if matches { - Outcome::Ok - } else { - Outcome::Failed { - reason: "the restored tree differs from the saved tree".into(), - } - }, - } -} - -/// Gives the record of a prune with the plan of the prune. -fn prune_record( - record: StepRecord, - pruned: &anyhow::Result>, - parameters: Value, -) -> StepRecord { - let record = record.with_parameters(parameters); - match pruned { - Ok(Some(report)) => record.with_details(prune_details(report), phase_walls(&report.phases)), - Ok(None) => record.with_details(json!({ "repository": null }), Box::default()), - Err(_) => record, - } -} - -fn prune_details(report: &PruneReport) -> Value { - json!({ - "packs_used": report.packs_used, - "packs_partly_used": report.packs_partly_used, - "packs_unused": report.packs_unused, - "packs_repacked": report.packs_repacked, - "packs_kept": report.packs_kept, - "marked_packs_deleted": report.marked_packs_deleted, - "marked_bytes_deleted": report.marked_bytes_deleted, - "marked_packs_kept": report.marked_packs_kept, - "bytes_used": report.bytes_used, - "bytes_unused": report.bytes_unused, - "bytes_removed": report.bytes_removed, - "bytes_repacked": report.bytes_repacked, - "bytes_repack_removed": report.bytes_repack_removed, - "index_files": report.index_files, - "index_files_rebuilt": report.index_files_rebuilt, - }) -} - -/// Gives the record with the value at the key in its details. -fn with_detail(record: StepRecord, key: &str, value: Value) -> StepRecord { - let details = match record.details.clone() { - Value::Object(mut details) => { - details.insert(key.to_string(), value); - Value::Object(details) - } - other => other, - }; - let phases = record.phases.clone(); - record.with_details(details, phases) -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/measure.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/measure.rs deleted file mode 100644 index ca55941104..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/measure.rs +++ /dev/null @@ -1,435 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! Measures one step of a phase: wall time, CPU time, memory, threads and blob storage requests. -//! -//! The values come from the process and from the cgroup v2 files of the container. A value that -//! the host does not give is `null` in the result. - -use super::report::{ - CgroupCpuTime, CpuTime, MemoryPeaks, StepRecord, StepStatus, ThreadCounts, millis, -}; -use super::requests::{MeasuredBlobStorage, summarize, written_and_read}; -use serde_json::Value; -use std::future::Future; -use std::path::Path; -use std::sync::Arc; -use std::sync::atomic::{AtomicBool, Ordering}; -use std::time::{Duration, Instant}; - -/// The time between two samples of the memory and the threads. -const SAMPLE_INTERVAL: Duration = Duration::from_millis(10); - -const PROC_STATUS: &str = "/proc/self/status"; -const PROC_CLEAR_REFS: &str = "/proc/self/clear_refs"; -const CGROUP_CPU_STAT: &str = "/sys/fs/cgroup/cpu.stat"; -const CGROUP_MEMORY_CURRENT: &str = "/sys/fs/cgroup/memory.current"; -const CGROUP_MEMORY_STAT: &str = "/sys/fs/cgroup/memory.stat"; -const CGROUP_MEMORY_EVENTS: &str = "/sys/fs/cgroup/memory.events"; - -/// Runs the step and measures it. -/// -/// The requests of the storage before the step are removed first, so the record holds only the -/// requests of the step. A line on the standard error tells that the step starts, so the log of a -/// pod that the kernel stopped shows the step that ran. -pub(super) async fn measure( - name: &'static str, - storage: &MeasuredBlobStorage, - step: impl Future>, -) -> (StepRecord, anyhow::Result) { - eprintln!("fs-snapshot-benchmark: the step {name} starts"); - let _ = storage.take(); - let before = ProcessSample::read(); - let peak_reset = reset_peak_rss(); - let sampler = Sampler::start(); - let started = Instant::now(); - let result = step.await; - let wall = started.elapsed(); - let peaks = sampler.stop(); - let after = ProcessSample::read(); - let records = storage.take(); - let (bytes_written, bytes_read) = written_and_read(&records); - let (rss_peak, rss_peak_source) = match after.status.peak_rss { - Some(peak) if peak_reset => (peak, "VmHWM"), - _ => (peaks.rss, "sampled VmRSS"), - }; - let record = StepRecord { - name, - status: match &result { - Ok(_) => StepStatus::Ok, - Err(error) => StepStatus::Error(format!("{error:#}").into()), - }, - parameters: Value::Object(Default::default()), - wall_ms: Some(millis(wall)), - cpu: Some(CpuTime { - user_ms: millis(after.user.saturating_sub(before.user)), - system_ms: millis(after.system.saturating_sub(before.system)), - cgroup: before - .cgroup_cpu - .zip(after.cgroup_cpu) - .map(|(before, after)| after.since(&before)), - }), - memory: Some(MemoryPeaks { - rss_start_bytes: before.status.rss.unwrap_or_default(), - rss_peak_bytes: rss_peak, - rss_peak_source, - cgroup_current_peak_bytes: peaks.cgroup_current, - cgroup_anon_peak_bytes: peaks.cgroup_anon, - cgroup_file_peak_bytes: peaks.cgroup_file, - }), - threads: Some(ThreadCounts { - start: before.status.threads.unwrap_or_default(), - peak: peaks.threads, - sample_interval_ms: SAMPLE_INTERVAL.as_millis() as u64, - }), - requests: summarize(&records), - bytes_written, - bytes_read, - phases: Box::default(), - details: Value::Object(Default::default()), - }; - (record, result) -} - -/// The values of `/proc/self/status` that a step uses. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub(super) struct StatusValues { - pub(super) rss: Option, - pub(super) peak_rss: Option, - pub(super) threads: Option, -} - -/// Reads `VmRSS`, `VmHWM` and `Threads` from the text of `/proc//status`. The sizes are in -/// bytes. -pub(super) fn parse_status(text: &str) -> StatusValues { - text.lines().filter_map(|line| line.split_once(':')).fold( - StatusValues::default(), - |values, (key, value)| { - let number = || value.split_whitespace().next()?.parse::().ok(); - match key { - "VmRSS" => StatusValues { - rss: number().map(|kib| kib * 1024), - ..values - }, - "VmHWM" => StatusValues { - peak_rss: number().map(|kib| kib * 1024), - ..values - }, - "Threads" => StatusValues { - threads: number(), - ..values - }, - _ => values, - } - }, - ) -} - -/// The CPU counters of a cgroup v2, in microseconds. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub(super) struct CgroupCpu { - pub(super) usage_usec: u64, - pub(super) user_usec: u64, - pub(super) system_usec: u64, - pub(super) nr_periods: u64, - pub(super) nr_throttled: u64, - pub(super) throttled_usec: u64, -} - -impl CgroupCpu { - fn since(&self, before: &Self) -> CgroupCpuTime { - let micros = - |after: u64, before: u64| millis(Duration::from_micros(after.saturating_sub(before))); - CgroupCpuTime { - usage_ms: micros(self.usage_usec, before.usage_usec), - user_ms: micros(self.user_usec, before.user_usec), - system_ms: micros(self.system_usec, before.system_usec), - nr_periods: self.nr_periods.saturating_sub(before.nr_periods), - nr_throttled: self.nr_throttled.saturating_sub(before.nr_throttled), - throttled_ms: micros(self.throttled_usec, before.throttled_usec), - } - } -} - -/// Reads the text of a cgroup v2 `cpu.stat`. A counter that the text does not have is zero; the -/// throttling counters are only there when the cgroup has a CPU limit controller. -pub(super) fn parse_cpu_stat(text: &str) -> CgroupCpu { - text.lines() - .filter_map(|line| line.split_once(' ')) - .filter_map(|(key, value)| value.trim().parse::().ok().map(|value| (key, value))) - .fold(CgroupCpu::default(), |counters, (key, value)| match key { - "usage_usec" => CgroupCpu { - usage_usec: value, - ..counters - }, - "user_usec" => CgroupCpu { - user_usec: value, - ..counters - }, - "system_usec" => CgroupCpu { - system_usec: value, - ..counters - }, - "nr_periods" => CgroupCpu { - nr_periods: value, - ..counters - }, - "nr_throttled" => CgroupCpu { - nr_throttled: value, - ..counters - }, - "throttled_usec" => CgroupCpu { - throttled_usec: value, - ..counters - }, - _ => counters, - }) -} - -/// Reads the value of a key of a cgroup v2 `memory.stat`, in bytes. -pub(super) fn memory_stat_value(text: &str, key: &str) -> Option { - text.lines() - .filter_map(|line| line.split_once(' ')) - .find(|(name, _)| *name == key) - .and_then(|(_, value)| value.trim().parse().ok()) -} - -/// Reads the counters of the text of a cgroup v2 `memory.events`, as a JSON object from the name -/// of each counter to its value. A line that has no counter is not in the object. -pub(super) fn parse_memory_events(text: &str) -> Value { - Value::Object( - text.lines() - .filter_map(|line| line.split_once(' ')) - .filter_map(|(key, value)| { - value - .trim() - .parse::() - .ok() - .map(|value| (key.to_string(), Value::from(value))) - }) - .collect(), - ) -} - -/// Gives the counters of the cgroup `memory.events` of the process, or `null` when the host does -/// not give them. -pub(super) fn memory_events() -> Value { - read_text(CGROUP_MEMORY_EVENTS).map_or(Value::Null, |text| parse_memory_events(&text)) -} - -/// The CPU time and the status of the process at one moment. -struct ProcessSample { - user: Duration, - system: Duration, - status: StatusValues, - cgroup_cpu: Option, -} - -impl ProcessSample { - fn read() -> Self { - let (user, system) = process_cpu_time(); - Self { - user, - system, - status: read_status(), - cgroup_cpu: read_text(CGROUP_CPU_STAT).map(|text| parse_cpu_stat(&text)), - } - } -} - -/// Gives the user and the system CPU time of all threads of the process. -fn process_cpu_time() -> (Duration, Duration) { - // SAFETY: `getrusage` writes one `rusage` record, which the zeroed value holds. - let usage = unsafe { - let mut usage = std::mem::zeroed::(); - if libc::getrusage(libc::RUSAGE_SELF, &mut usage) == 0 { - Some(usage) - } else { - None - } - }; - usage.map_or((Duration::ZERO, Duration::ZERO), |usage| { - (time_of(usage.ru_utime), time_of(usage.ru_stime)) - }) -} - -fn time_of(time: libc::timeval) -> Duration { - Duration::from_secs(u64::try_from(time.tv_sec).unwrap_or_default()) - + Duration::from_micros(u64::try_from(time.tv_usec).unwrap_or_default()) -} - -fn read_status() -> StatusValues { - read_text(PROC_STATUS) - .map(|text| parse_status(&text)) - .unwrap_or_default() -} - -fn read_text(path: &str) -> Option { - std::fs::read_to_string(Path::new(path)).ok() -} - -/// Sets `VmHWM` of the process to its current RSS, and tells whether that worked. -fn reset_peak_rss() -> bool { - std::fs::write(PROC_CLEAR_REFS, b"5").is_ok() -} - -/// The largest values that the sampler saw. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -struct Peaks { - rss: u64, - threads: u64, - cgroup_current: Option, - cgroup_anon: Option, - cgroup_file: Option, -} - -impl Peaks { - fn with(self, sample: Peaks) -> Self { - let max = |left: Option, right: Option| left.max(right); - Self { - rss: self.rss.max(sample.rss), - threads: self.threads.max(sample.threads), - cgroup_current: max(self.cgroup_current, sample.cgroup_current), - cgroup_anon: max(self.cgroup_anon, sample.cgroup_anon), - cgroup_file: max(self.cgroup_file, sample.cgroup_file), - } - } - - fn read() -> Self { - let status = read_status(); - let memory_stat = read_text(CGROUP_MEMORY_STAT); - Self { - rss: status.rss.unwrap_or_default(), - threads: status.threads.unwrap_or_default(), - cgroup_current: read_text(CGROUP_MEMORY_CURRENT) - .and_then(|text| text.trim().parse().ok()), - cgroup_anon: memory_stat - .as_deref() - .and_then(|text| memory_stat_value(text, "anon")), - cgroup_file: memory_stat - .as_deref() - .and_then(|text| memory_stat_value(text, "file")), - } - } -} - -/// A thread that samples the memory and the threads of the process until it is stopped. -struct Sampler { - stop: Arc, - thread: std::thread::JoinHandle, -} - -impl Sampler { - fn start() -> Self { - let stop = Arc::new(AtomicBool::new(false)); - let thread = std::thread::spawn({ - let stop = stop.clone(); - move || { - std::iter::from_fn(|| { - (!stop.load(Ordering::Acquire)).then(|| { - let sample = Peaks::read(); - std::thread::sleep(SAMPLE_INTERVAL); - sample - }) - }) - .fold(Peaks::default(), Peaks::with) - } - }); - Self { stop, thread } - } - - fn stop(self) -> Peaks { - self.stop.store(true, Ordering::Release); - self.thread.join().unwrap_or_default().with(Peaks::read()) - } -} - -#[cfg(test)] -mod tests { - use super::{ - CgroupCpu, StatusValues, memory_stat_value, parse_cpu_stat, parse_memory_events, - parse_status, - }; - use pretty_assertions::assert_eq; - use serde_json::json; - use test_r::test; - - #[test] - fn the_memory_events_give_each_counter() { - let text = "low 0\nhigh 12\nmax 34\noom 1\noom_kill 1\noom_group_kill 0\n"; - - assert_eq!( - (parse_memory_events(text), parse_memory_events("")), - ( - json!({ - "low": 0, - "high": 12, - "max": 34, - "oom": 1, - "oom_kill": 1, - "oom_group_kill": 0 - }), - json!({}) - ) - ); - } - - #[test] - fn the_status_gives_the_rss_the_peak_rss_and_the_threads() { - let text = "Name:\tbench\nVmPeak:\t 900 kB\nVmHWM:\t 300 kB\nVmRSS:\t 200 kB\nThreads:\t12\n"; - - assert_eq!( - (parse_status(text), parse_status("Name:\tbench\n")), - ( - StatusValues { - rss: Some(200 * 1024), - peak_rss: Some(300 * 1024), - threads: Some(12), - }, - StatusValues::default() - ) - ); - } - - #[test] - fn the_cpu_stat_gives_the_usage_and_the_throttling() { - let text = "usage_usec 1000\nuser_usec 700\nsystem_usec 300\nnr_periods 40\nnr_throttled 5\nthrottled_usec 2500\nnr_bursts 0\n"; - - assert_eq!( - parse_cpu_stat(text), - CgroupCpu { - usage_usec: 1000, - user_usec: 700, - system_usec: 300, - nr_periods: 40, - nr_throttled: 5, - throttled_usec: 2500, - } - ); - } - - #[test] - fn the_memory_stat_gives_the_value_of_a_key() { - let text = "anon 4096\nfile 8192\nfile_mapped 100\n"; - - assert_eq!( - ( - memory_stat_value(text, "anon"), - memory_stat_value(text, "file"), - memory_stat_value(text, "shmem") - ), - (Some(4096), Some(8192), None) - ); - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs deleted file mode 100644 index 7b05b1ad9e..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs +++ /dev/null @@ -1,1320 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! A benchmark of the save, restore and forget operations of the rustic repository over the blob -//! storage of the executor. -//! -//! A scenario has trees and phases. A phase is the work of one pod, and it runs on one tree. A -//! phase measures each of its steps, and writes its result as JSON into the blob storage and on -//! the standard output. A later phase of the same scenario and tree reads the result of an -//! earlier phase from the blob storage. -//! -//! Each repository and each result is in the namespace `InitialAgentFiles` of an environment -//! that only the benchmark uses, below the object prefix of the run. The repositories of a -//! scenario, a CPU setting and a tree are the repositories of its agents (see [`agents`]). - -mod agents; -mod capture; -pub mod cli; -mod concurrent; -mod history; -mod measure; -mod report; -mod requests; -mod scopes; -mod sqlite; -mod trees; -mod volume; - -use super::rustic::{ - ChangeDetection, Chunking, Compression, InspectReport, PhaseTime, Repository, RepositoryKey, - RepositorySettings, SaveSettings, -}; -use super::{SnapshotName, SnapshotScope}; -use crate::services::golem_config::DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE as STORAGE_CALL_DEADLINE; -use agents::{AgentStorage, FIRST_AGENT}; -use golem_common::model::environment::EnvironmentId; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; -use measure::measure; -use report::{FORMAT, Outcome, PhaseResult, PhaseWall, StepRecord, TreeFacts, millis}; -use requests::MeasuredBlobStorage; -use serde::Serialize; -use serde_json::{Map, Value, json}; -use std::num::{NonZeroI32, NonZeroU32, NonZeroUsize}; -use std::path::{Path, PathBuf}; -use std::sync::Arc; -use std::time::Instant; -use trees::{ - COMPRESSIBLE_1G, FILES_1G, FILES_128M, FILES_TINY, MODULES_128M, OBJECTS_1G, OBJECTS_128M, - SQLITE_1G, SQLITE_TINY, TreeSpec, -}; -use uuid::Uuid; - -/// The labels of the blob storage calls of the benchmark itself. -const TARGET_LABEL: &str = "filesystem_snapshot_benchmark"; - -/// The namespace of the UUIDs of the environments of the repositories. -const REPOSITORY_ENVIRONMENTS: Uuid = Uuid::from_u128(0x6f1c_5d2e_9a4b_4c3d_8e7f_0a1b_2c3d_4e5f); - -/// The names of the snapshots of the base scenario. -const COLD_SAVE: &str = "cold-save"; -const WARM_SAVE: &str = "warm-save"; - -/// A scenario: its trees and the phases that run on each tree, in order. -struct Scenario { - name: &'static str, - trees: &'static [TreeSpec], - phases: &'static [Phase], - /// The phases that run under the lower memory limit of the workflow. - memory_limited_phases: &'static [&'static str], -} - -/// A phase of a scenario: the work of one pod. -struct Phase { - name: &'static str, - kind: PhaseKind, - variant: &'static Variant, -} - -/// The repository that a phase uses, and the settings of its repositories and its saves. -#[derive(Debug)] -struct Variant { - /// The agent whose repository the phase uses. - agent: &'static str, - /// The phase whose result holds the hash that a restore of the phase compares with. - save_phase: &'static str, - /// The settings of the repositories and the saves of the phase. `None` is the defaults, and - /// then the steps of the phase record no settings. - settings: Option, -} - -impl Variant { - const fn new(agent: &'static str, save_phase: &'static str, settings: Settings) -> Self { - Self { - agent, - save_phase, - settings: Some(settings), - } - } -} - -/// The settings of a repository that a save makes, and of each save. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -struct Settings { - repository: RepositorySettings, - save: SaveSettings, -} - -impl Settings { - const DEFAULT: Self = Self { - repository: RepositorySettings::DEFAULT, - save: SaveSettings::DEFAULT, - }; - - /// The defaults, with the number of threads of each stage of a save. - const fn save_threads(threads: usize) -> Self { - Self { - save: SaveSettings { - threads: NonZeroUsize::new(threads), - ..SaveSettings::DEFAULT - }, - ..Self::DEFAULT - } - } - - /// The defaults, with the settings of the repository. - const fn repository(repository: RepositorySettings) -> Self { - Self { - repository, - ..Self::DEFAULT - } - } -} - -/// The variant of the phases of the earlier scenarios: the repository of the first agent, the -/// defaults, and no settings in the steps. -const BASE: Variant = Variant { - agent: FIRST_AGENT, - save_phase: "save", - settings: None, -}; - -/// The variant of most phases of the later scenarios: the repository of the first agent and the -/// defaults, which the steps record. -const DEFAULTS: Variant = Variant::new(FIRST_AGENT, "save", Settings::DEFAULT); - -/// What a phase does. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum PhaseKind { - /// A cold save of a new tree, a small change and a warm save, into the repository of the - /// first agent. - Save, - /// A cold restore of the warm save of the first agent with the number of reader threads. - /// `None` is the default of rustic. - Restore { - reader_threads: Option, - }, - /// The restores of the warm save into the repositories of `agents` agents at the same time. - ConcurrentRestore { - agents: usize, - reader_threads: Option, - }, - /// The cold saves of `agents` agents at the same time, a small change of the tree of each - /// agent, and then the warm saves of the agents at the same time. The repository of each agent - /// has the number of the agent after `x-`. - ConcurrentSave { agents: usize }, - /// The phase of [`PhaseKind::ConcurrentSave`], in which the repository of each agent has the - /// number of the agent after the name of the phase and `-`. - NamedConcurrentSave { agents: usize }, - /// A reflink capture of a new tree, a cold save, and two warm saves of a later capture, one - /// with each change detection (see [`capture`]). - Capture, - /// Twelve saves with a different change each, a forget of all but the two newest snapshots, - /// and the open of the repository after 1, 11 and 12 saves (see [`history`]). - History, - /// A copy of the repository of the history phase, two prunes with no grace period, an open - /// and a restore (see [`history`]). - Prune { fast_repack: bool }, - /// Saves after a clustered and after a scattered change of a SQLite tree (see [`sqlite`]). - SqliteChanges, - /// The cold saves of `saves` agents and the cold restores of `restores` agents, all at the - /// same time, with the number of reader threads of each restore. - Mixed { - saves: usize, - restores: usize, - reader_threads: Option, - }, - /// A copy of the repository of the save phase to another agent, and its deletion (see - /// [`scopes`]). - Scopes, - /// The phase of [`PhaseKind::Save`], and then an open of the repository, whose record gives - /// the settings that the repository has. - SaveAndOpen, -} - -const SAVE: Phase = Phase { - name: "save", - kind: PhaseKind::Save, - variant: &BASE, -}; - -const fn restore(name: &'static str, reader_threads: usize) -> Phase { - Phase { - name, - kind: PhaseKind::Restore { - reader_threads: NonZeroUsize::new(reader_threads), - }, - variant: &BASE, - } -} - -const fn concurrent_restore(name: &'static str, agents: usize, reader_threads: usize) -> Phase { - Phase { - name, - kind: PhaseKind::ConcurrentRestore { - agents, - reader_threads: NonZeroUsize::new(reader_threads), - }, - variant: &BASE, - } -} - -const fn concurrent_save(name: &'static str, agents: usize) -> Phase { - Phase { - name, - kind: PhaseKind::ConcurrentSave { agents }, - variant: &BASE, - } -} - -const BASE_PHASES: &[Phase] = &[ - SAVE, - Phase { - name: "restore", - kind: PhaseKind::Restore { - reader_threads: None, - }, - variant: &BASE, - }, -]; - -const RESTORE_THREADS_PHASES: &[Phase] = &[ - SAVE, - restore("restore-1", 1), - restore("restore-2", 2), - restore("restore-4", 4), - restore("restore-8", 8), - restore("restore-20", 20), -]; - -const CONCURRENT_RESTORE_1_PHASES: &[Phase] = &[ - SAVE, - concurrent_restore("restore-x1", 1, 1), - concurrent_restore("restore-x5", 5, 1), - concurrent_restore("restore-x10", 10, 1), - concurrent_restore("restore-x25", 25, 1), - concurrent_restore("restore-x50", 50, 1), - concurrent_restore("restore-x100", 100, 1), - concurrent_restore("restore-x200", 200, 1), -]; - -const CONCURRENT_RESTORE_4_PHASES: &[Phase] = &[ - SAVE, - concurrent_restore("restore-x1", 1, 4), - concurrent_restore("restore-x5", 5, 4), - concurrent_restore("restore-x10", 10, 4), - concurrent_restore("restore-x25", 25, 4), - concurrent_restore("restore-x50", 50, 4), - concurrent_restore("restore-x100", 100, 4), - concurrent_restore("restore-x200", 200, 4), -]; - -const CONCURRENT_RESTORE_20_PHASES: &[Phase] = &[ - SAVE, - concurrent_restore("restore-x1", 1, 20), - concurrent_restore("restore-x5", 5, 20), - concurrent_restore("restore-x10", 10, 20), - concurrent_restore("restore-x25", 25, 20), - concurrent_restore("restore-x50", 50, 20), - concurrent_restore("restore-x100", 100, 20), - concurrent_restore("restore-x200", 200, 20), -]; - -const CONCURRENT_SAVE_PHASES: &[Phase] = &[ - concurrent_save("save-x1", 1), - concurrent_save("save-x2", 2), - concurrent_save("save-x4", 4), - concurrent_save("save-x8", 8), -]; - -/// A phase of a later scenario with the defaults, which its steps record. -const fn with_defaults(name: &'static str, kind: PhaseKind) -> Phase { - Phase { - name, - kind, - variant: &DEFAULTS, - } -} - -/// A save phase of a later scenario: a cold save, a small change and a warm save into the -/// repository of the variant, and an open of the repository. -const fn save_phase(name: &'static str, variant: &'static Variant) -> Phase { - Phase { - name, - kind: PhaseKind::SaveAndOpen, - variant, - } -} - -const DEFAULT_PHASES: &[Phase] = &[ - with_defaults("save", PhaseKind::Save), - with_defaults( - "restore", - PhaseKind::Restore { - reader_threads: None, - }, - ), -]; - -const CAPTURE_PHASES: &[Phase] = &[with_defaults("capture", PhaseKind::Capture)]; - -/// The phase of the saves that the prune phases prune. -const HISTORY: Phase = with_defaults("history", PhaseKind::History); - -/// A prune phase, which prunes a copy of the repository of the history phase in the repository of -/// the agent with the name of the phase. -const fn prune_phase(name: &'static str, variant: &'static Variant, fast_repack: bool) -> Phase { - Phase { - name, - kind: PhaseKind::Prune { fast_repack }, - variant, - } -} - -const PRUNE: Variant = Variant::new("prune", "history", Settings::DEFAULT); -const PRUNE_FAST_REPACK: Variant = Variant::new("prune-fast-repack", "history", Settings::DEFAULT); - -const PRUNE_PHASES: &[Phase] = &[ - HISTORY, - prune_phase("prune", &PRUNE, false), - prune_phase("prune-fast-repack", &PRUNE_FAST_REPACK, true), -]; - -const REPOSITORY_OPEN_PHASES: &[Phase] = &[HISTORY, prune_phase("prune", &PRUNE, false)]; - -/// A phase of 4 concurrent saves whose repositories have the name of the phase as prefix. -const fn save_threads_phase(name: &'static str, variant: &'static Variant) -> Phase { - Phase { - name, - kind: PhaseKind::NamedConcurrentSave { agents: 4 }, - variant, - } -} - -const SAVE_THREADS_1: Variant = Variant::new(FIRST_AGENT, "save", Settings::save_threads(1)); -const SAVE_THREADS_2: Variant = Variant::new(FIRST_AGENT, "save", Settings::save_threads(2)); -const SAVE_THREADS_4: Variant = Variant::new(FIRST_AGENT, "save", Settings::save_threads(4)); - -const SAVE_THREADS_PHASES: &[Phase] = &[ - save_threads_phase("save-x4-t1", &SAVE_THREADS_1), - save_threads_phase("save-x4-t2", &SAVE_THREADS_2), - save_threads_phase("save-x4-t4", &SAVE_THREADS_4), - save_threads_phase("save-x4-tdefault", &DEFAULTS), -]; - -/// The number of saves and of restores of a mixed phase. -const MIXED_SAVES: usize = 4; -const MIXED_RESTORES: usize = 4; - -/// A mixed phase: the saves have the save threads of the variant, and the restores have the -/// reader threads. -const fn mixed_phase( - name: &'static str, - variant: &'static Variant, - reader_threads: usize, -) -> Phase { - Phase { - name, - kind: PhaseKind::Mixed { - saves: MIXED_SAVES, - restores: MIXED_RESTORES, - reader_threads: NonZeroUsize::new(reader_threads), - }, - variant, - } -} - -const MIXED_PHASES: &[Phase] = &[ - with_defaults("save", PhaseKind::Save), - mixed_phase("mixed-s2-r2", &SAVE_THREADS_2, 2), - mixed_phase("mixed-s2-r4", &SAVE_THREADS_2, 4), - mixed_phase("mixed-s4-r2", &SAVE_THREADS_4, 2), - mixed_phase("mixed-s4-r4", &SAVE_THREADS_4, 4), -]; - -/// The settings of a repository with fixed chunks of 64 KiB, which are 16 SQLite pages. -const FIXED_64K: RepositorySettings = RepositorySettings { - chunking: match NonZeroU32::new(64 * 1024) { - Some(size) => Chunking::Fixed(size), - None => Chunking::Rabin, - }, - ..RepositorySettings::DEFAULT -}; - -const SQLITE_RABIN: Variant = Variant::new("rabin", "save-rabin", Settings::DEFAULT); -const SQLITE_FIXED_64K: Variant = Variant::new( - "fixed-64k", - "save-fixed-64k", - Settings::repository(FIXED_64K), -); - -const SQLITE_CHANGES_PHASES: &[Phase] = &[ - Phase { - name: "save-rabin", - kind: PhaseKind::SqliteChanges, - variant: &SQLITE_RABIN, - }, - Phase { - name: "restore-rabin", - kind: PhaseKind::Restore { - reader_threads: None, - }, - variant: &SQLITE_RABIN, - }, - Phase { - name: "save-fixed-64k", - kind: PhaseKind::SqliteChanges, - variant: &SQLITE_FIXED_64K, - }, - Phase { - name: "restore-fixed-64k", - kind: PhaseKind::Restore { - reader_threads: None, - }, - variant: &SQLITE_FIXED_64K, - }, -]; - -/// The compression with the zstd level, or no compression for the level 0. -const fn zstd_level(level: i32) -> Compression { - match NonZeroI32::new(level) { - Some(level) => Compression::Level(level), - None => Compression::Off, - } -} - -const CPU_DEFAULT: Variant = Variant::new("default", "save-default", Settings::DEFAULT); -const CPU_VERIFY_OFF: Variant = Variant::new( - "verify-off", - "save-verify-off", - Settings::repository(RepositorySettings { - extra_verify: false, - ..RepositorySettings::DEFAULT - }), -); -const CPU_ZSTD_OFF: Variant = Variant::new( - "zstd-off", - "save-zstd-off", - Settings::repository(RepositorySettings { - compression: Compression::Off, - ..RepositorySettings::DEFAULT - }), -); -const CPU_ZSTD_1: Variant = Variant::new( - "zstd-1", - "save-zstd-1", - Settings::repository(RepositorySettings { - compression: zstd_level(1), - ..RepositorySettings::DEFAULT - }), -); -const CPU_ZSTD_9: Variant = Variant::new( - "zstd-9", - "save-zstd-9", - Settings::repository(RepositorySettings { - compression: zstd_level(9), - ..RepositorySettings::DEFAULT - }), -); - -const CPU_OPTIONS_PHASES: &[Phase] = &[ - save_phase("save-default", &CPU_DEFAULT), - save_phase("save-verify-off", &CPU_VERIFY_OFF), - save_phase("save-zstd-off", &CPU_ZSTD_OFF), - save_phase("save-zstd-1", &CPU_ZSTD_1), - save_phase("save-zstd-9", &CPU_ZSTD_9), -]; - -const SCOPES_PHASES: &[Phase] = &[ - with_defaults("save", PhaseKind::Save), - with_defaults("scopes", PhaseKind::Scopes), -]; - -/// The five trees of the base scenario. -const BASE_TREES: &[TreeSpec] = &[FILES_128M, FILES_1G, SQLITE_1G, OBJECTS_128M, OBJECTS_1G]; - -/// The 1 GiB trees of the prune and repository open scenarios. -const HISTORY_TREES: &[TreeSpec] = &[FILES_1G, SQLITE_1G, OBJECTS_1G]; - -/// The scenarios of the benchmark. -const SCENARIOS: &[Scenario] = &[ - Scenario { - name: "base", - trees: &[FILES_128M, FILES_1G, SQLITE_1G, OBJECTS_128M, OBJECTS_1G], - phases: BASE_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "smoke", - trees: &[FILES_TINY, SQLITE_TINY], - phases: BASE_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "restore-threads", - trees: &[FILES_1G], - phases: RESTORE_THREADS_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "memory-pressure", - trees: &[FILES_1G], - phases: BASE_PHASES, - memory_limited_phases: &["restore"], - }, - Scenario { - name: "concurrent-restore-1", - trees: &[FILES_128M, FILES_1G], - phases: CONCURRENT_RESTORE_1_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "concurrent-restore-4", - trees: &[FILES_128M, FILES_1G], - phases: CONCURRENT_RESTORE_4_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "concurrent-restore-20", - trees: &[FILES_128M, FILES_1G], - phases: CONCURRENT_RESTORE_20_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "concurrent-save", - trees: &[FILES_128M, FILES_1G, SQLITE_1G], - phases: CONCURRENT_SAVE_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "capture", - trees: BASE_TREES, - phases: CAPTURE_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "prune", - trees: HISTORY_TREES, - phases: PRUNE_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "save-threads", - trees: &[FILES_1G, SQLITE_1G], - phases: SAVE_THREADS_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "mixed", - trees: &[FILES_1G], - phases: MIXED_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "repository-open", - trees: HISTORY_TREES, - phases: REPOSITORY_OPEN_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "sqlite-changes", - trees: &[SQLITE_1G], - phases: SQLITE_CHANGES_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "tree-shape", - trees: &[FILES_128M, MODULES_128M], - phases: DEFAULT_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "cpu-options", - trees: &[COMPRESSIBLE_1G], - phases: CPU_OPTIONS_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "scopes", - trees: BASE_TREES, - phases: SCOPES_PHASES, - memory_limited_phases: &[], - }, -]; - -/// One line of a plan: the phases of one tree of a scenario, in order. -/// -/// `memory_limited_phases` names the phases that run under the lower memory limit of the -/// workflow. A line without such phases does not have the field. -#[derive(Clone, Debug, PartialEq, Eq, Serialize)] -struct PlanEntry { - scenario: &'static str, - tree: &'static str, - phases: Box<[&'static str]>, - #[serde(skip_serializing_if = "<[_]>::is_empty")] - memory_limited_phases: Box<[&'static str]>, -} - -/// Gives the plan of the scenarios with the names, or the first name that no scenario has. -fn plan(names: &[String]) -> Result, String> { - names - .iter() - .try_fold(Vec::new(), |mut entries, name| { - let scenario = scenario(name).ok_or_else(|| name.clone())?; - entries.extend(scenario.trees.iter().map(|tree| PlanEntry { - scenario: scenario.name, - tree: tree.name, - phases: scenario.phases.iter().map(|phase| phase.name).collect(), - memory_limited_phases: scenario.memory_limited_phases.into(), - })); - Ok(entries) - }) - .map(Vec::into_boxed_slice) -} - -fn scenario(name: &str) -> Option<&'static Scenario> { - SCENARIOS.iter().find(|scenario| scenario.name == name) -} - -/// The phase of one pod, found by its names. -#[derive(Clone, Copy)] -struct Selection { - scenario: &'static Scenario, - tree: &'static TreeSpec, - phase: &'static Phase, -} - -impl Selection { - /// Gives the phase with the names, or an error that says which name is unknown. - fn find(scenario: &str, tree: &str, phase: &str) -> Result { - let found = self::scenario(scenario) - .ok_or_else(|| format!("no scenario has the name {scenario:?}"))?; - Ok(Self { - scenario: found, - tree: found - .trees - .iter() - .find(|spec| spec.name == tree) - .ok_or_else(|| format!("the scenario {scenario:?} has no tree {tree:?}"))?, - phase: found - .phases - .iter() - .find(|spec| spec.name == phase) - .ok_or_else(|| format!("the scenario {scenario:?} has no phase {phase:?}"))?, - }) - } -} - -/// Tells whether the text can be a run id or a CPU setting: 1 to 64 ASCII letters, digits, `-` -/// or `_`. Each is a segment of an object key. -fn is_key_segment(text: &str) -> bool { - (1..=64).contains(&text.len()) - && text - .bytes() - .all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'_')) -} - -/// Gives the repository key of the run. The data of a run is synthetic and the workflow deletes -/// it after the run, so each pod of the run derives the same key from the run id. -fn repository_key(run_id: &str) -> RepositoryKey { - let mut bytes = [0; 64]; - blake3::Hasher::new_derive_key("golem fs-snapshot benchmark repository key") - .update(run_id.as_bytes()) - .finalize_xof() - .fill(&mut bytes); - RepositoryKey::new(bytes) -} - -/// Gives the namespace of the results of a run. -fn results_namespace() -> BlobStorageNamespace { - BlobStorageNamespace::InitialAgentFiles { - environment_id: EnvironmentId(Uuid::nil()), - } -} - -/// Gives the path of the result of a phase in the namespace of the results. -fn result_path(scenario: &str, cpu_setting: &str, tree: &str, phase: &str) -> Box { - PathBuf::from(format!( - "results/{scenario}/{cpu_setting}/{tree}/{phase}.json" - )) - .into_boxed_path() -} - -/// Gives the scope of the repository of a scenario, a CPU setting and a tree. -fn repository_scope(scenario: &str, cpu_setting: &str, tree: &str) -> SnapshotScope { - SnapshotScope(BlobStorageNamespace::InitialAgentFiles { - environment_id: EnvironmentId(Uuid::new_v5( - &REPOSITORY_ENVIRONMENTS, - format!("{scenario}/{cpu_setting}/{tree}").as_bytes(), - )), - }) -} - -/// What a phase gets. -struct PhaseContext { - run_id: Box, - cpu_setting: Box, - selection: Selection, - work_dir: Box, - storage: Arc, -} - -impl PhaseContext { - /// Gives the scope of the repositories of the agents of the phase. - fn scope(&self) -> SnapshotScope { - repository_scope( - self.selection.scenario.name, - &self.cpu_setting, - self.selection.tree.name, - ) - } - - fn variant(&self) -> &'static Variant { - self.selection.phase.variant - } - - fn settings(&self) -> Settings { - self.variant().settings.unwrap_or(Settings::DEFAULT) - } - - /// Gives the settings of each save of the phase. - fn save_settings(&self) -> SaveSettings { - self.settings().save - } - - /// Gives the repository of the agent, which a save makes with the settings of the phase. - fn agent_repository(&self, agent: &str) -> Repository { - Repository::new( - Arc::new(AgentStorage::new(self.storage.clone(), agent)), - self.scope(), - repository_key(&self.run_id), - STORAGE_CALL_DEADLINE, - ) - .with_settings(self.settings().repository) - } - - /// Gives the repository of the agent of the variant of the phase. - fn repository(&self) -> Repository { - self.agent_repository(self.variant().agent) - } - - fn result_path(&self, phase: &str) -> Box { - result_path( - self.selection.scenario.name, - &self.cpu_setting, - self.selection.tree.name, - phase, - ) - } -} - -/// What a phase gives. -struct PhaseOutcome { - tree_facts: TreeFacts, - steps: Vec, - outcome: Outcome, -} - -/// Runs one phase on the storage, writes its result into the storage, and gives the result and -/// whether the write succeeded. -/// -/// `environment` is recorded as it is, with the time of a first request of the storage, which -/// gets the credentials and a connection as a running executor already has them. The counters of -/// the cgroup `memory.events` at the start and at the end of the phase go into its `cgroup` -/// object. -async fn run_phase( - run_id: &str, - cpu_setting: &str, - selection: Selection, - work_dir: &Path, - storage: Arc, - environment: Value, -) -> (PhaseResult, anyhow::Result<()>) { - let environment = - with_cgroup_value(environment, "memory_events_start", measure::memory_events()); - let context = PhaseContext { - run_id: run_id.into(), - cpu_setting: cpu_setting.into(), - selection, - work_dir: work_dir.into(), - storage: Arc::new(MeasuredBlobStorage::new(storage)), - }; - let volume = volume::check(work_dir); - let started = Instant::now(); - let warm_up = context - .storage - .get_metadata( - TARGET_LABEL, - "warm_up", - results_namespace(), - Path::new("warm-up"), - ) - .await; - let environment = with_warm_up(environment, millis(started.elapsed()), warm_up.err()); - let outcome = run_kind(&context, selection.phase.kind).await; - let steps = outcome - .steps - .into_iter() - .map(|step| with_settings(step, selection.phase.variant)) - .collect::>(); - let environment = with_cgroup_value(environment, "memory_events_end", measure::memory_events()); - let result = PhaseResult { - format: FORMAT, - run_id: run_id.into(), - scenario: selection.scenario.name, - phase: selection.phase.name, - tree: selection.tree.name, - cpu_setting: cpu_setting.into(), - environment, - volume, - tree_facts: outcome.tree_facts, - steps, - outcome: outcome.outcome, - }; - let written = write_result(&context, &result).await; - (result, written) -} - -/// Runs the work of the kind of phase. -async fn run_kind(context: &PhaseContext, kind: PhaseKind) -> PhaseOutcome { - match kind { - PhaseKind::Save => base_save(context).await, - PhaseKind::Restore { reader_threads } => base_restore(context, reader_threads).await, - PhaseKind::ConcurrentRestore { - agents, - reader_threads, - } => concurrent::concurrent_restore(context, agents, reader_threads).await, - PhaseKind::ConcurrentSave { agents } => { - concurrent::concurrent_save(context, agents, &format!("x{agents}")).await - } - PhaseKind::NamedConcurrentSave { agents } => { - concurrent::concurrent_save(context, agents, context.selection.phase.name).await - } - PhaseKind::Capture => capture::capture(context).await, - PhaseKind::History => history::history(context).await, - PhaseKind::Prune { fast_repack } => history::prune(context, fast_repack).await, - PhaseKind::SqliteChanges => with_open(context, sqlite::sqlite_changes(context).await).await, - PhaseKind::Mixed { - saves, - restores, - reader_threads, - } => concurrent::mixed(context, saves, restores, reader_threads).await, - PhaseKind::Scopes => scopes::scopes(context).await, - PhaseKind::SaveAndOpen => with_open(context, base_save(context).await).await, - } -} - -/// Gives the outcome of a save phase with an open of the repository of the phase after it. The -/// open finds the warm save, and its record gives the settings that the repository has. A save -/// phase that failed gets a skipped open, and an open that fails or finds no repository fails the -/// phase. -async fn with_open(context: &PhaseContext, saved: PhaseOutcome) -> PhaseOutcome { - if saved.outcome != Outcome::Ok { - return PhaseOutcome { - steps: saved - .steps - .into_iter() - .chain(std::iter::once(StepRecord::skipped("open"))) - .collect(), - ..saved - }; - } - let repository = context.repository(); - let (record, inspected) = measure("open", &context.storage, async { - repository.inspect(&snapshot_name(WARM_SAVE)?).await - }) - .await; - let mut steps = saved.steps; - steps.push(inspect_record(record, &inspected)); - if matches!(inspected, Ok(Some(_))) { - PhaseOutcome { steps, ..saved } - } else { - failed(saved.tree_facts, steps, "open", &[]) - } -} - -/// Gives the record of an open with what it found: the number of snapshots, whether a snapshot -/// has the name, and the settings of the repository, as its config file gives them. -fn inspect_record( - record: StepRecord, - inspected: &anyhow::Result>, -) -> StepRecord { - match inspected { - Ok(Some(report)) => record.with_details( - json!({ - "snapshots": report.snapshots, - "found": report.found, - "settings": repository_parameters(&report.settings), - }), - phase_walls(&report.phases), - ), - Ok(None) => record.with_details(json!({ "repository": null }), Box::default()), - Err(_) => record, - } -} - -/// Gives the step with the settings of the variant in its parameters. A parameter that the step -/// already has stays. A variant without settings leaves the step as it is. -fn with_settings(step: StepRecord, variant: &Variant) -> StepRecord { - match (variant.settings, step.parameters.clone()) { - (Some(settings), Value::Object(parameters)) => { - let merged = settings_parameters(&settings) - .into_iter() - .chain(parameters) - .collect::>(); - step.with_parameters(Value::Object(merged)) - } - _ => step, - } -} - -/// Gives the settings as step parameters. A setting that is not set is `null`. -fn settings_parameters(settings: &Settings) -> Map { - [ - ("save_threads", json!(settings.save.threads)), - ( - "change_detection", - json!(change_detection_name(settings.save.detection)), - ), - ] - .into_iter() - .map(|(key, value)| (key.to_string(), value)) - .chain(repository_parameters(&settings.repository)) - .collect() -} - -/// Gives the settings of a repository as step parameters. A setting that is not set is `null`: -/// the default compression is the zstd level that rustic chooses when the config file has none. -fn repository_parameters(repository: &RepositorySettings) -> Map { - [ - ( - "chunker", - match repository.chunking { - Chunking::Rabin => json!("rabin"), - Chunking::Fixed(size) => json!(format!("fixed-{size}")), - }, - ), - ( - "compression", - match repository.compression { - Compression::Default => Value::Null, - Compression::Off => json!("off"), - Compression::Level(level) => json!(level.get()), - }, - ), - ("extra_verify", json!(repository.extra_verify)), - ] - .into_iter() - .map(|(key, value)| (key.to_string(), value)) - .collect() -} - -fn change_detection_name(detection: ChangeDetection) -> &'static str { - match detection { - ChangeDetection::Ctime => "ctime", - ChangeDetection::SizeMtime => "size-mtime", - } -} - -/// Gives the environment with the value at the key in its `cgroup` object. An environment that -/// is an object without a `cgroup` object gets one. -fn with_cgroup_value(environment: Value, key: &str, value: Value) -> Value { - match environment { - Value::Object(mut fields) => { - let cgroup = fields - .entry("cgroup") - .or_insert_with(|| Value::Object(Default::default())); - if let Value::Object(cgroup) = cgroup { - cgroup.insert(key.to_string(), value); - } - Value::Object(fields) - } - other => other, - } -} - -fn with_warm_up(environment: Value, warm_up_ms: f64, error: Option) -> Value { - match environment { - Value::Object(mut fields) => { - fields.insert("warm_up_ms".to_string(), json!(warm_up_ms)); - fields.insert( - "warm_up_error".to_string(), - json!(error.map(|error| format!("{error:#}"))), - ); - Value::Object(fields) - } - other => other, - } -} - -async fn write_result(context: &PhaseContext, result: &PhaseResult) -> anyhow::Result<()> { - let json = serde_json::to_vec(result)?; - context - .storage - .put_raw( - TARGET_LABEL, - "result", - results_namespace(), - &context.result_path(context.selection.phase.name), - &json, - ) - .await -} - -/// Gives the times of the parts of an operation as the phases of a step. -fn phase_walls(phases: &[PhaseTime]) -> Box<[PhaseWall]> { - phases - .iter() - .map(|time| PhaseWall { - name: phase_name(time.phase), - wall_ms: millis(time.wall), - }) - .collect() -} - -fn phase_name(phase: super::rustic::OperationPhase) -> &'static str { - use super::rustic::OperationPhase; - match phase { - OperationPhase::Create => "create", - OperationPhase::Open => "open", - OperationPhase::Lookup => "lookup", - OperationPhase::IndexLoad => "index_load", - OperationPhase::Backup => "backup", - OperationPhase::RestorePlan => "restore_plan", - OperationPhase::Restore => "restore", - OperationPhase::PrunePlan => "prune_plan", - OperationPhase::Prune => "prune", - } -} - -fn snapshot_name(text: &str) -> anyhow::Result { - Ok(SnapshotName::new(text)?) -} - -/// Gives the outcome of a phase whose step failed: the records so far, and a skipped record for -/// each step that did not run. -fn failed( - tree_facts: TreeFacts, - steps: Vec, - failed_step: &'static str, - skipped: &[&'static str], -) -> PhaseOutcome { - PhaseOutcome { - tree_facts, - steps: steps - .into_iter() - .chain(skipped.iter().copied().map(StepRecord::skipped)) - .collect(), - outcome: Outcome::Failed { - reason: format!("the step {failed_step} failed").into(), - }, - } -} - -/// The save phase of the base scenario: a cold save of a new tree, a small change of the tree, a -/// warm save, and the hash of the changed tree. -/// -/// `generate_tree` and `small_change` remove the pages of the tree from the page cache. No step -/// reads the content of a file between one of them and the save after it, so each save reads the -/// tree from the volume. The hash of the new tree is read after the cold save, and the hash of the -/// changed tree after the warm save. A save does not change the tree. -async fn base_save(context: &PhaseContext) -> PhaseOutcome { - let spec = context.selection.tree; - let tree = context.work_dir.join("tree"); - let repository = context.repository(); - let facts = TreeFacts { - name: spec.name, - content: Some(spec.content.label()), - page_cache: Some("dropped"), - ..TreeFacts::default() - }; - let storage = &context.storage; - let settings = context.save_settings(); - - let (record, generated) = measure("generate_tree", storage, trees::generate(spec, &tree)).await; - let mut steps = vec![record]; - let Ok(counts) = generated else { - return failed( - facts, - steps, - "generate_tree", - &["cold_save", "small_change", "warm_save", "hash_tree"], - ); - }; - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - ..facts - }; - - let (record, cold) = measure("cold_save", storage, async { - repository - .save_with(&snapshot_name(COLD_SAVE)?, &tree, settings) - .await - }) - .await; - steps.push(save_record(record, &cold)); - if cold.is_err() { - return failed( - facts, - steps, - "cold_save", - &["small_change", "warm_save", "hash_tree"], - ); - } - let facts = TreeFacts { - hash: trees::hash(&tree).await.ok().map(|(hash, _)| hash), - ..facts - }; - - let (record, changed) = measure("small_change", storage, trees::change(spec, &tree)).await; - steps.push(record.with_details( - changed.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )); - let Ok(change) = changed else { - return failed(facts, steps, "small_change", &["warm_save", "hash_tree"]); - }; - let facts = TreeFacts { - change: Some(change), - ..facts - }; - - let (record, warm) = measure("warm_save", storage, async { - repository - .save_with(&snapshot_name(WARM_SAVE)?, &tree, settings) - .await - }) - .await; - steps.push(save_record(record, &warm)); - if warm.is_err() { - return failed(facts, steps, "warm_save", &["hash_tree"]); - } - - let (record, hashed) = measure("hash_tree", storage, trees::hash(&tree)).await; - steps.push(record); - let Ok((hash, _)) = hashed else { - return failed(facts, steps, "hash_tree", &[]); - }; - PhaseOutcome { - tree_facts: TreeFacts { - hash_after_change: Some(hash), - ..facts - }, - steps, - outcome: Outcome::Ok, - } -} - -fn save_record(record: StepRecord, save: &anyhow::Result) -> StepRecord { - match save { - Ok(report) => record.with_details( - json!({ - "snapshot": report.snapshot, - "parent": report.parent, - "files_new": report.files_new, - "files_changed": report.files_changed, - "files_unmodified": report.files_unmodified, - "dirs_new": report.dirs_new, - "dirs_changed": report.dirs_changed, - "dirs_unmodified": report.dirs_unmodified, - "bytes_processed": report.bytes_processed, - "data_added": report.data_added, - "data_added_packed": report.data_added_packed, - "data_blobs": report.data_blobs, - "tree_blobs": report.tree_blobs, - }), - phase_walls(&report.phases), - ), - Err(_) => record, - } -} - -/// The restore phase of the base scenario: a cold restore of the warm save into an empty -/// directory, and a comparison of the hash of the restored tree with the hash that the save -/// phase recorded. -/// -/// `reader_threads` is the number of threads of the restore that read data, and the -/// `parameters` of the restore step record it. `None` is the default of rustic. -async fn base_restore( - context: &PhaseContext, - reader_threads: Option, -) -> PhaseOutcome { - let spec = context.selection.tree; - let into = context.work_dir.join("restore"); - let repository = context.repository(); - let storage = &context.storage; - let facts = TreeFacts { - name: spec.name, - ..TreeFacts::default() - }; - let expected = match saved_hash(context).await { - Ok(expected) => expected, - Err(error) => return without_save(facts, &error, &["cold_restore", "hash_tree"]), - }; - - let (record, restored) = measure("cold_restore", storage, async { - std::fs::create_dir(&into)?; - repository - .restore(&snapshot_name(WARM_SAVE)?, &into, reader_threads) - .await? - .ok_or_else(|| anyhow::anyhow!("no snapshot has the name {WARM_SAVE}")) - }) - .await; - let record = record.with_parameters(json!({ "reader_threads": reader_threads })); - let record = match &restored { - Ok(report) => record.with_details( - json!({ "files": report.files, "dirs": report.dirs, "bytes": report.bytes }), - phase_walls(&report.phases), - ), - Err(_) => record, - }; - let mut steps = vec![record]; - if restored.is_err() { - return failed(facts, steps, "cold_restore", &["hash_tree"]); - } - - let (record, hashed) = measure("hash_tree", storage, trees::hash(&into)).await; - let Ok((hash, counts)) = hashed else { - steps.push(record); - return failed(facts, steps, "hash_tree", &[]); - }; - let matches = *hash == *expected; - steps.push(record.with_details( - json!({ "hash": hash, "expected": expected, "matches": matches }), - Box::default(), - )); - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - hash: Some(hash), - ..facts - }; - PhaseOutcome { - tree_facts: facts, - steps, - outcome: if matches { - Outcome::Ok - } else { - Outcome::Failed { - reason: "the restored tree differs from the saved tree".into(), - } - }, - } -} - -/// Gives the outcome of a restore phase without a result of the save phase: a skipped record for -/// each step. -fn without_save( - tree_facts: TreeFacts, - error: &anyhow::Error, - skipped: &[&'static str], -) -> PhaseOutcome { - PhaseOutcome { - tree_facts, - steps: skipped.iter().copied().map(StepRecord::skipped).collect(), - outcome: Outcome::Failed { - reason: format!("the save phase gave no result to compare with: {error:#}").into(), - }, - } -} - -/// Reads the hash after the change from the result of the save phase of the variant of the -/// phase. -async fn saved_hash(context: &PhaseContext) -> anyhow::Result> { - let saved = context - .storage - .get_raw( - TARGET_LABEL, - "result", - results_namespace(), - &context.result_path(context.variant().save_phase), - ) - .await? - .ok_or_else(|| anyhow::anyhow!("the save phase wrote no result"))?; - let saved: Value = serde_json::from_slice(&saved)?; - if saved.pointer("/outcome/status") != Some(&json!("ok")) { - anyhow::bail!("the save phase failed"); - } - saved - .pointer("/tree_facts/hash_after_change") - .and_then(Value::as_str) - .map(Box::from) - .ok_or_else(|| anyhow::anyhow!("the result of the save phase has no hash")) -} - -#[cfg(test)] -mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/report.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/report.rs deleted file mode 100644 index 7bdeccbeab..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/report.rs +++ /dev/null @@ -1,301 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The result of one phase of the benchmark, as JSON. -//! -//! A new scenario adds steps with new names, parameters and details. It does not add or change a -//! field of these types. - -use serde::Serialize; -use serde_json::Value; -use std::time::Duration; - -/// The version of the result format. -pub(super) const FORMAT: &str = "golem-fs-snapshot-benchmark/1"; - -/// The result of one phase: the work of one pod. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct PhaseResult { - pub(super) format: &'static str, - pub(super) run_id: Box, - pub(super) scenario: &'static str, - pub(super) phase: &'static str, - pub(super) tree: &'static str, - pub(super) cpu_setting: Box, - pub(super) environment: Value, - pub(super) volume: Value, - pub(super) tree_facts: TreeFacts, - pub(super) steps: Box<[StepRecord]>, - pub(super) outcome: Outcome, -} - -/// What a phase knows about its tree. A field that the phase does not know is `null`. -#[derive(Clone, Debug, Default, PartialEq, Serialize)] -pub(super) struct TreeFacts { - pub(super) name: &'static str, - pub(super) files: Option, - pub(super) directories: Option, - pub(super) bytes: Option, - pub(super) content: Option<&'static str>, - pub(super) page_cache: Option<&'static str>, - pub(super) hash: Option>, - pub(super) hash_after_change: Option>, - pub(super) change: Option, -} - -/// The outcome of a phase. -#[derive(Clone, Debug, PartialEq, Serialize)] -#[serde(tag = "status", rename_all = "snake_case")] -pub(super) enum Outcome { - Ok, - Failed { reason: Box }, -} - -/// One measured step of a phase. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct StepRecord { - pub(super) name: &'static str, - pub(super) status: StepStatus, - pub(super) parameters: Value, - pub(super) wall_ms: Option, - pub(super) cpu: Option, - pub(super) memory: Option, - pub(super) threads: Option, - pub(super) requests: Box<[RequestSummary]>, - pub(super) bytes_written: u64, - pub(super) bytes_read: u64, - pub(super) phases: Box<[PhaseWall]>, - pub(super) details: Value, -} - -impl StepRecord { - /// Gives the record of a step that did not run, because an earlier step failed. - pub(super) fn skipped(name: &'static str) -> Self { - Self { - name, - status: StepStatus::Skipped, - parameters: Value::Object(Default::default()), - wall_ms: None, - cpu: None, - memory: None, - threads: None, - requests: Box::default(), - bytes_written: 0, - bytes_read: 0, - phases: Box::default(), - details: Value::Object(Default::default()), - } - } - - /// Gives the record with the parameters of the step. - pub(super) fn with_parameters(self, parameters: Value) -> Self { - Self { parameters, ..self } - } - - /// Gives the record with the details and the phases of the operation. - pub(super) fn with_details(self, details: Value, phases: Box<[PhaseWall]>) -> Self { - Self { - details, - phases, - ..self - } - } -} - -/// Whether a step succeeded. -#[derive(Clone, Debug, PartialEq, Serialize)] -#[serde(rename_all = "snake_case")] -pub(super) enum StepStatus { - Ok, - Error(Box), - Skipped, -} - -/// The CPU time of a step. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct CpuTime { - pub(super) user_ms: f64, - pub(super) system_ms: f64, - pub(super) cgroup: Option, -} - -/// The change of the CPU counters of the cgroup of the process during a step. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct CgroupCpuTime { - pub(super) usage_ms: f64, - pub(super) user_ms: f64, - pub(super) system_ms: f64, - pub(super) nr_periods: u64, - pub(super) nr_throttled: u64, - pub(super) throttled_ms: f64, -} - -/// The memory of a step. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct MemoryPeaks { - pub(super) rss_start_bytes: u64, - pub(super) rss_peak_bytes: u64, - pub(super) rss_peak_source: &'static str, - pub(super) cgroup_current_peak_bytes: Option, - pub(super) cgroup_anon_peak_bytes: Option, - pub(super) cgroup_file_peak_bytes: Option, -} - -/// The threads of the process during a step. The count includes the thread that samples it. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct ThreadCounts { - pub(super) start: u64, - pub(super) peak: u64, - pub(super) sample_interval_ms: u64, -} - -/// The blob storage requests of one call and one file type in a step. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct RequestSummary { - pub(super) call: &'static str, - pub(super) file_type: &'static str, - pub(super) count: u64, - pub(super) errors: u64, - pub(super) bytes: u64, - pub(super) time_ms: RequestTimes, -} - -/// The request times of one call and one file type, by the nearest rank. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct RequestTimes { - pub(super) min: f64, - pub(super) p50: f64, - pub(super) p90: f64, - pub(super) p99: f64, - pub(super) max: f64, - pub(super) total: f64, -} - -/// The time of one part of an operation. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct PhaseWall { - pub(super) name: &'static str, - pub(super) wall_ms: f64, -} - -/// Gives the duration in milliseconds, with the precision of a microsecond. -pub(super) fn millis(duration: Duration) -> f64 { - duration.as_micros() as f64 / 1_000.0 -} - -#[cfg(test)] -mod tests { - use super::{ - CgroupCpuTime, CpuTime, FORMAT, MemoryPeaks, Outcome, PhaseResult, PhaseWall, - RequestSummary, RequestTimes, StepRecord, StepStatus, ThreadCounts, TreeFacts, millis, - }; - use pretty_assertions::assert_eq; - use serde_json::json; - use std::time::Duration; - use test_r::test; - - fn result() -> PhaseResult { - let step = StepRecord { - name: "cold_save", - status: StepStatus::Ok, - parameters: json!({}), - wall_ms: Some(1500.25), - cpu: Some(CpuTime { - user_ms: 900.5, - system_ms: 100.0, - cgroup: Some(CgroupCpuTime { - usage_ms: 1000.0, - user_ms: 900.0, - system_ms: 100.0, - nr_periods: 15, - nr_throttled: 2, - throttled_ms: 30.5, - }), - }), - memory: Some(MemoryPeaks { - rss_start_bytes: 1000, - rss_peak_bytes: 5000, - rss_peak_source: "VmHWM", - cgroup_current_peak_bytes: Some(9000), - cgroup_anon_peak_bytes: Some(6000), - cgroup_file_peak_bytes: None, - }), - threads: Some(ThreadCounts { - start: 10, - peak: 40, - sample_interval_ms: 10, - }), - requests: Box::new([RequestSummary { - call: "put_raw", - file_type: "pack", - count: 3, - errors: 0, - bytes: 3000, - time_ms: RequestTimes { - min: 1.0, - p50: 2.0, - p90: 3.0, - p99: 3.0, - max: 3.0, - total: 6.0, - }, - }]), - bytes_written: 3000, - bytes_read: 0, - phases: Box::new([PhaseWall { - name: "backup", - wall_ms: 1400.0, - }]), - details: json!({ "snapshot": "abc" }), - }; - PhaseResult { - format: FORMAT, - run_id: "123-1".into(), - scenario: "base", - phase: "save", - tree: "files-128m", - cpu_setting: "limit-3".into(), - environment: json!({ "pod": "p" }), - volume: json!({ "check": { "status": "ok" } }), - tree_facts: TreeFacts { - name: "files-128m", - files: Some(10_000), - directories: Some(100), - bytes: Some(134_217_728), - content: Some("incompressible"), - page_cache: Some("dropped"), - hash: Some("h1".into()), - hash_after_change: Some("h2".into()), - change: Some(json!({ "files_rewritten": 10 })), - }, - steps: Box::new([step, StepRecord::skipped("warm_save")]), - outcome: Outcome::Failed { - reason: "the step warm_save failed".into(), - }, - } - } - - #[test] - fn the_result_format_is_stable() { - assert_eq!( - serde_json::to_string_pretty(&result()).unwrap(), - include_str!("golden/phase_result.json").trim_end() - ); - } - - #[test] - fn a_duration_is_given_in_milliseconds_with_microseconds() { - assert_eq!(millis(Duration::from_nanos(1_234_567_890)), 1234.567); - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/requests.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/requests.rs deleted file mode 100644 index 9101caedcf..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/requests.rs +++ /dev/null @@ -1,659 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! A blob storage that records each request that it passes to another blob storage. - -use super::agents::AGENTS; -use super::report::{RequestSummary, RequestTimes, millis}; -use async_trait::async_trait; -use bytes::Bytes; -use futures::stream::BoxStream; -use golem_service_base::replayable_stream::ErasedReplayableStream; -use golem_service_base::storage::blob::{ - BlobMetadata, BlobStorage, BlobStorageNamespace, ExistsResult, ListedBlob, PutIfAbsent, -}; -use std::future::Future; -use std::path::{Component, Path, PathBuf}; -use std::sync::{Arc, Mutex, PoisonError}; -use std::time::{Duration, Instant}; - -/// One request that the storage passed on. -#[derive(Clone, Debug, PartialEq)] -pub(super) struct RequestRecord { - pub(super) call: &'static str, - pub(super) file_type: &'static str, - /// The bytes that the request wrote or read. - pub(super) bytes: u64, - pub(super) written: bool, - pub(super) time: Duration, - pub(super) ok: bool, -} - -/// A blob storage that passes each call to `inner` and records it. -/// -/// A record holds the name of the call, the file type of the path, the bytes that the call wrote -/// or read, and the time from the start of the call to its end. The time includes the retries of -/// the inner storage. Each method of the trait is passed on, also a method with a default, so the -/// inner storage does each call in its own way. -#[derive(Debug)] -pub(super) struct MeasuredBlobStorage { - inner: Arc, - records: Mutex>, -} - -impl MeasuredBlobStorage { - pub(super) fn new(inner: Arc) -> Self { - Self { - inner, - records: Mutex::new(Vec::new()), - } - } - - /// Gives the records since the last call, and removes them. - pub(super) fn take(&self) -> Box<[RequestRecord]> { - std::mem::take(&mut *self.records.lock().unwrap_or_else(PoisonError::into_inner)) - .into_boxed_slice() - } - - async fn record( - &self, - call: &'static str, - path: &Path, - bytes: impl FnOnce(&T) -> (u64, bool), - future: impl Future>, - ) -> anyhow::Result { - let started = Instant::now(); - let result = future.await; - let time = started.elapsed(); - let (bytes, written) = result.as_ref().map_or((0, false), bytes); - self.records - .lock() - .unwrap_or_else(PoisonError::into_inner) - .push(RequestRecord { - call, - file_type: file_type(path), - bytes, - written, - time, - ok: result.is_ok(), - }); - result - } -} - -/// Gives the file type of a repository path from its first name: `config`, `pack`, `index`, -/// `snapshot`, `key`, or `other`. The path of the repository of an agent starts with -/// `agents/` (see [`super::agents`]), and the first name after it gives the type. -pub(super) fn file_type(path: &Path) -> &'static str { - let names = path - .components() - .filter_map(|component| match component { - Component::Normal(name) => Some(name.to_str()), - _ => None, - }) - .collect::>(); - let in_repository = match &*names { - [Some(AGENTS), Some(_), rest @ ..] => rest, - all => all, - }; - match in_repository.first() { - Some(Some("config")) => "config", - Some(Some("data")) => "pack", - Some(Some("index")) => "index", - Some(Some("snapshots")) => "snapshot", - Some(Some("keys")) => "key", - _ => "other", - } -} - -fn nothing(_: &T) -> (u64, bool) { - (0, false) -} - -fn read_bytes(data: &Option>) -> (u64, bool) { - (data.as_ref().map_or(0, |data| data.len() as u64), false) -} - -/// Gives a summary of the records for each call and file type, in the order of the call and the -/// file type. -pub(super) fn summarize(records: &[RequestRecord]) -> Box<[RequestSummary]> { - let mut sorted = records.to_vec(); - sorted.sort_by(|left, right| { - (left.call, left.file_type, left.time).cmp(&(right.call, right.file_type, right.time)) - }); - sorted - .chunk_by(|left, right| (left.call, left.file_type) == (right.call, right.file_type)) - .filter_map(|group| { - group.first().map(|first| RequestSummary { - call: first.call, - file_type: first.file_type, - count: group.len() as u64, - errors: group.iter().filter(|record| !record.ok).count() as u64, - bytes: group.iter().map(|record| record.bytes).sum(), - time_ms: times(&group.iter().map(|record| record.time).collect::>()), - }) - }) - .collect() -} - -/// Gives the minimum, the percentiles by the nearest rank, the maximum and the total of times -/// that are in ascending order. Each value of an empty list is zero. -pub(super) fn times(sorted: &[Duration]) -> RequestTimes { - let rank = |percent: usize| { - let index = (percent * sorted.len()).div_ceil(100).max(1) - 1; - sorted.get(index).copied().map(millis).unwrap_or_default() - }; - RequestTimes { - min: sorted.first().copied().map(millis).unwrap_or_default(), - p50: rank(50), - p90: rank(90), - p99: rank(99), - max: sorted.last().copied().map(millis).unwrap_or_default(), - total: millis(sorted.iter().sum()), - } -} - -/// Gives the bytes that the records wrote and the bytes that they read. -pub(super) fn written_and_read(records: &[RequestRecord]) -> (u64, u64) { - records.iter().fold((0, 0), |(written, read), record| { - if record.written { - (written + record.bytes, read) - } else { - (written, read + record.bytes) - } - }) -} - -#[async_trait] -impl BlobStorage for MeasuredBlobStorage { - async fn get_raw( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result>> { - self.record( - "get_raw", - path, - read_bytes, - self.inner.get_raw(target_label, op_label, namespace, path), - ) - .await - } - - async fn get_stream( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result>>> { - self.record( - "get_stream", - path, - nothing, - self.inner - .get_stream(target_label, op_label, namespace, path), - ) - .await - } - - async fn get_raw_slice( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - start: u64, - end: u64, - ) -> anyhow::Result>> { - self.record( - "get_raw_slice", - path, - read_bytes, - self.inner - .get_raw_slice(target_label, op_label, namespace, path, start, end), - ) - .await - } - - async fn get_metadata( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.record( - "get_metadata", - path, - nothing, - self.inner - .get_metadata(target_label, op_label, namespace, path), - ) - .await - } - - async fn put_raw( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - data: &[u8], - ) -> anyhow::Result<()> { - let length = data.len() as u64; - self.record( - "put_raw", - path, - |_| (length, true), - self.inner - .put_raw(target_label, op_label, namespace, path, data), - ) - .await - } - - async fn put_raw_if_absent( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - data: &[u8], - ) -> anyhow::Result { - let length = data.len() as u64; - self.record( - "put_raw_if_absent", - path, - |outcome| match outcome { - PutIfAbsent::Written => (length, true), - PutIfAbsent::AlreadyExists => (0, true), - }, - self.inner - .put_raw_if_absent(target_label, op_label, namespace, path, data), - ) - .await - } - - async fn put_stream( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - stream: &dyn ErasedReplayableStream>, Error = anyhow::Error>, - ) -> anyhow::Result<()> { - self.record( - "put_stream", - path, - nothing, - self.inner - .put_stream(target_label, op_label, namespace, path, stream), - ) - .await - } - - async fn delete( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result<()> { - self.record( - "delete", - path, - nothing, - self.inner.delete(target_label, op_label, namespace, path), - ) - .await - } - - async fn delete_many( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - paths: &[PathBuf], - ) -> anyhow::Result<()> { - self.record( - "delete_many", - paths.first().map_or(Path::new(""), PathBuf::as_path), - nothing, - self.inner - .delete_many(target_label, op_label, namespace, paths), - ) - .await - } - - async fn create_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result<()> { - self.record( - "create_dir", - path, - nothing, - self.inner - .create_dir(target_label, op_label, namespace, path), - ) - .await - } - - async fn list_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.record( - "list_dir", - path, - nothing, - self.inner.list_dir(target_label, op_label, namespace, path), - ) - .await - } - - async fn list_blobs_below( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.record( - "list_blobs_below", - path, - nothing, - self.inner - .list_blobs_below(target_label, op_label, namespace, path), - ) - .await - } - - async fn delete_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result { - self.record( - "delete_dir", - path, - nothing, - self.inner - .delete_dir(target_label, op_label, namespace, path), - ) - .await - } - - async fn exists( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result { - self.record( - "exists", - path, - nothing, - self.inner.exists(target_label, op_label, namespace, path), - ) - .await - } - - async fn copy( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - from: &Path, - to: &Path, - ) -> anyhow::Result<()> { - self.record( - "copy", - from, - nothing, - self.inner.copy(target_label, op_label, namespace, from, to), - ) - .await - } - - async fn r#move( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - from: &Path, - to: &Path, - ) -> anyhow::Result<()> { - self.record( - "move", - from, - nothing, - self.inner - .r#move(target_label, op_label, namespace, from, to), - ) - .await - } -} - -#[cfg(test)] -mod tests { - use super::{ - MeasuredBlobStorage, RequestRecord, file_type, summarize, times, written_and_read, - }; - use crate::filesystem_snapshot::benchmark::report::RequestTimes; - use golem_common::model::environment::EnvironmentId; - use golem_service_base::storage::blob::memory::InMemoryBlobStorage; - use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; - use pretty_assertions::assert_eq; - use std::path::Path; - use std::sync::Arc; - use std::time::Duration; - use test_r::test; - use uuid::Uuid; - - fn record(call: &'static str, file_type: &'static str, millis: u64) -> RequestRecord { - RequestRecord { - call, - file_type, - bytes: 10, - written: call == "put_raw", - time: Duration::from_millis(millis), - ok: millis != 0, - } - } - - #[test] - fn the_times_are_the_nearest_ranks_of_the_sorted_times() { - let hundred = (1..=100).map(Duration::from_millis).collect::>(); - let three = [1, 2, 3].map(Duration::from_millis); - - assert_eq!( - ( - times(&hundred), - times(&three), - times(&[Duration::from_millis(7)]) - ), - ( - RequestTimes { - min: 1.0, - p50: 50.0, - p90: 90.0, - p99: 99.0, - max: 100.0, - total: 5050.0, - }, - RequestTimes { - min: 1.0, - p50: 2.0, - p90: 3.0, - p99: 3.0, - max: 3.0, - total: 6.0, - }, - RequestTimes { - min: 7.0, - p50: 7.0, - p90: 7.0, - p99: 7.0, - max: 7.0, - total: 7.0, - } - ) - ); - } - - #[test] - fn a_summary_groups_the_records_by_call_and_file_type() { - let records = [ - record("put_raw", "pack", 3), - record("get_raw_slice", "pack", 5), - record("put_raw", "pack", 1), - record("put_raw", "index", 2), - record("put_raw", "pack", 0), - ]; - - let summary = summarize(&records); - - assert_eq!( - ( - summary - .iter() - .map(|summary| ( - summary.call, - summary.file_type, - summary.count, - summary.errors, - summary.bytes, - summary.time_ms.max - )) - .collect::>(), - written_and_read(&records) - ), - ( - vec![ - ("get_raw_slice", "pack", 1, 0, 10, 5.0), - ("put_raw", "index", 1, 0, 10, 2.0), - ("put_raw", "pack", 3, 1, 30, 3.0), - ], - (40, 10) - ) - ); - } - - #[test] - fn the_file_type_is_the_first_name_of_the_repository_path() { - assert_eq!( - [ - "config", - "data/ab/abcd", - "index/ab", - "snapshots/ab", - "keys/ab", - "results/base/save.json", - "", - "./data/ab/abcd", - "agents/0/config", - "agents/x8-7/data/ab/abcd", - "agents/0", - "agents" - ] - .map(|path| file_type(Path::new(path))), - [ - "config", "pack", "index", "snapshot", "key", "other", "other", "pack", "config", - "pack", "other", "other" - ] - ); - } - - #[test] - async fn the_storage_records_each_call_and_passes_it_on() { - let inner = Arc::new(InMemoryBlobStorage::new()); - let storage = MeasuredBlobStorage::new(inner.clone()); - let namespace = BlobStorageNamespace::InitialAgentFiles { - environment_id: EnvironmentId(Uuid::new_v4()), - }; - let pack = Path::new("data/ab/abcd"); - - storage - .put_raw("test", "test", namespace.clone(), pack, b"0123456789") - .await - .unwrap(); - let before = storage.take(); - let slice = storage - .get_raw_slice("test", "test", namespace.clone(), pack, 2, 4) - .await - .unwrap(); - let missing = storage - .get_raw("test", "test", namespace.clone(), Path::new("index/ef")) - .await - .unwrap(); - let after = storage.take(); - let empty = storage.take(); - let stored = inner - .get_raw("test", "test", namespace, pack) - .await - .unwrap(); - - assert_eq!( - ( - before - .iter() - .map(|record| ( - record.call, - record.file_type, - record.bytes, - record.written, - record.ok - )) - .collect::>(), - after - .iter() - .map(|record| ( - record.call, - record.file_type, - record.bytes, - record.written, - record.ok - )) - .collect::>(), - empty.len(), - slice, - missing, - stored, - ), - ( - vec![("put_raw", "pack", 10, true, true)], - vec![ - ("get_raw_slice", "pack", 3, false, true), - ("get_raw", "index", 0, false, true) - ], - 0, - Some(b"234".to_vec()), - None, - Some(b"0123456789".to_vec()), - ) - ); - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/scopes.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/scopes.rs deleted file mode 100644 index 8b450f00da..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/scopes.rs +++ /dev/null @@ -1,97 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The phase of the scopes scenario: the copy of the repository of an agent for a fork, and the -//! deletion of the repository of an agent. - -use super::agents::{FIRST_AGENT, agent_blobs, copy_agent, delete_agent}; -use super::measure::measure; -use super::report::{Outcome, TreeFacts}; -use super::{PhaseContext, PhaseOutcome, failed}; -use serde_json::{Value, json}; - -/// The agent that gets the copy of the repository of the first agent. -const FORK: &str = "fork"; - -/// The scopes phase: the repository of the save phase goes on the server to the fork agent, as -/// `copy_scope` copies a scope, and then the repository of the fork agent is deleted, as -/// `delete_scope` deletes a scope. -/// -/// A listing after the copy checks that the fork has each blob of the source with its size, and a -/// listing after the delete checks that the fork has no blob. Each listing is outside the measured -/// steps. -pub(super) async fn scopes(context: &PhaseContext) -> PhaseOutcome { - let storage = &context.storage; - let namespace = context.scope().0; - let facts = TreeFacts { - name: context.selection.tree.name, - ..TreeFacts::default() - }; - - let (record, copied) = measure( - "copy_scope", - storage, - copy_agent(storage.as_ref(), &namespace, FIRST_AGENT, FORK), - ) - .await; - let (source, fork) = ( - agent_blobs(storage.as_ref(), &namespace, FIRST_AGENT).await, - agent_blobs(storage.as_ref(), &namespace, FORK).await, - ); - let same = - matches!((&source, &fork), (Ok(source), Ok(fork)) if !source.is_empty() && source == fork); - let details = match copied.as_ref() { - Ok(Value::Object(details)) => { - let mut details = details.clone(); - details.insert("same_as_source".to_string(), json!(same)); - Value::Object(details) - } - _ => Value::Null, - }; - let mut steps = vec![record.with_details(details, Box::default())]; - if copied.is_err() { - return failed(facts, steps, "copy_scope", &["delete_scope"]); - } - - let (record, deleted) = measure( - "delete_scope", - storage, - delete_agent(storage.as_ref(), &namespace, FORK), - ) - .await; - let left = agent_blobs(storage.as_ref(), &namespace, FORK) - .await - .map(|blobs| blobs.len()); - steps.push(record.with_details( - json!({ "deleted": deleted.as_ref().ok(), "blobs_left": left.as_ref().ok() }), - Box::default(), - )); - if deleted.is_err() { - return failed(facts, steps, "delete_scope", &[]); - } - let outcome = match (same, left) { - (true, Ok(0)) => Outcome::Ok, - (false, _) => Outcome::Failed { - reason: "the copy of the repository differs from the repository".into(), - }, - (true, _) => Outcome::Failed { - reason: "the deleted repository still has blobs".into(), - }, - }; - PhaseOutcome { - tree_facts: facts, - steps, - outcome, - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/sqlite.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/sqlite.rs deleted file mode 100644 index 63ec9e9602..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/sqlite.rs +++ /dev/null @@ -1,141 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The save phase of the SQLite changes scenario. - -use super::measure::measure; -use super::report::{Outcome, TreeFacts}; -use super::{ - COLD_SAVE, PhaseContext, PhaseOutcome, WARM_SAVE, failed, save_record, snapshot_name, trees, -}; -use serde_json::{Value, json}; - -/// The name of the snapshot after the clustered change. -const CLUSTERED_SAVE: &str = "warm-save-clustered"; - -/// The save phase of a SQLite tree into the repository of the variant of the phase: a cold save, -/// an update of 100 consecutive rows and a warm save, then an update of 100 rows spread over the -/// database and a second warm save, and the hash of the tree. -/// -/// The chunker of the repository decides how many bytes each warm save adds. The second warm save -/// has the name of the warm save of the base scenario, so the restore phase of the base scenario -/// restores it. -pub(super) async fn sqlite_changes(context: &PhaseContext) -> PhaseOutcome { - let spec = context.selection.tree; - let tree = context.work_dir.join("tree"); - let repository = context.repository(); - let storage = &context.storage; - let settings = context.save_settings(); - let facts = TreeFacts { - name: spec.name, - content: Some(spec.content.label()), - page_cache: Some("dropped"), - ..TreeFacts::default() - }; - let later = [ - "cold_save", - "clustered_change", - "warm_save_clustered", - "scattered_change", - "warm_save_scattered", - "hash_tree", - ]; - - let (record, generated) = measure("generate_tree", storage, trees::generate(spec, &tree)).await; - let mut steps = vec![record]; - let Ok(counts) = generated else { - return failed(facts, steps, "generate_tree", &later); - }; - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - ..facts - }; - - let (record, cold) = measure("cold_save", storage, async { - repository - .save_with(&snapshot_name(COLD_SAVE)?, &tree, settings) - .await - }) - .await; - steps.push(save_record(record, &cold)); - if cold.is_err() { - return failed(facts, steps, "cold_save", &later[1..]); - } - - let (record, clustered) = measure( - "clustered_change", - storage, - trees::change_clustered(spec, &tree), - ) - .await; - steps.push(record.with_details( - clustered.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )); - if clustered.is_err() { - return failed(facts, steps, "clustered_change", &later[2..]); - } - - let (record, warm) = measure("warm_save_clustered", storage, async { - repository - .save_with(&snapshot_name(CLUSTERED_SAVE)?, &tree, settings) - .await - }) - .await; - steps.push(save_record(record, &warm).with_parameters(json!({ "change": "clustered" }))); - if warm.is_err() { - return failed(facts, steps, "warm_save_clustered", &later[3..]); - } - - let (record, scattered) = - measure("scattered_change", storage, trees::change(spec, &tree)).await; - steps.push(record.with_details( - scattered.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )); - let Ok(change) = scattered else { - return failed(facts, steps, "scattered_change", &later[4..]); - }; - let facts = TreeFacts { - change: Some(change), - ..facts - }; - - let (record, warm) = measure("warm_save_scattered", storage, async { - repository - .save_with(&snapshot_name(WARM_SAVE)?, &tree, settings) - .await - }) - .await; - steps.push(save_record(record, &warm).with_parameters(json!({ "change": "scattered" }))); - if warm.is_err() { - return failed(facts, steps, "warm_save_scattered", &later[5..]); - } - - let (record, hashed) = measure("hash_tree", storage, trees::hash(&tree)).await; - steps.push(record); - let Ok((hash, _)) = hashed else { - return failed(facts, steps, "hash_tree", &[]); - }; - PhaseOutcome { - tree_facts: TreeFacts { - hash_after_change: Some(hash), - ..facts - }, - steps, - outcome: Outcome::Ok, - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/tests.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/tests.rs deleted file mode 100644 index 60d238cae2..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/tests.rs +++ /dev/null @@ -1,1736 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -use super::agents::AgentStorage; -use super::report::{Outcome, PhaseResult, StepRecord, StepStatus}; -use super::requests::MeasuredBlobStorage; -use super::trees::{Content, FILES_TINY, SQLITE_TINY, TreeShape, TreeSpec, tree_hash}; -use super::{ - BASE, CPU_DEFAULT, CPU_OPTIONS_PHASES, CPU_ZSTD_OFF, Compression, DEFAULTS, HISTORY, PRUNE, - PRUNE_FAST_REPACK, Phase, PhaseContext, PhaseKind, PlanEntry, RESTORE_THREADS_PHASES, - Repository, RepositorySettings, SAVE, SAVE_THREADS_2, SQLITE_FIXED_64K, SQLITE_RABIN, - STORAGE_CALL_DEADLINE, SaveSettings, Scenario, Selection, WARM_SAVE, concurrent_restore, - concurrent_save, is_key_segment, mixed_phase, plan, prune_phase, repository_key, - repository_scope, result_path, run_phase, save_phase, save_threads_phase, snapshot_name, - with_defaults, with_settings, -}; -use async_trait::async_trait; -use bytes::Bytes; -use futures::stream::BoxStream; -use golem_service_base::replayable_stream::ErasedReplayableStream; -use golem_service_base::storage::blob::memory::InMemoryBlobStorage; -use golem_service_base::storage::blob::{ - BlobMetadata, BlobStorage, BlobStorageNamespace, ExistsResult, ListedBlob, PutIfAbsent, -}; -use pretty_assertions::assert_eq; -use serde_json::{Value, json}; -use std::num::{NonZeroI32, NonZeroUsize}; -use std::path::{Path, PathBuf}; -use std::sync::Arc; -use std::sync::atomic::{AtomicUsize, Ordering}; -use test_r::test; - -/// The phases of the scenario `restore-threads` on a tiny tree. -static RESTORE_THREADS_TINY: Scenario = Scenario { - name: "restore-threads-tiny", - trees: &[FILES_TINY], - phases: RESTORE_THREADS_PHASES, - memory_limited_phases: &[], -}; - -/// The phases of the concurrent scenarios with 3 agents, on a tiny tree. -static CONCURRENT_TINY: Scenario = Scenario { - name: "concurrent-tiny", - trees: &[FILES_TINY], - phases: &[ - SAVE, - concurrent_restore("restore-x3", 3, 2), - concurrent_save("save-x3", 3), - ], - memory_limited_phases: &[], -}; - -/// Runs the phase of the scenario on its first tree, in a new work directory, and gives the -/// result and the work directory. -async fn run_tiny( - scenario: &'static Scenario, - phase: &str, - storage: &Arc, -) -> (PhaseResult, tempfile::TempDir) { - run_tiny_with(scenario, phase, storage.clone(), |_| {}).await -} - -/// Runs the phase of the scenario on its first tree over the storage, in a new work directory -/// that `prepare` gets before the phase runs, and gives the result and the work directory. -async fn run_tiny_with( - scenario: &'static Scenario, - phase: &str, - storage: Arc, - prepare: impl FnOnce(&Path), -) -> (PhaseResult, tempfile::TempDir) { - let work = tempfile::tempdir().unwrap(); - prepare(work.path()); - let selection = Selection { - scenario, - tree: &scenario.trees[0], - phase: scenario - .phases - .iter() - .find(|candidate| candidate.name == phase) - .unwrap(), - }; - let (result, written) = run_phase( - "run-1", - "no-limit", - selection, - work.path(), - storage, - json!({}), - ) - .await; - written.unwrap(); - (result, work) -} - -/// Gives the files, the directories and the bytes of the tree facts of the result. -fn tree_counts(result: &PhaseResult) -> (Option, Option, Option) { - ( - result.tree_facts.files, - result.tree_facts.directories, - result.tree_facts.bytes, - ) -} - -/// The files, the directories and the bytes of a new `FILES_TINY` tree. -const FILES_TINY_COUNTS: (Option, Option, Option) = - (Some(100), Some(10), Some(1024 * 1024)); - -/// Gives the name of each step of the result that did not run. -fn skipped_steps(result: &PhaseResult) -> Vec<&'static str> { - result - .steps - .iter() - .filter(|step| step.status == StepStatus::Skipped) - .map(|step| step.name) - .collect() -} - -/// Gives the name of each step of the result and whether it succeeded. -fn step_states(result: &PhaseResult) -> Vec<(&'static str, bool)> { - result - .steps - .iter() - .map(|step| (step.name, step.status == StepStatus::Ok)) - .collect() -} - -fn step<'a>(result: &'a PhaseResult, name: &str) -> &'a StepRecord { - result.steps.iter().find(|step| step.name == name).unwrap() -} - -/// Gives the namespace of the repositories of the agents of a phase of `CONCURRENT_TINY`. -fn concurrent_tiny_namespace() -> BlobStorageNamespace { - repository_scope(CONCURRENT_TINY.name, "no-limit", FILES_TINY.name).0 -} - -#[test] -fn the_plan_gives_the_phases_of_each_tree_of_each_scenario() { - let names = |names: &[&str]| { - names - .iter() - .map(|name| name.to_string()) - .collect::>() - }; - let entry = |scenario, tree| PlanEntry { - scenario, - tree, - phases: Box::new(["save", "restore"]), - memory_limited_phases: Box::new([]), - }; - let restore_threads = PlanEntry { - phases: Box::new([ - "save", - "restore-1", - "restore-2", - "restore-4", - "restore-8", - "restore-20", - ]), - ..entry("restore-threads", "files-1g") - }; - let memory_pressure = PlanEntry { - memory_limited_phases: Box::new(["restore"]), - ..entry("memory-pressure", "files-1g") - }; - - assert_eq!( - ( - plan(&names(&["base", "smoke", "restore-threads", "memory-pressure"])).map(Vec::from), - plan(&names(&["base", "unknown"])).map(Vec::from), - serde_json::to_string(&entry("base", "files-1g")).unwrap(), - serde_json::to_string(&memory_pressure).unwrap(), - ), - ( - Ok(vec![ - entry("base", "files-128m"), - entry("base", "files-1g"), - entry("base", "sqlite-1g"), - entry("base", "objects-128m"), - entry("base", "objects-1g"), - entry("smoke", "files-tiny"), - entry("smoke", "sqlite-tiny"), - restore_threads.clone(), - memory_pressure.clone(), - ]), - Err("unknown".to_string()), - r#"{"scenario":"base","tree":"files-1g","phases":["save","restore"]}"#.to_string(), - r#"{"scenario":"memory-pressure","tree":"files-1g","phases":["save","restore"],"memory_limited_phases":["restore"]}"#.to_string(), - ) - ); -} - -#[test] -fn the_plan_gives_the_counts_of_the_concurrent_scenarios() { - let restores = |scenario| { - ["files-128m", "files-1g"].map(|tree| PlanEntry { - scenario, - tree, - phases: Box::new([ - "save", - "restore-x1", - "restore-x5", - "restore-x10", - "restore-x25", - "restore-x50", - "restore-x100", - "restore-x200", - ]), - memory_limited_phases: Box::new([]), - }) - }; - let saves = ["files-128m", "files-1g", "sqlite-1g"].map(|tree| PlanEntry { - scenario: "concurrent-save", - tree, - phases: Box::new(["save-x1", "save-x2", "save-x4", "save-x8"]), - memory_limited_phases: Box::new([]), - }); - let kinds = |scenario: &str| { - super::scenario(scenario) - .unwrap() - .phases - .iter() - .map(|phase| phase.kind) - .collect::>() - }; - let threads = |count| super::NonZeroUsize::new(count); - - assert_eq!( - ( - plan(&[ - "concurrent-restore-1".to_string(), - "concurrent-restore-4".to_string(), - "concurrent-restore-20".to_string(), - "concurrent-save".to_string(), - ]) - .map(Vec::from), - kinds("concurrent-restore-4")[1..3].to_vec(), - kinds("concurrent-restore-20").last().copied(), - kinds("concurrent-restore-1")[1], - kinds("concurrent-save"), - ), - ( - Ok([ - restores("concurrent-restore-1").to_vec(), - restores("concurrent-restore-4").to_vec(), - restores("concurrent-restore-20").to_vec(), - saves.to_vec(), - ] - .concat()), - vec![ - super::PhaseKind::ConcurrentRestore { - agents: 1, - reader_threads: threads(4) - }, - super::PhaseKind::ConcurrentRestore { - agents: 5, - reader_threads: threads(4) - }, - ], - Some(super::PhaseKind::ConcurrentRestore { - agents: 200, - reader_threads: threads(20) - }), - super::PhaseKind::ConcurrentRestore { - agents: 1, - reader_threads: threads(1) - }, - [1, 2, 4, 8] - .map(|agents| super::PhaseKind::ConcurrentSave { agents }) - .to_vec(), - ) - ); -} - -#[test] -fn a_selection_names_a_tree_and_a_phase_of_its_scenario() { - let found = |scenario, tree, phase| Selection::find(scenario, tree, phase).map(|_| ()); - - assert_eq!( - [ - found("base", "files-1g", "save"), - found("smoke", "sqlite-tiny", "restore"), - found("base", "objects-1g", "restore"), - found("restore-threads", "files-1g", "restore-8"), - found("memory-pressure", "files-1g", "restore"), - found("concurrent-restore-4", "files-128m", "restore-x200"), - found("concurrent-save", "sqlite-1g", "save-x8"), - found("base", "files-tiny", "save"), - found("base", "files-1g", "prune"), - found("restore-threads", "files-128m", "save"), - found("concurrent-save", "files-1g", "save"), - found("other", "files-1g", "save"), - ] - .map(|found| found.is_ok()), - [ - true, true, true, true, true, true, true, false, false, false, false, false - ] - ); -} - -#[test] -fn a_key_segment_has_1_to_64_ascii_letters_digits_dashes_or_underscores() { - assert_eq!( - [ - "12345-1", - "limit-3", - "no_limit", - "", - "a/b", - "a.b", - &"x".repeat(64), - &"x".repeat(65), - ] - .map(is_key_segment), - [true, true, true, false, false, false, true, false] - ); -} - -#[test] -fn each_run_id_gives_its_own_repository_key() { - assert_eq!( - ( - repository_key("1-1") == repository_key("1-1"), - repository_key("1-1") == repository_key("1-2"), - ), - (true, false) - ); -} - -#[test] -fn a_result_is_at_the_path_of_its_scenario_cpu_setting_tree_and_phase() { - assert_eq!( - &*result_path("base", "limit-3", "files-1g", "save"), - std::path::Path::new("results/base/limit-3/files-1g/save.json") - ); -} - -#[test] -async fn the_smoke_scenario_saves_and_restores_each_tree_with_the_same_hash() { - let storage = Arc::new(InMemoryBlobStorage::new()); - - let outcomes = futures::future::join_all(["files-tiny", "sqlite-tiny"].map(|tree| { - let storage = storage.clone(); - async move { - let save_pod = tempfile::tempdir().unwrap(); - let restore_pod = tempfile::tempdir().unwrap(); - let (save, save_written) = run_phase( - "run-1", - "no-limit", - Selection::find("smoke", tree, "save").unwrap(), - save_pod.path(), - storage.clone(), - json!({}), - ) - .await; - let (restore, restore_written) = run_phase( - "run-1", - "no-limit", - Selection::find("smoke", tree, "restore").unwrap(), - restore_pod.path(), - storage.clone(), - json!({}), - ) - .await; - let steps = |result: &super::report::PhaseResult| { - result - .steps - .iter() - .map(|step| (step.name, step.status == StepStatus::Ok)) - .collect::>() - }; - let hash_step = restore.steps.iter().find(|step| step.name == "hash_tree"); - let restore_step = restore - .steps - .iter() - .find(|step| step.name == "cold_restore"); - let cgroup_keys = |result: &PhaseResult| { - result.environment["cgroup"] - .as_object() - .map(|cgroup| cgroup.keys().cloned().collect::>()) - }; - ( - save.outcome.clone(), - steps(&save), - save_written.is_ok(), - restore.outcome.clone(), - steps(&restore), - restore_written.is_ok(), - hash_step.map(|step| step.details["matches"].clone()), - save.steps - .iter() - .find(|step| step.name == "cold_save") - .map(|step| { - step.requests - .iter() - .any(|request| request.call == "put_raw" && request.file_type == "pack") - }), - restore_step.map(|step| step.parameters.clone()), - cgroup_keys(&save), - ) - } - })) - .await; - let cgroup_keys = Some(vec![ - "memory_events_end".to_string(), - "memory_events_start".to_string(), - ]); - - assert_eq!( - outcomes, - vec![ - ( - Outcome::Ok, - vec![ - ("generate_tree", true), - ("cold_save", true), - ("small_change", true), - ("warm_save", true), - ("hash_tree", true), - ], - true, - Outcome::Ok, - vec![("cold_restore", true), ("hash_tree", true)], - true, - Some(json!(true)), - Some(true), - Some(json!({ "reader_threads": null })), - cgroup_keys, - ); - 2 - ] - ); -} - -#[test] -async fn each_restore_of_the_restore_threads_phases_records_its_reader_threads() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let (save, _save_pod) = run_tiny(&RESTORE_THREADS_TINY, "save", &storage).await; - - let restores = futures::future::join_all(["restore-1", "restore-20"].map(|phase| { - let storage = storage.clone(); - async move { - let (restore, _restore_pod) = run_tiny(&RESTORE_THREADS_TINY, phase, &storage).await; - ( - restore.outcome.clone(), - step(&restore, "cold_restore").parameters.clone(), - step(&restore, "hash_tree").details["matches"].clone(), - ) - } - })) - .await; - - assert_eq!( - (save.outcome, restores), - ( - Outcome::Ok, - vec![ - (Outcome::Ok, json!({ "reader_threads": 1 }), json!(true)), - (Outcome::Ok, json!({ "reader_threads": 20 }), json!(true)), - ] - ) - ); -} - -#[test] -async fn the_agents_of_a_concurrent_restore_get_the_first_repository_and_restore_it() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let (save, _save_pod) = run_tiny(&CONCURRENT_TINY, "save", &storage).await; - - let (first, first_pod) = run_tiny(&CONCURRENT_TINY, "restore-x3", &storage).await; - let (again, _again_pod) = run_tiny(&CONCURRENT_TINY, "restore-x3", &storage).await; - let configs = futures::future::join_all((0..4).map(|agent| { - let storage = storage.clone(); - async move { - storage - .get_metadata( - "test", - "test", - concurrent_tiny_namespace(), - &Path::new("agents").join(agent.to_string()).join("config"), - ) - .await - .unwrap() - .is_some() - } - })) - .await; - let restore = step(&first, "concurrent_restore"); - - assert_eq!( - ( - save.outcome, - first.outcome.clone(), - step_states(&first), - step(&first, "copy_scopes").details.clone(), - step(&again, "copy_scopes").details.clone(), - restore.parameters.clone(), - [ - &restore.details["agents"], - &restore.details["failed"], - &restore.details["first_error"] - ] - .map(Value::clone), - [ - &step(&first, "hash_trees").details["trees"], - &step(&first, "hash_trees").details["matches"] - ] - .map(Value::clone), - configs, - first_pod.path().join("restore").exists(), - ), - ( - Outcome::Ok, - Outcome::Ok, - vec![ - ("copy_scopes", true), - ("concurrent_restore", true), - ("hash_trees", true) - ], - json!({ "agents_copied": 2, "blobs_copied": restore_blobs(&storage).await * 2 }), - json!({ "agents_copied": 0, "blobs_copied": 0 }), - json!({ "agents": 3, "reader_threads": 2 }), - [json!(3), json!(0), Value::Null], - [json!(3), json!(3)], - vec![true, true, true, false], - false, - ) - ); -} - -/// Gives the number of blobs of the repository of the first agent of `CONCURRENT_TINY`. -async fn restore_blobs(storage: &InMemoryBlobStorage) -> usize { - storage - .list_blobs_below( - "test", - "test", - concurrent_tiny_namespace(), - Path::new("agents/0"), - ) - .await - .unwrap() - .len() -} - -#[test] -async fn the_agents_of_a_concurrent_save_each_save_their_tree_into_their_own_repository() { - let storage = Arc::new(InMemoryBlobStorage::new()); - - let (save, pod) = run_tiny(&CONCURRENT_TINY, "save-x3", &storage).await; - let restores = futures::future::join_all((0..3).map(|agent| { - let storage = storage.clone(); - let tree = pod.path().join("trees").join(agent.to_string()); - async move { - let repository = Repository::new( - Arc::new(AgentStorage::new(storage.clone(), &format!("x3-{agent}"))), - repository_scope(CONCURRENT_TINY.name, "no-limit", FILES_TINY.name), - repository_key("run-1"), - STORAGE_CALL_DEADLINE, - ); - let into = tempfile::tempdir().unwrap(); - repository - .restore(&snapshot_name(WARM_SAVE).unwrap(), into.path(), None) - .await - .unwrap(); - let snapshots = storage - .list_blobs_below( - "test", - "test", - concurrent_tiny_namespace(), - &Path::new("agents") - .join(format!("x3-{agent}")) - .join("snapshots"), - ) - .await - .unwrap() - .len(); - ( - snapshots, - tree_hash(into.path()).unwrap() == tree_hash(&tree).unwrap(), - ) - } - })) - .await; - let batch = |name| { - let step = step(&save, name); - ( - step.parameters.clone(), - [ - &step.details["agents"], - &step.details["failed"], - &step.details["first_error"], - ] - .map(Value::clone), - step.details["data_added"] - .as_u64() - .is_some_and(|added| added > 0), - ) - }; - let copied = &step(&save, "copy_trees").details; - - assert_eq!( - ( - save.outcome.clone(), - step_states(&save), - batch("concurrent_cold_save"), - batch("concurrent_warm_save"), - ( - copied["copies"].clone(), - copied["files_reflinked"].as_u64().unwrap_or_default() - + copied["files_copied"].as_u64().unwrap_or_default() - ), - step(&save, "small_change").details["trees"].clone(), - restores, - ), - ( - Outcome::Ok, - vec![ - ("generate_tree", true), - ("copy_trees", true), - ("concurrent_cold_save", true), - ("small_change", true), - ("concurrent_warm_save", true), - ], - ( - json!({ "agents": 3 }), - [json!(3), json!(0), Value::Null], - true - ), - ( - json!({ "agents": 3 }), - [json!(3), json!(0), Value::Null], - true - ), - (json!(2), 200), - json!(3), - vec![(2, true); 3], - ) - ); -} - -/// A tiny tree whose content compresses. -const COMPRESSIBLE_TINY: TreeSpec = TreeSpec { - name: "compressible-tiny", - shape: TreeShape::Files { - files: 100, - directories: 10, - bytes: 1024 * 1024, - }, - content: Content::Compressible, -}; - -static CAPTURE_TINY: Scenario = Scenario { - name: "capture-tiny", - trees: &[FILES_TINY], - phases: &[with_defaults("capture", PhaseKind::Capture)], - memory_limited_phases: &[], -}; - -static PRUNE_TINY: Scenario = Scenario { - name: "prune-tiny", - trees: &[FILES_TINY], - phases: &[ - HISTORY, - prune_phase("prune", &PRUNE, false), - prune_phase("prune-fast-repack", &PRUNE_FAST_REPACK, true), - ], - memory_limited_phases: &[], -}; - -static SQLITE_CHANGES_TINY: Scenario = Scenario { - name: "sqlite-changes-tiny", - trees: &[SQLITE_TINY], - phases: &[ - Phase { - name: "save-rabin", - kind: PhaseKind::SqliteChanges, - variant: &SQLITE_RABIN, - }, - Phase { - name: "save-fixed-64k", - kind: PhaseKind::SqliteChanges, - variant: &SQLITE_FIXED_64K, - }, - Phase { - name: "restore-fixed-64k", - kind: PhaseKind::Restore { - reader_threads: None, - }, - variant: &SQLITE_FIXED_64K, - }, - ], - memory_limited_phases: &[], -}; - -static MIXED_TINY: Scenario = Scenario { - name: "mixed-tiny", - trees: &[FILES_TINY], - phases: &[ - with_defaults("save", PhaseKind::Save), - mixed_phase("mixed-s2-r2", &SAVE_THREADS_2, 2), - save_threads_phase("save-x4-t2", &SAVE_THREADS_2), - ], - memory_limited_phases: &[], -}; - -static SCOPES_TINY: Scenario = Scenario { - name: "scopes-tiny", - trees: &[FILES_TINY], - phases: &[ - with_defaults("save", PhaseKind::Save), - with_defaults("scopes", PhaseKind::Scopes), - ], - memory_limited_phases: &[], -}; - -static CPU_OPTIONS_TINY: Scenario = Scenario { - name: "cpu-options-tiny", - trees: &[COMPRESSIBLE_TINY], - phases: &[ - save_phase("save-default", &CPU_DEFAULT), - save_phase("save-zstd-off", &CPU_ZSTD_OFF), - ], - memory_limited_phases: &[], -}; - -/// Gives the value at the JSON pointer of the details of the step. -fn detail(result: &PhaseResult, step_name: &str, pointer: &str) -> Value { - step(result, step_name) - .details - .pointer(pointer) - .cloned() - .unwrap_or(Value::Null) -} - -#[test] -fn the_plan_gives_the_phases_of_each_later_scenario() { - let entry = |scenario, tree, phases: &[&'static str]| PlanEntry { - scenario, - tree, - phases: phases.into(), - memory_limited_phases: Box::new([]), - }; - let base_trees = [ - "files-128m", - "files-1g", - "sqlite-1g", - "objects-128m", - "objects-1g", - ]; - let history_trees = ["files-1g", "sqlite-1g", "objects-1g"]; - let each = |scenario, trees: &[&'static str], phases: &[&'static str]| { - trees - .iter() - .map(|tree| entry(scenario, *tree, phases)) - .collect::>() - }; - let names = [ - "capture", - "prune", - "save-threads", - "mixed", - "repository-open", - "sqlite-changes", - "tree-shape", - "cpu-options", - "scopes", - ] - .map(str::to_string); - - assert_eq!( - plan(&names).map(Vec::from), - Ok([ - each("capture", &base_trees, &["capture"]), - each( - "prune", - &history_trees, - &["history", "prune", "prune-fast-repack"] - ), - each( - "save-threads", - &["files-1g", "sqlite-1g"], - &["save-x4-t1", "save-x4-t2", "save-x4-t4", "save-x4-tdefault"] - ), - each( - "mixed", - &["files-1g"], - &[ - "save", - "mixed-s2-r2", - "mixed-s2-r4", - "mixed-s4-r2", - "mixed-s4-r4" - ] - ), - each("repository-open", &history_trees, &["history", "prune"]), - each( - "sqlite-changes", - &["sqlite-1g"], - &[ - "save-rabin", - "restore-rabin", - "save-fixed-64k", - "restore-fixed-64k" - ] - ), - each( - "tree-shape", - &["files-128m", "modules-128m"], - &["save", "restore"] - ), - each( - "cpu-options", - &["compressible-1g"], - &[ - "save-default", - "save-verify-off", - "save-zstd-off", - "save-zstd-1", - "save-zstd-9" - ] - ), - each("scopes", &base_trees, &["save", "scopes"]), - ] - .concat()) - ); -} - -#[test] -fn a_later_phase_records_its_settings_and_an_earlier_phase_records_none() { - let record = || StepRecord::skipped("save").with_parameters(json!({ "agents": 4 })); - let own = - StepRecord::skipped("save").with_parameters(json!({ "change_detection": "size-mtime" })); - - assert_eq!( - ( - with_settings(record(), &BASE).parameters, - with_settings(record(), &SAVE_THREADS_2).parameters, - with_settings(own, &DEFAULTS).parameters["change_detection"].clone(), - with_settings(record(), &SQLITE_FIXED_64K).parameters["chunker"].clone(), - with_settings(record(), &CPU_ZSTD_OFF).parameters["compression"].clone(), - ), - ( - json!({ "agents": 4 }), - json!({ - "agents": 4, - "save_threads": 2, - "change_detection": "ctime", - "chunker": "rabin", - "compression": null, - "extra_verify": true, - }), - json!("size-mtime"), - json!("fixed-65536"), - json!("off"), - ) - ); -} - -#[test] -async fn a_capture_phase_reads_every_file_with_ctime_and_the_changed_files_with_size_and_mtime() { - let storage = Arc::new(InMemoryBlobStorage::new()); - - let (result, _pod) = run_tiny(&CAPTURE_TINY, "capture", &storage).await; - let counts = |name: &str| { - ( - detail(&result, name, "/files_new"), - detail(&result, name, "/files_changed"), - detail(&result, name, "/files_unmodified"), - ) - }; - let captured = |name: &str| { - detail(&result, name, "/files_reflinked") - .as_u64() - .unwrap_or_default() - + detail(&result, name, "/files_copied") - .as_u64() - .unwrap_or_default() - }; - - assert_eq!( - ( - result.outcome.clone(), - step_states(&result), - [ - captured("capture"), - captured("capture_full_read"), - captured("capture_size_mtime") - ], - counts("warm_save_full_read"), - counts("warm_save_size_mtime"), - step(&result, "warm_save_size_mtime").parameters["change_detection"].clone(), - tree_counts(&result), - result.tree_facts.change.as_ref() == Some(&step(&result, "small_change").details), - ), - ( - Outcome::Ok, - vec![ - ("generate_tree", true), - ("capture", true), - ("cold_save", true), - ("copy_scopes", true), - ("small_change", true), - ("capture_full_read", true), - ("warm_save_full_read", true), - ("capture_size_mtime", true), - ("warm_save_size_mtime", true), - ], - [100, 101, 101], - (json!(1), json!(100), json!(0)), - (json!(1), json!(10), json!(90)), - json!("size-mtime"), - FILES_TINY_COUNTS, - true, - ) - ); -} - -#[test] -async fn the_prune_phases_give_back_the_data_of_the_forgotten_snapshots_of_the_history() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let (history, _history_pod) = run_tiny(&PRUNE_TINY, "history", &storage).await; - - let prunes = futures::future::join_all(["prune", "prune-fast-repack"].map(|phase| { - let storage = storage.clone(); - async move { - let (result, _pod) = run_tiny(&PRUNE_TINY, phase, &storage).await; - ( - result.outcome.clone(), - step_states(&result), - step(&result, "prune_mark").parameters["fast_repack"].clone(), - detail(&result, "prune_mark", "/packs_repacked") - .as_u64() - .is_some_and(|packs| packs > 0), - detail(&result, "prune_delete", "/marked_packs_deleted") - .as_u64() - .is_some_and(|packs| packs > 0), - detail(&result, "prune_delete", "/repository/bytes_given_back") - .as_u64() - .is_some_and(|bytes| bytes > 0), - detail(&result, "open", "/snapshots"), - detail(&result, "hash_tree", "/matches"), - tree_counts(&result), - result.tree_facts.hash.as_deref().map(|hash| json!(hash)) - == Some(detail(&result, "hash_tree", "/hash")), - ) - } - })) - .await; - let opens = history - .steps - .iter() - .filter(|step| step.name == "open") - .map(|step| { - ( - step.parameters["saves"].clone(), - step.details["snapshots"].clone(), - ) - }) - .collect::>(); - let saves = history - .steps - .iter() - .filter(|step| step.name == "save") - .count(); - let prune = |fast_repack| { - ( - Outcome::Ok, - vec![ - ("copy_scopes", true), - ("prune_mark", true), - ("prune_delete", true), - ("open", true), - ("cold_restore", true), - ("hash_tree", true), - ], - json!(fast_repack), - true, - true, - true, - json!(2), - json!(true), - // The history adds one file of 10,486 bytes in each of its 11 rounds. - (Some(111), Some(10), Some(1024 * 1024 + 11 * 10_486)), - true, - ) - }; - - assert_eq!( - ( - history.outcome.clone(), - saves, - opens, - detail(&history, "forget", "/snapshots_forgotten"), - tree_counts(&history), - prunes, - ), - ( - Outcome::Ok, - 11, - vec![ - (json!(1), json!(1)), - (json!(11), json!(11)), - (json!(12), json!(12)), - ], - json!(10), - FILES_TINY_COUNTS, - vec![prune(false), prune(true)], - ) - ); -} - -#[test] -async fn fixed_chunks_save_less_after_a_clustered_change_and_restore_the_database() { - let storage = Arc::new(InMemoryBlobStorage::new()); - - let (rabin, _rabin_pod) = run_tiny(&SQLITE_CHANGES_TINY, "save-rabin", &storage).await; - let (fixed, _fixed_pod) = run_tiny(&SQLITE_CHANGES_TINY, "save-fixed-64k", &storage).await; - let (restore, _restore_pod) = - run_tiny(&SQLITE_CHANGES_TINY, "restore-fixed-64k", &storage).await; - let added = |result: &PhaseResult, name: &str| { - detail(result, name, "/data_added") - .as_u64() - .unwrap_or_default() - }; - - assert_eq!( - ( - rabin.outcome.clone(), - fixed.outcome.clone(), - step_states(&fixed), - added(&fixed, "warm_save_clustered") < added(&rabin, "warm_save_clustered"), - step(&fixed, "cold_save").parameters["chunker"].clone(), - detail(&fixed, "clustered_change", "/rows_updated"), - restore.outcome.clone(), - detail(&restore, "hash_tree", "/matches"), - (fixed.tree_facts.files, fixed.tree_facts.directories), - fixed - .tree_facts - .bytes - .is_some_and(|bytes| bytes >= 4 * 1024 * 1024), - fixed.tree_facts.change.as_ref() == Some(&step(&fixed, "scattered_change").details), - [ - detail(&rabin, "open", "/settings/chunker"), - detail(&fixed, "open", "/settings"), - ], - ), - ( - Outcome::Ok, - Outcome::Ok, - vec![ - ("generate_tree", true), - ("cold_save", true), - ("clustered_change", true), - ("warm_save_clustered", true), - ("scattered_change", true), - ("warm_save_scattered", true), - ("hash_tree", true), - ("open", true), - ], - true, - json!("fixed-65536"), - json!(100), - Outcome::Ok, - json!(true), - (Some(1), Some(0)), - true, - true, - [ - json!("rabin"), - json!({ "chunker": "fixed-65536", "compression": null, "extra_verify": true }), - ], - ) - ); -} - -#[test] -async fn a_mixed_phase_saves_and_restores_at_the_same_time_with_its_thread_counts() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let (save, _save_pod) = run_tiny(&MIXED_TINY, "save", &storage).await; - - let (mixed, _mixed_pod) = run_tiny(&MIXED_TINY, "mixed-s2-r2", &storage).await; - let (threads, _threads_pod) = run_tiny(&MIXED_TINY, "save-x4-t2", &storage).await; - let configs = futures::future::join_all( - [ - "agents/1", - "agents/4", - "agents/5", - "agents/mixed-s2-r2-3", - "agents/save-x4-t2-3", - ] - .map(|agent| { - let storage = storage.clone(); - async move { - storage - .get_metadata( - "test", - "test", - repository_scope(MIXED_TINY.name, "no-limit", FILES_TINY.name).0, - &Path::new(agent).join("config"), - ) - .await - .unwrap() - .is_some() - } - }), - ) - .await; - let step_mixed = step(&mixed, "mixed"); - - assert_eq!( - ( - save.outcome, - mixed.outcome.clone(), - step_states(&mixed), - [ - &step_mixed.parameters["saves"], - &step_mixed.parameters["restores"], - &step_mixed.parameters["save_threads"], - &step_mixed.parameters["reader_threads"], - ] - .map(Value::clone), - [ - &step_mixed.details["saves"]["agents"], - &step_mixed.details["saves"]["failed"], - &step_mixed.details["restores"]["agents"], - &step_mixed.details["restores"]["failed"], - ] - .map(Value::clone), - detail(&mixed, "hash_trees", "/matches"), - threads.outcome.clone(), - step(&threads, "concurrent_cold_save").parameters["save_threads"].clone(), - configs, - ), - ( - Outcome::Ok, - Outcome::Ok, - vec![ - ("copy_scopes", true), - ("generate_tree", true), - ("copy_trees", true), - ("mixed", true), - ("hash_trees", true), - ], - [json!(4), json!(4), json!(2), json!(2)], - [json!(4), json!(0), json!(4), json!(0)], - json!(4), - Outcome::Ok, - json!(2), - vec![true, true, false, true, true], - ) - ); -} - -#[test] -async fn a_scopes_phase_copies_the_repository_whole_and_deletes_the_copy_whole() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let (save, _save_pod) = run_tiny(&SCOPES_TINY, "save", &storage).await; - - let (scopes, _pod) = run_tiny(&SCOPES_TINY, "scopes", &storage).await; - - assert_eq!( - ( - save.outcome, - scopes.outcome.clone(), - step_states(&scopes), - detail(&scopes, "copy_scope", "/same_as_source"), - detail(&scopes, "copy_scope", "/blobs_copied") - .as_u64() - .is_some_and(|blobs| blobs > 0), - detail(&scopes, "delete_scope", "/blobs_left"), - ), - ( - Outcome::Ok, - Outcome::Ok, - vec![("copy_scope", true), ("delete_scope", true)], - json!(true), - true, - json!(0), - ) - ); -} - -#[test] -async fn a_cpu_options_phase_makes_its_repository_with_its_compression() { - let storage = Arc::new(InMemoryBlobStorage::new()); - - let (default, _default_pod) = run_tiny(&CPU_OPTIONS_TINY, "save-default", &storage).await; - let (off, _off_pod) = run_tiny(&CPU_OPTIONS_TINY, "save-zstd-off", &storage).await; - let packed = |result: &PhaseResult| { - ( - detail(result, "cold_save", "/data_added").as_u64(), - detail(result, "cold_save", "/data_added_packed").as_u64(), - ) - }; - let (default_added, default_packed) = packed(&default); - let (off_added, off_packed) = packed(&off); - - assert_eq!( - ( - default.outcome.clone(), - off.outcome.clone(), - default.tree_facts.content, - default_packed < default_added.map(|added| added * 3 / 4), - off_packed >= off_added, - step(&off, "cold_save").parameters["compression"].clone(), - step(&default, "open").details.clone(), - detail(&off, "open", "/settings"), - ), - ( - Outcome::Ok, - Outcome::Ok, - Some("compressible"), - true, - true, - json!("off"), - json!({ - "snapshots": 2, - "found": true, - "settings": { "chunker": "rabin", "compression": null, "extra_verify": true }, - }), - json!({ "chunker": "rabin", "compression": "off", "extra_verify": true }), - ) - ); -} - -/// A blob storage that passes each call to an in-memory storage, and fails each write of a -/// snapshot file below `prefix` after the first `allowed` of them. After `config_hidden_after` -/// such writes, it gives no metadata for the config of the agent, so the repository looks absent. -#[derive(Debug)] -struct FailingSnapshotWrites { - inner: Arc, - prefix: PathBuf, - config: PathBuf, - allowed: usize, - config_hidden_after: usize, - seen: AtomicUsize, -} - -impl FailingSnapshotWrites { - /// Gives a storage over `inner` that fails the writes of snapshot files of the agent after the - /// first `allowed` of them. - fn new(inner: Arc, agent: &str, allowed: usize) -> Arc { - Self::hiding_config(inner, agent, allowed, usize::MAX) - } - - /// Gives a storage over `inner` that fails the writes of snapshot files of the agent after the - /// first `allowed` of them, and hides the config of the agent after `config_hidden_after` - /// writes of snapshot files. - fn hiding_config( - inner: Arc, - agent: &str, - allowed: usize, - config_hidden_after: usize, - ) -> Arc { - let root = Path::new("agents").join(agent); - Arc::new(Self { - inner, - prefix: root.join("snapshots"), - config: root.join("config"), - allowed, - config_hidden_after, - seen: AtomicUsize::new(0), - }) - } - - /// Tells whether the write of the path fails. - fn fails(&self, path: &Path) -> bool { - path.starts_with(&self.prefix) && self.seen.fetch_add(1, Ordering::SeqCst) >= self.allowed - } -} - -#[async_trait] -impl BlobStorage for FailingSnapshotWrites { - async fn get_raw( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result>> { - self.inner - .get_raw(target_label, op_label, namespace, path) - .await - } - - async fn get_stream( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result>>> { - self.inner - .get_stream(target_label, op_label, namespace, path) - .await - } - - async fn get_raw_slice( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - start: u64, - end: u64, - ) -> anyhow::Result>> { - self.inner - .get_raw_slice(target_label, op_label, namespace, path, start, end) - .await - } - - async fn get_metadata( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - if path == self.config && self.seen.load(Ordering::SeqCst) >= self.config_hidden_after { - return Ok(None); - } - self.inner - .get_metadata(target_label, op_label, namespace, path) - .await - } - - async fn put_raw( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - data: &[u8], - ) -> anyhow::Result<()> { - anyhow::ensure!(!self.fails(path), "the write of {} fails", path.display()); - self.inner - .put_raw(target_label, op_label, namespace, path, data) - .await - } - - async fn put_raw_if_absent( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - data: &[u8], - ) -> anyhow::Result { - self.inner - .put_raw_if_absent(target_label, op_label, namespace, path, data) - .await - } - - async fn put_stream( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - stream: &dyn ErasedReplayableStream>, Error = anyhow::Error>, - ) -> anyhow::Result<()> { - self.inner - .put_stream(target_label, op_label, namespace, path, stream) - .await - } - - async fn delete( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result<()> { - self.inner - .delete(target_label, op_label, namespace, path) - .await - } - - async fn create_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result<()> { - self.inner - .create_dir(target_label, op_label, namespace, path) - .await - } - - async fn list_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.inner - .list_dir(target_label, op_label, namespace, path) - .await - } - - async fn list_blobs_below( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.inner - .list_blobs_below(target_label, op_label, namespace, path) - .await - } - - async fn delete_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result { - self.inner - .delete_dir(target_label, op_label, namespace, path) - .await - } - - async fn exists( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result { - self.inner - .exists(target_label, op_label, namespace, path) - .await - } -} - -/// The steps of the two forms of a capture phase, in the order in which they run. -const FORM_STEPS: [&str; 4] = [ - "capture_full_read", - "warm_save_full_read", - "capture_size_mtime", - "warm_save_size_mtime", -]; - -#[test] -async fn a_failed_capture_or_warm_save_of_a_form_skips_the_steps_after_it() { - // Form 0 saves into the repository of agent 0, after its cold save. Form 1 saves into the - // repository of agent 1, after the copy of the cold save, which the default `copy` of the - // storage writes. So in both repositories the warm save writes the second snapshot file. A - // capture fails when its directory exists before it. - let capture_fails = async |capture: &'static str| { - let (result, _pod) = run_tiny_with( - &CAPTURE_TINY, - "capture", - Arc::new(InMemoryBlobStorage::new()), - |work| std::fs::create_dir(work.join(capture)).unwrap(), - ) - .await; - (result.outcome.clone(), skipped_steps(&result)) - }; - let save_fails = async |agent: &str, allowed| { - let storage = - FailingSnapshotWrites::new(Arc::new(InMemoryBlobStorage::new()), agent, allowed); - let (result, _pod) = run_tiny_with(&CAPTURE_TINY, "capture", storage, |_| {}).await; - (result.outcome.clone(), skipped_steps(&result)) - }; - let failed = |step: &str| Outcome::Failed { - reason: format!("the step {step} failed").into(), - }; - - let outcomes = [ - capture_fails(FORM_STEPS[0]).await, - save_fails("0", 1).await, - capture_fails(FORM_STEPS[2]).await, - save_fails("1", 1).await, - ]; - - assert_eq!( - outcomes, - [ - (failed(FORM_STEPS[0]), FORM_STEPS[1..].to_vec()), - (failed(FORM_STEPS[1]), FORM_STEPS[2..].to_vec()), - (failed(FORM_STEPS[2]), FORM_STEPS[3..].to_vec()), - (failed(FORM_STEPS[3]), Vec::new()), - ] - ); -} - -/// Gives the namespace of the repositories of the agents of a phase of `MIXED_TINY`. -fn mixed_tiny_namespace() -> BlobStorageNamespace { - repository_scope(MIXED_TINY.name, "no-limit", FILES_TINY.name).0 -} - -#[test] -async fn a_mixed_phase_whose_restores_fail_fails_at_the_mixed_step_with_the_error_of_a_restore() { - // The repository of agent 1 has a config that no key opens, so the copy of the first - // repository leaves it out, and its restore fails. The saves go to other agents and succeed. - let storage = Arc::new(InMemoryBlobStorage::new()); - let (save, _save_pod) = run_tiny(&MIXED_TINY, "save", &storage).await; - storage - .put_raw( - "test", - "test", - mixed_tiny_namespace(), - Path::new("agents/1/config"), - b"not a config", - ) - .await - .unwrap(); - - let (mixed, _pod) = run_tiny(&MIXED_TINY, "mixed-s2-r2", &storage).await; - let details = &step(&mixed, "mixed").details; - - assert_eq!( - ( - save.outcome, - mixed.outcome.clone(), - step(&mixed, "mixed").status != StepStatus::Ok, - details["first_error"].is_string(), - [&details["saves"]["failed"], &details["restores"]["failed"]].map(Value::clone), - skipped_steps(&mixed), - tree_counts(&mixed), - ), - ( - Outcome::Ok, - Outcome::Failed { - reason: "the step mixed failed".into() - }, - true, - true, - [json!(0), json!(1)], - vec!["hash_trees"], - FILES_TINY_COUNTS, - ) - ); -} - -#[test] -async fn the_restores_of_a_mixed_phase_read_the_copies_and_not_the_first_repository() { - // The copies exist before the phase, so the phase copies nothing, and the first repository - // loses its snapshot files. A restore that reads the first repository finds no snapshot. - let storage = Arc::new(InMemoryBlobStorage::new()); - let (save, _save_pod) = run_tiny(&MIXED_TINY, "save", &storage).await; - super::agents::copy_first_agent(storage.as_ref(), &mixed_tiny_namespace(), 5) - .await - .unwrap(); - storage - .delete_dir( - "test", - "test", - mixed_tiny_namespace(), - Path::new("agents/0/snapshots"), - ) - .await - .unwrap(); - - let (mixed, _pod) = run_tiny(&MIXED_TINY, "mixed-s2-r2", &storage).await; - - assert_eq!( - ( - save.outcome, - mixed.outcome.clone(), - detail(&mixed, "copy_scopes", "/agents_copied"), - detail(&mixed, "hash_trees", "/matches"), - ), - (Outcome::Ok, Outcome::Ok, json!(0), json!(4)) - ); -} - -#[test] -fn each_cpu_options_variant_changes_only_its_own_setting_of_the_defaults() { - let defaults = RepositorySettings::DEFAULT; - let variants = CPU_OPTIONS_PHASES - .iter() - .map(|phase| { - ( - phase.name, - phase.variant.settings.map(|settings| settings.repository), - phase.variant.settings.map(|settings| settings.save), - ) - }) - .collect::>(); - let level = |level| Compression::Level(NonZeroI32::new(level).unwrap()); - - assert_eq!( - variants, - [ - ("save-default", defaults), - ( - "save-verify-off", - RepositorySettings { - extra_verify: false, - ..defaults - } - ), - ( - "save-zstd-off", - RepositorySettings { - compression: Compression::Off, - ..defaults - } - ), - ( - "save-zstd-1", - RepositorySettings { - compression: level(1), - ..defaults - } - ), - ( - "save-zstd-9", - RepositorySettings { - compression: level(9), - ..defaults - } - ), - ] - .map(|(name, repository)| (name, Some(repository), Some(SaveSettings::DEFAULT))) - .to_vec() - ); -} - -#[test] -fn a_save_threads_phase_saves_with_its_threads() { - let scenario = super::scenario("save-threads").unwrap(); - let context = |phase: &str| PhaseContext { - run_id: "run-1".into(), - cpu_setting: "no-limit".into(), - selection: Selection::find("save-threads", "files-1g", phase).unwrap(), - work_dir: Path::new("/nowhere").into(), - storage: Arc::new(MeasuredBlobStorage::new(Arc::new( - InMemoryBlobStorage::new(), - ))), - }; - - assert_eq!( - ( - scenario.phases.len(), - ["save-x4-t1", "save-x4-t2", "save-x4-t4", "save-x4-tdefault"] - .map(|phase| context(phase).save_settings()), - ), - ( - 4, - [ - NonZeroUsize::new(1), - NonZeroUsize::new(2), - NonZeroUsize::new(4), - None - ] - .map(|threads| SaveSettings { - threads, - ..SaveSettings::DEFAULT - }), - ) - ); -} - -#[test] -async fn a_save_and_open_phase_whose_save_fails_skips_the_open() { - // The cold save writes the first snapshot file of the repository, and the last warm save - // writes the last one: the second for the cpu-options phase, and the third for the - // sqlite-changes phase, which also saves after its clustered change. - let fail_last_save = async |scenario: &'static Scenario, phase, agent, snapshots: usize| { - let storage = - FailingSnapshotWrites::new(Arc::new(InMemoryBlobStorage::new()), agent, snapshots - 1); - let (result, _pod) = run_tiny_with(scenario, phase, storage, |_| {}).await; - (result.outcome.clone(), skipped_steps(&result)) - }; - - let outcomes = [ - fail_last_save(&CPU_OPTIONS_TINY, "save-default", "default", 2).await, - fail_last_save(&SQLITE_CHANGES_TINY, "save-rabin", "rabin", 3).await, - ]; - - assert_eq!( - outcomes, - [ - ( - Outcome::Failed { - reason: "the step warm_save failed".into() - }, - vec!["hash_tree", "open"] - ), - ( - Outcome::Failed { - reason: "the step warm_save_scattered failed".into() - }, - vec!["hash_tree", "open"] - ), - ] - ); -} - -#[test] -async fn a_save_and_open_phase_fails_at_the_open_when_the_open_finds_no_repository() { - // The config of the repository disappears after the second snapshot file, which the warm - // save writes last, so the open after the saves finds no repository. - let storage = FailingSnapshotWrites::hiding_config( - Arc::new(InMemoryBlobStorage::new()), - "default", - usize::MAX, - 2, - ); - - let (result, _pod) = run_tiny_with(&CPU_OPTIONS_TINY, "save-default", storage, |_| {}).await; - - assert_eq!( - ( - result.outcome.clone(), - step_states(&result).last().copied(), - step(&result, "open").details.clone(), - skipped_steps(&result), - ), - ( - Outcome::Failed { - reason: "the step open failed".into() - }, - Some(("open", true)), - json!({ "repository": null }), - Vec::<&str>::new(), - ) - ); -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/trees.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/trees.rs deleted file mode 100644 index 4f2e3daf67..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/trees.rs +++ /dev/null @@ -1,1324 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The trees of the benchmark: how a tree is made, how it changes, and its hash. -//! -//! The content of a tree does not compress, unless the tree has compressible content. After each -//! write, the pages of the written files leave the page cache, so a save reads them from the -//! volume. The upload of a reflink capture reads its files from the volume too, because a clone -//! does not share the page cache of its source. - -use anyhow::Context; -use futures::{StreamExt, TryStreamExt}; -use serde_json::{Value, json}; -use sqlx::sqlite::{SqliteConnectOptions, SqliteJournalMode, SqliteSynchronous}; -use sqlx::{ConnectOptions, Connection, Executor, SqliteConnection}; -use std::fs::{File, Metadata}; -use std::io; -use std::os::unix::ffi::OsStrExt; -use std::os::unix::fs::{MetadataExt, PermissionsExt}; -use std::path::{Path, PathBuf}; - -const MIB: u64 = 1024 * 1024; - -/// The name of the database file of a SQLite tree. -const DATABASE: &str = "database.sqlite"; - -/// The rows that one insert of a SQLite tree adds. -const ROWS_PER_INSERT: u64 = 1_000; - -/// The bytes of the payload of one row of a SQLite tree. -const ROW_PAYLOAD_BYTES: u64 = 1_024; - -/// The number of files that a small change rewrites, and the rows that it updates. -const CHANGED_FILES: u64 = 10; -const CHANGED_ROWS: u64 = 100; - -/// The number of files in each directory of a tree at the object limit, as in the other file -/// trees. -const FILES_PER_DIRECTORY: u64 = 100; - -/// The directories of the first level of a modules tree. Each has 10 directories, and each of -/// those has 10 directories that hold the files. -const MODULE_PACKAGES: u64 = 25; -const MODULE_FANOUT: u64 = 10; - -/// The bytes of a block of compressible content. Its second half is zero. -const COMPRESSIBLE_BLOCK: usize = 64; - -/// A tree that the benchmark makes. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) struct TreeSpec { - pub(super) name: &'static str, - pub(super) shape: TreeShape, - pub(super) content: Content, -} - -/// What the files of a tree hold. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) enum Content { - /// Bytes that do not compress. - Incompressible, - /// Blocks of 64 bytes, each with 32 bytes that do not compress and 32 zero bytes, so zstd - /// makes them about half as large. - Compressible, -} - -impl Content { - /// Gives the name of the content, which the result records. - pub(super) const fn label(self) -> &'static str { - match self { - Content::Incompressible => "incompressible", - Content::Compressible => "compressible", - } - } -} - -/// What a tree holds. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) enum TreeShape { - /// Files of about the same size, in directories of the same number of files. - Files { - files: u64, - directories: u64, - bytes: u64, - }, - /// One SQLite database of at least the size. - Sqlite { bytes: u64 }, - /// Files and directories that are `objects` filesystem objects together with the root, in - /// the layout that [`limit_layout`] gives. Its small change keeps the number of objects. - ObjectLimit { objects: u64, bytes: u64 }, - /// Files of about the same size in many small directories, as a `node_modules` tree has - /// them: the files fill the directories of the third level of [`MODULE_PACKAGES`] directories - /// with [`MODULE_FANOUT`] directories each, which again have [`MODULE_FANOUT`] directories. - Modules { files: u64, bytes: u64 }, -} - -pub(super) const FILES_128M: TreeSpec = TreeSpec { - name: "files-128m", - shape: TreeShape::Files { - files: 10_000, - directories: 100, - bytes: 128 * MIB, - }, - content: Content::Incompressible, -}; - -pub(super) const FILES_1G: TreeSpec = TreeSpec { - name: "files-1g", - shape: TreeShape::Files { - files: 10_000, - directories: 100, - bytes: 1024 * MIB, - }, - content: Content::Incompressible, -}; - -/// The tree at the object limit of an agent with 128 MiB of storage: 8,192 objects. -pub(super) const OBJECTS_128M: TreeSpec = TreeSpec { - name: "objects-128m", - shape: TreeShape::ObjectLimit { - objects: 8_192, - bytes: 128 * MIB, - }, - content: Content::Incompressible, -}; - -/// The tree at the object limit of an agent with 1 GiB of storage: 32,768 objects. -pub(super) const OBJECTS_1G: TreeSpec = TreeSpec { - name: "objects-1g", - shape: TreeShape::ObjectLimit { - objects: 32_768, - bytes: 1024 * MIB, - }, - content: Content::Incompressible, -}; - -pub(super) const SQLITE_1G: TreeSpec = TreeSpec { - name: "sqlite-1g", - shape: TreeShape::Sqlite { bytes: 1024 * MIB }, - content: Content::Incompressible, -}; - -/// The 10,000 files and 128 MiB of [`FILES_128M`] in 2,775 directories, 3 levels deep. -pub(super) const MODULES_128M: TreeSpec = TreeSpec { - name: "modules-128m", - shape: TreeShape::Modules { - files: 10_000, - bytes: 128 * MIB, - }, - content: Content::Incompressible, -}; - -/// The layout of [`FILES_1G`], with content that compresses to about half its size. -pub(super) const COMPRESSIBLE_1G: TreeSpec = TreeSpec { - name: "compressible-1g", - shape: TreeShape::Files { - files: 10_000, - directories: 100, - bytes: 1024 * MIB, - }, - content: Content::Compressible, -}; - -pub(super) const FILES_TINY: TreeSpec = TreeSpec { - name: "files-tiny", - shape: TreeShape::Files { - files: 100, - directories: 10, - bytes: MIB, - }, - content: Content::Incompressible, -}; - -pub(super) const SQLITE_TINY: TreeSpec = TreeSpec { - name: "sqlite-tiny", - shape: TreeShape::Sqlite { bytes: 4 * MIB }, - content: Content::Incompressible, -}; - -/// The number of files, directories and bytes of a tree, not counting its root. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub(super) struct TreeCounts { - pub(super) files: u64, - pub(super) directories: u64, - pub(super) bytes: u64, -} - -/// Gives the number of files and the number of directories of a tree whose files, directories -/// and root are `objects` objects together. -/// -/// Each directory holds at most [`FILES_PER_DIRECTORY`] files, so a directory and its files are -/// at most `FILES_PER_DIRECTORY + 1` objects. -pub(super) const fn limit_layout(objects: u64) -> (u64, u64) { - let below_root = objects.saturating_sub(1); - let directories = below_root.div_ceil(FILES_PER_DIRECTORY + 1); - (below_root - directories, directories) -} - -/// Makes the tree in the directory `root`, which must not exist. -pub(super) async fn generate(spec: &TreeSpec, root: &Path) -> anyhow::Result { - std::fs::create_dir(root).with_context(|| format!("create the tree {}", root.display()))?; - match tree_content(spec) { - TreeContent::Files(files) => { - in_blocking(root, move |root| generate_files(root, &files)).await - } - TreeContent::Sqlite { bytes } => generate_database(&root.join(DATABASE), bytes).await, - }?; - settle(root)?; - Ok(count(root)?) -} - -/// Changes a small part of the tree in `root`, and gives what changed. It is the first round of -/// [`change_round`]. -pub(super) async fn change(spec: &TreeSpec, root: &Path) -> anyhow::Result { - change_round(spec, root, 1).await -} - -/// Changes a small part of the tree in `root` for the round, and gives what changed. Each round -/// from 1 to 255 writes other content. -/// -/// A files tree gets new content in the same files in each round, and one new file. A tree at the -/// object limit also loses one file in each round, so it keeps its number of objects. A SQLite -/// tree gets new payload in rows spread over the database. -pub(super) async fn change_round(spec: &TreeSpec, root: &Path, round: u8) -> anyhow::Result { - let change = match tree_content(spec) { - TreeContent::Files(files) => { - in_blocking(root, move |root| change_files(root, &files, round)).await? - } - TreeContent::Sqlite { .. } => { - change_database(&root.join(DATABASE), SqliteChange::Scattered).await? - } - }; - settle(root)?; - Ok(change) -} - -/// Gives new payload to 100 consecutive rows in the middle of the database of a SQLite tree, and -/// gives what changed. The rows are on about 34 consecutive pages. -pub(super) async fn change_clustered(spec: &TreeSpec, root: &Path) -> anyhow::Result { - anyhow::ensure!( - matches!(tree_content(spec), TreeContent::Sqlite { .. }), - "the tree {} has no database", - spec.name - ); - let change = change_database(&root.join(DATABASE), SqliteChange::Clustered).await?; - settle(root)?; - Ok(change) -} - -/// Runs the work on the tree `root` on a blocking thread. -async fn in_blocking( - root: &Path, - work: impl FnOnce(&Path) -> anyhow::Result + Send + 'static, -) -> anyhow::Result { - let root = root.to_path_buf(); - tokio::task::spawn_blocking(move || work(&root)).await? -} - -/// Where the files of a tree are. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) enum Layout { - /// The files fill `directories` directories below the root in their order. - Flat { files: u64, directories: u64 }, - /// The files fill the directories of the third level of a modules tree in their order. - Modules { files: u64 }, -} - -impl Layout { - fn files(self) -> u64 { - match self { - Layout::Flat { files, .. } | Layout::Modules { files } => files, - } - } - - /// Gives the path of the file with the index, relative to the root of the tree. - pub(super) fn path(self, index: u64) -> Box { - match self { - Layout::Flat { files, directories } => file_path(index, files, directories), - Layout::Modules { files } => { - let leaves = MODULE_PACKAGES * MODULE_FANOUT * MODULE_FANOUT; - let leaf = index / files.div_ceil(leaves).max(1); - PathBuf::from(format!( - "p{:02}/s{}/l{}/f{index:05}", - leaf / (MODULE_FANOUT * MODULE_FANOUT), - leaf / MODULE_FANOUT % MODULE_FANOUT, - leaf % MODULE_FANOUT - )) - .into_boxed_path() - } - } - } -} - -/// The files of a tree that is not a SQLite tree. -#[derive(Clone, Copy, Debug)] -struct Files { - layout: Layout, - bytes: u64, - replace: Replace, - content: Content, -} - -/// What a tree holds: files, or one SQLite database. -#[derive(Clone, Copy, Debug)] -enum TreeContent { - Files(Files), - /// One SQLite database of at least the size. - Sqlite { - bytes: u64, - }, -} - -/// Gives what the tree holds. -fn tree_content(spec: &TreeSpec) -> TreeContent { - let (layout, bytes, replace) = match spec.shape { - TreeShape::Files { - files, - directories, - bytes, - } => (Layout::Flat { files, directories }, bytes, Replace::No), - TreeShape::ObjectLimit { objects, bytes } => { - let (files, directories) = limit_layout(objects); - (Layout::Flat { files, directories }, bytes, Replace::Yes) - } - TreeShape::Modules { files, bytes } => (Layout::Modules { files }, bytes, Replace::No), - TreeShape::Sqlite { bytes } => return TreeContent::Sqlite { bytes }, - }; - TreeContent::Files(Files { - layout, - bytes, - replace, - content: spec.content, - }) -} - -/// Gives the path of the file with the index in a flat tree, relative to the root of the tree. -fn file_path(index: u64, files: u64, directories: u64) -> Box { - let per_directory = files.div_ceil(directories.max(1)).max(1); - PathBuf::from(format!("d{:03}/f{index:05}", index / per_directory)).into_boxed_path() -} - -/// Gives the size of the file with the index. The sizes add up to `bytes`. -fn file_size(index: u64, files: u64, bytes: u64) -> u64 { - bytes / files.max(1) + u64::from(index < bytes % files.max(1)) -} - -/// Gives `size` bytes of the content, the same for the same path and generation. -fn content(path: &Path, generation: u8, size: u64, kind: Content) -> io::Result> { - let mut content = vec![0; usize::try_from(size).map_err(io::Error::other)?].into_boxed_slice(); - blake3::Hasher::new_derive_key("golem fs-snapshot benchmark file content") - .update(&[generation]) - .update(path.as_os_str().as_bytes()) - .finalize_xof() - .fill(&mut content); - if kind == Content::Compressible { - content.chunks_mut(COMPRESSIBLE_BLOCK).for_each(|block| { - block - .iter_mut() - .skip(COMPRESSIBLE_BLOCK / 2) - .for_each(|byte| *byte = 0) - }); - } - Ok(content) -} - -fn write_file( - root: &Path, - relative: &Path, - generation: u8, - size: u64, - kind: Content, -) -> io::Result<()> { - let path = root.join(relative); - std::fs::write(&path, content(relative, generation, size, kind)?)?; - std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o644)) -} - -/// Makes each directory of the relative path below `root` that does not exist, with the -/// permission bits `0o755`. -fn make_directories(root: &Path, relative: &Path) -> io::Result<()> { - relative - .ancestors() - .filter(|ancestor| !ancestor.as_os_str().is_empty()) - .collect::>() - .iter() - .rev() - .map(|ancestor| root.join(ancestor)) - .filter(|directory| !directory.exists()) - .try_for_each(|directory| { - std::fs::create_dir(&directory)?; - std::fs::set_permissions(&directory, std::fs::Permissions::from_mode(0o755)) - }) -} - -fn generate_files(root: &Path, files: &Files) -> anyhow::Result<()> { - let count = files.layout.files(); - (0..count).try_for_each(|index| { - let relative = files.layout.path(index); - if let Some(parent) = relative.parent() { - make_directories(root, parent)?; - } - write_file( - root, - &relative, - 0, - file_size(index, count, files.bytes), - files.content, - ) - })?; - Ok(()) -} - -/// Whether the small change of a files tree replaces a file, or only adds one. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum Replace { - /// The change deletes a file, and adds a file with a new name in its directory. So the tree - /// keeps its number of objects. - Yes, - /// The change adds a file in the first directory. - No, -} - -/// Gives the file name of the round: the name, and for a round after the first, the name with -/// the round. -fn file_name_of_round(name: &str, round: u8) -> String { - if round == 1 { - name.to_string() - } else { - format!("{name}-{round}") - } -} - -fn change_files(root: &Path, files: &Files, round: u8) -> anyhow::Result { - let count = files.layout.files(); - let size = |index| file_size(index, count, files.bytes); - let step = (count / CHANGED_FILES).max(1); - let rewritten = (0..CHANGED_FILES.min(count)) - .map(|position| position * step) - .try_fold(0_u64, |written, index| { - write_file( - root, - &files.layout.path(index), - round, - size(index), - files.content, - ) - .map(|()| written + size(index)) - })?; - // The added file has the size of the file with the index. Round `k` deletes the `k`-th file - // from the end. - let (added_path, added_index, deleted) = match files.replace { - Replace::Yes => { - let deleted = count.saturating_sub(u64::from(round)); - let path = files.layout.path(deleted); - std::fs::remove_file(root.join(&path))?; - ( - path.with_file_name(file_name_of_round("replaced", round)), - deleted, - 1, - ) - } - Replace::No => ( - files - .layout - .path(0) - .with_file_name(file_name_of_round("added", round)), - 0, - 0, - ), - }; - let added = size(added_index); - write_file(root, &added_path, 0, added, files.content)?; - Ok(json!({ - "files_rewritten": CHANGED_FILES.min(count), - "files_added": 1, - "files_deleted": deleted, - "rows_updated": 0, - "bytes": rewritten + added, - })) -} - -async fn connect(path: &Path) -> anyhow::Result { - Ok(SqliteConnectOptions::new() - .filename(path) - .create_if_missing(true) - .journal_mode(SqliteJournalMode::Delete) - .synchronous(SqliteSynchronous::Normal) - .page_size(4096) - .connect() - .await?) -} - -/// Inserts rows of random payload until the database file has at least `bytes` bytes. -async fn generate_database(path: &Path, bytes: u64) -> anyhow::Result<()> { - let mut connection = connect(path).await?; - connection - .execute("CREATE TABLE rows (id INTEGER PRIMARY KEY, payload BLOB NOT NULL)") - .await?; - // Each row takes more than its payload, so this number of inserts is more than enough. - let inserts = bytes / (ROWS_PER_INSERT * ROW_PAYLOAD_BYTES) + 2; - let connection = futures::stream::iter(0..inserts) - .map(Ok::<_, anyhow::Error>) - .try_fold(connection, |mut connection, _| async move { - if std::fs::metadata(path)?.len() < bytes { - sqlx::query( - "WITH RECURSIVE counter(n) AS (SELECT 1 UNION ALL SELECT n + 1 FROM counter WHERE n < ?1) \ - INSERT INTO rows (payload) SELECT randomblob(?2) FROM counter", - ) - .bind(ROWS_PER_INSERT as i64) - .bind(ROW_PAYLOAD_BYTES as i64) - .execute(&mut connection) - .await?; - } - Ok(connection) - }) - .await?; - connection.close().await?; - Ok(()) -} - -/// Which rows a change of a database updates. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum SqliteChange { - /// Rows spread over the whole table. - Scattered, - /// Consecutive rows in the middle of the table. - Clustered, -} - -/// Updates the payload of [`CHANGED_ROWS`] rows of the database. -async fn change_database(path: &Path, pattern: SqliteChange) -> anyhow::Result { - let mut connection = connect(path).await?; - let last: i64 = sqlx::query_scalar("SELECT max(id) FROM rows") - .fetch_one(&mut connection) - .await?; - let rows = CHANGED_ROWS as i64; - let (first, step) = match pattern { - SqliteChange::Scattered => (1, (last / rows).max(1)), - SqliteChange::Clustered => ((last / 2).max(1), 1), - }; - let ids = (0..rows) - .map(|position| (first + position * step).to_string()) - .collect::>() - .join(","); - let updated = sqlx::query(&format!( - "UPDATE rows SET payload = randomblob({ROW_PAYLOAD_BYTES}) WHERE id IN ({ids})" - )) - .execute(&mut connection) - .await? - .rows_affected(); - connection.close().await?; - Ok(json!({ - "files_rewritten": 0, - "files_added": 0, - "files_deleted": 0, - "rows_updated": updated, - "bytes": updated * ROW_PAYLOAD_BYTES, - })) -} - -/// How the files of a copy of a tree were made. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub(super) struct CopyCounts { - /// The files that share their data with the source, through a reflink. - pub(super) reflinked: u64, - /// The files whose bytes were copied. - pub(super) copied: u64, -} - -impl CopyCounts { - pub(super) fn with(self, other: Self) -> Self { - Self { - reflinked: self.reflinked + other.reflinked, - copied: self.copied + other.copied, - } - } -} - -/// Whether a copy of a tree keeps the modification times of its entries. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) enum Times { - /// Each file and directory of the copy, and its root, gets the modification time of its - /// source, as the capture of an agent tree gives them. - Keep, - /// The copy does not keep the modification times. - Drop, -} - -/// Copies the tree `from` into the directory `to`, which must not exist, with the permission -/// bits of each entry, and with the modification times when `times` keeps them. A file is a -/// reflink of its source where the filesystem has reflinks (`FICLONE`, for example on XFS), and a -/// copy of its bytes where it does not. The copy does not sync the volume. -pub(super) fn copy_tree(from: &Path, to: &Path, times: Times) -> anyhow::Result { - std::fs::create_dir(to).with_context(|| format!("create the tree {}", to.display()))?; - let (counts, directories) = walk( - from, - (CopyCounts::default(), Vec::new()), - &mut |(counts, mut directories), path, metadata| { - let target = to.join(path.strip_prefix(from).map_err(io::Error::other)?); - if metadata.is_dir() { - std::fs::create_dir(&target)?; - std::fs::set_permissions(&target, metadata.permissions())?; - if times == Times::Keep { - directories.push((target, metadata.modified()?)); - } - Ok((counts, directories)) - } else if metadata.is_file() { - let reflinked = copy_file(path, &target, metadata)?; - if times == Times::Keep { - File::options() - .write(true) - .open(&target)? - .set_modified(metadata.modified()?)?; - } - Ok(( - counts.with(if reflinked { - CopyCounts { - reflinked: 1, - copied: 0, - } - } else { - CopyCounts { - reflinked: 0, - copied: 1, - } - }), - directories, - )) - } else { - Err(io::Error::other(format!( - "the tree has an entry that is not a file or a directory: {}", - path.display() - ))) - } - }, - )?; - if times == Times::Keep { - // A directory gets its time after its entries, which change it, and the root last. - directories - .iter() - .rev() - .map(|(directory, modified)| (directory.as_path(), *modified)) - .chain(std::iter::once((to, std::fs::metadata(from)?.modified()?))) - .try_for_each(|(directory, modified)| File::open(directory)?.set_modified(modified))?; - } - Ok(counts) -} - -/// Copies the file `from` to the new file `to`, and tells whether the copy is a reflink. -/// -/// When the reflink fails, the filesystem cannot share the data of the files, so the copy of the -/// bytes after it does not share them either. -fn copy_file(from: &Path, to: &Path, metadata: &Metadata) -> io::Result { - let source = File::open(from)?; - let target = File::create_new(to)?; - match rustix::fs::ioctl_ficlone(&target, &source) { - Ok(()) => { - target.set_permissions(metadata.permissions())?; - Ok(true) - } - Err(_) => { - drop(target); - std::fs::copy(from, to)?; - Ok(false) - } - } -} - -/// Writes each change of the filesystem of `root` to the volume, and removes the pages of each -/// file below `root` from the page cache. -pub(super) fn settle(root: &Path) -> anyhow::Result<()> { - rustix::fs::syncfs(File::open(root)?)?; - walk(root, (), &mut |(), path, metadata| { - if metadata.is_file() { - rustix::fs::fadvise(File::open(path)?, 0, None, rustix::fs::Advice::DontNeed)?; - } - Ok(()) - })?; - Ok(()) -} - -fn count(root: &Path) -> io::Result { - walk(root, TreeCounts::default(), &mut |counts, _, metadata| { - Ok(if metadata.is_dir() { - TreeCounts { - directories: counts.directories + 1, - ..counts - } - } else if metadata.is_file() { - TreeCounts { - files: counts.files + 1, - bytes: counts.bytes + metadata.len(), - ..counts - } - } else { - counts - }) - }) -} - -/// Gives the hash of the tree below `root` and its counts, and reads the tree on a blocking thread. -pub(super) async fn hash(root: &Path) -> anyhow::Result<(Box, TreeCounts)> { - let root = root.to_path_buf(); - Ok(tokio::task::spawn_blocking(move || tree_hash(&root)).await??) -} - -/// Gives the hash of the tree below `root` and its counts. -/// -/// The hash covers the path, the kind, the permission bits and the modification time of each -/// entry, the size and the content of each file, and the target of each symlink. It does not -/// cover the owner or the metadata of `root`. -pub(super) fn tree_hash(root: &Path) -> io::Result<(Box, TreeCounts)> { - let hasher = walk( - root, - blake3::Hasher::new(), - &mut |mut hasher, path, metadata| { - let relative = path.strip_prefix(root).map_err(io::Error::other)?; - let name = relative.as_os_str().as_bytes(); - hasher.update(&(name.len() as u64).to_le_bytes()); - hasher.update(name); - hasher.update(&metadata.mode().to_le_bytes()); - hasher.update(&metadata.mtime().to_le_bytes()); - hasher.update(&metadata.mtime_nsec().to_le_bytes()); - if metadata.is_file() { - hasher.update(b"f"); - hasher.update(&metadata.len().to_le_bytes()); - let mut content = blake3::Hasher::new(); - content.update_reader(File::open(path)?)?; - hasher.update(content.finalize().as_bytes()); - } else if metadata.is_symlink() { - let target = std::fs::read_link(path)?; - hasher.update(b"l"); - hasher.update(&(target.as_os_str().len() as u64).to_le_bytes()); - hasher.update(target.as_os_str().as_bytes()); - } else if metadata.is_dir() { - hasher.update(b"d"); - } else { - hasher.update(b"o"); - } - Ok(hasher) - }, - )?; - Ok((hasher.finalize().to_hex().as_str().into(), count(root)?)) -} - -/// Visits each entry below `directory`, in the order of the names, parents before children, and -/// folds the visits into one value. Symlinks are not followed. -fn walk( - directory: &Path, - initial: T, - visit: &mut impl FnMut(T, &Path, &Metadata) -> io::Result, -) -> io::Result { - let mut names = std::fs::read_dir(directory)? - .map(|entry| entry.map(|entry| entry.file_name())) - .collect::>>()?; - names.sort(); - names.into_iter().try_fold(initial, |value, name| { - let path = directory.join(name); - let metadata = std::fs::symlink_metadata(&path)?; - let value = visit(value, &path, &metadata)?; - if metadata.is_dir() { - walk(&path, value, visit) - } else { - Ok(value) - } - }) -} - -#[cfg(test)] -mod tests { - use super::{ - COMPRESSIBLE_1G, COMPRESSIBLE_BLOCK, Content, DATABASE, FILES_1G, FILES_128M, FILES_TINY, - Layout, MIB, MODULES_128M, SQLITE_TINY, Times, TreeCounts, TreeShape, TreeSpec, change, - change_clustered, change_round, connect, content, copy_tree, file_path, file_size, - generate, limit_layout, tree_hash, walk, - }; - use pretty_assertions::assert_eq; - use sqlx::Connection; - use std::os::unix::fs::{MetadataExt, PermissionsExt}; - use std::path::{Path, PathBuf}; - use std::time::{Duration, SystemTime}; - use test_r::test; - - /// A tree at the object limit of 128 objects: 125 files in 2 directories, and the root. - const OBJECTS_TINY: TreeSpec = TreeSpec { - name: "objects-tiny", - shape: TreeShape::ObjectLimit { - objects: 128, - bytes: MIB, - }, - content: Content::Incompressible, - }; - - fn hash(root: &Path) -> Box { - tree_hash(root).unwrap().0 - } - - /// Gives the path, the permission bits and the content of each entry below `root`, in the - /// order of the paths. A directory has no content. - fn contents(root: &Path) -> Vec<(PathBuf, u32, Vec)> { - walk(root, Vec::new(), &mut |mut entries, path, metadata| { - let content = if metadata.is_file() { - std::fs::read(path)? - } else { - Vec::new() - }; - entries.push(( - path.strip_prefix(root).unwrap().to_path_buf(), - metadata.mode(), - content, - )); - Ok(entries) - }) - .unwrap() - } - - #[test] - fn the_limit_layout_counts_the_root_and_fills_each_directory() { - let layout = |objects| { - let (files, directories) = limit_layout(objects); - ( - files, - directories, - files + directories + 1, - file_path(files - 1, files, directories), - ) - }; - - assert_eq!( - [layout(8_192), layout(32_768), layout(128)], - [ - (8_109, 82, 8_192, Path::new("d081/f08108").into()), - (32_442, 325, 32_768, Path::new("d324/f32441").into()), - (125, 2, 128, Path::new("d001/f00124").into()), - ] - ); - } - - #[test] - async fn a_tree_at_the_object_limit_keeps_its_number_of_objects_after_its_change() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - - let counts = generate(&OBJECTS_TINY, &root).await.unwrap(); - let before = hash(&root); - let changed = change(&OBJECTS_TINY, &root).await.unwrap(); - let (after, after_counts) = tree_hash(&root).unwrap(); - - assert_eq!( - ( - counts, - after_counts, - [ - &changed["files_rewritten"], - &changed["files_added"], - &changed["files_deleted"] - ] - .map(|value| value.as_u64()), - root.join("d001/f00124").exists(), - root.join("d001/replaced").exists(), - before != after, - ), - ( - TreeCounts { - files: 125, - directories: 2, - bytes: MIB, - }, - TreeCounts { - files: 125, - directories: 2, - bytes: MIB, - }, - [Some(10), Some(1), Some(1)], - false, - true, - true, - ) - ); - } - - #[test] - async fn a_copy_of_a_tree_has_its_entries_permissions_and_content() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - let copy = work.path().join("copy"); - generate(&FILES_TINY, &root).await.unwrap(); - std::fs::set_permissions( - root.join("d000/f00000"), - std::fs::Permissions::from_mode(0o600), - ) - .unwrap(); - std::fs::set_permissions(root.join("d001"), std::fs::Permissions::from_mode(0o700)) - .unwrap(); - - let counts = copy_tree(&root, ©, Times::Drop).unwrap(); - - assert_eq!( - (counts.reflinked + counts.copied, contents(©)), - (100, contents(&root)) - ); - } - - #[test] - fn the_file_sizes_add_up_and_the_files_fill_the_directories() { - assert_eq!( - ( - (0..7).map(|index| file_size(index, 7, 100)).sum::(), - (0..7) - .map(|index| file_size(index, 7, 100)) - .collect::>(), - file_path(0, 100, 10), - file_path(19, 100, 10), - file_path(99, 100, 10), - ), - ( - 100, - vec![15, 15, 14, 14, 14, 14, 14], - Path::new("d000/f00000").into(), - Path::new("d001/f00019").into(), - Path::new("d009/f00099").into(), - ) - ); - } - - #[test] - async fn a_files_tree_has_its_files_directories_and_bytes_and_a_change_changes_it() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - - let counts = generate(&FILES_TINY, &root).await.unwrap(); - let before = hash(&root); - let changed = change(&FILES_TINY, &root).await.unwrap(); - let (after, after_counts) = tree_hash(&root).unwrap(); - - assert_eq!( - ( - counts, - changed["files_rewritten"].as_u64(), - changed["files_added"].as_u64(), - changed["files_deleted"].as_u64(), - after_counts.files, - before != after, - ), - ( - TreeCounts { - files: 100, - directories: 10, - bytes: 1024 * 1024, - }, - Some(10), - Some(1), - Some(0), - 101, - true, - ) - ); - } - - #[test] - async fn a_sqlite_tree_has_one_database_of_at_least_its_size() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - - let counts = generate(&SQLITE_TINY, &root).await.unwrap(); - let before = hash(&root); - let changed = change(&SQLITE_TINY, &root).await.unwrap(); - - assert_eq!( - ( - counts.files, - counts.bytes >= 4 * 1024 * 1024, - counts.bytes < 6 * 1024 * 1024, - changed["rows_updated"].as_u64(), - before != hash(&root), - ), - (1, true, true, Some(100), true) - ); - } - - #[test] - fn the_tree_hash_changes_with_each_kept_attribute() { - let work = tempfile::tempdir().unwrap(); - let root = work.path(); - let time = SystemTime::UNIX_EPOCH + Duration::new(1_700_000_000, 5); - let reset = |root: &Path| { - let _ = std::fs::remove_dir_all(root.join("dir")); - let _ = std::fs::remove_file(root.join("file")); - let _ = std::fs::remove_file(root.join("other")); - let _ = std::fs::remove_file(root.join("link")); - std::fs::create_dir(root.join("dir")).unwrap(); - std::fs::write(root.join("file"), b"content").unwrap(); - std::fs::set_permissions(root.join("file"), std::fs::Permissions::from_mode(0o644)) - .unwrap(); - std::os::unix::fs::symlink("file", root.join("link")).unwrap(); - std::fs::File::options() - .write(true) - .open(root.join("file")) - .unwrap() - .set_modified(time) - .unwrap(); - fs_set_times::set_times( - root.join("dir"), - None, - Some(fs_set_times::SystemTimeSpec::Absolute(time)), - ) - .unwrap(); - fs_set_times::set_symlink_times( - root.join("link"), - None, - Some(fs_set_times::SystemTimeSpec::Absolute(time)), - ) - .unwrap(); - }; - let changed = |change: &dyn Fn(&Path)| { - reset(root); - change(root); - hash(root) - }; - reset(root); - let original = hash(root); - - let hashes = [ - changed(&|_| {}), - changed(&|root| { - std::fs::write(root.join("file"), b"CONTENT").unwrap(); - std::fs::File::options() - .write(true) - .open(root.join("file")) - .unwrap() - .set_modified(time) - .unwrap(); - }), - changed(&|root| { - std::fs::set_permissions(root.join("file"), std::fs::Permissions::from_mode(0o600)) - .unwrap() - }), - changed(&|root| { - std::fs::File::options() - .write(true) - .open(root.join("file")) - .unwrap() - .set_modified(time + Duration::from_nanos(1)) - .unwrap() - }), - changed(&|root| { - std::fs::remove_file(root.join("link")).unwrap(); - std::os::unix::fs::symlink("elsewhere", root.join("link")).unwrap(); - fs_set_times::set_symlink_times( - root.join("link"), - None, - Some(fs_set_times::SystemTimeSpec::Absolute(time)), - ) - .unwrap(); - }), - changed(&|root| std::fs::rename(root.join("file"), root.join("other")).unwrap()), - changed(&|root| { - std::fs::create_dir(root.join("dir/empty")).unwrap(); - [root.join("dir/empty"), root.join("dir")] - .iter() - .for_each(|directory| { - fs_set_times::set_times( - directory, - None, - Some(fs_set_times::SystemTimeSpec::Absolute(time)), - ) - .unwrap() - }); - }), - ]; - - assert_eq!( - hashes - .iter() - .map(|hash| *hash == original) - .collect::>(), - vec![true, false, false, false, false, false, false] - ); - } - - #[test] - async fn a_copy_that_keeps_the_times_has_the_hash_of_its_source_and_new_inodes() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - let kept = work.path().join("kept"); - let dropped = work.path().join("dropped"); - generate(&FILES_TINY, &root).await.unwrap(); - let old = SystemTime::UNIX_EPOCH + Duration::from_secs(1_700_000_000); - walk(&root, (), &mut |(), path, _| { - std::fs::File::open(path).and_then(|file| file.set_modified(old)) - }) - .unwrap(); - std::fs::File::open(&root) - .unwrap() - .set_modified(old) - .unwrap(); - let inode = |root: &Path| std::fs::metadata(root.join("d000/f00000")).unwrap().ino(); - - let counts = copy_tree(&root, &kept, Times::Keep).unwrap(); - copy_tree(&root, &dropped, Times::Drop).unwrap(); - - assert_eq!( - ( - counts.reflinked + counts.copied, - hash(&kept) == hash(&root), - hash(&dropped) == hash(&root), - std::fs::metadata(&kept).unwrap().modified().unwrap(), - inode(&kept) == inode(&root), - ), - (100, true, false, old, false) - ); - } - - #[test] - async fn each_round_of_a_change_writes_other_content_and_keeps_the_object_limit() { - let work = tempfile::tempdir().unwrap(); - let files = work.path().join("files"); - let objects = work.path().join("objects"); - generate(&FILES_TINY, &files).await.unwrap(); - generate(&OBJECTS_TINY, &objects).await.unwrap(); - - let rounds = futures::future::join_all( - [(&FILES_TINY, &files), (&OBJECTS_TINY, &objects)].map(|(spec, root)| async move { - let first = change_round(spec, root, 1).await.unwrap(); - let after_first = tree_hash(root).unwrap(); - change_round(spec, root, 2).await.unwrap(); - let after_second = tree_hash(root).unwrap(); - ( - first["files_rewritten"].as_u64(), - first["bytes"].as_u64(), - after_first.0 != after_second.0, - after_first.1.files, - after_second.1.files, - ) - }), - ) - .await; - - assert_eq!( - ( - rounds, - files.join("d000/added-2").exists(), - objects.join("d001/replaced-2").exists(), - objects.join("d001/f00123").exists(), - ), - ( - // FILES_TINY: the files 0, 10, ..., 70 have 10,486 bytes and the files 80 and 90 - // have 10,485, and the added file has the 10,486 bytes of the file 0. - // OBJECTS_TINY: the files 0, 12, ..., 72 have 8,389 bytes, the files 84, 96 and - // 108 have 8,388, and the added file has the 8,388 bytes of the file 124. - vec![ - ( - Some(10), - Some(8 * 10_486 + 2 * 10_485 + 10_486), - true, - 101, - 102 - ), - ( - Some(10), - Some(7 * 8_389 + 3 * 8_388 + 8_388), - true, - 125, - 125 - ), - ], - true, - true, - false, - ) - ); - } - - #[test] - async fn a_modules_tree_has_its_files_in_three_levels_of_small_directories() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - let spec = TreeSpec { - name: "modules-small", - shape: TreeShape::Modules { - files: 10_000, - bytes: 10_000, - }, - content: Content::Incompressible, - }; - let layout = Layout::Modules { files: 10_000 }; - - let counts = generate(&spec, &root).await.unwrap(); - - assert_eq!( - (counts, layout.path(0), layout.path(4), layout.path(9_999),), - ( - TreeCounts { - files: 10_000, - directories: 2_775, - bytes: 10_000, - }, - Path::new("p00/s0/l0/f00000").into(), - Path::new("p00/s0/l1/f00004").into(), - Path::new("p24/s9/l9/f09999").into(), - ) - ); - } - - #[test] - fn compressible_content_has_a_zero_second_half_in_each_block() { - let compressible = content(Path::new("f"), 0, 1_000, Content::Compressible).unwrap(); - let incompressible = content(Path::new("f"), 0, 1_000, Content::Incompressible).unwrap(); - let zero_halves = |content: &[u8]| { - content.chunks(COMPRESSIBLE_BLOCK).all(|block| { - block - .iter() - .skip(COMPRESSIBLE_BLOCK / 2) - .all(|byte| *byte == 0) - }) - }; - - assert_eq!( - ( - compressible.len(), - zero_halves(&compressible), - zero_halves(&incompressible), - compressible - .chunks(COMPRESSIBLE_BLOCK) - .zip(incompressible.chunks(COMPRESSIBLE_BLOCK)) - .all(|(left, right)| left[..COMPRESSIBLE_BLOCK / 2] - == right[..COMPRESSIBLE_BLOCK / 2]), - ), - (1_000, true, false, true) - ); - } - - #[test] - async fn a_scattered_and_a_clustered_change_update_the_rows_of_their_pattern() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - let files = work.path().join("files"); - generate(&SQLITE_TINY, &root).await.unwrap(); - generate(&FILES_TINY, &files).await.unwrap(); - let payloads = async |root: &Path| { - let mut connection = connect(&root.join(DATABASE)).await.unwrap(); - let rows: Vec<(i64, Vec)> = - sqlx::query_as("SELECT id, payload FROM rows ORDER BY id") - .fetch_all(&mut connection) - .await - .unwrap(); - connection.close().await.unwrap(); - rows - }; - let updated = |before: &[(i64, Vec)], after: &[(i64, Vec)]| { - before - .iter() - .zip(after) - .filter(|(old, new)| old.1 != new.1) - .map(|(old, _)| old.0) - .collect::>() - }; - let before = payloads(&root).await; - let last = before.last().unwrap().0; - - let scattered = change(&SQLITE_TINY, &root).await.unwrap(); - let after_scattered = payloads(&root).await; - let clustered = change_clustered(&SQLITE_TINY, &root).await.unwrap(); - let after_clustered = payloads(&root).await; - let refused = change_clustered(&FILES_TINY, &files).await; - - assert_eq!( - ( - scattered["rows_updated"].as_u64(), - updated(&before, &after_scattered), - clustered["rows_updated"].as_u64(), - updated(&after_scattered, &after_clustered), - refused.is_err() - ), - ( - Some(100), - (0..100) - .map(|row| 1 + row * (last / 100)) - .collect::>(), - Some(100), - (last / 2..last / 2 + 100).collect::>(), - true - ) - ); - } - - #[test] - fn the_later_trees_have_the_files_and_bytes_of_the_trees_that_they_follow() { - let (modules_files, modules_bytes) = match MODULES_128M.shape { - TreeShape::Modules { files, bytes } => (files, bytes), - _ => (0, 0), - }; - let (files_files, files_bytes) = match FILES_128M.shape { - TreeShape::Files { files, bytes, .. } => (files, bytes), - _ => (1, 1), - }; - - assert_eq!( - ( - (modules_files, modules_bytes), - COMPRESSIBLE_1G.shape, - COMPRESSIBLE_1G.content, - FILES_1G.content, - ), - ( - (files_files, files_bytes), - FILES_1G.shape, - Content::Compressible, - Content::Incompressible, - ) - ); - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/volume.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/volume.rs deleted file mode 100644 index 2f5c4f0d09..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/volume.rs +++ /dev/null @@ -1,555 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! Records the filesystem of the benchmark volume, without privileges. -//! -//! The record holds the filesystem type, the mount, the project quota flags of the mount, the -//! answer of `quotactl_fd` with `Q_XGETQSTATV`, and the XFS geometry that `xfs_info` shows. The -//! verdict uses only the filesystem type and the mount options. - -use serde_json::{Value, json}; -use std::fs::File; -use std::os::fd::AsRawFd; -use std::path::{Path, PathBuf}; - -const XFS_SUPER_MAGIC: u64 = 0x5846_5342; - -/// The quota type and the command of `quotactl_fd`, as the executor uses them to check the -/// project quota state at its start. -const XQM_PRJQUOTA: u32 = 2; -const Q_XGETQSTATV: u32 = (b'X' as u32) << 8 | 8; -const FS_QSTATV_VERSION1: i8 = 1; -const FS_QUOTA_PDQ_ACCT: u16 = 1 << 4; -const FS_QUOTA_PDQ_ENFD: u16 = 1 << 5; - -/// The size of `struct xfs_fsop_geom` of `XFS_IOC_FSGEOMETRY`, and of `struct xfs_fsop_geom_v4` -/// of `XFS_IOC_FSGEOMETRY_V4`. -const GEOMETRY_BYTES: usize = 256; -const GEOMETRY_V4_BYTES: usize = 112; - -/// The names of the bits of `xfs_fsop_geom.flags`, `XFS_FSOP_GEOM_FLAGS_*`. -const GEOMETRY_FLAGS: [(u32, &str); 26] = [ - (1 << 0, "attr"), - (1 << 1, "nlink"), - (1 << 2, "quota"), - (1 << 3, "ialign"), - (1 << 4, "dalign"), - (1 << 5, "shared"), - (1 << 6, "extflg"), - (1 << 7, "dirv2"), - (1 << 8, "logv2"), - (1 << 9, "sector"), - (1 << 10, "attr2"), - (1 << 11, "projid32"), - (1 << 12, "dirv2ci"), - (1 << 14, "lazysb"), - (1 << 15, "v5sb"), - (1 << 16, "ftype"), - (1 << 17, "finobt"), - (1 << 18, "spinodes"), - (1 << 19, "rmapbt"), - (1 << 20, "reflink"), - (1 << 21, "bigtime"), - (1 << 22, "inobtcnt"), - (1 << 23, "nrext64"), - (1 << 24, "exchange_range"), - (1 << 25, "parent"), - (1 << 26, "metadir"), -]; - -/// One line of `/proc/self/mountinfo`. -#[derive(Clone, Debug, PartialEq, Eq)] -pub(super) struct Mount { - pub(super) mount_point: Box, - pub(super) filesystem_type: Box, - pub(super) source: Box, - pub(super) mount_options: Box, - pub(super) super_options: Box, -} - -/// The project quota flags of a mount. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) struct ProjectQuota { - pub(super) accounting: bool, - pub(super) enforcement: bool, -} - -/// Records the volume that holds `path`. -pub(super) fn check(path: &Path) -> Value { - let path = std::fs::canonicalize(path).unwrap_or_else(|_| path.to_path_buf()); - let magic = rustix::fs::statfs(&path).map(|statfs| statfs.f_type as u64); - let mount = std::fs::read_to_string("/proc/self/mountinfo") - .ok() - .and_then(|text| mount_of(&text, &path)); - let quota = mount - .as_ref() - .map(|mount| project_quota(&mount.super_options)); - let directory = File::open(&path); - json!({ - "path": path.display().to_string(), - "filesystem_type": magic.as_ref().map(|magic| filesystem_name(*magic)).unwrap_or("unknown"), - "filesystem_magic": magic.as_ref().map(|magic| format!("{magic:#x}")).unwrap_or_else(|error| error.to_string()), - "mount": mount.as_ref().map(|mount| json!({ - "mount_point": mount.mount_point.display().to_string(), - "filesystem_type": mount.filesystem_type, - "source": mount.source, - "mount_options": mount.mount_options, - "super_options": mount.super_options, - })), - "project_quota": quota.map(|quota| json!({ - "accounting": quota.accounting, - "enforcement": quota.enforcement, - "source": "mount options", - })), - "quotactl": match &directory { - Ok(directory) => quota_state(directory), - Err(error) => json!({ "error": error.to_string() }), - }, - "xfs_geometry": match &directory { - Ok(directory) => geometry(directory), - Err(error) => json!({ "error": error.to_string() }), - }, - "check": verdict(magic.ok(), quota), - }) -} - -fn filesystem_name(magic: u64) -> &'static str { - match magic { - XFS_SUPER_MAGIC => "xfs", - 0xef53 => "ext4", - 0x0102_1994 => "tmpfs", - 0x9123_683e => "btrfs", - 0x794c_7630 => "overlay", - _ => "other", - } -} - -/// Gives the mount whose mount point is the longest prefix of `path`, from the text of -/// `/proc/self/mountinfo`. -pub(super) fn mount_of(mountinfo: &str, path: &Path) -> Option { - mountinfo - .lines() - .filter_map(parse_mount) - .filter(|mount| path.starts_with(&mount.mount_point)) - .max_by_key(|mount| mount.mount_point.components().count()) -} - -fn parse_mount(line: &str) -> Option { - let (mount, filesystem) = line.split_once(" - ")?; - let mount = mount.split(' ').collect::>(); - let filesystem = filesystem.split(' ').collect::>(); - Some(Mount { - mount_point: PathBuf::from(unescape(mount.get(4)?)).into_boxed_path(), - mount_options: (*mount.get(5)?).into(), - filesystem_type: (*filesystem.first()?).into(), - source: (*filesystem.get(1)?).into(), - super_options: (*filesystem.get(2)?).into(), - }) -} - -/// Replaces the octal escapes of mountinfo, such as `\040` for a space. -fn unescape(text: &str) -> String { - let bytes = text.as_bytes(); - let (decoded, _) = bytes.iter().enumerate().fold( - (Vec::with_capacity(bytes.len()), 0_usize), - |(mut decoded, skip), (index, byte)| { - if skip > 0 { - return (decoded, skip - 1); - } - let escape = bytes - .get(index + 1..index + 4) - .filter(|digits| { - *byte == b'\\' && digits.iter().all(|digit| (b'0'..=b'7').contains(digit)) - }) - .and_then(|digits| std::str::from_utf8(digits).ok()) - .and_then(|digits| u8::from_str_radix(digits, 8).ok()); - match escape { - Some(value) => { - decoded.push(value); - (decoded, 3) - } - None => { - decoded.push(*byte); - (decoded, 0) - } - } - }, - ); - String::from_utf8_lossy(&decoded).into_owned() -} - -/// Gives the project quota flags from the super options of an XFS mount. The kernel shows -/// `prjquota` when it counts and enforces project quotas, and `pqnoenforce` when it only counts -/// them. -pub(super) fn project_quota(super_options: &str) -> ProjectQuota { - super_options.split(',').fold( - ProjectQuota { - accounting: false, - enforcement: false, - }, - |quota, option| match option { - "prjquota" | "pquota" => ProjectQuota { - accounting: true, - enforcement: true, - }, - "pqnoenforce" => ProjectQuota { - accounting: true, - ..quota - }, - _ => quota, - }, - ) -} - -/// Tells whether the volume is XFS with project quota accounting and enforcement, and why not. -pub(super) fn verdict(magic: Option, quota: Option) -> Value { - let reasons = [ - (magic != Some(XFS_SUPER_MAGIC)).then(|| { - format!( - "the filesystem type is {}, not xfs", - magic.map_or("unknown", filesystem_name) - ) - }), - match quota { - None => Some("the mount of the volume is not in /proc/self/mountinfo".to_string()), - Some(ProjectQuota { - accounting: false, .. - }) => Some("the mount has no project quota".to_string()), - Some(ProjectQuota { - enforcement: false, .. - }) => Some("the mount counts project quotas but does not enforce them".to_string()), - Some(_) => None, - }, - ] - .into_iter() - .flatten() - .collect::>(); - if reasons.is_empty() { - json!({ "status": "ok" }) - } else { - json!({ "status": "mismatch", "reasons": reasons }) - } -} - -/// The `fs_quota_statv` record of `Q_XGETQSTATV`, as in `linux/dqblk_xfs.h`. -#[repr(C)] -#[derive(Clone, Copy, Default)] -struct FsQuotaFileStatV { - qfs_ino: u64, - qfs_nblks: u64, - qfs_nextents: u32, - qfs_pad: u32, -} - -#[repr(C)] -#[derive(Clone, Copy, Default)] -struct FsQuotaStatV { - qs_version: i8, - qs_pad1: u8, - qs_flags: u16, - qs_incoredqs: u32, - qs_uquota: FsQuotaFileStatV, - qs_gquota: FsQuotaFileStatV, - qs_pquota: FsQuotaFileStatV, - qs_btimelimit: i32, - qs_itimelimit: i32, - qs_rtbtimelimit: i32, - qs_bwarnlimit: u16, - qs_iwarnlimit: u16, - qs_rtbwarnlimit: u16, - qs_pad3: u16, - qs_pad4: u32, - qs_pad2: [u64; 7], -} - -/// Asks the kernel for the project quota state of the filesystem of the directory. -fn quota_state(directory: &File) -> Value { - let mut state = FsQuotaStatV { - qs_version: FS_QSTATV_VERSION1, - ..FsQuotaStatV::default() - }; - // SAFETY: `Q_XGETQSTATV` writes one `fs_quota_statv` record, which `state` holds for the - // duration of the call. - let result = unsafe { - libc::syscall( - libc::SYS_quotactl_fd, - directory.as_raw_fd(), - (Q_XGETQSTATV << 8) | XQM_PRJQUOTA, - 0, - std::ptr::from_mut(&mut state).cast::(), - ) - }; - if result == -1 { - let error = std::io::Error::last_os_error(); - json!({ "error": { "errno": error.raw_os_error(), "message": error.to_string() } }) - } else { - json!({ "ok": { - "flags": format!("{:#x}", state.qs_flags), - "accounting": state.qs_flags & FS_QUOTA_PDQ_ACCT != 0, - "enforcement": state.qs_flags & FS_QUOTA_PDQ_ENFD != 0, - } }) - } -} - -/// Asks XFS for the geometry of the filesystem of the directory. -fn geometry(directory: &File) -> Value { - // SAFETY: The opcodes and the sizes come from `struct xfs_fsop_geom` and - // `struct xfs_fsop_geom_v4` of `xfs_fs.h`, and the kernel writes at most that many bytes. - let full = unsafe { - rustix::ioctl::ioctl( - directory, - rustix::ioctl::Getter::< - { rustix::ioctl::opcode::read::<[u8; GEOMETRY_BYTES]>(b'X', 126) }, - [u8; GEOMETRY_BYTES], - >::new(), - ) - }; - let result = full.map(|bytes| bytes.to_vec()).or_else(|error| { - if error == rustix::io::Errno::NOTTY { - // SAFETY: As above, for the older record. - unsafe { - rustix::ioctl::ioctl( - directory, - rustix::ioctl::Getter::< - { rustix::ioctl::opcode::read::<[u8; GEOMETRY_V4_BYTES]>(b'X', 124) }, - [u8; GEOMETRY_V4_BYTES], - >::new(), - ) - } - .map(|bytes| bytes.to_vec()) - } else { - Err(error) - } - }); - match result { - Ok(bytes) => decode_geometry(&bytes) - .unwrap_or_else(|| json!({ "error": "the geometry record is too short" })), - Err(error) => { - json!({ "error": { "errno": error.raw_os_error(), "message": error.to_string() } }) - } - } -} - -/// Reads the fields of `struct xfs_fsop_geom` that `xfs_info` shows. -pub(super) fn decode_geometry(bytes: &[u8]) -> Option { - let u32_at = |offset: usize| { - bytes - .get(offset..offset + 4) - .and_then(|field| field.try_into().ok()) - .map(u32::from_ne_bytes) - }; - let u64_at = |offset: usize| { - bytes - .get(offset..offset + 8) - .and_then(|field| field.try_into().ok()) - .map(u64::from_ne_bytes) - }; - let flags = u32_at(92)?; - Some(json!({ - "blocksize": u32_at(0)?, - "rtextsize": u32_at(4)?, - "agblocks": u32_at(8)?, - "agcount": u32_at(12)?, - "logblocks": u32_at(16)?, - "sectsize": u32_at(20)?, - "inodesize": u32_at(24)?, - "imaxpct": u32_at(28)?, - "datablocks": u64_at(32)?, - "rtblocks": u64_at(40)?, - "rtextents": u64_at(48)?, - "logstart": u64_at(56)?, - "sunit": u32_at(80)?, - "swidth": u32_at(84)?, - "version": u32_at(88)?, - "flags": format!("{flags:#x}"), - "flag_names": GEOMETRY_FLAGS - .iter() - .filter(|(bit, _)| flags & bit != 0) - .map(|(_, name)| *name) - .collect::>(), - "logsectsize": u32_at(96)?, - "rtsectsize": u32_at(100)?, - "dirblocksize": u32_at(104)?, - "logsunit": u32_at(108)?, - })) -} - -#[cfg(test)] -mod tests { - use super::{ - ProjectQuota, XFS_SUPER_MAGIC, check, decode_geometry, mount_of, project_quota, verdict, - }; - use pretty_assertions::assert_eq; - use serde_json::json; - use std::path::Path; - use test_r::test; - - const MOUNTINFO: &str = "\ -22 1 259:1 / / rw,relatime shared:1 - overlay overlay rw,lowerdir=/a,upperdir=/b -30 22 259:2 / /data rw,relatime shared:2 - xfs /dev/nvme1n1 rw,attr2,inode64,logbufs=8,logbsize=32k,prjquota -31 22 259:3 / /data2 rw,relatime - xfs /dev/nvme2n1 rw,attr2,inode64,pqnoenforce -32 22 259:4 / /with\\040space rw - ext4 /dev/sda1 rw -33 30 0:5 / /data/proc rw - proc proc rw -"; - - #[test] - fn the_mount_of_a_path_is_the_one_with_the_longest_mount_point() { - let mount_point = |path: &str| { - mount_of(MOUNTINFO, Path::new(path)).map(|mount| { - ( - mount.mount_point.display().to_string(), - mount.filesystem_type.to_string(), - ) - }) - }; - - assert_eq!( - ( - mount_point("/data/tree"), - mount_point("/data"), - mount_point("/data2/x"), - mount_point("/database"), - mount_point("/with space/x"), - mount_point("/data/proc/1"), - ), - ( - Some(("/data".to_string(), "xfs".to_string())), - Some(("/data".to_string(), "xfs".to_string())), - Some(("/data2".to_string(), "xfs".to_string())), - Some(("/".to_string(), "overlay".to_string())), - Some(("/with space".to_string(), "ext4".to_string())), - Some(("/data/proc".to_string(), "proc".to_string())), - ) - ); - } - - #[test] - fn the_super_options_give_the_project_quota_flags() { - assert_eq!( - [ - "rw,attr2,inode64,prjquota", - "rw,attr2,pqnoenforce", - "rw,attr2,inode64", - "rw,usrquota", - ] - .map(project_quota), - [ - ProjectQuota { - accounting: true, - enforcement: true - }, - ProjectQuota { - accounting: true, - enforcement: false - }, - ProjectQuota { - accounting: false, - enforcement: false - }, - ProjectQuota { - accounting: false, - enforcement: false - }, - ] - ); - } - - #[test] - fn the_verdict_accepts_only_xfs_that_counts_and_enforces_project_quotas() { - let enforced = ProjectQuota { - accounting: true, - enforcement: true, - }; - let counted = ProjectQuota { - accounting: true, - enforcement: false, - }; - - assert_eq!( - [ - verdict(Some(XFS_SUPER_MAGIC), Some(enforced)), - verdict(Some(XFS_SUPER_MAGIC), Some(counted)), - verdict(Some(0xef53), Some(enforced)), - verdict(None, None), - ] - .map(|verdict| verdict["status"].as_str().map(str::to_string)), - [ - Some("ok".to_string()), - Some("mismatch".to_string()), - Some("mismatch".to_string()), - Some("mismatch".to_string()), - ] - ); - } - - #[test] - fn the_geometry_gives_the_fields_that_xfs_info_shows() { - let bytes = [ - (0, 4096_u64, 4), - (12, 16, 4), - (32, 1_000_000, 8), - (80, 8, 4), - (92, (1 << 20) | (1 << 15) | (1 << 21), 4), - (104, 4096, 4), - ] - .iter() - .fold(vec![0_u8; 256], |mut bytes, (offset, value, size)| { - bytes[*offset..offset + size].copy_from_slice(&value.to_ne_bytes()[..*size]); - bytes - }); - - let geometry = decode_geometry(&bytes).unwrap(); - - assert_eq!( - ( - &geometry["blocksize"], - &geometry["agcount"], - &geometry["datablocks"], - &geometry["sunit"], - &geometry["flag_names"], - &geometry["dirblocksize"], - decode_geometry(&bytes[..100]).is_none(), - ), - ( - &json!(4096), - &json!(16), - &json!(1_000_000), - &json!(8), - &json!(["v5sb", "reflink", "bigtime"]), - &json!(4096), - true, - ) - ); - } - - #[test] - fn a_check_of_a_directory_records_each_part() { - let directory = tempfile::tempdir().unwrap(); - - let record = check(directory.path()); - - assert_eq!( - [ - "filesystem_type", - "mount", - "project_quota", - "quotactl", - "xfs_geometry", - "check" - ] - .map(|key| record.get(key).is_some()), - [true; 6] - ); - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/mod.rs b/golem-worker-executor/src/filesystem_snapshot/mod.rs index a3b0ad53d5..288cea9e6d 100644 --- a/golem-worker-executor/src/filesystem_snapshot/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/mod.rs @@ -25,8 +25,6 @@ use std::cmp::Reverse; use std::fmt::{Display, Formatter}; use std::path::Path; -#[cfg(all(target_os = "linux", any(test, feature = "fs-snapshot-benchmark")))] -pub(crate) mod benchmark; #[cfg(test)] mod contract_tests; mod memory; diff --git a/golem-worker-executor/src/fs_snapshot_benchmark.rs b/golem-worker-executor/src/fs_snapshot_benchmark.rs deleted file mode 100644 index 040e403897..0000000000 --- a/golem-worker-executor/src/fs_snapshot_benchmark.rs +++ /dev/null @@ -1,30 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The filesystem snapshot benchmark. It runs on Linux only, with the allocator of the executor. - -#[cfg(target_os = "linux")] -#[global_allocator] -static GLOBAL: mimalloc::MiMalloc = mimalloc::MiMalloc; - -#[cfg(target_os = "linux")] -fn main() -> std::process::ExitCode { - golem_worker_executor::fs_snapshot_benchmark_main() -} - -#[cfg(not(target_os = "linux"))] -fn main() -> std::process::ExitCode { - eprintln!("error: the filesystem snapshot benchmark runs on Linux only"); - std::process::ExitCode::from(2) -} diff --git a/golem-worker-executor/src/lib.rs b/golem-worker-executor/src/lib.rs index 363f1f2756..c11cc11943 100644 --- a/golem-worker-executor/src/lib.rs +++ b/golem-worker-executor/src/lib.rs @@ -35,10 +35,6 @@ pub mod workerctx; #[cfg(test)] pub mod span_test_support; -/// Runs the filesystem snapshot benchmark with the arguments of the process. -#[cfg(all(target_os = "linux", feature = "fs-snapshot-benchmark"))] -pub use filesystem_snapshot::benchmark::cli::main as fs_snapshot_benchmark_main; - #[cfg(test)] test_r::enable!(); From 900a4fe65623b8a053f6cca71bd545b9a774ec45 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 21:49:16 -0700 Subject: [PATCH 51/55] Remove the inspection and the repository settings of the rustic bridge --- .../src/filesystem_snapshot/rustic/mod.rs | 112 +++------------- .../src/filesystem_snapshot/rustic/tests.rs | 123 +----------------- 2 files changed, 17 insertions(+), 218 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 594499f248..be5ac51311 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -44,7 +44,7 @@ use bytesize::ByteSize; use golem_common::model::Timestamp; use golem_service_base::storage::blob::BlobStorage; use rustic_core::jiff::Span; -use rustic_core::repofile::{Chunker, ConfigFile, MasterKey, SnapshotFile}; +use rustic_core::repofile::{Chunker, MasterKey, SnapshotFile}; use rustic_core::{ BackupOptions, ConfigOptions, Credentials, KeyOptions, LimitOption, LocalDestination, LsOptions, Open, OpenStatus, ParentOptions, PathList, PruneOptions, PruneStats, @@ -310,18 +310,6 @@ pub(super) struct PruneReport { pub(super) phases: Box<[PhaseTime]>, } -/// What an inspection of a repository found. -#[derive(Clone, Debug, PartialEq, Eq)] -pub(super) struct InspectReport { - /// The number of snapshots of the repository. - pub(super) snapshots: u64, - /// Whether a snapshot has the name. - pub(super) found: bool, - /// The settings of the repository, as its config file gives them. - pub(super) settings: RepositorySettings, - pub(super) phases: Box<[PhaseTime]>, -} - /// The rustic repository of one scope in blob storage. /// /// Each operation opens the repository again, with the master key and without the rustic cache. @@ -333,7 +321,6 @@ pub(super) struct Repository { scope: SnapshotScope, key: RepositoryKey, deadline: Duration, - settings: RepositorySettings, } impl Repository { @@ -350,15 +337,9 @@ impl Repository { scope, key, deadline, - settings: RepositorySettings::default(), } } - /// Gives the repository with the settings that a save uses when it makes the repository. - pub(super) fn with_settings(self, settings: RepositorySettings) -> Self { - Self { settings, ..self } - } - /// Saves the directory tree `tree` as a snapshot with the name, with the default settings of /// a save. pub(super) async fn save( @@ -371,10 +352,10 @@ impl Repository { /// Saves the directory tree `tree` as a snapshot with the name. /// - /// A save in a scope without a repository makes the repository first, with the settings of - /// this value. The newest snapshot of the scope is the parent of the save, so the save reads - /// only the files that `settings` finds changed since that snapshot. The snapshot keeps the - /// paths relative to `tree`. + /// A save in a scope without a repository makes the repository first, with + /// [`RepositorySettings::DEFAULT`]. The newest snapshot of the scope is the parent of the save, + /// so the save reads only the files that `settings` finds changed since that snapshot. The + /// snapshot keeps the paths relative to `tree`. pub(super) async fn save_with( &self, name: &SnapshotName, @@ -383,11 +364,19 @@ impl Repository { ) -> anyhow::Result { let backend = self.backend()?; let key = self.key.clone(); - let repository_settings = self.settings; let name = name.clone(); let tree: Box = tree.into(); - run_blocking(move || save(backend, &key, &repository_settings, &settings, &name, &tree)) - .await + run_blocking(move || { + save( + backend, + &key, + &RepositorySettings::DEFAULT, + &settings, + &name, + &tree, + ) + }) + .await } /// Restores the newest snapshot with the name into the empty directory `into`. @@ -431,18 +420,6 @@ impl Repository { run_blocking(move || prune(backend, &key, &settings)).await } - /// Opens the repository, finds the snapshots with the name, and loads the index, as a - /// restore does before it reads data. The result is `None` when the scope has no repository. - pub(super) async fn inspect( - &self, - name: &SnapshotName, - ) -> anyhow::Result> { - let backend = self.backend()?; - let key = self.key.clone(); - let name = name.clone(); - run_blocking(move || inspect(backend, &key, &name)).await - } - /// Gives a backend over the namespace of the scope, which waits on the current runtime for at /// most the deadline. fn backend(&self) -> anyhow::Result> { @@ -583,35 +560,6 @@ fn prune( })) } -fn inspect( - backend: Arc, - key: &RepositoryKey, - name: &SnapshotName, -) -> anyhow::Result> { - let (repository, open) = timed(OperationPhase::Open, || open_existing(backend, key))?; - let Some(repository) = repository else { - return Ok(None); - }; - let ((snapshots, found), lookup) = timed(OperationPhase::Lookup, || { - repository.get_all_snapshots().map(|snapshots| { - ( - snapshots.len(), - snapshots - .iter() - .any(|snapshot| snapshot.label == name.as_str()), - ) - }) - })?; - let settings = repository_settings(repository.config())?; - let (_, index) = timed(OperationPhase::IndexLoad, || repository.to_indexed())?; - Ok(Some(InspectReport { - snapshots: u64::try_from(snapshots)?, - found, - settings, - phases: Box::new([open, lookup, index]), - })) -} - fn forget( backend: Arc, key: &RepositoryKey, @@ -706,34 +654,6 @@ fn config_options(settings: &RepositorySettings) -> ConfigOptions { } } -/// Gives the settings of a repository from its config file, or an error when the config file has -/// fixed chunks whose size is not a size of [`Chunking::Fixed`]. -fn repository_settings(config: &ConfigFile) -> anyhow::Result { - Ok(RepositorySettings { - chunking: match config.chunker() { - Chunker::Rabin => Chunking::Rabin, - Chunker::FixedSize => Chunking::Fixed( - u32::try_from(config.chunk_size()) - .ok() - .and_then(NonZeroU32::new) - .with_context(|| { - format!( - "the repository has fixed chunks of {} bytes, which is not 1 to {} bytes", - config.chunk_size(), - u32::MAX - ) - })?, - ), - }, - compression: match config.compression.map(NonZeroI32::new) { - None => Compression::Default, - Some(None) => Compression::Off, - Some(Some(level)) => Compression::Level(level), - }, - extra_verify: config.extra_verify(), - }) -} - /// The options of a save. /// /// A snapshot keeps the paths relative to the saved tree. The parent of a save is the newest diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index f367be34f8..f15c45fa55 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -24,7 +24,7 @@ use super::scripted::{Script, ScriptedBlobStorage}; use super::{ ChangeDetection, Chunking, Compression, OperationPhase, PruneSettings, RepackLimits, Repository, RepositoryKey, RepositorySettings, SaveSettings, backup_options, config_options, - open_existing, prune_options, repository_options, run_blocking, unopened, + open_existing, prune_options, repository_options, run_blocking, }; use crate::filesystem_snapshot::contract_tests::fixture::{ Scratch, Spec, fixture, listing, write_tree, @@ -1139,127 +1139,6 @@ fn each_setting_goes_into_its_rustic_option() { ); } -#[test] -async fn a_repository_keeps_the_settings_of_its_first_save_and_inspect_gives_them() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let (fixed_scope, default_scope) = (new_scope(), new_scope()); - let settings = RepositorySettings { - chunking: Chunking::Fixed(NonZeroU32::new(65_536).unwrap()), - compression: Compression::Off, - extra_verify: false, - }; - // 4 chunks of 64 KiB and one of 1 byte, each with other bytes. Rabin keeps a file below its - // smallest chunk of 512 KiB in one chunk. - let tree = Scratch::new(); - let content = (0..4 * 65_536 + 1) - .map(|index| (index % 251) as u8) - .collect::>(); - std::fs::write(tree.path().join("data"), &content).unwrap(); - let fixed = repository(&storage, &fixed_scope).with_settings(settings); - let default = repository(&storage, &default_scope); - - let fixed_save = fixed.save(&name("first"), tree.path()).await.unwrap(); - let default_save = default.save(&name("first"), tree.path()).await.unwrap(); - let reopened = repository(&storage, &fixed_scope) - .with_settings(RepositorySettings::default()) - .inspect(&name("first")) - .await - .unwrap() - .unwrap(); - let default_settings = default - .inspect(&name("first")) - .await - .unwrap() - .unwrap() - .settings; - - assert_eq!( - ( - fixed_save.data_blobs, - fixed_save.data_added_packed >= fixed_save.data_added, - default_save.data_blobs, - default_save.data_added_packed < default_save.data_added, - reopened.settings, - default_settings, - ), - (5, true, 1, true, settings, RepositorySettings::default()) - ); -} - -#[test] -async fn inspect_gives_an_error_for_fixed_chunks_that_no_setting_can_hold() { - // A repository that rustic makes with fixed chunks of 4 GiB has a chunk size that does not fit - // `Chunking::Fixed`. The inspection must not report it as a Rabin repository. - let storage = Arc::new(InMemoryBlobStorage::new()); - let scope = new_scope(); - let backend = Arc::new(BlobBackend::new( - storage.clone(), - scope.0.clone(), - Handle::current(), - STORAGE_CALL_DEADLINE, - )); - run_blocking(move || { - unopened(backend)?.init( - &rustic_core::Credentials::Masterkey(key().master_key()), - &rustic_core::KeyOptions::default(), - &rustic_core::ConfigOptions::default() - .set_chunker(rustic_core::repofile::Chunker::FixedSize) - .set_chunk_size(bytesize::ByteSize::b(1 << 32)), - )?; - Ok(()) - }) - .await - .unwrap(); - - let inspected = repository(&storage, &scope).inspect(&name("first")).await; - - assert_eq!( - inspected.map_err(|error| error.to_string()), - Err(format!( - "the repository has fixed chunks of {} bytes, which is not 1 to {} bytes", - 1_u64 << 32, - u32::MAX - )) - ); -} - -#[test] -async fn inspect_gives_the_snapshots_the_name_and_the_phases_and_nothing_without_a_repository() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let scope = new_scope(); - let repository = repository(&storage, &scope); - let tree = fixture_tree(); - - let before = repository.inspect(&name("first")).await.unwrap(); - repository.save(&name("first"), tree.path()).await.unwrap(); - repository.save(&name("second"), tree.path()).await.unwrap(); - let found = repository.inspect(&name("first")).await.unwrap().unwrap(); - let missing = repository.inspect(&name("third")).await.unwrap().unwrap(); - - assert_eq!( - ( - before, - (found.snapshots, found.found), - (missing.snapshots, missing.found), - found - .phases - .iter() - .map(|time| time.phase) - .collect::>(), - ), - ( - None, - (2, true), - (2, false), - vec![ - OperationPhase::Open, - OperationPhase::Lookup, - OperationPhase::IndexLoad - ], - ) - ); -} - #[test] async fn a_size_and_mtime_save_of_a_copied_tree_reads_no_file_and_a_ctime_save_reads_each() { // A copy gives each file a new inode and a new change time, and keeps its size and its From 2da0e0accc63f850e4cd6bdb52c477093c52fb38 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 22:12:16 -0700 Subject: [PATCH 52/55] Check that a new repository keeps the settings of its creation and that a bridge save uses the defaults --- .../src/filesystem_snapshot/rustic/tests.rs | 65 ++++++++++++++++++- 1 file changed, 64 insertions(+), 1 deletion(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index f15c45fa55..1405b955ea 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -24,7 +24,7 @@ use super::scripted::{Script, ScriptedBlobStorage}; use super::{ ChangeDetection, Chunking, Compression, OperationPhase, PruneSettings, RepackLimits, Repository, RepositoryKey, RepositorySettings, SaveSettings, backup_options, config_options, - open_existing, prune_options, repository_options, run_blocking, + open_existing, open_or_create, prune_options, repository_options, run_blocking, }; use crate::filesystem_snapshot::contract_tests::fixture::{ Scratch, Spec, fixture, listing, write_tree, @@ -1139,6 +1139,69 @@ fn each_setting_goes_into_its_rustic_option() { ); } +#[test] +async fn a_repository_keeps_the_settings_of_its_creation_and_a_bridge_save_uses_the_defaults() { + let storage = Arc::new(InMemoryBlobStorage::new()); + let (created_scope, saved_scope) = (new_scope(), new_scope()); + let settings = RepositorySettings { + chunking: Chunking::Fixed(NonZeroU32::new(65_536).unwrap()), + compression: Compression::Off, + extra_verify: false, + }; + let backend = Arc::new(BlobBackend::new( + storage.clone(), + created_scope.0.clone(), + Handle::current(), + STORAGE_CALL_DEADLINE, + )); + run_blocking(move || { + open_or_create(backend.clone(), &key(), &settings)?; + open_or_create(backend, &key(), &RepositorySettings::DEFAULT)?; + Ok(()) + }) + .await + .unwrap(); + repository(&storage, &saved_scope) + .save(&name("first"), fixture_tree().path()) + .await + .unwrap(); + let config = |scope: SnapshotScope| { + let storage = storage.clone(); + async move { + with_existing_repository(storage, &scope, STORAGE_CALL_DEADLINE, |repository| { + let config = repository.config(); + Ok(( + config.chunker(), + config.chunk_size(), + config.compression, + config.extra_verify(), + )) + }) + .await + } + }; + + let created = config(created_scope).await.unwrap(); + let saved = config(saved_scope).await.unwrap(); + + assert_eq!( + (created, (saved.0, saved.2, saved.3)), + ( + ( + rustic_core::repofile::Chunker::FixedSize, + 65_536, + Some(0), + false + ), + ( + rustic_core::repofile::Chunker::Rabin, + None, + RepositorySettings::DEFAULT.extra_verify + ), + ) + ); +} + #[test] async fn a_size_and_mtime_save_of_a_copied_tree_reads_no_file_and_a_ctime_save_reads_each() { // A copy gives each file a new inode and a new change time, and keeps its size and its From 7c2f96a034175422e86cf806b01e20d1081f904f Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 22:21:55 -0700 Subject: [PATCH 53/55] Check that the settings of a new repository set its chunks and compression --- .../src/filesystem_snapshot/rustic/tests.rs | 73 +++++++++++++------ 1 file changed, 51 insertions(+), 22 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index 1405b955ea..9adb1e40f5 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -1155,20 +1155,24 @@ async fn a_repository_keeps_the_settings_of_its_creation_and_a_bridge_save_uses_ STORAGE_CALL_DEADLINE, )); run_blocking(move || { - open_or_create(backend.clone(), &key(), &settings)?; - open_or_create(backend, &key(), &RepositorySettings::DEFAULT)?; + open_or_create(backend, &key(), &settings)?; Ok(()) }) .await .unwrap(); - repository(&storage, &saved_scope) - .save(&name("first"), fixture_tree().path()) - .await - .unwrap(); - let config = |scope: SnapshotScope| { - let storage = storage.clone(); - async move { - with_existing_repository(storage, &scope, STORAGE_CALL_DEADLINE, |repository| { + // 4 chunks of 64 KiB and one of 1 byte, each with other bytes. Rabin keeps a file below its + // smallest chunk of 512 KiB in one chunk. + let tree = Scratch::new(); + let content = (0..4 * 65_536 + 1) + .map(|index| (index % 251) as u8) + .collect::>(); + std::fs::write(tree.path().join("data"), &content).unwrap(); + let config = async |scope: &SnapshotScope| { + with_existing_repository( + storage.clone(), + scope, + STORAGE_CALL_DEADLINE, + |repository| { let config = repository.config(); Ok(( config.chunker(), @@ -1176,28 +1180,53 @@ async fn a_repository_keeps_the_settings_of_its_creation_and_a_bridge_save_uses_ config.compression, config.extra_verify(), )) - }) - .await - } + }, + ) + .await + .unwrap() }; - let created = config(created_scope).await.unwrap(); - let saved = config(saved_scope).await.unwrap(); + let fixed_save = repository(&storage, &created_scope) + .save(&name("first"), tree.path()) + .await + .unwrap(); + let default_save = repository(&storage, &saved_scope) + .save(&name("first"), tree.path()) + .await + .unwrap(); + let (fixed_chunker, fixed_chunk_size, fixed_compression, fixed_extra_verify) = + config(&created_scope).await; + let (default_chunker, _, default_compression, default_extra_verify) = + config(&saved_scope).await; assert_eq!( - (created, (saved.0, saved.2, saved.3)), ( ( + fixed_save.data_blobs, + fixed_save.data_added_packed >= fixed_save.data_added, + fixed_chunker, + fixed_chunk_size, + fixed_compression, + fixed_extra_verify, + ), + ( + default_save.data_blobs, + default_save.data_added_packed < default_save.data_added, + default_chunker, + default_compression, + default_extra_verify, + ), + ), + ( + ( + 5, + true, rustic_core::repofile::Chunker::FixedSize, 65_536, Some(0), - false - ), - ( - rustic_core::repofile::Chunker::Rabin, - None, - RepositorySettings::DEFAULT.extra_verify + false, ), + (1, true, rustic_core::repofile::Chunker::Rabin, None, true), ) ); } From 5a8237600a25efe23fae70cab24b568381ed0d92 Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 22:29:07 -0700 Subject: [PATCH 54/55] Make the save of the bridge create a repository with the default settings, and name the scopes of the settings test --- .../src/filesystem_snapshot/rustic/mod.rs | 15 ++---------- .../src/filesystem_snapshot/rustic/tests.rs | 24 ++++++++++++------- 2 files changed, 18 insertions(+), 21 deletions(-) diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index be5ac51311..fc97ff14f4 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -366,17 +366,7 @@ impl Repository { let key = self.key.clone(); let name = name.clone(); let tree: Box = tree.into(); - run_blocking(move || { - save( - backend, - &key, - &RepositorySettings::DEFAULT, - &settings, - &name, - &tree, - ) - }) - .await + run_blocking(move || save(backend, &key, &settings, &name, &tree)).await } /// Restores the newest snapshot with the name into the empty directory `into`. @@ -449,13 +439,12 @@ async fn run_blocking( fn save( backend: Arc, key: &RepositoryKey, - repository_settings: &RepositorySettings, settings: &SaveSettings, name: &SnapshotName, tree: &Path, ) -> anyhow::Result { let started = Instant::now(); - let (repository, opening) = open_or_create(backend, key, repository_settings)?; + let (repository, opening) = open_or_create(backend, key, &RepositorySettings::DEFAULT)?; let open = PhaseTime { phase: opening, wall: started.elapsed(), diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs index 9adb1e40f5..5849669a20 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests.rs @@ -1142,7 +1142,7 @@ fn each_setting_goes_into_its_rustic_option() { #[test] async fn a_repository_keeps_the_settings_of_its_creation_and_a_bridge_save_uses_the_defaults() { let storage = Arc::new(InMemoryBlobStorage::new()); - let (created_scope, saved_scope) = (new_scope(), new_scope()); + let (fixed_scope, default_scope) = (new_scope(), new_scope()); let settings = RepositorySettings { chunking: Chunking::Fixed(NonZeroU32::new(65_536).unwrap()), compression: Compression::Off, @@ -1150,7 +1150,7 @@ async fn a_repository_keeps_the_settings_of_its_creation_and_a_bridge_save_uses_ }; let backend = Arc::new(BlobBackend::new( storage.clone(), - created_scope.0.clone(), + fixed_scope.0.clone(), Handle::current(), STORAGE_CALL_DEADLINE, )); @@ -1186,18 +1186,18 @@ async fn a_repository_keeps_the_settings_of_its_creation_and_a_bridge_save_uses_ .unwrap() }; - let fixed_save = repository(&storage, &created_scope) + let fixed_save = repository(&storage, &fixed_scope) .save(&name("first"), tree.path()) .await .unwrap(); - let default_save = repository(&storage, &saved_scope) + let default_save = repository(&storage, &default_scope) .save(&name("first"), tree.path()) .await .unwrap(); let (fixed_chunker, fixed_chunk_size, fixed_compression, fixed_extra_verify) = - config(&created_scope).await; - let (default_chunker, _, default_compression, default_extra_verify) = - config(&saved_scope).await; + config(&fixed_scope).await; + let (default_chunker, default_chunk_size, default_compression, default_extra_verify) = + config(&default_scope).await; assert_eq!( ( @@ -1213,6 +1213,7 @@ async fn a_repository_keeps_the_settings_of_its_creation_and_a_bridge_save_uses_ default_save.data_blobs, default_save.data_added_packed < default_save.data_added, default_chunker, + default_chunk_size, default_compression, default_extra_verify, ), @@ -1226,7 +1227,14 @@ async fn a_repository_keeps_the_settings_of_its_creation_and_a_bridge_save_uses_ Some(0), false, ), - (1, true, rustic_core::repofile::Chunker::Rabin, None, true), + ( + 1, + true, + rustic_core::repofile::Chunker::Rabin, + 1_048_576, + None, + true, + ), ) ); } From f33aefbc7ad62d9fdfac5484066569239057cc1d Mon Sep 17 00:00:00 2001 From: Kaur Matas <33095685+kmatasfp@users.noreply.github.com> Date: Fri, 25 Sep 2026 23:03:21 -0700 Subject: [PATCH 55/55] Remove the SQLite journal of the registry service database and ignore its companion files --- .gitignore | 1 + golem_registry_service.db-wal | Bin 210152 -> 0 bytes 2 files changed, 1 insertion(+) delete mode 100644 golem_registry_service.db-wal diff --git a/.gitignore b/.gitignore index 94dabd9655..a1072e132d 100644 --- a/.gitignore +++ b/.gitignore @@ -9,6 +9,7 @@ golem-worker-executor/golem-wit/ zig-cache/ dump.rdb /golem_registry_service.db +/golem_registry_service.db-* test-components/*.wat test-components/**/*.wat test-components/*.wasm diff --git a/golem_registry_service.db-wal b/golem_registry_service.db-wal deleted file mode 100644 index 64121e8261e5b8854d46e920a6b68979365441ae..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 210152 zcmeI53w#vSy~oe&uE}nY439tnL02%8#0jw3yuj8HAjEA+P)w}c(uR;MVR4g<*$oe` zHeD54twJB@NA301+FEY4ddo*^@p}D~R(zHA6A-c1>#cgN_XDWteb6G-d(N3Nv$L~j z!cHGb=x+#uf$)r9`w009sH0T2KI5C8!X009sH0T2KI5csnRq|Z>M zXlM8odiV-_9$x`JeEtG|L4kk5RlBbxhax@OroOawXlP|;S9DFsKs>d8=2kSTt83bx zC>i~`M|RxTMaX%#KT8(Sf5eP2Bq#bkTpD`2)9C*p!(MP(`7JLmZ`)H>L?adTFrhy< zKmY_l00ck)1V8`;KmY_l00ck)1Wo}0QKpc2r^nwp5_*WR4cv8rKlQ}E8}H?~z@ zI#eF(jiq88@dX3PXn)n*s@lrA^D0C2!MSzw!?p9Pt7g~F4b|3F*UpAJuFYSM2!w01+b-TcEq8`tr#T5vE00@8p2!H?x zfB*=900@8p2!Oz;L%{3hdk9WvH8Si1Ywun2Vb#-L{G9GzU@K9#o;vkKO+WwyKmY_l z00ck)1V8`;KmY_l;3OxIP>M8vS(z4S(){)%{}J8cK(LKE9I!s}x4RN_;2L-9ZW1V8`;KmY_l00ck)1V8`;KmY{tAyDGe{4)On zbu6*5Uf`ZTb-#Jve^;t}y+B|yp+7i400ck)1V8`;KmY_jDgrldRQ%el>zYWQTH-I7 zzDU~cM0%S-6ieaGX4eXtWPOSx#Klgcu}%U)#yN||JB!MkMPr;rs*{NCdX?9y#3{`L zCjlWrXVElg(K*hd$kn}G70V`b@NRAfdCwLh-kD7a@4seSx`R~-!oIwBtKmY_l z00ck)1V8`;KmY_l00c%h0rozE91}Ak+i9(a47HzhyTHi)#8G+> z009sH0T2KI5C8!X009sH0T9R#aIg#bnTBB(7*jlY#`qPl|B~$^@Dx-30)zB~0|Y<- z1V8`;KmY_l00ck)1V8`;Mm~Y`SWiHUG&K>;uU-21cbATN{M_~>!vVy9--R0>e_kph80Q+5W2Eq$A?dO%5Q%4c`E0dDh#{8JNI1l znYvFb_#;~{;870{^#FDf*cl=W0w4eaAOHd&00JNY0w4eaAOHd<%37{mIbUU5ojvzV$!e+y!vks-}#kWepjj61&$K+xOx!3m00ck) z1V8`;KmY_l00ck)1WqXeJ}*%`UN7H6z+Z%6*ag~ez9qD`uJcRWzkuo^>Q;Kf0RkWZ z0w4eaAOHd&00JNY0w4ear!axE-&3UdLlI4B(){+-_X~B;1Ho46c_6rqc^(-4iTs4C zc3(>lMS8YPeQE2^(8^AqeEYv}PGF~+P-scY%CrD2iG9g`q&Jp|b;N^h$!Ndzk-t4h zd4|FnfjvVSW+njB@8GI^XaDfNKb?6@UoWtqsQXp;9h5*r6bOI-2!H?xfB*=900@8p z2!H?xoU{Zc_%wg6dw~)fEAuZ<$Le9M7nri_d(R&~|MF(OUcmD(32de(93U_f2;9`{ z@oTr2G?75H-N7d>eoo{CM6uNCY<8`X6F4Ci35xSq9-F@^A>-UAk9VV7=0Bl3hw?V?-ZU2PX#&6O&zAYQT7lIAq8$y@YRNS7N6Zb z{}Gt!<~>~Vf_vShD0HLDbO~8^|G}a;FMMzVQTViCf5aqzU{h5as;{f9V(IM(ZrrH& zwOiMj^I5jPXj=KZcl6Jk%`14 zBNK@!C08USkX!*FV!Fr`iOC^XU`_=xkr?=8A~DR%L}E~viNp{t6Nv#^CKAK5Oe6+l znMe%1GLaZ)Wg;=G%0yxim5IcVDHDkSQX6xKSYcr#G12aRs zO_`zWi%iQg<9TB-L;3}udY+p3oS1s-YYOC3PpGDTZe2D1``?~`-ZjqdY}f_<;emH{ zhX=gxsQ`v?xxmlEIr0T2KI5C8!X009sH0T2KI5C8!X7)b=!`v~k$NBR2*a!kyG zY^N0)GVB5`&v>)%n}0Ze59|UX>ElLOK>!3m00ck)1V8`;KmY_l00clFi+~%u0AG?~ z*ag1UcFo44Kf8AO#TPA&rJ}(#iT<9BR4`1&5RZo}2nJ~`Y&G6KOK!iW*Vc6r)+H_$ zwX91onSDhcTkbesy_E@`R^M0uPkmSY6X<6KOX)s%(C`gZGEh08JOC)TMb(LQe z9$i>Ed9rd{L#ktSJgTSD&&A@7tqlv8Hj1=Gv-rKZ0!>bD+twCY%x)hHwk&H6wzMr> zs&SQMw6i~&;)2%3OI!JcO(jzd$GgybbfltP?X=^xMw%NhZf$73&~}&T#=cm8G&!8g zqGc@?w=Qppw6q#v`Yz*zWTm=rd8D~v`HJ9Y8&?Ep85JwnE0LQ}uU7j&CVg8=j)UtA&v!_G!^0)2cX@-j1Fqi)M#YyNdR7#NxV&qfTGEqgQXh&IPKh3@(hN z)<-*2iT+?>O|U-^k0yC3_?5^(l#_;RMYXhD$Hr^utls2YjIJo{mr>T&g29E$mMv{; zXvxjY(ozl92G3->Q)@RgV_jO%&}M@|PUANe4JF#!^sy(|o=1C@wc$LnU>}k5{ly1Z zZoN5AQB>Y=dO>O19H-uqTbEEy)aY1IVd;6(lyo5zkMoAG-O%REGb<}D(Z`g}T|kGD zU?%Ti6dl!q@-u_CQ46EJ>1$c4P99$=9MtI1|9m% z!)Xo}=<713i4*Ets-LT?H_M(xS8Pp;PCKkyibGZ>y_^I6{n6f3yU{m>9lmnAX+FKP zyr_JRzo2x{v=g*x`mMoe5P|fkPb(~)F+;h!k$A3Mdr!0{ z(Z8v^FWTR}A<=(XwBLT0mbT``<&i};^r=LuBi`Q8*VoR!0@@Sl;o*`-d)NQDNjnog zeTiPy@G|^62ISd_R|*U(l$NRL z3gsuqKL7f$cV|}&DgmubFOq%dztvfcJP+}TfnkNx0$5$u5dN-0)4#xvE`M>&SMI<2 zS8N}FSKUSE6Z-?|Vf$SW1_2NN0T2KI5C8!X009sH0T2KI5I8vquq^|2Q;U~H+HD`) zzkuEPK?G*~#Pe@;wCn1(}9j;IntU z(e%#M2N&x81-1}%tLiJ>GO}xo(t`j9fB*=900@8p2!H?xfB*=L6ap!4k>)Qi*OVsB zZ(keVY`7r^wox|(*-y$RT($dJawyWXZR$%~hlW;mdgOcVg_i<5jkMoGvkgUfw)WNc z3kPCx>ii(snn=Wh%Qo~z`-gubZ!d~ZUKHV+z+RMuLQ7IsrUfVf`;z}iZ!8t-i1X@O zANkvJlxHZM5!f@NVP*m__Ypi8tGfU0pKO_^p3RgL^2y;1f1>`y z`75)Rh2(wTac-2yyHPH4qddlqvYNj#-?Wr38*TEw^F%kwrEZibxlumTjq+J;l&82+4!BVERQM_?^q2qS+5Ytu-2G49 zDLfUP3c@Duyg$+DZj?R5LkN5AQQo)bZ!pfY`0VES-O(oEJX0fhc@hv3~=TKd`B)4b|7xR3 zNUlgs4!HtiOa(HL82Du(G0e+EVo;Zf#1JkMi2++C62r4hBnD%dNDRF)kr-%YA~CGW zL}Cz?iNug86Nv#*BFY*L9pz&1lgh0_okW3%flVf|4r4O87_?*}F+|BkylctYAE5&q zO(%vQSyD0B$V7ZOH@`K1fXlB~GtV5no0~!2vqgw^W>doZuNmT9)(r7pYKC~nG()^E ziV*L9riAx65#pW9lmyKycEP)n8N+*z8R8wr4DmiR1S5O42hh&ORF z#9Or);tkmh@wRJ*cyl#EmKK^F#T#WND$@>)IKAnap-gKtqj&=|L%dCyq3nxH%Q545 zV=+Ve1)qAJn)#fVdhCDZ%cq`DP5s=uYX0}XB>}zRURI(wyR%^zI6J-Q=WmnbFZA^S z^nC<}>H7$FF%29b00JNY0w4eaAOHd&00JNY0w4eaCm#X!>H+&J8TtDN>`zDe`v`JO z%!F)bbu#P%HEP)>_ICa3AnXDs-zSX{fdB}A00@8p2!H?xfB*=900@9UE`h(;F0hx~ z)36I%a@TKG%|Dp9gRNQdcpfH!&Gdu=1V#b@>?5E~P%0D53Ce=*L_FFvyEC2`=n@|t z_7NB)f{aMMYn!o;z}kqgkHAU;`v|D(rOMQBqYhs&nQvd?3o7$f=EGXP%9-Vu`KG`= z0_M$X!2osU#YtlyL59RR`UE0d_Kkf6)+U2}1bW}G4nMM1gM9>IjKMwv0mAr|X*os= z`5W&e*j9b%P&v(PANz0E1%A4zdAF9@_&Bo*D5I7T`hx=mKmY_l00ck)1V8`;Kp-yy z*VlRd+Bepk4kE&-FF*Luk-aw-&R@gR+t>M$rMVl&6*4v3$-2l{c!-b*g^viC$h7Xr zg~CsSTqqta$c4gJgj^`RMaYE0UxZ9#dW?_>h0h3?$n+W^7YakaTqrC)a-p#K$c4gt zgj^{6N63Z3gM?ft4DoWI@FF1>3O^EZq3|Rj7Ybhza-r}hAr}fekW^^dukX;NvmEL; z+#|D_-{B^PUkPb?-O+<^kR#U+hHSY|7_Q|)y-PcE*UaDdPURN1A^E+jmkC98P(gFy z0bWiq76{iiue{hG{0rD$+93BY5DtaIHFIm&P3#G1l%K^Lc7dzjxohj1l5PLQ>;k^g zLqr`cK2+RIFK~bW2!H?xfB*=900@8p2!H?xj0^(l@xFl8+|)!gzxKe7W6Sp3weSmH z|H8AgzWU6QuNaQD`{D!Lbp2P(f>VdTDr@G2!=buig?2U;L)h>4eX?tJTJZ+7a=ldc zKTWf(VGh}Zs%z`RwZjUBDFe6ArN?>#T0}2{-SewKM!qCYo~?M*;IKkznJjM72bF+U zrWeWL5s*L0w>puP=OJD_IIK`w0ILgILb#2`unWAE?3wf9f7$UP=3l_4K0wr6)bHQ} z>S6nR5C#Dd009sH0T2KI5C8!X009sH0TB3z2{2azc3Y2+McPd*-0z^>_QCxQ+N}?? z2K==PhFu_Z`!&-B&s#N0w+q}%)SJ~k>JfG4M_f~62Ld1f0w4eaAOHd&00JNY0w4ea zf4>CI@ir^Q3I?aO9lmkiawU7Ig?G9)Vy2U?eDJ8=vVe6Z0)@+4009sH0T2KI5C8!X009sH0T4Jv2>852@v)PaojhJI z-$OvX7-akleDkg!H1_`P?H}s?1wJI|R@GO$W%P%q2mw?B1V8`;KmY_l00ck)1V8`; zKp>ky`gC8B=C5hilqSt@U!uNnAQq?Zp$T5d-4O(rvG-q{=rQ($t9D;Y4n=ylO?_$W z(9p_GpM3XwSoc|wT^K@AURs*+a$Xwy+W6*<{>!4YINiZE`ofUxC*}5BJr23*UJC5F zru`n8Z79OCwXeQ6Ahgn#@nj>AKasZ=#Q}hDPGB!eLZKxoE7Jn>UVG#bwG#%1V8`;KmY_l00ck)1V8`;KmY_l;NvC0TnyUZ^uYZN zS|a(Y7r5U+9%*^^0rxw|BeUN}u$@)PunYY1-N$~D{^H|L!!Gdge$>bv1V8`;KmY_l z00ck)1V8`;KmY`C2>fmA0{dt_`wY9l?Dgli{o>g7XRyr$9^YItY6(5z009sHfnf-2 z8SnFJ-wrmBK()n4$=^}SQ{_ZnKoraE&SuvNS#Y9lY9dLo8}+enlnEKr@2Z*rn^)35Dy^*?$Y?& zsnZNx)S4-~QRfx#x=T~&PMv8JvhMzaMRVSG=ubr9laqa8wS0$exTe0UYHlq{Z%=T2 zo!75@W36pgh?oOqLNVjXMCSY^7mC?TE);W@TqtHL zxlqhYGNG7-WFkVuoFfy88Ac{D=M%Y5%pP)~m@DK$F*C@8VjhqS#i%bAiZNa;6eGG^ zD8_HOP>j}cp%|OxLNOA{g<{;53&kia6=sd8!)S;R)Iq~KUP@Jr(NQk6j)ig!G4jcU zVw{r;^)5~1LhBeN*AOF?Tu7Y&DufS#G#4J=tKi00JNY0w4eaAOHd&00JNY0w4eae-#30 zRVmauwWem>Qdph3=Jw|A|NObzTCa;QUK(C}VO3XMB;2^VdrkLe=B;R4cS%>{r7exE zeHUKVv|_`i)#2K8-N~wYyNdw1EueaCxF%d(cWzZ}Rj9hEGE`L=4h5^i^K0hOpV{+_ zgI0?h^9;MdtpEMix37QX-pR~pqo{|ee}TVB#c>l5009sH0T2KI5C8!X009sH0T2Lz zkD0)DuNL4dmn`-OS|q*#$gm5n`cBFE?_M{vO1BFP5q0QeE;O8;sbhPz8lG~3;610{viFIm(0U@f&7s}69hm21V8`;KmY_l z00ck)1V8`;Mgjr0UckNyftH*39Web4-uPdG&tHCTf1Pd@I8M~#BcWy}BM5*12!H?x zfB*=900@8p2!H?xfPh;9UN2$(BeVPtzP$d)Gdf3g{(--bp!j9#caWZNfB*=900@8p z2!H?xfB*=900@A<2qchJJ-J@~o`2zr^*2{_ZRlOs*tM~~qkZ*;`qtK#y8hZtmv^pT zx}s`b!$kv$#i5qkw(!96P4gml2mTIT{;KM#YwN-%^741``yIR^@z8%fImTbW)(d#l z1Jv&zJ>dWW5C8!X009sH0T2KI5C8!X009sfNd!u~nqqM=$c-VMeEbehd`A22YtK|n z((MBG5cM8-9vn#@Jjx0JAOHd&00JNY0w4eaAOHd&aEcL_2i2peJ}%S{1V8`;KmY_l00ck)1V8`;KmY{(t_b+NMDZ}cgZ~eE CeACGQ