diff --git a/.gitignore b/.gitignore index 94dabd9655..a1072e132d 100644 --- a/.gitignore +++ b/.gitignore @@ -9,6 +9,7 @@ golem-worker-executor/golem-wit/ zig-cache/ dump.rdb /golem_registry_service.db +/golem_registry_service.db-* test-components/*.wat test-components/**/*.wat test-components/*.wasm diff --git a/Cargo.lock b/Cargo.lock index 943cc74d43..1e340eeb3a 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4606,7 +4606,6 @@ dependencies = [ "cap-std", "cap-time-ext", "chrono", - "clap", "criterion", "dashmap", "desert_rust", diff --git a/golem-worker-executor/Cargo.toml b/golem-worker-executor/Cargo.toml index d54d18649a..70039382c5 100644 --- a/golem-worker-executor/Cargo.toml +++ b/golem-worker-executor/Cargo.toml @@ -13,7 +13,6 @@ autotests = false [features] test-utils = [] -fs-snapshot-benchmark = ["dep:clap"] [lib] path = "src/lib.rs" @@ -25,12 +24,6 @@ path = "src/server.rs" harness = false test = false -[[bin]] -name = "fs-snapshot-benchmark" -path = "src/fs_snapshot_benchmark.rs" -required-features = ["fs-snapshot-benchmark"] -test = false - [[test]] name = "integration" path = "tests/lib.rs" @@ -61,7 +54,6 @@ cap-fs-ext = { workspace = true } cap-std = { workspace = true } cap-time-ext = { workspace = true } # keep in sync with wasmtime chrono = { workspace = true } -clap = { workspace = true, optional = true } dashmap = { workspace = true } desert_rust = { workspace = true } drop-stream = { workspace = true } @@ -160,7 +152,6 @@ golem-worker-executor-test-utils = { workspace = true } assert2 = { workspace = true } aws-config = { workspace = true } aws-sdk-s3 = { workspace = true } -clap = { workspace = true } criterion = { workspace = true } axum = { workspace = true } figment = { workspace = true, features = ["test"] } diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/agents.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/agents.rs deleted file mode 100644 index ac885922b3..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/agents.rs +++ /dev/null @@ -1,460 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The repositories of the agents of a scenario, a CPU setting and a tree. -//! -//! Each agent has its own repository, as in production. The repository of an agent is below the -//! path `agents/` of the namespace of its scenario, CPU setting and tree. Thus a copy of -//! the repository of one agent into another agent is a copy of blobs in one namespace, which the -//! S3 backend does in the bucket with `CopyObject`. - -use super::TARGET_LABEL; -use async_trait::async_trait; -use bytes::Bytes; -use futures::stream::BoxStream; -use futures::{StreamExt, TryStreamExt}; -use golem_service_base::replayable_stream::ErasedReplayableStream; -use golem_service_base::storage::blob::{ - BlobMetadata, BlobStorage, BlobStorageNamespace, ExistsResult, ListedBlob, PutIfAbsent, -}; -use serde_json::{Value, json}; -use std::path::{Path, PathBuf}; -use std::sync::Arc; - -/// The first name of the path of the repository of each agent. -pub(super) const AGENTS: &str = "agents"; - -/// The agent whose repository the phases copy to the other agents. -pub(super) const FIRST_AGENT: &str = "0"; - -/// The number of blob copies that are in progress at the same time. -const COPY_CONCURRENCY: usize = 64; - -/// The name of the file that a repository writes when it is made. -const CONFIG: &str = "config"; - -/// Gives the path of the repository of the agent in its namespace. -fn agent_root(agent: &str) -> PathBuf { - Path::new(AGENTS).join(agent) -} - -/// A blob storage that keeps the repository of one agent below its path in the namespace. -/// -/// Each call goes to `inner`, with the path below the path of the agent. Each path that `inner` -/// gives is relative to the path of the agent again. -#[derive(Debug)] -pub(super) struct AgentStorage { - inner: Arc, - root: Box, -} - -impl AgentStorage { - pub(super) fn new(inner: Arc, agent: &str) -> Self { - Self { - inner, - root: agent_root(agent).into_boxed_path(), - } - } - - fn path(&self, path: &Path) -> PathBuf { - self.root.join(path) - } - - /// Gives the path that `inner` gave, relative to the path of the agent. - fn relative(&self, path: &Path) -> anyhow::Result { - path.strip_prefix(&self.root) - .map(Path::to_path_buf) - .map_err(|_| { - anyhow::anyhow!( - "the storage gave the path {}, which is not below {}", - path.display(), - self.root.display() - ) - }) - } -} - -#[async_trait] -impl BlobStorage for AgentStorage { - async fn get_raw( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result>> { - self.inner - .get_raw(target_label, op_label, namespace, &self.path(path)) - .await - } - - async fn get_stream( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result>>> { - self.inner - .get_stream(target_label, op_label, namespace, &self.path(path)) - .await - } - - async fn get_raw_slice( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - start: u64, - end: u64, - ) -> anyhow::Result>> { - self.inner - .get_raw_slice( - target_label, - op_label, - namespace, - &self.path(path), - start, - end, - ) - .await - } - - async fn get_metadata( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.inner - .get_metadata(target_label, op_label, namespace, &self.path(path)) - .await - } - - async fn put_raw( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - data: &[u8], - ) -> anyhow::Result<()> { - self.inner - .put_raw(target_label, op_label, namespace, &self.path(path), data) - .await - } - - async fn put_raw_if_absent( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - data: &[u8], - ) -> anyhow::Result { - self.inner - .put_raw_if_absent(target_label, op_label, namespace, &self.path(path), data) - .await - } - - async fn put_stream( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - stream: &dyn ErasedReplayableStream>, Error = anyhow::Error>, - ) -> anyhow::Result<()> { - self.inner - .put_stream(target_label, op_label, namespace, &self.path(path), stream) - .await - } - - async fn delete( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result<()> { - self.inner - .delete(target_label, op_label, namespace, &self.path(path)) - .await - } - - async fn create_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result<()> { - self.inner - .create_dir(target_label, op_label, namespace, &self.path(path)) - .await - } - - async fn list_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.inner - .list_dir(target_label, op_label, namespace, &self.path(path)) - .await? - .iter() - .map(|path| self.relative(path)) - .collect() - } - - async fn list_blobs_below( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.inner - .list_blobs_below(target_label, op_label, namespace, &self.path(path)) - .await? - .iter() - .map(|blob| { - self.relative(&blob.path).map(|path| ListedBlob { - path: path.into_boxed_path(), - size: blob.size, - }) - }) - .collect() - } - - async fn delete_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result { - self.inner - .delete_dir(target_label, op_label, namespace, &self.path(path)) - .await - } - - async fn exists( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result { - self.inner - .exists(target_label, op_label, namespace, &self.path(path)) - .await - } -} - -/// Copies the repository of [`FIRST_AGENT`] to each agent from `1` to `agents - 1` that has no -/// repository, and gives the numbers of the copied repositories and blobs. -/// -/// The config of a repository is its last copied blob, so an agent with a config has each blob of -/// the repository, also after a copy that stopped. The copies of the other blobs of all agents -/// are in progress at the same time, up to [`COPY_CONCURRENCY`]. -pub(super) async fn copy_first_agent( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - agents: usize, -) -> anyhow::Result { - let source = &agent_root(FIRST_AGENT); - let blobs = storage - .list_blobs_below(TARGET_LABEL, "list_agent", namespace.clone(), source) - .await? - .iter() - .map(|blob| blob.path.strip_prefix(source).map(Path::to_path_buf)) - .collect::, _>>()?; - anyhow::ensure!( - blobs.iter().any(|path| path == Path::new(CONFIG)), - "the repository of the agent {FIRST_AGENT} has no {CONFIG}" - ); - let missing = futures::stream::iter(1..agents) - .map(|agent| async move { - let config = agent_root(&agent.to_string()).join(CONFIG); - storage - .get_metadata(TARGET_LABEL, "find_agent", namespace.clone(), &config) - .await - .map(|metadata| metadata.is_none().then_some(agent)) - }) - .buffered(COPY_CONCURRENCY) - .try_filter_map(|agent| async move { Ok(agent) }) - .try_collect::>() - .await?; - let (configs, others): (Vec<_>, Vec<_>) = missing - .iter() - .flat_map(|agent| blobs.iter().map(move |blob| (*agent, blob))) - .partition(|(_, blob)| *blob == Path::new(CONFIG)); - futures::stream::iter(others.iter().copied()) - .map(|(agent, blob)| copy_blob(storage, namespace, source, agent, blob)) - .buffer_unordered(COPY_CONCURRENCY) - .try_collect::<()>() - .await?; - futures::stream::iter(configs.iter().copied()) - .map(|(agent, blob)| copy_blob(storage, namespace, source, agent, blob)) - .buffer_unordered(COPY_CONCURRENCY) - .try_collect::<()>() - .await?; - Ok(json!({ - "agents_copied": missing.len(), - "blobs_copied": configs.len() + others.len(), - })) -} - -/// Copies the blob of the repository at `source` to the repository of the agent. -async fn copy_blob( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - source: &Path, - agent: usize, - blob: &Path, -) -> anyhow::Result<()> { - storage - .copy( - TARGET_LABEL, - "copy_agent", - namespace.clone(), - &source.join(blob), - &agent_root(&agent.to_string()).join(blob), - ) - .await -} - -/// The directories of a repository in the order in which a copy of the repository lists and -/// copies them. A save writes them in the opposite order, so a snapshot that the copy has also has -/// its index and its packs. The config goes last. -const COPY_ORDER: [&str; 3] = ["snapshots", "index", "data"]; - -/// Gives each blob of the repository of the agent, with the path relative to the repository, in -/// the order of the paths. -pub(super) async fn agent_blobs( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - agent: &str, -) -> anyhow::Result> { - let root = agent_root(agent); - let mut blobs = storage - .list_blobs_below(TARGET_LABEL, "list_agent", namespace.clone(), &root) - .await? - .iter() - .map(|blob| { - Ok(ListedBlob { - path: blob.path.strip_prefix(&root)?.into(), - size: blob.size, - }) - }) - .collect::>>()?; - blobs.sort_by(|left, right| left.path.cmp(&right.path)); - Ok(blobs.into_boxed_slice()) -} - -/// Gives the sum of the sizes of the blobs. -pub(super) fn total_bytes(blobs: &[ListedBlob]) -> u64 { - blobs.iter().map(|blob| blob.size).sum() -} - -/// Copies the repository of the agent `from` to the agent `to` on the server, as the copy of a -/// scope does it, and gives the numbers of the copied blobs and bytes. -/// -/// The copy lists the snapshot files, the index files and the packs, in this order, and then -/// copies them in the same order, each group after the one before it. The config goes last, so an -/// agent with a config has each blob of the copy. -pub(super) async fn copy_agent( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - from: &str, - to: &str, -) -> anyhow::Result { - let source = &agent_root(from); - let groups = futures::stream::iter(COPY_ORDER) - .then(|directory| async move { - storage - .list_blobs_below( - TARGET_LABEL, - "list_agent", - namespace.clone(), - &source.join(directory), - ) - .await - }) - .try_collect::>() - .await?; - let config = storage - .get_metadata( - TARGET_LABEL, - "find_agent", - namespace.clone(), - &source.join(CONFIG), - ) - .await? - .ok_or_else(|| anyhow::anyhow!("the repository of the agent {from} has no {CONFIG}"))?; - let groups = groups - .into_iter() - .chain(std::iter::once(Box::from([ListedBlob { - path: source.join(CONFIG).into_boxed_path(), - size: config.size, - }]))) - .collect::>(); - let target = &agent_root(to); - futures::stream::iter(groups.iter()) - .map(Ok) - .try_for_each(|group| async move { - futures::stream::iter(group.iter()) - .map(|blob| async move { - let relative = blob.path.strip_prefix(source)?; - storage - .copy( - TARGET_LABEL, - "copy_agent", - namespace.clone(), - &blob.path, - &target.join(relative), - ) - .await - }) - .buffer_unordered(COPY_CONCURRENCY) - .try_collect::<()>() - .await - }) - .await?; - Ok(json!({ - "blobs_copied": groups.iter().map(|group| group.len()).sum::(), - "bytes_copied": groups.iter().map(|group| total_bytes(group)).sum::(), - })) -} - -/// Deletes the repository of the agent, and tells whether it had one. -pub(super) async fn delete_agent( - storage: &dyn BlobStorage, - namespace: &BlobStorageNamespace, - agent: &str, -) -> anyhow::Result { - storage - .delete_dir( - TARGET_LABEL, - "delete_agent", - namespace.clone(), - &agent_root(agent), - ) - .await -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/capture.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/capture.rs deleted file mode 100644 index b0a8bbfb36..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/capture.rs +++ /dev/null @@ -1,250 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The phase of the capture scenario: the time of a capture, and the warm saves of a capture with -//! each change detection. -//! -//! A capture is a reflink copy of the tree of an agent that keeps the permissions and the -//! modification times, as the executor makes it before an upload. Each captured file is a new -//! inode with a new change time. So a save that compares change times reads every file of a -//! capture, and a save that compares only sizes and modification times reads the changed files. - -use super::agents::{FIRST_AGENT, copy_agent}; -use super::measure::measure; -use super::report::{Outcome, StepRecord, TreeFacts}; -use super::trees::{self, CopyCounts, Times}; -use super::{ - COLD_SAVE, PhaseContext, PhaseOutcome, WARM_SAVE, change_detection_name, failed, save_record, - snapshot_name, -}; -use crate::filesystem_snapshot::rustic::{ChangeDetection, SaveSettings}; -use futures::{StreamExt, TryStreamExt}; -use serde_json::{Value, json}; -use std::path::{Path, PathBuf}; - -/// The agent whose repository gets the warm save with change detection by size and modification -/// time. The first agent gets the warm save that compares change times. -const SIZE_MTIME_AGENT: &str = "1"; - -/// The capture phase: a new tree, a capture of it, and a cold save of the capture into the -/// repository of the first agent, which then goes to a second agent on the server. Then a small -/// change of the tree, and for each change detection a new capture and a warm save of it into the -/// repository of one of the agents. So both warm saves have the same parent, and each reads a -/// capture that no step read before. -/// -/// The time of a capture step is the time of the capture: the copy does not sync the volume. -pub(super) async fn capture(context: &PhaseContext) -> PhaseOutcome { - let spec = context.selection.tree; - let storage = &context.storage; - let tree = context.work_dir.join("tree"); - let facts = TreeFacts { - name: spec.name, - content: Some(spec.content.label()), - page_cache: Some("dropped"), - ..TreeFacts::default() - }; - let later = [ - "capture", - "cold_save", - "copy_scopes", - "small_change", - FORM_STEPS[0], - FORM_STEPS[1], - FORM_STEPS[2], - FORM_STEPS[3], - ]; - - let (record, generated) = measure("generate_tree", storage, trees::generate(spec, &tree)).await; - let mut steps = vec![record]; - let Ok(counts) = generated else { - return failed(facts, steps, "generate_tree", &later); - }; - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - ..facts - }; - - let first = context.work_dir.join("capture-cold"); - let (record, captured) = measure("capture", storage, capture_tree(&tree, &first)).await; - steps.push(capture_record(record, &captured)); - if captured.is_err() { - return failed(facts, steps, "capture", &later[1..]); - } - - let (record, cold) = measure("cold_save", storage, async { - context - .repository() - .save_with(&snapshot_name(COLD_SAVE)?, &first, SaveSettings::DEFAULT) - .await - }) - .await; - steps.push(save_record(record, &cold)); - if cold.is_err() { - return failed(facts, steps, "cold_save", &later[2..]); - } - - let (record, copied) = measure( - "copy_scopes", - storage, - copy_agent( - storage.as_ref(), - &context.scope().0, - FIRST_AGENT, - SIZE_MTIME_AGENT, - ), - ) - .await; - steps.push(record.with_details( - copied.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )); - if copied.is_err() { - return failed(facts, steps, "copy_scopes", &later[3..]); - } - - let (record, changed) = measure("small_change", storage, trees::change(spec, &tree)).await; - steps.push(record.with_details( - changed.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )); - let Ok(change) = changed else { - return failed(facts, steps, "small_change", &later[4..]); - }; - let facts = TreeFacts { - change: Some(change), - ..facts - }; - - match warm_forms(context, &tree, steps).await { - Ok(steps) => PhaseOutcome { - tree_facts: facts, - steps, - outcome: Outcome::Ok, - }, - Err((steps, step, skipped)) => failed(facts, steps, step, skipped), - } -} - -/// One form of a warm save. -#[derive(Clone, Copy, Debug)] -struct Form { - /// The name of the step that captures the tree for the save. - capture_step: &'static str, - /// The name of the step of the save. - save_step: &'static str, - /// The agent whose repository gets the save. - agent: &'static str, - detection: ChangeDetection, -} - -/// The forms of the warm saves, in the order in which they run. -const FORMS: [Form; 2] = [ - Form { - capture_step: "capture_full_read", - save_step: "warm_save_full_read", - agent: FIRST_AGENT, - detection: ChangeDetection::Ctime, - }, - Form { - capture_step: "capture_size_mtime", - save_step: "warm_save_size_mtime", - agent: SIZE_MTIME_AGENT, - detection: ChangeDetection::SizeMtime, - }, -]; - -/// The names of the steps of the forms, in the order in which they run. -static FORM_STEPS: [&str; 4] = [ - FORMS[0].capture_step, - FORMS[0].save_step, - FORMS[1].capture_step, - FORMS[1].save_step, -]; - -/// Runs the capture and the warm save of each form, one form after the other. A failed step gives -/// the records so far, the name of the step and the steps that did not run. -async fn warm_forms( - context: &PhaseContext, - tree: &Path, - steps: Vec, -) -> Result, (Vec, &'static str, &'static [&'static str])> { - let storage = &context.storage; - futures::stream::iter(FORMS.iter().copied().enumerate()) - .map(Ok) - .try_fold( - steps, - |mut steps, - ( - index, - Form { - capture_step, - save_step, - agent, - detection, - }, - )| async move { - let target: PathBuf = context.work_dir.join(capture_step); - let (record, captured) = - measure(capture_step, storage, capture_tree(tree, &target)).await; - steps.push(capture_record(record, &captured)); - if captured.is_err() { - return Err((steps, capture_step, &FORM_STEPS[2 * index + 1..])); - } - let settings = SaveSettings { - detection, - ..SaveSettings::DEFAULT - }; - let (record, warm) = measure(save_step, storage, async { - context - .agent_repository(agent) - .save_with(&snapshot_name(WARM_SAVE)?, &target, settings) - .await - }) - .await; - steps.push(save_record(record, &warm).with_parameters(json!({ - "change_detection": change_detection_name(detection), - "agent": agent, - }))); - if warm.is_err() { - return Err((steps, save_step, &FORM_STEPS[2 * index + 2..])); - } - Ok(steps) - }, - ) - .await -} - -/// Captures the tree `from` into the new directory `to` on a blocking thread. -async fn capture_tree(from: &Path, to: &Path) -> anyhow::Result { - let (from, to): (Box, Box) = (from.into(), to.into()); - tokio::task::spawn_blocking(move || trees::copy_tree(&from, &to, Times::Keep)).await? -} - -/// Gives the record of a capture with the number of files that are reflinks and copies. -fn capture_record(record: StepRecord, captured: &anyhow::Result) -> StepRecord { - record.with_details( - captured - .as_ref() - .map(|counts| { - json!({ - "files_reflinked": counts.reflinked, - "files_copied": counts.copied, - }) - }) - .unwrap_or(Value::Null), - Box::default(), - ) -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/cli.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/cli.rs deleted file mode 100644 index 203051a105..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/cli.rs +++ /dev/null @@ -1,357 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The command line of the filesystem snapshot benchmark. -//! -//! `plan` prints the phases of each tree of the scenarios, one JSON line for each tree. `run` runs -//! one phase and prints its result as the last line of the standard output. The blob storage -//! comes from the `GOLEM__BLOB_STORAGE__*` environment variables of the executor, with the object -//! prefix of the run. The exit code is 0 when each step succeeded, 1 when a step failed, the -//! restored tree differs or the result could not be written, and 2 for an error of the arguments -//! or of the configuration, or for an error of the environment that each later phase finds again, -//! such as an async runtime that does not start. - -use super::{Selection, is_key_segment, plan, run_phase}; -use clap::Parser; -use figment::Figment; -use figment::providers::{Env, Serialized}; -use golem_common::tracing::{OutputConfig, TracingConfig, init_tracing_with_default_env_filter}; -use golem_service_base::config::{BlobStorageConfig, S3BlobStorageConfig}; -use golem_service_base::storage::blob::BlobStorage; -use golem_service_base::storage::blob::fs::FileSystemBlobStorage; -use golem_service_base::storage::blob::s3::S3BlobStorage; -use serde_json::{Value, json}; -use std::path::{Path, PathBuf}; -use std::process::ExitCode; -use std::sync::Arc; - -/// The start of each object prefix that the benchmark accepts. -const OBJECT_PREFIX_START: &str = "fs-snapshot-bench/"; - -const ENV_PREFIX: &str = "GOLEM__BLOB_STORAGE__"; - -#[derive(Debug, PartialEq, Eq, Parser)] -#[command(name = "fs-snapshot-benchmark")] -enum Command { - /// Prints the phases of each tree of the scenarios, one JSON line for each tree. - Plan { - /// The names of the scenarios, separated by commas. - #[arg(long, value_delimiter = ',', required = true)] - scenarios: Vec, - }, - /// Runs one phase of one tree of a scenario. - Run(RunArguments), -} - -#[derive(Debug, PartialEq, Eq, clap::Args)] -struct RunArguments { - /// The id of the run: 1 to 64 ASCII letters, digits, `-` or `_`. - #[arg(long)] - run_id: String, - #[arg(long)] - scenario: String, - #[arg(long)] - tree: String, - #[arg(long)] - phase: String, - /// The label of the CPU setting of the pod, which the result records. - #[arg(long)] - cpu_setting: String, - /// The directory on the benchmark volume. - #[arg(long)] - work_dir: PathBuf, - /// The object prefix of the run: `fs-snapshot-bench/`, with or without one `/` at the - /// end. - #[arg(long)] - object_prefix: String, -} - -/// Runs the command of the arguments of the process. -pub fn main() -> ExitCode { - match Command::try_parse() { - Ok(Command::Plan { scenarios }) => print_plan(&scenarios), - Ok(Command::Run(arguments)) => run(arguments), - Err(error) => { - let _ = error.print(); - ExitCode::from(u8::try_from(error.exit_code()).unwrap_or(2)) - } - } -} - -fn print_plan(scenarios: &[String]) -> ExitCode { - match plan(scenarios) { - Ok(entries) => { - entries - .iter() - .for_each(|entry| println!("{}", serde_json::to_string(entry).unwrap_or_default())); - ExitCode::SUCCESS - } - Err(name) => usage_error(&format!("no scenario has the name {name:?}")), - } -} - -fn usage_error(message: &str) -> ExitCode { - eprintln!("error: {message}"); - ExitCode::from(2) -} - -fn run(arguments: RunArguments) -> ExitCode { - let selection = match validate(&arguments) { - Ok(selection) => selection, - Err(message) => return usage_error(&message), - }; - let config = match storage_config(&arguments.object_prefix) { - Ok(config) => config, - Err(message) => return usage_error(&message), - }; - let _ = rustls::crypto::ring::default_provider().install_default(); - let _ = init_tracing_with_default_env_filter(&TracingConfig { - stdout: OutputConfig::disabled(), - stderr: OutputConfig::text(), - ..TracingConfig::local_dev("fs-snapshot-benchmark") - }); - let runtime = match crate::bootstrap::create_runtime() { - Ok(runtime) => runtime, - Err(error) => { - eprintln!("error: failed to start the async runtime: {error}"); - return ExitCode::from(2); - } - }; - runtime.block_on(async { - let storage = match storage(&config).await { - Ok(storage) => storage, - Err(error) => { - eprintln!("error: failed to open the blob storage: {error:#}"); - return ExitCode::from(2); - } - }; - let (result, written) = run_phase( - &arguments.run_id, - &arguments.cpu_setting, - selection, - &arguments.work_dir, - storage, - environment(&config), - ) - .await; - println!("{}", serde_json::to_string(&result).unwrap_or_default()); - if let Err(error) = &written { - eprintln!("error: failed to write the result into the blob storage: {error:#}"); - } - if written.is_ok() && result.outcome == super::report::Outcome::Ok { - ExitCode::SUCCESS - } else { - ExitCode::from(1) - } - }) -} - -/// Checks the arguments of `run`, and gives the phase that they name. -fn validate(arguments: &RunArguments) -> Result { - if !is_key_segment(&arguments.run_id) { - return Err(format!( - "the run id {:?} does not have 1 to 64 ASCII letters, digits, `-` or `_`", - arguments.run_id - )); - } - if !is_key_segment(&arguments.cpu_setting) { - return Err(format!( - "the CPU setting {:?} does not have 1 to 64 ASCII letters, digits, `-` or `_`", - arguments.cpu_setting - )); - } - let expected = format!("{OBJECT_PREFIX_START}{}", arguments.run_id); - if arguments.object_prefix != expected && arguments.object_prefix != format!("{expected}/") { - return Err(format!( - "the object prefix {:?} is not {expected:?}, the prefix of the run", - arguments.object_prefix - )); - } - Selection::find(&arguments.scenario, &arguments.tree, &arguments.phase) -} - -/// Reads the blob storage configuration of the executor from the environment, with the object -/// prefix of the run. -fn storage_config(object_prefix: &str) -> Result { - let config = Figment::from(Serialized::defaults(BlobStorageConfig::default_s3())) - .merge(Env::prefixed(ENV_PREFIX).split("__")) - .extract::() - .map_err(|error| format!("the blob storage configuration is not valid: {error}"))?; - match config { - BlobStorageConfig::S3(config) => Ok(BlobStorageConfig::S3(S3BlobStorageConfig { - object_prefix: object_prefix.trim_end_matches('/').to_string(), - ..config - })), - BlobStorageConfig::LocalFileSystem(config) => { - Ok(BlobStorageConfig::LocalFileSystem(config)) - } - _ => { - Err("the benchmark supports the blob storage types S3 and LocalFileSystem".to_string()) - } - } -} - -async fn storage(config: &BlobStorageConfig) -> anyhow::Result> { - match config { - BlobStorageConfig::S3(config) => Ok(Arc::new(S3BlobStorage::new(config.clone()).await)), - BlobStorageConfig::LocalFileSystem(config) => { - Ok(Arc::new(FileSystemBlobStorage::new(&config.root).await?)) - } - _ => anyhow::bail!("the benchmark supports the blob storage types S3 and LocalFileSystem"), - } -} - -/// Records the pod, the host and the storage of the run. -fn environment(config: &BlobStorageConfig) -> Value { - let read = |path: &str| { - std::fs::read_to_string(Path::new(path)) - .ok() - .map(|text| text.trim().to_string()) - }; - json!({ - "pod": std::env::var("POD_NAME").ok(), - "node": std::env::var("NODE_NAME").ok(), - "kernel": read("/proc/sys/kernel/osrelease"), - "available_parallelism": std::thread::available_parallelism().map(|count| count.get()).ok(), - "tokio_workers": tokio::runtime::Handle::try_current().ok().map(|handle| handle.metrics().num_workers()), - "rayon_threads": rayon::current_num_threads(), - "cgroup": { - "cpu_max": read("/sys/fs/cgroup/cpu.max"), - "cpuset": read("/sys/fs/cgroup/cpuset.cpus.effective"), - "memory_max": read("/sys/fs/cgroup/memory.max").and_then(|text| text.parse::().ok()), - }, - "storage": match config { - BlobStorageConfig::S3(config) => json!({ - "type": "S3", - "bucket": config.initial_agent_files_bucket, - "region": config.region, - "object_prefix": config.object_prefix, - "retries": { - "max_attempts": config.retries.max_attempts, - "min_delay_ms": config.retries.min_delay.as_millis() as u64, - "max_delay_ms": config.retries.max_delay.as_millis() as u64, - "multiplier": config.retries.multiplier, - }, - }), - BlobStorageConfig::LocalFileSystem(config) => json!({ - "type": "LocalFileSystem", - "root": config.root.display().to_string(), - }), - _ => Value::Null, - }, - }) -} - -#[cfg(test)] -mod tests { - use super::{Command, RunArguments, storage_config, validate}; - use clap::Parser; - use golem_service_base::config::BlobStorageConfig; - use pretty_assertions::assert_eq; - use std::path::PathBuf; - use test_r::test; - - fn run_arguments() -> RunArguments { - RunArguments { - run_id: "123-1".to_string(), - scenario: "base".to_string(), - tree: "files-1g".to_string(), - phase: "save".to_string(), - cpu_setting: "limit-3".to_string(), - work_dir: PathBuf::from("/data"), - object_prefix: "fs-snapshot-bench/123-1".to_string(), - } - } - - #[test] - fn the_plan_and_run_command_lines_of_the_workflow_parse() { - let plan = - Command::try_parse_from(["fs-snapshot-benchmark", "plan", "--scenarios", "base,smoke"]); - let run = Command::try_parse_from([ - "fs-snapshot-benchmark", - "run", - "--run-id", - "123-1", - "--scenario", - "base", - "--tree", - "files-1g", - "--phase", - "save", - "--cpu-setting", - "limit-3", - "--work-dir", - "/data", - "--object-prefix", - "fs-snapshot-bench/123-1", - ]); - let missing = Command::try_parse_from(["fs-snapshot-benchmark", "plan"]) - .map_err(|error| error.exit_code()); - - assert_eq!( - (plan.ok(), run.ok(), missing), - ( - Some(Command::Plan { - scenarios: vec!["base".to_string(), "smoke".to_string()], - }), - Some(Command::Run(run_arguments())), - Err(2), - ) - ); - } - - #[test] - fn a_run_needs_a_valid_run_id_cpu_setting_prefix_and_selection() { - let with = |change: fn(&mut RunArguments)| { - let mut arguments = run_arguments(); - change(&mut arguments); - validate(&arguments).is_ok() - }; - - assert_eq!( - [ - with(|_| {}), - with(|arguments| arguments.run_id = "a/b".to_string()), - with(|arguments| arguments.cpu_setting = String::new()), - with(|arguments| arguments.object_prefix = "release".to_string()), - with(|arguments| arguments.object_prefix = "fs-snapshot-bench/".to_string()), - with( - |arguments| arguments.object_prefix = "fs-snapshot-bench/other-run".to_string() - ), - with(|arguments| { - arguments.object_prefix = "fs-snapshot-bench/123-1/extra".to_string() - }), - with(|arguments| arguments.object_prefix = "fs-snapshot-bench/123-1//".to_string()), - with(|arguments| arguments.object_prefix = "fs-snapshot-bench/123-1/".to_string()), - with(|arguments| arguments.tree = "files-tiny".to_string()), - ], - [ - true, false, false, false, false, false, false, false, true, false - ] - ); - } - - #[test] - fn the_object_prefix_of_the_run_replaces_the_prefix_of_the_configuration() { - let config = storage_config("fs-snapshot-bench/123-1/").unwrap(); - - assert_eq!( - match config { - BlobStorageConfig::S3(config) => Some(config.object_prefix), - _ => None, - }, - Some("fs-snapshot-bench/123-1".to_string()) - ); - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/concurrent.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/concurrent.rs deleted file mode 100644 index f9c9c451f9..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/concurrent.rs +++ /dev/null @@ -1,621 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The phases that run the operations of several agents at the same time, in one pod. -//! -//! Each agent has its own repository (see [`super::agents`]). A measured step starts the -//! operations of all agents at the same time on the async runtime. Each operation runs on a -//! blocking thread of the runtime, as the executor runs it. The requests of all agents go into the -//! record of the step. - -use super::agents::copy_first_agent; -use super::measure::measure; -use super::report::{Outcome, StepRecord, StepStatus, TreeFacts}; -use super::requests::times; -use super::trees::{self, CopyCounts}; -use super::{ - COLD_SAVE, PhaseContext, PhaseOutcome, WARM_SAVE, failed, saved_hash, snapshot_name, - without_save, -}; -use crate::filesystem_snapshot::rustic::SaveReport; -use futures::{StreamExt, TryStreamExt}; -use serde_json::{Map, Value, json}; -use std::future::Future; -use std::num::NonZeroUsize; -use std::path::{Path, PathBuf}; -use std::time::{Duration, Instant}; - -/// The result of the operation of one agent in a batch. -struct AgentRun { - /// The time from the start of the operation to its end. - time: Duration, - /// The time from the start of the batch to the end of the operation. - finished: Duration, - result: anyhow::Result, -} - -/// Starts the operations at the same time, and gives the result of each, in the order of the -/// operations. -async fn batch( - operations: impl IntoIterator>>, -) -> Box<[AgentRun]> { - let started = Instant::now(); - futures::future::join_all(operations.into_iter().map(|operation| async move { - let begin = Instant::now(); - let result = operation.await; - AgentRun { - time: begin.elapsed(), - finished: started.elapsed(), - result, - } - })) - .await - .into_boxed_slice() -} - -/// Gives the details of a batch: the number of agents, the times of their operations, the number -/// of operations that failed, and the error of the operation that failed first. -fn batch_details(runs: &[AgentRun]) -> Map { - let mut sorted = runs.iter().map(|run| run.time).collect::>(); - sorted.sort(); - let first_error = runs - .iter() - .filter_map(|run| run.result.as_ref().err().map(|error| (run.finished, error))) - .min_by_key(|(finished, _)| *finished) - .map(|(_, error)| format!("{error:#}")); - let details = json!({ - "agents": runs.len(), - "time_ms": times(&sorted), - "failed": runs.iter().filter(|run| run.result.is_err()).count(), - "first_error": first_error, - }); - match details { - Value::Object(fields) => fields, - _ => Map::new(), - } -} - -/// Gives the record of the step of a batch with the parameters and the details. A batch with a -/// failed operation is a failed step, with the error of the operation that failed first. -fn batch_record(record: StepRecord, parameters: Value, details: Map) -> StepRecord { - let status = match details.get("first_error").and_then(Value::as_str) { - Some(error) => StepStatus::Error(error.into()), - None => record.status.clone(), - }; - StepRecord { - status, - ..record - .with_parameters(parameters) - .with_details(Value::Object(details), Box::default()) - } -} - -/// Tells whether each operation of the batch succeeded. -fn all_ok(runs: &anyhow::Result]>>) -> bool { - runs.as_ref() - .is_ok_and(|runs| runs.iter().all(|run| run.result.is_ok())) -} - -/// A restore phase of several agents: the repository of the first agent goes to each agent that -/// has none, the agents restore the warm save at the same time, and the hash of each restored -/// tree is compared with the hash that the save phase recorded. -/// -/// The phase deletes the restored trees at its end, so the next phase has the space of the -/// volume. -pub(super) async fn concurrent_restore( - context: &PhaseContext, - agents: usize, - reader_threads: Option, -) -> PhaseOutcome { - let into = context.work_dir.join("restore"); - let outcome = restore_agents(context, agents, reader_threads, &into).await; - let _ = tokio::fs::remove_dir_all(&into).await; - outcome -} - -async fn restore_agents( - context: &PhaseContext, - agents: usize, - reader_threads: Option, - into: &Path, -) -> PhaseOutcome { - let storage = &context.storage; - let facts = TreeFacts { - name: context.selection.tree.name, - ..TreeFacts::default() - }; - let parameters = json!({ "agents": agents, "reader_threads": reader_threads }); - let expected = match saved_hash(context).await { - Ok(expected) => expected, - Err(error) => { - return without_save( - facts, - &error, - &["copy_scopes", "concurrent_restore", "hash_trees"], - ); - } - }; - - let (record, copied) = measure( - "copy_scopes", - storage, - copy_first_agent(storage.as_ref(), &context.scope().0, agents), - ) - .await; - let mut steps = vec![record.with_details( - copied.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )]; - if copied.is_err() { - return failed( - facts, - steps, - "copy_scopes", - &["concurrent_restore", "hash_trees"], - ); - } - - let targets = (0..agents) - .map(|agent| into.join(agent.to_string())) - .collect::>(); - let (record, restored) = measure("concurrent_restore", storage, async { - targets.iter().try_for_each(std::fs::create_dir_all)?; - let name = snapshot_name(WARM_SAVE)?; - let name = &name; - Ok(batch( - targets - .iter() - .enumerate() - .map(|(agent, target)| async move { - context - .agent_repository(&agent.to_string()) - .restore(name, target, reader_threads) - .await? - .ok_or_else(|| anyhow::anyhow!("no snapshot has the name {WARM_SAVE}")) - }), - ) - .await) - }) - .await; - steps.push(match &restored { - Ok(runs) => batch_record(record, parameters, batch_details(runs)), - Err(_) => record.with_parameters(parameters), - }); - if !all_ok(&restored) { - return failed(facts, steps, "concurrent_restore", &["hash_trees"]); - } - - compare_hashes(context, facts, steps, &targets, &expected).await -} - -/// Measures the hash of each restored tree, compares each hash with the expected hash, and gives -/// the outcome of the phase with the record of that step. -async fn compare_hashes( - context: &PhaseContext, - facts: TreeFacts, - mut steps: Vec, - targets: &[PathBuf], - expected: &str, -) -> PhaseOutcome { - let (record, hashed) = measure("hash_trees", &context.storage, hash_trees(targets)).await; - let Ok(hashes) = hashed else { - steps.push(record); - return failed(facts, steps, "hash_trees", &[]); - }; - let matches = hashes - .iter() - .filter(|(hash, _)| **hash == *expected) - .count(); - steps.push(record.with_details( - json!({ "trees": hashes.len(), "matches": matches, "expected": expected }), - Box::default(), - )); - let facts = match hashes.first() { - Some((hash, counts)) => TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - hash: Some(hash.clone()), - ..facts - }, - None => facts, - }; - PhaseOutcome { - tree_facts: facts, - steps, - outcome: if matches == hashes.len() { - Outcome::Ok - } else { - Outcome::Failed { - reason: format!( - "{} of {} restored trees differ from the saved tree", - hashes.len() - matches, - hashes.len() - ) - .into(), - } - }, - } -} - -/// Gives the hash of each tree, in the order of the trees. The number of trees that are read at -/// the same time is the number of CPUs that the process can use. -async fn hash_trees(roots: &[PathBuf]) -> anyhow::Result, trees::TreeCounts)]>> { - let parallel = std::thread::available_parallelism().map_or(1, NonZeroUsize::get); - Ok(futures::stream::iter(roots) - .map(|root| trees::hash(root)) - .buffered(parallel) - .try_collect::>() - .await? - .into_boxed_slice()) -} - -/// A save phase of several agents: a new tree and a copy of it for each other agent, the cold -/// saves of all agents at the same time, the small change of each tree, and the warm saves of all -/// agents at the same time. -/// -/// Each agent of the phase has a repository of its own, whose name is the number of the agent -/// after `prefix`. The other phases of the scenario do not use these repositories, so each cold -/// save makes a repository. The copies of the tree are reflinks where the volume has them. After -/// the copies, no page of a tree is in the page cache, so each save reads its tree from the -/// volume. -pub(super) async fn concurrent_save( - context: &PhaseContext, - agents: usize, - prefix: &str, -) -> PhaseOutcome { - let spec = context.selection.tree; - let storage = &context.storage; - let roots = tree_roots(context, agents); - let names = (0..agents) - .map(|agent| format!("{prefix}-{agent}")) - .collect::>(); - let parameters = json!({ "agents": agents }); - let facts = TreeFacts { - name: spec.name, - content: Some(spec.content.label()), - page_cache: Some("dropped"), - ..TreeFacts::default() - }; - let later = [ - "copy_trees", - "concurrent_cold_save", - "small_change", - "concurrent_warm_save", - ]; - - let (record, generated) = - measure("generate_tree", storage, generate_first(context, &roots)).await; - let mut steps = vec![record]; - let Ok(counts) = generated else { - return failed(facts, steps, "generate_tree", &later); - }; - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - ..facts - }; - - let (record, copied) = measure("copy_trees", storage, copy_trees(&roots)).await; - steps.push( - record.with_details( - copied - .as_ref() - .map(|counts| { - json!({ - "copies": agents.saturating_sub(1), - "files_reflinked": counts.reflinked, - "files_copied": counts.copied, - }) - }) - .unwrap_or(Value::Null), - Box::default(), - ), - ); - if copied.is_err() { - return failed(facts, steps, "copy_trees", &later[1..]); - } - - let (record, cold) = measure( - "concurrent_cold_save", - storage, - save_agents(context, &names, &roots, COLD_SAVE), - ) - .await; - steps.push(save_batch_record(record, parameters.clone(), &cold)); - if !all_ok(&cold) { - return failed(facts, steps, "concurrent_cold_save", &later[2..]); - } - - let (record, changed) = measure("small_change", storage, async { - futures::stream::iter(roots.iter()) - .then(|root| trees::change(spec, root)) - .try_collect::>() - .await - }) - .await; - let change = changed - .as_ref() - .ok() - .and_then(|changes| changes.first().cloned()); - steps.push(record.with_details(json!({ "trees": agents, "change": change }), Box::default())); - if changed.is_err() { - return failed(facts, steps, "small_change", &later[3..]); - } - let facts = TreeFacts { change, ..facts }; - - let (record, warm) = measure( - "concurrent_warm_save", - storage, - save_agents(context, &names, &roots, WARM_SAVE), - ) - .await; - steps.push(save_batch_record(record, parameters, &warm)); - if !all_ok(&warm) { - return failed(facts, steps, "concurrent_warm_save", &[]); - } - PhaseOutcome { - tree_facts: facts, - steps, - outcome: Outcome::Ok, - } -} - -/// Gives the root of the tree of each of the agents. -fn tree_roots(context: &PhaseContext, agents: usize) -> Box<[PathBuf]> { - (0..agents) - .map(|agent| context.work_dir.join("trees").join(agent.to_string())) - .collect() -} - -/// Makes the tree of the phase at the first root. -async fn generate_first( - context: &PhaseContext, - roots: &[PathBuf], -) -> anyhow::Result { - let first = roots - .first() - .ok_or_else(|| anyhow::anyhow!("a phase of agents needs one agent or more"))?; - std::fs::create_dir_all(context.work_dir.join("trees"))?; - trees::generate(context.selection.tree, first).await -} - -/// Copies the first tree to each other root, and then removes the pages of each tree from the -/// page cache. -async fn copy_trees(roots: &[PathBuf]) -> anyhow::Result { - let roots = roots.to_vec(); - tokio::task::spawn_blocking(move || { - let counts = match roots.split_first() { - Some((first, others)) => { - others - .iter() - .try_fold(CopyCounts::default(), |counts, root| { - trees::copy_tree(first, root, trees::Times::Drop) - .map(|copied| counts.with(copied)) - })? - } - None => CopyCounts::default(), - }; - roots.iter().try_for_each(|root| trees::settle(root))?; - Ok(counts) - }) - .await? -} - -/// Saves the tree of each agent into the repository of the agent with the snapshot name, all at -/// the same time. -async fn save_agents( - context: &PhaseContext, - names: &[String], - roots: &[PathBuf], - snapshot: &str, -) -> anyhow::Result]>> { - let name = snapshot_name(snapshot)?; - let name = &name; - let settings = context.save_settings(); - Ok( - batch(names.iter().zip(roots).map(|(agent, root)| async move { - context - .agent_repository(agent) - .save_with(name, root, settings) - .await - })) - .await, - ) -} - -/// Gives the record of a step of saves: the details of the batch, and the sums of the data that -/// the saves added. -fn save_batch_record( - record: StepRecord, - parameters: Value, - runs: &anyhow::Result]>>, -) -> StepRecord { - match runs { - Ok(runs) => batch_record(record, parameters, save_batch_details(runs)), - Err(_) => record.with_parameters(parameters), - } -} - -/// Gives the details of a batch of saves, and the sums of the data that the saves added. -fn save_batch_details(runs: &[AgentRun]) -> Map { - let reports = runs - .iter() - .filter_map(|run| run.result.as_ref().ok()) - .collect::>(); - let mut details = batch_details(runs); - details.insert( - "data_added".to_string(), - json!(reports.iter().map(|report| report.data_added).sum::()), - ); - details.insert( - "data_added_packed".to_string(), - json!( - reports - .iter() - .map(|report| report.data_added_packed) - .sum::() - ), - ); - details -} - -/// A mixed phase: the repository of the first agent goes to each of `restores` agents that has -/// none, then `saves` agents save a new tree cold and the `restores` agents restore the warm save -/// cold, all at the same time, and the hash of each restored tree is compared with the hash that -/// the save phase recorded. -/// -/// The saves use the save threads of the variant of the phase, and the restores use -/// `reader_threads`. So the step gives the memory of one choice of both values. The time of a -/// restore is the time until its files are readable, not until they are durable, because nothing -/// syncs the volume. The phase deletes the restored trees at its end. -pub(super) async fn mixed( - context: &PhaseContext, - saves: usize, - restores: usize, - reader_threads: Option, -) -> PhaseOutcome { - let into = context.work_dir.join("restore"); - let outcome = mix(context, saves, restores, reader_threads, &into).await; - let _ = tokio::fs::remove_dir_all(&into).await; - outcome -} - -async fn mix( - context: &PhaseContext, - saves: usize, - restores: usize, - reader_threads: Option, - into: &Path, -) -> PhaseOutcome { - let storage = &context.storage; - let spec = context.selection.tree; - let facts = TreeFacts { - name: spec.name, - content: Some(spec.content.label()), - page_cache: Some("dropped"), - ..TreeFacts::default() - }; - let parameters = json!({ - "saves": saves, - "restores": restores, - "reader_threads": reader_threads, - }); - let later = ["generate_tree", "copy_trees", "mixed", "hash_trees"]; - let expected = match saved_hash(context).await { - Ok(expected) => expected, - Err(error) => { - return without_save( - facts, - &error, - &[ - "copy_scopes", - "generate_tree", - "copy_trees", - "mixed", - "hash_trees", - ], - ); - } - }; - - let (record, copied) = measure( - "copy_scopes", - storage, - copy_first_agent(storage.as_ref(), &context.scope().0, restores + 1), - ) - .await; - let mut steps = vec![record.with_details( - copied.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )]; - if copied.is_err() { - return failed(facts, steps, "copy_scopes", &later); - } - - let roots = tree_roots(context, saves); - let (record, generated) = - measure("generate_tree", storage, generate_first(context, &roots)).await; - steps.push(record); - let Ok(counts) = generated else { - return failed(facts, steps, "generate_tree", &later[1..]); - }; - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - ..facts - }; - let (record, copied) = measure("copy_trees", storage, copy_trees(&roots)).await; - steps.push(record); - if copied.is_err() { - return failed(facts, steps, "copy_trees", &later[2..]); - } - - let names = (0..saves) - .map(|agent| format!("{}-{agent}", context.selection.phase.name)) - .collect::>(); - let targets = (1..=restores) - .map(|agent| into.join(agent.to_string())) - .collect::>(); - let (record, ran) = measure("mixed", storage, async { - targets.iter().try_for_each(std::fs::create_dir_all)?; - let name = snapshot_name(WARM_SAVE)?; - let name = &name; - let restoring = batch( - targets - .iter() - .enumerate() - .map(|(index, target)| async move { - context - .agent_repository(&(index + 1).to_string()) - .restore(name, target, reader_threads) - .await? - .ok_or_else(|| anyhow::anyhow!("no snapshot has the name {WARM_SAVE}")) - }), - ); - let (saved, restored) = - futures::join!(save_agents(context, &names, &roots, COLD_SAVE), restoring); - Ok((saved?, restored)) - }) - .await; - let failed_mix = match &ran { - Ok((saved, restored)) => { - let save_details = save_batch_details(saved); - let restore_details = batch_details(restored); - let first_error = save_details - .get("first_error") - .filter(|error| !error.is_null()) - .or_else(|| restore_details.get("first_error")) - .cloned() - .unwrap_or(Value::Null); - let mut details = Map::new(); - details.insert("first_error".to_string(), first_error); - details.insert("saves".to_string(), Value::Object(save_details)); - details.insert("restores".to_string(), Value::Object(restore_details)); - steps.push(batch_record(record, parameters, details)); - !(saved.iter().all(|run| run.result.is_ok()) - && restored.iter().all(|run| run.result.is_ok())) - } - Err(_) => { - steps.push(record.with_parameters(parameters)); - true - } - }; - if failed_mix { - return failed(facts, steps, "mixed", &later[3..]); - } - compare_hashes(context, facts, steps, &targets, &expected).await -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/golden/phase_result.json b/golem-worker-executor/src/filesystem_snapshot/benchmark/golden/phase_result.json deleted file mode 100644 index 3ced4cd170..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/golden/phase_result.json +++ /dev/null @@ -1,108 +0,0 @@ -{ - "format": "golem-fs-snapshot-benchmark/1", - "run_id": "123-1", - "scenario": "base", - "phase": "save", - "tree": "files-128m", - "cpu_setting": "limit-3", - "environment": { - "pod": "p" - }, - "volume": { - "check": { - "status": "ok" - } - }, - "tree_facts": { - "name": "files-128m", - "files": 10000, - "directories": 100, - "bytes": 134217728, - "content": "incompressible", - "page_cache": "dropped", - "hash": "h1", - "hash_after_change": "h2", - "change": { - "files_rewritten": 10 - } - }, - "steps": [ - { - "name": "cold_save", - "status": "ok", - "parameters": {}, - "wall_ms": 1500.25, - "cpu": { - "user_ms": 900.5, - "system_ms": 100.0, - "cgroup": { - "usage_ms": 1000.0, - "user_ms": 900.0, - "system_ms": 100.0, - "nr_periods": 15, - "nr_throttled": 2, - "throttled_ms": 30.5 - } - }, - "memory": { - "rss_start_bytes": 1000, - "rss_peak_bytes": 5000, - "rss_peak_source": "VmHWM", - "cgroup_current_peak_bytes": 9000, - "cgroup_anon_peak_bytes": 6000, - "cgroup_file_peak_bytes": null - }, - "threads": { - "start": 10, - "peak": 40, - "sample_interval_ms": 10 - }, - "requests": [ - { - "call": "put_raw", - "file_type": "pack", - "count": 3, - "errors": 0, - "bytes": 3000, - "time_ms": { - "min": 1.0, - "p50": 2.0, - "p90": 3.0, - "p99": 3.0, - "max": 3.0, - "total": 6.0 - } - } - ], - "bytes_written": 3000, - "bytes_read": 0, - "phases": [ - { - "name": "backup", - "wall_ms": 1400.0 - } - ], - "details": { - "snapshot": "abc" - } - }, - { - "name": "warm_save", - "status": "skipped", - "parameters": {}, - "wall_ms": null, - "cpu": null, - "memory": null, - "threads": null, - "requests": [], - "bytes_written": 0, - "bytes_read": 0, - "phases": [], - "details": {} - } - ], - "outcome": { - "status": "failed", - "reason": "the step warm_save failed" - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/history.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/history.rs deleted file mode 100644 index b4b163cac3..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/history.rs +++ /dev/null @@ -1,392 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The phases of the prune and repository open scenarios. -//! -//! The history phase makes a repository with a history of saves in the repository of the first -//! agent. Each prune phase copies that repository on the server into the repository of its own -//! agent, and prunes the copy. So each prune starts from the same repository. - -use super::agents::{FIRST_AGENT, agent_blobs, copy_agent, total_bytes}; -use super::measure::measure; -use super::report::{Outcome, StepRecord, TreeFacts}; -use super::{ - COLD_SAVE, PhaseContext, PhaseOutcome, failed, inspect_record, phase_walls, save_record, - saved_hash, snapshot_name, trees, without_save, -}; -use crate::filesystem_snapshot::rustic::{PruneReport, PruneSettings, RepackLimits, Repository}; -use futures::{StreamExt, TryStreamExt}; -use serde_json::{Value, json}; -use std::time::Duration; - -/// The number of saves after the cold save: the warm save and 10 more. -const ROUNDS: u8 = 11; - -/// The saves after which the history phase opens the repository. -const OPEN_AFTER: [u8; 3] = [1, 11, 12]; - -/// The number of newest snapshots that the forget keeps. -const KEEP: u8 = 2; - -/// The name of the snapshot of the save of the round, from 1 to [`ROUNDS`]. -fn snapshot_of_round(round: u8) -> Box { - format!("round-{round:02}").into() -} - -/// The name of the newest snapshot of the history. -fn newest() -> Box { - snapshot_of_round(ROUNDS) -} - -/// The settings of both prunes of a prune phase: no grace period, and no limit that leaves unused -/// data or stops a repack. -const fn prune_settings(fast_repack: bool) -> PruneSettings { - PruneSettings { - fast_repack, - keep_delete: Duration::ZERO, - repack: RepackLimits::Unlimited, - } -} - -/// The history phase: a cold save of a new tree, then a small change and a save in each of -/// [`ROUNDS`] rounds, each round with other content, a forget of every snapshot but the -/// [`KEEP`] newest, and the hash of the tree. -/// -/// The phase opens the repository after the saves in [`OPEN_AFTER`]. Each open finds the newest -/// snapshot by its name and loads the index, as a restore does. -pub(super) async fn history(context: &PhaseContext) -> PhaseOutcome { - let spec = context.selection.tree; - let tree = context.work_dir.join("tree"); - let repository = context.repository(); - let storage = &context.storage; - let settings = context.save_settings(); - let facts = TreeFacts { - name: spec.name, - content: Some(spec.content.label()), - page_cache: Some("dropped"), - ..TreeFacts::default() - }; - - let (record, generated) = measure("generate_tree", storage, trees::generate(spec, &tree)).await; - let mut steps = vec![record]; - let Ok(counts) = generated else { - return failed(facts, steps, "generate_tree", &["cold_save"]); - }; - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - ..facts - }; - - let (record, cold) = measure("cold_save", storage, async { - repository - .save_with(&snapshot_name(COLD_SAVE)?, &tree, settings) - .await - }) - .await; - steps.push(save_record(record, &cold)); - if cold.is_err() { - return failed(facts, steps, "cold_save", &[]); - } - let steps = open_step(context, &repository, COLD_SAVE, 1, steps).await; - - let (repository_ref, tree_ref) = (&repository, tree.as_path()); - let saved = futures::stream::iter(1..=ROUNDS) - .map(Ok::<_, Vec>) - .try_fold(steps, |steps, round| async move { - save_round(context, repository_ref, tree_ref, round, steps).await - }) - .await; - let steps = match saved { - Ok(steps) => steps, - Err(steps) => { - let step = steps.last().map_or("round", |step| step.name); - return failed(facts, steps, step, &[]); - } - }; - - let forgotten = (0..=ROUNDS - KEEP) - .map(|round| match round { - 0 => Box::from(COLD_SAVE), - round => snapshot_of_round(round), - }) - .collect::>(); - let (record, forgot) = measure("forget", storage, async { - futures::stream::iter(forgotten.iter()) - .then(|name| async { repository.forget(&snapshot_name(name)?).await }) - .try_fold(0_u64, |total, count| async move { Ok(total + count) }) - .await - }) - .await; - let mut steps = steps; - steps.push(record.with_details( - json!({ "snapshots_forgotten": forgot.as_ref().ok(), "snapshots_kept": KEEP }), - Box::default(), - )); - if forgot.is_err() { - return failed(facts, steps, "forget", &["hash_tree"]); - } - - let (record, hashed) = measure("hash_tree", storage, trees::hash(&tree)).await; - steps.push(record); - let Ok((hash, _)) = hashed else { - return failed(facts, steps, "hash_tree", &[]); - }; - PhaseOutcome { - tree_facts: TreeFacts { - hash_after_change: Some(hash), - ..facts - }, - steps, - outcome: Outcome::Ok, - } -} - -/// Changes the tree for the round and saves it, and opens the repository after the save when -/// [`OPEN_AFTER`] names it. A failed step gives the records so far as the error. -async fn save_round( - context: &PhaseContext, - repository: &Repository, - tree: &std::path::Path, - round: u8, - mut steps: Vec, -) -> Result, Vec> { - let storage = &context.storage; - let spec = context.selection.tree; - let parameters = json!({ "round": round }); - let (record, changed) = measure( - "small_change", - storage, - trees::change_round(spec, tree, round), - ) - .await; - steps.push(record.with_parameters(parameters.clone()).with_details( - changed.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )); - if changed.is_err() { - return Err(steps); - } - let name = snapshot_of_round(round); - let (record, saved) = measure("save", storage, async { - repository - .save_with(&snapshot_name(&name)?, tree, context.save_settings()) - .await - }) - .await; - steps.push(save_record(record, &saved).with_parameters(parameters)); - if saved.is_err() { - return Err(steps); - } - Ok(open_step(context, repository, &name, round + 1, steps).await) -} - -/// Opens the repository and finds the snapshot with the name, when the number of saves is in -/// [`OPEN_AFTER`], and gives the steps with the record of the open. A failed open is recorded and -/// does not stop the phase. -async fn open_step( - context: &PhaseContext, - repository: &Repository, - name: &str, - saves: u8, - mut steps: Vec, -) -> Vec { - if OPEN_AFTER.contains(&saves) { - let (record, inspected) = measure("open", &context.storage, async { - repository.inspect(&snapshot_name(name)?).await - }) - .await; - steps.push(inspect_record(record, &inspected).with_parameters(json!({ "saves": saves }))); - } - steps -} - -/// A prune phase: the repository of the history phase goes on the server to the agent of the -/// variant of the phase, two prunes with no grace period run on it, the first to mark the packs -/// that no snapshot uses and the second to delete them, then an open, and a restore of the newest -/// snapshot whose hash is compared with the hash of the history phase. -/// -/// The details of the second prune give the listed bytes of the repository before the first prune -/// and after the second, and the difference, which is what the prunes gave back. The restore time -/// is the time until the files are readable, not until they are durable. -pub(super) async fn prune(context: &PhaseContext, fast_repack: bool) -> PhaseOutcome { - let into = context.work_dir.join("restore"); - let storage = &context.storage; - let namespace = context.scope().0; - let agent = context.variant().agent; - let repository = context.repository(); - let settings = prune_settings(fast_repack); - let parameters = json!({ - "fast_repack": fast_repack, - "keep_delete_s": settings.keep_delete.as_secs(), - "max_unused": "0%", - "max_repack": "unlimited", - }); - let facts = TreeFacts { - name: context.selection.tree.name, - ..TreeFacts::default() - }; - let later = [ - "copy_scopes", - "prune_mark", - "prune_delete", - "open", - "cold_restore", - "hash_tree", - ]; - let expected = match saved_hash(context).await { - Ok(expected) => expected, - Err(error) => return without_save(facts, &error, &later), - }; - - let (record, copied) = measure( - "copy_scopes", - storage, - copy_agent(storage.as_ref(), &namespace, FIRST_AGENT, agent), - ) - .await; - let mut steps = vec![record.with_details( - copied.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )]; - if copied.is_err() { - return failed(facts, steps, "copy_scopes", &later[1..]); - } - let before = agent_blobs(storage.as_ref(), &namespace, agent).await; - - let (record, marked) = measure("prune_mark", storage, repository.prune(settings)).await; - steps.push(prune_record(record, &marked, parameters.clone())); - if marked.is_err() { - return failed(facts, steps, "prune_mark", &later[2..]); - } - let (record, deleted) = measure("prune_delete", storage, repository.prune(settings)).await; - let after = agent_blobs(storage.as_ref(), &namespace, agent).await; - let record = prune_record(record, &deleted, parameters); - let given_back = match (&before, &after) { - (Ok(before), Ok(after)) => json!({ - "bytes_before": total_bytes(before), - "bytes_after": total_bytes(after), - "bytes_given_back": total_bytes(before).saturating_sub(total_bytes(after)), - "blobs_before": before.len(), - "blobs_after": after.len(), - }), - _ => Value::Null, - }; - steps.push(with_detail(record, "repository", given_back)); - if deleted.is_err() { - return failed(facts, steps, "prune_delete", &later[3..]); - } - - let newest = newest(); - let (record, inspected) = measure("open", storage, async { - repository.inspect(&snapshot_name(&newest)?).await - }) - .await; - steps.push(inspect_record(record, &inspected).with_parameters(json!({ "saves": "pruned" }))); - - let (record, restored) = measure("cold_restore", storage, async { - std::fs::create_dir(&into)?; - repository - .restore(&snapshot_name(&newest)?, &into, None) - .await? - .ok_or_else(|| anyhow::anyhow!("no snapshot has the name {newest}")) - }) - .await; - steps.push(match &restored { - Ok(report) => record.with_details( - json!({ "files": report.files, "dirs": report.dirs, "bytes": report.bytes }), - phase_walls(&report.phases), - ), - Err(_) => record, - }); - if restored.is_err() { - return failed(facts, steps, "cold_restore", &later[5..]); - } - let (record, hashed) = measure("hash_tree", storage, trees::hash(&into)).await; - let _ = tokio::fs::remove_dir_all(&into).await; - let Ok((hash, counts)) = hashed else { - steps.push(record); - return failed(facts, steps, "hash_tree", &[]); - }; - let matches = *hash == *expected; - steps.push(record.with_details( - json!({ "hash": hash, "expected": expected, "matches": matches }), - Box::default(), - )); - PhaseOutcome { - tree_facts: TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - hash: Some(hash), - ..facts - }, - steps, - outcome: if matches { - Outcome::Ok - } else { - Outcome::Failed { - reason: "the restored tree differs from the saved tree".into(), - } - }, - } -} - -/// Gives the record of a prune with the plan of the prune. -fn prune_record( - record: StepRecord, - pruned: &anyhow::Result>, - parameters: Value, -) -> StepRecord { - let record = record.with_parameters(parameters); - match pruned { - Ok(Some(report)) => record.with_details(prune_details(report), phase_walls(&report.phases)), - Ok(None) => record.with_details(json!({ "repository": null }), Box::default()), - Err(_) => record, - } -} - -fn prune_details(report: &PruneReport) -> Value { - json!({ - "packs_used": report.packs_used, - "packs_partly_used": report.packs_partly_used, - "packs_unused": report.packs_unused, - "packs_repacked": report.packs_repacked, - "packs_kept": report.packs_kept, - "marked_packs_deleted": report.marked_packs_deleted, - "marked_bytes_deleted": report.marked_bytes_deleted, - "marked_packs_kept": report.marked_packs_kept, - "bytes_used": report.bytes_used, - "bytes_unused": report.bytes_unused, - "bytes_removed": report.bytes_removed, - "bytes_repacked": report.bytes_repacked, - "bytes_repack_removed": report.bytes_repack_removed, - "index_files": report.index_files, - "index_files_rebuilt": report.index_files_rebuilt, - }) -} - -/// Gives the record with the value at the key in its details. -fn with_detail(record: StepRecord, key: &str, value: Value) -> StepRecord { - let details = match record.details.clone() { - Value::Object(mut details) => { - details.insert(key.to_string(), value); - Value::Object(details) - } - other => other, - }; - let phases = record.phases.clone(); - record.with_details(details, phases) -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/measure.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/measure.rs deleted file mode 100644 index ca55941104..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/measure.rs +++ /dev/null @@ -1,435 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! Measures one step of a phase: wall time, CPU time, memory, threads and blob storage requests. -//! -//! The values come from the process and from the cgroup v2 files of the container. A value that -//! the host does not give is `null` in the result. - -use super::report::{ - CgroupCpuTime, CpuTime, MemoryPeaks, StepRecord, StepStatus, ThreadCounts, millis, -}; -use super::requests::{MeasuredBlobStorage, summarize, written_and_read}; -use serde_json::Value; -use std::future::Future; -use std::path::Path; -use std::sync::Arc; -use std::sync::atomic::{AtomicBool, Ordering}; -use std::time::{Duration, Instant}; - -/// The time between two samples of the memory and the threads. -const SAMPLE_INTERVAL: Duration = Duration::from_millis(10); - -const PROC_STATUS: &str = "/proc/self/status"; -const PROC_CLEAR_REFS: &str = "/proc/self/clear_refs"; -const CGROUP_CPU_STAT: &str = "/sys/fs/cgroup/cpu.stat"; -const CGROUP_MEMORY_CURRENT: &str = "/sys/fs/cgroup/memory.current"; -const CGROUP_MEMORY_STAT: &str = "/sys/fs/cgroup/memory.stat"; -const CGROUP_MEMORY_EVENTS: &str = "/sys/fs/cgroup/memory.events"; - -/// Runs the step and measures it. -/// -/// The requests of the storage before the step are removed first, so the record holds only the -/// requests of the step. A line on the standard error tells that the step starts, so the log of a -/// pod that the kernel stopped shows the step that ran. -pub(super) async fn measure( - name: &'static str, - storage: &MeasuredBlobStorage, - step: impl Future>, -) -> (StepRecord, anyhow::Result) { - eprintln!("fs-snapshot-benchmark: the step {name} starts"); - let _ = storage.take(); - let before = ProcessSample::read(); - let peak_reset = reset_peak_rss(); - let sampler = Sampler::start(); - let started = Instant::now(); - let result = step.await; - let wall = started.elapsed(); - let peaks = sampler.stop(); - let after = ProcessSample::read(); - let records = storage.take(); - let (bytes_written, bytes_read) = written_and_read(&records); - let (rss_peak, rss_peak_source) = match after.status.peak_rss { - Some(peak) if peak_reset => (peak, "VmHWM"), - _ => (peaks.rss, "sampled VmRSS"), - }; - let record = StepRecord { - name, - status: match &result { - Ok(_) => StepStatus::Ok, - Err(error) => StepStatus::Error(format!("{error:#}").into()), - }, - parameters: Value::Object(Default::default()), - wall_ms: Some(millis(wall)), - cpu: Some(CpuTime { - user_ms: millis(after.user.saturating_sub(before.user)), - system_ms: millis(after.system.saturating_sub(before.system)), - cgroup: before - .cgroup_cpu - .zip(after.cgroup_cpu) - .map(|(before, after)| after.since(&before)), - }), - memory: Some(MemoryPeaks { - rss_start_bytes: before.status.rss.unwrap_or_default(), - rss_peak_bytes: rss_peak, - rss_peak_source, - cgroup_current_peak_bytes: peaks.cgroup_current, - cgroup_anon_peak_bytes: peaks.cgroup_anon, - cgroup_file_peak_bytes: peaks.cgroup_file, - }), - threads: Some(ThreadCounts { - start: before.status.threads.unwrap_or_default(), - peak: peaks.threads, - sample_interval_ms: SAMPLE_INTERVAL.as_millis() as u64, - }), - requests: summarize(&records), - bytes_written, - bytes_read, - phases: Box::default(), - details: Value::Object(Default::default()), - }; - (record, result) -} - -/// The values of `/proc/self/status` that a step uses. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub(super) struct StatusValues { - pub(super) rss: Option, - pub(super) peak_rss: Option, - pub(super) threads: Option, -} - -/// Reads `VmRSS`, `VmHWM` and `Threads` from the text of `/proc//status`. The sizes are in -/// bytes. -pub(super) fn parse_status(text: &str) -> StatusValues { - text.lines().filter_map(|line| line.split_once(':')).fold( - StatusValues::default(), - |values, (key, value)| { - let number = || value.split_whitespace().next()?.parse::().ok(); - match key { - "VmRSS" => StatusValues { - rss: number().map(|kib| kib * 1024), - ..values - }, - "VmHWM" => StatusValues { - peak_rss: number().map(|kib| kib * 1024), - ..values - }, - "Threads" => StatusValues { - threads: number(), - ..values - }, - _ => values, - } - }, - ) -} - -/// The CPU counters of a cgroup v2, in microseconds. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub(super) struct CgroupCpu { - pub(super) usage_usec: u64, - pub(super) user_usec: u64, - pub(super) system_usec: u64, - pub(super) nr_periods: u64, - pub(super) nr_throttled: u64, - pub(super) throttled_usec: u64, -} - -impl CgroupCpu { - fn since(&self, before: &Self) -> CgroupCpuTime { - let micros = - |after: u64, before: u64| millis(Duration::from_micros(after.saturating_sub(before))); - CgroupCpuTime { - usage_ms: micros(self.usage_usec, before.usage_usec), - user_ms: micros(self.user_usec, before.user_usec), - system_ms: micros(self.system_usec, before.system_usec), - nr_periods: self.nr_periods.saturating_sub(before.nr_periods), - nr_throttled: self.nr_throttled.saturating_sub(before.nr_throttled), - throttled_ms: micros(self.throttled_usec, before.throttled_usec), - } - } -} - -/// Reads the text of a cgroup v2 `cpu.stat`. A counter that the text does not have is zero; the -/// throttling counters are only there when the cgroup has a CPU limit controller. -pub(super) fn parse_cpu_stat(text: &str) -> CgroupCpu { - text.lines() - .filter_map(|line| line.split_once(' ')) - .filter_map(|(key, value)| value.trim().parse::().ok().map(|value| (key, value))) - .fold(CgroupCpu::default(), |counters, (key, value)| match key { - "usage_usec" => CgroupCpu { - usage_usec: value, - ..counters - }, - "user_usec" => CgroupCpu { - user_usec: value, - ..counters - }, - "system_usec" => CgroupCpu { - system_usec: value, - ..counters - }, - "nr_periods" => CgroupCpu { - nr_periods: value, - ..counters - }, - "nr_throttled" => CgroupCpu { - nr_throttled: value, - ..counters - }, - "throttled_usec" => CgroupCpu { - throttled_usec: value, - ..counters - }, - _ => counters, - }) -} - -/// Reads the value of a key of a cgroup v2 `memory.stat`, in bytes. -pub(super) fn memory_stat_value(text: &str, key: &str) -> Option { - text.lines() - .filter_map(|line| line.split_once(' ')) - .find(|(name, _)| *name == key) - .and_then(|(_, value)| value.trim().parse().ok()) -} - -/// Reads the counters of the text of a cgroup v2 `memory.events`, as a JSON object from the name -/// of each counter to its value. A line that has no counter is not in the object. -pub(super) fn parse_memory_events(text: &str) -> Value { - Value::Object( - text.lines() - .filter_map(|line| line.split_once(' ')) - .filter_map(|(key, value)| { - value - .trim() - .parse::() - .ok() - .map(|value| (key.to_string(), Value::from(value))) - }) - .collect(), - ) -} - -/// Gives the counters of the cgroup `memory.events` of the process, or `null` when the host does -/// not give them. -pub(super) fn memory_events() -> Value { - read_text(CGROUP_MEMORY_EVENTS).map_or(Value::Null, |text| parse_memory_events(&text)) -} - -/// The CPU time and the status of the process at one moment. -struct ProcessSample { - user: Duration, - system: Duration, - status: StatusValues, - cgroup_cpu: Option, -} - -impl ProcessSample { - fn read() -> Self { - let (user, system) = process_cpu_time(); - Self { - user, - system, - status: read_status(), - cgroup_cpu: read_text(CGROUP_CPU_STAT).map(|text| parse_cpu_stat(&text)), - } - } -} - -/// Gives the user and the system CPU time of all threads of the process. -fn process_cpu_time() -> (Duration, Duration) { - // SAFETY: `getrusage` writes one `rusage` record, which the zeroed value holds. - let usage = unsafe { - let mut usage = std::mem::zeroed::(); - if libc::getrusage(libc::RUSAGE_SELF, &mut usage) == 0 { - Some(usage) - } else { - None - } - }; - usage.map_or((Duration::ZERO, Duration::ZERO), |usage| { - (time_of(usage.ru_utime), time_of(usage.ru_stime)) - }) -} - -fn time_of(time: libc::timeval) -> Duration { - Duration::from_secs(u64::try_from(time.tv_sec).unwrap_or_default()) - + Duration::from_micros(u64::try_from(time.tv_usec).unwrap_or_default()) -} - -fn read_status() -> StatusValues { - read_text(PROC_STATUS) - .map(|text| parse_status(&text)) - .unwrap_or_default() -} - -fn read_text(path: &str) -> Option { - std::fs::read_to_string(Path::new(path)).ok() -} - -/// Sets `VmHWM` of the process to its current RSS, and tells whether that worked. -fn reset_peak_rss() -> bool { - std::fs::write(PROC_CLEAR_REFS, b"5").is_ok() -} - -/// The largest values that the sampler saw. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -struct Peaks { - rss: u64, - threads: u64, - cgroup_current: Option, - cgroup_anon: Option, - cgroup_file: Option, -} - -impl Peaks { - fn with(self, sample: Peaks) -> Self { - let max = |left: Option, right: Option| left.max(right); - Self { - rss: self.rss.max(sample.rss), - threads: self.threads.max(sample.threads), - cgroup_current: max(self.cgroup_current, sample.cgroup_current), - cgroup_anon: max(self.cgroup_anon, sample.cgroup_anon), - cgroup_file: max(self.cgroup_file, sample.cgroup_file), - } - } - - fn read() -> Self { - let status = read_status(); - let memory_stat = read_text(CGROUP_MEMORY_STAT); - Self { - rss: status.rss.unwrap_or_default(), - threads: status.threads.unwrap_or_default(), - cgroup_current: read_text(CGROUP_MEMORY_CURRENT) - .and_then(|text| text.trim().parse().ok()), - cgroup_anon: memory_stat - .as_deref() - .and_then(|text| memory_stat_value(text, "anon")), - cgroup_file: memory_stat - .as_deref() - .and_then(|text| memory_stat_value(text, "file")), - } - } -} - -/// A thread that samples the memory and the threads of the process until it is stopped. -struct Sampler { - stop: Arc, - thread: std::thread::JoinHandle, -} - -impl Sampler { - fn start() -> Self { - let stop = Arc::new(AtomicBool::new(false)); - let thread = std::thread::spawn({ - let stop = stop.clone(); - move || { - std::iter::from_fn(|| { - (!stop.load(Ordering::Acquire)).then(|| { - let sample = Peaks::read(); - std::thread::sleep(SAMPLE_INTERVAL); - sample - }) - }) - .fold(Peaks::default(), Peaks::with) - } - }); - Self { stop, thread } - } - - fn stop(self) -> Peaks { - self.stop.store(true, Ordering::Release); - self.thread.join().unwrap_or_default().with(Peaks::read()) - } -} - -#[cfg(test)] -mod tests { - use super::{ - CgroupCpu, StatusValues, memory_stat_value, parse_cpu_stat, parse_memory_events, - parse_status, - }; - use pretty_assertions::assert_eq; - use serde_json::json; - use test_r::test; - - #[test] - fn the_memory_events_give_each_counter() { - let text = "low 0\nhigh 12\nmax 34\noom 1\noom_kill 1\noom_group_kill 0\n"; - - assert_eq!( - (parse_memory_events(text), parse_memory_events("")), - ( - json!({ - "low": 0, - "high": 12, - "max": 34, - "oom": 1, - "oom_kill": 1, - "oom_group_kill": 0 - }), - json!({}) - ) - ); - } - - #[test] - fn the_status_gives_the_rss_the_peak_rss_and_the_threads() { - let text = "Name:\tbench\nVmPeak:\t 900 kB\nVmHWM:\t 300 kB\nVmRSS:\t 200 kB\nThreads:\t12\n"; - - assert_eq!( - (parse_status(text), parse_status("Name:\tbench\n")), - ( - StatusValues { - rss: Some(200 * 1024), - peak_rss: Some(300 * 1024), - threads: Some(12), - }, - StatusValues::default() - ) - ); - } - - #[test] - fn the_cpu_stat_gives_the_usage_and_the_throttling() { - let text = "usage_usec 1000\nuser_usec 700\nsystem_usec 300\nnr_periods 40\nnr_throttled 5\nthrottled_usec 2500\nnr_bursts 0\n"; - - assert_eq!( - parse_cpu_stat(text), - CgroupCpu { - usage_usec: 1000, - user_usec: 700, - system_usec: 300, - nr_periods: 40, - nr_throttled: 5, - throttled_usec: 2500, - } - ); - } - - #[test] - fn the_memory_stat_gives_the_value_of_a_key() { - let text = "anon 4096\nfile 8192\nfile_mapped 100\n"; - - assert_eq!( - ( - memory_stat_value(text, "anon"), - memory_stat_value(text, "file"), - memory_stat_value(text, "shmem") - ), - (Some(4096), Some(8192), None) - ); - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs deleted file mode 100644 index 7b05b1ad9e..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/mod.rs +++ /dev/null @@ -1,1320 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! A benchmark of the save, restore and forget operations of the rustic repository over the blob -//! storage of the executor. -//! -//! A scenario has trees and phases. A phase is the work of one pod, and it runs on one tree. A -//! phase measures each of its steps, and writes its result as JSON into the blob storage and on -//! the standard output. A later phase of the same scenario and tree reads the result of an -//! earlier phase from the blob storage. -//! -//! Each repository and each result is in the namespace `InitialAgentFiles` of an environment -//! that only the benchmark uses, below the object prefix of the run. The repositories of a -//! scenario, a CPU setting and a tree are the repositories of its agents (see [`agents`]). - -mod agents; -mod capture; -pub mod cli; -mod concurrent; -mod history; -mod measure; -mod report; -mod requests; -mod scopes; -mod sqlite; -mod trees; -mod volume; - -use super::rustic::{ - ChangeDetection, Chunking, Compression, InspectReport, PhaseTime, Repository, RepositoryKey, - RepositorySettings, SaveSettings, -}; -use super::{SnapshotName, SnapshotScope}; -use crate::services::golem_config::DEFAULT_FILESYSTEM_SNAPSHOT_STORAGE_CALL_DEADLINE as STORAGE_CALL_DEADLINE; -use agents::{AgentStorage, FIRST_AGENT}; -use golem_common::model::environment::EnvironmentId; -use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; -use measure::measure; -use report::{FORMAT, Outcome, PhaseResult, PhaseWall, StepRecord, TreeFacts, millis}; -use requests::MeasuredBlobStorage; -use serde::Serialize; -use serde_json::{Map, Value, json}; -use std::num::{NonZeroI32, NonZeroU32, NonZeroUsize}; -use std::path::{Path, PathBuf}; -use std::sync::Arc; -use std::time::Instant; -use trees::{ - COMPRESSIBLE_1G, FILES_1G, FILES_128M, FILES_TINY, MODULES_128M, OBJECTS_1G, OBJECTS_128M, - SQLITE_1G, SQLITE_TINY, TreeSpec, -}; -use uuid::Uuid; - -/// The labels of the blob storage calls of the benchmark itself. -const TARGET_LABEL: &str = "filesystem_snapshot_benchmark"; - -/// The namespace of the UUIDs of the environments of the repositories. -const REPOSITORY_ENVIRONMENTS: Uuid = Uuid::from_u128(0x6f1c_5d2e_9a4b_4c3d_8e7f_0a1b_2c3d_4e5f); - -/// The names of the snapshots of the base scenario. -const COLD_SAVE: &str = "cold-save"; -const WARM_SAVE: &str = "warm-save"; - -/// A scenario: its trees and the phases that run on each tree, in order. -struct Scenario { - name: &'static str, - trees: &'static [TreeSpec], - phases: &'static [Phase], - /// The phases that run under the lower memory limit of the workflow. - memory_limited_phases: &'static [&'static str], -} - -/// A phase of a scenario: the work of one pod. -struct Phase { - name: &'static str, - kind: PhaseKind, - variant: &'static Variant, -} - -/// The repository that a phase uses, and the settings of its repositories and its saves. -#[derive(Debug)] -struct Variant { - /// The agent whose repository the phase uses. - agent: &'static str, - /// The phase whose result holds the hash that a restore of the phase compares with. - save_phase: &'static str, - /// The settings of the repositories and the saves of the phase. `None` is the defaults, and - /// then the steps of the phase record no settings. - settings: Option, -} - -impl Variant { - const fn new(agent: &'static str, save_phase: &'static str, settings: Settings) -> Self { - Self { - agent, - save_phase, - settings: Some(settings), - } - } -} - -/// The settings of a repository that a save makes, and of each save. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -struct Settings { - repository: RepositorySettings, - save: SaveSettings, -} - -impl Settings { - const DEFAULT: Self = Self { - repository: RepositorySettings::DEFAULT, - save: SaveSettings::DEFAULT, - }; - - /// The defaults, with the number of threads of each stage of a save. - const fn save_threads(threads: usize) -> Self { - Self { - save: SaveSettings { - threads: NonZeroUsize::new(threads), - ..SaveSettings::DEFAULT - }, - ..Self::DEFAULT - } - } - - /// The defaults, with the settings of the repository. - const fn repository(repository: RepositorySettings) -> Self { - Self { - repository, - ..Self::DEFAULT - } - } -} - -/// The variant of the phases of the earlier scenarios: the repository of the first agent, the -/// defaults, and no settings in the steps. -const BASE: Variant = Variant { - agent: FIRST_AGENT, - save_phase: "save", - settings: None, -}; - -/// The variant of most phases of the later scenarios: the repository of the first agent and the -/// defaults, which the steps record. -const DEFAULTS: Variant = Variant::new(FIRST_AGENT, "save", Settings::DEFAULT); - -/// What a phase does. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum PhaseKind { - /// A cold save of a new tree, a small change and a warm save, into the repository of the - /// first agent. - Save, - /// A cold restore of the warm save of the first agent with the number of reader threads. - /// `None` is the default of rustic. - Restore { - reader_threads: Option, - }, - /// The restores of the warm save into the repositories of `agents` agents at the same time. - ConcurrentRestore { - agents: usize, - reader_threads: Option, - }, - /// The cold saves of `agents` agents at the same time, a small change of the tree of each - /// agent, and then the warm saves of the agents at the same time. The repository of each agent - /// has the number of the agent after `x-`. - ConcurrentSave { agents: usize }, - /// The phase of [`PhaseKind::ConcurrentSave`], in which the repository of each agent has the - /// number of the agent after the name of the phase and `-`. - NamedConcurrentSave { agents: usize }, - /// A reflink capture of a new tree, a cold save, and two warm saves of a later capture, one - /// with each change detection (see [`capture`]). - Capture, - /// Twelve saves with a different change each, a forget of all but the two newest snapshots, - /// and the open of the repository after 1, 11 and 12 saves (see [`history`]). - History, - /// A copy of the repository of the history phase, two prunes with no grace period, an open - /// and a restore (see [`history`]). - Prune { fast_repack: bool }, - /// Saves after a clustered and after a scattered change of a SQLite tree (see [`sqlite`]). - SqliteChanges, - /// The cold saves of `saves` agents and the cold restores of `restores` agents, all at the - /// same time, with the number of reader threads of each restore. - Mixed { - saves: usize, - restores: usize, - reader_threads: Option, - }, - /// A copy of the repository of the save phase to another agent, and its deletion (see - /// [`scopes`]). - Scopes, - /// The phase of [`PhaseKind::Save`], and then an open of the repository, whose record gives - /// the settings that the repository has. - SaveAndOpen, -} - -const SAVE: Phase = Phase { - name: "save", - kind: PhaseKind::Save, - variant: &BASE, -}; - -const fn restore(name: &'static str, reader_threads: usize) -> Phase { - Phase { - name, - kind: PhaseKind::Restore { - reader_threads: NonZeroUsize::new(reader_threads), - }, - variant: &BASE, - } -} - -const fn concurrent_restore(name: &'static str, agents: usize, reader_threads: usize) -> Phase { - Phase { - name, - kind: PhaseKind::ConcurrentRestore { - agents, - reader_threads: NonZeroUsize::new(reader_threads), - }, - variant: &BASE, - } -} - -const fn concurrent_save(name: &'static str, agents: usize) -> Phase { - Phase { - name, - kind: PhaseKind::ConcurrentSave { agents }, - variant: &BASE, - } -} - -const BASE_PHASES: &[Phase] = &[ - SAVE, - Phase { - name: "restore", - kind: PhaseKind::Restore { - reader_threads: None, - }, - variant: &BASE, - }, -]; - -const RESTORE_THREADS_PHASES: &[Phase] = &[ - SAVE, - restore("restore-1", 1), - restore("restore-2", 2), - restore("restore-4", 4), - restore("restore-8", 8), - restore("restore-20", 20), -]; - -const CONCURRENT_RESTORE_1_PHASES: &[Phase] = &[ - SAVE, - concurrent_restore("restore-x1", 1, 1), - concurrent_restore("restore-x5", 5, 1), - concurrent_restore("restore-x10", 10, 1), - concurrent_restore("restore-x25", 25, 1), - concurrent_restore("restore-x50", 50, 1), - concurrent_restore("restore-x100", 100, 1), - concurrent_restore("restore-x200", 200, 1), -]; - -const CONCURRENT_RESTORE_4_PHASES: &[Phase] = &[ - SAVE, - concurrent_restore("restore-x1", 1, 4), - concurrent_restore("restore-x5", 5, 4), - concurrent_restore("restore-x10", 10, 4), - concurrent_restore("restore-x25", 25, 4), - concurrent_restore("restore-x50", 50, 4), - concurrent_restore("restore-x100", 100, 4), - concurrent_restore("restore-x200", 200, 4), -]; - -const CONCURRENT_RESTORE_20_PHASES: &[Phase] = &[ - SAVE, - concurrent_restore("restore-x1", 1, 20), - concurrent_restore("restore-x5", 5, 20), - concurrent_restore("restore-x10", 10, 20), - concurrent_restore("restore-x25", 25, 20), - concurrent_restore("restore-x50", 50, 20), - concurrent_restore("restore-x100", 100, 20), - concurrent_restore("restore-x200", 200, 20), -]; - -const CONCURRENT_SAVE_PHASES: &[Phase] = &[ - concurrent_save("save-x1", 1), - concurrent_save("save-x2", 2), - concurrent_save("save-x4", 4), - concurrent_save("save-x8", 8), -]; - -/// A phase of a later scenario with the defaults, which its steps record. -const fn with_defaults(name: &'static str, kind: PhaseKind) -> Phase { - Phase { - name, - kind, - variant: &DEFAULTS, - } -} - -/// A save phase of a later scenario: a cold save, a small change and a warm save into the -/// repository of the variant, and an open of the repository. -const fn save_phase(name: &'static str, variant: &'static Variant) -> Phase { - Phase { - name, - kind: PhaseKind::SaveAndOpen, - variant, - } -} - -const DEFAULT_PHASES: &[Phase] = &[ - with_defaults("save", PhaseKind::Save), - with_defaults( - "restore", - PhaseKind::Restore { - reader_threads: None, - }, - ), -]; - -const CAPTURE_PHASES: &[Phase] = &[with_defaults("capture", PhaseKind::Capture)]; - -/// The phase of the saves that the prune phases prune. -const HISTORY: Phase = with_defaults("history", PhaseKind::History); - -/// A prune phase, which prunes a copy of the repository of the history phase in the repository of -/// the agent with the name of the phase. -const fn prune_phase(name: &'static str, variant: &'static Variant, fast_repack: bool) -> Phase { - Phase { - name, - kind: PhaseKind::Prune { fast_repack }, - variant, - } -} - -const PRUNE: Variant = Variant::new("prune", "history", Settings::DEFAULT); -const PRUNE_FAST_REPACK: Variant = Variant::new("prune-fast-repack", "history", Settings::DEFAULT); - -const PRUNE_PHASES: &[Phase] = &[ - HISTORY, - prune_phase("prune", &PRUNE, false), - prune_phase("prune-fast-repack", &PRUNE_FAST_REPACK, true), -]; - -const REPOSITORY_OPEN_PHASES: &[Phase] = &[HISTORY, prune_phase("prune", &PRUNE, false)]; - -/// A phase of 4 concurrent saves whose repositories have the name of the phase as prefix. -const fn save_threads_phase(name: &'static str, variant: &'static Variant) -> Phase { - Phase { - name, - kind: PhaseKind::NamedConcurrentSave { agents: 4 }, - variant, - } -} - -const SAVE_THREADS_1: Variant = Variant::new(FIRST_AGENT, "save", Settings::save_threads(1)); -const SAVE_THREADS_2: Variant = Variant::new(FIRST_AGENT, "save", Settings::save_threads(2)); -const SAVE_THREADS_4: Variant = Variant::new(FIRST_AGENT, "save", Settings::save_threads(4)); - -const SAVE_THREADS_PHASES: &[Phase] = &[ - save_threads_phase("save-x4-t1", &SAVE_THREADS_1), - save_threads_phase("save-x4-t2", &SAVE_THREADS_2), - save_threads_phase("save-x4-t4", &SAVE_THREADS_4), - save_threads_phase("save-x4-tdefault", &DEFAULTS), -]; - -/// The number of saves and of restores of a mixed phase. -const MIXED_SAVES: usize = 4; -const MIXED_RESTORES: usize = 4; - -/// A mixed phase: the saves have the save threads of the variant, and the restores have the -/// reader threads. -const fn mixed_phase( - name: &'static str, - variant: &'static Variant, - reader_threads: usize, -) -> Phase { - Phase { - name, - kind: PhaseKind::Mixed { - saves: MIXED_SAVES, - restores: MIXED_RESTORES, - reader_threads: NonZeroUsize::new(reader_threads), - }, - variant, - } -} - -const MIXED_PHASES: &[Phase] = &[ - with_defaults("save", PhaseKind::Save), - mixed_phase("mixed-s2-r2", &SAVE_THREADS_2, 2), - mixed_phase("mixed-s2-r4", &SAVE_THREADS_2, 4), - mixed_phase("mixed-s4-r2", &SAVE_THREADS_4, 2), - mixed_phase("mixed-s4-r4", &SAVE_THREADS_4, 4), -]; - -/// The settings of a repository with fixed chunks of 64 KiB, which are 16 SQLite pages. -const FIXED_64K: RepositorySettings = RepositorySettings { - chunking: match NonZeroU32::new(64 * 1024) { - Some(size) => Chunking::Fixed(size), - None => Chunking::Rabin, - }, - ..RepositorySettings::DEFAULT -}; - -const SQLITE_RABIN: Variant = Variant::new("rabin", "save-rabin", Settings::DEFAULT); -const SQLITE_FIXED_64K: Variant = Variant::new( - "fixed-64k", - "save-fixed-64k", - Settings::repository(FIXED_64K), -); - -const SQLITE_CHANGES_PHASES: &[Phase] = &[ - Phase { - name: "save-rabin", - kind: PhaseKind::SqliteChanges, - variant: &SQLITE_RABIN, - }, - Phase { - name: "restore-rabin", - kind: PhaseKind::Restore { - reader_threads: None, - }, - variant: &SQLITE_RABIN, - }, - Phase { - name: "save-fixed-64k", - kind: PhaseKind::SqliteChanges, - variant: &SQLITE_FIXED_64K, - }, - Phase { - name: "restore-fixed-64k", - kind: PhaseKind::Restore { - reader_threads: None, - }, - variant: &SQLITE_FIXED_64K, - }, -]; - -/// The compression with the zstd level, or no compression for the level 0. -const fn zstd_level(level: i32) -> Compression { - match NonZeroI32::new(level) { - Some(level) => Compression::Level(level), - None => Compression::Off, - } -} - -const CPU_DEFAULT: Variant = Variant::new("default", "save-default", Settings::DEFAULT); -const CPU_VERIFY_OFF: Variant = Variant::new( - "verify-off", - "save-verify-off", - Settings::repository(RepositorySettings { - extra_verify: false, - ..RepositorySettings::DEFAULT - }), -); -const CPU_ZSTD_OFF: Variant = Variant::new( - "zstd-off", - "save-zstd-off", - Settings::repository(RepositorySettings { - compression: Compression::Off, - ..RepositorySettings::DEFAULT - }), -); -const CPU_ZSTD_1: Variant = Variant::new( - "zstd-1", - "save-zstd-1", - Settings::repository(RepositorySettings { - compression: zstd_level(1), - ..RepositorySettings::DEFAULT - }), -); -const CPU_ZSTD_9: Variant = Variant::new( - "zstd-9", - "save-zstd-9", - Settings::repository(RepositorySettings { - compression: zstd_level(9), - ..RepositorySettings::DEFAULT - }), -); - -const CPU_OPTIONS_PHASES: &[Phase] = &[ - save_phase("save-default", &CPU_DEFAULT), - save_phase("save-verify-off", &CPU_VERIFY_OFF), - save_phase("save-zstd-off", &CPU_ZSTD_OFF), - save_phase("save-zstd-1", &CPU_ZSTD_1), - save_phase("save-zstd-9", &CPU_ZSTD_9), -]; - -const SCOPES_PHASES: &[Phase] = &[ - with_defaults("save", PhaseKind::Save), - with_defaults("scopes", PhaseKind::Scopes), -]; - -/// The five trees of the base scenario. -const BASE_TREES: &[TreeSpec] = &[FILES_128M, FILES_1G, SQLITE_1G, OBJECTS_128M, OBJECTS_1G]; - -/// The 1 GiB trees of the prune and repository open scenarios. -const HISTORY_TREES: &[TreeSpec] = &[FILES_1G, SQLITE_1G, OBJECTS_1G]; - -/// The scenarios of the benchmark. -const SCENARIOS: &[Scenario] = &[ - Scenario { - name: "base", - trees: &[FILES_128M, FILES_1G, SQLITE_1G, OBJECTS_128M, OBJECTS_1G], - phases: BASE_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "smoke", - trees: &[FILES_TINY, SQLITE_TINY], - phases: BASE_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "restore-threads", - trees: &[FILES_1G], - phases: RESTORE_THREADS_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "memory-pressure", - trees: &[FILES_1G], - phases: BASE_PHASES, - memory_limited_phases: &["restore"], - }, - Scenario { - name: "concurrent-restore-1", - trees: &[FILES_128M, FILES_1G], - phases: CONCURRENT_RESTORE_1_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "concurrent-restore-4", - trees: &[FILES_128M, FILES_1G], - phases: CONCURRENT_RESTORE_4_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "concurrent-restore-20", - trees: &[FILES_128M, FILES_1G], - phases: CONCURRENT_RESTORE_20_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "concurrent-save", - trees: &[FILES_128M, FILES_1G, SQLITE_1G], - phases: CONCURRENT_SAVE_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "capture", - trees: BASE_TREES, - phases: CAPTURE_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "prune", - trees: HISTORY_TREES, - phases: PRUNE_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "save-threads", - trees: &[FILES_1G, SQLITE_1G], - phases: SAVE_THREADS_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "mixed", - trees: &[FILES_1G], - phases: MIXED_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "repository-open", - trees: HISTORY_TREES, - phases: REPOSITORY_OPEN_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "sqlite-changes", - trees: &[SQLITE_1G], - phases: SQLITE_CHANGES_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "tree-shape", - trees: &[FILES_128M, MODULES_128M], - phases: DEFAULT_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "cpu-options", - trees: &[COMPRESSIBLE_1G], - phases: CPU_OPTIONS_PHASES, - memory_limited_phases: &[], - }, - Scenario { - name: "scopes", - trees: BASE_TREES, - phases: SCOPES_PHASES, - memory_limited_phases: &[], - }, -]; - -/// One line of a plan: the phases of one tree of a scenario, in order. -/// -/// `memory_limited_phases` names the phases that run under the lower memory limit of the -/// workflow. A line without such phases does not have the field. -#[derive(Clone, Debug, PartialEq, Eq, Serialize)] -struct PlanEntry { - scenario: &'static str, - tree: &'static str, - phases: Box<[&'static str]>, - #[serde(skip_serializing_if = "<[_]>::is_empty")] - memory_limited_phases: Box<[&'static str]>, -} - -/// Gives the plan of the scenarios with the names, or the first name that no scenario has. -fn plan(names: &[String]) -> Result, String> { - names - .iter() - .try_fold(Vec::new(), |mut entries, name| { - let scenario = scenario(name).ok_or_else(|| name.clone())?; - entries.extend(scenario.trees.iter().map(|tree| PlanEntry { - scenario: scenario.name, - tree: tree.name, - phases: scenario.phases.iter().map(|phase| phase.name).collect(), - memory_limited_phases: scenario.memory_limited_phases.into(), - })); - Ok(entries) - }) - .map(Vec::into_boxed_slice) -} - -fn scenario(name: &str) -> Option<&'static Scenario> { - SCENARIOS.iter().find(|scenario| scenario.name == name) -} - -/// The phase of one pod, found by its names. -#[derive(Clone, Copy)] -struct Selection { - scenario: &'static Scenario, - tree: &'static TreeSpec, - phase: &'static Phase, -} - -impl Selection { - /// Gives the phase with the names, or an error that says which name is unknown. - fn find(scenario: &str, tree: &str, phase: &str) -> Result { - let found = self::scenario(scenario) - .ok_or_else(|| format!("no scenario has the name {scenario:?}"))?; - Ok(Self { - scenario: found, - tree: found - .trees - .iter() - .find(|spec| spec.name == tree) - .ok_or_else(|| format!("the scenario {scenario:?} has no tree {tree:?}"))?, - phase: found - .phases - .iter() - .find(|spec| spec.name == phase) - .ok_or_else(|| format!("the scenario {scenario:?} has no phase {phase:?}"))?, - }) - } -} - -/// Tells whether the text can be a run id or a CPU setting: 1 to 64 ASCII letters, digits, `-` -/// or `_`. Each is a segment of an object key. -fn is_key_segment(text: &str) -> bool { - (1..=64).contains(&text.len()) - && text - .bytes() - .all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'_')) -} - -/// Gives the repository key of the run. The data of a run is synthetic and the workflow deletes -/// it after the run, so each pod of the run derives the same key from the run id. -fn repository_key(run_id: &str) -> RepositoryKey { - let mut bytes = [0; 64]; - blake3::Hasher::new_derive_key("golem fs-snapshot benchmark repository key") - .update(run_id.as_bytes()) - .finalize_xof() - .fill(&mut bytes); - RepositoryKey::new(bytes) -} - -/// Gives the namespace of the results of a run. -fn results_namespace() -> BlobStorageNamespace { - BlobStorageNamespace::InitialAgentFiles { - environment_id: EnvironmentId(Uuid::nil()), - } -} - -/// Gives the path of the result of a phase in the namespace of the results. -fn result_path(scenario: &str, cpu_setting: &str, tree: &str, phase: &str) -> Box { - PathBuf::from(format!( - "results/{scenario}/{cpu_setting}/{tree}/{phase}.json" - )) - .into_boxed_path() -} - -/// Gives the scope of the repository of a scenario, a CPU setting and a tree. -fn repository_scope(scenario: &str, cpu_setting: &str, tree: &str) -> SnapshotScope { - SnapshotScope(BlobStorageNamespace::InitialAgentFiles { - environment_id: EnvironmentId(Uuid::new_v5( - &REPOSITORY_ENVIRONMENTS, - format!("{scenario}/{cpu_setting}/{tree}").as_bytes(), - )), - }) -} - -/// What a phase gets. -struct PhaseContext { - run_id: Box, - cpu_setting: Box, - selection: Selection, - work_dir: Box, - storage: Arc, -} - -impl PhaseContext { - /// Gives the scope of the repositories of the agents of the phase. - fn scope(&self) -> SnapshotScope { - repository_scope( - self.selection.scenario.name, - &self.cpu_setting, - self.selection.tree.name, - ) - } - - fn variant(&self) -> &'static Variant { - self.selection.phase.variant - } - - fn settings(&self) -> Settings { - self.variant().settings.unwrap_or(Settings::DEFAULT) - } - - /// Gives the settings of each save of the phase. - fn save_settings(&self) -> SaveSettings { - self.settings().save - } - - /// Gives the repository of the agent, which a save makes with the settings of the phase. - fn agent_repository(&self, agent: &str) -> Repository { - Repository::new( - Arc::new(AgentStorage::new(self.storage.clone(), agent)), - self.scope(), - repository_key(&self.run_id), - STORAGE_CALL_DEADLINE, - ) - .with_settings(self.settings().repository) - } - - /// Gives the repository of the agent of the variant of the phase. - fn repository(&self) -> Repository { - self.agent_repository(self.variant().agent) - } - - fn result_path(&self, phase: &str) -> Box { - result_path( - self.selection.scenario.name, - &self.cpu_setting, - self.selection.tree.name, - phase, - ) - } -} - -/// What a phase gives. -struct PhaseOutcome { - tree_facts: TreeFacts, - steps: Vec, - outcome: Outcome, -} - -/// Runs one phase on the storage, writes its result into the storage, and gives the result and -/// whether the write succeeded. -/// -/// `environment` is recorded as it is, with the time of a first request of the storage, which -/// gets the credentials and a connection as a running executor already has them. The counters of -/// the cgroup `memory.events` at the start and at the end of the phase go into its `cgroup` -/// object. -async fn run_phase( - run_id: &str, - cpu_setting: &str, - selection: Selection, - work_dir: &Path, - storage: Arc, - environment: Value, -) -> (PhaseResult, anyhow::Result<()>) { - let environment = - with_cgroup_value(environment, "memory_events_start", measure::memory_events()); - let context = PhaseContext { - run_id: run_id.into(), - cpu_setting: cpu_setting.into(), - selection, - work_dir: work_dir.into(), - storage: Arc::new(MeasuredBlobStorage::new(storage)), - }; - let volume = volume::check(work_dir); - let started = Instant::now(); - let warm_up = context - .storage - .get_metadata( - TARGET_LABEL, - "warm_up", - results_namespace(), - Path::new("warm-up"), - ) - .await; - let environment = with_warm_up(environment, millis(started.elapsed()), warm_up.err()); - let outcome = run_kind(&context, selection.phase.kind).await; - let steps = outcome - .steps - .into_iter() - .map(|step| with_settings(step, selection.phase.variant)) - .collect::>(); - let environment = with_cgroup_value(environment, "memory_events_end", measure::memory_events()); - let result = PhaseResult { - format: FORMAT, - run_id: run_id.into(), - scenario: selection.scenario.name, - phase: selection.phase.name, - tree: selection.tree.name, - cpu_setting: cpu_setting.into(), - environment, - volume, - tree_facts: outcome.tree_facts, - steps, - outcome: outcome.outcome, - }; - let written = write_result(&context, &result).await; - (result, written) -} - -/// Runs the work of the kind of phase. -async fn run_kind(context: &PhaseContext, kind: PhaseKind) -> PhaseOutcome { - match kind { - PhaseKind::Save => base_save(context).await, - PhaseKind::Restore { reader_threads } => base_restore(context, reader_threads).await, - PhaseKind::ConcurrentRestore { - agents, - reader_threads, - } => concurrent::concurrent_restore(context, agents, reader_threads).await, - PhaseKind::ConcurrentSave { agents } => { - concurrent::concurrent_save(context, agents, &format!("x{agents}")).await - } - PhaseKind::NamedConcurrentSave { agents } => { - concurrent::concurrent_save(context, agents, context.selection.phase.name).await - } - PhaseKind::Capture => capture::capture(context).await, - PhaseKind::History => history::history(context).await, - PhaseKind::Prune { fast_repack } => history::prune(context, fast_repack).await, - PhaseKind::SqliteChanges => with_open(context, sqlite::sqlite_changes(context).await).await, - PhaseKind::Mixed { - saves, - restores, - reader_threads, - } => concurrent::mixed(context, saves, restores, reader_threads).await, - PhaseKind::Scopes => scopes::scopes(context).await, - PhaseKind::SaveAndOpen => with_open(context, base_save(context).await).await, - } -} - -/// Gives the outcome of a save phase with an open of the repository of the phase after it. The -/// open finds the warm save, and its record gives the settings that the repository has. A save -/// phase that failed gets a skipped open, and an open that fails or finds no repository fails the -/// phase. -async fn with_open(context: &PhaseContext, saved: PhaseOutcome) -> PhaseOutcome { - if saved.outcome != Outcome::Ok { - return PhaseOutcome { - steps: saved - .steps - .into_iter() - .chain(std::iter::once(StepRecord::skipped("open"))) - .collect(), - ..saved - }; - } - let repository = context.repository(); - let (record, inspected) = measure("open", &context.storage, async { - repository.inspect(&snapshot_name(WARM_SAVE)?).await - }) - .await; - let mut steps = saved.steps; - steps.push(inspect_record(record, &inspected)); - if matches!(inspected, Ok(Some(_))) { - PhaseOutcome { steps, ..saved } - } else { - failed(saved.tree_facts, steps, "open", &[]) - } -} - -/// Gives the record of an open with what it found: the number of snapshots, whether a snapshot -/// has the name, and the settings of the repository, as its config file gives them. -fn inspect_record( - record: StepRecord, - inspected: &anyhow::Result>, -) -> StepRecord { - match inspected { - Ok(Some(report)) => record.with_details( - json!({ - "snapshots": report.snapshots, - "found": report.found, - "settings": repository_parameters(&report.settings), - }), - phase_walls(&report.phases), - ), - Ok(None) => record.with_details(json!({ "repository": null }), Box::default()), - Err(_) => record, - } -} - -/// Gives the step with the settings of the variant in its parameters. A parameter that the step -/// already has stays. A variant without settings leaves the step as it is. -fn with_settings(step: StepRecord, variant: &Variant) -> StepRecord { - match (variant.settings, step.parameters.clone()) { - (Some(settings), Value::Object(parameters)) => { - let merged = settings_parameters(&settings) - .into_iter() - .chain(parameters) - .collect::>(); - step.with_parameters(Value::Object(merged)) - } - _ => step, - } -} - -/// Gives the settings as step parameters. A setting that is not set is `null`. -fn settings_parameters(settings: &Settings) -> Map { - [ - ("save_threads", json!(settings.save.threads)), - ( - "change_detection", - json!(change_detection_name(settings.save.detection)), - ), - ] - .into_iter() - .map(|(key, value)| (key.to_string(), value)) - .chain(repository_parameters(&settings.repository)) - .collect() -} - -/// Gives the settings of a repository as step parameters. A setting that is not set is `null`: -/// the default compression is the zstd level that rustic chooses when the config file has none. -fn repository_parameters(repository: &RepositorySettings) -> Map { - [ - ( - "chunker", - match repository.chunking { - Chunking::Rabin => json!("rabin"), - Chunking::Fixed(size) => json!(format!("fixed-{size}")), - }, - ), - ( - "compression", - match repository.compression { - Compression::Default => Value::Null, - Compression::Off => json!("off"), - Compression::Level(level) => json!(level.get()), - }, - ), - ("extra_verify", json!(repository.extra_verify)), - ] - .into_iter() - .map(|(key, value)| (key.to_string(), value)) - .collect() -} - -fn change_detection_name(detection: ChangeDetection) -> &'static str { - match detection { - ChangeDetection::Ctime => "ctime", - ChangeDetection::SizeMtime => "size-mtime", - } -} - -/// Gives the environment with the value at the key in its `cgroup` object. An environment that -/// is an object without a `cgroup` object gets one. -fn with_cgroup_value(environment: Value, key: &str, value: Value) -> Value { - match environment { - Value::Object(mut fields) => { - let cgroup = fields - .entry("cgroup") - .or_insert_with(|| Value::Object(Default::default())); - if let Value::Object(cgroup) = cgroup { - cgroup.insert(key.to_string(), value); - } - Value::Object(fields) - } - other => other, - } -} - -fn with_warm_up(environment: Value, warm_up_ms: f64, error: Option) -> Value { - match environment { - Value::Object(mut fields) => { - fields.insert("warm_up_ms".to_string(), json!(warm_up_ms)); - fields.insert( - "warm_up_error".to_string(), - json!(error.map(|error| format!("{error:#}"))), - ); - Value::Object(fields) - } - other => other, - } -} - -async fn write_result(context: &PhaseContext, result: &PhaseResult) -> anyhow::Result<()> { - let json = serde_json::to_vec(result)?; - context - .storage - .put_raw( - TARGET_LABEL, - "result", - results_namespace(), - &context.result_path(context.selection.phase.name), - &json, - ) - .await -} - -/// Gives the times of the parts of an operation as the phases of a step. -fn phase_walls(phases: &[PhaseTime]) -> Box<[PhaseWall]> { - phases - .iter() - .map(|time| PhaseWall { - name: phase_name(time.phase), - wall_ms: millis(time.wall), - }) - .collect() -} - -fn phase_name(phase: super::rustic::OperationPhase) -> &'static str { - use super::rustic::OperationPhase; - match phase { - OperationPhase::Create => "create", - OperationPhase::Open => "open", - OperationPhase::Lookup => "lookup", - OperationPhase::IndexLoad => "index_load", - OperationPhase::Backup => "backup", - OperationPhase::RestorePlan => "restore_plan", - OperationPhase::Restore => "restore", - OperationPhase::PrunePlan => "prune_plan", - OperationPhase::Prune => "prune", - } -} - -fn snapshot_name(text: &str) -> anyhow::Result { - Ok(SnapshotName::new(text)?) -} - -/// Gives the outcome of a phase whose step failed: the records so far, and a skipped record for -/// each step that did not run. -fn failed( - tree_facts: TreeFacts, - steps: Vec, - failed_step: &'static str, - skipped: &[&'static str], -) -> PhaseOutcome { - PhaseOutcome { - tree_facts, - steps: steps - .into_iter() - .chain(skipped.iter().copied().map(StepRecord::skipped)) - .collect(), - outcome: Outcome::Failed { - reason: format!("the step {failed_step} failed").into(), - }, - } -} - -/// The save phase of the base scenario: a cold save of a new tree, a small change of the tree, a -/// warm save, and the hash of the changed tree. -/// -/// `generate_tree` and `small_change` remove the pages of the tree from the page cache. No step -/// reads the content of a file between one of them and the save after it, so each save reads the -/// tree from the volume. The hash of the new tree is read after the cold save, and the hash of the -/// changed tree after the warm save. A save does not change the tree. -async fn base_save(context: &PhaseContext) -> PhaseOutcome { - let spec = context.selection.tree; - let tree = context.work_dir.join("tree"); - let repository = context.repository(); - let facts = TreeFacts { - name: spec.name, - content: Some(spec.content.label()), - page_cache: Some("dropped"), - ..TreeFacts::default() - }; - let storage = &context.storage; - let settings = context.save_settings(); - - let (record, generated) = measure("generate_tree", storage, trees::generate(spec, &tree)).await; - let mut steps = vec![record]; - let Ok(counts) = generated else { - return failed( - facts, - steps, - "generate_tree", - &["cold_save", "small_change", "warm_save", "hash_tree"], - ); - }; - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - ..facts - }; - - let (record, cold) = measure("cold_save", storage, async { - repository - .save_with(&snapshot_name(COLD_SAVE)?, &tree, settings) - .await - }) - .await; - steps.push(save_record(record, &cold)); - if cold.is_err() { - return failed( - facts, - steps, - "cold_save", - &["small_change", "warm_save", "hash_tree"], - ); - } - let facts = TreeFacts { - hash: trees::hash(&tree).await.ok().map(|(hash, _)| hash), - ..facts - }; - - let (record, changed) = measure("small_change", storage, trees::change(spec, &tree)).await; - steps.push(record.with_details( - changed.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )); - let Ok(change) = changed else { - return failed(facts, steps, "small_change", &["warm_save", "hash_tree"]); - }; - let facts = TreeFacts { - change: Some(change), - ..facts - }; - - let (record, warm) = measure("warm_save", storage, async { - repository - .save_with(&snapshot_name(WARM_SAVE)?, &tree, settings) - .await - }) - .await; - steps.push(save_record(record, &warm)); - if warm.is_err() { - return failed(facts, steps, "warm_save", &["hash_tree"]); - } - - let (record, hashed) = measure("hash_tree", storage, trees::hash(&tree)).await; - steps.push(record); - let Ok((hash, _)) = hashed else { - return failed(facts, steps, "hash_tree", &[]); - }; - PhaseOutcome { - tree_facts: TreeFacts { - hash_after_change: Some(hash), - ..facts - }, - steps, - outcome: Outcome::Ok, - } -} - -fn save_record(record: StepRecord, save: &anyhow::Result) -> StepRecord { - match save { - Ok(report) => record.with_details( - json!({ - "snapshot": report.snapshot, - "parent": report.parent, - "files_new": report.files_new, - "files_changed": report.files_changed, - "files_unmodified": report.files_unmodified, - "dirs_new": report.dirs_new, - "dirs_changed": report.dirs_changed, - "dirs_unmodified": report.dirs_unmodified, - "bytes_processed": report.bytes_processed, - "data_added": report.data_added, - "data_added_packed": report.data_added_packed, - "data_blobs": report.data_blobs, - "tree_blobs": report.tree_blobs, - }), - phase_walls(&report.phases), - ), - Err(_) => record, - } -} - -/// The restore phase of the base scenario: a cold restore of the warm save into an empty -/// directory, and a comparison of the hash of the restored tree with the hash that the save -/// phase recorded. -/// -/// `reader_threads` is the number of threads of the restore that read data, and the -/// `parameters` of the restore step record it. `None` is the default of rustic. -async fn base_restore( - context: &PhaseContext, - reader_threads: Option, -) -> PhaseOutcome { - let spec = context.selection.tree; - let into = context.work_dir.join("restore"); - let repository = context.repository(); - let storage = &context.storage; - let facts = TreeFacts { - name: spec.name, - ..TreeFacts::default() - }; - let expected = match saved_hash(context).await { - Ok(expected) => expected, - Err(error) => return without_save(facts, &error, &["cold_restore", "hash_tree"]), - }; - - let (record, restored) = measure("cold_restore", storage, async { - std::fs::create_dir(&into)?; - repository - .restore(&snapshot_name(WARM_SAVE)?, &into, reader_threads) - .await? - .ok_or_else(|| anyhow::anyhow!("no snapshot has the name {WARM_SAVE}")) - }) - .await; - let record = record.with_parameters(json!({ "reader_threads": reader_threads })); - let record = match &restored { - Ok(report) => record.with_details( - json!({ "files": report.files, "dirs": report.dirs, "bytes": report.bytes }), - phase_walls(&report.phases), - ), - Err(_) => record, - }; - let mut steps = vec![record]; - if restored.is_err() { - return failed(facts, steps, "cold_restore", &["hash_tree"]); - } - - let (record, hashed) = measure("hash_tree", storage, trees::hash(&into)).await; - let Ok((hash, counts)) = hashed else { - steps.push(record); - return failed(facts, steps, "hash_tree", &[]); - }; - let matches = *hash == *expected; - steps.push(record.with_details( - json!({ "hash": hash, "expected": expected, "matches": matches }), - Box::default(), - )); - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - hash: Some(hash), - ..facts - }; - PhaseOutcome { - tree_facts: facts, - steps, - outcome: if matches { - Outcome::Ok - } else { - Outcome::Failed { - reason: "the restored tree differs from the saved tree".into(), - } - }, - } -} - -/// Gives the outcome of a restore phase without a result of the save phase: a skipped record for -/// each step. -fn without_save( - tree_facts: TreeFacts, - error: &anyhow::Error, - skipped: &[&'static str], -) -> PhaseOutcome { - PhaseOutcome { - tree_facts, - steps: skipped.iter().copied().map(StepRecord::skipped).collect(), - outcome: Outcome::Failed { - reason: format!("the save phase gave no result to compare with: {error:#}").into(), - }, - } -} - -/// Reads the hash after the change from the result of the save phase of the variant of the -/// phase. -async fn saved_hash(context: &PhaseContext) -> anyhow::Result> { - let saved = context - .storage - .get_raw( - TARGET_LABEL, - "result", - results_namespace(), - &context.result_path(context.variant().save_phase), - ) - .await? - .ok_or_else(|| anyhow::anyhow!("the save phase wrote no result"))?; - let saved: Value = serde_json::from_slice(&saved)?; - if saved.pointer("/outcome/status") != Some(&json!("ok")) { - anyhow::bail!("the save phase failed"); - } - saved - .pointer("/tree_facts/hash_after_change") - .and_then(Value::as_str) - .map(Box::from) - .ok_or_else(|| anyhow::anyhow!("the result of the save phase has no hash")) -} - -#[cfg(test)] -mod tests; diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/report.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/report.rs deleted file mode 100644 index 7bdeccbeab..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/report.rs +++ /dev/null @@ -1,301 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The result of one phase of the benchmark, as JSON. -//! -//! A new scenario adds steps with new names, parameters and details. It does not add or change a -//! field of these types. - -use serde::Serialize; -use serde_json::Value; -use std::time::Duration; - -/// The version of the result format. -pub(super) const FORMAT: &str = "golem-fs-snapshot-benchmark/1"; - -/// The result of one phase: the work of one pod. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct PhaseResult { - pub(super) format: &'static str, - pub(super) run_id: Box, - pub(super) scenario: &'static str, - pub(super) phase: &'static str, - pub(super) tree: &'static str, - pub(super) cpu_setting: Box, - pub(super) environment: Value, - pub(super) volume: Value, - pub(super) tree_facts: TreeFacts, - pub(super) steps: Box<[StepRecord]>, - pub(super) outcome: Outcome, -} - -/// What a phase knows about its tree. A field that the phase does not know is `null`. -#[derive(Clone, Debug, Default, PartialEq, Serialize)] -pub(super) struct TreeFacts { - pub(super) name: &'static str, - pub(super) files: Option, - pub(super) directories: Option, - pub(super) bytes: Option, - pub(super) content: Option<&'static str>, - pub(super) page_cache: Option<&'static str>, - pub(super) hash: Option>, - pub(super) hash_after_change: Option>, - pub(super) change: Option, -} - -/// The outcome of a phase. -#[derive(Clone, Debug, PartialEq, Serialize)] -#[serde(tag = "status", rename_all = "snake_case")] -pub(super) enum Outcome { - Ok, - Failed { reason: Box }, -} - -/// One measured step of a phase. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct StepRecord { - pub(super) name: &'static str, - pub(super) status: StepStatus, - pub(super) parameters: Value, - pub(super) wall_ms: Option, - pub(super) cpu: Option, - pub(super) memory: Option, - pub(super) threads: Option, - pub(super) requests: Box<[RequestSummary]>, - pub(super) bytes_written: u64, - pub(super) bytes_read: u64, - pub(super) phases: Box<[PhaseWall]>, - pub(super) details: Value, -} - -impl StepRecord { - /// Gives the record of a step that did not run, because an earlier step failed. - pub(super) fn skipped(name: &'static str) -> Self { - Self { - name, - status: StepStatus::Skipped, - parameters: Value::Object(Default::default()), - wall_ms: None, - cpu: None, - memory: None, - threads: None, - requests: Box::default(), - bytes_written: 0, - bytes_read: 0, - phases: Box::default(), - details: Value::Object(Default::default()), - } - } - - /// Gives the record with the parameters of the step. - pub(super) fn with_parameters(self, parameters: Value) -> Self { - Self { parameters, ..self } - } - - /// Gives the record with the details and the phases of the operation. - pub(super) fn with_details(self, details: Value, phases: Box<[PhaseWall]>) -> Self { - Self { - details, - phases, - ..self - } - } -} - -/// Whether a step succeeded. -#[derive(Clone, Debug, PartialEq, Serialize)] -#[serde(rename_all = "snake_case")] -pub(super) enum StepStatus { - Ok, - Error(Box), - Skipped, -} - -/// The CPU time of a step. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct CpuTime { - pub(super) user_ms: f64, - pub(super) system_ms: f64, - pub(super) cgroup: Option, -} - -/// The change of the CPU counters of the cgroup of the process during a step. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct CgroupCpuTime { - pub(super) usage_ms: f64, - pub(super) user_ms: f64, - pub(super) system_ms: f64, - pub(super) nr_periods: u64, - pub(super) nr_throttled: u64, - pub(super) throttled_ms: f64, -} - -/// The memory of a step. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct MemoryPeaks { - pub(super) rss_start_bytes: u64, - pub(super) rss_peak_bytes: u64, - pub(super) rss_peak_source: &'static str, - pub(super) cgroup_current_peak_bytes: Option, - pub(super) cgroup_anon_peak_bytes: Option, - pub(super) cgroup_file_peak_bytes: Option, -} - -/// The threads of the process during a step. The count includes the thread that samples it. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct ThreadCounts { - pub(super) start: u64, - pub(super) peak: u64, - pub(super) sample_interval_ms: u64, -} - -/// The blob storage requests of one call and one file type in a step. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct RequestSummary { - pub(super) call: &'static str, - pub(super) file_type: &'static str, - pub(super) count: u64, - pub(super) errors: u64, - pub(super) bytes: u64, - pub(super) time_ms: RequestTimes, -} - -/// The request times of one call and one file type, by the nearest rank. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct RequestTimes { - pub(super) min: f64, - pub(super) p50: f64, - pub(super) p90: f64, - pub(super) p99: f64, - pub(super) max: f64, - pub(super) total: f64, -} - -/// The time of one part of an operation. -#[derive(Clone, Debug, PartialEq, Serialize)] -pub(super) struct PhaseWall { - pub(super) name: &'static str, - pub(super) wall_ms: f64, -} - -/// Gives the duration in milliseconds, with the precision of a microsecond. -pub(super) fn millis(duration: Duration) -> f64 { - duration.as_micros() as f64 / 1_000.0 -} - -#[cfg(test)] -mod tests { - use super::{ - CgroupCpuTime, CpuTime, FORMAT, MemoryPeaks, Outcome, PhaseResult, PhaseWall, - RequestSummary, RequestTimes, StepRecord, StepStatus, ThreadCounts, TreeFacts, millis, - }; - use pretty_assertions::assert_eq; - use serde_json::json; - use std::time::Duration; - use test_r::test; - - fn result() -> PhaseResult { - let step = StepRecord { - name: "cold_save", - status: StepStatus::Ok, - parameters: json!({}), - wall_ms: Some(1500.25), - cpu: Some(CpuTime { - user_ms: 900.5, - system_ms: 100.0, - cgroup: Some(CgroupCpuTime { - usage_ms: 1000.0, - user_ms: 900.0, - system_ms: 100.0, - nr_periods: 15, - nr_throttled: 2, - throttled_ms: 30.5, - }), - }), - memory: Some(MemoryPeaks { - rss_start_bytes: 1000, - rss_peak_bytes: 5000, - rss_peak_source: "VmHWM", - cgroup_current_peak_bytes: Some(9000), - cgroup_anon_peak_bytes: Some(6000), - cgroup_file_peak_bytes: None, - }), - threads: Some(ThreadCounts { - start: 10, - peak: 40, - sample_interval_ms: 10, - }), - requests: Box::new([RequestSummary { - call: "put_raw", - file_type: "pack", - count: 3, - errors: 0, - bytes: 3000, - time_ms: RequestTimes { - min: 1.0, - p50: 2.0, - p90: 3.0, - p99: 3.0, - max: 3.0, - total: 6.0, - }, - }]), - bytes_written: 3000, - bytes_read: 0, - phases: Box::new([PhaseWall { - name: "backup", - wall_ms: 1400.0, - }]), - details: json!({ "snapshot": "abc" }), - }; - PhaseResult { - format: FORMAT, - run_id: "123-1".into(), - scenario: "base", - phase: "save", - tree: "files-128m", - cpu_setting: "limit-3".into(), - environment: json!({ "pod": "p" }), - volume: json!({ "check": { "status": "ok" } }), - tree_facts: TreeFacts { - name: "files-128m", - files: Some(10_000), - directories: Some(100), - bytes: Some(134_217_728), - content: Some("incompressible"), - page_cache: Some("dropped"), - hash: Some("h1".into()), - hash_after_change: Some("h2".into()), - change: Some(json!({ "files_rewritten": 10 })), - }, - steps: Box::new([step, StepRecord::skipped("warm_save")]), - outcome: Outcome::Failed { - reason: "the step warm_save failed".into(), - }, - } - } - - #[test] - fn the_result_format_is_stable() { - assert_eq!( - serde_json::to_string_pretty(&result()).unwrap(), - include_str!("golden/phase_result.json").trim_end() - ); - } - - #[test] - fn a_duration_is_given_in_milliseconds_with_microseconds() { - assert_eq!(millis(Duration::from_nanos(1_234_567_890)), 1234.567); - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/requests.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/requests.rs deleted file mode 100644 index 9101caedcf..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/requests.rs +++ /dev/null @@ -1,659 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! A blob storage that records each request that it passes to another blob storage. - -use super::agents::AGENTS; -use super::report::{RequestSummary, RequestTimes, millis}; -use async_trait::async_trait; -use bytes::Bytes; -use futures::stream::BoxStream; -use golem_service_base::replayable_stream::ErasedReplayableStream; -use golem_service_base::storage::blob::{ - BlobMetadata, BlobStorage, BlobStorageNamespace, ExistsResult, ListedBlob, PutIfAbsent, -}; -use std::future::Future; -use std::path::{Component, Path, PathBuf}; -use std::sync::{Arc, Mutex, PoisonError}; -use std::time::{Duration, Instant}; - -/// One request that the storage passed on. -#[derive(Clone, Debug, PartialEq)] -pub(super) struct RequestRecord { - pub(super) call: &'static str, - pub(super) file_type: &'static str, - /// The bytes that the request wrote or read. - pub(super) bytes: u64, - pub(super) written: bool, - pub(super) time: Duration, - pub(super) ok: bool, -} - -/// A blob storage that passes each call to `inner` and records it. -/// -/// A record holds the name of the call, the file type of the path, the bytes that the call wrote -/// or read, and the time from the start of the call to its end. The time includes the retries of -/// the inner storage. Each method of the trait is passed on, also a method with a default, so the -/// inner storage does each call in its own way. -#[derive(Debug)] -pub(super) struct MeasuredBlobStorage { - inner: Arc, - records: Mutex>, -} - -impl MeasuredBlobStorage { - pub(super) fn new(inner: Arc) -> Self { - Self { - inner, - records: Mutex::new(Vec::new()), - } - } - - /// Gives the records since the last call, and removes them. - pub(super) fn take(&self) -> Box<[RequestRecord]> { - std::mem::take(&mut *self.records.lock().unwrap_or_else(PoisonError::into_inner)) - .into_boxed_slice() - } - - async fn record( - &self, - call: &'static str, - path: &Path, - bytes: impl FnOnce(&T) -> (u64, bool), - future: impl Future>, - ) -> anyhow::Result { - let started = Instant::now(); - let result = future.await; - let time = started.elapsed(); - let (bytes, written) = result.as_ref().map_or((0, false), bytes); - self.records - .lock() - .unwrap_or_else(PoisonError::into_inner) - .push(RequestRecord { - call, - file_type: file_type(path), - bytes, - written, - time, - ok: result.is_ok(), - }); - result - } -} - -/// Gives the file type of a repository path from its first name: `config`, `pack`, `index`, -/// `snapshot`, `key`, or `other`. The path of the repository of an agent starts with -/// `agents/` (see [`super::agents`]), and the first name after it gives the type. -pub(super) fn file_type(path: &Path) -> &'static str { - let names = path - .components() - .filter_map(|component| match component { - Component::Normal(name) => Some(name.to_str()), - _ => None, - }) - .collect::>(); - let in_repository = match &*names { - [Some(AGENTS), Some(_), rest @ ..] => rest, - all => all, - }; - match in_repository.first() { - Some(Some("config")) => "config", - Some(Some("data")) => "pack", - Some(Some("index")) => "index", - Some(Some("snapshots")) => "snapshot", - Some(Some("keys")) => "key", - _ => "other", - } -} - -fn nothing(_: &T) -> (u64, bool) { - (0, false) -} - -fn read_bytes(data: &Option>) -> (u64, bool) { - (data.as_ref().map_or(0, |data| data.len() as u64), false) -} - -/// Gives a summary of the records for each call and file type, in the order of the call and the -/// file type. -pub(super) fn summarize(records: &[RequestRecord]) -> Box<[RequestSummary]> { - let mut sorted = records.to_vec(); - sorted.sort_by(|left, right| { - (left.call, left.file_type, left.time).cmp(&(right.call, right.file_type, right.time)) - }); - sorted - .chunk_by(|left, right| (left.call, left.file_type) == (right.call, right.file_type)) - .filter_map(|group| { - group.first().map(|first| RequestSummary { - call: first.call, - file_type: first.file_type, - count: group.len() as u64, - errors: group.iter().filter(|record| !record.ok).count() as u64, - bytes: group.iter().map(|record| record.bytes).sum(), - time_ms: times(&group.iter().map(|record| record.time).collect::>()), - }) - }) - .collect() -} - -/// Gives the minimum, the percentiles by the nearest rank, the maximum and the total of times -/// that are in ascending order. Each value of an empty list is zero. -pub(super) fn times(sorted: &[Duration]) -> RequestTimes { - let rank = |percent: usize| { - let index = (percent * sorted.len()).div_ceil(100).max(1) - 1; - sorted.get(index).copied().map(millis).unwrap_or_default() - }; - RequestTimes { - min: sorted.first().copied().map(millis).unwrap_or_default(), - p50: rank(50), - p90: rank(90), - p99: rank(99), - max: sorted.last().copied().map(millis).unwrap_or_default(), - total: millis(sorted.iter().sum()), - } -} - -/// Gives the bytes that the records wrote and the bytes that they read. -pub(super) fn written_and_read(records: &[RequestRecord]) -> (u64, u64) { - records.iter().fold((0, 0), |(written, read), record| { - if record.written { - (written + record.bytes, read) - } else { - (written, read + record.bytes) - } - }) -} - -#[async_trait] -impl BlobStorage for MeasuredBlobStorage { - async fn get_raw( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result>> { - self.record( - "get_raw", - path, - read_bytes, - self.inner.get_raw(target_label, op_label, namespace, path), - ) - .await - } - - async fn get_stream( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result>>> { - self.record( - "get_stream", - path, - nothing, - self.inner - .get_stream(target_label, op_label, namespace, path), - ) - .await - } - - async fn get_raw_slice( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - start: u64, - end: u64, - ) -> anyhow::Result>> { - self.record( - "get_raw_slice", - path, - read_bytes, - self.inner - .get_raw_slice(target_label, op_label, namespace, path, start, end), - ) - .await - } - - async fn get_metadata( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.record( - "get_metadata", - path, - nothing, - self.inner - .get_metadata(target_label, op_label, namespace, path), - ) - .await - } - - async fn put_raw( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - data: &[u8], - ) -> anyhow::Result<()> { - let length = data.len() as u64; - self.record( - "put_raw", - path, - |_| (length, true), - self.inner - .put_raw(target_label, op_label, namespace, path, data), - ) - .await - } - - async fn put_raw_if_absent( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - data: &[u8], - ) -> anyhow::Result { - let length = data.len() as u64; - self.record( - "put_raw_if_absent", - path, - |outcome| match outcome { - PutIfAbsent::Written => (length, true), - PutIfAbsent::AlreadyExists => (0, true), - }, - self.inner - .put_raw_if_absent(target_label, op_label, namespace, path, data), - ) - .await - } - - async fn put_stream( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - stream: &dyn ErasedReplayableStream>, Error = anyhow::Error>, - ) -> anyhow::Result<()> { - self.record( - "put_stream", - path, - nothing, - self.inner - .put_stream(target_label, op_label, namespace, path, stream), - ) - .await - } - - async fn delete( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result<()> { - self.record( - "delete", - path, - nothing, - self.inner.delete(target_label, op_label, namespace, path), - ) - .await - } - - async fn delete_many( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - paths: &[PathBuf], - ) -> anyhow::Result<()> { - self.record( - "delete_many", - paths.first().map_or(Path::new(""), PathBuf::as_path), - nothing, - self.inner - .delete_many(target_label, op_label, namespace, paths), - ) - .await - } - - async fn create_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result<()> { - self.record( - "create_dir", - path, - nothing, - self.inner - .create_dir(target_label, op_label, namespace, path), - ) - .await - } - - async fn list_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.record( - "list_dir", - path, - nothing, - self.inner.list_dir(target_label, op_label, namespace, path), - ) - .await - } - - async fn list_blobs_below( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.record( - "list_blobs_below", - path, - nothing, - self.inner - .list_blobs_below(target_label, op_label, namespace, path), - ) - .await - } - - async fn delete_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result { - self.record( - "delete_dir", - path, - nothing, - self.inner - .delete_dir(target_label, op_label, namespace, path), - ) - .await - } - - async fn exists( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result { - self.record( - "exists", - path, - nothing, - self.inner.exists(target_label, op_label, namespace, path), - ) - .await - } - - async fn copy( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - from: &Path, - to: &Path, - ) -> anyhow::Result<()> { - self.record( - "copy", - from, - nothing, - self.inner.copy(target_label, op_label, namespace, from, to), - ) - .await - } - - async fn r#move( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - from: &Path, - to: &Path, - ) -> anyhow::Result<()> { - self.record( - "move", - from, - nothing, - self.inner - .r#move(target_label, op_label, namespace, from, to), - ) - .await - } -} - -#[cfg(test)] -mod tests { - use super::{ - MeasuredBlobStorage, RequestRecord, file_type, summarize, times, written_and_read, - }; - use crate::filesystem_snapshot::benchmark::report::RequestTimes; - use golem_common::model::environment::EnvironmentId; - use golem_service_base::storage::blob::memory::InMemoryBlobStorage; - use golem_service_base::storage::blob::{BlobStorage, BlobStorageNamespace}; - use pretty_assertions::assert_eq; - use std::path::Path; - use std::sync::Arc; - use std::time::Duration; - use test_r::test; - use uuid::Uuid; - - fn record(call: &'static str, file_type: &'static str, millis: u64) -> RequestRecord { - RequestRecord { - call, - file_type, - bytes: 10, - written: call == "put_raw", - time: Duration::from_millis(millis), - ok: millis != 0, - } - } - - #[test] - fn the_times_are_the_nearest_ranks_of_the_sorted_times() { - let hundred = (1..=100).map(Duration::from_millis).collect::>(); - let three = [1, 2, 3].map(Duration::from_millis); - - assert_eq!( - ( - times(&hundred), - times(&three), - times(&[Duration::from_millis(7)]) - ), - ( - RequestTimes { - min: 1.0, - p50: 50.0, - p90: 90.0, - p99: 99.0, - max: 100.0, - total: 5050.0, - }, - RequestTimes { - min: 1.0, - p50: 2.0, - p90: 3.0, - p99: 3.0, - max: 3.0, - total: 6.0, - }, - RequestTimes { - min: 7.0, - p50: 7.0, - p90: 7.0, - p99: 7.0, - max: 7.0, - total: 7.0, - } - ) - ); - } - - #[test] - fn a_summary_groups_the_records_by_call_and_file_type() { - let records = [ - record("put_raw", "pack", 3), - record("get_raw_slice", "pack", 5), - record("put_raw", "pack", 1), - record("put_raw", "index", 2), - record("put_raw", "pack", 0), - ]; - - let summary = summarize(&records); - - assert_eq!( - ( - summary - .iter() - .map(|summary| ( - summary.call, - summary.file_type, - summary.count, - summary.errors, - summary.bytes, - summary.time_ms.max - )) - .collect::>(), - written_and_read(&records) - ), - ( - vec![ - ("get_raw_slice", "pack", 1, 0, 10, 5.0), - ("put_raw", "index", 1, 0, 10, 2.0), - ("put_raw", "pack", 3, 1, 30, 3.0), - ], - (40, 10) - ) - ); - } - - #[test] - fn the_file_type_is_the_first_name_of_the_repository_path() { - assert_eq!( - [ - "config", - "data/ab/abcd", - "index/ab", - "snapshots/ab", - "keys/ab", - "results/base/save.json", - "", - "./data/ab/abcd", - "agents/0/config", - "agents/x8-7/data/ab/abcd", - "agents/0", - "agents" - ] - .map(|path| file_type(Path::new(path))), - [ - "config", "pack", "index", "snapshot", "key", "other", "other", "pack", "config", - "pack", "other", "other" - ] - ); - } - - #[test] - async fn the_storage_records_each_call_and_passes_it_on() { - let inner = Arc::new(InMemoryBlobStorage::new()); - let storage = MeasuredBlobStorage::new(inner.clone()); - let namespace = BlobStorageNamespace::InitialAgentFiles { - environment_id: EnvironmentId(Uuid::new_v4()), - }; - let pack = Path::new("data/ab/abcd"); - - storage - .put_raw("test", "test", namespace.clone(), pack, b"0123456789") - .await - .unwrap(); - let before = storage.take(); - let slice = storage - .get_raw_slice("test", "test", namespace.clone(), pack, 2, 4) - .await - .unwrap(); - let missing = storage - .get_raw("test", "test", namespace.clone(), Path::new("index/ef")) - .await - .unwrap(); - let after = storage.take(); - let empty = storage.take(); - let stored = inner - .get_raw("test", "test", namespace, pack) - .await - .unwrap(); - - assert_eq!( - ( - before - .iter() - .map(|record| ( - record.call, - record.file_type, - record.bytes, - record.written, - record.ok - )) - .collect::>(), - after - .iter() - .map(|record| ( - record.call, - record.file_type, - record.bytes, - record.written, - record.ok - )) - .collect::>(), - empty.len(), - slice, - missing, - stored, - ), - ( - vec![("put_raw", "pack", 10, true, true)], - vec![ - ("get_raw_slice", "pack", 3, false, true), - ("get_raw", "index", 0, false, true) - ], - 0, - Some(b"234".to_vec()), - None, - Some(b"0123456789".to_vec()), - ) - ); - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/scopes.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/scopes.rs deleted file mode 100644 index 8b450f00da..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/scopes.rs +++ /dev/null @@ -1,97 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The phase of the scopes scenario: the copy of the repository of an agent for a fork, and the -//! deletion of the repository of an agent. - -use super::agents::{FIRST_AGENT, agent_blobs, copy_agent, delete_agent}; -use super::measure::measure; -use super::report::{Outcome, TreeFacts}; -use super::{PhaseContext, PhaseOutcome, failed}; -use serde_json::{Value, json}; - -/// The agent that gets the copy of the repository of the first agent. -const FORK: &str = "fork"; - -/// The scopes phase: the repository of the save phase goes on the server to the fork agent, as -/// `copy_scope` copies a scope, and then the repository of the fork agent is deleted, as -/// `delete_scope` deletes a scope. -/// -/// A listing after the copy checks that the fork has each blob of the source with its size, and a -/// listing after the delete checks that the fork has no blob. Each listing is outside the measured -/// steps. -pub(super) async fn scopes(context: &PhaseContext) -> PhaseOutcome { - let storage = &context.storage; - let namespace = context.scope().0; - let facts = TreeFacts { - name: context.selection.tree.name, - ..TreeFacts::default() - }; - - let (record, copied) = measure( - "copy_scope", - storage, - copy_agent(storage.as_ref(), &namespace, FIRST_AGENT, FORK), - ) - .await; - let (source, fork) = ( - agent_blobs(storage.as_ref(), &namespace, FIRST_AGENT).await, - agent_blobs(storage.as_ref(), &namespace, FORK).await, - ); - let same = - matches!((&source, &fork), (Ok(source), Ok(fork)) if !source.is_empty() && source == fork); - let details = match copied.as_ref() { - Ok(Value::Object(details)) => { - let mut details = details.clone(); - details.insert("same_as_source".to_string(), json!(same)); - Value::Object(details) - } - _ => Value::Null, - }; - let mut steps = vec![record.with_details(details, Box::default())]; - if copied.is_err() { - return failed(facts, steps, "copy_scope", &["delete_scope"]); - } - - let (record, deleted) = measure( - "delete_scope", - storage, - delete_agent(storage.as_ref(), &namespace, FORK), - ) - .await; - let left = agent_blobs(storage.as_ref(), &namespace, FORK) - .await - .map(|blobs| blobs.len()); - steps.push(record.with_details( - json!({ "deleted": deleted.as_ref().ok(), "blobs_left": left.as_ref().ok() }), - Box::default(), - )); - if deleted.is_err() { - return failed(facts, steps, "delete_scope", &[]); - } - let outcome = match (same, left) { - (true, Ok(0)) => Outcome::Ok, - (false, _) => Outcome::Failed { - reason: "the copy of the repository differs from the repository".into(), - }, - (true, _) => Outcome::Failed { - reason: "the deleted repository still has blobs".into(), - }, - }; - PhaseOutcome { - tree_facts: facts, - steps, - outcome, - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/sqlite.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/sqlite.rs deleted file mode 100644 index 63ec9e9602..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/sqlite.rs +++ /dev/null @@ -1,141 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The save phase of the SQLite changes scenario. - -use super::measure::measure; -use super::report::{Outcome, TreeFacts}; -use super::{ - COLD_SAVE, PhaseContext, PhaseOutcome, WARM_SAVE, failed, save_record, snapshot_name, trees, -}; -use serde_json::{Value, json}; - -/// The name of the snapshot after the clustered change. -const CLUSTERED_SAVE: &str = "warm-save-clustered"; - -/// The save phase of a SQLite tree into the repository of the variant of the phase: a cold save, -/// an update of 100 consecutive rows and a warm save, then an update of 100 rows spread over the -/// database and a second warm save, and the hash of the tree. -/// -/// The chunker of the repository decides how many bytes each warm save adds. The second warm save -/// has the name of the warm save of the base scenario, so the restore phase of the base scenario -/// restores it. -pub(super) async fn sqlite_changes(context: &PhaseContext) -> PhaseOutcome { - let spec = context.selection.tree; - let tree = context.work_dir.join("tree"); - let repository = context.repository(); - let storage = &context.storage; - let settings = context.save_settings(); - let facts = TreeFacts { - name: spec.name, - content: Some(spec.content.label()), - page_cache: Some("dropped"), - ..TreeFacts::default() - }; - let later = [ - "cold_save", - "clustered_change", - "warm_save_clustered", - "scattered_change", - "warm_save_scattered", - "hash_tree", - ]; - - let (record, generated) = measure("generate_tree", storage, trees::generate(spec, &tree)).await; - let mut steps = vec![record]; - let Ok(counts) = generated else { - return failed(facts, steps, "generate_tree", &later); - }; - let facts = TreeFacts { - files: Some(counts.files), - directories: Some(counts.directories), - bytes: Some(counts.bytes), - ..facts - }; - - let (record, cold) = measure("cold_save", storage, async { - repository - .save_with(&snapshot_name(COLD_SAVE)?, &tree, settings) - .await - }) - .await; - steps.push(save_record(record, &cold)); - if cold.is_err() { - return failed(facts, steps, "cold_save", &later[1..]); - } - - let (record, clustered) = measure( - "clustered_change", - storage, - trees::change_clustered(spec, &tree), - ) - .await; - steps.push(record.with_details( - clustered.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )); - if clustered.is_err() { - return failed(facts, steps, "clustered_change", &later[2..]); - } - - let (record, warm) = measure("warm_save_clustered", storage, async { - repository - .save_with(&snapshot_name(CLUSTERED_SAVE)?, &tree, settings) - .await - }) - .await; - steps.push(save_record(record, &warm).with_parameters(json!({ "change": "clustered" }))); - if warm.is_err() { - return failed(facts, steps, "warm_save_clustered", &later[3..]); - } - - let (record, scattered) = - measure("scattered_change", storage, trees::change(spec, &tree)).await; - steps.push(record.with_details( - scattered.as_ref().ok().cloned().unwrap_or(Value::Null), - Box::default(), - )); - let Ok(change) = scattered else { - return failed(facts, steps, "scattered_change", &later[4..]); - }; - let facts = TreeFacts { - change: Some(change), - ..facts - }; - - let (record, warm) = measure("warm_save_scattered", storage, async { - repository - .save_with(&snapshot_name(WARM_SAVE)?, &tree, settings) - .await - }) - .await; - steps.push(save_record(record, &warm).with_parameters(json!({ "change": "scattered" }))); - if warm.is_err() { - return failed(facts, steps, "warm_save_scattered", &later[5..]); - } - - let (record, hashed) = measure("hash_tree", storage, trees::hash(&tree)).await; - steps.push(record); - let Ok((hash, _)) = hashed else { - return failed(facts, steps, "hash_tree", &[]); - }; - PhaseOutcome { - tree_facts: TreeFacts { - hash_after_change: Some(hash), - ..facts - }, - steps, - outcome: Outcome::Ok, - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/tests.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/tests.rs deleted file mode 100644 index 60d238cae2..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/tests.rs +++ /dev/null @@ -1,1736 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -use super::agents::AgentStorage; -use super::report::{Outcome, PhaseResult, StepRecord, StepStatus}; -use super::requests::MeasuredBlobStorage; -use super::trees::{Content, FILES_TINY, SQLITE_TINY, TreeShape, TreeSpec, tree_hash}; -use super::{ - BASE, CPU_DEFAULT, CPU_OPTIONS_PHASES, CPU_ZSTD_OFF, Compression, DEFAULTS, HISTORY, PRUNE, - PRUNE_FAST_REPACK, Phase, PhaseContext, PhaseKind, PlanEntry, RESTORE_THREADS_PHASES, - Repository, RepositorySettings, SAVE, SAVE_THREADS_2, SQLITE_FIXED_64K, SQLITE_RABIN, - STORAGE_CALL_DEADLINE, SaveSettings, Scenario, Selection, WARM_SAVE, concurrent_restore, - concurrent_save, is_key_segment, mixed_phase, plan, prune_phase, repository_key, - repository_scope, result_path, run_phase, save_phase, save_threads_phase, snapshot_name, - with_defaults, with_settings, -}; -use async_trait::async_trait; -use bytes::Bytes; -use futures::stream::BoxStream; -use golem_service_base::replayable_stream::ErasedReplayableStream; -use golem_service_base::storage::blob::memory::InMemoryBlobStorage; -use golem_service_base::storage::blob::{ - BlobMetadata, BlobStorage, BlobStorageNamespace, ExistsResult, ListedBlob, PutIfAbsent, -}; -use pretty_assertions::assert_eq; -use serde_json::{Value, json}; -use std::num::{NonZeroI32, NonZeroUsize}; -use std::path::{Path, PathBuf}; -use std::sync::Arc; -use std::sync::atomic::{AtomicUsize, Ordering}; -use test_r::test; - -/// The phases of the scenario `restore-threads` on a tiny tree. -static RESTORE_THREADS_TINY: Scenario = Scenario { - name: "restore-threads-tiny", - trees: &[FILES_TINY], - phases: RESTORE_THREADS_PHASES, - memory_limited_phases: &[], -}; - -/// The phases of the concurrent scenarios with 3 agents, on a tiny tree. -static CONCURRENT_TINY: Scenario = Scenario { - name: "concurrent-tiny", - trees: &[FILES_TINY], - phases: &[ - SAVE, - concurrent_restore("restore-x3", 3, 2), - concurrent_save("save-x3", 3), - ], - memory_limited_phases: &[], -}; - -/// Runs the phase of the scenario on its first tree, in a new work directory, and gives the -/// result and the work directory. -async fn run_tiny( - scenario: &'static Scenario, - phase: &str, - storage: &Arc, -) -> (PhaseResult, tempfile::TempDir) { - run_tiny_with(scenario, phase, storage.clone(), |_| {}).await -} - -/// Runs the phase of the scenario on its first tree over the storage, in a new work directory -/// that `prepare` gets before the phase runs, and gives the result and the work directory. -async fn run_tiny_with( - scenario: &'static Scenario, - phase: &str, - storage: Arc, - prepare: impl FnOnce(&Path), -) -> (PhaseResult, tempfile::TempDir) { - let work = tempfile::tempdir().unwrap(); - prepare(work.path()); - let selection = Selection { - scenario, - tree: &scenario.trees[0], - phase: scenario - .phases - .iter() - .find(|candidate| candidate.name == phase) - .unwrap(), - }; - let (result, written) = run_phase( - "run-1", - "no-limit", - selection, - work.path(), - storage, - json!({}), - ) - .await; - written.unwrap(); - (result, work) -} - -/// Gives the files, the directories and the bytes of the tree facts of the result. -fn tree_counts(result: &PhaseResult) -> (Option, Option, Option) { - ( - result.tree_facts.files, - result.tree_facts.directories, - result.tree_facts.bytes, - ) -} - -/// The files, the directories and the bytes of a new `FILES_TINY` tree. -const FILES_TINY_COUNTS: (Option, Option, Option) = - (Some(100), Some(10), Some(1024 * 1024)); - -/// Gives the name of each step of the result that did not run. -fn skipped_steps(result: &PhaseResult) -> Vec<&'static str> { - result - .steps - .iter() - .filter(|step| step.status == StepStatus::Skipped) - .map(|step| step.name) - .collect() -} - -/// Gives the name of each step of the result and whether it succeeded. -fn step_states(result: &PhaseResult) -> Vec<(&'static str, bool)> { - result - .steps - .iter() - .map(|step| (step.name, step.status == StepStatus::Ok)) - .collect() -} - -fn step<'a>(result: &'a PhaseResult, name: &str) -> &'a StepRecord { - result.steps.iter().find(|step| step.name == name).unwrap() -} - -/// Gives the namespace of the repositories of the agents of a phase of `CONCURRENT_TINY`. -fn concurrent_tiny_namespace() -> BlobStorageNamespace { - repository_scope(CONCURRENT_TINY.name, "no-limit", FILES_TINY.name).0 -} - -#[test] -fn the_plan_gives_the_phases_of_each_tree_of_each_scenario() { - let names = |names: &[&str]| { - names - .iter() - .map(|name| name.to_string()) - .collect::>() - }; - let entry = |scenario, tree| PlanEntry { - scenario, - tree, - phases: Box::new(["save", "restore"]), - memory_limited_phases: Box::new([]), - }; - let restore_threads = PlanEntry { - phases: Box::new([ - "save", - "restore-1", - "restore-2", - "restore-4", - "restore-8", - "restore-20", - ]), - ..entry("restore-threads", "files-1g") - }; - let memory_pressure = PlanEntry { - memory_limited_phases: Box::new(["restore"]), - ..entry("memory-pressure", "files-1g") - }; - - assert_eq!( - ( - plan(&names(&["base", "smoke", "restore-threads", "memory-pressure"])).map(Vec::from), - plan(&names(&["base", "unknown"])).map(Vec::from), - serde_json::to_string(&entry("base", "files-1g")).unwrap(), - serde_json::to_string(&memory_pressure).unwrap(), - ), - ( - Ok(vec![ - entry("base", "files-128m"), - entry("base", "files-1g"), - entry("base", "sqlite-1g"), - entry("base", "objects-128m"), - entry("base", "objects-1g"), - entry("smoke", "files-tiny"), - entry("smoke", "sqlite-tiny"), - restore_threads.clone(), - memory_pressure.clone(), - ]), - Err("unknown".to_string()), - r#"{"scenario":"base","tree":"files-1g","phases":["save","restore"]}"#.to_string(), - r#"{"scenario":"memory-pressure","tree":"files-1g","phases":["save","restore"],"memory_limited_phases":["restore"]}"#.to_string(), - ) - ); -} - -#[test] -fn the_plan_gives_the_counts_of_the_concurrent_scenarios() { - let restores = |scenario| { - ["files-128m", "files-1g"].map(|tree| PlanEntry { - scenario, - tree, - phases: Box::new([ - "save", - "restore-x1", - "restore-x5", - "restore-x10", - "restore-x25", - "restore-x50", - "restore-x100", - "restore-x200", - ]), - memory_limited_phases: Box::new([]), - }) - }; - let saves = ["files-128m", "files-1g", "sqlite-1g"].map(|tree| PlanEntry { - scenario: "concurrent-save", - tree, - phases: Box::new(["save-x1", "save-x2", "save-x4", "save-x8"]), - memory_limited_phases: Box::new([]), - }); - let kinds = |scenario: &str| { - super::scenario(scenario) - .unwrap() - .phases - .iter() - .map(|phase| phase.kind) - .collect::>() - }; - let threads = |count| super::NonZeroUsize::new(count); - - assert_eq!( - ( - plan(&[ - "concurrent-restore-1".to_string(), - "concurrent-restore-4".to_string(), - "concurrent-restore-20".to_string(), - "concurrent-save".to_string(), - ]) - .map(Vec::from), - kinds("concurrent-restore-4")[1..3].to_vec(), - kinds("concurrent-restore-20").last().copied(), - kinds("concurrent-restore-1")[1], - kinds("concurrent-save"), - ), - ( - Ok([ - restores("concurrent-restore-1").to_vec(), - restores("concurrent-restore-4").to_vec(), - restores("concurrent-restore-20").to_vec(), - saves.to_vec(), - ] - .concat()), - vec![ - super::PhaseKind::ConcurrentRestore { - agents: 1, - reader_threads: threads(4) - }, - super::PhaseKind::ConcurrentRestore { - agents: 5, - reader_threads: threads(4) - }, - ], - Some(super::PhaseKind::ConcurrentRestore { - agents: 200, - reader_threads: threads(20) - }), - super::PhaseKind::ConcurrentRestore { - agents: 1, - reader_threads: threads(1) - }, - [1, 2, 4, 8] - .map(|agents| super::PhaseKind::ConcurrentSave { agents }) - .to_vec(), - ) - ); -} - -#[test] -fn a_selection_names_a_tree_and_a_phase_of_its_scenario() { - let found = |scenario, tree, phase| Selection::find(scenario, tree, phase).map(|_| ()); - - assert_eq!( - [ - found("base", "files-1g", "save"), - found("smoke", "sqlite-tiny", "restore"), - found("base", "objects-1g", "restore"), - found("restore-threads", "files-1g", "restore-8"), - found("memory-pressure", "files-1g", "restore"), - found("concurrent-restore-4", "files-128m", "restore-x200"), - found("concurrent-save", "sqlite-1g", "save-x8"), - found("base", "files-tiny", "save"), - found("base", "files-1g", "prune"), - found("restore-threads", "files-128m", "save"), - found("concurrent-save", "files-1g", "save"), - found("other", "files-1g", "save"), - ] - .map(|found| found.is_ok()), - [ - true, true, true, true, true, true, true, false, false, false, false, false - ] - ); -} - -#[test] -fn a_key_segment_has_1_to_64_ascii_letters_digits_dashes_or_underscores() { - assert_eq!( - [ - "12345-1", - "limit-3", - "no_limit", - "", - "a/b", - "a.b", - &"x".repeat(64), - &"x".repeat(65), - ] - .map(is_key_segment), - [true, true, true, false, false, false, true, false] - ); -} - -#[test] -fn each_run_id_gives_its_own_repository_key() { - assert_eq!( - ( - repository_key("1-1") == repository_key("1-1"), - repository_key("1-1") == repository_key("1-2"), - ), - (true, false) - ); -} - -#[test] -fn a_result_is_at_the_path_of_its_scenario_cpu_setting_tree_and_phase() { - assert_eq!( - &*result_path("base", "limit-3", "files-1g", "save"), - std::path::Path::new("results/base/limit-3/files-1g/save.json") - ); -} - -#[test] -async fn the_smoke_scenario_saves_and_restores_each_tree_with_the_same_hash() { - let storage = Arc::new(InMemoryBlobStorage::new()); - - let outcomes = futures::future::join_all(["files-tiny", "sqlite-tiny"].map(|tree| { - let storage = storage.clone(); - async move { - let save_pod = tempfile::tempdir().unwrap(); - let restore_pod = tempfile::tempdir().unwrap(); - let (save, save_written) = run_phase( - "run-1", - "no-limit", - Selection::find("smoke", tree, "save").unwrap(), - save_pod.path(), - storage.clone(), - json!({}), - ) - .await; - let (restore, restore_written) = run_phase( - "run-1", - "no-limit", - Selection::find("smoke", tree, "restore").unwrap(), - restore_pod.path(), - storage.clone(), - json!({}), - ) - .await; - let steps = |result: &super::report::PhaseResult| { - result - .steps - .iter() - .map(|step| (step.name, step.status == StepStatus::Ok)) - .collect::>() - }; - let hash_step = restore.steps.iter().find(|step| step.name == "hash_tree"); - let restore_step = restore - .steps - .iter() - .find(|step| step.name == "cold_restore"); - let cgroup_keys = |result: &PhaseResult| { - result.environment["cgroup"] - .as_object() - .map(|cgroup| cgroup.keys().cloned().collect::>()) - }; - ( - save.outcome.clone(), - steps(&save), - save_written.is_ok(), - restore.outcome.clone(), - steps(&restore), - restore_written.is_ok(), - hash_step.map(|step| step.details["matches"].clone()), - save.steps - .iter() - .find(|step| step.name == "cold_save") - .map(|step| { - step.requests - .iter() - .any(|request| request.call == "put_raw" && request.file_type == "pack") - }), - restore_step.map(|step| step.parameters.clone()), - cgroup_keys(&save), - ) - } - })) - .await; - let cgroup_keys = Some(vec![ - "memory_events_end".to_string(), - "memory_events_start".to_string(), - ]); - - assert_eq!( - outcomes, - vec![ - ( - Outcome::Ok, - vec![ - ("generate_tree", true), - ("cold_save", true), - ("small_change", true), - ("warm_save", true), - ("hash_tree", true), - ], - true, - Outcome::Ok, - vec![("cold_restore", true), ("hash_tree", true)], - true, - Some(json!(true)), - Some(true), - Some(json!({ "reader_threads": null })), - cgroup_keys, - ); - 2 - ] - ); -} - -#[test] -async fn each_restore_of_the_restore_threads_phases_records_its_reader_threads() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let (save, _save_pod) = run_tiny(&RESTORE_THREADS_TINY, "save", &storage).await; - - let restores = futures::future::join_all(["restore-1", "restore-20"].map(|phase| { - let storage = storage.clone(); - async move { - let (restore, _restore_pod) = run_tiny(&RESTORE_THREADS_TINY, phase, &storage).await; - ( - restore.outcome.clone(), - step(&restore, "cold_restore").parameters.clone(), - step(&restore, "hash_tree").details["matches"].clone(), - ) - } - })) - .await; - - assert_eq!( - (save.outcome, restores), - ( - Outcome::Ok, - vec![ - (Outcome::Ok, json!({ "reader_threads": 1 }), json!(true)), - (Outcome::Ok, json!({ "reader_threads": 20 }), json!(true)), - ] - ) - ); -} - -#[test] -async fn the_agents_of_a_concurrent_restore_get_the_first_repository_and_restore_it() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let (save, _save_pod) = run_tiny(&CONCURRENT_TINY, "save", &storage).await; - - let (first, first_pod) = run_tiny(&CONCURRENT_TINY, "restore-x3", &storage).await; - let (again, _again_pod) = run_tiny(&CONCURRENT_TINY, "restore-x3", &storage).await; - let configs = futures::future::join_all((0..4).map(|agent| { - let storage = storage.clone(); - async move { - storage - .get_metadata( - "test", - "test", - concurrent_tiny_namespace(), - &Path::new("agents").join(agent.to_string()).join("config"), - ) - .await - .unwrap() - .is_some() - } - })) - .await; - let restore = step(&first, "concurrent_restore"); - - assert_eq!( - ( - save.outcome, - first.outcome.clone(), - step_states(&first), - step(&first, "copy_scopes").details.clone(), - step(&again, "copy_scopes").details.clone(), - restore.parameters.clone(), - [ - &restore.details["agents"], - &restore.details["failed"], - &restore.details["first_error"] - ] - .map(Value::clone), - [ - &step(&first, "hash_trees").details["trees"], - &step(&first, "hash_trees").details["matches"] - ] - .map(Value::clone), - configs, - first_pod.path().join("restore").exists(), - ), - ( - Outcome::Ok, - Outcome::Ok, - vec![ - ("copy_scopes", true), - ("concurrent_restore", true), - ("hash_trees", true) - ], - json!({ "agents_copied": 2, "blobs_copied": restore_blobs(&storage).await * 2 }), - json!({ "agents_copied": 0, "blobs_copied": 0 }), - json!({ "agents": 3, "reader_threads": 2 }), - [json!(3), json!(0), Value::Null], - [json!(3), json!(3)], - vec![true, true, true, false], - false, - ) - ); -} - -/// Gives the number of blobs of the repository of the first agent of `CONCURRENT_TINY`. -async fn restore_blobs(storage: &InMemoryBlobStorage) -> usize { - storage - .list_blobs_below( - "test", - "test", - concurrent_tiny_namespace(), - Path::new("agents/0"), - ) - .await - .unwrap() - .len() -} - -#[test] -async fn the_agents_of_a_concurrent_save_each_save_their_tree_into_their_own_repository() { - let storage = Arc::new(InMemoryBlobStorage::new()); - - let (save, pod) = run_tiny(&CONCURRENT_TINY, "save-x3", &storage).await; - let restores = futures::future::join_all((0..3).map(|agent| { - let storage = storage.clone(); - let tree = pod.path().join("trees").join(agent.to_string()); - async move { - let repository = Repository::new( - Arc::new(AgentStorage::new(storage.clone(), &format!("x3-{agent}"))), - repository_scope(CONCURRENT_TINY.name, "no-limit", FILES_TINY.name), - repository_key("run-1"), - STORAGE_CALL_DEADLINE, - ); - let into = tempfile::tempdir().unwrap(); - repository - .restore(&snapshot_name(WARM_SAVE).unwrap(), into.path(), None) - .await - .unwrap(); - let snapshots = storage - .list_blobs_below( - "test", - "test", - concurrent_tiny_namespace(), - &Path::new("agents") - .join(format!("x3-{agent}")) - .join("snapshots"), - ) - .await - .unwrap() - .len(); - ( - snapshots, - tree_hash(into.path()).unwrap() == tree_hash(&tree).unwrap(), - ) - } - })) - .await; - let batch = |name| { - let step = step(&save, name); - ( - step.parameters.clone(), - [ - &step.details["agents"], - &step.details["failed"], - &step.details["first_error"], - ] - .map(Value::clone), - step.details["data_added"] - .as_u64() - .is_some_and(|added| added > 0), - ) - }; - let copied = &step(&save, "copy_trees").details; - - assert_eq!( - ( - save.outcome.clone(), - step_states(&save), - batch("concurrent_cold_save"), - batch("concurrent_warm_save"), - ( - copied["copies"].clone(), - copied["files_reflinked"].as_u64().unwrap_or_default() - + copied["files_copied"].as_u64().unwrap_or_default() - ), - step(&save, "small_change").details["trees"].clone(), - restores, - ), - ( - Outcome::Ok, - vec![ - ("generate_tree", true), - ("copy_trees", true), - ("concurrent_cold_save", true), - ("small_change", true), - ("concurrent_warm_save", true), - ], - ( - json!({ "agents": 3 }), - [json!(3), json!(0), Value::Null], - true - ), - ( - json!({ "agents": 3 }), - [json!(3), json!(0), Value::Null], - true - ), - (json!(2), 200), - json!(3), - vec![(2, true); 3], - ) - ); -} - -/// A tiny tree whose content compresses. -const COMPRESSIBLE_TINY: TreeSpec = TreeSpec { - name: "compressible-tiny", - shape: TreeShape::Files { - files: 100, - directories: 10, - bytes: 1024 * 1024, - }, - content: Content::Compressible, -}; - -static CAPTURE_TINY: Scenario = Scenario { - name: "capture-tiny", - trees: &[FILES_TINY], - phases: &[with_defaults("capture", PhaseKind::Capture)], - memory_limited_phases: &[], -}; - -static PRUNE_TINY: Scenario = Scenario { - name: "prune-tiny", - trees: &[FILES_TINY], - phases: &[ - HISTORY, - prune_phase("prune", &PRUNE, false), - prune_phase("prune-fast-repack", &PRUNE_FAST_REPACK, true), - ], - memory_limited_phases: &[], -}; - -static SQLITE_CHANGES_TINY: Scenario = Scenario { - name: "sqlite-changes-tiny", - trees: &[SQLITE_TINY], - phases: &[ - Phase { - name: "save-rabin", - kind: PhaseKind::SqliteChanges, - variant: &SQLITE_RABIN, - }, - Phase { - name: "save-fixed-64k", - kind: PhaseKind::SqliteChanges, - variant: &SQLITE_FIXED_64K, - }, - Phase { - name: "restore-fixed-64k", - kind: PhaseKind::Restore { - reader_threads: None, - }, - variant: &SQLITE_FIXED_64K, - }, - ], - memory_limited_phases: &[], -}; - -static MIXED_TINY: Scenario = Scenario { - name: "mixed-tiny", - trees: &[FILES_TINY], - phases: &[ - with_defaults("save", PhaseKind::Save), - mixed_phase("mixed-s2-r2", &SAVE_THREADS_2, 2), - save_threads_phase("save-x4-t2", &SAVE_THREADS_2), - ], - memory_limited_phases: &[], -}; - -static SCOPES_TINY: Scenario = Scenario { - name: "scopes-tiny", - trees: &[FILES_TINY], - phases: &[ - with_defaults("save", PhaseKind::Save), - with_defaults("scopes", PhaseKind::Scopes), - ], - memory_limited_phases: &[], -}; - -static CPU_OPTIONS_TINY: Scenario = Scenario { - name: "cpu-options-tiny", - trees: &[COMPRESSIBLE_TINY], - phases: &[ - save_phase("save-default", &CPU_DEFAULT), - save_phase("save-zstd-off", &CPU_ZSTD_OFF), - ], - memory_limited_phases: &[], -}; - -/// Gives the value at the JSON pointer of the details of the step. -fn detail(result: &PhaseResult, step_name: &str, pointer: &str) -> Value { - step(result, step_name) - .details - .pointer(pointer) - .cloned() - .unwrap_or(Value::Null) -} - -#[test] -fn the_plan_gives_the_phases_of_each_later_scenario() { - let entry = |scenario, tree, phases: &[&'static str]| PlanEntry { - scenario, - tree, - phases: phases.into(), - memory_limited_phases: Box::new([]), - }; - let base_trees = [ - "files-128m", - "files-1g", - "sqlite-1g", - "objects-128m", - "objects-1g", - ]; - let history_trees = ["files-1g", "sqlite-1g", "objects-1g"]; - let each = |scenario, trees: &[&'static str], phases: &[&'static str]| { - trees - .iter() - .map(|tree| entry(scenario, *tree, phases)) - .collect::>() - }; - let names = [ - "capture", - "prune", - "save-threads", - "mixed", - "repository-open", - "sqlite-changes", - "tree-shape", - "cpu-options", - "scopes", - ] - .map(str::to_string); - - assert_eq!( - plan(&names).map(Vec::from), - Ok([ - each("capture", &base_trees, &["capture"]), - each( - "prune", - &history_trees, - &["history", "prune", "prune-fast-repack"] - ), - each( - "save-threads", - &["files-1g", "sqlite-1g"], - &["save-x4-t1", "save-x4-t2", "save-x4-t4", "save-x4-tdefault"] - ), - each( - "mixed", - &["files-1g"], - &[ - "save", - "mixed-s2-r2", - "mixed-s2-r4", - "mixed-s4-r2", - "mixed-s4-r4" - ] - ), - each("repository-open", &history_trees, &["history", "prune"]), - each( - "sqlite-changes", - &["sqlite-1g"], - &[ - "save-rabin", - "restore-rabin", - "save-fixed-64k", - "restore-fixed-64k" - ] - ), - each( - "tree-shape", - &["files-128m", "modules-128m"], - &["save", "restore"] - ), - each( - "cpu-options", - &["compressible-1g"], - &[ - "save-default", - "save-verify-off", - "save-zstd-off", - "save-zstd-1", - "save-zstd-9" - ] - ), - each("scopes", &base_trees, &["save", "scopes"]), - ] - .concat()) - ); -} - -#[test] -fn a_later_phase_records_its_settings_and_an_earlier_phase_records_none() { - let record = || StepRecord::skipped("save").with_parameters(json!({ "agents": 4 })); - let own = - StepRecord::skipped("save").with_parameters(json!({ "change_detection": "size-mtime" })); - - assert_eq!( - ( - with_settings(record(), &BASE).parameters, - with_settings(record(), &SAVE_THREADS_2).parameters, - with_settings(own, &DEFAULTS).parameters["change_detection"].clone(), - with_settings(record(), &SQLITE_FIXED_64K).parameters["chunker"].clone(), - with_settings(record(), &CPU_ZSTD_OFF).parameters["compression"].clone(), - ), - ( - json!({ "agents": 4 }), - json!({ - "agents": 4, - "save_threads": 2, - "change_detection": "ctime", - "chunker": "rabin", - "compression": null, - "extra_verify": true, - }), - json!("size-mtime"), - json!("fixed-65536"), - json!("off"), - ) - ); -} - -#[test] -async fn a_capture_phase_reads_every_file_with_ctime_and_the_changed_files_with_size_and_mtime() { - let storage = Arc::new(InMemoryBlobStorage::new()); - - let (result, _pod) = run_tiny(&CAPTURE_TINY, "capture", &storage).await; - let counts = |name: &str| { - ( - detail(&result, name, "/files_new"), - detail(&result, name, "/files_changed"), - detail(&result, name, "/files_unmodified"), - ) - }; - let captured = |name: &str| { - detail(&result, name, "/files_reflinked") - .as_u64() - .unwrap_or_default() - + detail(&result, name, "/files_copied") - .as_u64() - .unwrap_or_default() - }; - - assert_eq!( - ( - result.outcome.clone(), - step_states(&result), - [ - captured("capture"), - captured("capture_full_read"), - captured("capture_size_mtime") - ], - counts("warm_save_full_read"), - counts("warm_save_size_mtime"), - step(&result, "warm_save_size_mtime").parameters["change_detection"].clone(), - tree_counts(&result), - result.tree_facts.change.as_ref() == Some(&step(&result, "small_change").details), - ), - ( - Outcome::Ok, - vec![ - ("generate_tree", true), - ("capture", true), - ("cold_save", true), - ("copy_scopes", true), - ("small_change", true), - ("capture_full_read", true), - ("warm_save_full_read", true), - ("capture_size_mtime", true), - ("warm_save_size_mtime", true), - ], - [100, 101, 101], - (json!(1), json!(100), json!(0)), - (json!(1), json!(10), json!(90)), - json!("size-mtime"), - FILES_TINY_COUNTS, - true, - ) - ); -} - -#[test] -async fn the_prune_phases_give_back_the_data_of_the_forgotten_snapshots_of_the_history() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let (history, _history_pod) = run_tiny(&PRUNE_TINY, "history", &storage).await; - - let prunes = futures::future::join_all(["prune", "prune-fast-repack"].map(|phase| { - let storage = storage.clone(); - async move { - let (result, _pod) = run_tiny(&PRUNE_TINY, phase, &storage).await; - ( - result.outcome.clone(), - step_states(&result), - step(&result, "prune_mark").parameters["fast_repack"].clone(), - detail(&result, "prune_mark", "/packs_repacked") - .as_u64() - .is_some_and(|packs| packs > 0), - detail(&result, "prune_delete", "/marked_packs_deleted") - .as_u64() - .is_some_and(|packs| packs > 0), - detail(&result, "prune_delete", "/repository/bytes_given_back") - .as_u64() - .is_some_and(|bytes| bytes > 0), - detail(&result, "open", "/snapshots"), - detail(&result, "hash_tree", "/matches"), - tree_counts(&result), - result.tree_facts.hash.as_deref().map(|hash| json!(hash)) - == Some(detail(&result, "hash_tree", "/hash")), - ) - } - })) - .await; - let opens = history - .steps - .iter() - .filter(|step| step.name == "open") - .map(|step| { - ( - step.parameters["saves"].clone(), - step.details["snapshots"].clone(), - ) - }) - .collect::>(); - let saves = history - .steps - .iter() - .filter(|step| step.name == "save") - .count(); - let prune = |fast_repack| { - ( - Outcome::Ok, - vec![ - ("copy_scopes", true), - ("prune_mark", true), - ("prune_delete", true), - ("open", true), - ("cold_restore", true), - ("hash_tree", true), - ], - json!(fast_repack), - true, - true, - true, - json!(2), - json!(true), - // The history adds one file of 10,486 bytes in each of its 11 rounds. - (Some(111), Some(10), Some(1024 * 1024 + 11 * 10_486)), - true, - ) - }; - - assert_eq!( - ( - history.outcome.clone(), - saves, - opens, - detail(&history, "forget", "/snapshots_forgotten"), - tree_counts(&history), - prunes, - ), - ( - Outcome::Ok, - 11, - vec![ - (json!(1), json!(1)), - (json!(11), json!(11)), - (json!(12), json!(12)), - ], - json!(10), - FILES_TINY_COUNTS, - vec![prune(false), prune(true)], - ) - ); -} - -#[test] -async fn fixed_chunks_save_less_after_a_clustered_change_and_restore_the_database() { - let storage = Arc::new(InMemoryBlobStorage::new()); - - let (rabin, _rabin_pod) = run_tiny(&SQLITE_CHANGES_TINY, "save-rabin", &storage).await; - let (fixed, _fixed_pod) = run_tiny(&SQLITE_CHANGES_TINY, "save-fixed-64k", &storage).await; - let (restore, _restore_pod) = - run_tiny(&SQLITE_CHANGES_TINY, "restore-fixed-64k", &storage).await; - let added = |result: &PhaseResult, name: &str| { - detail(result, name, "/data_added") - .as_u64() - .unwrap_or_default() - }; - - assert_eq!( - ( - rabin.outcome.clone(), - fixed.outcome.clone(), - step_states(&fixed), - added(&fixed, "warm_save_clustered") < added(&rabin, "warm_save_clustered"), - step(&fixed, "cold_save").parameters["chunker"].clone(), - detail(&fixed, "clustered_change", "/rows_updated"), - restore.outcome.clone(), - detail(&restore, "hash_tree", "/matches"), - (fixed.tree_facts.files, fixed.tree_facts.directories), - fixed - .tree_facts - .bytes - .is_some_and(|bytes| bytes >= 4 * 1024 * 1024), - fixed.tree_facts.change.as_ref() == Some(&step(&fixed, "scattered_change").details), - [ - detail(&rabin, "open", "/settings/chunker"), - detail(&fixed, "open", "/settings"), - ], - ), - ( - Outcome::Ok, - Outcome::Ok, - vec![ - ("generate_tree", true), - ("cold_save", true), - ("clustered_change", true), - ("warm_save_clustered", true), - ("scattered_change", true), - ("warm_save_scattered", true), - ("hash_tree", true), - ("open", true), - ], - true, - json!("fixed-65536"), - json!(100), - Outcome::Ok, - json!(true), - (Some(1), Some(0)), - true, - true, - [ - json!("rabin"), - json!({ "chunker": "fixed-65536", "compression": null, "extra_verify": true }), - ], - ) - ); -} - -#[test] -async fn a_mixed_phase_saves_and_restores_at_the_same_time_with_its_thread_counts() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let (save, _save_pod) = run_tiny(&MIXED_TINY, "save", &storage).await; - - let (mixed, _mixed_pod) = run_tiny(&MIXED_TINY, "mixed-s2-r2", &storage).await; - let (threads, _threads_pod) = run_tiny(&MIXED_TINY, "save-x4-t2", &storage).await; - let configs = futures::future::join_all( - [ - "agents/1", - "agents/4", - "agents/5", - "agents/mixed-s2-r2-3", - "agents/save-x4-t2-3", - ] - .map(|agent| { - let storage = storage.clone(); - async move { - storage - .get_metadata( - "test", - "test", - repository_scope(MIXED_TINY.name, "no-limit", FILES_TINY.name).0, - &Path::new(agent).join("config"), - ) - .await - .unwrap() - .is_some() - } - }), - ) - .await; - let step_mixed = step(&mixed, "mixed"); - - assert_eq!( - ( - save.outcome, - mixed.outcome.clone(), - step_states(&mixed), - [ - &step_mixed.parameters["saves"], - &step_mixed.parameters["restores"], - &step_mixed.parameters["save_threads"], - &step_mixed.parameters["reader_threads"], - ] - .map(Value::clone), - [ - &step_mixed.details["saves"]["agents"], - &step_mixed.details["saves"]["failed"], - &step_mixed.details["restores"]["agents"], - &step_mixed.details["restores"]["failed"], - ] - .map(Value::clone), - detail(&mixed, "hash_trees", "/matches"), - threads.outcome.clone(), - step(&threads, "concurrent_cold_save").parameters["save_threads"].clone(), - configs, - ), - ( - Outcome::Ok, - Outcome::Ok, - vec![ - ("copy_scopes", true), - ("generate_tree", true), - ("copy_trees", true), - ("mixed", true), - ("hash_trees", true), - ], - [json!(4), json!(4), json!(2), json!(2)], - [json!(4), json!(0), json!(4), json!(0)], - json!(4), - Outcome::Ok, - json!(2), - vec![true, true, false, true, true], - ) - ); -} - -#[test] -async fn a_scopes_phase_copies_the_repository_whole_and_deletes_the_copy_whole() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let (save, _save_pod) = run_tiny(&SCOPES_TINY, "save", &storage).await; - - let (scopes, _pod) = run_tiny(&SCOPES_TINY, "scopes", &storage).await; - - assert_eq!( - ( - save.outcome, - scopes.outcome.clone(), - step_states(&scopes), - detail(&scopes, "copy_scope", "/same_as_source"), - detail(&scopes, "copy_scope", "/blobs_copied") - .as_u64() - .is_some_and(|blobs| blobs > 0), - detail(&scopes, "delete_scope", "/blobs_left"), - ), - ( - Outcome::Ok, - Outcome::Ok, - vec![("copy_scope", true), ("delete_scope", true)], - json!(true), - true, - json!(0), - ) - ); -} - -#[test] -async fn a_cpu_options_phase_makes_its_repository_with_its_compression() { - let storage = Arc::new(InMemoryBlobStorage::new()); - - let (default, _default_pod) = run_tiny(&CPU_OPTIONS_TINY, "save-default", &storage).await; - let (off, _off_pod) = run_tiny(&CPU_OPTIONS_TINY, "save-zstd-off", &storage).await; - let packed = |result: &PhaseResult| { - ( - detail(result, "cold_save", "/data_added").as_u64(), - detail(result, "cold_save", "/data_added_packed").as_u64(), - ) - }; - let (default_added, default_packed) = packed(&default); - let (off_added, off_packed) = packed(&off); - - assert_eq!( - ( - default.outcome.clone(), - off.outcome.clone(), - default.tree_facts.content, - default_packed < default_added.map(|added| added * 3 / 4), - off_packed >= off_added, - step(&off, "cold_save").parameters["compression"].clone(), - step(&default, "open").details.clone(), - detail(&off, "open", "/settings"), - ), - ( - Outcome::Ok, - Outcome::Ok, - Some("compressible"), - true, - true, - json!("off"), - json!({ - "snapshots": 2, - "found": true, - "settings": { "chunker": "rabin", "compression": null, "extra_verify": true }, - }), - json!({ "chunker": "rabin", "compression": "off", "extra_verify": true }), - ) - ); -} - -/// A blob storage that passes each call to an in-memory storage, and fails each write of a -/// snapshot file below `prefix` after the first `allowed` of them. After `config_hidden_after` -/// such writes, it gives no metadata for the config of the agent, so the repository looks absent. -#[derive(Debug)] -struct FailingSnapshotWrites { - inner: Arc, - prefix: PathBuf, - config: PathBuf, - allowed: usize, - config_hidden_after: usize, - seen: AtomicUsize, -} - -impl FailingSnapshotWrites { - /// Gives a storage over `inner` that fails the writes of snapshot files of the agent after the - /// first `allowed` of them. - fn new(inner: Arc, agent: &str, allowed: usize) -> Arc { - Self::hiding_config(inner, agent, allowed, usize::MAX) - } - - /// Gives a storage over `inner` that fails the writes of snapshot files of the agent after the - /// first `allowed` of them, and hides the config of the agent after `config_hidden_after` - /// writes of snapshot files. - fn hiding_config( - inner: Arc, - agent: &str, - allowed: usize, - config_hidden_after: usize, - ) -> Arc { - let root = Path::new("agents").join(agent); - Arc::new(Self { - inner, - prefix: root.join("snapshots"), - config: root.join("config"), - allowed, - config_hidden_after, - seen: AtomicUsize::new(0), - }) - } - - /// Tells whether the write of the path fails. - fn fails(&self, path: &Path) -> bool { - path.starts_with(&self.prefix) && self.seen.fetch_add(1, Ordering::SeqCst) >= self.allowed - } -} - -#[async_trait] -impl BlobStorage for FailingSnapshotWrites { - async fn get_raw( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result>> { - self.inner - .get_raw(target_label, op_label, namespace, path) - .await - } - - async fn get_stream( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result>>> { - self.inner - .get_stream(target_label, op_label, namespace, path) - .await - } - - async fn get_raw_slice( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - start: u64, - end: u64, - ) -> anyhow::Result>> { - self.inner - .get_raw_slice(target_label, op_label, namespace, path, start, end) - .await - } - - async fn get_metadata( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - if path == self.config && self.seen.load(Ordering::SeqCst) >= self.config_hidden_after { - return Ok(None); - } - self.inner - .get_metadata(target_label, op_label, namespace, path) - .await - } - - async fn put_raw( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - data: &[u8], - ) -> anyhow::Result<()> { - anyhow::ensure!(!self.fails(path), "the write of {} fails", path.display()); - self.inner - .put_raw(target_label, op_label, namespace, path, data) - .await - } - - async fn put_raw_if_absent( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - data: &[u8], - ) -> anyhow::Result { - self.inner - .put_raw_if_absent(target_label, op_label, namespace, path, data) - .await - } - - async fn put_stream( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - stream: &dyn ErasedReplayableStream>, Error = anyhow::Error>, - ) -> anyhow::Result<()> { - self.inner - .put_stream(target_label, op_label, namespace, path, stream) - .await - } - - async fn delete( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result<()> { - self.inner - .delete(target_label, op_label, namespace, path) - .await - } - - async fn create_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result<()> { - self.inner - .create_dir(target_label, op_label, namespace, path) - .await - } - - async fn list_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.inner - .list_dir(target_label, op_label, namespace, path) - .await - } - - async fn list_blobs_below( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result> { - self.inner - .list_blobs_below(target_label, op_label, namespace, path) - .await - } - - async fn delete_dir( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result { - self.inner - .delete_dir(target_label, op_label, namespace, path) - .await - } - - async fn exists( - &self, - target_label: &'static str, - op_label: &'static str, - namespace: BlobStorageNamespace, - path: &Path, - ) -> anyhow::Result { - self.inner - .exists(target_label, op_label, namespace, path) - .await - } -} - -/// The steps of the two forms of a capture phase, in the order in which they run. -const FORM_STEPS: [&str; 4] = [ - "capture_full_read", - "warm_save_full_read", - "capture_size_mtime", - "warm_save_size_mtime", -]; - -#[test] -async fn a_failed_capture_or_warm_save_of_a_form_skips_the_steps_after_it() { - // Form 0 saves into the repository of agent 0, after its cold save. Form 1 saves into the - // repository of agent 1, after the copy of the cold save, which the default `copy` of the - // storage writes. So in both repositories the warm save writes the second snapshot file. A - // capture fails when its directory exists before it. - let capture_fails = async |capture: &'static str| { - let (result, _pod) = run_tiny_with( - &CAPTURE_TINY, - "capture", - Arc::new(InMemoryBlobStorage::new()), - |work| std::fs::create_dir(work.join(capture)).unwrap(), - ) - .await; - (result.outcome.clone(), skipped_steps(&result)) - }; - let save_fails = async |agent: &str, allowed| { - let storage = - FailingSnapshotWrites::new(Arc::new(InMemoryBlobStorage::new()), agent, allowed); - let (result, _pod) = run_tiny_with(&CAPTURE_TINY, "capture", storage, |_| {}).await; - (result.outcome.clone(), skipped_steps(&result)) - }; - let failed = |step: &str| Outcome::Failed { - reason: format!("the step {step} failed").into(), - }; - - let outcomes = [ - capture_fails(FORM_STEPS[0]).await, - save_fails("0", 1).await, - capture_fails(FORM_STEPS[2]).await, - save_fails("1", 1).await, - ]; - - assert_eq!( - outcomes, - [ - (failed(FORM_STEPS[0]), FORM_STEPS[1..].to_vec()), - (failed(FORM_STEPS[1]), FORM_STEPS[2..].to_vec()), - (failed(FORM_STEPS[2]), FORM_STEPS[3..].to_vec()), - (failed(FORM_STEPS[3]), Vec::new()), - ] - ); -} - -/// Gives the namespace of the repositories of the agents of a phase of `MIXED_TINY`. -fn mixed_tiny_namespace() -> BlobStorageNamespace { - repository_scope(MIXED_TINY.name, "no-limit", FILES_TINY.name).0 -} - -#[test] -async fn a_mixed_phase_whose_restores_fail_fails_at_the_mixed_step_with_the_error_of_a_restore() { - // The repository of agent 1 has a config that no key opens, so the copy of the first - // repository leaves it out, and its restore fails. The saves go to other agents and succeed. - let storage = Arc::new(InMemoryBlobStorage::new()); - let (save, _save_pod) = run_tiny(&MIXED_TINY, "save", &storage).await; - storage - .put_raw( - "test", - "test", - mixed_tiny_namespace(), - Path::new("agents/1/config"), - b"not a config", - ) - .await - .unwrap(); - - let (mixed, _pod) = run_tiny(&MIXED_TINY, "mixed-s2-r2", &storage).await; - let details = &step(&mixed, "mixed").details; - - assert_eq!( - ( - save.outcome, - mixed.outcome.clone(), - step(&mixed, "mixed").status != StepStatus::Ok, - details["first_error"].is_string(), - [&details["saves"]["failed"], &details["restores"]["failed"]].map(Value::clone), - skipped_steps(&mixed), - tree_counts(&mixed), - ), - ( - Outcome::Ok, - Outcome::Failed { - reason: "the step mixed failed".into() - }, - true, - true, - [json!(0), json!(1)], - vec!["hash_trees"], - FILES_TINY_COUNTS, - ) - ); -} - -#[test] -async fn the_restores_of_a_mixed_phase_read_the_copies_and_not_the_first_repository() { - // The copies exist before the phase, so the phase copies nothing, and the first repository - // loses its snapshot files. A restore that reads the first repository finds no snapshot. - let storage = Arc::new(InMemoryBlobStorage::new()); - let (save, _save_pod) = run_tiny(&MIXED_TINY, "save", &storage).await; - super::agents::copy_first_agent(storage.as_ref(), &mixed_tiny_namespace(), 5) - .await - .unwrap(); - storage - .delete_dir( - "test", - "test", - mixed_tiny_namespace(), - Path::new("agents/0/snapshots"), - ) - .await - .unwrap(); - - let (mixed, _pod) = run_tiny(&MIXED_TINY, "mixed-s2-r2", &storage).await; - - assert_eq!( - ( - save.outcome, - mixed.outcome.clone(), - detail(&mixed, "copy_scopes", "/agents_copied"), - detail(&mixed, "hash_trees", "/matches"), - ), - (Outcome::Ok, Outcome::Ok, json!(0), json!(4)) - ); -} - -#[test] -fn each_cpu_options_variant_changes_only_its_own_setting_of_the_defaults() { - let defaults = RepositorySettings::DEFAULT; - let variants = CPU_OPTIONS_PHASES - .iter() - .map(|phase| { - ( - phase.name, - phase.variant.settings.map(|settings| settings.repository), - phase.variant.settings.map(|settings| settings.save), - ) - }) - .collect::>(); - let level = |level| Compression::Level(NonZeroI32::new(level).unwrap()); - - assert_eq!( - variants, - [ - ("save-default", defaults), - ( - "save-verify-off", - RepositorySettings { - extra_verify: false, - ..defaults - } - ), - ( - "save-zstd-off", - RepositorySettings { - compression: Compression::Off, - ..defaults - } - ), - ( - "save-zstd-1", - RepositorySettings { - compression: level(1), - ..defaults - } - ), - ( - "save-zstd-9", - RepositorySettings { - compression: level(9), - ..defaults - } - ), - ] - .map(|(name, repository)| (name, Some(repository), Some(SaveSettings::DEFAULT))) - .to_vec() - ); -} - -#[test] -fn a_save_threads_phase_saves_with_its_threads() { - let scenario = super::scenario("save-threads").unwrap(); - let context = |phase: &str| PhaseContext { - run_id: "run-1".into(), - cpu_setting: "no-limit".into(), - selection: Selection::find("save-threads", "files-1g", phase).unwrap(), - work_dir: Path::new("/nowhere").into(), - storage: Arc::new(MeasuredBlobStorage::new(Arc::new( - InMemoryBlobStorage::new(), - ))), - }; - - assert_eq!( - ( - scenario.phases.len(), - ["save-x4-t1", "save-x4-t2", "save-x4-t4", "save-x4-tdefault"] - .map(|phase| context(phase).save_settings()), - ), - ( - 4, - [ - NonZeroUsize::new(1), - NonZeroUsize::new(2), - NonZeroUsize::new(4), - None - ] - .map(|threads| SaveSettings { - threads, - ..SaveSettings::DEFAULT - }), - ) - ); -} - -#[test] -async fn a_save_and_open_phase_whose_save_fails_skips_the_open() { - // The cold save writes the first snapshot file of the repository, and the last warm save - // writes the last one: the second for the cpu-options phase, and the third for the - // sqlite-changes phase, which also saves after its clustered change. - let fail_last_save = async |scenario: &'static Scenario, phase, agent, snapshots: usize| { - let storage = - FailingSnapshotWrites::new(Arc::new(InMemoryBlobStorage::new()), agent, snapshots - 1); - let (result, _pod) = run_tiny_with(scenario, phase, storage, |_| {}).await; - (result.outcome.clone(), skipped_steps(&result)) - }; - - let outcomes = [ - fail_last_save(&CPU_OPTIONS_TINY, "save-default", "default", 2).await, - fail_last_save(&SQLITE_CHANGES_TINY, "save-rabin", "rabin", 3).await, - ]; - - assert_eq!( - outcomes, - [ - ( - Outcome::Failed { - reason: "the step warm_save failed".into() - }, - vec!["hash_tree", "open"] - ), - ( - Outcome::Failed { - reason: "the step warm_save_scattered failed".into() - }, - vec!["hash_tree", "open"] - ), - ] - ); -} - -#[test] -async fn a_save_and_open_phase_fails_at_the_open_when_the_open_finds_no_repository() { - // The config of the repository disappears after the second snapshot file, which the warm - // save writes last, so the open after the saves finds no repository. - let storage = FailingSnapshotWrites::hiding_config( - Arc::new(InMemoryBlobStorage::new()), - "default", - usize::MAX, - 2, - ); - - let (result, _pod) = run_tiny_with(&CPU_OPTIONS_TINY, "save-default", storage, |_| {}).await; - - assert_eq!( - ( - result.outcome.clone(), - step_states(&result).last().copied(), - step(&result, "open").details.clone(), - skipped_steps(&result), - ), - ( - Outcome::Failed { - reason: "the step open failed".into() - }, - Some(("open", true)), - json!({ "repository": null }), - Vec::<&str>::new(), - ) - ); -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/trees.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/trees.rs deleted file mode 100644 index 4f2e3daf67..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/trees.rs +++ /dev/null @@ -1,1324 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The trees of the benchmark: how a tree is made, how it changes, and its hash. -//! -//! The content of a tree does not compress, unless the tree has compressible content. After each -//! write, the pages of the written files leave the page cache, so a save reads them from the -//! volume. The upload of a reflink capture reads its files from the volume too, because a clone -//! does not share the page cache of its source. - -use anyhow::Context; -use futures::{StreamExt, TryStreamExt}; -use serde_json::{Value, json}; -use sqlx::sqlite::{SqliteConnectOptions, SqliteJournalMode, SqliteSynchronous}; -use sqlx::{ConnectOptions, Connection, Executor, SqliteConnection}; -use std::fs::{File, Metadata}; -use std::io; -use std::os::unix::ffi::OsStrExt; -use std::os::unix::fs::{MetadataExt, PermissionsExt}; -use std::path::{Path, PathBuf}; - -const MIB: u64 = 1024 * 1024; - -/// The name of the database file of a SQLite tree. -const DATABASE: &str = "database.sqlite"; - -/// The rows that one insert of a SQLite tree adds. -const ROWS_PER_INSERT: u64 = 1_000; - -/// The bytes of the payload of one row of a SQLite tree. -const ROW_PAYLOAD_BYTES: u64 = 1_024; - -/// The number of files that a small change rewrites, and the rows that it updates. -const CHANGED_FILES: u64 = 10; -const CHANGED_ROWS: u64 = 100; - -/// The number of files in each directory of a tree at the object limit, as in the other file -/// trees. -const FILES_PER_DIRECTORY: u64 = 100; - -/// The directories of the first level of a modules tree. Each has 10 directories, and each of -/// those has 10 directories that hold the files. -const MODULE_PACKAGES: u64 = 25; -const MODULE_FANOUT: u64 = 10; - -/// The bytes of a block of compressible content. Its second half is zero. -const COMPRESSIBLE_BLOCK: usize = 64; - -/// A tree that the benchmark makes. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) struct TreeSpec { - pub(super) name: &'static str, - pub(super) shape: TreeShape, - pub(super) content: Content, -} - -/// What the files of a tree hold. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) enum Content { - /// Bytes that do not compress. - Incompressible, - /// Blocks of 64 bytes, each with 32 bytes that do not compress and 32 zero bytes, so zstd - /// makes them about half as large. - Compressible, -} - -impl Content { - /// Gives the name of the content, which the result records. - pub(super) const fn label(self) -> &'static str { - match self { - Content::Incompressible => "incompressible", - Content::Compressible => "compressible", - } - } -} - -/// What a tree holds. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) enum TreeShape { - /// Files of about the same size, in directories of the same number of files. - Files { - files: u64, - directories: u64, - bytes: u64, - }, - /// One SQLite database of at least the size. - Sqlite { bytes: u64 }, - /// Files and directories that are `objects` filesystem objects together with the root, in - /// the layout that [`limit_layout`] gives. Its small change keeps the number of objects. - ObjectLimit { objects: u64, bytes: u64 }, - /// Files of about the same size in many small directories, as a `node_modules` tree has - /// them: the files fill the directories of the third level of [`MODULE_PACKAGES`] directories - /// with [`MODULE_FANOUT`] directories each, which again have [`MODULE_FANOUT`] directories. - Modules { files: u64, bytes: u64 }, -} - -pub(super) const FILES_128M: TreeSpec = TreeSpec { - name: "files-128m", - shape: TreeShape::Files { - files: 10_000, - directories: 100, - bytes: 128 * MIB, - }, - content: Content::Incompressible, -}; - -pub(super) const FILES_1G: TreeSpec = TreeSpec { - name: "files-1g", - shape: TreeShape::Files { - files: 10_000, - directories: 100, - bytes: 1024 * MIB, - }, - content: Content::Incompressible, -}; - -/// The tree at the object limit of an agent with 128 MiB of storage: 8,192 objects. -pub(super) const OBJECTS_128M: TreeSpec = TreeSpec { - name: "objects-128m", - shape: TreeShape::ObjectLimit { - objects: 8_192, - bytes: 128 * MIB, - }, - content: Content::Incompressible, -}; - -/// The tree at the object limit of an agent with 1 GiB of storage: 32,768 objects. -pub(super) const OBJECTS_1G: TreeSpec = TreeSpec { - name: "objects-1g", - shape: TreeShape::ObjectLimit { - objects: 32_768, - bytes: 1024 * MIB, - }, - content: Content::Incompressible, -}; - -pub(super) const SQLITE_1G: TreeSpec = TreeSpec { - name: "sqlite-1g", - shape: TreeShape::Sqlite { bytes: 1024 * MIB }, - content: Content::Incompressible, -}; - -/// The 10,000 files and 128 MiB of [`FILES_128M`] in 2,775 directories, 3 levels deep. -pub(super) const MODULES_128M: TreeSpec = TreeSpec { - name: "modules-128m", - shape: TreeShape::Modules { - files: 10_000, - bytes: 128 * MIB, - }, - content: Content::Incompressible, -}; - -/// The layout of [`FILES_1G`], with content that compresses to about half its size. -pub(super) const COMPRESSIBLE_1G: TreeSpec = TreeSpec { - name: "compressible-1g", - shape: TreeShape::Files { - files: 10_000, - directories: 100, - bytes: 1024 * MIB, - }, - content: Content::Compressible, -}; - -pub(super) const FILES_TINY: TreeSpec = TreeSpec { - name: "files-tiny", - shape: TreeShape::Files { - files: 100, - directories: 10, - bytes: MIB, - }, - content: Content::Incompressible, -}; - -pub(super) const SQLITE_TINY: TreeSpec = TreeSpec { - name: "sqlite-tiny", - shape: TreeShape::Sqlite { bytes: 4 * MIB }, - content: Content::Incompressible, -}; - -/// The number of files, directories and bytes of a tree, not counting its root. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub(super) struct TreeCounts { - pub(super) files: u64, - pub(super) directories: u64, - pub(super) bytes: u64, -} - -/// Gives the number of files and the number of directories of a tree whose files, directories -/// and root are `objects` objects together. -/// -/// Each directory holds at most [`FILES_PER_DIRECTORY`] files, so a directory and its files are -/// at most `FILES_PER_DIRECTORY + 1` objects. -pub(super) const fn limit_layout(objects: u64) -> (u64, u64) { - let below_root = objects.saturating_sub(1); - let directories = below_root.div_ceil(FILES_PER_DIRECTORY + 1); - (below_root - directories, directories) -} - -/// Makes the tree in the directory `root`, which must not exist. -pub(super) async fn generate(spec: &TreeSpec, root: &Path) -> anyhow::Result { - std::fs::create_dir(root).with_context(|| format!("create the tree {}", root.display()))?; - match tree_content(spec) { - TreeContent::Files(files) => { - in_blocking(root, move |root| generate_files(root, &files)).await - } - TreeContent::Sqlite { bytes } => generate_database(&root.join(DATABASE), bytes).await, - }?; - settle(root)?; - Ok(count(root)?) -} - -/// Changes a small part of the tree in `root`, and gives what changed. It is the first round of -/// [`change_round`]. -pub(super) async fn change(spec: &TreeSpec, root: &Path) -> anyhow::Result { - change_round(spec, root, 1).await -} - -/// Changes a small part of the tree in `root` for the round, and gives what changed. Each round -/// from 1 to 255 writes other content. -/// -/// A files tree gets new content in the same files in each round, and one new file. A tree at the -/// object limit also loses one file in each round, so it keeps its number of objects. A SQLite -/// tree gets new payload in rows spread over the database. -pub(super) async fn change_round(spec: &TreeSpec, root: &Path, round: u8) -> anyhow::Result { - let change = match tree_content(spec) { - TreeContent::Files(files) => { - in_blocking(root, move |root| change_files(root, &files, round)).await? - } - TreeContent::Sqlite { .. } => { - change_database(&root.join(DATABASE), SqliteChange::Scattered).await? - } - }; - settle(root)?; - Ok(change) -} - -/// Gives new payload to 100 consecutive rows in the middle of the database of a SQLite tree, and -/// gives what changed. The rows are on about 34 consecutive pages. -pub(super) async fn change_clustered(spec: &TreeSpec, root: &Path) -> anyhow::Result { - anyhow::ensure!( - matches!(tree_content(spec), TreeContent::Sqlite { .. }), - "the tree {} has no database", - spec.name - ); - let change = change_database(&root.join(DATABASE), SqliteChange::Clustered).await?; - settle(root)?; - Ok(change) -} - -/// Runs the work on the tree `root` on a blocking thread. -async fn in_blocking( - root: &Path, - work: impl FnOnce(&Path) -> anyhow::Result + Send + 'static, -) -> anyhow::Result { - let root = root.to_path_buf(); - tokio::task::spawn_blocking(move || work(&root)).await? -} - -/// Where the files of a tree are. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) enum Layout { - /// The files fill `directories` directories below the root in their order. - Flat { files: u64, directories: u64 }, - /// The files fill the directories of the third level of a modules tree in their order. - Modules { files: u64 }, -} - -impl Layout { - fn files(self) -> u64 { - match self { - Layout::Flat { files, .. } | Layout::Modules { files } => files, - } - } - - /// Gives the path of the file with the index, relative to the root of the tree. - pub(super) fn path(self, index: u64) -> Box { - match self { - Layout::Flat { files, directories } => file_path(index, files, directories), - Layout::Modules { files } => { - let leaves = MODULE_PACKAGES * MODULE_FANOUT * MODULE_FANOUT; - let leaf = index / files.div_ceil(leaves).max(1); - PathBuf::from(format!( - "p{:02}/s{}/l{}/f{index:05}", - leaf / (MODULE_FANOUT * MODULE_FANOUT), - leaf / MODULE_FANOUT % MODULE_FANOUT, - leaf % MODULE_FANOUT - )) - .into_boxed_path() - } - } - } -} - -/// The files of a tree that is not a SQLite tree. -#[derive(Clone, Copy, Debug)] -struct Files { - layout: Layout, - bytes: u64, - replace: Replace, - content: Content, -} - -/// What a tree holds: files, or one SQLite database. -#[derive(Clone, Copy, Debug)] -enum TreeContent { - Files(Files), - /// One SQLite database of at least the size. - Sqlite { - bytes: u64, - }, -} - -/// Gives what the tree holds. -fn tree_content(spec: &TreeSpec) -> TreeContent { - let (layout, bytes, replace) = match spec.shape { - TreeShape::Files { - files, - directories, - bytes, - } => (Layout::Flat { files, directories }, bytes, Replace::No), - TreeShape::ObjectLimit { objects, bytes } => { - let (files, directories) = limit_layout(objects); - (Layout::Flat { files, directories }, bytes, Replace::Yes) - } - TreeShape::Modules { files, bytes } => (Layout::Modules { files }, bytes, Replace::No), - TreeShape::Sqlite { bytes } => return TreeContent::Sqlite { bytes }, - }; - TreeContent::Files(Files { - layout, - bytes, - replace, - content: spec.content, - }) -} - -/// Gives the path of the file with the index in a flat tree, relative to the root of the tree. -fn file_path(index: u64, files: u64, directories: u64) -> Box { - let per_directory = files.div_ceil(directories.max(1)).max(1); - PathBuf::from(format!("d{:03}/f{index:05}", index / per_directory)).into_boxed_path() -} - -/// Gives the size of the file with the index. The sizes add up to `bytes`. -fn file_size(index: u64, files: u64, bytes: u64) -> u64 { - bytes / files.max(1) + u64::from(index < bytes % files.max(1)) -} - -/// Gives `size` bytes of the content, the same for the same path and generation. -fn content(path: &Path, generation: u8, size: u64, kind: Content) -> io::Result> { - let mut content = vec![0; usize::try_from(size).map_err(io::Error::other)?].into_boxed_slice(); - blake3::Hasher::new_derive_key("golem fs-snapshot benchmark file content") - .update(&[generation]) - .update(path.as_os_str().as_bytes()) - .finalize_xof() - .fill(&mut content); - if kind == Content::Compressible { - content.chunks_mut(COMPRESSIBLE_BLOCK).for_each(|block| { - block - .iter_mut() - .skip(COMPRESSIBLE_BLOCK / 2) - .for_each(|byte| *byte = 0) - }); - } - Ok(content) -} - -fn write_file( - root: &Path, - relative: &Path, - generation: u8, - size: u64, - kind: Content, -) -> io::Result<()> { - let path = root.join(relative); - std::fs::write(&path, content(relative, generation, size, kind)?)?; - std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o644)) -} - -/// Makes each directory of the relative path below `root` that does not exist, with the -/// permission bits `0o755`. -fn make_directories(root: &Path, relative: &Path) -> io::Result<()> { - relative - .ancestors() - .filter(|ancestor| !ancestor.as_os_str().is_empty()) - .collect::>() - .iter() - .rev() - .map(|ancestor| root.join(ancestor)) - .filter(|directory| !directory.exists()) - .try_for_each(|directory| { - std::fs::create_dir(&directory)?; - std::fs::set_permissions(&directory, std::fs::Permissions::from_mode(0o755)) - }) -} - -fn generate_files(root: &Path, files: &Files) -> anyhow::Result<()> { - let count = files.layout.files(); - (0..count).try_for_each(|index| { - let relative = files.layout.path(index); - if let Some(parent) = relative.parent() { - make_directories(root, parent)?; - } - write_file( - root, - &relative, - 0, - file_size(index, count, files.bytes), - files.content, - ) - })?; - Ok(()) -} - -/// Whether the small change of a files tree replaces a file, or only adds one. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum Replace { - /// The change deletes a file, and adds a file with a new name in its directory. So the tree - /// keeps its number of objects. - Yes, - /// The change adds a file in the first directory. - No, -} - -/// Gives the file name of the round: the name, and for a round after the first, the name with -/// the round. -fn file_name_of_round(name: &str, round: u8) -> String { - if round == 1 { - name.to_string() - } else { - format!("{name}-{round}") - } -} - -fn change_files(root: &Path, files: &Files, round: u8) -> anyhow::Result { - let count = files.layout.files(); - let size = |index| file_size(index, count, files.bytes); - let step = (count / CHANGED_FILES).max(1); - let rewritten = (0..CHANGED_FILES.min(count)) - .map(|position| position * step) - .try_fold(0_u64, |written, index| { - write_file( - root, - &files.layout.path(index), - round, - size(index), - files.content, - ) - .map(|()| written + size(index)) - })?; - // The added file has the size of the file with the index. Round `k` deletes the `k`-th file - // from the end. - let (added_path, added_index, deleted) = match files.replace { - Replace::Yes => { - let deleted = count.saturating_sub(u64::from(round)); - let path = files.layout.path(deleted); - std::fs::remove_file(root.join(&path))?; - ( - path.with_file_name(file_name_of_round("replaced", round)), - deleted, - 1, - ) - } - Replace::No => ( - files - .layout - .path(0) - .with_file_name(file_name_of_round("added", round)), - 0, - 0, - ), - }; - let added = size(added_index); - write_file(root, &added_path, 0, added, files.content)?; - Ok(json!({ - "files_rewritten": CHANGED_FILES.min(count), - "files_added": 1, - "files_deleted": deleted, - "rows_updated": 0, - "bytes": rewritten + added, - })) -} - -async fn connect(path: &Path) -> anyhow::Result { - Ok(SqliteConnectOptions::new() - .filename(path) - .create_if_missing(true) - .journal_mode(SqliteJournalMode::Delete) - .synchronous(SqliteSynchronous::Normal) - .page_size(4096) - .connect() - .await?) -} - -/// Inserts rows of random payload until the database file has at least `bytes` bytes. -async fn generate_database(path: &Path, bytes: u64) -> anyhow::Result<()> { - let mut connection = connect(path).await?; - connection - .execute("CREATE TABLE rows (id INTEGER PRIMARY KEY, payload BLOB NOT NULL)") - .await?; - // Each row takes more than its payload, so this number of inserts is more than enough. - let inserts = bytes / (ROWS_PER_INSERT * ROW_PAYLOAD_BYTES) + 2; - let connection = futures::stream::iter(0..inserts) - .map(Ok::<_, anyhow::Error>) - .try_fold(connection, |mut connection, _| async move { - if std::fs::metadata(path)?.len() < bytes { - sqlx::query( - "WITH RECURSIVE counter(n) AS (SELECT 1 UNION ALL SELECT n + 1 FROM counter WHERE n < ?1) \ - INSERT INTO rows (payload) SELECT randomblob(?2) FROM counter", - ) - .bind(ROWS_PER_INSERT as i64) - .bind(ROW_PAYLOAD_BYTES as i64) - .execute(&mut connection) - .await?; - } - Ok(connection) - }) - .await?; - connection.close().await?; - Ok(()) -} - -/// Which rows a change of a database updates. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum SqliteChange { - /// Rows spread over the whole table. - Scattered, - /// Consecutive rows in the middle of the table. - Clustered, -} - -/// Updates the payload of [`CHANGED_ROWS`] rows of the database. -async fn change_database(path: &Path, pattern: SqliteChange) -> anyhow::Result { - let mut connection = connect(path).await?; - let last: i64 = sqlx::query_scalar("SELECT max(id) FROM rows") - .fetch_one(&mut connection) - .await?; - let rows = CHANGED_ROWS as i64; - let (first, step) = match pattern { - SqliteChange::Scattered => (1, (last / rows).max(1)), - SqliteChange::Clustered => ((last / 2).max(1), 1), - }; - let ids = (0..rows) - .map(|position| (first + position * step).to_string()) - .collect::>() - .join(","); - let updated = sqlx::query(&format!( - "UPDATE rows SET payload = randomblob({ROW_PAYLOAD_BYTES}) WHERE id IN ({ids})" - )) - .execute(&mut connection) - .await? - .rows_affected(); - connection.close().await?; - Ok(json!({ - "files_rewritten": 0, - "files_added": 0, - "files_deleted": 0, - "rows_updated": updated, - "bytes": updated * ROW_PAYLOAD_BYTES, - })) -} - -/// How the files of a copy of a tree were made. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub(super) struct CopyCounts { - /// The files that share their data with the source, through a reflink. - pub(super) reflinked: u64, - /// The files whose bytes were copied. - pub(super) copied: u64, -} - -impl CopyCounts { - pub(super) fn with(self, other: Self) -> Self { - Self { - reflinked: self.reflinked + other.reflinked, - copied: self.copied + other.copied, - } - } -} - -/// Whether a copy of a tree keeps the modification times of its entries. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) enum Times { - /// Each file and directory of the copy, and its root, gets the modification time of its - /// source, as the capture of an agent tree gives them. - Keep, - /// The copy does not keep the modification times. - Drop, -} - -/// Copies the tree `from` into the directory `to`, which must not exist, with the permission -/// bits of each entry, and with the modification times when `times` keeps them. A file is a -/// reflink of its source where the filesystem has reflinks (`FICLONE`, for example on XFS), and a -/// copy of its bytes where it does not. The copy does not sync the volume. -pub(super) fn copy_tree(from: &Path, to: &Path, times: Times) -> anyhow::Result { - std::fs::create_dir(to).with_context(|| format!("create the tree {}", to.display()))?; - let (counts, directories) = walk( - from, - (CopyCounts::default(), Vec::new()), - &mut |(counts, mut directories), path, metadata| { - let target = to.join(path.strip_prefix(from).map_err(io::Error::other)?); - if metadata.is_dir() { - std::fs::create_dir(&target)?; - std::fs::set_permissions(&target, metadata.permissions())?; - if times == Times::Keep { - directories.push((target, metadata.modified()?)); - } - Ok((counts, directories)) - } else if metadata.is_file() { - let reflinked = copy_file(path, &target, metadata)?; - if times == Times::Keep { - File::options() - .write(true) - .open(&target)? - .set_modified(metadata.modified()?)?; - } - Ok(( - counts.with(if reflinked { - CopyCounts { - reflinked: 1, - copied: 0, - } - } else { - CopyCounts { - reflinked: 0, - copied: 1, - } - }), - directories, - )) - } else { - Err(io::Error::other(format!( - "the tree has an entry that is not a file or a directory: {}", - path.display() - ))) - } - }, - )?; - if times == Times::Keep { - // A directory gets its time after its entries, which change it, and the root last. - directories - .iter() - .rev() - .map(|(directory, modified)| (directory.as_path(), *modified)) - .chain(std::iter::once((to, std::fs::metadata(from)?.modified()?))) - .try_for_each(|(directory, modified)| File::open(directory)?.set_modified(modified))?; - } - Ok(counts) -} - -/// Copies the file `from` to the new file `to`, and tells whether the copy is a reflink. -/// -/// When the reflink fails, the filesystem cannot share the data of the files, so the copy of the -/// bytes after it does not share them either. -fn copy_file(from: &Path, to: &Path, metadata: &Metadata) -> io::Result { - let source = File::open(from)?; - let target = File::create_new(to)?; - match rustix::fs::ioctl_ficlone(&target, &source) { - Ok(()) => { - target.set_permissions(metadata.permissions())?; - Ok(true) - } - Err(_) => { - drop(target); - std::fs::copy(from, to)?; - Ok(false) - } - } -} - -/// Writes each change of the filesystem of `root` to the volume, and removes the pages of each -/// file below `root` from the page cache. -pub(super) fn settle(root: &Path) -> anyhow::Result<()> { - rustix::fs::syncfs(File::open(root)?)?; - walk(root, (), &mut |(), path, metadata| { - if metadata.is_file() { - rustix::fs::fadvise(File::open(path)?, 0, None, rustix::fs::Advice::DontNeed)?; - } - Ok(()) - })?; - Ok(()) -} - -fn count(root: &Path) -> io::Result { - walk(root, TreeCounts::default(), &mut |counts, _, metadata| { - Ok(if metadata.is_dir() { - TreeCounts { - directories: counts.directories + 1, - ..counts - } - } else if metadata.is_file() { - TreeCounts { - files: counts.files + 1, - bytes: counts.bytes + metadata.len(), - ..counts - } - } else { - counts - }) - }) -} - -/// Gives the hash of the tree below `root` and its counts, and reads the tree on a blocking thread. -pub(super) async fn hash(root: &Path) -> anyhow::Result<(Box, TreeCounts)> { - let root = root.to_path_buf(); - Ok(tokio::task::spawn_blocking(move || tree_hash(&root)).await??) -} - -/// Gives the hash of the tree below `root` and its counts. -/// -/// The hash covers the path, the kind, the permission bits and the modification time of each -/// entry, the size and the content of each file, and the target of each symlink. It does not -/// cover the owner or the metadata of `root`. -pub(super) fn tree_hash(root: &Path) -> io::Result<(Box, TreeCounts)> { - let hasher = walk( - root, - blake3::Hasher::new(), - &mut |mut hasher, path, metadata| { - let relative = path.strip_prefix(root).map_err(io::Error::other)?; - let name = relative.as_os_str().as_bytes(); - hasher.update(&(name.len() as u64).to_le_bytes()); - hasher.update(name); - hasher.update(&metadata.mode().to_le_bytes()); - hasher.update(&metadata.mtime().to_le_bytes()); - hasher.update(&metadata.mtime_nsec().to_le_bytes()); - if metadata.is_file() { - hasher.update(b"f"); - hasher.update(&metadata.len().to_le_bytes()); - let mut content = blake3::Hasher::new(); - content.update_reader(File::open(path)?)?; - hasher.update(content.finalize().as_bytes()); - } else if metadata.is_symlink() { - let target = std::fs::read_link(path)?; - hasher.update(b"l"); - hasher.update(&(target.as_os_str().len() as u64).to_le_bytes()); - hasher.update(target.as_os_str().as_bytes()); - } else if metadata.is_dir() { - hasher.update(b"d"); - } else { - hasher.update(b"o"); - } - Ok(hasher) - }, - )?; - Ok((hasher.finalize().to_hex().as_str().into(), count(root)?)) -} - -/// Visits each entry below `directory`, in the order of the names, parents before children, and -/// folds the visits into one value. Symlinks are not followed. -fn walk( - directory: &Path, - initial: T, - visit: &mut impl FnMut(T, &Path, &Metadata) -> io::Result, -) -> io::Result { - let mut names = std::fs::read_dir(directory)? - .map(|entry| entry.map(|entry| entry.file_name())) - .collect::>>()?; - names.sort(); - names.into_iter().try_fold(initial, |value, name| { - let path = directory.join(name); - let metadata = std::fs::symlink_metadata(&path)?; - let value = visit(value, &path, &metadata)?; - if metadata.is_dir() { - walk(&path, value, visit) - } else { - Ok(value) - } - }) -} - -#[cfg(test)] -mod tests { - use super::{ - COMPRESSIBLE_1G, COMPRESSIBLE_BLOCK, Content, DATABASE, FILES_1G, FILES_128M, FILES_TINY, - Layout, MIB, MODULES_128M, SQLITE_TINY, Times, TreeCounts, TreeShape, TreeSpec, change, - change_clustered, change_round, connect, content, copy_tree, file_path, file_size, - generate, limit_layout, tree_hash, walk, - }; - use pretty_assertions::assert_eq; - use sqlx::Connection; - use std::os::unix::fs::{MetadataExt, PermissionsExt}; - use std::path::{Path, PathBuf}; - use std::time::{Duration, SystemTime}; - use test_r::test; - - /// A tree at the object limit of 128 objects: 125 files in 2 directories, and the root. - const OBJECTS_TINY: TreeSpec = TreeSpec { - name: "objects-tiny", - shape: TreeShape::ObjectLimit { - objects: 128, - bytes: MIB, - }, - content: Content::Incompressible, - }; - - fn hash(root: &Path) -> Box { - tree_hash(root).unwrap().0 - } - - /// Gives the path, the permission bits and the content of each entry below `root`, in the - /// order of the paths. A directory has no content. - fn contents(root: &Path) -> Vec<(PathBuf, u32, Vec)> { - walk(root, Vec::new(), &mut |mut entries, path, metadata| { - let content = if metadata.is_file() { - std::fs::read(path)? - } else { - Vec::new() - }; - entries.push(( - path.strip_prefix(root).unwrap().to_path_buf(), - metadata.mode(), - content, - )); - Ok(entries) - }) - .unwrap() - } - - #[test] - fn the_limit_layout_counts_the_root_and_fills_each_directory() { - let layout = |objects| { - let (files, directories) = limit_layout(objects); - ( - files, - directories, - files + directories + 1, - file_path(files - 1, files, directories), - ) - }; - - assert_eq!( - [layout(8_192), layout(32_768), layout(128)], - [ - (8_109, 82, 8_192, Path::new("d081/f08108").into()), - (32_442, 325, 32_768, Path::new("d324/f32441").into()), - (125, 2, 128, Path::new("d001/f00124").into()), - ] - ); - } - - #[test] - async fn a_tree_at_the_object_limit_keeps_its_number_of_objects_after_its_change() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - - let counts = generate(&OBJECTS_TINY, &root).await.unwrap(); - let before = hash(&root); - let changed = change(&OBJECTS_TINY, &root).await.unwrap(); - let (after, after_counts) = tree_hash(&root).unwrap(); - - assert_eq!( - ( - counts, - after_counts, - [ - &changed["files_rewritten"], - &changed["files_added"], - &changed["files_deleted"] - ] - .map(|value| value.as_u64()), - root.join("d001/f00124").exists(), - root.join("d001/replaced").exists(), - before != after, - ), - ( - TreeCounts { - files: 125, - directories: 2, - bytes: MIB, - }, - TreeCounts { - files: 125, - directories: 2, - bytes: MIB, - }, - [Some(10), Some(1), Some(1)], - false, - true, - true, - ) - ); - } - - #[test] - async fn a_copy_of_a_tree_has_its_entries_permissions_and_content() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - let copy = work.path().join("copy"); - generate(&FILES_TINY, &root).await.unwrap(); - std::fs::set_permissions( - root.join("d000/f00000"), - std::fs::Permissions::from_mode(0o600), - ) - .unwrap(); - std::fs::set_permissions(root.join("d001"), std::fs::Permissions::from_mode(0o700)) - .unwrap(); - - let counts = copy_tree(&root, ©, Times::Drop).unwrap(); - - assert_eq!( - (counts.reflinked + counts.copied, contents(©)), - (100, contents(&root)) - ); - } - - #[test] - fn the_file_sizes_add_up_and_the_files_fill_the_directories() { - assert_eq!( - ( - (0..7).map(|index| file_size(index, 7, 100)).sum::(), - (0..7) - .map(|index| file_size(index, 7, 100)) - .collect::>(), - file_path(0, 100, 10), - file_path(19, 100, 10), - file_path(99, 100, 10), - ), - ( - 100, - vec![15, 15, 14, 14, 14, 14, 14], - Path::new("d000/f00000").into(), - Path::new("d001/f00019").into(), - Path::new("d009/f00099").into(), - ) - ); - } - - #[test] - async fn a_files_tree_has_its_files_directories_and_bytes_and_a_change_changes_it() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - - let counts = generate(&FILES_TINY, &root).await.unwrap(); - let before = hash(&root); - let changed = change(&FILES_TINY, &root).await.unwrap(); - let (after, after_counts) = tree_hash(&root).unwrap(); - - assert_eq!( - ( - counts, - changed["files_rewritten"].as_u64(), - changed["files_added"].as_u64(), - changed["files_deleted"].as_u64(), - after_counts.files, - before != after, - ), - ( - TreeCounts { - files: 100, - directories: 10, - bytes: 1024 * 1024, - }, - Some(10), - Some(1), - Some(0), - 101, - true, - ) - ); - } - - #[test] - async fn a_sqlite_tree_has_one_database_of_at_least_its_size() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - - let counts = generate(&SQLITE_TINY, &root).await.unwrap(); - let before = hash(&root); - let changed = change(&SQLITE_TINY, &root).await.unwrap(); - - assert_eq!( - ( - counts.files, - counts.bytes >= 4 * 1024 * 1024, - counts.bytes < 6 * 1024 * 1024, - changed["rows_updated"].as_u64(), - before != hash(&root), - ), - (1, true, true, Some(100), true) - ); - } - - #[test] - fn the_tree_hash_changes_with_each_kept_attribute() { - let work = tempfile::tempdir().unwrap(); - let root = work.path(); - let time = SystemTime::UNIX_EPOCH + Duration::new(1_700_000_000, 5); - let reset = |root: &Path| { - let _ = std::fs::remove_dir_all(root.join("dir")); - let _ = std::fs::remove_file(root.join("file")); - let _ = std::fs::remove_file(root.join("other")); - let _ = std::fs::remove_file(root.join("link")); - std::fs::create_dir(root.join("dir")).unwrap(); - std::fs::write(root.join("file"), b"content").unwrap(); - std::fs::set_permissions(root.join("file"), std::fs::Permissions::from_mode(0o644)) - .unwrap(); - std::os::unix::fs::symlink("file", root.join("link")).unwrap(); - std::fs::File::options() - .write(true) - .open(root.join("file")) - .unwrap() - .set_modified(time) - .unwrap(); - fs_set_times::set_times( - root.join("dir"), - None, - Some(fs_set_times::SystemTimeSpec::Absolute(time)), - ) - .unwrap(); - fs_set_times::set_symlink_times( - root.join("link"), - None, - Some(fs_set_times::SystemTimeSpec::Absolute(time)), - ) - .unwrap(); - }; - let changed = |change: &dyn Fn(&Path)| { - reset(root); - change(root); - hash(root) - }; - reset(root); - let original = hash(root); - - let hashes = [ - changed(&|_| {}), - changed(&|root| { - std::fs::write(root.join("file"), b"CONTENT").unwrap(); - std::fs::File::options() - .write(true) - .open(root.join("file")) - .unwrap() - .set_modified(time) - .unwrap(); - }), - changed(&|root| { - std::fs::set_permissions(root.join("file"), std::fs::Permissions::from_mode(0o600)) - .unwrap() - }), - changed(&|root| { - std::fs::File::options() - .write(true) - .open(root.join("file")) - .unwrap() - .set_modified(time + Duration::from_nanos(1)) - .unwrap() - }), - changed(&|root| { - std::fs::remove_file(root.join("link")).unwrap(); - std::os::unix::fs::symlink("elsewhere", root.join("link")).unwrap(); - fs_set_times::set_symlink_times( - root.join("link"), - None, - Some(fs_set_times::SystemTimeSpec::Absolute(time)), - ) - .unwrap(); - }), - changed(&|root| std::fs::rename(root.join("file"), root.join("other")).unwrap()), - changed(&|root| { - std::fs::create_dir(root.join("dir/empty")).unwrap(); - [root.join("dir/empty"), root.join("dir")] - .iter() - .for_each(|directory| { - fs_set_times::set_times( - directory, - None, - Some(fs_set_times::SystemTimeSpec::Absolute(time)), - ) - .unwrap() - }); - }), - ]; - - assert_eq!( - hashes - .iter() - .map(|hash| *hash == original) - .collect::>(), - vec![true, false, false, false, false, false, false] - ); - } - - #[test] - async fn a_copy_that_keeps_the_times_has_the_hash_of_its_source_and_new_inodes() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - let kept = work.path().join("kept"); - let dropped = work.path().join("dropped"); - generate(&FILES_TINY, &root).await.unwrap(); - let old = SystemTime::UNIX_EPOCH + Duration::from_secs(1_700_000_000); - walk(&root, (), &mut |(), path, _| { - std::fs::File::open(path).and_then(|file| file.set_modified(old)) - }) - .unwrap(); - std::fs::File::open(&root) - .unwrap() - .set_modified(old) - .unwrap(); - let inode = |root: &Path| std::fs::metadata(root.join("d000/f00000")).unwrap().ino(); - - let counts = copy_tree(&root, &kept, Times::Keep).unwrap(); - copy_tree(&root, &dropped, Times::Drop).unwrap(); - - assert_eq!( - ( - counts.reflinked + counts.copied, - hash(&kept) == hash(&root), - hash(&dropped) == hash(&root), - std::fs::metadata(&kept).unwrap().modified().unwrap(), - inode(&kept) == inode(&root), - ), - (100, true, false, old, false) - ); - } - - #[test] - async fn each_round_of_a_change_writes_other_content_and_keeps_the_object_limit() { - let work = tempfile::tempdir().unwrap(); - let files = work.path().join("files"); - let objects = work.path().join("objects"); - generate(&FILES_TINY, &files).await.unwrap(); - generate(&OBJECTS_TINY, &objects).await.unwrap(); - - let rounds = futures::future::join_all( - [(&FILES_TINY, &files), (&OBJECTS_TINY, &objects)].map(|(spec, root)| async move { - let first = change_round(spec, root, 1).await.unwrap(); - let after_first = tree_hash(root).unwrap(); - change_round(spec, root, 2).await.unwrap(); - let after_second = tree_hash(root).unwrap(); - ( - first["files_rewritten"].as_u64(), - first["bytes"].as_u64(), - after_first.0 != after_second.0, - after_first.1.files, - after_second.1.files, - ) - }), - ) - .await; - - assert_eq!( - ( - rounds, - files.join("d000/added-2").exists(), - objects.join("d001/replaced-2").exists(), - objects.join("d001/f00123").exists(), - ), - ( - // FILES_TINY: the files 0, 10, ..., 70 have 10,486 bytes and the files 80 and 90 - // have 10,485, and the added file has the 10,486 bytes of the file 0. - // OBJECTS_TINY: the files 0, 12, ..., 72 have 8,389 bytes, the files 84, 96 and - // 108 have 8,388, and the added file has the 8,388 bytes of the file 124. - vec![ - ( - Some(10), - Some(8 * 10_486 + 2 * 10_485 + 10_486), - true, - 101, - 102 - ), - ( - Some(10), - Some(7 * 8_389 + 3 * 8_388 + 8_388), - true, - 125, - 125 - ), - ], - true, - true, - false, - ) - ); - } - - #[test] - async fn a_modules_tree_has_its_files_in_three_levels_of_small_directories() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - let spec = TreeSpec { - name: "modules-small", - shape: TreeShape::Modules { - files: 10_000, - bytes: 10_000, - }, - content: Content::Incompressible, - }; - let layout = Layout::Modules { files: 10_000 }; - - let counts = generate(&spec, &root).await.unwrap(); - - assert_eq!( - (counts, layout.path(0), layout.path(4), layout.path(9_999),), - ( - TreeCounts { - files: 10_000, - directories: 2_775, - bytes: 10_000, - }, - Path::new("p00/s0/l0/f00000").into(), - Path::new("p00/s0/l1/f00004").into(), - Path::new("p24/s9/l9/f09999").into(), - ) - ); - } - - #[test] - fn compressible_content_has_a_zero_second_half_in_each_block() { - let compressible = content(Path::new("f"), 0, 1_000, Content::Compressible).unwrap(); - let incompressible = content(Path::new("f"), 0, 1_000, Content::Incompressible).unwrap(); - let zero_halves = |content: &[u8]| { - content.chunks(COMPRESSIBLE_BLOCK).all(|block| { - block - .iter() - .skip(COMPRESSIBLE_BLOCK / 2) - .all(|byte| *byte == 0) - }) - }; - - assert_eq!( - ( - compressible.len(), - zero_halves(&compressible), - zero_halves(&incompressible), - compressible - .chunks(COMPRESSIBLE_BLOCK) - .zip(incompressible.chunks(COMPRESSIBLE_BLOCK)) - .all(|(left, right)| left[..COMPRESSIBLE_BLOCK / 2] - == right[..COMPRESSIBLE_BLOCK / 2]), - ), - (1_000, true, false, true) - ); - } - - #[test] - async fn a_scattered_and_a_clustered_change_update_the_rows_of_their_pattern() { - let work = tempfile::tempdir().unwrap(); - let root = work.path().join("tree"); - let files = work.path().join("files"); - generate(&SQLITE_TINY, &root).await.unwrap(); - generate(&FILES_TINY, &files).await.unwrap(); - let payloads = async |root: &Path| { - let mut connection = connect(&root.join(DATABASE)).await.unwrap(); - let rows: Vec<(i64, Vec)> = - sqlx::query_as("SELECT id, payload FROM rows ORDER BY id") - .fetch_all(&mut connection) - .await - .unwrap(); - connection.close().await.unwrap(); - rows - }; - let updated = |before: &[(i64, Vec)], after: &[(i64, Vec)]| { - before - .iter() - .zip(after) - .filter(|(old, new)| old.1 != new.1) - .map(|(old, _)| old.0) - .collect::>() - }; - let before = payloads(&root).await; - let last = before.last().unwrap().0; - - let scattered = change(&SQLITE_TINY, &root).await.unwrap(); - let after_scattered = payloads(&root).await; - let clustered = change_clustered(&SQLITE_TINY, &root).await.unwrap(); - let after_clustered = payloads(&root).await; - let refused = change_clustered(&FILES_TINY, &files).await; - - assert_eq!( - ( - scattered["rows_updated"].as_u64(), - updated(&before, &after_scattered), - clustered["rows_updated"].as_u64(), - updated(&after_scattered, &after_clustered), - refused.is_err() - ), - ( - Some(100), - (0..100) - .map(|row| 1 + row * (last / 100)) - .collect::>(), - Some(100), - (last / 2..last / 2 + 100).collect::>(), - true - ) - ); - } - - #[test] - fn the_later_trees_have_the_files_and_bytes_of_the_trees_that_they_follow() { - let (modules_files, modules_bytes) = match MODULES_128M.shape { - TreeShape::Modules { files, bytes } => (files, bytes), - _ => (0, 0), - }; - let (files_files, files_bytes) = match FILES_128M.shape { - TreeShape::Files { files, bytes, .. } => (files, bytes), - _ => (1, 1), - }; - - assert_eq!( - ( - (modules_files, modules_bytes), - COMPRESSIBLE_1G.shape, - COMPRESSIBLE_1G.content, - FILES_1G.content, - ), - ( - (files_files, files_bytes), - FILES_1G.shape, - Content::Compressible, - Content::Incompressible, - ) - ); - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/benchmark/volume.rs b/golem-worker-executor/src/filesystem_snapshot/benchmark/volume.rs deleted file mode 100644 index 2f5c4f0d09..0000000000 --- a/golem-worker-executor/src/filesystem_snapshot/benchmark/volume.rs +++ /dev/null @@ -1,555 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! Records the filesystem of the benchmark volume, without privileges. -//! -//! The record holds the filesystem type, the mount, the project quota flags of the mount, the -//! answer of `quotactl_fd` with `Q_XGETQSTATV`, and the XFS geometry that `xfs_info` shows. The -//! verdict uses only the filesystem type and the mount options. - -use serde_json::{Value, json}; -use std::fs::File; -use std::os::fd::AsRawFd; -use std::path::{Path, PathBuf}; - -const XFS_SUPER_MAGIC: u64 = 0x5846_5342; - -/// The quota type and the command of `quotactl_fd`, as the executor uses them to check the -/// project quota state at its start. -const XQM_PRJQUOTA: u32 = 2; -const Q_XGETQSTATV: u32 = (b'X' as u32) << 8 | 8; -const FS_QSTATV_VERSION1: i8 = 1; -const FS_QUOTA_PDQ_ACCT: u16 = 1 << 4; -const FS_QUOTA_PDQ_ENFD: u16 = 1 << 5; - -/// The size of `struct xfs_fsop_geom` of `XFS_IOC_FSGEOMETRY`, and of `struct xfs_fsop_geom_v4` -/// of `XFS_IOC_FSGEOMETRY_V4`. -const GEOMETRY_BYTES: usize = 256; -const GEOMETRY_V4_BYTES: usize = 112; - -/// The names of the bits of `xfs_fsop_geom.flags`, `XFS_FSOP_GEOM_FLAGS_*`. -const GEOMETRY_FLAGS: [(u32, &str); 26] = [ - (1 << 0, "attr"), - (1 << 1, "nlink"), - (1 << 2, "quota"), - (1 << 3, "ialign"), - (1 << 4, "dalign"), - (1 << 5, "shared"), - (1 << 6, "extflg"), - (1 << 7, "dirv2"), - (1 << 8, "logv2"), - (1 << 9, "sector"), - (1 << 10, "attr2"), - (1 << 11, "projid32"), - (1 << 12, "dirv2ci"), - (1 << 14, "lazysb"), - (1 << 15, "v5sb"), - (1 << 16, "ftype"), - (1 << 17, "finobt"), - (1 << 18, "spinodes"), - (1 << 19, "rmapbt"), - (1 << 20, "reflink"), - (1 << 21, "bigtime"), - (1 << 22, "inobtcnt"), - (1 << 23, "nrext64"), - (1 << 24, "exchange_range"), - (1 << 25, "parent"), - (1 << 26, "metadir"), -]; - -/// One line of `/proc/self/mountinfo`. -#[derive(Clone, Debug, PartialEq, Eq)] -pub(super) struct Mount { - pub(super) mount_point: Box, - pub(super) filesystem_type: Box, - pub(super) source: Box, - pub(super) mount_options: Box, - pub(super) super_options: Box, -} - -/// The project quota flags of a mount. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) struct ProjectQuota { - pub(super) accounting: bool, - pub(super) enforcement: bool, -} - -/// Records the volume that holds `path`. -pub(super) fn check(path: &Path) -> Value { - let path = std::fs::canonicalize(path).unwrap_or_else(|_| path.to_path_buf()); - let magic = rustix::fs::statfs(&path).map(|statfs| statfs.f_type as u64); - let mount = std::fs::read_to_string("/proc/self/mountinfo") - .ok() - .and_then(|text| mount_of(&text, &path)); - let quota = mount - .as_ref() - .map(|mount| project_quota(&mount.super_options)); - let directory = File::open(&path); - json!({ - "path": path.display().to_string(), - "filesystem_type": magic.as_ref().map(|magic| filesystem_name(*magic)).unwrap_or("unknown"), - "filesystem_magic": magic.as_ref().map(|magic| format!("{magic:#x}")).unwrap_or_else(|error| error.to_string()), - "mount": mount.as_ref().map(|mount| json!({ - "mount_point": mount.mount_point.display().to_string(), - "filesystem_type": mount.filesystem_type, - "source": mount.source, - "mount_options": mount.mount_options, - "super_options": mount.super_options, - })), - "project_quota": quota.map(|quota| json!({ - "accounting": quota.accounting, - "enforcement": quota.enforcement, - "source": "mount options", - })), - "quotactl": match &directory { - Ok(directory) => quota_state(directory), - Err(error) => json!({ "error": error.to_string() }), - }, - "xfs_geometry": match &directory { - Ok(directory) => geometry(directory), - Err(error) => json!({ "error": error.to_string() }), - }, - "check": verdict(magic.ok(), quota), - }) -} - -fn filesystem_name(magic: u64) -> &'static str { - match magic { - XFS_SUPER_MAGIC => "xfs", - 0xef53 => "ext4", - 0x0102_1994 => "tmpfs", - 0x9123_683e => "btrfs", - 0x794c_7630 => "overlay", - _ => "other", - } -} - -/// Gives the mount whose mount point is the longest prefix of `path`, from the text of -/// `/proc/self/mountinfo`. -pub(super) fn mount_of(mountinfo: &str, path: &Path) -> Option { - mountinfo - .lines() - .filter_map(parse_mount) - .filter(|mount| path.starts_with(&mount.mount_point)) - .max_by_key(|mount| mount.mount_point.components().count()) -} - -fn parse_mount(line: &str) -> Option { - let (mount, filesystem) = line.split_once(" - ")?; - let mount = mount.split(' ').collect::>(); - let filesystem = filesystem.split(' ').collect::>(); - Some(Mount { - mount_point: PathBuf::from(unescape(mount.get(4)?)).into_boxed_path(), - mount_options: (*mount.get(5)?).into(), - filesystem_type: (*filesystem.first()?).into(), - source: (*filesystem.get(1)?).into(), - super_options: (*filesystem.get(2)?).into(), - }) -} - -/// Replaces the octal escapes of mountinfo, such as `\040` for a space. -fn unescape(text: &str) -> String { - let bytes = text.as_bytes(); - let (decoded, _) = bytes.iter().enumerate().fold( - (Vec::with_capacity(bytes.len()), 0_usize), - |(mut decoded, skip), (index, byte)| { - if skip > 0 { - return (decoded, skip - 1); - } - let escape = bytes - .get(index + 1..index + 4) - .filter(|digits| { - *byte == b'\\' && digits.iter().all(|digit| (b'0'..=b'7').contains(digit)) - }) - .and_then(|digits| std::str::from_utf8(digits).ok()) - .and_then(|digits| u8::from_str_radix(digits, 8).ok()); - match escape { - Some(value) => { - decoded.push(value); - (decoded, 3) - } - None => { - decoded.push(*byte); - (decoded, 0) - } - } - }, - ); - String::from_utf8_lossy(&decoded).into_owned() -} - -/// Gives the project quota flags from the super options of an XFS mount. The kernel shows -/// `prjquota` when it counts and enforces project quotas, and `pqnoenforce` when it only counts -/// them. -pub(super) fn project_quota(super_options: &str) -> ProjectQuota { - super_options.split(',').fold( - ProjectQuota { - accounting: false, - enforcement: false, - }, - |quota, option| match option { - "prjquota" | "pquota" => ProjectQuota { - accounting: true, - enforcement: true, - }, - "pqnoenforce" => ProjectQuota { - accounting: true, - ..quota - }, - _ => quota, - }, - ) -} - -/// Tells whether the volume is XFS with project quota accounting and enforcement, and why not. -pub(super) fn verdict(magic: Option, quota: Option) -> Value { - let reasons = [ - (magic != Some(XFS_SUPER_MAGIC)).then(|| { - format!( - "the filesystem type is {}, not xfs", - magic.map_or("unknown", filesystem_name) - ) - }), - match quota { - None => Some("the mount of the volume is not in /proc/self/mountinfo".to_string()), - Some(ProjectQuota { - accounting: false, .. - }) => Some("the mount has no project quota".to_string()), - Some(ProjectQuota { - enforcement: false, .. - }) => Some("the mount counts project quotas but does not enforce them".to_string()), - Some(_) => None, - }, - ] - .into_iter() - .flatten() - .collect::>(); - if reasons.is_empty() { - json!({ "status": "ok" }) - } else { - json!({ "status": "mismatch", "reasons": reasons }) - } -} - -/// The `fs_quota_statv` record of `Q_XGETQSTATV`, as in `linux/dqblk_xfs.h`. -#[repr(C)] -#[derive(Clone, Copy, Default)] -struct FsQuotaFileStatV { - qfs_ino: u64, - qfs_nblks: u64, - qfs_nextents: u32, - qfs_pad: u32, -} - -#[repr(C)] -#[derive(Clone, Copy, Default)] -struct FsQuotaStatV { - qs_version: i8, - qs_pad1: u8, - qs_flags: u16, - qs_incoredqs: u32, - qs_uquota: FsQuotaFileStatV, - qs_gquota: FsQuotaFileStatV, - qs_pquota: FsQuotaFileStatV, - qs_btimelimit: i32, - qs_itimelimit: i32, - qs_rtbtimelimit: i32, - qs_bwarnlimit: u16, - qs_iwarnlimit: u16, - qs_rtbwarnlimit: u16, - qs_pad3: u16, - qs_pad4: u32, - qs_pad2: [u64; 7], -} - -/// Asks the kernel for the project quota state of the filesystem of the directory. -fn quota_state(directory: &File) -> Value { - let mut state = FsQuotaStatV { - qs_version: FS_QSTATV_VERSION1, - ..FsQuotaStatV::default() - }; - // SAFETY: `Q_XGETQSTATV` writes one `fs_quota_statv` record, which `state` holds for the - // duration of the call. - let result = unsafe { - libc::syscall( - libc::SYS_quotactl_fd, - directory.as_raw_fd(), - (Q_XGETQSTATV << 8) | XQM_PRJQUOTA, - 0, - std::ptr::from_mut(&mut state).cast::(), - ) - }; - if result == -1 { - let error = std::io::Error::last_os_error(); - json!({ "error": { "errno": error.raw_os_error(), "message": error.to_string() } }) - } else { - json!({ "ok": { - "flags": format!("{:#x}", state.qs_flags), - "accounting": state.qs_flags & FS_QUOTA_PDQ_ACCT != 0, - "enforcement": state.qs_flags & FS_QUOTA_PDQ_ENFD != 0, - } }) - } -} - -/// Asks XFS for the geometry of the filesystem of the directory. -fn geometry(directory: &File) -> Value { - // SAFETY: The opcodes and the sizes come from `struct xfs_fsop_geom` and - // `struct xfs_fsop_geom_v4` of `xfs_fs.h`, and the kernel writes at most that many bytes. - let full = unsafe { - rustix::ioctl::ioctl( - directory, - rustix::ioctl::Getter::< - { rustix::ioctl::opcode::read::<[u8; GEOMETRY_BYTES]>(b'X', 126) }, - [u8; GEOMETRY_BYTES], - >::new(), - ) - }; - let result = full.map(|bytes| bytes.to_vec()).or_else(|error| { - if error == rustix::io::Errno::NOTTY { - // SAFETY: As above, for the older record. - unsafe { - rustix::ioctl::ioctl( - directory, - rustix::ioctl::Getter::< - { rustix::ioctl::opcode::read::<[u8; GEOMETRY_V4_BYTES]>(b'X', 124) }, - [u8; GEOMETRY_V4_BYTES], - >::new(), - ) - } - .map(|bytes| bytes.to_vec()) - } else { - Err(error) - } - }); - match result { - Ok(bytes) => decode_geometry(&bytes) - .unwrap_or_else(|| json!({ "error": "the geometry record is too short" })), - Err(error) => { - json!({ "error": { "errno": error.raw_os_error(), "message": error.to_string() } }) - } - } -} - -/// Reads the fields of `struct xfs_fsop_geom` that `xfs_info` shows. -pub(super) fn decode_geometry(bytes: &[u8]) -> Option { - let u32_at = |offset: usize| { - bytes - .get(offset..offset + 4) - .and_then(|field| field.try_into().ok()) - .map(u32::from_ne_bytes) - }; - let u64_at = |offset: usize| { - bytes - .get(offset..offset + 8) - .and_then(|field| field.try_into().ok()) - .map(u64::from_ne_bytes) - }; - let flags = u32_at(92)?; - Some(json!({ - "blocksize": u32_at(0)?, - "rtextsize": u32_at(4)?, - "agblocks": u32_at(8)?, - "agcount": u32_at(12)?, - "logblocks": u32_at(16)?, - "sectsize": u32_at(20)?, - "inodesize": u32_at(24)?, - "imaxpct": u32_at(28)?, - "datablocks": u64_at(32)?, - "rtblocks": u64_at(40)?, - "rtextents": u64_at(48)?, - "logstart": u64_at(56)?, - "sunit": u32_at(80)?, - "swidth": u32_at(84)?, - "version": u32_at(88)?, - "flags": format!("{flags:#x}"), - "flag_names": GEOMETRY_FLAGS - .iter() - .filter(|(bit, _)| flags & bit != 0) - .map(|(_, name)| *name) - .collect::>(), - "logsectsize": u32_at(96)?, - "rtsectsize": u32_at(100)?, - "dirblocksize": u32_at(104)?, - "logsunit": u32_at(108)?, - })) -} - -#[cfg(test)] -mod tests { - use super::{ - ProjectQuota, XFS_SUPER_MAGIC, check, decode_geometry, mount_of, project_quota, verdict, - }; - use pretty_assertions::assert_eq; - use serde_json::json; - use std::path::Path; - use test_r::test; - - const MOUNTINFO: &str = "\ -22 1 259:1 / / rw,relatime shared:1 - overlay overlay rw,lowerdir=/a,upperdir=/b -30 22 259:2 / /data rw,relatime shared:2 - xfs /dev/nvme1n1 rw,attr2,inode64,logbufs=8,logbsize=32k,prjquota -31 22 259:3 / /data2 rw,relatime - xfs /dev/nvme2n1 rw,attr2,inode64,pqnoenforce -32 22 259:4 / /with\\040space rw - ext4 /dev/sda1 rw -33 30 0:5 / /data/proc rw - proc proc rw -"; - - #[test] - fn the_mount_of_a_path_is_the_one_with_the_longest_mount_point() { - let mount_point = |path: &str| { - mount_of(MOUNTINFO, Path::new(path)).map(|mount| { - ( - mount.mount_point.display().to_string(), - mount.filesystem_type.to_string(), - ) - }) - }; - - assert_eq!( - ( - mount_point("/data/tree"), - mount_point("/data"), - mount_point("/data2/x"), - mount_point("/database"), - mount_point("/with space/x"), - mount_point("/data/proc/1"), - ), - ( - Some(("/data".to_string(), "xfs".to_string())), - Some(("/data".to_string(), "xfs".to_string())), - Some(("/data2".to_string(), "xfs".to_string())), - Some(("/".to_string(), "overlay".to_string())), - Some(("/with space".to_string(), "ext4".to_string())), - Some(("/data/proc".to_string(), "proc".to_string())), - ) - ); - } - - #[test] - fn the_super_options_give_the_project_quota_flags() { - assert_eq!( - [ - "rw,attr2,inode64,prjquota", - "rw,attr2,pqnoenforce", - "rw,attr2,inode64", - "rw,usrquota", - ] - .map(project_quota), - [ - ProjectQuota { - accounting: true, - enforcement: true - }, - ProjectQuota { - accounting: true, - enforcement: false - }, - ProjectQuota { - accounting: false, - enforcement: false - }, - ProjectQuota { - accounting: false, - enforcement: false - }, - ] - ); - } - - #[test] - fn the_verdict_accepts_only_xfs_that_counts_and_enforces_project_quotas() { - let enforced = ProjectQuota { - accounting: true, - enforcement: true, - }; - let counted = ProjectQuota { - accounting: true, - enforcement: false, - }; - - assert_eq!( - [ - verdict(Some(XFS_SUPER_MAGIC), Some(enforced)), - verdict(Some(XFS_SUPER_MAGIC), Some(counted)), - verdict(Some(0xef53), Some(enforced)), - verdict(None, None), - ] - .map(|verdict| verdict["status"].as_str().map(str::to_string)), - [ - Some("ok".to_string()), - Some("mismatch".to_string()), - Some("mismatch".to_string()), - Some("mismatch".to_string()), - ] - ); - } - - #[test] - fn the_geometry_gives_the_fields_that_xfs_info_shows() { - let bytes = [ - (0, 4096_u64, 4), - (12, 16, 4), - (32, 1_000_000, 8), - (80, 8, 4), - (92, (1 << 20) | (1 << 15) | (1 << 21), 4), - (104, 4096, 4), - ] - .iter() - .fold(vec![0_u8; 256], |mut bytes, (offset, value, size)| { - bytes[*offset..offset + size].copy_from_slice(&value.to_ne_bytes()[..*size]); - bytes - }); - - let geometry = decode_geometry(&bytes).unwrap(); - - assert_eq!( - ( - &geometry["blocksize"], - &geometry["agcount"], - &geometry["datablocks"], - &geometry["sunit"], - &geometry["flag_names"], - &geometry["dirblocksize"], - decode_geometry(&bytes[..100]).is_none(), - ), - ( - &json!(4096), - &json!(16), - &json!(1_000_000), - &json!(8), - &json!(["v5sb", "reflink", "bigtime"]), - &json!(4096), - true, - ) - ); - } - - #[test] - fn a_check_of_a_directory_records_each_part() { - let directory = tempfile::tempdir().unwrap(); - - let record = check(directory.path()); - - assert_eq!( - [ - "filesystem_type", - "mount", - "project_quota", - "quotactl", - "xfs_geometry", - "check" - ] - .map(|key| record.get(key).is_some()), - [true; 6] - ); - } -} diff --git a/golem-worker-executor/src/filesystem_snapshot/mod.rs b/golem-worker-executor/src/filesystem_snapshot/mod.rs index aa2f8dc5ef..2228ba255b 100644 --- a/golem-worker-executor/src/filesystem_snapshot/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/mod.rs @@ -25,8 +25,6 @@ use std::cmp::Reverse; use std::fmt::{Display, Formatter}; use std::path::Path; -#[cfg(all(target_os = "linux", any(test, feature = "fs-snapshot-benchmark")))] -pub(crate) mod benchmark; #[cfg(test)] mod contract_tests; mod memory; diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs index 792181a44f..652121f8a3 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/mod.rs @@ -40,7 +40,7 @@ use bytesize::ByteSize; use golem_common::model::Timestamp; use golem_service_base::storage::blob::BlobStorage; use rustic_core::jiff::Span; -use rustic_core::repofile::{Chunker, ConfigFile, MasterKey, SnapshotFile}; +use rustic_core::repofile::{Chunker, MasterKey, SnapshotFile}; use rustic_core::{ BackupOptions, ConfigOptions, Credentials, KeyOptions, LimitOption, LocalDestination, LsOptions, Open, OpenStatus, ParentOptions, PathList, PruneOptions, PruneStats, @@ -306,18 +306,6 @@ pub(super) struct PruneReport { pub(super) phases: Box<[PhaseTime]>, } -/// What an inspection of a repository found. -#[derive(Clone, Debug, PartialEq, Eq)] -pub(super) struct InspectReport { - /// The number of snapshots of the repository. - pub(super) snapshots: u64, - /// Whether a snapshot has the name. - pub(super) found: bool, - /// The settings of the repository, as its config file gives them. - pub(super) settings: RepositorySettings, - pub(super) phases: Box<[PhaseTime]>, -} - /// The rustic repository of one scope in blob storage. /// /// Each operation opens the repository again, with the master key and without the rustic cache. @@ -329,7 +317,6 @@ pub(super) struct Repository { scope: SnapshotScope, key: RepositoryKey, deadline: Duration, - settings: RepositorySettings, } impl Repository { @@ -346,15 +333,9 @@ impl Repository { scope, key, deadline, - settings: RepositorySettings::default(), } } - /// Gives the repository with the settings that a save uses when it makes the repository. - pub(super) fn with_settings(self, settings: RepositorySettings) -> Self { - Self { settings, ..self } - } - /// Saves the directory tree `tree` as a snapshot with the name, with the default settings of /// a save. pub(super) async fn save( @@ -367,10 +348,10 @@ impl Repository { /// Saves the directory tree `tree` as a snapshot with the name. /// - /// A save in a scope without a repository makes the repository first, with the settings of - /// this value. The newest snapshot of the scope is the parent of the save, so the save reads - /// only the files that `settings` finds changed since that snapshot. The snapshot keeps the - /// paths relative to `tree`. + /// A save in a scope without a repository makes the repository first, with + /// [`RepositorySettings::DEFAULT`]. The newest snapshot of the scope is the parent of the save, + /// so the save reads only the files that `settings` finds changed since that snapshot. The + /// snapshot keeps the paths relative to `tree`. pub(super) async fn save_with( &self, name: &SnapshotName, @@ -379,11 +360,9 @@ impl Repository { ) -> anyhow::Result { let backend = self.backend()?; let key = self.key.clone(); - let repository_settings = self.settings; let name = name.clone(); let tree: Box = tree.into(); - run_blocking(move || save(backend, &key, &repository_settings, &settings, &name, &tree)) - .await + run_blocking(move || save(backend, &key, &settings, &name, &tree)).await } /// Restores the newest snapshot with the name into the empty directory `into`. @@ -427,18 +406,6 @@ impl Repository { run_blocking(move || prune(backend, &key, &settings)).await } - /// Opens the repository, finds the snapshots with the name, and loads the index, as a - /// restore does before it reads data. The result is `None` when the scope has no repository. - pub(super) async fn inspect( - &self, - name: &SnapshotName, - ) -> anyhow::Result> { - let backend = self.backend()?; - let key = self.key.clone(); - let name = name.clone(); - run_blocking(move || inspect(backend, &key, &name)).await - } - /// Gives a backend over the namespace of the scope, which waits on the current runtime for at /// most the deadline. fn backend(&self) -> anyhow::Result> { @@ -468,13 +435,12 @@ async fn run_blocking( fn save( backend: Arc, key: &RepositoryKey, - repository_settings: &RepositorySettings, settings: &SaveSettings, name: &SnapshotName, tree: &Path, ) -> anyhow::Result { let started = Instant::now(); - let (repository, opening) = open_or_create(backend, key, repository_settings)?; + let (repository, opening) = open_or_create(backend, key, &RepositorySettings::DEFAULT)?; let open = PhaseTime { phase: opening, wall: started.elapsed(), @@ -579,35 +545,6 @@ fn prune( })) } -fn inspect( - backend: Arc, - key: &RepositoryKey, - name: &SnapshotName, -) -> anyhow::Result> { - let (repository, open) = timed(OperationPhase::Open, || open_existing(backend, key))?; - let Some(repository) = repository else { - return Ok(None); - }; - let ((snapshots, found), lookup) = timed(OperationPhase::Lookup, || { - repository.get_all_snapshots().map(|snapshots| { - ( - snapshots.len(), - snapshots - .iter() - .any(|snapshot| snapshot.label == name.as_str()), - ) - }) - })?; - let settings = repository_settings(repository.config())?; - let (_, index) = timed(OperationPhase::IndexLoad, || repository.to_indexed())?; - Ok(Some(InspectReport { - snapshots: u64::try_from(snapshots)?, - found, - settings, - phases: Box::new([open, lookup, index]), - })) -} - fn forget( backend: Arc, key: &RepositoryKey, @@ -702,34 +639,6 @@ fn config_options(settings: &RepositorySettings) -> ConfigOptions { } } -/// Gives the settings of a repository from its config file, or an error when the config file has -/// fixed chunks whose size is not a size of [`Chunking::Fixed`]. -fn repository_settings(config: &ConfigFile) -> anyhow::Result { - Ok(RepositorySettings { - chunking: match config.chunker() { - Chunker::Rabin => Chunking::Rabin, - Chunker::FixedSize => Chunking::Fixed( - u32::try_from(config.chunk_size()) - .ok() - .and_then(NonZeroU32::new) - .with_context(|| { - format!( - "the repository has fixed chunks of {} bytes, which is not 1 to {} bytes", - config.chunk_size(), - u32::MAX - ) - })?, - ), - }, - compression: match config.compression.map(NonZeroI32::new) { - None => Compression::Default, - Some(None) => Compression::Off, - Some(Some(level)) => Compression::Level(level), - }, - extra_verify: config.extra_verify(), - }) -} - /// The options of a save. /// /// A snapshot keeps the paths relative to the saved tree. The parent of a save is the newest diff --git a/golem-worker-executor/src/filesystem_snapshot/rustic/tests/mod.rs b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/mod.rs index 33239e3407..d989627604 100644 --- a/golem-worker-executor/src/filesystem_snapshot/rustic/tests/mod.rs +++ b/golem-worker-executor/src/filesystem_snapshot/rustic/tests/mod.rs @@ -27,7 +27,7 @@ use super::backend::BlobBackend; use super::{ ChangeDetection, Chunking, Compression, OperationPhase, PruneSettings, RepackLimits, Repository, RepositoryKey, RepositorySettings, SaveSettings, backup_options, config_options, - open_existing, prune_options, repository_options, run_blocking, unopened, + open_existing, open_or_create, prune_options, repository_options, run_blocking, }; use crate::filesystem_snapshot::contract_tests::fixture::{ Scratch, Spec, fixture, listing, write_tree, @@ -1151,7 +1151,7 @@ fn each_setting_goes_into_its_rustic_option() { } #[test] -async fn a_repository_keeps_the_settings_of_its_first_save_and_inspect_gives_them() { +async fn a_repository_keeps_the_settings_of_its_creation_and_a_bridge_save_uses_the_defaults() { let storage = Arc::new(InMemoryBlobStorage::new()); let (fixed_scope, default_scope) = (new_scope(), new_scope()); let settings = RepositorySettings { @@ -1159,114 +1159,93 @@ async fn a_repository_keeps_the_settings_of_its_first_save_and_inspect_gives_the compression: Compression::Off, extra_verify: false, }; - // 4 chunks of 64 KiB and one of 1 byte, each with other bytes. Rabin keeps a file below its - // smallest chunk of 512 KiB in one chunk. - let tree = Scratch::new(); - let content = (0..4 * 65_536 + 1) - .map(|index| (index % 251) as u8) - .collect::>(); - std::fs::write(tree.path().join("data"), &content).unwrap(); - let fixed = repository(&storage, &fixed_scope).with_settings(settings); - let default = repository(&storage, &default_scope); - - let fixed_save = fixed.save(&name("first"), tree.path()).await.unwrap(); - let default_save = default.save(&name("first"), tree.path()).await.unwrap(); - let reopened = repository(&storage, &fixed_scope) - .with_settings(RepositorySettings::default()) - .inspect(&name("first")) - .await - .unwrap() - .unwrap(); - let default_settings = default - .inspect(&name("first")) - .await - .unwrap() - .unwrap() - .settings; - - assert_eq!( - ( - fixed_save.data_blobs, - fixed_save.data_added_packed >= fixed_save.data_added, - default_save.data_blobs, - default_save.data_added_packed < default_save.data_added, - reopened.settings, - default_settings, - ), - (5, true, 1, true, settings, RepositorySettings::default()) - ); -} - -#[test] -async fn inspect_gives_an_error_for_fixed_chunks_that_no_setting_can_hold() { - // A repository that rustic makes with fixed chunks of 4 GiB has a chunk size that does not fit - // `Chunking::Fixed`. The inspection must not report it as a Rabin repository. - let storage = Arc::new(InMemoryBlobStorage::new()); - let scope = new_scope(); let backend = Arc::new(BlobBackend::new( storage.clone(), - scope.0.clone(), + fixed_scope.0.clone(), Handle::current(), STORAGE_CALL_DEADLINE, )); run_blocking(move || { - unopened(backend)?.init( - &rustic_core::Credentials::Masterkey(key().master_key()), - &rustic_core::KeyOptions::default(), - &rustic_core::ConfigOptions::default() - .set_chunker(rustic_core::repofile::Chunker::FixedSize) - .set_chunk_size(bytesize::ByteSize::b(1 << 32)), - )?; + open_or_create(backend, &key(), &settings)?; Ok(()) }) .await .unwrap(); + // 4 chunks of 64 KiB and one of 1 byte, each with other bytes. Rabin keeps a file below its + // smallest chunk of 512 KiB in one chunk. + let tree = Scratch::new(); + let content = (0..4 * 65_536 + 1) + .map(|index| (index % 251) as u8) + .collect::>(); + std::fs::write(tree.path().join("data"), &content).unwrap(); + let config = async |scope: &SnapshotScope| { + with_existing_repository( + storage.clone(), + scope, + STORAGE_CALL_DEADLINE, + |repository| { + let config = repository.config(); + Ok(( + config.chunker(), + config.chunk_size(), + config.compression, + config.extra_verify(), + )) + }, + ) + .await + .unwrap() + }; - let inspected = repository(&storage, &scope).inspect(&name("first")).await; - - assert_eq!( - inspected.map_err(|error| error.to_string()), - Err(format!( - "the repository has fixed chunks of {} bytes, which is not 1 to {} bytes", - 1_u64 << 32, - u32::MAX - )) - ); -} - -#[test] -async fn inspect_gives_the_snapshots_the_name_and_the_phases_and_nothing_without_a_repository() { - let storage = Arc::new(InMemoryBlobStorage::new()); - let scope = new_scope(); - let repository = repository(&storage, &scope); - let tree = fixture_tree(); - - let before = repository.inspect(&name("first")).await.unwrap(); - repository.save(&name("first"), tree.path()).await.unwrap(); - repository.save(&name("second"), tree.path()).await.unwrap(); - let found = repository.inspect(&name("first")).await.unwrap().unwrap(); - let missing = repository.inspect(&name("third")).await.unwrap().unwrap(); + let fixed_save = repository(&storage, &fixed_scope) + .save(&name("first"), tree.path()) + .await + .unwrap(); + let default_save = repository(&storage, &default_scope) + .save(&name("first"), tree.path()) + .await + .unwrap(); + let (fixed_chunker, fixed_chunk_size, fixed_compression, fixed_extra_verify) = + config(&fixed_scope).await; + let (default_chunker, default_chunk_size, default_compression, default_extra_verify) = + config(&default_scope).await; assert_eq!( ( - before, - (found.snapshots, found.found), - (missing.snapshots, missing.found), - found - .phases - .iter() - .map(|time| time.phase) - .collect::>(), + ( + fixed_save.data_blobs, + fixed_save.data_added_packed >= fixed_save.data_added, + fixed_chunker, + fixed_chunk_size, + fixed_compression, + fixed_extra_verify, + ), + ( + default_save.data_blobs, + default_save.data_added_packed < default_save.data_added, + default_chunker, + default_chunk_size, + default_compression, + default_extra_verify, + ), ), ( - None, - (2, true), - (2, false), - vec![ - OperationPhase::Open, - OperationPhase::Lookup, - OperationPhase::IndexLoad - ], + ( + 5, + true, + rustic_core::repofile::Chunker::FixedSize, + 65_536, + Some(0), + false, + ), + ( + 1, + true, + rustic_core::repofile::Chunker::Rabin, + 1_048_576, + None, + true, + ), ) ); } diff --git a/golem-worker-executor/src/fs_snapshot_benchmark.rs b/golem-worker-executor/src/fs_snapshot_benchmark.rs deleted file mode 100644 index 040e403897..0000000000 --- a/golem-worker-executor/src/fs_snapshot_benchmark.rs +++ /dev/null @@ -1,30 +0,0 @@ -// Copyright 2024-2026 Golem Cloud -// -// Licensed under the Golem Source License v1.1 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://license.golem.cloud/LICENSE -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -//! The filesystem snapshot benchmark. It runs on Linux only, with the allocator of the executor. - -#[cfg(target_os = "linux")] -#[global_allocator] -static GLOBAL: mimalloc::MiMalloc = mimalloc::MiMalloc; - -#[cfg(target_os = "linux")] -fn main() -> std::process::ExitCode { - golem_worker_executor::fs_snapshot_benchmark_main() -} - -#[cfg(not(target_os = "linux"))] -fn main() -> std::process::ExitCode { - eprintln!("error: the filesystem snapshot benchmark runs on Linux only"); - std::process::ExitCode::from(2) -} diff --git a/golem-worker-executor/src/lib.rs b/golem-worker-executor/src/lib.rs index 363f1f2756..c11cc11943 100644 --- a/golem-worker-executor/src/lib.rs +++ b/golem-worker-executor/src/lib.rs @@ -35,10 +35,6 @@ pub mod workerctx; #[cfg(test)] pub mod span_test_support; -/// Runs the filesystem snapshot benchmark with the arguments of the process. -#[cfg(all(target_os = "linux", feature = "fs-snapshot-benchmark"))] -pub use filesystem_snapshot::benchmark::cli::main as fs_snapshot_benchmark_main; - #[cfg(test)] test_r::enable!(); diff --git a/golem_registry_service.db-wal b/golem_registry_service.db-wal deleted file mode 100644 index 64121e8261..0000000000 Binary files a/golem_registry_service.db-wal and /dev/null differ