diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 0ccece7..c10b3f0 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -29,6 +29,8 @@ jobs: run: python3 -m unittest discover -s tests -p 'test_*.py' - name: Compare performance with the pinned reference run: python3 scripts/benchmark.py + - name: Compare linear sieve with its pinned reference + run: python3 scripts/benchmark.py --config benchmarks/linear.json --output test-results/performance/linear - name: Upload performance evidence if: always() uses: actions/upload-artifact@v7 diff --git a/AGENTS.md b/AGENTS.md index f0d1dcf..93a1c5d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -7,3 +7,12 @@ - Never commit `RELEASE_PLAN.md`. - Explain progress to the user throughout the work. - See `docs/performance.md` for benchmark execution and baseline changes. + +# Code and documentation + +- Keep code simple, concise, and easy for humans to read. Avoid unnecessary abstractions without sacrificing performance. +- Write direct, concise documentation. Avoid em dashes, excessive semicolons, filler, and verbose AI-style prose. +- Never change `README.md` without the programmer's explicit authorization. +- Name implementation reports and plans in `docs` as `YYYY-MM-DD-topic.md` and include the date in a heading. +- Use the actual date of the documented event or result. Do not present historical results as current checks. +- Keep versioned release-note filenames. Permanent guides may also keep their existing names. diff --git a/benchmarks/harness.rs b/benchmarks/harness.rs index 946072c..e80a04d 100644 --- a/benchmarks/harness.rs +++ b/benchmarks/harness.rs @@ -19,7 +19,7 @@ fn sample() { .parse() .unwrap(), ); - let primes = tauri::async_runtime::block_on(super::calculate(limit)); + let primes = tauri::async_runtime::block_on(super::benchmark_calculate(limit)); let (count, last) = match limit { 1_000 => (168, 997), 100_000 => (9_592, 99_991), @@ -32,9 +32,9 @@ fn sample() { assert!(primes.windows(2).all(|pair| pair[0] < pair[1])); let run = || match operation.as_str() { "calculate" => { - black_box(tauri::async_runtime::block_on(super::calculate(black_box( - limit, - )))); + black_box(tauri::async_runtime::block_on(super::benchmark_calculate( + black_box(limit), + ))); } "serialize" => { black_box(serde_json::to_vec(black_box(&primes)).unwrap()); diff --git a/benchmarks/linear.json b/benchmarks/linear.json new file mode 100644 index 0000000..ac0a022 --- /dev/null +++ b/benchmarks/linear.json @@ -0,0 +1,33 @@ +{ + "baseline": "c6a0ac9665146f48381378f365c62214e0c2a903", + "rounds": 9, + "sample_ms": 300, + "max_slowdown": 0.2, + "cases": [ + { + "operation": "calculate", + "limit": 1000 + }, + { + "operation": "calculate", + "limit": 100000 + }, + { + "operation": "calculate", + "limit": 1000000 + }, + { + "operation": "calculate", + "limit": 10000000 + }, + { + "operation": "serialize", + "limit": 100000 + }, + { + "operation": "serialize", + "limit": 1000000 + } + ], + "command": "calculate_linear" +} diff --git a/docs/performance.md b/docs/performance.md index 1c5f81b..6c300b2 100644 --- a/docs/performance.md +++ b/docs/performance.md @@ -6,6 +6,11 @@ command and JSON serialization with commit The reference is pinned in `benchmarks/config.json`: moving `dev` or `main` does not silently reset the budget and allow a series of small slowdowns to accumulate. +The linear sieve has a separate reference in `benchmarks/linear.json`, pinned to +its first implementation, `c6a0ac9665146f48381378f365c62214e0c2a903`. Both algorithms +use the same workloads and regression budget. The linear sieve is an educational +alternative and does not need to match Eratosthenes in speed. + Both revisions are compiled with `cargo test --release --locked`, their own Cargo manifests and lockfiles, and the same installed Rust toolchain. A shared measurement harness is injected into temporary copies; it never replaces the @@ -40,6 +45,7 @@ With Python 3.10+, Git, Rust and the normal Tauri native build dependencies: ```sh python3 -m unittest discover -s tests -p 'test_*.py' python3 scripts/benchmark.py +python3 scripts/benchmark.py --config benchmarks/linear.json --output test-results/performance/linear ``` The candidate includes local Rust edits. The baseline commit must be available @@ -52,6 +58,11 @@ For a deliberate investigation, use `--baseline `. Update the pinned baseline only in a reviewed task explaining the accepted performance tradeoff, and retain the before/after report. Do not refresh it automatically after a merge. +To compare algorithms directly, add `--baseline-command calculate` to the linear +command above. This compares the candidate's linear sieve with primal from the +linear reference commit. The report names both commands. A slower alternative +can fail this diagnostic comparison without failing its own regression check. + ## Branch workflow Start each task from an up-to-date `dev` and create a dedicated branch. Commit diff --git a/scripts/benchmark.py b/scripts/benchmark.py index 19c1769..08cb9fb 100644 --- a/scripts/benchmark.py +++ b/scripts/benchmark.py @@ -14,6 +14,7 @@ import tempfile ROOT = Path(__file__).resolve().parents[1] +COMMANDS = ("calculate", "calculate_linear") def command(args, **kwargs): @@ -21,6 +22,8 @@ def command(args, **kwargs): def validate_config(config): + if config.get("command", "calculate") not in COMMANDS: + raise ValueError("Unsupported calculation command") if not isinstance(config["rounds"], int) or config["rounds"] < 5: raise ValueError("At least five rounds are required") if not 100 <= config["sample_ms"] <= 10000: @@ -48,7 +51,7 @@ def assess(baseline, candidate, max_slowdown): "regression": regression, "ratios": ratios} -def prepare(destination, revision=None): +def prepare(destination, revision=None, command_name="calculate"): if revision: archive = subprocess.check_output(["git", "archive", revision], cwd=ROOT) with tarfile.open(fileobj=io.BytesIO(archive)) as source: @@ -66,6 +69,7 @@ def prepare(destination, revision=None): (assets / "index.html").write_text("Benchmark") main = destination / "src-tauri/src/main.rs" with main.open("a") as output: + output.write(f'\n#[cfg(test)]\nuse {command_name} as benchmark_calculate;\n') output.write('\n#[cfg(test)]\n#[path = "benchmark_harness.rs"]\nmod performance;\n') shutil.copyfile(ROOT / "benchmarks/harness.rs", main.with_name("benchmark_harness.rs")) @@ -101,17 +105,23 @@ def measure(binary, case, sample_ms): def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--baseline", help="Override the pinned baseline for a deliberate comparison") + parser.add_argument("--config", type=Path, default=ROOT / "benchmarks/config.json") + parser.add_argument("--baseline-command", choices=COMMANDS, + help="Compare different algorithms without changing their regression gates") parser.add_argument("--output", type=Path, default=ROOT / "test-results/performance") args = parser.parse_args() - config = json.loads((ROOT / "benchmarks/config.json").read_text()) + config = json.loads(args.config.read_text()) validate_config(config) + candidate_command = config.get("command", "calculate") + baseline_command = args.baseline_command or candidate_command baseline = command(["git", "rev-parse", "--verify", (args.baseline or config["baseline"]) + "^{commit}"], cwd=ROOT) output = args.output.resolve() output.mkdir(parents=True, exist_ok=True) report = {"baseline": baseline, "candidate": command(["git", "rev-parse", "HEAD"], cwd=ROOT), "dirty": bool(command(["git", "status", "--porcelain"], cwd=ROOT)), "platform": platform.platform(), "rustc": command(["rustc", "-Vv"]), - "config": config, "results": []} + "config": config, "baseline_command": baseline_command, + "candidate_command": candidate_command, "results": []} # Compile with all available CPUs; pin only the subsequent measurements. target = Path(os.environ.get("CARGO_TARGET_DIR", ROOT / "src-tauri/target")).resolve() with tempfile.TemporaryDirectory(prefix="graphprime-benchmark-") as temporary: @@ -121,7 +131,7 @@ def main(): print(f"Building {label} in release mode...", flush=True) source = work / label source.mkdir() - prepare(source, revision) + prepare(source, revision, baseline_command if label == "baseline" else candidate_command) binaries[label] = work / f"{label}-benchmark" build(source, target, binaries[label], output / f"{label}-build.jsonl") if hasattr(os, "sched_getaffinity"): @@ -139,7 +149,8 @@ def main(): print(f'{case["operation"]}/{case["limit"]}: {result["slowdown"]:+.1%}' f' {"FAIL" if result["regression"] else "PASS"}', flush=True) (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") - lines = ["## Performance comparison", "", f"Baseline: `{baseline}`", "", + lines = [f"## Performance comparison: {candidate_command}", "", + f"Baseline: `{baseline}` (`{baseline_command}`)", "", "| Workload | Baseline (ms) | Candidate (ms) | Change | Result |", "| --- | ---: | ---: | ---: | --- |"] for result in report["results"]: diff --git a/src-tauri/src/main.rs b/src-tauri/src/main.rs index 48b4222..016e701 100644 --- a/src-tauri/src/main.rs +++ b/src-tauri/src/main.rs @@ -6,7 +6,7 @@ fn main() { tauri::Builder::default() .plugin(tauri_plugin_clipboard_manager::init()) - .invoke_handler(tauri::generate_handler![calculate]) + .invoke_handler(tauri::generate_handler![calculate, calculate_linear]) .run(tauri::generate_context!()) .expect("error while running tauri application"); } @@ -26,9 +26,37 @@ async fn calculate(x: u64) -> Vec { .collect() } +// Based on Jayadev Misra's smallest-divisor invariant: https://www.cs.utexas.edu/~misra/scannedPdf.dir/ProgramExplanation.pdf +#[tauri::command] +async fn calculate_linear(x: u64) -> Vec { + if x < 2 { + return Vec::new(); + } + + let limit = x as usize; + let mut composite = vec![false; limit + 1]; + let mut primes = Vec::new(); + for n in 2..=limit { + if !composite[n] { + primes.push(n as u64); + } + for &prime in &primes { + let p = prime as usize; + if p > limit / n { + break; + } + composite[n * p] = true; + if n % p == 0 { + break; + } + } + } + primes +} + #[cfg(test)] mod tests { - use super::calculate; + use super::{calculate, calculate_linear}; #[test] fn calculates_small_sequences_and_boundaries() { @@ -59,6 +87,22 @@ mod tests { expected, "limit {limit}" ); + assert_eq!( + tauri::async_runtime::block_on(calculate_linear(limit)), + expected, + "linear limit {limit}" + ); + } + } + + #[test] + fn linear_sieve_matches_eratosthenes_for_large_limits() { + for limit in [100_000, 1_000_000] { + assert_eq!( + tauri::async_runtime::block_on(calculate_linear(limit)), + tauri::async_runtime::block_on(calculate(limit)), + "limit {limit}" + ); } } diff --git a/src/App.svelte b/src/App.svelte index 9ebb7af..c3ad95d 100644 --- a/src/App.svelte +++ b/src/App.svelte @@ -23,6 +23,7 @@ let calculationTime = $state(0); let compositeNumbers = $state(74); let chartType = $state("frappe"); + let algorithm = $state("calculate"); let chartFullscreen = $state(false); let editorFullscreen = $state(false); @@ -49,7 +50,7 @@ const calculationStart = Date.now(); try { - primes = await invoke("calculate", { x: chosenFinalValue }); + primes = await invoke(algorithm, { x: chosenFinalValue }); chartType = chosenFinalValue >= 10000 ? "dygraph" : "frappe"; calculationTime = (Date.now() - calculationStart) / 1000; lastFinalValue = chosenFinalValue; @@ -72,15 +73,26 @@

- +
+ + +
+
+ + +
@@ -94,7 +106,25 @@ {/if} {#if primes} - +
+ + +
+

+ {algorithm === "calculate" ? "Sieve of Eratosthenes" : "Linear sieve"} +

+

+ {#if algorithm === "calculate"} + Marks multiples of each prime to find all primes up to your limit. Its work grows as O(n + log log n). The optimized implementation is the default for fast calculations. + {:else} + Marks each composite once using its smallest prime factor. Its work grows as O(n), but + it may use more memory and run slower than the optimized Eratosthenes sieve. + {/if} + Both algorithms return the same exact primes. Larger limits require more time and memory. +

+
+
@@ -144,16 +174,77 @@ font-weight: 400; } + .stats-cards { + display: grid; + gap: 1em; + margin: 1em; + } + + .stats-cards > :global(.card) { + margin: 0; + min-width: 0; + } + + @media (min-width: 1100px) { + .stats-cards { + grid-template-columns: 1fr 1fr; + } + } + .input-group { display: flex; justify-content: center; - align-items: stretch; + align-items: end; gap: 10px; } + .limit-field, + .algorithm-picker { + display: flex; + flex-direction: column; + gap: 6px; + text-align: left; + } + + .limit-field { + flex: 1; + min-width: 0; + } + + label { + font-size: 0.875em; + } + + .input, + .algorithm-picker select, + .button { + box-sizing: border-box; + height: 40px; + } + + .algorithm-picker select { + padding: 0.4rem 2rem 0.4rem 0.6rem; + border: 1px solid var(--border-color); + border-radius: 5px; + font: inherit; + color: var(--body-color); + background: var(--card-bg); + appearance: none; + background-image: + linear-gradient(45deg, transparent 50%, currentColor 50%), + linear-gradient(135deg, currentColor 50%, transparent 50%); + background-position: + calc(100% - 14px) calc(50% + 1px), + calc(100% - 9px) calc(50% + 1px); + background-size: + 5px 5px, + 5px 5px; + background-repeat: no-repeat; + } + .input { - max-width: 60%; - flex-grow: 5; + width: 100%; + padding: 0.4rem 0.6rem; border: 1px solid var(--border-color); border-radius: 5px; font-size: 1.2em; @@ -165,8 +256,7 @@ .button { min-width: 100px; - max-width: 20%; - flex-grow: 1; + padding: 0 12px; border: none; border-radius: 5px; font-size: 1.2em; diff --git a/src/Components/Cards/Stats.svelte b/src/Components/Cards/Stats.svelte index 5d26ac2..9ed6f90 100644 --- a/src/Components/Cards/Stats.svelte +++ b/src/Components/Cards/Stats.svelte @@ -14,8 +14,8 @@ } = $props(); -
-

Stats

+
+

Stats

{primes.length} prime numbers calculated up to {lastFinalValue}

diff --git a/tests/smoke-linux.py b/tests/smoke-linux.py index 536b3fd..1686a39 100644 --- a/tests/smoke-linux.py +++ b/tests/smoke-linux.py @@ -150,45 +150,59 @@ def ready(): screenshot("initial") print("PASS: bundled CSS, dynamic editor styles, initial SVG graph", flush=True) - for limit in [1, 2, 30, 1000, 10000, 100000, 100]: - calculate(limit) - if limit == 1: - assert "No prime numbers in this range." in js("return document.body.innerText") - else: - graph = ".chart canvas" if limit >= 10000 else ".frappe-chart svg" - wait_for(lambda: js("return !!document.querySelector(arguments[0])", graph), "graph renders") - if limit == 10000: - js("window.smokeCanvas = document.querySelector('.chart canvas')") - if limit == 100000: - assert js("return window.smokeCanvas === document.querySelector('.chart canvas')") - click('[aria-label="Toggle graph fullscreen"]') - wait_for(lambda: js("return !!document.querySelector('.fullscreen .chart')"), "scientific fullscreen") - click('[aria-label="Toggle graph fullscreen"]') - wait_for(lambda: js("return !document.querySelector('.fullscreen')"), "exit scientific fullscreen") - js("document.querySelector('.chart').scrollIntoView()") - screenshot("scientific") - - check_clipboard(expected_primes(100)) - print("PASS: native clipboard", flush=True) - - js("window.smokeEditor = document.querySelector('.cm-editor')") - for label in ["Toggle primes fullscreen", "Toggle graph fullscreen"]: - click(f'[aria-label="{label}"]') - wait_for(lambda: js("return document.querySelectorAll('.fullscreen').length === 1"), label) - screenshot(label.lower().replace(" ", "-")) - click(f'[aria-label="{label}"]') - wait_for(lambda: js("return !document.querySelector('.fullscreen')"), "exit fullscreen") - - assert js("return window.smokeEditor === document.querySelector('.cm-editor')") - for chart_type in ["dygraph", "frappe", "dygraph", "frappe"]: + assert js("return document.querySelector('#algorithm').value") == "calculate" + for algorithm, heading in [("calculate", "Sieve of Eratosthenes"), + ("calculate_linear", "Linear sieve")]: + previous_stats = js("return document.querySelector('[aria-labelledby=stats-heading]').textContent") js(""" - const select = document.querySelector('select'); + const select = document.querySelector('#algorithm'); select.value = arguments[0]; select.dispatchEvent(new Event('change', {bubbles: true})); - """, chart_type) - selector = ".chart canvas" if chart_type == "dygraph" else ".frappe-chart svg" - wait_for(lambda: js("return !!document.querySelector(arguments[0])", selector), "chart switch") - print("PASS: fullscreen and repeated chart switching", flush=True) + """, algorithm) + wait_for(lambda: js("return document.querySelector('#algorithm-heading').textContent.trim()") == heading, + "algorithm explanation updates") + assert js("return document.querySelector('[aria-labelledby=stats-heading]').textContent") == previous_stats + js("window.scrollTo(0, 0)") + screenshot(algorithm) + for limit in [1, 2, 30, 1000, 10000, 100000, 100]: + calculate(limit) + if limit == 1: + assert "No prime numbers in this range." in js("return document.body.innerText") + else: + graph = ".chart canvas" if limit >= 10000 else ".frappe-chart svg" + wait_for(lambda: js("return !!document.querySelector(arguments[0])", graph), "graph renders") + if limit == 10000: + js("window.smokeCanvas = document.querySelector('.chart canvas')") + if limit == 100000: + assert js("return window.smokeCanvas === document.querySelector('.chart canvas')") + click('[aria-label="Toggle graph fullscreen"]') + wait_for(lambda: js("return !!document.querySelector('.fullscreen .chart')"), "scientific fullscreen") + click('[aria-label="Toggle graph fullscreen"]') + wait_for(lambda: js("return !document.querySelector('.fullscreen')"), "exit scientific fullscreen") + js("document.querySelector('.chart').scrollIntoView()") + screenshot(f"{algorithm}-scientific") + + check_clipboard(expected_primes(100)) + print("PASS: native clipboard", flush=True) + + js("window.smokeEditor = document.querySelector('.cm-editor')") + for label in ["Toggle primes fullscreen", "Toggle graph fullscreen"]: + click(f'[aria-label="{label}"]') + wait_for(lambda: js("return document.querySelectorAll('.fullscreen').length === 1"), label) + screenshot(algorithm + "-" + label.lower().replace(" ", "-")) + click(f'[aria-label="{label}"]') + wait_for(lambda: js("return !document.querySelector('.fullscreen')"), "exit fullscreen") + + assert js("return window.smokeEditor === document.querySelector('.cm-editor')") + for chart_type in ["dygraph", "frappe", "dygraph", "frappe"]: + js(""" + const select = document.querySelector('[aria-label="Graph type"]'); + select.value = arguments[0]; + select.dispatchEvent(new Event('change', {bubbles: true})); + """, chart_type) + selector = ".chart canvas" if chart_type == "dygraph" else ".frappe-chart svg" + wait_for(lambda: js("return !!document.querySelector(arguments[0])", selector), "chart switch") + print("PASS: fullscreen and repeated chart switching", flush=True) js("window.scrollTo(0, 0)") screenshot("final") diff --git a/tests/test_benchmark.py b/tests/test_benchmark.py index 35d7703..b5fb5a3 100644 --- a/tests/test_benchmark.py +++ b/tests/test_benchmark.py @@ -16,10 +16,15 @@ def test_invalid_configuration_cannot_disable_the_gate(self): config = json.loads((benchmark.ROOT / "benchmarks/config.json").read_text()) benchmark.validate_config(config) for key, value in [("rounds", 0), ("sample_ms", 0), ("max_slowdown", float("nan")), - ("max_slowdown", 1), ("cases", [])]: + ("max_slowdown", 1), ("cases", []), ("command", "unknown")]: with self.subTest(key=key, value=value), self.assertRaises(ValueError): benchmark.validate_config(dict(config, **{key: value})) + def test_both_algorithm_configurations_are_valid(self): + for name in ("config.json", "linear.json"): + config = json.loads((benchmark.ROOT / "benchmarks" / name).read_text()) + benchmark.validate_config(config) + def test_equal_performance_and_improvements_pass(self): for candidate in ([100] * 9, [70] * 9): self.assertFalse(benchmark.assess([100] * 9, candidate, 0.2)["regression"])