From 4aad751c54514c6a96832d91efb96ab1074dff7d Mon Sep 17 00:00:00 2001 From: Ray Liu Date: Wed, 23 Sep 2026 00:41:20 -0400 Subject: [PATCH] [5/6][load] publish shard-10 ramp run --- .../analysis.md | 5 +- .../analysis.md | 5 +- .../analysis.md | 107 +++++ .../result.json | 400 ++++++++++++++++++ .../run.json | 76 ++++ .../analysis.md | 5 +- .../analysis.md | 5 +- 7 files changed, 599 insertions(+), 4 deletions(-) create mode 100644 docs/performance/runs/2026-09-23-adhoc-shard-10-ramp-33c95bf/analysis.md create mode 100644 docs/performance/runs/2026-09-23-adhoc-shard-10-ramp-33c95bf/result.json create mode 100644 docs/performance/runs/2026-09-23-adhoc-shard-10-ramp-33c95bf/run.json diff --git a/docs/performance/runs/2026-09-23-adhoc-lean-probe-capacity-33c95bf/analysis.md b/docs/performance/runs/2026-09-23-adhoc-lean-probe-capacity-33c95bf/analysis.md index c5f5542..5343c4f 100644 --- a/docs/performance/runs/2026-09-23-adhoc-lean-probe-capacity-33c95bf/analysis.md +++ b/docs/performance/runs/2026-09-23-adhoc-lean-probe-capacity-33c95bf/analysis.md @@ -11,7 +11,7 @@ retained diagnostics. ## Sequence -This is the second of four runs published on 2026-09-23: +This is the second of the runs published on 2026-09-23: 1. [Harness capacity](../2026-09-23-north-star-steady-v1-harness-capacity-33c95bf/analysis.md): `kq-bench` saturated the host near 13,000 tasks/s, mostly on its own @@ -22,6 +22,9 @@ This is the second of four runs published on 2026-09-23: 4. [Share group with 1000 members](../2026-09-23-adhoc-share-group-1000-members-33c95bf/analysis.md): revisit this run's 20,000 RPS concurrency series with the member cap raised from 200 to 1000. +5. [Shard-10 ramp](../2026-09-23-adhoc-shard-10-ramp-33c95bf/analysis.md): + ramp arrival with 1000 members x concurrency 10; 20,000 RPS is the highest + rate near the Redis reference, limited by slot capacity. ## Purpose diff --git a/docs/performance/runs/2026-09-23-adhoc-queue-latency-diagnosis-33c95bf/analysis.md b/docs/performance/runs/2026-09-23-adhoc-queue-latency-diagnosis-33c95bf/analysis.md index e647e0f..b66bec0 100644 --- a/docs/performance/runs/2026-09-23-adhoc-queue-latency-diagnosis-33c95bf/analysis.md +++ b/docs/performance/runs/2026-09-23-adhoc-queue-latency-diagnosis-33c95bf/analysis.md @@ -9,7 +9,7 @@ retained diagnostics. ## Sequence -This is the third of four runs published on 2026-09-23: +This is the third of the runs published on 2026-09-23: 1. [Harness capacity](../2026-09-23-north-star-steady-v1-harness-capacity-33c95bf/analysis.md): `kq-bench` saturated the host near 13,000 tasks/s. @@ -20,6 +20,9 @@ This is the third of four runs published on 2026-09-23: 4. [Share group with 1000 members](../2026-09-23-adhoc-share-group-1000-members-33c95bf/analysis.md): follows suggestion 2 below by raising the member cap to 1000 and running 1000 members x concurrency 10 at 20,000 RPS. +5. [Shard-10 ramp](../2026-09-23-adhoc-shard-10-ramp-33c95bf/analysis.md): + ramp arrival with 1000 members x concurrency 10; 20,000 RPS is the highest + rate near the Redis reference, limited by slot capacity. ## Purpose diff --git a/docs/performance/runs/2026-09-23-adhoc-shard-10-ramp-33c95bf/analysis.md b/docs/performance/runs/2026-09-23-adhoc-shard-10-ramp-33c95bf/analysis.md new file mode 100644 index 0000000..a7e0906 --- /dev/null +++ b/docs/performance/runs/2026-09-23-adhoc-shard-10-ramp-33c95bf/analysis.md @@ -0,0 +1,107 @@ +# Run Analysis + +Run: [run.json](run.json) + +Results: [result.json](result.json) + +Kafka: [compose.yaml](../2026-09-23-adhoc-share-group-1000-members-33c95bf/compose.yaml) +(`group.share.max.size=1000`, `group.share.partition.max.record.locks=4000`) + +This document is generated by AI from the run metadata, machine results, and +retained diagnostics. + +## Sequence + +This is the fifth of the runs published on 2026-09-23: + +1. [Harness capacity](../2026-09-23-north-star-steady-v1-harness-capacity-33c95bf/analysis.md): + `kq-bench` saturated the host near 13,000 tasks/s. +2. [Lean-probe capacity](../2026-09-23-adhoc-lean-probe-capacity-33c95bf/analysis.md): + kq reached 150,000 tasks/s, but queue p50 stayed near 465 ms. +3. [Queue-latency diagnosis](../2026-09-23-adhoc-queue-latency-diagnosis-33c95bf/analysis.md): + queue time comes from the worker generation wait. +4. [Share group with 1000 members](../2026-09-23-adhoc-share-group-1000-members-33c95bf/analysis.md): + 1000 members x concurrency 10 sustained 20,000 RPS with low queue time at + 8 to 32 partitions and collapsed at 2 and 4. +5. **This run.** Hold the run 4 topology fixed and ramp arrival at 8, 16, 32, + and 64 partitions to find the highest rate that keeps queue time near the + Redis reference. + +## Purpose + +Find the highest arrival rate at which 1000 share-group members x concurrency +10 keep north-star queue time near the Redis reference (36 / 77 / 100 ms +p50 / p95 / p99), ramping arrival at 8, 16, 32, and 64 ready partitions with +the record-lock window fixed at 4000. + +A point meets the target when every task completes exactly once, enqueue +reaches at least 99% of the requested rate, completions end within 3 s of +production, and queue p50 / p95 / p99 are at most 50 / 150 / 300 ms. + +## Observations + +| Partitions | Requested RPS | Median completions/s | Tail after production | Queue p50 / p95 / p99 / max | Outcome | +| ---: | ---: | ---: | ---: | ---: | --- | +| 8 | 20,000 | 20,013 | 1 s | 14 / 48 / 99 / 589 ms | meets target | +| 8 | 21,000 | 21,042 | 1 s | 50 / 160 / 243 / 633 ms | sustained; p95 over limit | +| 8 | 22,000 | 20,911 | 5 s | 1,431 / 3,205 / 3,813 / 3,978 ms | overloaded | +| 16 | 20,000 | 20,010 | 1 s | 24 / 70 / 101 / 837 ms | meets target | +| 16 | 21,000 | 20,938 | 4 s | 60 / 597 / 2,815 / 3,485 ms | overloaded | +| 32 | 20,000 | 19,996 | 1 s | 42 / 133 / 207 / 544 ms | meets target | +| 32 | 21,000 | 20,884 | 14 s | 94 / 465 / 11,391 / 13,639 ms | overloaded | +| 64 | 18,000 | 18,003 | 1 s | 39 / 125 / 203 / 724 ms | meets target | +| 64 | 19,000 | 18,992 | 2 s | 51 / 171 / 421 / 1,435 ms | sustained; over limit | +| 64 | 20,000 | 20,005 | 1 s | 73 / 225 / 361 / 903 ms | sustained; over limit | + +- Highest rate meeting the target: 20,000 RPS at 8, 16, and 32 partitions; + 18,000 RPS at 64 partitions. +- Every overloaded point plateaued at 20,884 to 20,938 median completions/s + regardless of partition count. +- At a fixed 20,000 RPS, queue p50 rose monotonically with partition count: + 14, 24, 42, and 73 ms at 8, 16, 32, and 64 partitions. +- All 12,120,000 tasks across all points completed with zero duplicates and + zero missing IDs. +- Kafka used 5.7 to 6.3 cores at sustained points and 7.9 cores at the + 8-partition 22,000 RPS overload. Workers used about 1.6 cores and producers + about 1.4 cores. The host was not CPU-saturated. +- The 8-partition 20,000 RPS point (14 / 48 / 99 ms) was below the Redis + reference at p50 and p95 and matched it at p99. The same configuration in + run 4 measured 14 / 104 / 203 ms, so p95 and p99 varied between otherwise + identical runs. + +## Interpretation + +- The ceiling is worker slot capacity, not Kafka partitions or host CPU. + About 20,900 completions/s across 10,000 slots is about 2.1 tasks/s per + slot, or about 480 ms per generation of 10 against a 250 ms average task. + Roughly half of each slot's time is spent waiting for the slowest task in + its generation. +- Because `group.share.max.size` cannot exceed 1000, slots can only grow by + raising per-member concurrency, which lengthens each generation. +- More partitions than needed increase queue time. Hypothesis: each member's + small fetches are spread across more partitions and fetch sessions. 8 to 16 + partitions is the best range for this topology. +- Overload raises Kafka CPU, consistent with the collapse feedback + hypothesized in run 4, though no point here collapsed. + +## Suggestions + +1. Measure concurrency 20 with the same ramp to see how much capacity larger + generations add and at what latency cost. +2. Prioritize a worker that refills slots as tasks complete. At about 4 + tasks/s per slot, the same 10,000 slots could plausibly support 35,000 to + 40,000 tasks/s; this estimate is untested. +3. Repeat the 8- and 16-partition 20,000 RPS points several times to bound + run-to-run variation in p95 and p99. + +## Caveats + +- Single Dockerized Kafka broker on localhost, RF=1; one run per point with a + 60 s arrival window. +- Same probe as runs 2 to 4: ready success path only, sleep-only handlers, + 200-byte payloads, queue time measured from just before `Enqueue`. +- The ramp stopped at the first failing point per partition count, so some + partition counts have only two points. +- The acceptance limits are an engineering judgment of "close to" the Redis + reference, not a production SLO. The Redis figures were not measured with + this workload or harness. diff --git a/docs/performance/runs/2026-09-23-adhoc-shard-10-ramp-33c95bf/result.json b/docs/performance/runs/2026-09-23-adhoc-shard-10-ramp-33c95bf/result.json new file mode 100644 index 0000000..c062c0b --- /dev/null +++ b/docs/performance/runs/2026-09-23-adhoc-shard-10-ramp-33c95bf/result.json @@ -0,0 +1,400 @@ +{ + "correctness": { + "produced": 12120000, + "completed": 12120000, + "duplicates": 0, + "missing": 0 + }, + "summary_by_partitions": [ + { + "ready_partitions": 8, + "highest_rps_meeting_target": 20000, + "tested_rps": [ + 20000, + 21000, + 22000 + ], + "maximum_median_completion_rps": 21042 + }, + { + "ready_partitions": 16, + "highest_rps_meeting_target": 20000, + "tested_rps": [ + 20000, + 21000 + ], + "maximum_median_completion_rps": 20938 + }, + { + "ready_partitions": 32, + "highest_rps_meeting_target": 20000, + "tested_rps": [ + 20000, + 21000 + ], + "maximum_median_completion_rps": 20884 + }, + { + "ready_partitions": 64, + "highest_rps_meeting_target": 18000, + "tested_rps": [ + 18000, + 19000, + 20000 + ], + "maximum_median_completion_rps": 20005 + } + ], + "points": [ + { + "label": "ramp-p8-20k", + "requested_rps": 20000, + "duration_seconds": 60, + "ready_partitions": 8, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 3, + "enqueue_goroutines_per_client": 512, + "achieved_enqueue_rps": 20000, + "enqueue_errors": 0, + "completion_rps_median_1s": 20013, + "completion_rps_peak_1s": 21086, + "seconds_with_completions_after_production": 1, + "queue_p50_ms": 14, + "queue_p95_ms": 48, + "queue_p99_ms": 99, + "queue_max_ms": 589, + "correctness": { + "produced": 1200000, + "completed": 1200000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 5.71, + "producer": 1.31, + "workers": 1.56 + }, + "sustained": true, + "meets_latency_target": true + }, + { + "label": "ramp-p8-21k", + "requested_rps": 21000, + "duration_seconds": 60, + "ready_partitions": 8, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 3, + "enqueue_goroutines_per_client": 512, + "achieved_enqueue_rps": 21000, + "enqueue_errors": 0, + "completion_rps_median_1s": 21042, + "completion_rps_peak_1s": 21275, + "seconds_with_completions_after_production": 1, + "queue_p50_ms": 50, + "queue_p95_ms": 160, + "queue_p99_ms": 243, + "queue_max_ms": 633, + "correctness": { + "produced": 1260000, + "completed": 1260000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 6.1, + "producer": 1.48, + "workers": 1.6 + }, + "sustained": true, + "meets_latency_target": false + }, + { + "label": "ramp-p8-22k", + "requested_rps": 22000, + "duration_seconds": 60, + "ready_partitions": 8, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 3, + "enqueue_goroutines_per_client": 512, + "achieved_enqueue_rps": 21999, + "enqueue_errors": 0, + "completion_rps_median_1s": 20911, + "completion_rps_peak_1s": 21348, + "seconds_with_completions_after_production": 5, + "queue_p50_ms": 1431, + "queue_p95_ms": 3205, + "queue_p99_ms": 3813, + "queue_max_ms": 3978, + "correctness": { + "produced": 1320000, + "completed": 1320000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 7.9, + "producer": 0.94, + "workers": 1.37 + }, + "sustained": false, + "meets_latency_target": false + }, + { + "label": "ramp-p16-20k", + "requested_rps": 20000, + "duration_seconds": 60, + "ready_partitions": 16, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 3, + "enqueue_goroutines_per_client": 512, + "achieved_enqueue_rps": 20000, + "enqueue_errors": 0, + "completion_rps_median_1s": 20010, + "completion_rps_peak_1s": 20876, + "seconds_with_completions_after_production": 1, + "queue_p50_ms": 24, + "queue_p95_ms": 70, + "queue_p99_ms": 101, + "queue_max_ms": 837, + "correctness": { + "produced": 1200000, + "completed": 1200000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 5.97, + "producer": 1.43, + "workers": 1.54 + }, + "sustained": true, + "meets_latency_target": true + }, + { + "label": "ramp-p16-21k", + "requested_rps": 21000, + "duration_seconds": 60, + "ready_partitions": 16, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 3, + "enqueue_goroutines_per_client": 512, + "achieved_enqueue_rps": 21000, + "enqueue_errors": 0, + "completion_rps_median_1s": 20938, + "completion_rps_peak_1s": 21232, + "seconds_with_completions_after_production": 4, + "queue_p50_ms": 60, + "queue_p95_ms": 597, + "queue_p99_ms": 2815, + "queue_max_ms": 3485, + "correctness": { + "produced": 1260000, + "completed": 1260000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 6.24, + "producer": 1.5, + "workers": 1.62 + }, + "sustained": false, + "meets_latency_target": false + }, + { + "label": "ramp-p32-20k", + "requested_rps": 20000, + "duration_seconds": 60, + "ready_partitions": 32, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 3, + "enqueue_goroutines_per_client": 512, + "achieved_enqueue_rps": 20000, + "enqueue_errors": 0, + "completion_rps_median_1s": 19996, + "completion_rps_peak_1s": 20284, + "seconds_with_completions_after_production": 1, + "queue_p50_ms": 42, + "queue_p95_ms": 133, + "queue_p99_ms": 207, + "queue_max_ms": 544, + "correctness": { + "produced": 1200000, + "completed": 1200000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 5.82, + "producer": 1.39, + "workers": 1.6 + }, + "sustained": true, + "meets_latency_target": true + }, + { + "label": "ramp-p32-21k", + "requested_rps": 21000, + "duration_seconds": 60, + "ready_partitions": 32, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 3, + "enqueue_goroutines_per_client": 512, + "achieved_enqueue_rps": 21000, + "enqueue_errors": 0, + "completion_rps_median_1s": 20884, + "completion_rps_peak_1s": 21171, + "seconds_with_completions_after_production": 14, + "queue_p50_ms": 94, + "queue_p95_ms": 465, + "queue_p99_ms": 11391, + "queue_max_ms": 13639, + "correctness": { + "produced": 1260000, + "completed": 1260000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 6.3, + "producer": 1.54, + "workers": 1.61 + }, + "sustained": false, + "meets_latency_target": false + }, + { + "label": "ramp-p64-18k", + "requested_rps": 18000, + "duration_seconds": 60, + "ready_partitions": 64, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 2, + "enqueue_goroutines_per_client": 512, + "achieved_enqueue_rps": 18000, + "enqueue_errors": 0, + "completion_rps_median_1s": 18003, + "completion_rps_peak_1s": 19778, + "seconds_with_completions_after_production": 1, + "queue_p50_ms": 39, + "queue_p95_ms": 125, + "queue_p99_ms": 203, + "queue_max_ms": 724, + "correctness": { + "produced": 1080000, + "completed": 1080000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 5.9, + "producer": 1.37, + "workers": 1.6 + }, + "sustained": true, + "meets_latency_target": true + }, + { + "label": "ramp-p64-19k", + "requested_rps": 19000, + "duration_seconds": 60, + "ready_partitions": 64, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 2, + "enqueue_goroutines_per_client": 512, + "achieved_enqueue_rps": 19000, + "enqueue_errors": 0, + "completion_rps_median_1s": 18992, + "completion_rps_peak_1s": 20714, + "seconds_with_completions_after_production": 2, + "queue_p50_ms": 51, + "queue_p95_ms": 171, + "queue_p99_ms": 421, + "queue_max_ms": 1435, + "correctness": { + "produced": 1140000, + "completed": 1140000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 5.81, + "producer": 1.36, + "workers": 1.63 + }, + "sustained": true, + "meets_latency_target": false + }, + { + "label": "ramp-p64-20k", + "requested_rps": 20000, + "duration_seconds": 60, + "ready_partitions": 64, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 3, + "enqueue_goroutines_per_client": 512, + "achieved_enqueue_rps": 20000, + "enqueue_errors": 0, + "completion_rps_median_1s": 20005, + "completion_rps_peak_1s": 20314, + "seconds_with_completions_after_production": 1, + "queue_p50_ms": 73, + "queue_p95_ms": 225, + "queue_p99_ms": 361, + "queue_max_ms": 903, + "correctness": { + "produced": 1200000, + "completed": 1200000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 5.92, + "producer": 1.4, + "workers": 1.62 + }, + "sustained": true, + "meets_latency_target": false + } + ] +} diff --git a/docs/performance/runs/2026-09-23-adhoc-shard-10-ramp-33c95bf/run.json b/docs/performance/runs/2026-09-23-adhoc-shard-10-ramp-33c95bf/run.json new file mode 100644 index 0000000..8de2b7c --- /dev/null +++ b/docs/performance/runs/2026-09-23-adhoc-shard-10-ramp-33c95bf/run.json @@ -0,0 +1,76 @@ +{ + "run_id": "2026-09-23-adhoc-shard-10-ramp-33c95bf", + "created_at": "2026-09-23T01:55:00Z", + "revision": "33c95bf74b5c", + "purpose": "Find the highest arrival rate at which 1000 share-group members x concurrency 10 keep north-star queue time near the Redis reference (36 / 77 / 100 ms), ramping arrival at 8, 16, 32, and 64 ready partitions with the record-lock window fixed at 4000.", + "scenario": null, + "profile": null, + "runner": "../2026-09-23-adhoc-lean-probe-capacity-33c95bf/probe/main.go via probe/run.sh with COMPOSE=../2026-09-23-adhoc-share-group-1000-members-33c95bf/compose.yaml SETTLE=40", + "command": "COMPOSE=../2026-09-23-adhoc-share-group-1000-members-33c95bf/compose.yaml SETTLE=40 ../2026-09-23-adhoc-lean-probe-capacity-33c95bf/probe/run.sh