From 3bedaa9494625b0499cef98c06cae2ce11f91396 Mon Sep 17 00:00:00 2001 From: Ray Liu Date: Sun, 20 Sep 2026 19:59:49 -0400 Subject: [PATCH] [8/9][load] publish worker concurrency comparison --- .../analysis.md | 95 +++++++ .../result.json | 238 ++++++++++++++++++ .../run.json | 80 ++++++ 3 files changed, 413 insertions(+) create mode 100644 docs/performance/runs/2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923/analysis.md create mode 100644 docs/performance/runs/2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923/result.json create mode 100644 docs/performance/runs/2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923/run.json diff --git a/docs/performance/runs/2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923/analysis.md b/docs/performance/runs/2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923/analysis.md new file mode 100644 index 0000000..cbc391d --- /dev/null +++ b/docs/performance/runs/2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923/analysis.md @@ -0,0 +1,95 @@ +# Run Analysis + +Run: `2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923` + +Results: [result.json](result.json) + +This document is generated by AI from the run metadata and machine results. + +## Purpose + +Measure the effect of fixed in-process worker concurrency while holding worker +processes, producers, movers, partitions, workload distribution, and Kafka +environment constant. + +## Observations + +| Concurrency | Requested RPS | Worker RPS | Queue p95 | Drain | +| ---: | ---: | ---: | ---: | ---: | +| 1 | 200 | 200.5 | 5.4 ms | 1 ms | +| 1 | 300 | 300.7 | 72.8 ms | 7 ms | +| 1 | 325 | 314.3 | 377.8 ms | 367 ms | +| 1 | 400 | 309.6 | 2877.8 ms | 2.96 s | +| 1 | 800 | 285.5 | 17286.3 ms | 18.06 s | +| 2 | 200 | 200.5 | 2.4 ms | 2 ms | +| 2 | 300 | 300.6 | 6.4 ms | 6 ms | +| 2 | 325 | 325.3 | 7.0 ms | 12 ms | +| 2 | 400 | 400.6 | 86.1 ms | 11 ms | +| 2 | 800 | 458.6 | 7086.1 ms | 7.47 s | +| 4 | 200 | 200.6 | 2.6 ms | 1 ms | +| 4 | 300 | 300.6 | 5.4 ms | 7 ms | +| 4 | 325 | 325.4 | 5.9 ms | 15 ms | +| 4 | 400 | 401.0 | 7.1 ms | 11 ms | +| 4 | 800 | 756.1 | 547.0 ms | 607 ms | +| 8 | 200 | 200.6 | 3.0 ms | 2 ms | +| 8 | 300 | 300.7 | 5.5 ms | 6 ms | +| 8 | 325 | 325.5 | 5.9 ms | 14 ms | +| 8 | 400 | 400.8 | 6.8 ms | 11 ms | +| 8 | 800 | 801.2 | 10.9 ms | 11 ms | + +- Concurrency 1 reproduced the serial knee: 300 RPS was sustainable and + 325 RPS began sustained backlog growth. +- Concurrency 2 sustained 400 RPS, but saturated below 800 RPS at approximately + 459 handler completions per second. +- Concurrency 4 reached approximately 756 handler completions per second at the + 800 RPS point, leaving 547 ms queue p95 and a 607 ms drain. +- Concurrency 8 sustained the highest tested arrival rate of 800 RPS with + 10.9 ms queue p95 and an 11 ms drain. +- At 800 requested RPS, concurrency 8 delivered 2.81 times the observed worker + throughput of concurrency 1 and reduced queue p95 by 99.94%. +- All 81,000 measured tasks completed with zero missing, duplicate, + outstanding, or dead-lettered workload IDs. + +## Interpretation + +- Serial handler execution remains the dominant ready-path bottleneck for this + workload. A small fixed pool removes it without increasing worker process + count. +- Concurrency 2 is sufficient through 400 RPS for this execution-time + distribution. Concurrency 8 is required to demonstrate headroom at 800 RPS. +- Throughput does not scale linearly at the overloaded points. Generation + barriers, acknowledgement round trips, Kafka fetch behavior, and the workload + duration all contribute to the observed shape. +- The current lognormal workload specifies a 5 ms average and 20 ms p99. Across + these points, measured execution p95 ranged from 12.3 to 13.0 ms and measured + p99 ranged from 18.9 to 21.0 ms. +- The fixed-generation design is sufficient for this short-tailed workload. A + heavier-tailed workload is still needed before deciding whether continuous + refill justifies acquisition-renewal complexity. + +## Suggestions + +1. Use a modest explicit concurrency based on expected handler latency and + downstream capacity; do not default every deployment to 8 solely from this + local result. +2. Add a heavy-tail comparison where one task in a generation is much slower + than its peers. Use that evidence to decide whether continuous refill is the + next worker change. +3. Add process CPU and memory telemetry before comparing efficiency rather than + throughput and latency alone. +4. Add acquisition renewal before supporting handlers that may approach the + Kafka share-record lock duration. + +## Caveats + +- Each point ran once on a local arm64 laptop with one Dockerized Kafka broker, + one ready partition, two producers, two worker processes, and one retry mover. +- The highest tested arrival rate was 800 RPS, so this run establishes only + that concurrency 8 sustains at least 800 RPS; it does not locate its capacity + ceiling. +- The harness does not collect worker-process CPU or memory. The environment + metadata describes the runner, not per-process resource consumption. +- Fixed generations wait for the slowest task before polling again. This + workload's short tail does not stress that limitation. +- The implementation does not renew record acquisitions. The measured handlers + remain far below Kafka's acquisition lock duration. diff --git a/docs/performance/runs/2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923/result.json b/docs/performance/runs/2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923/result.json new file mode 100644 index 0000000..28aa3e2 --- /dev/null +++ b/docs/performance/runs/2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923/result.json @@ -0,0 +1,238 @@ +{ + "capacity": [ + { + "worker_concurrency": 1, + "highest_sustainable_tested_rps": 300, + "first_overloaded_tested_rps": 325, + "maximum_observed_active_worker_rps": 314.274 + }, + { + "worker_concurrency": 2, + "highest_sustainable_tested_rps": 400, + "first_overloaded_tested_rps": 800, + "maximum_observed_active_worker_rps": 458.645 + }, + { + "worker_concurrency": 4, + "highest_sustainable_tested_rps": 400, + "first_overloaded_tested_rps": 800, + "maximum_observed_active_worker_rps": 756.119 + }, + { + "worker_concurrency": 8, + "highest_sustainable_tested_rps": 800, + "first_overloaded_tested_rps": null, + "maximum_observed_active_worker_rps": 801.179 + } + ], + "correctness": { + "completed": 81000, + "dead_lettered": 0, + "duplicates": 0, + "missing": 0, + "outstanding": 0, + "produced": 81000 + }, + "sweep": [ + { + "worker_concurrency": 1, + "requested_rps": 200, + "achieved_enqueue_rps": 200.038, + "active_worker_rps": 200.477, + "queue_p95_ms": 5.379, + "delivery_p95_ms": 15.605, + "drain_ms": 1.382, + "source_run": "2026-09-21-ready-success-v1-00bb923-2" + }, + { + "worker_concurrency": 1, + "requested_rps": 300, + "achieved_enqueue_rps": 300.033, + "active_worker_rps": 300.691, + "queue_p95_ms": 72.826, + "delivery_p95_ms": 77.846, + "drain_ms": 7.076, + "source_run": "2026-09-21-ready-success-v1-00bb923-3" + }, + { + "worker_concurrency": 1, + "requested_rps": 325, + "achieved_enqueue_rps": 325.066, + "active_worker_rps": 314.274, + "queue_p95_ms": 377.783, + "delivery_p95_ms": 382.422, + "drain_ms": 367.281, + "source_run": "2026-09-21-ready-success-v1-00bb923-4" + }, + { + "worker_concurrency": 1, + "requested_rps": 400, + "achieved_enqueue_rps": 400.068, + "active_worker_rps": 309.563, + "queue_p95_ms": 2877.831, + "delivery_p95_ms": 2880.632, + "drain_ms": 2958.539, + "source_run": "2026-09-21-ready-success-v1-00bb923-5" + }, + { + "worker_concurrency": 1, + "requested_rps": 800, + "achieved_enqueue_rps": 800.012, + "active_worker_rps": 285.46, + "queue_p95_ms": 17286.266, + "delivery_p95_ms": 17288.53, + "drain_ms": 18063.615, + "source_run": "2026-09-21-ready-success-v1-00bb923-6" + }, + { + "worker_concurrency": 2, + "requested_rps": 200, + "achieved_enqueue_rps": 200.063, + "active_worker_rps": 200.479, + "queue_p95_ms": 2.419, + "delivery_p95_ms": 13.624, + "drain_ms": 1.664, + "source_run": "2026-09-21-ready-success-v1-00bb923-7" + }, + { + "worker_concurrency": 2, + "requested_rps": 300, + "achieved_enqueue_rps": 300.024, + "active_worker_rps": 300.625, + "queue_p95_ms": 6.427, + "delivery_p95_ms": 15.536, + "drain_ms": 6.398, + "source_run": "2026-09-21-ready-success-v1-00bb923-8" + }, + { + "worker_concurrency": 2, + "requested_rps": 325, + "achieved_enqueue_rps": 324.96, + "active_worker_rps": 325.346, + "queue_p95_ms": 7.015, + "delivery_p95_ms": 15.585, + "drain_ms": 11.519, + "source_run": "2026-09-21-ready-success-v1-00bb923-9" + }, + { + "worker_concurrency": 2, + "requested_rps": 400, + "achieved_enqueue_rps": 399.99, + "active_worker_rps": 400.606, + "queue_p95_ms": 86.076, + "delivery_p95_ms": 91.467, + "drain_ms": 11.097, + "source_run": "2026-09-21-ready-success-v1-00bb923-10" + }, + { + "worker_concurrency": 2, + "requested_rps": 800, + "achieved_enqueue_rps": 800.047, + "active_worker_rps": 458.645, + "queue_p95_ms": 7086.146, + "delivery_p95_ms": 7090.451, + "drain_ms": 7466.942, + "source_run": "2026-09-21-ready-success-v1-00bb923-11" + }, + { + "worker_concurrency": 4, + "requested_rps": 200, + "achieved_enqueue_rps": 200.067, + "active_worker_rps": 200.573, + "queue_p95_ms": 2.621, + "delivery_p95_ms": 13.404, + "drain_ms": 1.412, + "source_run": "2026-09-21-ready-success-v1-00bb923-12" + }, + { + "worker_concurrency": 4, + "requested_rps": 300, + "achieved_enqueue_rps": 300.028, + "active_worker_rps": 300.604, + "queue_p95_ms": 5.359, + "delivery_p95_ms": 14.868, + "drain_ms": 6.5, + "source_run": "2026-09-21-ready-success-v1-00bb923-13" + }, + { + "worker_concurrency": 4, + "requested_rps": 325, + "achieved_enqueue_rps": 325.041, + "active_worker_rps": 325.386, + "queue_p95_ms": 5.87, + "delivery_p95_ms": 15.12, + "drain_ms": 14.979, + "source_run": "2026-09-21-ready-success-v1-00bb923-14" + }, + { + "worker_concurrency": 4, + "requested_rps": 400, + "achieved_enqueue_rps": 400.011, + "active_worker_rps": 400.989, + "queue_p95_ms": 7.103, + "delivery_p95_ms": 16.218, + "drain_ms": 10.768, + "source_run": "2026-09-21-ready-success-v1-00bb923-15" + }, + { + "worker_concurrency": 4, + "requested_rps": 800, + "achieved_enqueue_rps": 799.902, + "active_worker_rps": 756.119, + "queue_p95_ms": 547.007, + "delivery_p95_ms": 551.424, + "drain_ms": 607.121, + "source_run": "2026-09-21-ready-success-v1-00bb923-16" + }, + { + "worker_concurrency": 8, + "requested_rps": 200, + "achieved_enqueue_rps": 200.085, + "active_worker_rps": 200.593, + "queue_p95_ms": 3.007, + "delivery_p95_ms": 13.95, + "drain_ms": 1.509, + "source_run": "2026-09-21-ready-success-v1-00bb923-17" + }, + { + "worker_concurrency": 8, + "requested_rps": 300, + "achieved_enqueue_rps": 300.036, + "active_worker_rps": 300.68, + "queue_p95_ms": 5.504, + "delivery_p95_ms": 14.716, + "drain_ms": 5.83, + "source_run": "2026-09-21-ready-success-v1-00bb923-18" + }, + { + "worker_concurrency": 8, + "requested_rps": 325, + "achieved_enqueue_rps": 325.055, + "active_worker_rps": 325.451, + "queue_p95_ms": 5.92, + "delivery_p95_ms": 14.953, + "drain_ms": 14.372, + "source_run": "2026-09-21-ready-success-v1-00bb923-19" + }, + { + "worker_concurrency": 8, + "requested_rps": 400, + "achieved_enqueue_rps": 400.0, + "active_worker_rps": 400.798, + "queue_p95_ms": 6.761, + "delivery_p95_ms": 15.425, + "drain_ms": 10.7, + "source_run": "2026-09-21-ready-success-v1-00bb923-20" + }, + { + "worker_concurrency": 8, + "requested_rps": 800, + "achieved_enqueue_rps": 799.948, + "active_worker_rps": 801.179, + "queue_p95_ms": 10.893, + "delivery_p95_ms": 18.742, + "drain_ms": 10.886, + "source_run": "2026-09-21-ready-success-v1-00bb923-21" + } + ] +} diff --git a/docs/performance/runs/2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923/run.json b/docs/performance/runs/2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923/run.json new file mode 100644 index 0000000..b74b41e --- /dev/null +++ b/docs/performance/runs/2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923/run.json @@ -0,0 +1,80 @@ +{ + "run_id": "2026-09-21-ready-success-v1-worker-concurrency-sweep-00bb923", + "created_at": "2026-09-21T00:31:04Z", + "revision": "00bb923fc477", + "purpose": "Measure ready-path throughput and latency as fixed in-process worker concurrency increases.", + "scenario": { + "id": "ready-success-v1", + "version": 1 + }, + "profile": "local-baseline", + "environment": { + "architecture": "arm64", + "ci_provider": "", + "cpu_count": 12, + "go_version": "go version go1.27.0 darwin/arm64", + "kafka_version": "apache/kafka:4.3.1", + "memory_bytes": 22544384, + "operating_system": "Darwin", + "python_version": "3.13.9", + "runner": "rays-MacBook-Pro.local" + }, + "topology": { + "movers": 1, + "producers": 2, + "ready_partitions": 1, + "retry_partitions": 4, + "worker_concurrency_sweep": [ + 1, + 2, + 4, + 8 + ], + "workers": 2 + }, + "workload": { + "arrival": { + "duration_seconds": 10, + "target_rps_sweep": [ + 200, + 300, + 325, + 400, + 800 + ] + }, + "execution_time_ms": { + "average": 5, + "p99": 20 + }, + "failures": { + "mode": "none", + "rate": 0 + }, + "random_seed": 42, + "warmup_tasks": 20 + }, + "source_runs": [ + "2026-09-21-ready-success-v1-00bb923-2", + "2026-09-21-ready-success-v1-00bb923-3", + "2026-09-21-ready-success-v1-00bb923-4", + "2026-09-21-ready-success-v1-00bb923-5", + "2026-09-21-ready-success-v1-00bb923-6", + "2026-09-21-ready-success-v1-00bb923-7", + "2026-09-21-ready-success-v1-00bb923-8", + "2026-09-21-ready-success-v1-00bb923-9", + "2026-09-21-ready-success-v1-00bb923-10", + "2026-09-21-ready-success-v1-00bb923-11", + "2026-09-21-ready-success-v1-00bb923-12", + "2026-09-21-ready-success-v1-00bb923-13", + "2026-09-21-ready-success-v1-00bb923-14", + "2026-09-21-ready-success-v1-00bb923-15", + "2026-09-21-ready-success-v1-00bb923-16", + "2026-09-21-ready-success-v1-00bb923-17", + "2026-09-21-ready-success-v1-00bb923-18", + "2026-09-21-ready-success-v1-00bb923-19", + "2026-09-21-ready-success-v1-00bb923-20", + "2026-09-21-ready-success-v1-00bb923-21" + ], + "notes": "All source runs use revision 00bb923fc477. Only topology.worker_concurrency and workload.arrival.target_rps varied." +}