From 7ef5a8e2ec05e20c4ef0b321cb472975df86f55f Mon Sep 17 00:00:00 2001 From: Ray Liu Date: Wed, 23 Sep 2026 00:41:20 -0400 Subject: [PATCH] [4/6][load] publish 1000-member share group run --- .../analysis.md | 14 + .../probe/run.sh | 6 +- .../analysis.md | 14 + .../analysis.md | 122 ++++ .../compose.yaml | 24 + .../result.json | 541 ++++++++++++++++++ .../run.json | 62 ++ .../analysis.md | 14 + 8 files changed, 794 insertions(+), 3 deletions(-) create mode 100644 docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/analysis.md create mode 100644 docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/compose.yaml create mode 100644 docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/result.json create mode 100644 docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/run.json diff --git a/docs/performance/runs/2026-09-23-adhoc-lean-probe-capacity-33c95bf/analysis.md b/docs/performance/runs/2026-09-23-adhoc-lean-probe-capacity-33c95bf/analysis.md index 125a899..c5f5542 100644 --- a/docs/performance/runs/2026-09-23-adhoc-lean-probe-capacity-33c95bf/analysis.md +++ b/docs/performance/runs/2026-09-23-adhoc-lean-probe-capacity-33c95bf/analysis.md @@ -9,6 +9,20 @@ Probe: [probe/main.go](probe/main.go), [probe/run.sh](probe/run.sh) This document is generated by AI from the run metadata, machine results, and retained diagnostics. +## Sequence + +This is the second of four runs published on 2026-09-23: + +1. [Harness capacity](../2026-09-23-north-star-steady-v1-harness-capacity-33c95bf/analysis.md): + `kq-bench` saturated the host near 13,000 tasks/s, mostly on its own + producer and Python accounting overhead. +2. **This run.** Remove the harness overhead and find kq's own ceiling. +3. [Queue-latency diagnosis](../2026-09-23-adhoc-queue-latency-diagnosis-33c95bf/analysis.md): + explain the flat roughly 465 ms queue p50 observed here. +4. [Share group with 1000 members](../2026-09-23-adhoc-share-group-1000-members-33c95bf/analysis.md): + revisit this run's 20,000 RPS concurrency series with the member cap + raised from 200 to 1000. + ## Purpose Measure kq's own capacity ceiling on the north-star execution-time diff --git a/docs/performance/runs/2026-09-23-adhoc-lean-probe-capacity-33c95bf/probe/run.sh b/docs/performance/runs/2026-09-23-adhoc-lean-probe-capacity-33c95bf/probe/run.sh index 6f151a3..e0ef8df 100755 --- a/docs/performance/runs/2026-09-23-adhoc-lean-probe-capacity-33c95bf/probe/run.sh +++ b/docs/performance/runs/2026-09-23-adhoc-lean-probe-capacity-33c95bf/probe/run.sh @@ -1,6 +1,6 @@ #!/bin/zsh # usage: run.sh LABEL RPS DUR PARTITIONS WORKER_PROCS WORKERS_PER_PROC CONCURRENCY [PRODUCER_CLIENTS] [GOROUTINES] -# env: EXEC=northstar|const: OUT= +# env: EXEC=northstar|const: OUT= COMPOSE= SETTLE= set -u HERE=${0:A:h} REPO=${HERE:h:h:h:h:h} @@ -8,14 +8,14 @@ LABEL=$1 RPS=$2 DUR=$3 PART=$4 WP=$5 WPP=$6 C=$7 PC=${8:-2} PG=${9:-512} EXEC=${ D=${OUT:-${TMPDIR:-/tmp}/kq-probe}/$LABEL; rm -rf $D; mkdir -p $D BIN=$D/probe (cd $REPO && go build -o $BIN $HERE/main.go) -KC=(docker compose -f $REPO/bench/compose.yaml -p kqprobe) +KC=(docker compose -f ${COMPOSE:-$REPO/bench/compose.yaml} -p kqprobe) $KC up -d --wait >/dev/null 2>&1 $KC exec -T kafka /opt/kafka/bin/kafka-topics.sh --bootstrap-server localhost:19092 --create --topic probe-ready --partitions $PART --replication-factor 1 >/dev/null $KC exec -T kafka /opt/kafka/bin/kafka-configs.sh --bootstrap-server localhost:19092 --alter --entity-type groups --entity-name kq.probe.workers --add-config share.auto.offset.reset=earliest >/dev/null for i in $(seq 0 $((WP-1))); do $BIN work -workers $WPP -concurrency $C -exec $EXEC -out $D/ids-$i.bin > $D/work-$i.out 2> $D/work-$i.err & done -sleep 12 # let share group assignment settle +sleep ${SETTLE:-12} # let share group assignment settle $BIN produce -clients $PC -goroutines $PG -rps $RPS -dur ${DUR}s > $D/produce.out 2>&1 wait echo "== $LABEL rps=$RPS dur=$DUR part=$PART procs=$WP x workers=$WPP x C=$C exec=$EXEC" diff --git a/docs/performance/runs/2026-09-23-adhoc-queue-latency-diagnosis-33c95bf/analysis.md b/docs/performance/runs/2026-09-23-adhoc-queue-latency-diagnosis-33c95bf/analysis.md index c016349..e647e0f 100644 --- a/docs/performance/runs/2026-09-23-adhoc-queue-latency-diagnosis-33c95bf/analysis.md +++ b/docs/performance/runs/2026-09-23-adhoc-queue-latency-diagnosis-33c95bf/analysis.md @@ -7,6 +7,20 @@ Results: [result.json](result.json) This document is generated by AI from the run metadata, machine results, and retained diagnostics. +## Sequence + +This is the third of four runs published on 2026-09-23: + +1. [Harness capacity](../2026-09-23-north-star-steady-v1-harness-capacity-33c95bf/analysis.md): + `kq-bench` saturated the host near 13,000 tasks/s. +2. [Lean-probe capacity](../2026-09-23-adhoc-lean-probe-capacity-33c95bf/analysis.md): + kq reached 150,000 tasks/s, but queue p50 stayed near 465 ms regardless of + load at concurrency 1000. +3. **This run.** Find what causes that queue time. +4. [Share group with 1000 members](../2026-09-23-adhoc-share-group-1000-members-33c95bf/analysis.md): + follows suggestion 2 below by raising the member cap to 1000 and running + 1000 members x concurrency 10 at 20,000 RPS. + ## Purpose Explain the roughly 465 ms queue p50 observed at worker concurrency 1000 by diff --git a/docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/analysis.md b/docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/analysis.md new file mode 100644 index 0000000..43e5499 --- /dev/null +++ b/docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/analysis.md @@ -0,0 +1,122 @@ +# Run Analysis + +Run: [run.json](run.json) + +Results: [result.json](result.json) + +Kafka: [compose.yaml](compose.yaml) (`bench/compose.yaml` plus +`group.share.max.size=1000` and `group.share.partition.max.record.locks=4000`) + +This document is generated by AI from the run metadata, machine results, and +retained diagnostics. + +## Sequence + +This is the fourth and latest of four runs published on 2026-09-23: + +1. [Harness capacity](../2026-09-23-north-star-steady-v1-harness-capacity-33c95bf/analysis.md): + `kq-bench` saturated the host near 13,000 tasks/s. +2. [Lean-probe capacity](../2026-09-23-adhoc-lean-probe-capacity-33c95bf/analysis.md): + kq reached 150,000 tasks/s, but queue p50 stayed near 465 ms. +3. [Queue-latency diagnosis](../2026-09-23-adhoc-queue-latency-diagnosis-33c95bf/analysis.md): + queue time comes from the worker generation wait; many small generations + are fast, but the default 200-member cap limits their throughput. +4. **This run.** Raise the member cap to 1000 and test whether 1000 small + generations sustain 20,000 RPS with low queue time. + +## Purpose + +Test whether raising the share-group member limit to its maximum (1000) lets +many small worker generations (1000 members x concurrency 10, emulating a +sharded consumer) sustain 20,000 RPS with low queue time, and measure +sensitivity to ready partition count. + +## Observations + +20,000 RPS for 60 s (1,200,000 tasks). 10 processes x 100 workers x +concurrency 10 = 1000 share-group members and 10,000 slots: + +| Ready partitions | Outcome | Median / peak completions/s | Queue p50 / p95 / p99 / max | Kafka / worker cores | +| ---: | --- | ---: | ---: | ---: | +| 2 | collapsed; 60,959 completed | 256 / 7,274 | 3,627 / 56,544 / 68,915 / 72,032 ms | 7.21 / 0.25 | +| 4 | collapsed; 230,552 completed | 1,136 / 16,123 | 2,354 / 40,253 / 55,050 / 60,916 ms | 7.56 / 0.29 | +| 8 | sustained | 20,027 / 21,002 | 14 / 104 / 203 / 621 ms | 5.80 / 1.62 | +| 16 | sustained | 20,012 / 20,933 | 24 / 89 / 179 / 1,170 ms | 5.88 / 1.63 | +| 32 | sustained | 20,032 / 20,896 | 43 / 158 / 272 / 665 ms | 5.82 / 1.62 | + +- `group.share.max.size=1000` is the broker maximum; 1001 is rejected at + startup. `group.share.partition.max.record.locks` was raised to 4000, the + default value of its upper bound + (`group.share.max.partition.max.record.locks`). +- The member cap applies to the whole share group regardless of partition + count. +- With 2 and 4 partitions, completions peaked in the first seconds of + production (7,274/s and 16,123/s), then decayed continuously to about + 200/s and 520/s while producers kept enqueuing 20,000/s. Completions + stopped entirely after about 72 to 76 s. Kafka used 7.2 to 7.6 cores. +- With 8, 16, and 32 partitions, all 3,600,000 tasks completed with zero + duplicates and zero missing IDs, and completions matched arrival. +- For comparison at 20,000 RPS: the lean-probe run's best result under the + default 200-member cap (180 members x concurrency 100) was + 115 / 410 / 1,291 ms queue p50 / p95 / p99. +- 1000 members cost about 1 more Kafka core and 1 more worker core than the + 32-member x concurrency 1000 topology in the lean-probe run (4.81 / 0.67). + +## Interpretation + +- Many small generations work once the member cap is raised. At 8 or 16 + partitions, 1000 members x concurrency 10 sustained the 20,000 RPS north-star + rate with queue p50 of 14 to 24 ms and p99 of 179 to 203 ms, roughly an + order of magnitude better than the 200-member result. +- Partition count bounds throughput independently of member count. + Hypothesis: a share partition's in-flight window runs from its oldest + unacknowledged record and holds at most `record.locks` records, so each + partition sustains roughly `4000 / time-from-acquire-to-ack of the oldest + record`. That time is about 1 s on this workload, which gives about 8,000/s + for 2 partitions and 16,000/s for 4. Those match the observed peaks of + 7,274 and 16,123 but are not confirmed from broker metrics. +- Exceeding that bound caused collapse rather than a plateau. Hypothesis: + most member fetches return few or no records and are retried at once. That + raises broker load, which slows acknowledgement and window advance, which + lowers the bound further. Rising Kafka CPU during falling throughput is + consistent with this. Why completions stopped entirely is unexplained. +- 32 partitions had higher latency than 8 or 16. Hypothesis: each member's + 10-record fetches are spread across more partitions and fetch sessions. + Untested. +- Each member still waits for its generation of 10, so the remaining p99 + (about 180 to 200 ms) likely includes the generation wait. The + acquisition-while-busy hypothesis from run 3 is also unresolved. +- At 1000 members the cap now binds at about 20,000 to 25,000 tasks/s per + queue for concurrency 10 on this workload. Higher rates need larger + generations (worse tail latency), multiple queues, or a worker that refills + continuously. + +## Suggestions + +1. Design the sharded consumer around 1000 members, about 10 slots per shard, + and a partition count sized so that + `partitions x 4000 / expected acquire-to-ack time` comfortably exceeds peak + arrival. +2. Document partition sizing as a correctness-of-operation requirement; the + failure mode is collapse, not gradual degradation. +3. Confirm the in-flight window model before relying on it: run 2 partitions + at 5,000 RPS (below the estimated bound) and at 20,000 RPS with 200 members, + and capture broker share-partition metrics during a collapse. +4. Sweep shard concurrency (5, 10, 20) at 8 and 16 partitions to locate the + latency optimum. +5. Add these share-group settings and a partition-sizing check to the + eventual GCP benchmark, since multi-broker behavior may differ. + +## Caveats + +- Single Dockerized Kafka broker on localhost, RF=1; one run per point with a + 60 s arrival window. +- Same probe as runs 2 and 3: ready success path only, sleep-only handlers, + 200-byte payloads, queue time measured from just before `Enqueue`. +- For collapsed points, `correctness.missing` counts tasks not completed when + the probe exited, including tasks still in Kafka. Worker processes exit after + 15 s without completions, so these points do not show whether the system + would have recovered. +- 1000 members ran as 1000 `kq.Worker` instances across 10 processes; a real + sharded consumer inside one worker may have different client and connection + overhead. diff --git a/docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/compose.yaml b/docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/compose.yaml new file mode 100644 index 0000000..19b4ef0 --- /dev/null +++ b/docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/compose.yaml @@ -0,0 +1,24 @@ +services: + kafka: + image: apache/kafka:4.3.1 + ports: + - "19092:19092" + environment: + KAFKA_NODE_ID: 1 + KAFKA_PROCESS_ROLES: broker,controller + KAFKA_LISTENERS: PLAINTEXT://:19092,CONTROLLER://:19093 + KAFKA_ADVERTISED_LISTENERS: PLAINTEXT://localhost:19092 + KAFKA_CONTROLLER_LISTENER_NAMES: CONTROLLER + KAFKA_LISTENER_SECURITY_PROTOCOL_MAP: CONTROLLER:PLAINTEXT,PLAINTEXT:PLAINTEXT + KAFKA_CONTROLLER_QUORUM_VOTERS: 1@localhost:19093 + KAFKA_OFFSETS_TOPIC_REPLICATION_FACTOR: 1 + KAFKA_SHARE_COORDINATOR_STATE_TOPIC_REPLICATION_FACTOR: 1 + KAFKA_SHARE_COORDINATOR_STATE_TOPIC_MIN_ISR: 1 + KAFKA_GROUP_SHARE_MIN_RECORD_LOCK_DURATION_MS: 1000 + KAFKA_GROUP_SHARE_MAX_SIZE: 1000 + KAFKA_GROUP_SHARE_PARTITION_MAX_RECORD_LOCKS: 4000 + healthcheck: + test: ["CMD-SHELL", "/opt/kafka/bin/kafka-broker-api-versions.sh --bootstrap-server localhost:19092 >/dev/null 2>&1"] + interval: 2s + timeout: 5s + retries: 30 diff --git a/docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/result.json b/docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/result.json new file mode 100644 index 0000000..b16b8c2 --- /dev/null +++ b/docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/result.json @@ -0,0 +1,541 @@ +{ + "correctness": { + "produced": 3600000, + "completed": 3600000, + "duplicates": 0, + "missing": 0 + }, + "collapsed_points": [ + "m1000-p2", + "m1000-p4" + ], + "points": [ + { + "label": "m1000-p2", + "requested_rps": 20000, + "duration_seconds": 60, + "ready_partitions": 2, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 2, + "enqueue_goroutines_per_client": 512, + "execution": "northstar", + "achieved_enqueue_rps": 20000, + "enqueue_errors": 0, + "completion_rps_median_1s": 256, + "completion_rps_max_1s": 7274, + "seconds_with_completions_after_production": null, + "queue_p50_ms": 3627, + "queue_p95_ms": 56544, + "queue_p99_ms": 68915, + "queue_max_ms": 72032, + "correctness": { + "produced": 1200000, + "completed": 60959, + "duplicates": 0, + "missing": 1142165 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 7.21, + "producer": 1.19, + "workers": 0.25 + }, + "completion_rps_series_1s": [ + 7274, + 5144, + 6638, + 5355, + 4457, + 3029, + 2386, + 1861, + 1512, + 1392, + 1195, + 1012, + 996, + 856, + 796, + 707, + 626, + 588, + 559, + 529, + 515, + 462, + 454, + 447, + 405, + 427, + 385, + 384, + 312, + 323, + 326, + 292, + 259, + 285, + 279, + 270, + 254, + 270, + 266, + 246, + 219, + 233, + 190, + 221, + 218, + 186, + 219, + 222, + 218, + 215, + 216, + 181, + 186, + 197, + 225, + 198, + 194, + 228, + 217, + 210, + 194, + 210, + 182, + 195, + 185, + 213, + 193, + 191, + 194, + 180, + 199, + 187, + 204, + 197, + 205, + 64 + ], + "completion_rps_peak_1s": 7274, + "sustained": false, + "unconsumed_tasks_at_exit": 1139041 + }, + { + "label": "m1000-p4", + "requested_rps": 20000, + "duration_seconds": 60, + "ready_partitions": 4, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 2, + "enqueue_goroutines_per_client": 512, + "execution": "northstar", + "achieved_enqueue_rps": 20000, + "enqueue_errors": 0, + "completion_rps_median_1s": 1136, + "completion_rps_max_1s": 16123, + "seconds_with_completions_after_production": null, + "queue_p50_ms": 2354, + "queue_p95_ms": 40253, + "queue_p99_ms": 55050, + "queue_max_ms": 60916, + "correctness": { + "produced": 1200000, + "completed": 230552, + "duplicates": 0, + "missing": 1000339 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 7.56, + "producer": 0.99, + "workers": 0.29 + }, + "completion_rps_series_1s": [ + 7851, + 14960, + 14560, + 11427, + 16123, + 13839, + 15551, + 12523, + 12067, + 9814, + 8698, + 7684, + 6152, + 5493, + 4601, + 3754, + 3777, + 3387, + 2899, + 2760, + 2645, + 2400, + 2261, + 1960, + 1870, + 1787, + 1773, + 1678, + 1514, + 1536, + 1340, + 1307, + 1205, + 1295, + 1247, + 1119, + 1154, + 1090, + 1059, + 1043, + 1017, + 925, + 942, + 921, + 846, + 850, + 780, + 794, + 785, + 764, + 722, + 726, + 690, + 627, + 694, + 680, + 637, + 645, + 636, + 629, + 563, + 582, + 522, + 538, + 539, + 515, + 527, + 532, + 522, + 497, + 520, + 182 + ], + "completion_rps_peak_1s": 16123, + "sustained": false, + "unconsumed_tasks_at_exit": 969448 + }, + { + "label": "m1000-p8", + "requested_rps": 20000, + "duration_seconds": 60, + "ready_partitions": 8, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 2, + "enqueue_goroutines_per_client": 512, + "execution": "northstar", + "achieved_enqueue_rps": 20000, + "enqueue_errors": 0, + "completion_rps_median_1s": 20027, + "completion_rps_max_1s": 21002, + "seconds_with_completions_after_production": 2, + "queue_p50_ms": 14, + "queue_p95_ms": 104, + "queue_p99_ms": 203, + "queue_max_ms": 621, + "correctness": { + "produced": 1200000, + "completed": 1200000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 5.8, + "producer": 1.39, + "workers": 1.62 + }, + "completion_rps_series_1s": [ + 1368, + 9701, + 20683, + 21002, + 20793, + 20695, + 20525, + 20025, + 20068, + 19951, + 20137, + 19817, + 19971, + 20232, + 19869, + 19980, + 19533, + 20259, + 20160, + 20131, + 20072, + 19883, + 19997, + 20032, + 20100, + 19945, + 19842, + 20213, + 19946, + 19837, + 20043, + 20158, + 20006, + 19909, + 19880, + 19958, + 20101, + 20115, + 19840, + 20033, + 20036, + 19943, + 20009, + 19978, + 20100, + 19938, + 20084, + 19999, + 20087, + 19891, + 20081, + 20029, + 20024, + 19737, + 20027, + 20027, + 19928, + 20008, + 20038, + 20041, + 17882, + 7296, + 7 + ], + "completion_rps_peak_1s": 21002, + "sustained": true + }, + { + "label": "m1000-p16", + "requested_rps": 20000, + "duration_seconds": 60, + "ready_partitions": 16, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 2, + "enqueue_goroutines_per_client": 512, + "execution": "northstar", + "achieved_enqueue_rps": 20000, + "enqueue_errors": 0, + "completion_rps_median_1s": 20012, + "completion_rps_max_1s": 20933, + "seconds_with_completions_after_production": 1, + "queue_p50_ms": 24, + "queue_p95_ms": 89, + "queue_p99_ms": 179, + "queue_max_ms": 1170, + "correctness": { + "produced": 1200000, + "completed": 1200000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 5.88, + "producer": 1.43, + "workers": 1.63 + }, + "completion_rps_series_1s": [ + 7668, + 20706, + 20933, + 20524, + 20441, + 20261, + 19968, + 20055, + 20090, + 19929, + 19949, + 20139, + 19840, + 20154, + 19986, + 19984, + 20017, + 19991, + 20005, + 19909, + 19963, + 19970, + 20138, + 20048, + 20035, + 19872, + 20012, + 20086, + 19932, + 19964, + 20084, + 20005, + 19919, + 19984, + 19890, + 20163, + 20019, + 20025, + 19887, + 19987, + 19946, + 20072, + 19980, + 20105, + 19847, + 20126, + 20100, + 19812, + 20102, + 20044, + 20042, + 19853, + 20047, + 20064, + 19936, + 19923, + 20019, + 19883, + 20236, + 19969, + 9355, + 7 + ], + "completion_rps_peak_1s": 20933, + "sustained": true + }, + { + "label": "m1000-p32", + "requested_rps": 20000, + "duration_seconds": 60, + "ready_partitions": 32, + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "total_slots": 10000, + "producer_clients": 2, + "enqueue_goroutines_per_client": 512, + "execution": "northstar", + "achieved_enqueue_rps": 20000, + "enqueue_errors": 0, + "completion_rps_median_1s": 20032, + "completion_rps_max_1s": 20896, + "seconds_with_completions_after_production": 1, + "queue_p50_ms": 43, + "queue_p95_ms": 158, + "queue_p99_ms": 272, + "queue_max_ms": 665, + "correctness": { + "produced": 1200000, + "completed": 1200000, + "duplicates": 0, + "missing": 0 + }, + "cpu_cores_10s_window": { + "kafka_docker_vm": 5.82, + "producer": 1.41, + "workers": 1.62 + }, + "completion_rps_series_1s": [ + 5598, + 20553, + 20896, + 20569, + 20484, + 20371, + 20236, + 20138, + 20173, + 19838, + 20184, + 20010, + 19984, + 20069, + 20093, + 19872, + 20061, + 19938, + 19958, + 20032, + 19973, + 20033, + 19821, + 20152, + 19978, + 19935, + 20028, + 19978, + 20015, + 20039, + 19979, + 20095, + 19956, + 19899, + 19928, + 20008, + 20062, + 20136, + 19878, + 20054, + 20032, + 19768, + 20097, + 20076, + 20062, + 19967, + 20183, + 19780, + 20116, + 19829, + 20173, + 19986, + 20005, + 20101, + 19891, + 20039, + 19976, + 19854, + 20230, + 20005, + 10814, + 12 + ], + "completion_rps_peak_1s": 20896, + "sustained": true + } + ] +} diff --git a/docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/run.json b/docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/run.json new file mode 100644 index 0000000..904f724 --- /dev/null +++ b/docs/performance/runs/2026-09-23-adhoc-share-group-1000-members-33c95bf/run.json @@ -0,0 +1,62 @@ +{ + "run_id": "2026-09-23-adhoc-share-group-1000-members-33c95bf", + "created_at": "2026-09-23T01:28:00Z", + "revision": "33c95bf74b5c", + "purpose": "Test whether raising the share-group member limit to its maximum (1000) lets many small worker generations (1000 members x concurrency 10, emulating a sharded consumer) sustain 20,000 RPS with low queue time, and measure sensitivity to ready partition count.", + "scenario": null, + "profile": null, + "runner": "../2026-09-23-adhoc-lean-probe-capacity-33c95bf/probe/main.go via probe/run.sh with COMPOSE=compose.yaml SETTLE=40", + "command": "COMPOSE=compose.yaml SETTLE=40 ../2026-09-23-adhoc-lean-probe-capacity-33c95bf/probe/run.sh m1000-p 20000 60 10 100 10", + "environment": { + "architecture": "arm64", + "ci_provider": "", + "cpu_count": 12, + "go_version": "go version go1.27.0 darwin/arm64", + "kafka_version": "apache/kafka:4.3.1", + "memory_bytes": 22003712, + "operating_system": "Darwin", + "python_version": "3.13.12", + "runner": "rays-MacBook-Pro.local", + "docker_vm_cpus": 12, + "docker_vm_memory_bytes": 8216776704 + }, + "kafka_overrides": { + "group.share.max.size": 1000, + "group.share.partition.max.record.locks": 4000, + "validation": "group.share.max.size=1001 is rejected at broker startup: 'Value must be no more than 1000'. group.share.max.partition.max.record.locks defaults to 4000, the effective ceiling for record locks." + }, + "topology": { + "ready_partitions_sweep": [ + 2, + 4, + 8, + 16, + 32 + ], + "worker_processes": 10, + "workers_per_process": 100, + "worker_concurrency": 10, + "share_group_members": 1000, + "retry_partitions": 0, + "movers": 0 + }, + "workload": { + "arrival": { + "target_rps": 20000, + "duration_seconds": 60 + }, + "execution_time_ms": { + "average": 250, + "p95": 475, + "p99": 646, + "distribution": "lognormal, sigma=ln(646/475)/(2.326-1.645), mean 250 ms, seeded by task ID" + }, + "failures": { + "rate": 0, + "mode": "none", + "duration_ms": 0 + }, + "payload_bytes": 200 + }, + "notes": "One run per point. Kafka recreated per point from compose.yaml (bench/compose.yaml plus the two overrides). Workers started 40 s before producers. Worker processes exit after 15 s without completions, so collapsed points end with unconsumed tasks still in Kafka." +} diff --git a/docs/performance/runs/2026-09-23-north-star-steady-v1-harness-capacity-33c95bf/analysis.md b/docs/performance/runs/2026-09-23-north-star-steady-v1-harness-capacity-33c95bf/analysis.md index acf7644..04e0914 100644 --- a/docs/performance/runs/2026-09-23-north-star-steady-v1-harness-capacity-33c95bf/analysis.md +++ b/docs/performance/runs/2026-09-23-north-star-steady-v1-harness-capacity-33c95bf/analysis.md @@ -7,6 +7,20 @@ Results: [result.json](result.json) This document is generated by AI from the run metadata, machine results, and retained diagnostics. +## Sequence + +This is the first of four runs published on 2026-09-23: + +1. **This run.** Push `kq-bench` on the north-star workload until something + breaks. +2. [Lean-probe capacity](../2026-09-23-adhoc-lean-probe-capacity-33c95bf/analysis.md): + because this run saturated on harness overhead, measure kq with a minimal + load generator instead. +3. [Queue-latency diagnosis](../2026-09-23-adhoc-queue-latency-diagnosis-33c95bf/analysis.md): + explain the roughly 475 ms queue p50 seen here and in run 2. +4. [Share group with 1000 members](../2026-09-23-adhoc-share-group-1000-members-33c95bf/analysis.md): + test many small worker generations at the raised member cap. + ## Purpose Locate the local capacity ceiling of the `kq-bench` harness on the