From 9d2fdd7d8f8ce25aa562a38c4bf2da56951ec2fe Mon Sep 17 00:00:00 2001 From: Bernardo Heynemann Date: Tue, 8 Sep 2026 09:30:18 -0300 Subject: [PATCH 1/2] fix: stop counting settle time in throughput benchmark window The 1000-subscriber throughput benchmark slept a fixed 500ms inside its timed window, so the reported msg/s figures were understated and varied with benchtime. Replace the sleep with a wait-until-delivered loop: the clock now stops only when all b.N * 1000 deliveries have reached the subscriber handlers, and the reported metric is true end-to-end delivery throughput. A 60s deadline with b.Fatal guards against a stuck delivery path. Measured effect on Apple M4 Pro: direct-goroutines direct-match fan-out goes from ~28M claimed msg/s (publish-side, skewed window) to 9.9-10.7M end-to-end deliveries/s; worker pool delivers 12.4-12.8M/s at this subscriber count. --- throughput_benchmark_test.go | 35 ++++++++++++++++++++++------------- 1 file changed, 22 insertions(+), 13 deletions(-) diff --git a/throughput_benchmark_test.go b/throughput_benchmark_test.go index 6890332..2213674 100644 --- a/throughput_benchmark_test.go +++ b/throughput_benchmark_test.go @@ -1,6 +1,7 @@ package blazesub_test import ( + "runtime" "sync/atomic" "testing" "time" @@ -90,7 +91,9 @@ func BenchmarkThroughputWith1000Subscribers(b *testing.B) { // Start with a fresh count handler.count.Store(0) - // Record start time for our own throughput calculation + // The clock runs until every published message has been delivered + // to all subscriber handlers, so the reported throughput is + // end-to-end delivery throughput, not publish-side dispatch. startTime := time.Now() // Run the benchmark @@ -98,27 +101,33 @@ func BenchmarkThroughputWith1000Subscribers(b *testing.B) { bus.Publish(pubTopic, payload) } - // Wait for message processing to complete (conservative wait time) - time.Sleep(500 * time.Millisecond) + // Wait until every published message reached every subscriber. + target := int64(b.N) * subscriberCount + deadline := time.Now().Add(60 * time.Second) + for handler.count.Load() < target { + if time.Now().After(deadline) { + b.Fatalf( + "delivery incomplete: %d/%d deliveries after 60s", + handler.count.Load(), + target, + ) + } + runtime.Gosched() + } - // Calculate actual throughput (messages per second) + // Calculate actual throughput (message deliveries per second) elapsedSeconds := time.Since(startTime).Seconds() - messagesSent := b.N - messagesReceived := handler.count.Load() - - // Because each message goes to 1000 subscribers, we multiply by 1000 - messagesPerSecond := float64(messagesSent) * subscriberCount / elapsedSeconds + deliveries := handler.count.Load() + messagesPerSecond := float64(deliveries) / elapsedSeconds // Report custom metrics b.ReportMetric(messagesPerSecond, "msg/s") - b.ReportMetric(float64(messagesReceived)/float64(messagesSent*subscriberCount)*100, "delivery_%") // Also log for easy viewing b.Logf( - "Throughput: %.2f msgs/sec, Total: %d sent, %d received in %.2f seconds", + "Throughput: %.2f msgs/sec, Total: %d deliveries in %.2f seconds", messagesPerSecond, - messagesSent*subscriberCount, - messagesReceived, + deliveries, elapsedSeconds, ) }) From 911ac739adfa7f7964d10cd871a6f0c3a73623fe Mon Sep 17 00:00:00 2001 From: Bernardo Heynemann Date: Tue, 8 Sep 2026 09:30:34 -0300 Subject: [PATCH 2/2] docs: replace unsubstantiated performance claims with measured numbers Re-benchmarked BlazeSub against MochiMQTT after fixing the throughput benchmark's timing flaw and rewrote every performance claim in README and PERFORMANCE.md to match measured results on Apple M4 Pro (Go 1.26.6, benchtime=3s, count=3): - Fan-out to 1000 subscribers: 9.9-10.7M deliveries/s (direct goroutines), 12.4-12.8M/s (worker pool), reported as ranges. - vs MochiMQTT in-process routing: ~1.04x faster single publish, ~1.9x faster concurrent publish, 7-11x less memory per publish, but 3.5x slower subscribe/unsubscribe churn. - Added a Tradeoffs section to README stating where MochiMQTT wins. - Dropped the "34% faster than MQTT" latency bullet (no latency benchmark exists to back it) and the false "30-50x faster", "84.7M msg/s", and "95% less memory" claims. - Reworked delivery-mode guidance to reflect the measured crossover: direct goroutines ~2x faster at small counts, worker pool slightly faster at 1000 subscribers. - Added methodology (in-process routing core, no network I/O, fan-out counts deliveries, hardware, date) and a reproduction command. Stale claims remaining in BENCHMARK.md and USER_GUIDE.md are tracked in a follow-up issue. --- PERFORMANCE.md | 48 ++++++++++++++++++++++++++++-------------------- README.md | 36 +++++++++++++++++++++++++++--------- 2 files changed, 55 insertions(+), 29 deletions(-) diff --git a/PERFORMANCE.md b/PERFORMANCE.md index 6feb647..1396257 100644 --- a/PERFORMANCE.md +++ b/PERFORMANCE.md @@ -4,16 +4,28 @@ This document provides performance metrics and recommendations for BlazeSub user ## Message Throughput Benchmarks -Our benchmarks show extraordinary performance, demonstrating BlazeSub's capability to handle high-volume messaging: +All figures below are end-to-end delivery throughput: the clock stops only when every published message has reached every subscriber handler. -| Scenario | Direct Goroutines | Worker Pool | Improvement over MQTT | -| ----------------------- | ----------------- | ---------------- | --------------------- | -| Direct match messages | 84.7 million/sec | 77.1 million/sec | 30-50x faster | -| Wildcard match messages | 83.5 million/sec | 73.8 million/sec | 1,000-5,000x faster | -| Memory usage | ~115 B/op | ~114 B/op | 95% less memory | -| Allocations | 2 allocs/op | 2 allocs/op | 80% fewer allocations | +**Fan-out to 1000 subscribers** (deliveries per second, Apple M4 Pro, Go 1.26.6, 3 runs each): -These results represent the number of message deliveries per second when publishing to 1000 subscribers, demonstrating BlazeSub's exceptional throughput capacity even under high subscription load. +| Scenario | Direct Goroutines | Worker Pool | +| ----------------------- | ----------------- | ---------------- | +| Direct match | 9.9–10.7M/s | 12.4–12.6M/s | +| Wildcard match | 9.8–10.5M/s | 12.7–12.8M/s | + +**vs MochiMQTT** (5000 subscriptions across 1000 topics, 20% wildcard-matching publishes; in-process routing with no network I/O — this measures routing speed, not end-to-end broker throughput): + +| Benchmark | BlazeSub (best) | MochiMQTT | BlazeSub advantage | +| ----------------------- | ----------------- | ---------------- | ----------------------------- | +| Single publish | 1,396 ns/op | 1,442 ns/op | ~1.04x faster, 7x less memory | +| Concurrent publish | 325 ns/op | 637 ns/op | ~1.9x faster, 11x less memory | +| Subscribe/unsubscribe | 5,195 ns/op | 1,487 ns/op | 3.5x slower | + +Reproduce with: + +```bash +go test -run='^$' -bench 'BenchmarkThroughputWith1000Subscribers|BenchmarkBusVsMQTT$|BenchmarkBusVsMQTTConcurrent|BenchmarkBusVsMQTTSubscribeUnsubscribe' -benchtime=3s -benchmem -count=3 . +``` ## Delivery Mode Comparison @@ -23,9 +35,8 @@ BlazeSub offers two delivery modes, each with different performance characterist **Advantages:** -- Up to 84.7 million messages/second to 1000 subscribers -- 10-13% faster than worker pool mode -- Lowest possible latency +- Up to 10.7 million message deliveries/second to 1000 subscribers +- ~2x faster than the worker pool at small subscriber counts; at 1000 subscribers the worker pool delivers slightly faster **Best for:** @@ -52,7 +63,7 @@ bus, err := blazesub.NewBusOf[MyCustomType](config) **Advantages:** -- Still delivers 77.1 million messages/second to 1000 subscribers +- Up to 12.8 million message deliveries/second to 1000 subscribers - Better resource management - Protection against goroutine explosion @@ -99,8 +110,8 @@ BlazeSub is designed for minimal memory usage. To optimize further: 1. **Worker Pool vs Direct Goroutines**: - - Direct goroutines: Faster but creates more goroutines - - Worker pool: Slightly slower but better memory management + - Direct goroutines: ~2x faster at small subscriber counts, slightly slower at 1000 subscribers + - Worker pool: the fast fan-out path at 1000 subscribers and predictable resource usage 2. **Topic Design Impact**: @@ -187,16 +198,13 @@ bus, err := blazesub.NewBusOf[CompactEvent](config) ## Performance Scaling -BlazeSub performance scales with different workloads: +We re-measured scaling up to 1000 subscribers; larger counts are not claimed because we have not benchmarked them. | Subscribers | Direct Goroutines | Worker Pool | | ----------- | ----------------- | ---------------- | -| 10 | 95.1 million/sec | 92.5 million/sec | -| 100 | 92.8 million/sec | 88.3 million/sec | -| 1,000 | 84.7 million/sec | 77.1 million/sec | -| 10,000 | 62.3 million/sec | 51.9 million/sec | +| 1,000 | 9.9–10.7M/s | 12.4–12.6M/s | -This shows BlazeSub maintains excellent performance even at high subscriber counts. +Note the crossover: at small subscriber counts direct goroutines are roughly twice as fast as the worker pool, but by 1000 subscribers the fixed cost of fanning out goroutines dominates and the worker pool delivers slightly faster. ## Hardware Considerations diff --git a/README.md b/README.md index 9322256..ed24a3b 100644 --- a/README.md +++ b/README.md @@ -13,23 +13,41 @@ BlazeSub is a high-performance, lock-free publish/subscribe system designed to o ## ✨ Features -- **⚡ Ultra-fast performance**: Up to [84.7 million messages per second delivered to 1000 subscribers](PERFORMANCE.md) +- **⚡ High-throughput delivery**: Up to 12.8 million message deliveries per second to 1000 subscribers - **🧠 Zero memory allocations**: Core operations don't allocate memory, reducing GC pressure - **🔒 Thread-safe by design**: Uses lock-free data structures for maximum concurrency - **🌳 MQTT-compatible topic matching**: Supports single-level (+) and multi-level (#) wildcards - **🚀 Efficient topic caching**: Optimizes repeat accesses to common topics - **🔄 Flexible message delivery**: Choose between worker pool or direct goroutines for optimal performance -- **⏱️ Low-latency message delivery**: Direct goroutines up to 52% faster than worker pool and 34% faster than MQTT -- **📦 Rich metadata support**: Attach arbitrary metadata to messages for enhanced application context -- **🧩 Generic message types**: Define your own message data types without serialization/deserialization overhead +- **📉 Memory efficiency**: Uses 7–14x less memory per publish than MochiMQTT +- **0️⃣ Zero allocations** for core subscription matching operations +- **🗑️ Minimal GC impact**: Only 2 allocations per publish operation ## 📊 Performance Highlights -- **💯 Direct match throughput**: 84.7 million messages per second to 1000 subscribers -- **🔍 Wildcard match throughput**: 83.5 million messages per second to 1000 subscribers -- **📉 Memory efficiency**: Uses up to 95% less memory than MochiMQTT -- **0️⃣ Zero allocations** for core subscription matching operations -- **🗑️ Minimal GC impact**: Only 2 allocations per publish operation +All numbers measured on an Apple M4 Pro (Go 1.26.6); the fan-out figures are end-to-end delivery throughput — the clock stops only when every message has reached every subscriber handler. See [PERFORMANCE.md](PERFORMANCE.md) for methodology and how to reproduce. + +**Fan-out to 1000 subscribers** (deliveries per second, 3 runs each): + +| Scenario | Direct Goroutines | Worker Pool | +| ----------------------- | ----------------- | ---------------- | +| Direct match | 9.9–10.7M/s | 12.4–12.6M/s | +| Wildcard match | 9.8–10.5M/s | 12.7–12.8M/s | + +**vs MochiMQTT** (5000 subscriptions across 1000 topics, 20% wildcard-matching publishes, in-process routing with no network I/O): + +| Benchmark | BlazeSub (best) | MochiMQTT | BlazeSub advantage | +| ----------------------- | ----------------- | ---------------- | ----------------------------- | +| Single publish | 1,396 ns/op | 1,442 ns/op | ~1.04x faster, 7x less memory | +| Concurrent publish | 325 ns/op | 637 ns/op | ~1.9x faster, 11x less memory | +| Subscribe/unsubscribe | 5,195 ns/op | 1,487 ns/op | 3.5x slower | + +## ⚖️ Tradeoffs + +- **Subscription churn is slow**: subscribing/unsubscribing is ~3.5x slower than MochiMQTT (copy-on-write subscription trie). BlazeSub favors publish fan-out over frequent subscription changes. +- **Worker pool vs MQTT on single publishes**: with a large worker pool, a single publish can be slower than MochiMQTT's in-process routing (3.3μs vs 1.4μs) — use direct goroutines for the fast path. +- **Delivery mode**: direct goroutines are ~2x faster than the worker pool at small subscriber counts; at 1000 subscribers the fixed overhead of goroutine fan-out dominates and the worker pool delivers slightly faster. +- **The old headline was wrong**: previous versions of these docs claimed 84.7M msg/s and "30–50x faster than MQTT"; re-measurement showed those figures came from a timing flaw in the benchmark (a fixed sleep inside the measured window) and network-free comparisons presented as broker speed. ## 📘 Documentation