Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions CHANGES.md
Original file line number Diff line number Diff line change
@@ -1,6 +1,16 @@
# p99.Zig CHANGES <!-- omit in toc -->


## 0.0.3 - 26th July 2026

Added `binary-scaling` optimisation

* Implemented optional $2^{32}$ fixed-point binary scaling float-avoidance optimizations from `p99.Rust` as a comptime build option (`-Dbinary-scaling`);
* Added `binary-scaling` option to `build.zig` and passed it to the root module as `build_options`;
* Updated integer-based percentile methods (`valueAtP90`, `valueAtP95`, `valueAtP99`, etc.) to use pre-encoded $2^{32}$ fixed-point multipliers when `binary_scaling` is enabled;
* Bumped package version to `0.0.3`;


## 0.0.2 - 26th July 2026

GitHub Actions CI
Expand Down
1 change: 1 addition & 0 deletions NEWS.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@

| Date | News Item |
| --------------------- | ----------------------------------------- |
| 26th July 2026 | Release 0.0.3: Optional 2^32 binary scaling optimizations added as a comptime build option. |
| 26th July 2026 | Release 0.0.2: GitHub Actions CI support added for Linux, macOS, and Windows. |
| 26th July 2026 | Release 0.0.1: Core Histogram implementation with unit tests and benchmarks is complete. |

Expand Down
36 changes: 36 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,9 @@ Very low-cost measuring of performance percentiles, for Zig
- [Introduction](#introduction)
- [How It Works](#how-it-works)
- [Installation](#installation)
- [Build Options](#build-options)
- [Minimal Example](#minimal-example)
- [Benchmarks](#benchmarks)
- [Project Information](#project-information)
- [Where to get help](#where-to-get-help)
- [Contribution guidelines](#contribution-guidelines)
Expand Down Expand Up @@ -81,6 +83,23 @@ exe.root_module.addImport("p99", p99_dep.module("p99"));
```


## Build Options

`p99.Zig` supports the following compile-time build options:

* **`binary-scaling`** *(bool, default: `false`)*: Replaces integer division in the integer-based percentile methods (`valueAtP90`, `valueAtP95`, `valueAtP99`, etc.) with $2^{32}$ fixed-point binary scaling. Each percentile multiplier (e.g., `0.90` for p90) is pre-encoded as a `u32` constant and the target rank is computed via a single multiplication and a 32-bit right-shift, avoiding the cost of integer division entirely. This yields a significant speedup for percentile queries with a negligible loss of accuracy (the scaled multiplier differs from the true value by less than $10^{-9}$). The generic `valueAtPercentile(f64)` method is unaffected by this feature.

To enable this option in your project, pass it when fetching the dependency in your `build.zig`:

```zig
const p99_dep = b.dependency("p99", .{
.target = target,
.optimize = optimize,
.@"binary-scaling" = true,
});
```


## Minimal Example

Here is a minimal example demonstrating how to use `p99.Zig`:
Expand Down Expand Up @@ -115,6 +134,23 @@ pub fn main(init: std.process.Init) !void {
```


## Benchmarks

A benchmark suite is included in `benches/benchmark_histogram.zig` to measure performance.

To run the benchmarks in **ReleaseFast** mode (without binary scaling):

```bash
zig build bench -Doptimize=ReleaseFast
```

To run the benchmarks with the **binary-scaling** optimization enabled:

```bash
zig build bench -Dbinary-scaling=true -Doptimize=ReleaseFast
```


## Project Information


Expand Down
2 changes: 1 addition & 1 deletion TODO.md
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@

## Performance improvements

- [ ] Implement $2^{32}$ fixed-point binary scaling float-avoidance optimizations from `p99.Rust` as a comptime build option.
- [x] Implement $2^{32}$ fixed-point binary scaling float-avoidance optimizations from `p99.Rust` as a comptime build option.


## Packaging improvements
Expand Down
211 changes: 177 additions & 34 deletions benches/benchmark_histogram.zig
Original file line number Diff line number Diff line change
Expand Up @@ -11,73 +11,216 @@ pub fn main(init: std.process.Init) !void {

// 1. Benchmark: Pushing sequential events
{
var h = p99.Histogram{};
const count = 100_000;
const trials = 50;
var total_ns: u64 = 0;

const start = std.Io.Clock.awake.now(io);
var i: u64 = 1;
while (i <= count) : (i += 1) {
_ = h.pushEventTimeNs(i);
// Warmup phase (to heat caches and scale CPU frequency)
{
var warmup_h = p99.Histogram{};
var i: u64 = 1;
while (i <= 10_000) : (i += 1) {
var val = i;
std.mem.doNotOptimizeAway(&val);
_ = warmup_h.pushEventTimeNs(val);
}
std.mem.doNotOptimizeAway(&warmup_h);
}
const elapsed = start.untilNow(io, .awake);
const ns = elapsed.nanoseconds;
const avg_ns = @as(f64, @floatFromInt(ns)) / @as(f64, @floatFromInt(count));

// Multiple trials for statistical stability
var t: usize = 0;
while (t < trials) : (t += 1) {
var h = p99.Histogram{};
const start = std.Io.Clock.awake.now(io);
var i: u64 = 1;
while (i <= count) : (i += 1) {
var val = i;
std.mem.doNotOptimizeAway(&val); // Force LLVM to treat input as dynamic runtime value!
_ = h.pushEventTimeNs(val);
}
const elapsed = start.untilNow(io, .awake);
std.mem.doNotOptimizeAway(&h); // Prevent LLVM from deleting the loop!
total_ns += @as(u64, @intCast(elapsed.nanoseconds));
}

const avg_ns = @as(f64, @floatFromInt(total_ns)) / @as(f64, @floatFromInt(count * trials));

try stdout.print("Benchmark: Push Sequential Events\n", .{});
try stdout.print(" Total events: {d}\n", .{count});
try stdout.print(" Total time: {d} ns\n", .{ns});
try stdout.print(" Total events: {d} (across {d} trials)\n", .{ count * trials, trials });
try stdout.print(" Average/push: {d:.2} ns\n\n", .{avg_ns});
}

// 2. Benchmark: Pushing random events (using a simple LCG PRNG for speed)
// 2. Benchmark: Pushing random events (1ns to 10s wide-range)
{
var h = p99.Histogram{};
const count = 100_000;
const trials = 50;
var total_ns: u64 = 0;
var seed: u64 = 12345;

const start = std.Io.Clock.awake.now(io);
// Warmup
{
var warmup_h = p99.Histogram{};
var i: u64 = 1;
while (i <= 10_000) : (i += 1) {
seed = seed *% 6364136223846793005 +% 1442695040888963407;
var val = (seed % 10_000_000_000) + 1;
std.mem.doNotOptimizeAway(&val);
_ = warmup_h.pushEventTimeNs(val);
}
std.mem.doNotOptimizeAway(&warmup_h);
}

var t: usize = 0;
while (t < trials) : (t += 1) {
var h = p99.Histogram{};
const start = std.Io.Clock.awake.now(io);
var i: u64 = 1;
while (i <= count) : (i += 1) {
seed = seed *% 6364136223846793005 +% 1442695040888963407;
var val = (seed % 10_000_000_000) + 1;
std.mem.doNotOptimizeAway(&val); // Force LLVM to treat input as dynamic runtime value!
_ = h.pushEventTimeNs(val);
}
const elapsed = start.untilNow(io, .awake);
std.mem.doNotOptimizeAway(&h); // Prevent loop deletion
total_ns += @as(u64, @intCast(elapsed.nanoseconds));
}

const avg_ns = @as(f64, @floatFromInt(total_ns)) / @as(f64, @floatFromInt(count * trials));

try stdout.print("Benchmark: Push Random Events (1ns to 10s wide-range)\n", .{});
try stdout.print(" Total events: {d} (across {d} trials)\n", .{ count * trials, trials });
try stdout.print(" Average/push: {d:.2} ns\n\n", .{avg_ns});
}

// 3. Benchmark: Percentile Queries (valueAtP99 on 10s wide-range)
{
var h = p99.Histogram{};
const count = 100_000;
var seed: u64 = 12345;
var i: u64 = 1;
while (i <= count) : (i += 1) {
// Simple LCG PRNG
seed = seed *% 6364136223846793005 +% 1442695040888963407;
const val = (seed % 1_000_000) + 1;
const val = (seed % 10_000_000_000) + 1;
_ = h.pushEventTimeNs(val);
}
const elapsed = start.untilNow(io, .awake);
const ns = elapsed.nanoseconds;
const avg_ns = @as(f64, @floatFromInt(ns)) / @as(f64, @floatFromInt(count));
std.mem.doNotOptimizeAway(&h);

try stdout.print("Benchmark: Push Random Events (1ns to 1ms)\n", .{});
try stdout.print(" Total events: {d}\n", .{count});
try stdout.print(" Total time: {d} ns\n", .{ns});
try stdout.print(" Average/push: {d:.2} ns\n\n", .{avg_ns});
const query_count = 10_000;
const trials = 50;
var total_ns: u64 = 0;

// Warmup
{
var q: u64 = 0;
while (q < 1_000) : (q += 1) {
var val = h.valueAtP99();
std.mem.doNotOptimizeAway(&val);
}
}

var t: usize = 0;
while (t < trials) : (t += 1) {
const start = std.Io.Clock.awake.now(io);
var q: u64 = 0;
while (q < query_count) : (q += 1) {
std.mem.doNotOptimizeAway(&h); // Force LLVM to assume 'h' might be modified, preventing loop hoisting/caching!
var val = h.valueAtP99();
std.mem.doNotOptimizeAway(&val); // Prevent query loop deletion
}
const elapsed = start.untilNow(io, .awake);
total_ns += @as(u64, @intCast(elapsed.nanoseconds));
}

const avg_ns = @as(f64, @floatFromInt(total_ns)) / @as(f64, @floatFromInt(query_count * trials));

try stdout.print("Benchmark: Percentile Queries (valueAtP99 on 10s wide-range)\n", .{});
try stdout.print(" Total queries: {d} (across {d} trials)\n", .{ query_count * trials, trials });
try stdout.print(" Average/query: {d:.2} ns\n\n", .{avg_ns});
}

// 3. Benchmark: Percentile Queries
// 4. Benchmark: Clear
{
const trials = 50;
const loop_count = 10_000;
var total_ns: u64 = 0;

// Warmup
{
var warmup_h = p99.Histogram{};
var i: usize = 0;
while (i < 1_000) : (i += 1) {
warmup_h.clear();
std.mem.doNotOptimizeAway(&warmup_h);
}
}

var t: usize = 0;
while (t < trials) : (t += 1) {
var h = p99.Histogram{};
_ = h.pushEventTimeNs(100);
_ = h.pushEventTimeNs(200);

const start = std.Io.Clock.awake.now(io);
var i: usize = 0;
while (i < loop_count) : (i += 1) {
h.clear();
std.mem.doNotOptimizeAway(&h);
}
const elapsed = start.untilNow(io, .awake);
total_ns += @as(u64, @intCast(elapsed.nanoseconds));
}

const avg_ns = @as(f64, @floatFromInt(total_ns)) / @as(f64, @floatFromInt(loop_count * trials));

try stdout.print("Benchmark: Clear\n", .{});
try stdout.print(" Total clears: {d} (across {d} trials)\n", .{ loop_count * trials, trials });
try stdout.print(" Average/clear: {d:.2} ns\n\n", .{avg_ns});
}

// 5. Benchmark: Generic Percentile Queries (valueAtPercentile)
{
var h = p99.Histogram{};
const count = 100_000;
var seed: u64 = 12345;
var i: u64 = 1;
while (i <= count) : (i += 1) {
seed = seed *% 6364136223846793005 +% 1442695040888963407;
const val = (seed % 1_000_000) + 1;
const val = (seed % 10_000_000_000) + 1; // 10-second wide-range
_ = h.pushEventTimeNs(val);
}
std.mem.doNotOptimizeAway(&h);

const query_count = 10_000;
const start = std.Io.Clock.awake.now(io);
var q: u64 = 0;
while (q < query_count) : (q += 1) {
_ = h.valueAtP99();
const trials = 50;
var total_ns: u64 = 0;

// Warmup
{
var q: u64 = 0;
while (q < 1_000) : (q += 1) {
var val = h.valueAtPercentile(99.0);
std.mem.doNotOptimizeAway(&val);
}
}
const elapsed = start.untilNow(io, .awake);
const ns = elapsed.nanoseconds;
const avg_ns = @as(f64, @floatFromInt(ns)) / @as(f64, @floatFromInt(query_count));

try stdout.print("Benchmark: Percentile Queries (valueAtP99)\n", .{});
try stdout.print(" Total queries: {d}\n", .{query_count});
try stdout.print(" Total time: {d} ns\n", .{ns});
var t: usize = 0;
while (t < trials) : (t += 1) {
const start = std.Io.Clock.awake.now(io);
var q: u64 = 0;
while (q < query_count) : (q += 1) {
std.mem.doNotOptimizeAway(&h);
var val = h.valueAtPercentile(99.0);
std.mem.doNotOptimizeAway(&val);
}
const elapsed = start.untilNow(io, .awake);
total_ns += @as(u64, @intCast(elapsed.nanoseconds));
}

const avg_ns = @as(f64, @floatFromInt(total_ns)) / @as(f64, @floatFromInt(query_count * trials));

try stdout.print("Benchmark: Generic Percentile Queries (valueAtPercentile(99.0))\n", .{});
try stdout.print(" Total queries: {d} (across {d} trials)\n", .{ query_count * trials, trials });
try stdout.print(" Average/query: {d:.2} ns\n\n", .{avg_ns});
}

Expand Down
12 changes: 12 additions & 0 deletions build.zig
Original file line number Diff line number Diff line change
Expand Up @@ -4,13 +4,25 @@ pub fn build(b: *std.Build) void {
const target = b.standardTargetOptions(.{});
const optimize = b.standardOptimizeOption(.{});

// Build option for binary scaling (mirroring Rust's feature flag)
const binary_scaling = b.option(
bool,
"binary-scaling",
"Enable 2^32 fixed-point binary scaling for integer-based percentile queries (default: false)",
) orelse false;

// Create a module for other packages to import
const p99_module = b.addModule("p99", .{
.root_source_file = b.path("src/root.zig"),
.target = target,
.optimize = optimize,
});

// Add build options to the module
const options = b.addOptions();
options.addOption(bool, "binary_scaling", binary_scaling);
p99_module.addOptions("build_options", options);

// Static library using the module
const lib = b.addLibrary(.{
.name = "p99",
Expand Down
2 changes: 1 addition & 1 deletion build.zig.zon
Original file line number Diff line number Diff line change
Expand Up @@ -8,5 +8,5 @@
"LICENSE",
"README.md",
},
.version = "0.0.2",
.version = "0.0.3",
}
Loading
Loading