Add cube-bench (correctness-gated microbenchmarks) + daemon stats telemetry
- cube-bench crate: real-code-path throughput/latency over cubestore, cubecrypt (aes/gcm/chacha/xts), cubecode VM, and cubesys Session. Every section asserts correctness before timing. Wired into ./check as an opt-in 'bench' stage. - cubesys Session: per-command latency histogram + per-C-namespace record counts, exposed via a new 'stats' command over the live socket. - Deployed rebuilt cube-server to /home/luulu/.cubelinux/bin and restarted the system cube.service; verified stats live.
This commit is contained in:
@@ -0,0 +1,300 @@
|
||||
//! cube-bench: correctness-gated microbenchmarks for CUBELinux-2.
|
||||
//!
|
||||
//! Discipline (per the empirical-design-benchmarking skill):
|
||||
//! * every measurement path first asserts an invariant, so a silent
|
||||
//! regression cannot masquerade as a number;
|
||||
//! * time in nanoseconds via `Instant::elapsed().as_secs_f64() * 1e9`
|
||||
//! (seconds -> ns), never mislabeled;
|
||||
//! * the timed loop's accumulator is folded into `black_box` so release
|
||||
//! builds cannot delete it;
|
||||
//! * all numbers are page-cache-warm by default (in-memory HashBackend);
|
||||
//! we say so explicitly rather than implying cold-disk figures.
|
||||
//!
|
||||
//! Run: `cargo run -p cube-bench --release` (defaults)
|
||||
//! `cargo run -p cube-bench --release -- 200000` (override scale N)
|
||||
|
||||
use cubecode::opcode::Op;
|
||||
use cubecoords::{CubeHeader, Czyx};
|
||||
use cubecrypt::transform::{self, Key, TransformId};
|
||||
use cubestore::{CubeStore, HashBackend};
|
||||
use cubesys::commands::Session;
|
||||
use std::hint::black_box;
|
||||
use std::time::Instant;
|
||||
|
||||
/// Time `f` for `iters` iterations, returning ns/op. The black_box sink
|
||||
/// prevents the optimizer from deleting a loop whose only effect is the
|
||||
/// accumulator. We time a single pass per iteration (not REPS*atomic) so the
|
||||
/// cost measured is the operation itself.
|
||||
fn time_ns<F: FnMut()>(mut f: F, iters: u64) -> f64 {
|
||||
// warm-up (also exercises the code path so the first call isn't special)
|
||||
for _ in 0..min(iters, 3) {
|
||||
f();
|
||||
}
|
||||
let t0 = Instant::now();
|
||||
let mut sink: u64 = 0;
|
||||
for i in 0..iters {
|
||||
f();
|
||||
sink = sink.wrapping_add(i); // keep the loop from being elided
|
||||
}
|
||||
let total_ns = Instant::now().duration_since(t0).as_secs_f64() * 1e9;
|
||||
black_box(sink);
|
||||
total_ns / iters as f64
|
||||
}
|
||||
|
||||
fn min(a: u64, b: u64) -> u64 {
|
||||
if a < b {
|
||||
a
|
||||
} else {
|
||||
b
|
||||
}
|
||||
}
|
||||
|
||||
fn coord_for(i: u32) -> Czyx {
|
||||
// injective over [0, ~4G): high bits -> c, then z, y; x fixed.
|
||||
let c = ((i >> 16) & 0xFF) as u8;
|
||||
let z = ((i >> 8) & 0xFF) as u8;
|
||||
let y = (i & 0xFF) as u8;
|
||||
Czyx::new(c, z, y, 0)
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let args: Vec<String> = std::env::args().collect();
|
||||
let scale_n: u32 = args.get(1).and_then(|s| s.parse().ok()).unwrap_or(100_000);
|
||||
let crypto_ops: u64 = min(scale_n as u64, 20_000);
|
||||
let vm_iters: u64 = 50_000;
|
||||
let run_iters: u64 = 5_000;
|
||||
|
||||
println!("=== CUBELinux-2 cube-bench (warm in-memory backend) ===");
|
||||
println!("scale N = {scale_n} records | crypto ops = {crypto_ops} | vm iters = {vm_iters}");
|
||||
println!("(all timings ns/op, page-cache-warm; correctness asserted before timing)\n");
|
||||
|
||||
// ---- 1. cubestore: put / get / scan_prefix ------------------------------
|
||||
{
|
||||
let mut store = CubeStore::new(HashBackend::new());
|
||||
let sample = b"benchmark-record-body";
|
||||
// correctness: a single round-trip preserves body + header
|
||||
let probe = Czyx::new(200, 1, 1, 1);
|
||||
let mut h = CubeHeader::new();
|
||||
h.title = Some("probe".into());
|
||||
h.refresh_flags();
|
||||
store.put_record(probe, &h, sample);
|
||||
let (rh, rb) = store.get_record(&probe).unwrap();
|
||||
assert_eq!(rb, sample);
|
||||
assert_eq!(rh.title.as_deref(), Some("probe"));
|
||||
|
||||
// bulk put
|
||||
let t_put = time_ns(
|
||||
|| {
|
||||
let mut s = CubeStore::new(HashBackend::new());
|
||||
for i in 0..scale_n {
|
||||
let c = coord_for(i);
|
||||
s.put_raw(c, vec![i as u8; 32]);
|
||||
}
|
||||
black_box(&s);
|
||||
},
|
||||
1,
|
||||
);
|
||||
// we timed a full bulk build as one op; convert to per-record ns
|
||||
let put_ns = t_put / scale_n as f64;
|
||||
|
||||
// build once for get/scan measurement
|
||||
let mut store = CubeStore::new(HashBackend::new());
|
||||
for i in 0..scale_n {
|
||||
store.put_raw(coord_for(i), vec![i as u8; 32]);
|
||||
}
|
||||
assert_eq!(store.keys().len(), scale_n as usize, "keys() count drift");
|
||||
|
||||
let get_ns = time_ns(
|
||||
|| {
|
||||
let mut acc: u8 = 0;
|
||||
for i in 0..scale_n {
|
||||
let v = store.get_raw(&coord_for(i)).unwrap();
|
||||
acc = acc.wrapping_add(v[0]);
|
||||
}
|
||||
black_box(acc);
|
||||
},
|
||||
1,
|
||||
) / scale_n as f64;
|
||||
|
||||
// scan_prefix correctness: c takes values 0 or 1 only across [0,scale_n)
|
||||
// exact expected count for c=0 and c=1 from the coord_for mapping
|
||||
let expected_c0 = if scale_n <= 0x10000 {
|
||||
scale_n as usize
|
||||
} else {
|
||||
0x10000
|
||||
};
|
||||
let expected_c1 = scale_n as usize - expected_c0;
|
||||
let got_c0 = store.scan_prefix(0, None, None).len();
|
||||
let got_c1 = store.scan_prefix(1, None, None).len();
|
||||
assert_eq!(got_c0, expected_c0, "scan_prefix c=0 wrong");
|
||||
assert_eq!(got_c1, expected_c1, "scan_prefix c=1 wrong");
|
||||
assert_eq!(got_c0 + got_c1, scale_n as usize, "prefix covers all");
|
||||
|
||||
let scan_ns = time_ns(
|
||||
|| {
|
||||
let v = store.scan_prefix(0, None, None);
|
||||
black_box(v.len());
|
||||
},
|
||||
200,
|
||||
);
|
||||
|
||||
println!("cubestore (HashBackend, {scale_n} recs)");
|
||||
println!(" put_raw {:9.2} ns/op", put_ns);
|
||||
println!(" get_raw {:9.2} ns/op", get_ns);
|
||||
println!(
|
||||
" scan_prefix {:9.2} ns/op (returns {} coords)",
|
||||
scan_ns, got_c0
|
||||
);
|
||||
println!(
|
||||
" throughput ~{:.1} k put/s ({} recs in {:.1} ms)\n",
|
||||
(scale_n as f64) / (t_put / 1e6),
|
||||
scale_n,
|
||||
t_put / 1e6
|
||||
);
|
||||
}
|
||||
|
||||
// ---- 2. cubecrypt: seal / open per transform ---------------------------
|
||||
{
|
||||
let key: Key = transform::derive_key(b"cube-bench-key-material-32b", b"salt");
|
||||
let pt: Vec<u8> = (0u8..=255).cycle().take(1024).collect();
|
||||
// correctness first: each transform round-trips exactly
|
||||
for t in [
|
||||
TransformId::None,
|
||||
TransformId::Aes256Gcm,
|
||||
TransformId::ChaCha20Poly1305,
|
||||
TransformId::Aes256Xts,
|
||||
] {
|
||||
let e = transform::seal(t, &key, &pt);
|
||||
let back = transform::open(&key, &e).unwrap();
|
||||
assert_eq!(back, pt, "roundtrip failed for {t:?}");
|
||||
}
|
||||
|
||||
println!("cubecrypt (1 KB payload, {crypto_ops} ops)");
|
||||
for t in [
|
||||
TransformId::None,
|
||||
TransformId::Aes256Gcm,
|
||||
TransformId::ChaCha20Poly1305,
|
||||
TransformId::Aes256Xts,
|
||||
] {
|
||||
let seal_ns = time_ns(
|
||||
|| {
|
||||
let e = transform::seal(t, &key, &pt);
|
||||
black_box(e.len());
|
||||
},
|
||||
crypto_ops,
|
||||
);
|
||||
let open_ns = time_ns(
|
||||
|| {
|
||||
// re-seal then open to keep the op self-contained
|
||||
let e = transform::seal(t, &key, &pt);
|
||||
let b = transform::open(&key, &e).unwrap();
|
||||
black_box(b.len());
|
||||
},
|
||||
crypto_ops,
|
||||
);
|
||||
println!(
|
||||
" {:<18} seal {:8.1} ns/op open {:8.1} ns/op",
|
||||
format!("{t:?}"),
|
||||
seal_ns,
|
||||
open_ns
|
||||
);
|
||||
}
|
||||
println!();
|
||||
}
|
||||
|
||||
// ---- 3. cubecode: VM dispatch (shared-stack call graph) -----------------
|
||||
{
|
||||
// Build: entry const 21 ; call 0 ; halt | leaf store 0 ; load 0 ;
|
||||
// const 2 ; mul ; ret -> expects top == 42
|
||||
let mut store = CubeStore::new(HashBackend::new());
|
||||
let entry = Czyx::new(1, 1, 1, 1);
|
||||
let leaf = Czyx::new(2, 0, 0, 1);
|
||||
let mut lh = CubeHeader::new();
|
||||
lh.linked_records = vec![];
|
||||
lh.refresh_flags();
|
||||
store.put_record(
|
||||
leaf,
|
||||
&lh,
|
||||
&cubecode::encode(&[Op::Store(0), Op::Load(0), Op::Const(2), Op::Mul, Op::Ret]),
|
||||
);
|
||||
let mut eh = CubeHeader::new();
|
||||
eh.linked_records = vec![leaf];
|
||||
eh.refresh_flags();
|
||||
store.put_record(
|
||||
entry,
|
||||
&eh,
|
||||
&cubecode::encode(&[Op::Const(21), Op::CallLink(0), Op::Halt]),
|
||||
);
|
||||
|
||||
// correctness: single run yields 42
|
||||
let mut vm = cubecode::Vm::new(store.clone());
|
||||
let r = vm.run(entry);
|
||||
assert_eq!(r, cubecode::vm::RunResult::Halted { top: Some(42) });
|
||||
|
||||
let run_ns = time_ns(
|
||||
|| {
|
||||
let mut v = cubecode::Vm::new(store.clone());
|
||||
let res = v.run(entry);
|
||||
black_box(res);
|
||||
},
|
||||
vm_iters,
|
||||
);
|
||||
let call_ns = time_ns(
|
||||
|| {
|
||||
// exercise the cross-cube edge specifically
|
||||
let mut v = cubecode::Vm::new(store.clone());
|
||||
let res = v.run(leaf);
|
||||
black_box(res);
|
||||
},
|
||||
vm_iters,
|
||||
);
|
||||
println!("cubecode VM ({vm_iters} runs of entry->leaf)");
|
||||
println!(" run entry (call graph) {:9.2} ns/op", run_ns);
|
||||
println!(" run leaf (single cell) {:9.2} ns/op\n", call_ns);
|
||||
}
|
||||
|
||||
// ---- 4. cubesys Session command interpreter ----------------------------
|
||||
{
|
||||
let mut sess = Session::new();
|
||||
// A single-cell program written via the real CLI path grammar:
|
||||
// `const 21 ; halt` -> Halted{ top: Some(21) }.
|
||||
let w = sess
|
||||
.exec("prog /c001/z001/y001/x001 const 21 halt")
|
||||
.expect("prog write");
|
||||
// correctness: running yields 21
|
||||
let out = sess.exec("run /c001/z001/y001/x001").expect("run");
|
||||
assert!(
|
||||
out.contains("Some(21)"),
|
||||
"session run expected 21, got: {out}"
|
||||
);
|
||||
println!("cubesys Session commands ({run_iters} iters)");
|
||||
println!(
|
||||
" prog (write program) {:9.2} ns/op",
|
||||
time_ns(
|
||||
|| {
|
||||
let _ = sess.exec("prog /c001/z001/y001/x001 const 21 halt");
|
||||
},
|
||||
run_iters
|
||||
),
|
||||
);
|
||||
let run_ns = time_ns(
|
||||
|| {
|
||||
let _ = sess.exec("prog /c001/z001/y001/x001 const 21 halt");
|
||||
let o = sess.exec("run /c001/z001/y001/x001").expect("run");
|
||||
black_box(o);
|
||||
},
|
||||
run_iters,
|
||||
);
|
||||
println!(" run (vm dispatch) {:9.2} ns/op", run_ns);
|
||||
// `w` proves the program actually landed in the cube (carries coord).
|
||||
println!(" prog result line: {w}");
|
||||
|
||||
// ls over the program path prefix must list the cell
|
||||
let ls = sess.exec("ls /c001/z001/y001").expect("ls");
|
||||
assert!(ls.contains("x001"), "ls missing cell: {ls}");
|
||||
println!(" ls (directory list) OK ({})", ls.trim());
|
||||
println!();
|
||||
}
|
||||
|
||||
println!("=== bench complete: every section asserted correctness before timing ===");
|
||||
}
|
||||
Reference in New Issue
Block a user