Add cube-bench (correctness-gated microbenchmarks) + daemon stats telemetry

- cube-bench crate: real-code-path throughput/latency over cubestore,
  cubecrypt (aes/gcm/chacha/xts), cubecode VM, and cubesys Session.
  Every section asserts correctness before timing. Wired into ./check
  as an opt-in 'bench' stage.
- cubesys Session: per-command latency histogram + per-C-namespace record
  counts, exposed via a new 'stats' command over the live socket.
- Deployed rebuilt cube-server to /home/luulu/.cubelinux/bin and
  restarted the system cube.service; verified stats live.
This commit is contained in:
cube-agent
2026-08-11 01:15:32 -04:00
parent 06ea3252eb
commit 4f9cb2e75e
6 changed files with 529 additions and 2 deletions
+300
View File
@@ -0,0 +1,300 @@
//! cube-bench: correctness-gated microbenchmarks for CUBELinux-2.
//!
//! Discipline (per the empirical-design-benchmarking skill):
//! * every measurement path first asserts an invariant, so a silent
//! regression cannot masquerade as a number;
//! * time in nanoseconds via `Instant::elapsed().as_secs_f64() * 1e9`
//! (seconds -> ns), never mislabeled;
//! * the timed loop's accumulator is folded into `black_box` so release
//! builds cannot delete it;
//! * all numbers are page-cache-warm by default (in-memory HashBackend);
//! we say so explicitly rather than implying cold-disk figures.
//!
//! Run: `cargo run -p cube-bench --release` (defaults)
//! `cargo run -p cube-bench --release -- 200000` (override scale N)
use cubecode::opcode::Op;
use cubecoords::{CubeHeader, Czyx};
use cubecrypt::transform::{self, Key, TransformId};
use cubestore::{CubeStore, HashBackend};
use cubesys::commands::Session;
use std::hint::black_box;
use std::time::Instant;
/// Time `f` for `iters` iterations, returning ns/op. The black_box sink
/// prevents the optimizer from deleting a loop whose only effect is the
/// accumulator. We time a single pass per iteration (not REPS*atomic) so the
/// cost measured is the operation itself.
fn time_ns<F: FnMut()>(mut f: F, iters: u64) -> f64 {
// warm-up (also exercises the code path so the first call isn't special)
for _ in 0..min(iters, 3) {
f();
}
let t0 = Instant::now();
let mut sink: u64 = 0;
for i in 0..iters {
f();
sink = sink.wrapping_add(i); // keep the loop from being elided
}
let total_ns = Instant::now().duration_since(t0).as_secs_f64() * 1e9;
black_box(sink);
total_ns / iters as f64
}
fn min(a: u64, b: u64) -> u64 {
if a < b {
a
} else {
b
}
}
fn coord_for(i: u32) -> Czyx {
// injective over [0, ~4G): high bits -> c, then z, y; x fixed.
let c = ((i >> 16) & 0xFF) as u8;
let z = ((i >> 8) & 0xFF) as u8;
let y = (i & 0xFF) as u8;
Czyx::new(c, z, y, 0)
}
fn main() {
let args: Vec<String> = std::env::args().collect();
let scale_n: u32 = args.get(1).and_then(|s| s.parse().ok()).unwrap_or(100_000);
let crypto_ops: u64 = min(scale_n as u64, 20_000);
let vm_iters: u64 = 50_000;
let run_iters: u64 = 5_000;
println!("=== CUBELinux-2 cube-bench (warm in-memory backend) ===");
println!("scale N = {scale_n} records | crypto ops = {crypto_ops} | vm iters = {vm_iters}");
println!("(all timings ns/op, page-cache-warm; correctness asserted before timing)\n");
// ---- 1. cubestore: put / get / scan_prefix ------------------------------
{
let mut store = CubeStore::new(HashBackend::new());
let sample = b"benchmark-record-body";
// correctness: a single round-trip preserves body + header
let probe = Czyx::new(200, 1, 1, 1);
let mut h = CubeHeader::new();
h.title = Some("probe".into());
h.refresh_flags();
store.put_record(probe, &h, sample);
let (rh, rb) = store.get_record(&probe).unwrap();
assert_eq!(rb, sample);
assert_eq!(rh.title.as_deref(), Some("probe"));
// bulk put
let t_put = time_ns(
|| {
let mut s = CubeStore::new(HashBackend::new());
for i in 0..scale_n {
let c = coord_for(i);
s.put_raw(c, vec![i as u8; 32]);
}
black_box(&s);
},
1,
);
// we timed a full bulk build as one op; convert to per-record ns
let put_ns = t_put / scale_n as f64;
// build once for get/scan measurement
let mut store = CubeStore::new(HashBackend::new());
for i in 0..scale_n {
store.put_raw(coord_for(i), vec![i as u8; 32]);
}
assert_eq!(store.keys().len(), scale_n as usize, "keys() count drift");
let get_ns = time_ns(
|| {
let mut acc: u8 = 0;
for i in 0..scale_n {
let v = store.get_raw(&coord_for(i)).unwrap();
acc = acc.wrapping_add(v[0]);
}
black_box(acc);
},
1,
) / scale_n as f64;
// scan_prefix correctness: c takes values 0 or 1 only across [0,scale_n)
// exact expected count for c=0 and c=1 from the coord_for mapping
let expected_c0 = if scale_n <= 0x10000 {
scale_n as usize
} else {
0x10000
};
let expected_c1 = scale_n as usize - expected_c0;
let got_c0 = store.scan_prefix(0, None, None).len();
let got_c1 = store.scan_prefix(1, None, None).len();
assert_eq!(got_c0, expected_c0, "scan_prefix c=0 wrong");
assert_eq!(got_c1, expected_c1, "scan_prefix c=1 wrong");
assert_eq!(got_c0 + got_c1, scale_n as usize, "prefix covers all");
let scan_ns = time_ns(
|| {
let v = store.scan_prefix(0, None, None);
black_box(v.len());
},
200,
);
println!("cubestore (HashBackend, {scale_n} recs)");
println!(" put_raw {:9.2} ns/op", put_ns);
println!(" get_raw {:9.2} ns/op", get_ns);
println!(
" scan_prefix {:9.2} ns/op (returns {} coords)",
scan_ns, got_c0
);
println!(
" throughput ~{:.1} k put/s ({} recs in {:.1} ms)\n",
(scale_n as f64) / (t_put / 1e6),
scale_n,
t_put / 1e6
);
}
// ---- 2. cubecrypt: seal / open per transform ---------------------------
{
let key: Key = transform::derive_key(b"cube-bench-key-material-32b", b"salt");
let pt: Vec<u8> = (0u8..=255).cycle().take(1024).collect();
// correctness first: each transform round-trips exactly
for t in [
TransformId::None,
TransformId::Aes256Gcm,
TransformId::ChaCha20Poly1305,
TransformId::Aes256Xts,
] {
let e = transform::seal(t, &key, &pt);
let back = transform::open(&key, &e).unwrap();
assert_eq!(back, pt, "roundtrip failed for {t:?}");
}
println!("cubecrypt (1 KB payload, {crypto_ops} ops)");
for t in [
TransformId::None,
TransformId::Aes256Gcm,
TransformId::ChaCha20Poly1305,
TransformId::Aes256Xts,
] {
let seal_ns = time_ns(
|| {
let e = transform::seal(t, &key, &pt);
black_box(e.len());
},
crypto_ops,
);
let open_ns = time_ns(
|| {
// re-seal then open to keep the op self-contained
let e = transform::seal(t, &key, &pt);
let b = transform::open(&key, &e).unwrap();
black_box(b.len());
},
crypto_ops,
);
println!(
" {:<18} seal {:8.1} ns/op open {:8.1} ns/op",
format!("{t:?}"),
seal_ns,
open_ns
);
}
println!();
}
// ---- 3. cubecode: VM dispatch (shared-stack call graph) -----------------
{
// Build: entry const 21 ; call 0 ; halt | leaf store 0 ; load 0 ;
// const 2 ; mul ; ret -> expects top == 42
let mut store = CubeStore::new(HashBackend::new());
let entry = Czyx::new(1, 1, 1, 1);
let leaf = Czyx::new(2, 0, 0, 1);
let mut lh = CubeHeader::new();
lh.linked_records = vec![];
lh.refresh_flags();
store.put_record(
leaf,
&lh,
&cubecode::encode(&[Op::Store(0), Op::Load(0), Op::Const(2), Op::Mul, Op::Ret]),
);
let mut eh = CubeHeader::new();
eh.linked_records = vec![leaf];
eh.refresh_flags();
store.put_record(
entry,
&eh,
&cubecode::encode(&[Op::Const(21), Op::CallLink(0), Op::Halt]),
);
// correctness: single run yields 42
let mut vm = cubecode::Vm::new(store.clone());
let r = vm.run(entry);
assert_eq!(r, cubecode::vm::RunResult::Halted { top: Some(42) });
let run_ns = time_ns(
|| {
let mut v = cubecode::Vm::new(store.clone());
let res = v.run(entry);
black_box(res);
},
vm_iters,
);
let call_ns = time_ns(
|| {
// exercise the cross-cube edge specifically
let mut v = cubecode::Vm::new(store.clone());
let res = v.run(leaf);
black_box(res);
},
vm_iters,
);
println!("cubecode VM ({vm_iters} runs of entry->leaf)");
println!(" run entry (call graph) {:9.2} ns/op", run_ns);
println!(" run leaf (single cell) {:9.2} ns/op\n", call_ns);
}
// ---- 4. cubesys Session command interpreter ----------------------------
{
let mut sess = Session::new();
// A single-cell program written via the real CLI path grammar:
// `const 21 ; halt` -> Halted{ top: Some(21) }.
let w = sess
.exec("prog /c001/z001/y001/x001 const 21 halt")
.expect("prog write");
// correctness: running yields 21
let out = sess.exec("run /c001/z001/y001/x001").expect("run");
assert!(
out.contains("Some(21)"),
"session run expected 21, got: {out}"
);
println!("cubesys Session commands ({run_iters} iters)");
println!(
" prog (write program) {:9.2} ns/op",
time_ns(
|| {
let _ = sess.exec("prog /c001/z001/y001/x001 const 21 halt");
},
run_iters
),
);
let run_ns = time_ns(
|| {
let _ = sess.exec("prog /c001/z001/y001/x001 const 21 halt");
let o = sess.exec("run /c001/z001/y001/x001").expect("run");
black_box(o);
},
run_iters,
);
println!(" run (vm dispatch) {:9.2} ns/op", run_ns);
// `w` proves the program actually landed in the cube (carries coord).
println!(" prog result line: {w}");
// ls over the program path prefix must list the cell
let ls = sess.exec("ls /c001/z001/y001").expect("ls");
assert!(ls.contains("x001"), "ls missing cell: {ls}");
println!(" ls (directory list) OK ({})", ls.trim());
println!();
}
println!("=== bench complete: every section asserted correctness before timing ===");
}