Compare commits
26
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
35fcb6c7fe | ||
|
|
2586c2ed0d | ||
|
|
3d5d213c75 | ||
|
|
8c24a7ff0d | ||
|
|
8db279a4d8 | ||
|
|
fc057820fa | ||
|
|
9fb4169e52 | ||
|
|
4cfdbfe63b | ||
|
|
e0218ec96b | ||
|
|
b3dc55392e | ||
|
|
b8a7f615ff | ||
|
|
1f7b9fe0de | ||
|
|
aab316a9a1 | ||
|
|
e968b3964e | ||
|
|
daa6289e2d | ||
|
|
e53ae04033 | ||
|
|
7938966d6a | ||
|
|
da0a19786e | ||
|
|
1081b5f1e2 | ||
|
|
72e72fe7a8 | ||
|
|
29ae3b53f1 | ||
|
|
9586e114e7 | ||
|
|
71552fc157 | ||
|
|
41e437c07a | ||
|
|
0267831b18 | ||
|
|
a111d7e8b2 |
@@ -14,7 +14,7 @@ NAME = CUBELinux
|
||||
# reject a version whose first character is not numeric, so a release spelled `CUBELinux.0.6` is
|
||||
# one the machine cannot load modules for. The box's own kernel works around this the same way
|
||||
# (`6.19.3-cube+`), so CUBELinux does too: the Linux base, then `-cubelinux`, then our version.
|
||||
CUBELINUX_VERSION = 6.19.3-cubelinux0.6
|
||||
CUBELINUX_VERSION = 6.19.3-cubelinux0.7
|
||||
|
||||
# *DOCUMENTATION*
|
||||
# To see a list of typical targets execute "make help"
|
||||
|
||||
+87
-18
@@ -30,6 +30,10 @@ pub const VERSION_V1: u8 = 1;
|
||||
pub const VERSION_V2: u8 = 2;
|
||||
/// The addressed format: the same, plus a space table, a fixed-size index, and packed values.
|
||||
pub const VERSION_V3: u8 = 3;
|
||||
/// v3, plus a 16-bit class mask in each index entry, so a scan by flag is a seek rather than a
|
||||
/// walk. The mask is the flag substrate (DESIGN-flag-vocabularies.md): a raw `u16` whose bits are a
|
||||
/// vocabulary's business, written at put time and read by `CUBE_OP_FLAG_SCAN`.
|
||||
pub const VERSION_V4: u8 = 4;
|
||||
|
||||
pub const HEADER_LEN_V1: usize = 6;
|
||||
pub const HEADER_LEN_V2: usize = 6 + 8 + 8;
|
||||
@@ -39,19 +43,28 @@ pub const HEADER_LEN_V3: usize = 4 + 1 + 1 + 8 + 8 + 8 + 8 + 8;
|
||||
|
||||
pub const SPACE_ID_LEN: usize = 32;
|
||||
pub const RAW_KEY_LEN: usize = 24;
|
||||
/// The class mask's width: one `u16` per record.
|
||||
pub const FLAGS_LEN: usize = 2;
|
||||
/// A packed record's fixed part: `space | key | value_len(u64)`, with the value behind it.
|
||||
pub const RECORD_FIXED: usize = SPACE_ID_LEN + RAW_KEY_LEN + 8;
|
||||
/// A v3 index entry: `key | value_off(u64) | value_len(u64)`.
|
||||
pub const INDEX_ENTRY: usize = RAW_KEY_LEN + 8 + 8;
|
||||
/// A v4 index entry: `key | flags(u16) | value_off(u64) | value_len(u64)`.
|
||||
pub const INDEX_ENTRY_V4: usize = RAW_KEY_LEN + FLAGS_LEN + 8 + 8;
|
||||
/// A v3 space-table row: `space | first index(u64) | records(u64)`.
|
||||
pub const SPACE_ENTRY: usize = SPACE_ID_LEN + 8 + 8;
|
||||
|
||||
/// The log's framing, which the fold and the readers both parse.
|
||||
pub const WAL_MAGIC: &[u8; 4] = b"CUBW";
|
||||
/// The original log entry: no class mask.
|
||||
pub const WAL_VERSION: u8 = 1;
|
||||
/// The flagged log entry: the entry carries a `u16` class mask before its length.
|
||||
pub const WAL_VERSION_V2: u8 = 2;
|
||||
pub const WAL_HEADER_LEN: usize = 6;
|
||||
/// `op(1) | crc(4) | space(32) | key(24) | len(4)`, with the value behind it.
|
||||
pub const ENTRY_FIXED: usize = 1 + 4 + SPACE_ID_LEN + RAW_KEY_LEN + 4;
|
||||
/// v2's entry: the same, with a `flags(2)` field before the length.
|
||||
pub const ENTRY_FIXED_V2: usize = 1 + 4 + SPACE_ID_LEN + RAW_KEY_LEN + FLAGS_LEN + 4;
|
||||
|
||||
/// `CUBE_OP_PUT`: an entry that stores a value.
|
||||
pub const WAL_OP_WRITE: u8 = 1;
|
||||
@@ -74,7 +87,7 @@ impl Header {
|
||||
pub fn records_off(&self) -> usize {
|
||||
if self.version == VERSION_V1 {
|
||||
HEADER_LEN_V1
|
||||
} else if self.version == VERSION_V3 {
|
||||
} else if self.version == VERSION_V3 || self.version == VERSION_V4 {
|
||||
HEADER_LEN_V3
|
||||
} else {
|
||||
HEADER_LEN_V2
|
||||
@@ -134,7 +147,7 @@ pub fn parse_header(bytes: &[u8]) -> Result<Header, Bad> {
|
||||
record_count: Some(record_count),
|
||||
})
|
||||
}
|
||||
VERSION_V3 => {
|
||||
VERSION_V3 | VERSION_V4 => {
|
||||
if bytes.len() < HEADER_LEN_V3 {
|
||||
return Err(Bad::Magic);
|
||||
}
|
||||
@@ -153,6 +166,9 @@ pub fn parse_header(bytes: &[u8]) -> Result<Header, Bad> {
|
||||
/// The v3 tables' geometry: where the index is, where the values start, and how many of each.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct V3 {
|
||||
/// The version this geometry belongs to: `VERSION_V3` or `VERSION_V4`, which differ only in
|
||||
/// the index entry's stride (the class mask adds two bytes).
|
||||
pub version: u8,
|
||||
pub record_count: u64,
|
||||
pub space_count: u64,
|
||||
pub index_off: u64,
|
||||
@@ -160,16 +176,27 @@ pub struct V3 {
|
||||
}
|
||||
|
||||
impl V3 {
|
||||
/// The width of one index entry for this geometry's version.
|
||||
pub fn index_stride(&self) -> usize {
|
||||
if self.version == VERSION_V4 {
|
||||
INDEX_ENTRY_V4
|
||||
} else {
|
||||
INDEX_ENTRY
|
||||
}
|
||||
}
|
||||
|
||||
/// Read the tables' geometry, refusing anything that does not add up.
|
||||
///
|
||||
/// The three equalities here are the whole point of the format: a reader computes a record's
|
||||
/// place as `index_off + i * INDEX_ENTRY`, and that is only a place if the index really starts
|
||||
/// place as `index_off + i * stride`, and that is only a place if the index really starts
|
||||
/// there and really is that wide.
|
||||
pub fn decode(bytes: &[u8]) -> Result<Self, Bad> {
|
||||
if bytes.len() < HEADER_LEN_V3 || &bytes[0..4] != MAGIC || bytes[4] != VERSION_V3 {
|
||||
let version = bytes.get(4).copied().ok_or(Bad::Magic)?;
|
||||
if bytes.len() < HEADER_LEN_V3 || &bytes[0..4] != MAGIC || (version != VERSION_V3 && version != VERSION_V4) {
|
||||
return Err(Bad::Magic);
|
||||
}
|
||||
let geometry = V3 {
|
||||
version,
|
||||
record_count: le_u64(bytes, 14),
|
||||
space_count: le_u64(bytes, 22),
|
||||
index_off: le_u64(bytes, 30),
|
||||
@@ -180,7 +207,7 @@ impl V3 {
|
||||
return Err(Bad::Extent);
|
||||
}
|
||||
let table_end = HEADER_LEN_V3 as u64 + geometry.space_count * SPACE_ENTRY as u64;
|
||||
let index_end = geometry.index_off + geometry.record_count * INDEX_ENTRY as u64;
|
||||
let index_end = geometry.index_off + geometry.record_count * geometry.index_stride() as u64;
|
||||
if geometry.index_off != table_end
|
||||
|| geometry.values_off != index_end
|
||||
|| geometry.values_off > image_bytes
|
||||
@@ -192,12 +219,18 @@ impl V3 {
|
||||
|
||||
/// Write the header this geometry describes. `image_bytes` is filled in by the caller's
|
||||
/// arithmetic, since only it knows how long the values are.
|
||||
///
|
||||
/// The version written is the geometry's own rather than a constant. A v4 geometry has a
|
||||
/// 42-byte index, and a header saying v3 would tell every reader to walk it 40 bytes at a time
|
||||
/// — one wrong byte that silently mis-addresses the whole store. This had no callers when it
|
||||
/// was written, so the mistake cost nothing; it has one now, and finding it before the first
|
||||
/// v4 image is written is the whole value of fixing it here.
|
||||
pub fn encode(&self, out: &mut [u8], curve: u8, image_bytes: u64) -> Result<usize, Bad> {
|
||||
if out.len() < HEADER_LEN_V3 {
|
||||
return Err(Bad::Extent);
|
||||
}
|
||||
out[0..4].copy_from_slice(MAGIC);
|
||||
out[4] = VERSION_V3;
|
||||
out[4] = self.version;
|
||||
out[5] = curve;
|
||||
out[6..14].copy_from_slice(&image_bytes.to_le_bytes());
|
||||
out[14..22].copy_from_slice(&self.record_count.to_le_bytes());
|
||||
@@ -229,14 +262,24 @@ impl V3 {
|
||||
if at >= self.record_count {
|
||||
return None;
|
||||
}
|
||||
let off = self.index_off as usize + at as usize * INDEX_ENTRY;
|
||||
if off + INDEX_ENTRY > bytes.len() {
|
||||
let stride = self.index_stride();
|
||||
let off = self.index_off as usize + at as usize * stride;
|
||||
if off + stride > bytes.len() {
|
||||
return None;
|
||||
}
|
||||
// The flag field exists only in v4; a v3 entry reads as a zero mask, which is honest —
|
||||
// "no class" — and matches nothing in a scan.
|
||||
let flags = if self.version == VERSION_V4 {
|
||||
le_u16(bytes, off + RAW_KEY_LEN)
|
||||
} else {
|
||||
0
|
||||
};
|
||||
let vo = off + RAW_KEY_LEN + if self.version == VERSION_V4 { FLAGS_LEN } else { 0 };
|
||||
let entry = IndexEntry {
|
||||
key: bytes[off..off + RAW_KEY_LEN].try_into().ok()?,
|
||||
value_off: le_u64(bytes, off + RAW_KEY_LEN),
|
||||
value_len: le_u64(bytes, off + RAW_KEY_LEN + 8),
|
||||
flags,
|
||||
value_off: le_u64(bytes, vo),
|
||||
value_len: le_u64(bytes, vo + 8),
|
||||
};
|
||||
// A value that is not inside the image is a truncated image, not an empty one.
|
||||
if entry.value_off < self.values_off
|
||||
@@ -258,10 +301,12 @@ impl V3 {
|
||||
}
|
||||
}
|
||||
|
||||
/// A v3 index entry: the key, and where its value lies.
|
||||
/// An addressed index entry: the key, its class mask, and where its value lies.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct IndexEntry<'a> {
|
||||
pub key: &'a [u8; RAW_KEY_LEN],
|
||||
/// The class mask (v4), or zero (v3, where no mask was written).
|
||||
pub flags: u16,
|
||||
pub value_off: u64,
|
||||
pub value_len: u64,
|
||||
}
|
||||
@@ -304,16 +349,25 @@ pub struct WalEntry<'a> {
|
||||
pub space: &'a [u8; SPACE_ID_LEN],
|
||||
pub key: &'a [u8; RAW_KEY_LEN],
|
||||
pub op: u8,
|
||||
/// The class mask (WAL version 2), or zero (version 1, where no mask was written).
|
||||
pub flags: u16,
|
||||
pub value: &'a [u8],
|
||||
pub next: usize,
|
||||
}
|
||||
|
||||
/// Read one log entry at `off`.
|
||||
/// Read one log entry at `off`. The log slice includes its header, so the version at `log[4]`
|
||||
/// decides the entry's stride: version 2 carries a two-byte class mask before the length.
|
||||
///
|
||||
/// The checksum covers space, key, length and value, so a torn tail is stopped at rather than
|
||||
/// applied — the rule the fold and every reader share.
|
||||
/// The checksum covers space, key, the mask, length and value, so a torn tail is stopped at rather
|
||||
/// than applied — the rule the fold and every reader share.
|
||||
pub fn wal_entry(log: &[u8], off: usize) -> Option<WalEntry<'_>> {
|
||||
if off + ENTRY_FIXED > log.len() {
|
||||
let version = log.get(4).copied().unwrap_or(WAL_VERSION);
|
||||
let (fixed, len_at) = if version == WAL_VERSION_V2 {
|
||||
(ENTRY_FIXED_V2, RAW_KEY_LEN + FLAGS_LEN + SPACE_ID_LEN + 5)
|
||||
} else {
|
||||
(ENTRY_FIXED, 61)
|
||||
};
|
||||
if off + fixed > log.len() {
|
||||
return None;
|
||||
}
|
||||
let op = log[off];
|
||||
@@ -321,19 +375,25 @@ pub fn wal_entry(log: &[u8], off: usize) -> Option<WalEntry<'_>> {
|
||||
return None;
|
||||
}
|
||||
let crc = le_u32(log, off + 1);
|
||||
let len = le_u32(log, off + 61) as usize;
|
||||
let frame_end = off + ENTRY_FIXED + len;
|
||||
let len = le_u32(log, off + len_at) as usize;
|
||||
let frame_end = off + fixed + len;
|
||||
if frame_end > log.len() {
|
||||
return None;
|
||||
}
|
||||
if crc32(&log[off + 5..frame_end]) != crc {
|
||||
return None;
|
||||
}
|
||||
let flags = if version == WAL_VERSION_V2 {
|
||||
le_u16(log, off + SPACE_ID_LEN + RAW_KEY_LEN + 5)
|
||||
} else {
|
||||
0
|
||||
};
|
||||
Some(WalEntry {
|
||||
space: log[off + 5..off + 37].try_into().ok()?,
|
||||
key: log[off + 37..off + 61].try_into().ok()?,
|
||||
op,
|
||||
value: &log[off + ENTRY_FIXED..frame_end],
|
||||
flags,
|
||||
value: &log[off + fixed..frame_end],
|
||||
next: frame_end,
|
||||
})
|
||||
}
|
||||
@@ -357,6 +417,15 @@ pub fn le_u32(bytes: &[u8], at: usize) -> u32 {
|
||||
u32::from_le_bytes(w)
|
||||
}
|
||||
|
||||
/// A little-endian `u16` at `at`, with the same rule.
|
||||
pub fn le_u16(bytes: &[u8], at: usize) -> u16 {
|
||||
let mut w = [0u8; 2];
|
||||
if at + 2 <= bytes.len() {
|
||||
w.copy_from_slice(&bytes[at..at + 2]);
|
||||
}
|
||||
u16::from_le_bytes(w)
|
||||
}
|
||||
|
||||
/// CRC-32 (IEEE 802.3), bitwise.
|
||||
///
|
||||
/// A corruption check, not a security check: it catches a torn write or a flipped bit, and says
|
||||
|
||||
+103
-7
@@ -30,10 +30,17 @@
|
||||
* encoding, which must produce exactly the key a userspace reader decodes.
|
||||
*/
|
||||
int cubelinux_kernel_put(const __u8 *space, __u64 x, __u64 y, __u64 z,
|
||||
const void *value, size_t len);
|
||||
const void *value, size_t len, __u16 flags);
|
||||
ssize_t cubelinux_kernel_get(const __u8 *space, __u64 x, __u64 y, __u64 z,
|
||||
void *buf, size_t len);
|
||||
void *buf, size_t len, __u16 *out_flags);
|
||||
int cubelinux_kernel_del(const __u8 *space, __u64 x, __u64 y, __u64 z);
|
||||
/*
|
||||
* Capture a request this layer refused, as a classed record in the events space. The refusals worth
|
||||
* recording are the ones that leave the store usable — a caller asking for something the interface
|
||||
* does not offer — because those are the moments the machine said no and carried on. It takes the
|
||||
* driver's write lock itself, so it must be called from here and not from inside an op.
|
||||
*/
|
||||
int cubelinux_kernel_capture_refusal(__u64 op, int errno, const __u8 *kind, size_t kind_len);
|
||||
int cubelinux_kernel_sync(void);
|
||||
|
||||
/*
|
||||
@@ -56,6 +63,15 @@ int cubelinux_kernel_range(const __u8 *space,
|
||||
__u64 cursor, void *buf, size_t cap,
|
||||
__u64 *out_len, __u64 *out_cursor);
|
||||
|
||||
/*
|
||||
* The flag scan (CUBE_OP_FLAG_SCAN). The mask and its mode travel as plain numbers: which bits mean
|
||||
* what is a vocabulary's business, and the kernel never interprets one — it compares masks, which is
|
||||
* what lets a new vocabulary attach without a format change.
|
||||
*/
|
||||
int cubelinux_kernel_flag_scan(const __u8 *space, __u16 mask, __u16 mode,
|
||||
__u32 every_space, __u64 cursor, void *buf, size_t cap,
|
||||
__u64 *out_len, __u64 *out_cursor);
|
||||
|
||||
/* The store device path, resolved from the `cube_store=` boot parameter at boot. */
|
||||
const char *cubelinux_store_device(void);
|
||||
|
||||
@@ -147,8 +163,10 @@ static long cube_args_op(unsigned int op, void __user *uargs)
|
||||
* The size is the caller's, and it must be the one this kernel implements: a caller
|
||||
* built against a later block would otherwise have fields silently ignored.
|
||||
*/
|
||||
if (args.size != sizeof(struct cube_args))
|
||||
if (args.size != sizeof(struct cube_args)) {
|
||||
cubelinux_kernel_capture_refusal(op, EINVAL, "bad-size", 8);
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
switch (op) {
|
||||
case CUBE_OP_PUT:
|
||||
@@ -158,12 +176,15 @@ static long cube_args_op(unsigned int op, void __user *uargs)
|
||||
case CUBE_OP_SYNC:
|
||||
break;
|
||||
default:
|
||||
cubelinux_kernel_capture_refusal(op, EINVAL, "bad-op", 6);
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
if (op == CUBE_OP_PUT || op == CUBE_OP_GET) {
|
||||
if (args.len > CUBE_MAX_VALUE)
|
||||
if (args.len > CUBE_MAX_VALUE) {
|
||||
cubelinux_kernel_capture_refusal(op, E2BIG, "too-big", 7);
|
||||
return -E2BIG;
|
||||
}
|
||||
if (args.len > 0) {
|
||||
buf = kvmalloc(args.len, GFP_KERNEL);
|
||||
if (!buf)
|
||||
@@ -179,14 +200,16 @@ static long cube_args_op(unsigned int op, void __user *uargs)
|
||||
break;
|
||||
}
|
||||
ret = cubelinux_kernel_put(args.coord.space, args.coord.x,
|
||||
args.coord.y, args.coord.z, buf, args.len);
|
||||
args.coord.y, args.coord.z, buf, args.len,
|
||||
args.flags);
|
||||
break;
|
||||
|
||||
case CUBE_OP_GET: {
|
||||
ssize_t got;
|
||||
|
||||
got = cubelinux_kernel_get(args.coord.space, args.coord.x,
|
||||
args.coord.y, args.coord.z, buf, args.len);
|
||||
args.coord.y, args.coord.z, buf, args.len,
|
||||
&args.flags);
|
||||
if (got < 0) {
|
||||
ret = got;
|
||||
break;
|
||||
@@ -253,8 +276,24 @@ static long cube_enum_op(unsigned int op, void __user *uargs)
|
||||
__u8 found[32];
|
||||
|
||||
ret = cubelinux_kernel_spaces(e.cursor, found);
|
||||
/*
|
||||
* The driver answers with one space written and 0, or with a negative errno. A
|
||||
* positive value is neither, and it must not be read as either: the buffer was not
|
||||
* written, so a caller handed this answer copies a space nobody found.
|
||||
*
|
||||
* This arm is the only one that advances the cursor by itself. Every other walk
|
||||
* leaves the cursor where the caller put it, so a bogus return there ends the walk
|
||||
* on the client's no-progress rule. Here it would hand the walk a cursor that keeps
|
||||
* moving over a space that is not there — which is not a hypothetical: a kernel error
|
||||
* code that had lost its sign did exactly that, and the walk served ~200,000 invented
|
||||
* records a minute until the machine stopped making progress. The cause is fixed
|
||||
* where it was (the driver's errno conversions, cubelinux_store.rs); this is the
|
||||
* bound that keeps a defect of that shape from becoming a loop again.
|
||||
*/
|
||||
if (ret < 0)
|
||||
return ret;
|
||||
if (ret != 0)
|
||||
return -EINVAL;
|
||||
memcpy(e.space, found, sizeof(found));
|
||||
/* An index here, not a count of records: hand back the one after this space. */
|
||||
e.cursor = e.cursor + 1;
|
||||
@@ -361,7 +400,62 @@ static long cube_range_op(void __user *uargs)
|
||||
}
|
||||
|
||||
/*
|
||||
* One syscall, three argument blocks. They share a prefix — `size`, then `op` — so the size the
|
||||
* The flag scan — CUBE_OP_FLAG_SCAN — which travels in `struct cube_flag_scan_args`.
|
||||
*
|
||||
* The walks' shape again, because it is the walks' contract: a batch fills the caller's buffer and
|
||||
* returns how much was used plus the cursor to pass next; a record that does not fit ends the
|
||||
* batch; a record that cannot fit in any buffer the caller offered comes back as -ERANGE with `len`
|
||||
* saying what it would need.
|
||||
*
|
||||
* The one thing a caller must know beyond the walk's rules: the cursor counts the records that
|
||||
* **matched**, not the records examined, because the records the mask rejected are not answers.
|
||||
*/
|
||||
static long cube_flag_scan_op(void __user *uargs)
|
||||
{
|
||||
struct cube_flag_scan_args f;
|
||||
void *buf = NULL;
|
||||
long ret = 0;
|
||||
u64 out_len = 0, out_cursor = 0;
|
||||
|
||||
if (copy_from_user(&f, uargs, sizeof(f)))
|
||||
return -EFAULT;
|
||||
if (f.size != sizeof(struct cube_flag_scan_args) || f.op != CUBE_OP_FLAG_SCAN)
|
||||
return -EINVAL;
|
||||
if (f.mode != CUBE_FLAG_ANY && f.mode != CUBE_FLAG_ALL)
|
||||
return -EINVAL;
|
||||
if (f.every_space != CUBE_SPACE_ONE && f.every_space != CUBE_SPACE_EVERY)
|
||||
return -EINVAL;
|
||||
|
||||
if (f.len > CUBE_MAX_WALK)
|
||||
return -E2BIG;
|
||||
if (f.len > 0) {
|
||||
buf = kvmalloc(f.len, GFP_KERNEL);
|
||||
if (!buf)
|
||||
return -ENOMEM;
|
||||
}
|
||||
|
||||
ret = cubelinux_kernel_flag_scan(f.space, f.mask, f.mode, f.every_space,
|
||||
f.cursor, buf, f.len, &out_len, &out_cursor);
|
||||
if (ret == 0) {
|
||||
if (out_len > 0 && copy_to_user((void __user *)f.value, buf, out_len))
|
||||
ret = -EFAULT;
|
||||
f.len = out_len;
|
||||
f.cursor = out_cursor;
|
||||
if (copy_to_user(uargs, &f, sizeof(f)))
|
||||
ret = -EFAULT;
|
||||
} else if (ret == -ERANGE) {
|
||||
/* Nothing was written; `len` now says how much one record needs. */
|
||||
f.len = out_len;
|
||||
if (copy_to_user(uargs, &f, sizeof(f)))
|
||||
ret = -EFAULT;
|
||||
}
|
||||
|
||||
kvfree(buf);
|
||||
return ret;
|
||||
}
|
||||
|
||||
/*
|
||||
* One syscall, four argument blocks. They share a prefix — `size`, then `op` — so the size the
|
||||
* caller declares is what says which one arrived. That is the whole point of putting `size`
|
||||
* first: an interface that cannot grow has to be replaced, and this one grows by being given a
|
||||
* new block with a new size.
|
||||
@@ -378,5 +472,7 @@ SYSCALL_DEFINE2(cube, unsigned int, op, void __user *, uargs)
|
||||
return cube_enum_op(op, uargs);
|
||||
if (size == sizeof(struct cube_range_args))
|
||||
return cube_range_op(uargs);
|
||||
if (size == sizeof(struct cube_flag_scan_args))
|
||||
return cube_flag_scan_op(uargs);
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
+2127
-302
File diff suppressed because it is too large
Load Diff
+103
-6
@@ -34,6 +34,23 @@ struct cube_args {
|
||||
* out: on -ERANGE, the bytes that would be needed;
|
||||
* on success for CUBE_OP_GET, the bytes read.
|
||||
*/
|
||||
__u16 flags; /* in: CUBE_OP_PUT — the class mask to stamp on the record;
|
||||
* out: CUBE_OP_GET — the mask the record carries.
|
||||
*
|
||||
* One field for both directions because it is one thing: the
|
||||
* class of this record. A write states it, a read learns it,
|
||||
* and neither is a special case of the other. 0 is "no class",
|
||||
* which is what every record written before the field existed
|
||||
* reads as — so not classifying is not writing a special value,
|
||||
* and a read that found nothing leaves 0 rather than a stale
|
||||
* class for the caller to believe.
|
||||
*
|
||||
* A store whose image is the legacy packed layout has no field
|
||||
* to put a mask in, and drops it: that layout cannot carry a
|
||||
* class, and saying otherwise would be a lie about the bytes on
|
||||
* disk. A read from such a store answers 0 for the same reason.
|
||||
*/
|
||||
__u16 reserved; /* must be 0 */
|
||||
};
|
||||
|
||||
#define CUBE_OP_PUT 1 /* store bytes at a coordinate */
|
||||
@@ -43,15 +60,21 @@ struct cube_args {
|
||||
#define CUBE_OP_ENUM 5 /* walk the records of a space, in batches */
|
||||
#define CUBE_OP_SPACES 6 /* walk the spaces that hold records */
|
||||
#define CUBE_OP_RANGE 7 /* walk the records of a space that lie in a box */
|
||||
#define CUBE_OP_FLAG_SCAN 8 /* walk the records of a space whose class mask matches */
|
||||
|
||||
/*
|
||||
* The walk's argument block: its own block rather than a wider `cube_args`, because it needs a
|
||||
* cursor and a buffer, and the coordinate would otherwise be both an input and an output.
|
||||
*
|
||||
* Records are packed as `key(24) | value_len(u32, little-endian) | value`, in the store's own
|
||||
* order — space first, then key — which is the order a checkpoint writes them and the order the
|
||||
* userspace store returns them, so a kernel listing and a userspace listing can be compared
|
||||
* directly. The space is not repeated per record: the caller named it.
|
||||
* Records are packed as `key(24) | flags(2, little-endian) | value_len(u32, little-endian) |
|
||||
* value`, in the store's own order — space first, then key — which is the order a checkpoint writes
|
||||
* them and the order the userspace store returns them, so a kernel listing and a userspace listing
|
||||
* can be compared directly. The space is not repeated per record: the caller named it.
|
||||
*
|
||||
* The class mask is in the frame because a listing has to be able to say what a record *is*, and a
|
||||
* caller that cannot see it has to open the store itself to find out — which is the second reader
|
||||
* of the format this interface exists to make unnecessary. A record whose layout carries no mask (a
|
||||
* packed v1/v2 image) reads as zero: "no class", the same answer every other reader gives.
|
||||
*
|
||||
* A walk ends when the cursor stops moving, and that is the only end signal: a batch holds as
|
||||
* many whole records as fit, so most batches come back short, and reading a short batch as the
|
||||
@@ -80,8 +103,9 @@ struct cube_enum_args {
|
||||
* replaced.
|
||||
*
|
||||
* A region is a box, inclusive on both corners. Records come back packed exactly as CUBE_OP_ENUM
|
||||
* packs them — `key(24) | value_len(u32, little-endian) | value`, in the store's own order — so a
|
||||
* kernel region answer and a userspace one can be compared byte for byte.
|
||||
* packs them — `key(24) | flags(2, little-endian) | value_len(u32, little-endian) | value`, in the
|
||||
* store's own order — so a kernel region answer and a userspace one can be compared byte for
|
||||
* byte.
|
||||
*
|
||||
* The cursor is a COUNT OF RECORDS ALREADY RETURNED, and it is the end-of-walk signal for the same
|
||||
* reason as the walk's: a batch holds as many whole records as fit, so most batches come back
|
||||
@@ -89,6 +113,9 @@ struct cube_enum_args {
|
||||
* counts — a walk counts the records of the space, a region walk counts the records *in the box*,
|
||||
* because those are the records it returns.
|
||||
*
|
||||
* Records come back in the same frame the walk uses, class mask included:
|
||||
* `key(24) | flags(2) | value_len(u32) | value`.
|
||||
*
|
||||
* Implementation, because it is what makes this cheap: **the kernel seeks.** The box's two corner
|
||||
* keys bound every key inside it (`cube_format::key_span`), so a v3 image's sorted index is
|
||||
* binary-searched for the foot of that span and read forward to its head. The span is a BOUND, not
|
||||
@@ -111,4 +138,74 @@ struct cube_range_args {
|
||||
*/
|
||||
};
|
||||
|
||||
/*
|
||||
* The flag scan's argument block: `CUBE_OP_FLAG_SCAN`, the class-mask half of the store's
|
||||
* classification substrate (DESIGN-flag-vocabularies.md). Its own block for the walks' reason — it
|
||||
* needs a cursor and a buffer — and versioned by `size` like the other three.
|
||||
*
|
||||
* `mask` is a raw 16-bit class mask and this interface does not interpret a bit of it: the
|
||||
* vocabulary that owns the bits (events today, sealing / lineage / lifecycle later) is the only
|
||||
* thing that knows what they mean. `mode` says how to read the mask:
|
||||
*
|
||||
* CUBE_FLAG_ANY the record shares at least one bit with `mask` — "every error"
|
||||
* CUBE_FLAG_ALL the record carries every bit of `mask` — "every Wi-Fi error"
|
||||
*
|
||||
* A `mask` of zero matches *nothing*, not everything: naming no class is asking no question, and a
|
||||
* scan that answered a walk's worth of records to an empty question would be a walk wearing a
|
||||
* scan's hat.
|
||||
*
|
||||
* The cursor counts **matches already returned**, as the region walk's counts records in its box,
|
||||
* and for the same reason: the records examined before a match are not matches, so the index
|
||||
* position of the cursor-th match is not arithmetic. A batch holds as many whole records as fit, so
|
||||
* most batches come back short; reading a short batch as the end truncates the answer. A finished
|
||||
* scan answers with no records and the cursor unchanged.
|
||||
*
|
||||
* Records come back as `key(24) | flags(2, little-endian) | value_len(u32, little-endian) | value`
|
||||
* — the walk's frame with the class mask in it. The mask travels because a scan's answer has to say
|
||||
* what class each record answered with: a record can carry bits beyond the one asked for, and no
|
||||
* other operation returns a mask.
|
||||
*
|
||||
* Under CUBE_SPACE_EVERY the frame gains a leading `space(32)`, because it has to: a walk's frame
|
||||
* leaves the space out on the grounds that the caller named it, and a caller who named no space
|
||||
* cannot be told which one a record came from any other way. A coordinate is meaningless without
|
||||
* its space, so an every-space answer that omitted it would be unusable rather than merely terse.
|
||||
* The space leads because the (space, key) pair it forms is the order records are stored in and
|
||||
* returned in.
|
||||
*/
|
||||
struct cube_flag_scan_args {
|
||||
__u32 size; /* sizeof(struct cube_flag_scan_args) as the caller built it */
|
||||
__u32 op; /* CUBE_OP_FLAG_SCAN */
|
||||
__u8 space[32]; /* in: the space to scan, or ignored with CUBE_SPACE_EVERY */
|
||||
__u16 mask; /* in: the class mask to match; 0 matches nothing */
|
||||
__u16 mode; /* in: CUBE_FLAG_ANY or CUBE_FLAG_ALL */
|
||||
__u32 every_space; /* in: CUBE_SPACE_ONE or CUBE_SPACE_EVERY — see below */
|
||||
__u64 cursor; /* in: 0 to start, or what the last call returned;
|
||||
* out: what to pass next — see the end-of-scan rule above
|
||||
*/
|
||||
__u64 value; /* user pointer: where to put the records */
|
||||
__u64 len; /* in: the buffer's capacity;
|
||||
* out: bytes written, or on -ERANGE what would be needed
|
||||
*/
|
||||
};
|
||||
|
||||
#define CUBE_FLAG_ANY 0 /* the record shares at least one bit with the mask */
|
||||
#define CUBE_FLAG_ALL 1 /* the record carries every bit of the mask */
|
||||
|
||||
/*
|
||||
* The scan's scope. A space is a hard partition, so this is a choice between two different
|
||||
* questions and never a filter that can be widened by accident:
|
||||
*
|
||||
* CUBE_SPACE_ONE the records of `space` — "this class here"
|
||||
* CUBE_SPACE_EVERY the records of every space — "this class anywhere"
|
||||
*
|
||||
* It is a field and not a reserved space id because there is no such id to reserve: every 32-byte
|
||||
* value is a legitimate space (root `0x00` and edge `0xFF…FF` are both in use), so a sentinel would
|
||||
* be a space somebody could name. The field occupies what was padding, so `sizeof` is unchanged —
|
||||
* which matters, because `size` is what says which argument block arrived and `cube_args` is
|
||||
* exactly eight bytes wider. A caller that zeroes its block (and every caller does) gets
|
||||
* CUBE_SPACE_ONE, which is the narrower question.
|
||||
*/
|
||||
#define CUBE_SPACE_ONE 0 /* scan only `space` */
|
||||
#define CUBE_SPACE_EVERY 1 /* scan every space */
|
||||
|
||||
#endif /* _UAPI_LINUX_CUBE_H */
|
||||
|
||||
Reference in New Issue
Block a user