The seventh operation: `cube(2)` gains CUBE_OP_RANGE, a bounded walk of one space's records that lie in a box. Its argument block is its own (`cube_range_args`, versioned by `size` like the walk's), and the box travels as its six numbers for the reason a coordinate does — the key it has to become is the driver's business. The operation rests on the span that the shared file already owns. `key_span(lo, hi)` bounds every key in the box because the interleave is monotone on each axis, so a v3 image is **sought**: the fixed-stride index is binary-searched for the span's foot (`Addressed::lower_bound`) and read forward to its head, merging the log's edits exactly as a walk does. A packed v1/v2 image has no index to search, so its space is walked with the same span used only to stop early — and the contract is the same either way, so a caller is not told which path it got. The trap that shaped the code, and the reason it is written the way it is: **the span is a bound, not the set.** Keys of points outside the box fall inside it (Z-order amplification), so every candidate is decoded and tested against the box before it is returned — which is what `morton_decode`, the interleave's inverse, is for, now in the shared file with the same kind of hand-pinned tests the interleave has. And the cursor counts the records *in the box*, not the records of the space, because those are the records the walk returns. The over-coverage — records examined versus records returned — is the number this operation is meant to publish, and it is not wired to the caller yet: the cost gate measures it by comparing against the userspace store, which counts the same thing.
115 lines
5.5 KiB
C
115 lines
5.5 KiB
C
/* SPDX-License-Identifier: GPL-2.0 WITH Linux-syscall-note */
|
|
/*
|
|
* CUBELinux: the coordinate interface.
|
|
*
|
|
* The kernel's native interface to the store. Operations are the verbs the command language
|
|
* already defines — one language, and this is its kernel form (DESIGN-cube-interface.md).
|
|
*
|
|
* A coordinate is not a name and not a path: it is *where* a record is, and knowing it is the
|
|
* authorisation to use it. Nothing here resolves a name, and nothing enumerates.
|
|
*/
|
|
#ifndef _UAPI_LINUX_CUBE_H
|
|
#define _UAPI_LINUX_CUBE_H
|
|
|
|
#include <linux/types.h>
|
|
|
|
/* Which cube, and where in it. 32-byte space: unguessable, deliberately. */
|
|
struct cube_coord {
|
|
__u8 space[32];
|
|
__u64 x;
|
|
__u64 y;
|
|
__u64 z;
|
|
};
|
|
|
|
/*
|
|
* The argument block. `size` first, and checked: an interface that cannot grow is an
|
|
* interface that has to be replaced, and syscall numbers are permanent.
|
|
*/
|
|
struct cube_args {
|
|
__u32 size; /* sizeof(struct cube_args) as the caller built it */
|
|
__u32 op; /* CUBE_OP_* */
|
|
struct cube_coord coord; /* unused by CUBE_OP_SYNC */
|
|
__u64 value; /* user pointer: bytes to write, or where to put them */
|
|
__u64 len; /* in: bytes offered, or the buffer's capacity.
|
|
* out: on -ERANGE, the bytes that would be needed;
|
|
* on success for CUBE_OP_GET, the bytes read.
|
|
*/
|
|
};
|
|
|
|
#define CUBE_OP_PUT 1 /* store bytes at a coordinate */
|
|
#define CUBE_OP_GET 2 /* read them back */
|
|
#define CUBE_OP_DEL 3 /* remove the record */
|
|
#define CUBE_OP_SYNC 4 /* fold the log into the image */
|
|
#define CUBE_OP_ENUM 5 /* walk the records of a space, in batches */
|
|
#define CUBE_OP_SPACES 6 /* walk the spaces that hold records */
|
|
#define CUBE_OP_RANGE 7 /* walk the records of a space that lie in a box */
|
|
|
|
/*
|
|
* The walk's argument block: its own block rather than a wider `cube_args`, because it needs a
|
|
* cursor and a buffer, and the coordinate would otherwise be both an input and an output.
|
|
*
|
|
* Records are packed as `key(24) | value_len(u32, little-endian) | value`, in the store's own
|
|
* order — space first, then key — which is the order a checkpoint writes them and the order the
|
|
* userspace store returns them, so a kernel listing and a userspace listing can be compared
|
|
* directly. The space is not repeated per record: the caller named it.
|
|
*
|
|
* A walk ends when the cursor stops moving, and that is the only end signal: a batch holds as
|
|
* many whole records as fit, so most batches come back short, and reading a short batch as the
|
|
* end truncates a listing to its first batch. CUBE_OP_ENUM answers a finished walk with no
|
|
* records and the cursor unchanged; CUBE_OP_SPACES answers it with -ENOENT, because asking for
|
|
* the space after the last one is asking for a space that is not there. -ERANGE keeps its usual
|
|
* meaning: not one whole record fits, and `len` says how much one needs.
|
|
*/
|
|
struct cube_enum_args {
|
|
__u32 size; /* sizeof(struct cube_enum_args) as the caller built it */
|
|
__u32 op; /* CUBE_OP_ENUM or CUBE_OP_SPACES */
|
|
__u8 space[32]; /* in: the space to walk; out: the space found (CUBE_OP_SPACES) */
|
|
__u64 cursor; /* in: 0 to start, or what the last call returned;
|
|
* out: what to pass next — see the end-of-walk rule above
|
|
*/
|
|
__u64 value; /* user pointer: where to put the records */
|
|
__u64 len; /* in: the buffer's capacity;
|
|
* out: bytes written, or on -ERANGE what would be needed
|
|
*/
|
|
};
|
|
|
|
/*
|
|
* The region walk's argument block: its own block, for the walk's reason — it needs a box, a cursor
|
|
* and a buffer, and the coordinate would otherwise be both an input and an output. It is versioned
|
|
* by `size` like the other two, so this interface grows by gaining a block rather than by being
|
|
* replaced.
|
|
*
|
|
* A region is a box, inclusive on both corners. Records come back packed exactly as CUBE_OP_ENUM
|
|
* packs them — `key(24) | value_len(u32, little-endian) | value`, in the store's own order — so a
|
|
* kernel region answer and a userspace one can be compared byte for byte.
|
|
*
|
|
* The cursor is a COUNT OF RECORDS ALREADY RETURNED, and it is the end-of-walk signal for the same
|
|
* reason as the walk's: a batch holds as many whole records as fit, so most batches come back
|
|
* short, and reading a short batch as the end truncates the answer. The one difference is what it
|
|
* counts — a walk counts the records of the space, a region walk counts the records *in the box*,
|
|
* because those are the records it returns.
|
|
*
|
|
* Implementation, because it is what makes this cheap: **the kernel seeks.** The box's two corner
|
|
* keys bound every key inside it (`cube_format::key_span`), so a v3 image's sorted index is
|
|
* binary-searched for the foot of that span and read forward to its head. The span is a BOUND, not
|
|
* the set — records *outside* the box also have keys inside it (Z-order amplification) — so each
|
|
* candidate is decoded and tested against the box before it is returned. More records may therefore
|
|
* be examined than are returned, and that over-coverage is the honest cost of the span.
|
|
*/
|
|
struct cube_range_args {
|
|
__u32 size; /* sizeof(struct cube_range_args) as the caller built it */
|
|
__u32 op; /* CUBE_OP_RANGE */
|
|
__u8 space[32]; /* in: the space to search */
|
|
__u64 lo[3]; /* in: the region's near corner, inclusive */
|
|
__u64 hi[3]; /* in: the region's far corner, inclusive */
|
|
__u64 cursor; /* in: 0 to start, or what the last call returned;
|
|
* out: what to pass next — see the end-of-walk rule above
|
|
*/
|
|
__u64 value; /* user pointer: where to put the records */
|
|
__u64 len; /* in: the buffer's capacity;
|
|
* out: bytes written, or on -ERANGE what would be needed
|
|
*/
|
|
};
|
|
|
|
#endif /* _UAPI_LINUX_CUBE_H */
|