/* SPDX-License-Identifier: GPL-2.0 WITH Linux-syscall-note */ /* * CUBELinux: the coordinate interface. * * The kernel's native interface to the store. Operations are the verbs the command language * already defines — one language, and this is its kernel form (DESIGN-cube-interface.md). * * A coordinate is not a name and not a path: it is *where* a record is, and knowing it is the * authorisation to use it. Nothing here resolves a name, and nothing enumerates. */ #ifndef _UAPI_LINUX_CUBE_H #define _UAPI_LINUX_CUBE_H #include /* Which cube, and where in it. 32-byte space: unguessable, deliberately. */ struct cube_coord { __u8 space[32]; __u64 x; __u64 y; __u64 z; }; /* * The argument block. `size` first, and checked: an interface that cannot grow is an * interface that has to be replaced, and syscall numbers are permanent. */ struct cube_args { __u32 size; /* sizeof(struct cube_args) as the caller built it */ __u32 op; /* CUBE_OP_* */ struct cube_coord coord; /* unused by CUBE_OP_SYNC */ __u64 value; /* user pointer: bytes to write, or where to put them */ __u64 len; /* in: bytes offered, or the buffer's capacity. * out: on -ERANGE, the bytes that would be needed; * on success for CUBE_OP_GET, the bytes read. */ }; #define CUBE_OP_PUT 1 /* store bytes at a coordinate */ #define CUBE_OP_GET 2 /* read them back */ #define CUBE_OP_DEL 3 /* remove the record */ #define CUBE_OP_SYNC 4 /* fold the log into the image */ #define CUBE_OP_ENUM 5 /* walk the records of a space, in batches */ #define CUBE_OP_SPACES 6 /* walk the spaces that hold records */ #define CUBE_OP_RANGE 7 /* walk the records of a space that lie in a box */ /* * The walk's argument block: its own block rather than a wider `cube_args`, because it needs a * cursor and a buffer, and the coordinate would otherwise be both an input and an output. * * Records are packed as `key(24) | value_len(u32, little-endian) | value`, in the store's own * order — space first, then key — which is the order a checkpoint writes them and the order the * userspace store returns them, so a kernel listing and a userspace listing can be compared * directly. The space is not repeated per record: the caller named it. * * A walk ends when the cursor stops moving, and that is the only end signal: a batch holds as * many whole records as fit, so most batches come back short, and reading a short batch as the * end truncates a listing to its first batch. CUBE_OP_ENUM answers a finished walk with no * records and the cursor unchanged; CUBE_OP_SPACES answers it with -ENOENT, because asking for * the space after the last one is asking for a space that is not there. -ERANGE keeps its usual * meaning: not one whole record fits, and `len` says how much one needs. */ struct cube_enum_args { __u32 size; /* sizeof(struct cube_enum_args) as the caller built it */ __u32 op; /* CUBE_OP_ENUM or CUBE_OP_SPACES */ __u8 space[32]; /* in: the space to walk; out: the space found (CUBE_OP_SPACES) */ __u64 cursor; /* in: 0 to start, or what the last call returned; * out: what to pass next — see the end-of-walk rule above */ __u64 value; /* user pointer: where to put the records */ __u64 len; /* in: the buffer's capacity; * out: bytes written, or on -ERANGE what would be needed */ }; /* * The region walk's argument block: its own block, for the walk's reason — it needs a box, a cursor * and a buffer, and the coordinate would otherwise be both an input and an output. It is versioned * by `size` like the other two, so this interface grows by gaining a block rather than by being * replaced. * * A region is a box, inclusive on both corners. Records come back packed exactly as CUBE_OP_ENUM * packs them — `key(24) | value_len(u32, little-endian) | value`, in the store's own order — so a * kernel region answer and a userspace one can be compared byte for byte. * * The cursor is a COUNT OF RECORDS ALREADY RETURNED, and it is the end-of-walk signal for the same * reason as the walk's: a batch holds as many whole records as fit, so most batches come back * short, and reading a short batch as the end truncates the answer. The one difference is what it * counts — a walk counts the records of the space, a region walk counts the records *in the box*, * because those are the records it returns. * * Implementation, because it is what makes this cheap: **the kernel seeks.** The box's two corner * keys bound every key inside it (`cube_format::key_span`), so a v3 image's sorted index is * binary-searched for the foot of that span and read forward to its head. The span is a BOUND, not * the set — records *outside* the box also have keys inside it (Z-order amplification) — so each * candidate is decoded and tested against the box before it is returned. More records may therefore * be examined than are returned, and that over-coverage is the honest cost of the span. */ struct cube_range_args { __u32 size; /* sizeof(struct cube_range_args) as the caller built it */ __u32 op; /* CUBE_OP_RANGE */ __u8 space[32]; /* in: the space to search */ __u64 lo[3]; /* in: the region's near corner, inclusive */ __u64 hi[3]; /* in: the region's far corner, inclusive */ __u64 cursor; /* in: 0 to start, or what the last call returned; * out: what to pass next — see the end-of-walk rule above */ __u64 value; /* user pointer: where to put the records */ __u64 len; /* in: the buffer's capacity; * out: bytes written, or on -ERANGE what would be needed */ }; #endif /* _UAPI_LINUX_CUBE_H */