API reference — core#
Scalar/device vocabulary and the shape machinery everything else indexes
through: data types, devices, errors, sample indexing and the layout
mappings a View (API reference — memory) is built from.
aether::DType#
-
struct DType#
Thin wrapper around
DLDataType(code/bits/lanes) — constexpr constructible and safe to use inAETHER_DEVICEHOST()code.Public Functions
-
inline constexpr DType(std::uint8_t code, std::uint8_t bits, std::uint16_t lanes = 1)#
Construct from a raw
(code, bits, lanes)triple.
-
inline explicit constexpr DType(DLDataType dl)#
Construct by wrapping an existing
DLDataTypeverbatim.
-
inline constexpr std::uint8_t code() const#
DLPack type code (see
DLDataTypeCodein dlpack.h).
-
inline constexpr std::uint8_t bits() const#
Number of bits per lane.
-
inline constexpr std::uint16_t lanes() const#
Number of lanes (1 for a plain scalar).
Public Members
-
DLDataType raw = {}#
The wrapped DLPack data-type descriptor.
-
inline constexpr DType(std::uint8_t code, std::uint8_t bits, std::uint16_t lanes = 1)#
aether::Device#
-
struct Device#
Thin wrapper around
DLDevice(device_type/device_id) — constexpr constructible and safe to use inAETHER_DEVICEHOST()code.Public Functions
-
inline constexpr Device(DLDeviceType type, std::int32_t id = 0)#
Construct from a device kind and (optional) device index.
-
inline explicit constexpr Device(DLDevice dl)#
Construct by wrapping an existing
DLDeviceverbatim.
-
inline constexpr DLDeviceType type() const#
The DLPack device kind.
-
inline constexpr std::int32_t id() const#
The device index (0 for CPU / pinned / managed memory).
-
inline constexpr bool is_cpu() const#
trueforkDLCPU.
-
inline constexpr bool is_cuda() const#
trueforkDLCUDA(device memory).
-
inline constexpr bool is_cuda_host() const#
trueforkDLCUDAHost(pinned host memory).
Public Members
-
DLDevice raw = {kDLCPU, 0}#
The wrapped DLPack device descriptor.
-
inline constexpr Device(DLDeviceType type, std::int32_t id = 0)#
aether::Error#
-
class Error : public std::runtime_error#
aether’s exception type.
aether::SampleIndex / aether::BundleIndex#
-
struct SampleIndex#
Sample index: both the global (grid-wide) index and the work (block-local / shared-memory) index for one sample.
Both members are
offset_t(32-bit). This is the HEAD of the address chain — astd::size_there forcesblockIdx.x * blockDim.x + threadIdx.xinto a 64-bit multiply-add and then a truncation at the firstmappingcall, which is precisely the seamindex/Offset.hdocuments. The GPU built-ins (threadIdx.x& co.) are alreadyunsigned, so the three- argument factory is an EXACT match with no conversion at all.Public Functions
-
inline constexpr offset_t global() const#
The global (grid-wide) sample index.
-
inline constexpr offset_t work() const#
The block-local index used for work (shared) memory arrays.
Public Members
-
offset_t globalIdx_#
Public data members (POD layout).
Public Static Functions
-
static inline constexpr SampleIndex make(offset_t threadIdxx, offset_t blockIdxx, offset_t blockDimx)#
Construct from GPU built-in thread/block indices.
- Parameters:
threadIdxx –
threadIdx.x.blockIdxx –
blockIdx.x.blockDimx –
blockDim.x.
- Returns:
SampleIndexwithglobal() == blockIdxx*blockDimx + threadIdxxandwork() == threadIdxx.
-
static inline constexpr SampleIndex make(std::size_t index)#
Construct from a flat host index —
global() == work() == index. Keeps astd::size_tparameter so the ubiquitous host spellingfor (std::size_t idx = ...) SampleIndex::make(idx)needs no cast at the call site; the narrowing happens once, here.
-
inline constexpr offset_t global() const#
-
template<std::size_t W>
struct BundleIndex# A contiguous group of
Wsamples, one thread’s worth of bundled work.Wis expected to be 2 or 4 (the 128-bit-LDG/STG-eligible widthsbackend/cuda/bundle/LoadStore.htargets), though nothing here hard-codes that — other widths simply never hit the vectorized memory path and always take the scalar per-lane fallback (still correct).Public Functions
-
inline constexpr offset_t base() const#
The global sample index of lane 0.
-
inline constexpr SampleIndex lane(std::size_t k) const#
The scalar
SampleIndexfor lanek(0 <= k < W):global() == base()+k,work() == k(the lane selector — see file docstring).
Public Members
-
offset_t base_#
The global sample index of lane 0.
Public Static Functions
-
static inline constexpr BundleIndex make(offset_t threadIdxx, offset_t blockIdxx, offset_t blockDimx)#
Construct from GPU built-in thread/block indices — mirrors
SampleIndex::make’s three-argument device factory exactly, scaled byW: threadtin blockb(block sized) owns samples[W*(b*d+t), W*(b*d+t)+W).
-
static inline constexpr BundleIndex make(std::size_t i)#
Construct from a flat host bundle index — bundle
icovers samples[W*i, W*i+W). Keeps astd::size_tparameter so the ubiquitous host spelling needs no cast at the call site (same convention asSampleIndex::make(std::size_t)).
Public Static Attributes
-
struct Mask#
Per-lane active mask for a bundle straddling the end of an
n-sample array (PacketMask-stylefirstN).
-
inline constexpr offset_t base() const#
Shape and layout#
-
template<std::size_t... Es>
class extents# Compile-time-and-runtime shape:
Es...are the modes, each either a static extent oraether::dyn.The constructor takes exactly
rank_dynamic()std::size_tvalues, in order, one perdynslot.Public Functions
-
inline constexpr offset_t extent(std::size_t i) const#
The runtime extent of mode
i— resolvesdynmodes from stored values. Returnsoffset_t: this is the value the row-major fold multiplies by, so it is part of the narrow hot chain. Storage staysstd::size_t(the constructor’s API is unchanged) — truncating a stored 64-bit value at the read is free on both targets (it is the low half of the register pair), whereas doing the arithmetic wide is not.Use
extentWide()where the untruncated value is the point, i.e. the span guard.
-
inline constexpr std::size_t extentWide(std::size_t i) const#
The runtime extent of mode
iat fullstd::size_twidth — the accessorrequired_span_size()(and therefore the span guard) reads, so an over-large shape is still visible as over-large rather than silently wrapped by the very narrowing the guard exists to police.
-
constexpr extents() = default#
Default-constructible when
rank_dynamic() == 0(all-static extents).
-
template<class ...DynVals>
inline explicit constexpr extents(DynVals... vals)# Construct from exactly
rank_dynamic()dynamic-mode values, in order.
Public Static Functions
-
static inline constexpr std::size_t rank()#
Number of modes (mdspan-mirrored name).
-
static inline constexpr std::size_t rank_dynamic()#
Number of
dynmodes (mdspan-mirrored name).
-
static inline constexpr std::size_t static_extent(std::size_t i)#
The compile-time extent of mode
i(dynif that mode is dynamic).
-
inline constexpr offset_t extent(std::size_t i) const#
-
struct layout_right#
Row-major layout policy: the rightmost index varies fastest. Default layout for
ViewandArray.-
template<class Extents>
class mapping# Public Functions
-
template<class ...Idxs>
inline constexpr offset_t operator()(Idxs... idxs) const# Row-major fold:
((idx0)*extent(1) + idx1)*extent(2) + idx2 ..., entirely inoffset_t.
-
inline constexpr std::size_t required_span_size() const#
Product of every extent — the number of elements this mapping spans, at FULL width (the bounds guard’s input; see the file note).
-
template<class ...Idxs>
-
template<class Extents>
-
struct layout_stride#
Strided layout policy: an explicit per-mode stride (in elements) replaces the row-major formula. This is the shape foreign (e.g. DLPack) strides land in — construct via
.mapping(extents,
strides)
-
template<class Extents>
class mapping# Public Functions
-
template<class ...Idxs>
inline constexpr offset_t operator()(Idxs... idxs) const# dot(idxs, strides), inoffset_t(same narrow chain aslayout_right; the strides themselves stay stored at full width sorequired_span_size()below can still see an out-of-range span and the bounds guard can reject it).
-
inline constexpr std::size_t required_span_size() const#
0if any extent is0; otherwise1 + sum((extent(i)-1) * stride(i))— the one-past-the-last offset a compliant strided layout can address (mdspan’s own formula).
-
template<class ...Idxs>
-
template<class Extents>