API reference — filtering#
Prefix-scan, scatter, and array-slicing for sample compaction. The two scan
engines live in the device namespaces as eagle::cuda::Scan and
eagle::cpu::Scan; eagle::filtering keeps the containers, the
compaction helper, and the native graph nodes.
eagle::cuda::Scan#
-
struct Scan#
Perform the Prefix sum (scan) algorithm.
Public Static Functions
-
template<typename T, typename OP, bool Inclusive, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
static inline void enqueue(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0, const cudaStream_t &stream = 0)# Enqueue the full scan on
stream, without capturing it.The launch sequence
graphcaptures, issued straight ontostream: nothing is allocated, nothing is synchronised and nothing is read back to the host, so a caller that is ALREADY capturingstream(a graph-building consumer) records it into its own graph, and an uncaptured caller simply runs it asynchronously.blockSumsmust hold at leastarr.samples()elements. The exclusive variant ends with a device-to-device copy; the inclusive one is kernels only.
-
template<typename T, typename OP, bool Inclusive, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
static inline cudaGraph_t graph(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0, const cudaStream_t &stream = 0)# Capture the full scan algorithm (
enqueue) into a graph.
-
template<typename T, typename OP, bool Inclusive, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
static inline void launch(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0, const cudaStream_t &stream = 0)# Instantiate and launch.
-
template<typename T, typename OP, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
static inline cudaGraph_t inclusiveGraph(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0, const cudaStream_t &stream = 0)# Run a full inclusive scan algorithm.
-
template<typename T, typename OP, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
static inline cudaGraph_t exclusiveGraph(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0, const cudaStream_t &stream = 0)# Run a full exclusive scan algorithm.
-
template<typename T, typename OP, bool Inclusive, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
eagle::cpu::Scan#
-
struct Scan#
OpenMP parallelized prefix sum (scan).
Public Static Functions
-
template<typename T, typename OP, bool Inclusive, unsigned int blockSize_ = 1024>
static inline void scan(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0)# Run the overall scan.
Context-aware: nested in a parallel region,
scanBlock_andapplyOffsets_work-share viaomp forand the sequential prefix is wrapped inomp single. Outside a parallel region, each step opens its ownparallel for.
-
template<typename T, typename OP, bool Inclusive, unsigned int blockSize_ = 1024>
Namespace eagle::filtering#
-
namespace filtering#
Enums
Functions
-
inline std::size_t compactScratchBytes(std::int64_t n)#
Bytes of scratch the compaction of
nsamples needs: threeidx_tplanes ofn(predicate, inclusive scan, scan block sums).
-
inline void compactDevice(const std::uint8_t *mask, std::uint32_t flags, std::int32_t *indexMap, std::uint32_t *count, void *scratch, std::int64_t n, cudaStream_t stream)#
Enqueue the compaction of
nsamples onstream(device memory).Kernels only, issued in order on
stream: predicate, inclusive scan, scatter. Legal whilestreamis being captured.
-
inline void compactDevice(const std::uint8_t *mask, std::uint32_t flags, std::int32_t *indexMap, std::uint32_t *count, void *scratch, std::int64_t n, cudaStream_t stream, const ReorderTrigger *trig)#
compactDevicethat also computes the reorder triggertrig(device words).trig == nullptris the call above, unchanged.One extra kernel (zeroing
live32); the predicate ballots and the scatter writesfirefrom the lane that writes the count. Capturable likewise.
-
inline void compactHost(const std::uint8_t *mask, std::uint32_t flags, std::int32_t *indexMap, std::uint32_t *count, void *scratch, std::int64_t n)#
The host face of
compactDeviceover host memory (OpenMP twins).
-
inline void compactHost(const std::uint8_t *mask, std::uint32_t flags, std::int32_t *indexMap, std::uint32_t *count, void *scratch, std::int64_t n, const ReorderTrigger *trig)#
compactHostthat also computes the reorder triggertrig(host words), by the same rule as the device face.trig == nullptris the call above, unchanged.
-
inline std::size_t reorderScratchBytes(std::int64_t n, std::uint32_t maxElemBytes)#
Bytes of scratch a reorder (or restore) of
nsamples needs: the compaction’s scan scratch, then one staging plane of the widest element (permincluded, so at least 4 bytes).
-
inline void reorderDevice(const ReorderPlane *planes, std::uint32_t nPlanes, const std::uint8_t *mask, std::uint32_t flags, std::int32_t *perm, std::int32_t *inv, std::int32_t *indexMap, std::uint32_t *count, std::uint32_t *span, std::uint32_t *fire, void *scratch, std::int64_t n, cudaStream_t stream)#
Enqueue the reorder of
nsamples onstream(device memory).Kernels only: predicate + inclusive scan of the mask, then stage + place for every plane and for
perm,invfromperm, the identity map, and last the words (count = span = kept,fire = 0). Legal whilestreamis being captured; with afireword it is a no-op on the device while*fire == 0(the scan still runs, into scratch).
-
inline void restoreDevice(const ReorderPlane *planes, std::uint32_t nPlanes, std::int32_t *perm, std::int32_t *inv, void *scratch, std::int64_t n, cudaStream_t stream)#
Enqueue the un-permutation of every plane on
stream:sampleiback at sloti;permandinv:= identity.A gather per plane over all
nslots. The index map, count and span are NOT touched (they describe the permuted layout): recompute the map and set the span back tonafterwards.
-
inline void reorderHost(const ReorderPlane *planes, std::uint32_t nPlanes, const std::uint8_t *mask, std::uint32_t flags, std::int32_t *perm, std::int32_t *inv, std::int32_t *indexMap, std::uint32_t *count, std::uint32_t *span, std::uint32_t *fire, void *scratch, std::int64_t n)#
The host face of
reorderDeviceover host memory (OpenMP); the same moves, so the two faces agree to the bit.
-
inline void restoreHost(const ReorderPlane *planes, std::uint32_t nPlanes, std::int32_t *perm, std::int32_t *inv, void *scratch, std::int64_t n)#
The host face of
restoreDeviceover host memory (OpenMP).
Variables
-
std::uint32_t kCompactMaskIsDrop = 1u#
Mask flag: a set byte drops the sample (otherwise it keeps it).
-
std::int64_t kCompactMaxSamples = INT32_MAX#
The largest sample count the
int32index map can address.
-
std::uint32_t kReorderMinSpan = 4096u#
The smallest span a reorder is worth firing for: below it the fixed launch cost of the moves exceeds what the restored coalescing saves.
-
std::uint32_t kReorderMaxElemBytes = 16#
The largest element a plane may carry, in bytes.
-
template<typename MyClass>
class FilteringSlice : public eagle::util::Slice<MyClass># - #include <Slice.h>
More flexible slice version that incorporates a scanner.
Public Types
Public Functions
-
FilteringSlice() = delete#
Default constructor is forbidden.
-
FilteringSlice(FilteringSlice &other) = delete#
Copy constructor is forbidden.
-
inline FilteringSlice(FilteringSlice &&other)#
Move constructor.
-
FilteringSlice &operator=(FilteringSlice &other) = delete#
Copy assignment is forbidden.
-
inline FilteringSlice &operator=(FilteringSlice &&other)#
Move assignment operator.
-
inline void hostUpdate()#
Run the scan and scatter on host.
-
inline cuda::CapturedGraph scanGraph()#
Capture the scan algorithm into a
CapturedGraph.Tags every kernel with
idealBlockSize = 0(grid-only mutation) — the warp-shuffle reduction layout depends on the capturedblockSizeand would be corrupted by re-tuning.
-
inline cuda::CapturedGraph scatterGraph()#
Capture the scatter kernel into a
CapturedGraphwith capEAGLE_BLOCKSIZE(element-wise; full re-tune OK).
-
inline void deviceUpdate(const bool &includeRetrieval = false)#
Run a full device scan.
-
inline const StreamT &stream() const#
Return a const reference to the CUDA stream associated to this slice.
-
inline void stream(const StreamT &stream)#
Set the cuda stream.
-
inline void upload(const StreamT &stream = 0, const bool &includeSliceData = true)#
Async memcpy from host to device - indexes.
-
inline void download(const StreamT &stream = 0, const bool &includeSliceData = true)#
Async memcpy from device to host.
-
inline void clearDevice()#
Mark the device copies stale (see
util::Slice::clearDevicefor why this no longer frees anything).
-
inline FilteringSlice clone() const#
Clone this slice.
-
FilteringSlice() = delete#
-
template<typename MyClass>
class FilteringSliceNode# - #include <FilteringSliceNode.h>
Stream compaction (predicate -> scan -> scatter) as a native node — the arena-backed twin of
FilteringSlice::deviceUpdate/hostUpdate, with both a CUDA and a host face.Builds the index map of the samples a predicate keeps, sourcing the scanner’s IN/OUT/BLOCKSUMS (3N) working buffer from the graph’s
ScratchArena. That whole 3N is INTERNAL to the node — IN is a copy of the predicate, OUT is consumed by the scatter, BLOCKSUMS is scan-internal — so a chain of interleaved compaction nodes shares one 3N slot (peak, not sum). The sliced object’s index map + active count (the compaction OUTPUT) are caller-owned payload, never arena scratch. The CUDA face (buildInto) captures predicate-copy -> exclusive scan (kFixedSize) ->CUDAscatter; the host face (runHost) runs the OpenMP twins over the same 3N slot: copy ->cpu::Scan->OMPscatter(the exactFilteringSlice::hostUpdatepath).- Template Parameters:
MyClass – the sliced object type (as
FilteringSlice<MyClass>).
Public Types
-
using StreamT = nativeStream_t#
Mode-agnostic stream type (
cudaStream_tunder CUDA,intinEAGLE_CPU_ONLY).
Public Functions
-
inline FilteringSliceNode(const SliceGRef &slice, const idx_t *predicate, idx_t n, const StreamT &stream = 0)#
- Parameters:
slice – View of the sliced object (
slice.deviceRef()/slice.hostRef()).predicate – Pointer to the N-element 0/1 keep-mask (device memory for the CUDA face, host memory for the host face).
n – Sample count.
stream – Capture stream (CUDA face).
-
inline void reserveScratch(ScratchArena &arena)#
Phase 1: reserve the scanner’s contiguous IN+OUT+BLOCKSUMS (3N).
-
inline idx_t buildInto(cuda::Graph &g, const std::vector<idx_t> &deps)#
Phase 2 (CUDA): capture predicate-copy -> scan -> scatter over the arena 3N buffer, wiring
depsto the first node. Returns the scatter node.
-
inline void runHost(cpu::Graph &g)#
Host face: run the OpenMP compaction twins over the committed arena’s 3N slot (host pointers), reconstructing the same SoA scanner view the CUDA face builds — copy predicate -> exclusive
cpu::Scan->OMPscatter(theFilteringSlice::hostUpdatepath). Each stage is outer-OpenMP over sample tiles, SIMD-shaped within.
-
template<typename ScannerT>
class RefScanner# - #include <Scanner.h>
Reference scanner to simplify the access to the scan products.
There is no separate
work/MaybeVolatilepair:aether::View(aether/view/View.h) is already the lightweight descriptor, so which storage it points at is just whichChunktheViewwas built over. Shared-memory (work) references areaether::make_work_viewat the point of use, not a flavour of this class, so eagle carries noWRef/VolatileRefalias.Public Types
-
using ParentT = typename aether::Array<DataT, HANDLESIZE>::ViewT#
The packed (component, sample) view this reference wraps.
Public Functions
-
inline componentT inputs() const#
Return a component view used to manipulate the scan inputs.
-
inline componentT results() const#
Return a component view exposing the scan results.
-
inline componentT blockSums() const#
Return the per-block partial sums component view.
Public Static Functions
-
static inline RefScanner make(const ParentT &parent)#
Factory method to construct from the packed view.
-
using ParentT = typename aether::Array<DataT, HANDLESIZE>::ViewT#
-
template<typename SliceT>
class RefSlice : public eagle::util::RefSlice<SliceT::ParentT># - #include <Slice.h>
Reference to the slice.
There is no separate
work/MaybeVolatilepair — seeRefScanner. The ref-levelupload/download/fetchforwarders went with the ones they forwarded to (util::RefSlice); a view-to-view async copy is transport work.Public Types
-
struct ReorderPlane#
- #include <Reorder.h>
A per-sample plane to move: its address and element size.
-
struct ReorderTrigger#
- #include <Compact.h>
The locality trigger a compaction can compute on the way (the input of the occasional physical reorder,
Reorder.h).The predicate counts
live32, the number of aligned 32-sample groups that hold at least one kept sample (one warp ballot per group, one atomic per block), and the lane that writes the count writesfire = count < theta * 32 * live32 && count < span && span >= 4096: the kept samples fill less thanthetaof the warps they occupy, a reorder would shrink the span, and the span is large enough for the moves to pay (kReorderMinSpan). Every pointer is a one-word buffer in the compaction’s memory space;live32is overwritten (it needs no initial value),spanis read.
-
template<typename DataT_, typename OP, bool Inclusive_>
class Scanner : public aether::Array<DataT_, HANDLESIZE># - #include <Scanner.h>
Parallel prefix-scan (prefix-sum) owning container.
Computes an inclusive or exclusive prefix scan over the
inputs()component and stores the result inresults(). Supports both CPU (hostRun()) and GPU (deviceRun()) execution. A CUDA graph node is available viagraph()for integration into larger CUDA graphs.- Template Parameters:
DataT_ – Element type of the scan (e.g.
int).OP – Scan operator; typically
aether::SumOp<DataT_>.Inclusive_ –
truefor an inclusive scan;falsefor exclusive.
Public Types
-
using GRef = RefScanner<Self>#
Expose the reference type (one tier; see
RefScanner).
-
using StreamT = nativeStream_t#
Expose the stream type.
Public Functions
-
inline explicit Scanner(const idx_t &n)#
Construct a scanner over
nsamples per component.Zero-fills:
aether::Array(n)deliberately leaves storage uninitialised, so this constructor zero-fills explicitly througheagle::makeArray.
-
inline Scanner clone() const#
Deep copy — see
eagle::cloneArray.
-
inline idx_t size() const#
Samples per component — NOT
aether::Array::size(), which is the TOTAL element count (HANDLESIZE * samples()).
-
inline void hostRun()#
Run the scan on host.
-
inline cudaGraph_t graph()#
Create a cuda graph that can be used to launch the execution of the scan algorithm.
-
inline void deviceRun(const bool &includeRetrieval = false)#
Run a device scan.
-
inline const StreamT &stream() const#
Return a const reference to the CUDA stream associated to this scanner.
-
inline void retrieveResults()#
Retrieve the results (transfer only the part relative to the outputs).
This copy touches an aether array on BOTH ends, so it routes through
aether::copyAsync(View, View, Stream)(aether/view/Copy.h) rather than a rawcudaMemcpyAsync— aether owns residency and the legal device-pair matrix (CUDA -> CUDAHost is async-capable), eagle owns only WHEN the move happens.component<OUT>is what makes the site expressible at all: it offsets by the SOURCE mapping’s own component pitch —capacity(), which aether quantises, neversamples()— and extends over exactlysamples()elements, so the hand-computeddata() + OUT * pitchandsz * sizeof(DataT)of the raw form are both read off the view instead of re-derived. A host/device pitch disagreement now throws (extents/strides mismatch) instead of transferring the wrong bytes.
-
inline void clearDevice()#
Mark the device copy stale.
Unlike an implementation that releases the device allocation here,
aether::Arrayowns both chunks for its whole lifetime and exposes no way to drop one, so this now only invalidates the ready flag — the nextensureInitDevice_()re-uploads exactly as before, but the device memory is not returned in between.
-
template<typename DataT, typename OP, bool Inclusive>
class ScanNode# - #include <ScanNode.h>
A parallel prefix scan as a native node — the arena-backed twin of
cuda::Scan::graph, with both a CUDA and a host face.Scans
inputintooutput(inclusive or exclusive prefix), sourcing only its BLOCKSUMS working buffer from the graph’sScratchArena.inputandoutputare caller-owned views (GRefs) — the scan’s I/O, not scratch — so a chain of scans shares a single BLOCKSUMS slot (peak, not sum). The CUDA face (buildInto) captures thecuda::Scantree; the host face (runHost) runs the OpenMP twincpu::Scan::scanover the same BLOCKSUMS slot (host pointer). Both sharereserveScratch.Contributed transparently with
graph.addNative(ScanNode{...}, deps)(CUDA) orhostGraph.addNative(ScanNode{...}, deps)(host). Holds only a pair of GRefs + a size + stream + the scratch handle, so it is copyable — asaddNativerequires.- Template Parameters:
DataT – element type.
OP – scan operator (as
cuda::Scan/cpu::Scan).Inclusive –
trueinclusive,falseexclusive.
Public Types
-
using StreamT = nativeStream_t#
Mode-agnostic stream type (
cudaStream_tunder CUDA,intinEAGLE_CPU_ONLY).
Public Functions
-
inline ScanNode(const CRefT &input, const GRefT &output, const StreamT &stream = 0)#
- Parameters:
input – Read-only view of the array to scan (e.g.
arr.deviceView().as_const()).output – View the prefix result is written to (size == input).
stream – Stream the scan tree is captured on (CUDA face).
-
inline void reserveScratch(ScratchArena &arena)#
Phase 1: reserve the BLOCKSUMS working buffer (N elements, as the Scanner’s BLOCKSUMS component).
-
inline idx_t buildInto(cuda::Graph &g, const std::vector<idx_t> &deps)#
Phase 2 (CUDA): capture the scan tree over
input/outputwith the committed arena’s BLOCKSUMS, wiringdeps. The scan tree is layout-locked (re-deriving the grid would overrun the recursion-level blockSums), so the child is added with a uniformkFixedSizecap. Returns the added node.
-
inline void runHost(cpu::Graph &g)#
Host face: run the OpenMP prefix-scan twin over
input_intooutput_, using the committed arena’s BLOCKSUMS slot (host pointer) as the scan’s per-block workspace — the exactScanner::hostRunpath. Outer OpenMP over sample tiles, SIMD within.
-
namespace detail#
Functions
-
inline void checkCompact(std::int64_t n)#
Validate a compaction request; throws
std::invalid_argument.
-
inline void checkTrigger(const ReorderTrigger &t)#
Validate a trigger; throws
std::invalid_argument.
-
inline std::uint32_t triggerFires(std::uint32_t count, std::uint32_t live32, std::uint32_t span, float theta)#
The trigger rule, shared by both faces so they agree to the bit.
-
template<bool Drop>
void compactPredicate(const std::uint8_t *mask, GRefArrT<idx_t> keep)# keep[i] = Drop ? !mask[i] : mask[i] != 0(asidx_t0/1).
-
template<typename MapT>
void compactScatter(CRefArrT<idx_t> keep, CRefArrT<idx_t> inclusive, MapT *indexMap, std::uint32_t *count)# Scatter every kept sample to
indexMap[inclusive[i] - 1]and write the count (inclusive[n - 1]) from the last lane. (A template, like every kernel defined in an eagle header, so each translation unit’s copy is the same inline definition.).
-
template<bool Drop>
void compactPredicateLive32(const std::uint8_t *mask, GRefArrT<idx_t> keep, std::uint32_t *live32)# compactPredicatethat also counts the 32-sample groups holding a kept sample into*live32(one ballot per warp, one atomic per block; the launch block is a multiple of 32, so a warp IS an aligned group).
-
template<typename MapT>
void compactScatterTrigger(CRefArrT<idx_t> keep, CRefArrT<idx_t> inclusive, MapT *indexMap, std::uint32_t *count, const std::uint32_t *live32, const std::uint32_t *span, std::uint32_t *fire, float theta)# compactScatterwhose last lane also writes the trigger.
-
template<typename WordT>
void compactZero(WordT *word)# *word = 0(the trigger’s group count before the predicate).
-
template<typename CountT>
void compactEmpty(CountT *count)# *count = 0— the whole compaction of an empty batch.
-
inline std::size_t reorderStageOffset(std::int64_t n)#
Byte offset of the staging plane inside the reorder scratch.
-
inline bool supportedElem(std::uint32_t b)#
-
inline void checkReorder(const ReorderPlane *planes, std::uint32_t nPlanes, const std::uint8_t *mask, std::int64_t n, bool needMask)#
Validate a reorder request; throws
std::invalid_argument.
-
inline std::int64_t reorderDst(bool kept, std::int64_t i, std::int64_t incl, std::int64_t cnt)#
Where slot
igoes: kept ->incl - 1, dropped -> after the kept ones, in order.cntis the kept count over[0, span).
-
inline std::int64_t reorderThread()#
-
template<typename T>
void reorderStage(const T *plane, T *stage, const std::uint32_t *span, const std::uint32_t *fire, std::int64_t n)# stage[i] = plane[i]over[0, *span)([0, n)without a span).
-
template<typename T>
void reorderPlace(const T *stage, T *plane, const idx_t *keep, const idx_t *incl, const std::uint32_t *span, const std::uint32_t *fire)# Scatter the staged
[0, *span)to their reordered slots.
-
template<typename IndexT>
void reorderPermInv(const IndexT *perm, IndexT *inv, const std::uint32_t *span, const std::uint32_t *fire)# inv[perm[s]] = sover[0, *span)(the moved slots).
-
template<typename IndexT>
void reorderIdentity(IndexT *indexMap, const idx_t *incl, const std::uint32_t *span, const std::uint32_t *fire)# map[t] = tover[0, kept), the kept count over[0, *span).
-
template<typename WordT>
void reorderFin(const idx_t *incl, WordT *count, WordT *span, WordT *fire)# The last kernel:
count = span = kept,fire = 0.
-
template<typename WordT>
void reorderEmpty(WordT *count, WordT *span, WordT *fire)# The whole reorder of an empty batch:
count = span = 0,fire = 0.
-
template<typename T, typename IndexT>
void restoreScatter(const T *stage, T *plane, const IndexT *perm, std::int64_t n)# plane[perm[s]] = stage[s]over[0, n): back to sample order.
-
template<typename IndexT>
void restoreIdentity(IndexT *perm, IndexT *inv, std::int64_t n)# perm[i] = inv[i] = iover[0, n).
-
template<typename T>
inline void reorderPlaneDevice(void *data, void *stage, const idx_t *keep, const idx_t *incl, const std::uint32_t *span, const std::uint32_t *fire, std::int64_t n, unsigned nb, cudaStream_t stream)# Stage + place one plane of element type
T.
-
template<typename T>
inline void restorePlaneDevice(void *data, void *stage, const std::int32_t *perm, std::int64_t n, unsigned nb, cudaStream_t stream)# Stage + scatter-back one plane of element type
T.
-
template<typename F>
inline void withElem(std::uint32_t bytes, F &&f)# Call
f.template operator()<T>()with the element type ofbytes.
-
template<typename T>
inline void reorderPlaneHost(void *data, void *stage, const idx_t *keep, const idx_t *incl, std::int64_t sp)#
-
template<typename T>
inline void restorePlaneHost(void *data, void *stage, const std::int32_t *perm, std::int64_t n)#
-
template<typename T, typename OP>
void inclusiveScanBlock(CRefArrT<T> arr, GRefArrT<T> out, GRefArrT<T> blockSums, const T init)# Run a per-block inclusive scan.
SASS register footprint: ~17 regs (sm_61).
(256, 4)documents the cap consumed byLauncher::setLogicalSizere-tuning.
-
struct Bytes16#
- #include <Reorder.h>
A 16-byte element, moved as one value.
-
struct ScanCore#
-
inline void checkCompact(std::int64_t n)#
-
inline std::size_t compactScratchBytes(std::int64_t n)#