API reference — filtering

Contents

API reference — filtering#

Prefix-scan, scatter, and array-slicing for sample compaction. The two scan engines live in the device namespaces as eagle::cuda::Scan and eagle::cpu::Scan; eagle::filtering keeps the containers, the compaction helper, and the native graph nodes.

eagle::cuda::Scan#

struct Scan#

Perform the Prefix sum (scan) algorithm.

Public Static Functions

template<typename T, typename OP, bool Inclusive, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
static inline void enqueue(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0, const cudaStream_t &stream = 0)#

Enqueue the full scan on stream, without capturing it.

The launch sequence graph captures, issued straight onto stream: nothing is allocated, nothing is synchronised and nothing is read back to the host, so a caller that is ALREADY capturing stream (a graph-building consumer) records it into its own graph, and an uncaptured caller simply runs it asynchronously. blockSums must hold at least arr.samples() elements. The exclusive variant ends with a device-to-device copy; the inclusive one is kernels only.

template<typename T, typename OP, bool Inclusive, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
static inline cudaGraph_t graph(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0, const cudaStream_t &stream = 0)#

Capture the full scan algorithm (enqueue) into a graph.

template<typename T, typename OP, bool Inclusive, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
static inline void launch(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0, const cudaStream_t &stream = 0)#

Instantiate and launch.

template<typename T, typename OP, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
static inline cudaGraph_t inclusiveGraph(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0, const cudaStream_t &stream = 0)#

Run a full inclusive scan algorithm.

template<typename T, typename OP, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
static inline cudaGraph_t exclusiveGraph(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0, const cudaStream_t &stream = 0)#

Run a full exclusive scan algorithm.

template<typename T, typename OP, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
static inline void inclusive(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0, const cudaStream_t &stream = 0)#

Instantiate and launch an inclusive scan.

template<typename T, typename OP, unsigned int blockSize_ = filtering::detail::ScanCore::blockSize>
static inline void exclusive(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0, const cudaStream_t &stream = 0)#

Instantiate and launch an exclusive scan.

eagle::cpu::Scan#

struct Scan#

OpenMP parallelized prefix sum (scan).

Public Static Functions

template<typename T, typename OP, bool Inclusive, unsigned int blockSize_ = 1024>
static inline void scan(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0)#

Run the overall scan.

Context-aware: nested in a parallel region, scanBlock_ and applyOffsets_ work-share via omp for and the sequential prefix is wrapped in omp single. Outside a parallel region, each step opens its own parallel for.

template<typename T, typename OP, unsigned int blockSize_ = 1024>
static inline void inclusive(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0)#

Run the inclusive scan.

template<typename T, typename OP, unsigned int blockSize_ = 1024>
static inline void exclusive(const CRefArrT<T> arr, GRefArrT<T> results, GRefArrT<T> blockSums, const T &init = 0)#

Run the exclusive scan.

Namespace eagle::filtering#

namespace filtering#

Enums

enum Handle#

Handle elements.

Values:

enumerator IN#
enumerator OUT#
enumerator BLOCKSUMS#
enumerator HANDLESIZE#

Functions

inline std::size_t compactScratchBytes(std::int64_t n)#

Bytes of scratch the compaction of n samples needs: three idx_t planes of n (predicate, inclusive scan, scan block sums).

inline void compactDevice(const std::uint8_t *mask, std::uint32_t flags, std::int32_t *indexMap, std::uint32_t *count, void *scratch, std::int64_t n, cudaStream_t stream)#

Enqueue the compaction of n samples on stream (device memory).

Kernels only, issued in order on stream: predicate, inclusive scan, scatter. Legal while stream is being captured.

inline void compactDevice(const std::uint8_t *mask, std::uint32_t flags, std::int32_t *indexMap, std::uint32_t *count, void *scratch, std::int64_t n, cudaStream_t stream, const ReorderTrigger *trig)#

compactDevice that also computes the reorder trigger trig (device words). trig == nullptr is the call above, unchanged.

One extra kernel (zeroing live32); the predicate ballots and the scatter writes fire from the lane that writes the count. Capturable likewise.

inline void compactHost(const std::uint8_t *mask, std::uint32_t flags, std::int32_t *indexMap, std::uint32_t *count, void *scratch, std::int64_t n)#

The host face of compactDevice over host memory (OpenMP twins).

inline void compactHost(const std::uint8_t *mask, std::uint32_t flags, std::int32_t *indexMap, std::uint32_t *count, void *scratch, std::int64_t n, const ReorderTrigger *trig)#

compactHost that also computes the reorder trigger trig (host words), by the same rule as the device face. trig == nullptr is the call above, unchanged.

inline std::size_t reorderScratchBytes(std::int64_t n, std::uint32_t maxElemBytes)#

Bytes of scratch a reorder (or restore) of n samples needs: the compaction’s scan scratch, then one staging plane of the widest element (perm included, so at least 4 bytes).

inline void reorderDevice(const ReorderPlane *planes, std::uint32_t nPlanes, const std::uint8_t *mask, std::uint32_t flags, std::int32_t *perm, std::int32_t *inv, std::int32_t *indexMap, std::uint32_t *count, std::uint32_t *span, std::uint32_t *fire, void *scratch, std::int64_t n, cudaStream_t stream)#

Enqueue the reorder of n samples on stream (device memory).

Kernels only: predicate + inclusive scan of the mask, then stage + place for every plane and for perm, inv from perm, the identity map, and last the words (count = span = kept, fire = 0). Legal while stream is being captured; with a fire word it is a no-op on the device while *fire == 0 (the scan still runs, into scratch).

inline void restoreDevice(const ReorderPlane *planes, std::uint32_t nPlanes, std::int32_t *perm, std::int32_t *inv, void *scratch, std::int64_t n, cudaStream_t stream)#

Enqueue the un-permutation of every plane on stream: sample i back at slot i; perm and inv := identity.

A gather per plane over all n slots. The index map, count and span are NOT touched (they describe the permuted layout): recompute the map and set the span back to n afterwards.

inline void reorderHost(const ReorderPlane *planes, std::uint32_t nPlanes, const std::uint8_t *mask, std::uint32_t flags, std::int32_t *perm, std::int32_t *inv, std::int32_t *indexMap, std::uint32_t *count, std::uint32_t *span, std::uint32_t *fire, void *scratch, std::int64_t n)#

The host face of reorderDevice over host memory (OpenMP); the same moves, so the two faces agree to the bit.

inline void restoreHost(const ReorderPlane *planes, std::uint32_t nPlanes, std::int32_t *perm, std::int32_t *inv, void *scratch, std::int64_t n)#

The host face of restoreDevice over host memory (OpenMP).

Variables

std::uint32_t kCompactMaskIsDrop = 1u#

Mask flag: a set byte drops the sample (otherwise it keeps it).

std::int64_t kCompactMaxSamples = INT32_MAX#

The largest sample count the int32 index map can address.

std::uint32_t kReorderMinSpan = 4096u#

The smallest span a reorder is worth firing for: below it the fixed launch cost of the moves exceeds what the restored coalescing saves.

std::uint32_t kReorderMaxElemBytes = 16#

The largest element a plane may carry, in bytes.

template<typename MyClass>
class FilteringSlice : public eagle::util::Slice<MyClass>#
#include <Slice.h>

More flexible slice version that incorporates a scanner.

Public Types

using ScannerT = Scanner<idx_t, aether::SumOp<idx_t>, false>#

Expose the scanner type.

using T = MyClass#

Expose inherently sliced type.

using GRef = RefSlice<Self>#

Expose the one non-owning reference tier.

Public Functions

FilteringSlice() = delete#

Default constructor is forbidden.

inline FilteringSlice(T *myobj = nullptr)#

Construct from the given object pointer.

inline FilteringSlice(T &myobj)#

Construct from the given object.

inline FilteringSlice(ParentT &&parent, ScannerT &&scanner)#

Move construct from data types.

FilteringSlice(FilteringSlice &other) = delete#

Copy constructor is forbidden.

inline FilteringSlice(FilteringSlice &&other)#

Move constructor.

FilteringSlice &operator=(FilteringSlice &other) = delete#

Copy assignment is forbidden.

inline FilteringSlice &operator=(FilteringSlice &&other)#

Move assignment operator.

inline ScannerT &scanner()#

Expose the scanner.

inline void hostUpdate()#

Run the scan and scatter on host.

inline GRef hostRef() const#

Return a host reference.

inline GRef deviceRef() const#

Return a device Reference.

inline cuda::CapturedGraph scanGraph()#

Capture the scan algorithm into a CapturedGraph.

Tags every kernel with idealBlockSize = 0 (grid-only mutation) — the warp-shuffle reduction layout depends on the captured blockSize and would be corrupted by re-tuning.

inline cuda::CapturedGraph scatterGraph()#

Capture the scatter kernel into a CapturedGraph with cap EAGLE_BLOCKSIZE (element-wise; full re-tune OK).

inline void deviceUpdate(const bool &includeRetrieval = false)#

Run a full device scan.

inline const StreamT &stream() const#

Return a const reference to the CUDA stream associated to this slice.

inline void stream(const StreamT &stream)#

Set the cuda stream.

inline void upload(const StreamT &stream = 0, const bool &includeSliceData = true)#

Async memcpy from host to device - indexes.

inline void download(const StreamT &stream = 0, const bool &includeSliceData = true)#

Async memcpy from device to host.

inline void clearDevice()#

Mark the device copies stale (see util::Slice::clearDevice for why this no longer frees anything).

inline FilteringSlice clone() const#

Clone this slice.

template<typename MyClass>
class FilteringSliceNode#
#include <FilteringSliceNode.h>

Stream compaction (predicate -> scan -> scatter) as a native node — the arena-backed twin of FilteringSlice::deviceUpdate / hostUpdate, with both a CUDA and a host face.

Builds the index map of the samples a predicate keeps, sourcing the scanner’s IN/OUT/BLOCKSUMS (3N) working buffer from the graph’s ScratchArena. That whole 3N is INTERNAL to the node — IN is a copy of the predicate, OUT is consumed by the scatter, BLOCKSUMS is scan-internal — so a chain of interleaved compaction nodes shares one 3N slot (peak, not sum). The sliced object’s index map + active count (the compaction OUTPUT) are caller-owned payload, never arena scratch. The CUDA face (buildInto) captures predicate-copy -> exclusive scan (kFixedSize) -> CUDAscatter; the host face (runHost) runs the OpenMP twins over the same 3N slot: copy -> cpu::Scan -> OMPscatter (the exact FilteringSlice::hostUpdate path).

Template Parameters:

MyClass – the sliced object type (as FilteringSlice<MyClass>).

Public Types

using StreamT = nativeStream_t#

Mode-agnostic stream type (cudaStream_t under CUDA, int in EAGLE_CPU_ONLY).

Public Functions

inline FilteringSliceNode(const SliceGRef &slice, const idx_t *predicate, idx_t n, const StreamT &stream = 0)#
Parameters:
  • slice – View of the sliced object (slice.deviceRef() / slice.hostRef()).

  • predicate – Pointer to the N-element 0/1 keep-mask (device memory for the CUDA face, host memory for the host face).

  • n – Sample count.

  • stream – Capture stream (CUDA face).

inline void reserveScratch(ScratchArena &arena)#

Phase 1: reserve the scanner’s contiguous IN+OUT+BLOCKSUMS (3N).

inline idx_t buildInto(cuda::Graph &g, const std::vector<idx_t> &deps)#

Phase 2 (CUDA): capture predicate-copy -> scan -> scatter over the arena 3N buffer, wiring deps to the first node. Returns the scatter node.

inline void runHost(cpu::Graph &g)#

Host face: run the OpenMP compaction twins over the committed arena’s 3N slot (host pointers), reconstructing the same SoA scanner view the CUDA face builds — copy predicate -> exclusive cpu::Scan -> OMPscatter (the FilteringSlice::hostUpdate path). Each stage is outer-OpenMP over sample tiles, SIMD-shaped within.

template<typename ScannerT>
class RefScanner#
#include <Scanner.h>

Reference scanner to simplify the access to the scan products.

There is no separate work/MaybeVolatile pair: aether::View (aether/view/View.h) is already the lightweight descriptor, so which storage it points at is just which Chunk the View was built over. Shared-memory (work) references are aether::make_work_view at the point of use, not a flavour of this class, so eagle carries no WRef/VolatileRef alias.

Public Types

using ParentT = typename aether::Array<DataT, HANDLESIZE>::ViewT#

The packed (component, sample) view this reference wraps.

using componentT = GRefArrT<DataT>#

A single scan component, as a scalar view.

Public Functions

inline const ParentT &packed() const#

The packed view underlying every component accessor.

inline componentT inputs() const#

Return a component view used to manipulate the scan inputs.

inline componentT results() const#

Return a component view exposing the scan results.

inline componentT blockSums() const#

Return the per-block partial sums component view.

inline idx_t size() const#

Number of samples in each component.

inline idx_t numElements() const#

Return the number of independent elements in the scan results.

aether’s View has no back(); the last sample is v(v.samples()-1).

Public Members

ParentT data_#

Data member made public for PODification.

Public Static Functions

static inline RefScanner make(const ParentT &parent)#

Factory method to construct from the packed view.

template<typename SliceT>
class RefSlice : public eagle::util::RefSlice<SliceT::ParentT>#
#include <Slice.h>

Reference to the slice.

There is no separate work/MaybeVolatile pair — see RefScanner. The ref-level upload/download/fetch forwarders went with the ones they forwarded to (util::RefSlice); a view-to-view async copy is transport work.

Public Types

using ScannerT = typename SliceT::ScannerT::GRef#

Expose the scanner reference type.

using T = typename SliceT::T::GRef#

Expose inherently sliced type.

Public Functions

inline ScannerT &scanner()#

Expose the scanner.

inline RefSlice clone() const#

Clone this reference object.

inline ViewT operator[](const SampleIndex &i)#

Expose element access operator.

inline decltype(auto) operator[](const SampleIndex &i) const#

Expose element access operator.

Public Static Functions

static inline RefSlice make(const ParentT &parent, const ScannerT &scanner)#

Factory method to construct from parent type and scanner.

struct ReorderPlane#
#include <Reorder.h>

A per-sample plane to move: its address and element size.

Public Members

void *data#

n elements

std::uint32_t elemBytes#

1, 2, 4, 8 or 16

struct ReorderTrigger#
#include <Compact.h>

The locality trigger a compaction can compute on the way (the input of the occasional physical reorder, Reorder.h).

The predicate counts live32, the number of aligned 32-sample groups that hold at least one kept sample (one warp ballot per group, one atomic per block), and the lane that writes the count writes fire = count < theta * 32 * live32 && count < span && span >= 4096: the kept samples fill less than theta of the warps they occupy, a reorder would shrink the span, and the span is large enough for the moves to pay (kReorderMinSpan). Every pointer is a one-word buffer in the compaction’s memory space; live32 is overwritten (it needs no initial value), span is read.

Public Members

std::uint32_t *live32#

OUT: groups of 32 holding a kept sample.

const std::uint32_t *span#

samples [0, span) may still be kept

std::uint32_t *fire#

OUT: 1 when a reorder pays, else 0.

float theta#

the locality threshold, in (0, 1)

template<typename DataT_, typename OP, bool Inclusive_>
class Scanner : public aether::Array<DataT_, HANDLESIZE>#
#include <Scanner.h>

Parallel prefix-scan (prefix-sum) owning container.

Computes an inclusive or exclusive prefix scan over the inputs() component and stores the result in results(). Supports both CPU (hostRun()) and GPU (deviceRun()) execution. A CUDA graph node is available via graph() for integration into larger CUDA graphs.

Template Parameters:
  • DataT_ – Element type of the scan (e.g. int).

  • OP – Scan operator; typically aether::SumOp<DataT_>.

  • Inclusive_ – true for an inclusive scan; false for exclusive.

Public Types

using GRef = RefScanner<Self>#

Expose the reference type (one tier; see RefScanner).

using StreamT = nativeStream_t#

Expose the stream type.

Public Functions

inline explicit Scanner(const idx_t &n)#

Construct a scanner over n samples per component.

Zero-fills: aether::Array(n) deliberately leaves storage uninitialised, so this constructor zero-fills explicitly through eagle::makeArray.

inline Scanner(Scanner &&other)#

Move constructor is allowed.

inline Scanner &operator=(Scanner &&other)#

Move assignment is allowed.

inline Scanner clone() const#

Deep copy — see eagle::cloneArray.

inline idx_t size() const#

Samples per component — NOT aether::Array::size(), which is the TOTAL element count (HANDLESIZE * samples()).

inline void hostRun()#

Run the scan on host.

inline cudaGraph_t graph()#

Create a cuda graph that can be used to launch the execution of the scan algorithm.

inline void deviceRun(const bool &includeRetrieval = false)#

Run a device scan.

inline const StreamT &stream() const#

Return a const reference to the CUDA stream associated to this scanner.

inline void stream(const StreamT &stream)#

Set the cuda stream.

inline void retrieveResults()#

Retrieve the results (transfer only the part relative to the outputs).

This copy touches an aether array on BOTH ends, so it routes through aether::copyAsync(View, View, Stream) (aether/view/Copy.h) rather than a raw cudaMemcpyAsync — aether owns residency and the legal device-pair matrix (CUDA -> CUDAHost is async-capable), eagle owns only WHEN the move happens.

component<OUT> is what makes the site expressible at all: it offsets by the SOURCE mapping’s own component pitch — capacity(), which aether quantises, never samples() — and extends over exactly samples() elements, so the hand-computed data() + OUT * pitch and sz * sizeof(DataT) of the raw form are both read off the view instead of re-derived. A host/device pitch disagreement now throws (extents/strides mismatch) instead of transferring the wrong bytes.

inline GRef hostRef() const#

Return a host reference.

inline GRef deviceRef() const#

Return a device reference.

inline void upload(const StreamT &stream)#

Upload the data marking device ready.

inline void clearDevice()#

Mark the device copy stale.

Unlike an implementation that releases the device allocation here, aether::Array owns both chunks for its whole lifetime and exposes no way to drop one, so this now only invalidates the ready flag — the next ensureInitDevice_() re-uploads exactly as before, but the device memory is not returned in between.

template<typename DataT, typename OP, bool Inclusive>
class ScanNode#
#include <ScanNode.h>

A parallel prefix scan as a native node — the arena-backed twin of cuda::Scan::graph, with both a CUDA and a host face.

Scans input into output (inclusive or exclusive prefix), sourcing only its BLOCKSUMS working buffer from the graph’s ScratchArena. input and output are caller-owned views (GRefs) — the scan’s I/O, not scratch — so a chain of scans shares a single BLOCKSUMS slot (peak, not sum). The CUDA face (buildInto) captures the cuda::Scan tree; the host face (runHost) runs the OpenMP twin cpu::Scan::scan over the same BLOCKSUMS slot (host pointer). Both share reserveScratch.

Contributed transparently with graph.addNative(ScanNode{...}, deps) (CUDA) or hostGraph.addNative(ScanNode{...}, deps) (host). Holds only a pair of GRefs + a size + stream + the scratch handle, so it is copyable — as addNative requires.

Template Parameters:
  • DataT – element type.

  • OP – scan operator (as cuda::Scan / cpu::Scan).

  • Inclusive – true inclusive, false exclusive.

Public Types

using StreamT = nativeStream_t#

Mode-agnostic stream type (cudaStream_t under CUDA, int in EAGLE_CPU_ONLY).

Public Functions

inline ScanNode(const CRefT &input, const GRefT &output, const StreamT &stream = 0)#
Parameters:
  • input – Read-only view of the array to scan (e.g. arr.deviceView().as_const()).

  • output – View the prefix result is written to (size == input).

  • stream – Stream the scan tree is captured on (CUDA face).

inline void reserveScratch(ScratchArena &arena)#

Phase 1: reserve the BLOCKSUMS working buffer (N elements, as the Scanner’s BLOCKSUMS component).

inline idx_t buildInto(cuda::Graph &g, const std::vector<idx_t> &deps)#

Phase 2 (CUDA): capture the scan tree over input/output with the committed arena’s BLOCKSUMS, wiring deps. The scan tree is layout-locked (re-deriving the grid would overrun the recursion-level blockSums), so the child is added with a uniform kFixedSize cap. Returns the added node.

inline void runHost(cpu::Graph &g)#

Host face: run the OpenMP prefix-scan twin over input_ into output_, using the committed arena’s BLOCKSUMS slot (host pointer) as the scan’s per-block workspace — the exact Scanner::hostRun path. Outer OpenMP over sample tiles, SIMD within.

namespace detail#

Functions

inline void checkCompact(std::int64_t n)#

Validate a compaction request; throws std::invalid_argument.

inline void checkTrigger(const ReorderTrigger &t)#

Validate a trigger; throws std::invalid_argument.

inline std::uint32_t triggerFires(std::uint32_t count, std::uint32_t live32, std::uint32_t span, float theta)#

The trigger rule, shared by both faces so they agree to the bit.

template<bool Drop>
void compactPredicate(const std::uint8_t *mask, GRefArrT<idx_t> keep)#

keep[i] = Drop ? !mask[i] : mask[i] != 0 (as idx_t 0/1).

template<typename MapT>
void compactScatter(CRefArrT<idx_t> keep, CRefArrT<idx_t> inclusive, MapT *indexMap, std::uint32_t *count)#

Scatter every kept sample to indexMap[inclusive[i] - 1] and write the count (inclusive[n - 1]) from the last lane. (A template, like every kernel defined in an eagle header, so each translation unit’s copy is the same inline definition.).

template<bool Drop>
void compactPredicateLive32(const std::uint8_t *mask, GRefArrT<idx_t> keep, std::uint32_t *live32)#

compactPredicate that also counts the 32-sample groups holding a kept sample into *live32 (one ballot per warp, one atomic per block; the launch block is a multiple of 32, so a warp IS an aligned group).

template<typename MapT>
void compactScatterTrigger(CRefArrT<idx_t> keep, CRefArrT<idx_t> inclusive, MapT *indexMap, std::uint32_t *count, const std::uint32_t *live32, const std::uint32_t *span, std::uint32_t *fire, float theta)#

compactScatter whose last lane also writes the trigger.

template<typename WordT>
void compactZero(WordT *word)#

*word = 0 (the trigger’s group count before the predicate).

template<typename CountT>
void compactEmpty(CountT *count)#

*count = 0 — the whole compaction of an empty batch.

inline std::size_t reorderStageOffset(std::int64_t n)#

Byte offset of the staging plane inside the reorder scratch.

inline bool supportedElem(std::uint32_t b)#
inline void checkReorder(const ReorderPlane *planes, std::uint32_t nPlanes, const std::uint8_t *mask, std::int64_t n, bool needMask)#

Validate a reorder request; throws std::invalid_argument.

inline std::int64_t reorderDst(bool kept, std::int64_t i, std::int64_t incl, std::int64_t cnt)#

Where slot i goes: kept -> incl - 1, dropped -> after the kept ones, in order. cnt is the kept count over [0, span).

inline std::int64_t reorderThread()#
template<typename T>
void reorderStage(const T *plane, T *stage, const std::uint32_t *span, const std::uint32_t *fire, std::int64_t n)#

stage[i] = plane[i] over [0, *span) ([0, n) without a span).

template<typename T>
void reorderPlace(const T *stage, T *plane, const idx_t *keep, const idx_t *incl, const std::uint32_t *span, const std::uint32_t *fire)#

Scatter the staged [0, *span) to their reordered slots.

template<typename IndexT>
void reorderPermInv(const IndexT *perm, IndexT *inv, const std::uint32_t *span, const std::uint32_t *fire)#

inv[perm[s]] = s over [0, *span) (the moved slots).

template<typename IndexT>
void reorderIdentity(IndexT *indexMap, const idx_t *incl, const std::uint32_t *span, const std::uint32_t *fire)#

map[t] = t over [0, kept), the kept count over [0, *span).

template<typename WordT>
void reorderFin(const idx_t *incl, WordT *count, WordT *span, WordT *fire)#

The last kernel: count = span = kept, fire = 0.

template<typename WordT>
void reorderEmpty(WordT *count, WordT *span, WordT *fire)#

The whole reorder of an empty batch: count = span = 0, fire = 0.

template<typename T, typename IndexT>
void restoreScatter(const T *stage, T *plane, const IndexT *perm, std::int64_t n)#

plane[perm[s]] = stage[s] over [0, n): back to sample order.

template<typename IndexT>
void restoreIdentity(IndexT *perm, IndexT *inv, std::int64_t n)#

perm[i] = inv[i] = i over [0, n).

template<typename T>
inline void reorderPlaneDevice(void *data, void *stage, const idx_t *keep, const idx_t *incl, const std::uint32_t *span, const std::uint32_t *fire, std::int64_t n, unsigned nb, cudaStream_t stream)#

Stage + place one plane of element type T.

template<typename T>
inline void restorePlaneDevice(void *data, void *stage, const std::int32_t *perm, std::int64_t n, unsigned nb, cudaStream_t stream)#

Stage + scatter-back one plane of element type T.

template<typename F>
inline void withElem(std::uint32_t bytes, F &&f)#

Call f.template operator()<T>() with the element type of bytes.

template<typename T>
inline void reorderPlaneHost(void *data, void *stage, const idx_t *keep, const idx_t *incl, std::int64_t sp)#
template<typename T>
inline void restorePlaneHost(void *data, void *stage, const std::int32_t *perm, std::int64_t n)#
template<typename T, typename OP>
void inclusiveScanBlock(CRefArrT<T> arr, GRefArrT<T> out, GRefArrT<T> blockSums, const T init)#

Run a per-block inclusive scan.

SASS register footprint: ~17 regs (sm_61). (256, 4) documents the cap consumed by Launcher::setLogicalSize re-tuning.

template<typename T, typename OP>
void applyOffsets(GRefArrT<T> out, CRefArrT<T> blockSums)#

Kernel to add the block offsets to the scanned indexes.

SASS register footprint: ~6 regs (sm_61).

template<typename T, typename OP>
void rightShift(GRefArrT<T> to, CRefArrT<T> from, const T init)#

Kernel to apply right-shifting to retrieve exclusive scan results.

SASS register footprint: ~6 regs (sm_61).

struct Bytes16#
#include <Reorder.h>

A 16-byte element, moved as one value.

struct ScanCore#

Public Static Functions

template<typename T, typename OP>
static inline T warpInclusive(T val, const idx_t &lane)#

Warp-level scanning.