11 #include <cudf/utilities/export.hpp>
14 #include <cuda/std/span>
15 #include <cuda/stream>
26 namespace CUDF_EXPORT
cudf {
34 namespace hashing::detail {
35 template <
typename Key>
40 template <
template <
typename>
class Hasher>
129 std::int32_t precision = 12,
131 nan_policy nan_handling = nan_policy::NAN_IS_NULL,
133 cuda::mr::any_resource<cuda::mr::device_accessible> mr =
160 nan_policy nan_handling = nan_policy::NAN_IS_NULL,
162 cuda::mr::any_resource<cuda::mr::device_accessible> mr =
181 std::int32_t precision,
183 nan_policy nan_handling = nan_policy::NAN_IS_NULL);
233 void merge(cuda::std::span<cuda::std::byte const> sketch_span,
253 [[nodiscard]] cuda::std::span<cuda::std::byte>
sketch() noexcept;
264 [[nodiscard]] cuda::std::span<cuda::std::
byte const> sketch() const noexcept;
285 [[nodiscard]] std::int32_t precision() const noexcept;
295 [[nodiscard]]
double standard_error() const noexcept;
303 [[nodiscard]] static std::
size_t sketch_bytes(std::int32_t precision);
310 [[nodiscard]] static std::
size_t sketch_alignment();
Object-oriented HyperLogLog sketch for approximate distinct counting.
approx_distinct_count(table_view const &input, std::int32_t precision=12, null_policy null_handling=null_policy::EXCLUDE, nan_policy nan_handling=nan_policy::NAN_IS_NULL, cuda::stream_ref stream=cudf::get_default_stream(), cuda::mr::any_resource< cuda::mr::device_accessible > mr=cudf::get_current_device_resource_ref())
Constructs an approximate distinct count sketch from a table with specified precision.
void add(table_view const &input, cuda::stream_ref stream=cudf::get_default_stream())
Adds rows from a table to the sketch.
void merge(approx_distinct_count const &other, cuda::stream_ref stream=cudf::get_default_stream())
Merges another sketch into this sketch.
std::size_t estimate(cuda::stream_ref stream=cudf::get_default_stream()) const
Estimates the approximate number of distinct rows in the sketch.
approx_distinct_count(approx_distinct_count &&)=default
Default move constructor.
approx_distinct_count & operator=(approx_distinct_count &&)=default
Move assignment operator.
approx_distinct_count(cuda::std::span< cuda::std::byte > sketch_span, std::int32_t precision, null_policy null_handling=null_policy::EXCLUDE, nan_policy nan_handling=nan_policy::NAN_IS_NULL)
Constructs a non-owning sketch that operates on user-allocated storage.
void merge(cuda::std::span< cuda::std::byte const > sketch_span, cuda::stream_ref stream=cudf::get_default_stream())
Merges a sketch from raw bytes into this sketch.
approx_distinct_count(table_view const &input, desired_standard_error error, null_policy null_handling=null_policy::EXCLUDE, nan_policy nan_handling=nan_policy::NAN_IS_NULL, cuda::stream_ref stream=cudf::get_default_stream(), cuda::mr::any_resource< cuda::mr::device_accessible > mr=cudf::get_current_device_resource_ref())
Constructs an approximate distinct count sketch from a table with specified standard error.
cuda::std::span< cuda::std::byte > sketch() noexcept
Gets the raw sketch bytes for serialization or external merging.
A set of cudf::column_view's of the same size.
APIs for querying the default CUDA stream and per-thread default stream status.
cuda::stream_ref const get_default_stream()
Get the current default stream.
rmm::device_async_resource_ref get_current_device_resource_ref()
Get the current device memory resource reference.
null_policy
Enum to specify whether to include nulls or exclude nulls.
nan_policy
Enum to treat NaN floating point value as null or non-null element.
APIs for getting and setting the current device memory resource.
Strong type wrapper for the desired standard error constructor parameter.
double value
The requested standard error value (must be positive)
constexpr desired_standard_error(double v)
Constructs a desired_standard_error with the given value.
Class definitions for (mutable)_table_view
Type declarations for libcudf.