approx_distinct_count.hpp
Go to the documentation of this file.
1 /*
2  * SPDX-FileCopyrightText: Copyright (c) 2025-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
3  * SPDX-License-Identifier: Apache-2.0
4  */
5 
6 #pragma once
7 
9 #include <cudf/types.hpp>
11 #include <cudf/utilities/export.hpp>
13 
14 #include <rmm/cuda_stream_view.hpp>
15 
16 #include <cuda/std/span>
17 
18 #include <cstddef>
19 #include <cstdint>
20 #include <memory>
21 
27 namespace CUDF_EXPORT cudf {
28 
34 // Forward declarations
35 namespace hashing::detail {
36 template <typename Key>
37 struct XXHash_64;
38 }
39 
40 namespace detail {
41 template <template <typename> class Hasher>
43 }
44 
92  public:
93  using impl_type =
95 
109  double value;
110 
115  explicit constexpr desired_standard_error(double v) : value{v} {}
116  };
117 
130  std::int32_t precision = 12,
131  null_policy null_handling = null_policy::EXCLUDE,
132  nan_policy nan_handling = nan_policy::NAN_IS_NULL,
134  cuda::mr::any_resource<cuda::mr::device_accessible> mr =
136 
160  null_policy null_handling = null_policy::EXCLUDE,
161  nan_policy nan_handling = nan_policy::NAN_IS_NULL,
163  cuda::mr::any_resource<cuda::mr::device_accessible> mr =
165 
181  approx_distinct_count(cuda::std::span<cuda::std::byte> sketch_span,
182  std::int32_t precision,
183  null_policy null_handling = null_policy::EXCLUDE,
184  nan_policy nan_handling = nan_policy::NAN_IS_NULL);
185 
187 
189  approx_distinct_count& operator=(approx_distinct_count const&) = delete;
197 
205 
218  void merge(approx_distinct_count const& other,
220 
234  void merge(cuda::std::span<cuda::std::byte const> sketch_span,
236 
243  [[nodiscard]] std::size_t estimate(
245 
255  [[nodiscard]] cuda::std::span<cuda::std::byte> sketch() noexcept;
256 
266  [[nodiscard]] cuda::std::span<cuda::std::byte const> sketch() const noexcept;
267 
273  [[nodiscard]] null_policy null_handling() const noexcept;
274 
280  [[nodiscard]] nan_policy nan_handling() const noexcept;
281 
287  [[nodiscard]] std::int32_t precision() const noexcept;
288 
297  [[nodiscard]] double standard_error() const noexcept;
298 
305  [[nodiscard]] static std::size_t sketch_bytes(std::int32_t precision);
306 
312  [[nodiscard]] static std::size_t sketch_alignment();
313 
314  private:
315  std::unique_ptr<impl_type> _impl;
316 };
317 
320 } // namespace CUDF_EXPORT cudf
Object-oriented HyperLogLog sketch for approximate distinct counting.
approx_distinct_count(table_view const &input, desired_standard_error error, null_policy null_handling=null_policy::EXCLUDE, nan_policy nan_handling=nan_policy::NAN_IS_NULL, rmm::cuda_stream_view stream=cudf::get_default_stream(), cuda::mr::any_resource< cuda::mr::device_accessible > mr=cudf::get_current_device_resource_ref())
Constructs an approximate distinct count sketch from a table with specified standard error.
std::size_t estimate(rmm::cuda_stream_view stream=cudf::get_default_stream()) const
Estimates the approximate number of distinct rows in the sketch.
void merge(approx_distinct_count const &other, rmm::cuda_stream_view stream=cudf::get_default_stream())
Merges another sketch into this sketch.
approx_distinct_count(approx_distinct_count &&)=default
Default move constructor.
approx_distinct_count & operator=(approx_distinct_count &&)=default
Move assignment operator.
approx_distinct_count(cuda::std::span< cuda::std::byte > sketch_span, std::int32_t precision, null_policy null_handling=null_policy::EXCLUDE, nan_policy nan_handling=nan_policy::NAN_IS_NULL)
Constructs a non-owning sketch that operates on user-allocated storage.
void merge(cuda::std::span< cuda::std::byte const > sketch_span, rmm::cuda_stream_view stream=cudf::get_default_stream())
Merges a sketch from raw bytes into this sketch.
approx_distinct_count(table_view const &input, std::int32_t precision=12, null_policy null_handling=null_policy::EXCLUDE, nan_policy nan_handling=nan_policy::NAN_IS_NULL, rmm::cuda_stream_view stream=cudf::get_default_stream(), cuda::mr::any_resource< cuda::mr::device_accessible > mr=cudf::get_current_device_resource_ref())
Constructs an approximate distinct count sketch from a table with specified precision.
cuda::std::span< cuda::std::byte > sketch() noexcept
Gets the raw sketch bytes for serialization or external merging.
void add(table_view const &input, rmm::cuda_stream_view stream=cudf::get_default_stream())
Adds rows from a table to the sketch.
A set of cudf::column_view's of the same size.
Definition: table_view.hpp:206
APIs for querying the default CUDA stream and per-thread default stream status.
rmm::cuda_stream_view const get_default_stream()
Get the current default stream.
rmm::device_async_resource_ref get_current_device_resource_ref()
Get the current device memory resource reference.
null_policy
Enum to specify whether to include nulls or exclude nulls.
Definition: types.hpp:107
nan_policy
Enum to treat NaN floating point value as null or non-null element.
Definition: types.hpp:115
APIs for getting and setting the current device memory resource.
cuDF interfaces
Definition: host_udf.hpp:26
Strong type wrapper for the desired standard error constructor parameter.
double value
The requested standard error value (must be positive)
constexpr desired_standard_error(double v)
Constructs a desired_standard_error with the given value.
Class definitions for (mutable)_table_view
Type declarations for libcudf.