hash_join.hpp
Go to the documentation of this file.
1 /*
2  * SPDX-FileCopyrightText: Copyright (c) 2025-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
3  * SPDX-License-Identifier: Apache-2.0
4  */
5 
6 #pragma once
7 
8 #include <cudf/hashing.hpp>
9 #include <cudf/join/join.hpp>
11 #include <cudf/types.hpp>
13 #include <cudf/utilities/export.hpp>
15 #include <cudf/utilities/span.hpp>
16 
17 #include <rmm/cuda_stream_view.hpp>
18 #include <rmm/device_uvector.hpp>
19 
20 #include <optional>
21 #include <utility>
22 
29 namespace CUDF_EXPORT cudf {
30 
36 // forward declaration
37 namespace hashing::detail {
41 template <typename T>
43 } // namespace hashing::detail
44 
45 namespace detail {
49 template <typename T>
50 class hash_join;
51 } // namespace detail
52 
61 enum class nullable_join : bool { YES, NO };
62 
70 class hash_join {
71  public:
74 
75  hash_join() = delete;
76  ~hash_join();
77  hash_join(hash_join const&) = delete;
78  hash_join(hash_join&&) = delete;
79  hash_join& operator=(hash_join const&) = delete;
80  hash_join& operator=(hash_join&&) = delete;
81 
96  null_equality compare_nulls,
98  cuda::mr::any_resource<cuda::mr::device_accessible> mr =
100 
121  null_equality compare_nulls,
122  double load_factor,
124  cuda::mr::any_resource<cuda::mr::device_accessible> mr =
126 
145  [[nodiscard]] std::pair<std::unique_ptr<rmm::device_uvector<size_type>>,
146  std::unique_ptr<rmm::device_uvector<size_type>>>
148  std::optional<std::size_t> output_size = {},
151 
170  [[nodiscard]] std::pair<std::unique_ptr<rmm::device_uvector<size_type>>,
171  std::unique_ptr<rmm::device_uvector<size_type>>>
173  std::optional<std::size_t> output_size = {},
176 
195  [[nodiscard]] std::pair<std::unique_ptr<rmm::device_uvector<size_type>>,
196  std::unique_ptr<rmm::device_uvector<size_type>>>
198  std::optional<std::size_t> output_size = {},
201 
215  [[nodiscard]] std::size_t inner_join_size(
216  cudf::table_view const& left, rmm::cuda_stream_view stream = cudf::get_default_stream()) const;
217 
231  [[nodiscard]] std::size_t left_join_size(
232  cudf::table_view const& left, rmm::cuda_stream_view stream = cudf::get_default_stream()) const;
233 
249  [[nodiscard]] std::size_t full_join_size(
250  cudf::table_view const& left,
253 
276  cudf::table_view const& left,
279 
301  cudf::table_view const& left,
304 
326  cudf::table_view const& left,
329 
350  [[nodiscard]] std::pair<std::unique_ptr<rmm::device_uvector<size_type>>,
351  std::unique_ptr<rmm::device_uvector<size_type>>>
353  cudf::join_partition_context const& context,
356 
376  [[nodiscard]] std::pair<std::unique_ptr<rmm::device_uvector<size_type>>,
377  std::unique_ptr<rmm::device_uvector<size_type>>>
379  cudf::join_partition_context const& context,
382 
406  [[nodiscard]] std::pair<std::unique_ptr<rmm::device_uvector<size_type>>,
407  std::unique_ptr<rmm::device_uvector<size_type>>>
409  cudf::join_partition_context const& context,
412 
432  [[nodiscard]] static std::pair<std::unique_ptr<rmm::device_uvector<size_type>>,
433  std::unique_ptr<rmm::device_uvector<size_type>>>
437  size_type left_table_num_rows,
438  size_type right_table_num_rows,
441 
442  private:
443  std::unique_ptr<impl_type const> _impl;
444 };
445  // end of group
447 
448 } // namespace CUDF_EXPORT cudf
Forward declaration for our hash join.
Definition: hash_join.hpp:50
Hash join that builds a hash table with the right table on construction and probes results in subsequ...
Definition: hash_join.hpp:70
hash_join(cudf::table_view const &right, null_equality compare_nulls, rmm::cuda_stream_view stream=cudf::get_default_stream(), cuda::mr::any_resource< cuda::mr::device_accessible > mr=cudf::get_current_device_resource_ref())
Construct a hash join object for subsequent probe calls.
std::pair< std::unique_ptr< rmm::device_uvector< size_type > >, std::unique_ptr< rmm::device_uvector< size_type > > > partitioned_full_join(cudf::join_partition_context const &context, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref()) const
Performs a full join probe on a partition of the probe table.
std::size_t inner_join_size(cudf::table_view const &left, rmm::cuda_stream_view stream=cudf::get_default_stream()) const
typename cudf::detail::hash_join< cudf::hashing::detail::MurmurHash3_x86_32< cudf::hash_value_type > > impl_type
Implementation type.
Definition: hash_join.hpp:73
hash_join(cudf::table_view const &right, nullable_join has_nulls, null_equality compare_nulls, double load_factor, rmm::cuda_stream_view stream=cudf::get_default_stream(), cuda::mr::any_resource< cuda::mr::device_accessible > mr=cudf::get_current_device_resource_ref())
Construct a hash join object for subsequent probe calls.
cudf::join_match_context inner_join_match_context(cudf::table_view const &left, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref()) const
Returns context information about matches between the left and right tables.
std::pair< std::unique_ptr< rmm::device_uvector< size_type > >, std::unique_ptr< rmm::device_uvector< size_type > > > inner_join(cudf::table_view const &left, std::optional< std::size_t > output_size={}, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref()) const
std::size_t full_join_size(cudf::table_view const &left, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref()) const
static std::pair< std::unique_ptr< rmm::device_uvector< size_type > >, std::unique_ptr< rmm::device_uvector< size_type > > > finalize_partitioned_full_join(cudf::host_span< cudf::device_span< size_type const > const > left_partials, cudf::host_span< cudf::device_span< size_type const > const > right_partials, size_type left_table_num_rows, size_type right_table_num_rows, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref())
Finalizes a partitioned full join by concatenating all per-partition results and appending the unmatc...
std::size_t left_join_size(cudf::table_view const &left, rmm::cuda_stream_view stream=cudf::get_default_stream()) const
cudf::join_match_context full_join_match_context(cudf::table_view const &left, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref()) const
Returns context information about matches between the left and right tables.
std::pair< std::unique_ptr< rmm::device_uvector< size_type > >, std::unique_ptr< rmm::device_uvector< size_type > > > left_join(cudf::table_view const &left, std::optional< std::size_t > output_size={}, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref()) const
std::pair< std::unique_ptr< rmm::device_uvector< size_type > >, std::unique_ptr< rmm::device_uvector< size_type > > > full_join(cudf::table_view const &left, std::optional< std::size_t > output_size={}, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref()) const
std::pair< std::unique_ptr< rmm::device_uvector< size_type > >, std::unique_ptr< rmm::device_uvector< size_type > > > partitioned_inner_join(cudf::join_partition_context const &context, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref()) const
Performs an inner join on a partition of the probe table.
cudf::join_match_context left_join_match_context(cudf::table_view const &left, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref()) const
Returns context information about matches between the left and right tables.
std::pair< std::unique_ptr< rmm::device_uvector< size_type > >, std::unique_ptr< rmm::device_uvector< size_type > > > partitioned_left_join(cudf::join_partition_context const &context, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref()) const
Performs a left join on a partition of the probe table.
Forward declaration for our Murmur Hash 3 implementation.
Definition: hash_join.hpp:42
A set of cudf::column_view's of the same size.
Definition: table_view.hpp:206
APIs for querying the default CUDA stream and per-thread default stream status.
nullable_join
The enum class to specify if any of the input join tables (right table and any later left table) has ...
Definition: hash_join.hpp:61
rmm::cuda_stream_view const get_default_stream()
Get the current default stream.
rmm::device_async_resource_ref get_current_device_resource_ref()
Get the current device memory resource reference.
cuda::mr::resource_ref< cuda::mr::device_accessible > device_async_resource_ref
cuda::std::span< T, Extent > device_span
Device span is an alias of cuda::std::span.
Definition: span.hpp:296
null_equality
Enum to consider two nulls as equal or unequal.
Definition: types.hpp:140
int32_t size_type
Row index type for columns and tables.
Definition: types.hpp:84
APIs for computing hash values of columns and tables using various hash algorithms.
Common types and utilities shared by cuDF's join APIs.
APIs for getting and setting the current device memory resource.
cuDF interfaces
Definition: host_udf.hpp:26
bool has_nulls(table_view const &view)
Returns True if the table has nulls in any of its columns.
APIs for spans.
Host span, a non-owning view over a contiguous sequence of host-accessible elements.
Definition: span.hpp:65
Holds context information about matches between tables during a join operation.
Definition: join.hpp:77
Stores context information for partitioned join operations.
Definition: join.hpp:116
Class definitions for (mutable)_table_view
Type declarations for libcudf.