contiguous_split.hpp
Go to the documentation of this file.
1 /*
2  * SPDX-FileCopyrightText: Copyright (c) 2023-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
3  * SPDX-License-Identifier: Apache-2.0
4  */
5 
6 #pragma once
7 
8 #include <cudf/packed_types.hpp>
9 #include <cudf/types.hpp>
10 #include <cudf/utilities/export.hpp>
12 
13 #include <cstdint>
14 #include <memory>
15 #include <span>
16 #include <vector>
17 
23 namespace CUDF_EXPORT cudf {
24 
70 std::vector<packed_table> contiguous_split(
71  cudf::table_view const& input,
72  std::vector<size_type> const& splits,
75 
76 namespace detail {
77 
83 struct contiguous_split_state;
84 } // namespace detail
85 
144  public:
155  explicit chunked_pack(
156  cudf::table_view const& input,
157  std::size_t user_buffer_size,
160 
166 
172  [[nodiscard]] std::size_t get_total_contiguous_size() const;
173 
179  [[nodiscard]] bool has_next() const;
180 
194  [[nodiscard]] std::size_t next(cudf::device_span<uint8_t> const& user_buffer);
195 
201  [[nodiscard]] std::unique_ptr<std::vector<uint8_t>> build_metadata() const;
202 
222  [[nodiscard]] static std::unique_ptr<chunked_pack> create(
223  cudf::table_view const& input,
224  std::size_t user_buffer_size,
227 
228  private:
229  // internal state of contiguous split
230  std::unique_ptr<detail::contiguous_split_state> state;
231 };
232 
249 
262 std::size_t packed_size(
263  cudf::table_view const& input,
266 
280 std::vector<uint8_t> pack_metadata(table_view const& table,
281  uint8_t const* contiguous_buffer,
282  size_t buffer_size);
283 
299 
317 table_view unpack(uint8_t const* metadata, uint8_t const* gpu_data);
318 
342  public:
354  class column_view {
355  public:
359  [[nodiscard]] data_type type() const;
360 
364  [[nodiscard]] size_type num_rows() const;
365 
369  [[nodiscard]] size_type null_count() const;
370 
374  [[nodiscard]] size_type num_children() const;
375 
383  [[nodiscard]] column_view child(size_type i) const;
384 
385  private:
386  friend class packed_metadata_view;
387  data_type _type{type_id::EMPTY};
388  size_type _size{};
389  size_type _null_count{};
390  size_type _num_children{};
391  // Span from this entry to the end of the metadata buffer (needed for child traversal).
392  std::span<std::uint8_t const> _buffer;
393  explicit column_view(std::span<std::uint8_t const> buffer);
394  };
395 
403  explicit packed_metadata_view(std::span<std::uint8_t const> buffer);
404 
408  [[nodiscard]] size_type num_columns() const;
409 
415  [[nodiscard]] size_type num_rows() const;
416 
424  [[nodiscard]] column_view column(size_type i) const;
425 
426  private:
427  // Span from the first top-level column entry to the end of the metadata buffer.
428  std::span<std::uint8_t const> _entries;
429  size_type _num_columns{};
430  // Table row count, read directly from the serialized table header.
431  size_type _num_rows{};
432 };
433 
435 } // namespace CUDF_EXPORT cudf
Perform a chunked "pack" operation of the input table_view using a user provided buffer of size user_...
std::size_t get_total_contiguous_size() const
Obtain the total size of the contiguously packed table_view.
std::size_t next(cudf::device_span< uint8_t > const &user_buffer)
Packs the next chunk into user_buffer. This should be called as long as has_next returns true....
~chunked_pack()
Destructor that will be implemented as default. Declared with definition here because contiguous_spli...
chunked_pack(cudf::table_view const &input, std::size_t user_buffer_size, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref temp_mr=cudf::get_current_device_resource_ref())
Construct a chunked_pack class.
std::unique_ptr< std::vector< uint8_t > > build_metadata() const
Build the opaque metadata for all added columns.
static std::unique_ptr< chunked_pack > create(cudf::table_view const &input, std::size_t user_buffer_size, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref temp_mr=cudf::get_current_device_resource_ref())
Creates a chunked_pack instance to perform a "pack" of the table_view "input", where a buffer of user...
bool has_next() const
Function to check if there are chunks left to be copied.
Indicator for the logical data type of an element in a column.
Definition: types.hpp:278
A non-owning view of a single column's metadata within packed column data.
column_view child(size_type i) const
A view of the i-th child column's metadata.
A non-owning view over the host metadata produced by cudf::pack.
size_type num_rows() const
The number of rows in the table.
column_view column(size_type i) const
A view of the i-th top-level column's metadata.
size_type num_columns() const
packed_metadata_view(std::span< std::uint8_t const > buffer)
Construct a view from a metadata byte buffer.
A set of cudf::column_view's of the same size.
Definition: table_view.hpp:206
A set of cudf::column's of the same size.
Definition: table.hpp:31
std::vector< packed_table > contiguous_split(cudf::table_view const &input, std::vector< size_type > const &splits, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref())
Performs a deep-copy split of a table_view into a vector of packed_table where each packed_table is u...
packed_columns pack(cudf::table_view const &input, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref())
Deep-copy a table_view into a serialized contiguous memory format.
std::size_t packed_size(cudf::table_view const &input, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref temp_mr=cudf::get_current_device_resource_ref())
Compute the size in bytes of the contiguous memory buffer needed to pack the input table.
table_view unpack(uint8_t const *metadata, uint8_t const *gpu_data)
Deserialize the result of cudf::pack.
std::vector< uint8_t > pack_metadata(table_view const &table, uint8_t const *contiguous_buffer, size_t buffer_size)
Produce the metadata used for packing a table stored in a contiguous buffer.
rmm::cuda_stream_view const get_default_stream()
Get the current default stream.
rmm::device_async_resource_ref get_current_device_resource_ref()
Get the current device memory resource reference.
cuda::mr::resource_ref< cuda::mr::device_accessible > device_async_resource_ref
cuda::std::span< T, Extent > device_span
Device span is an alias of cuda::std::span.
Definition: span.hpp:296
int32_t size_type
Row index type for columns and tables.
Definition: types.hpp:76
APIs for getting and setting the current device memory resource.
cuDF interfaces
Definition: host_udf.hpp:26
Packed table and column types for serialization.
Column data in a serialized format.
Type declarations for libcudf.