parquet.hpp
Go to the documentation of this file.
1 /*
2  * SPDX-FileCopyrightText: Copyright (c) 2020-2025, NVIDIA CORPORATION.
3  * SPDX-License-Identifier: Apache-2.0
4  */
5 
6 #pragma once
7 
9 #include <cudf/io/detail/parquet.hpp>
10 #include <cudf/io/types.hpp>
12 #include <cudf/types.hpp>
13 #include <cudf/utilities/export.hpp>
15 
16 #include <memory>
17 #include <optional>
18 #include <string>
19 #include <utility>
20 #include <vector>
21 
22 namespace CUDF_EXPORT cudf {
23 namespace io {
30 constexpr size_t default_row_group_size_bytes =
31  std::numeric_limits<size_t>::max();
32 constexpr size_type default_row_group_size_rows = 1'000'000;
33 constexpr size_t default_max_page_size_bytes = 512 * 1024;
35 constexpr int32_t default_column_index_truncate_length = 64;
36 constexpr size_t default_max_dictionary_size = 1024 * 1024;
38 
48 [[nodiscard]] bool is_supported_read_parquet(compression_type compression);
49 
59 [[nodiscard]] bool is_supported_write_parquet(compression_type compression);
60 
62 
67  source_info _source;
68 
69  // Path in schema of column to read; `nullopt` is all
70  std::optional<std::vector<std::string>> _columns;
71 
72  // List of individual row groups to read (ignored if empty)
73  std::vector<std::vector<size_type>> _row_groups;
74  // Number of rows to skip from the start; Parquet stores the number of rows as int64_t
75  int64_t _skip_rows = 0;
76  // Number of rows to read; `nullopt` is all
77  std::optional<int64_t> _num_rows;
78 
79  // Read row groups that start at or after this byte offset into the source
80  size_t _skip_bytes = 0;
81  // Read row groups that start before _num_bytes bytes after _skip_bytes into the source
82  std::optional<size_t> _num_bytes;
83 
84  // Predicate filter as AST to filter output rows.
85  std::optional<std::reference_wrapper<ast::expression const>> _filter;
86 
87  // Whether to store string data as categorical type
88  bool _convert_strings_to_categories = false;
89  // Whether to use PANDAS metadata to load columns
90  bool _use_pandas_metadata = true;
91  // Whether to read and use ARROW schema
92  bool _use_arrow_schema = true;
93  // Whether to allow reading matching select columns from mismatched Parquet files.
94  bool _allow_mismatched_pq_schemas = false;
95  // Whether to ignore non-existent projected columns
96  bool _ignore_missing_columns = true;
97  // Cast timestamp columns to a specific type
98  data_type _timestamp_type{type_id::EMPTY};
99  // Whether to use JIT compilation for filtering
100  bool _use_jit_filter = false;
101 
102  std::optional<std::vector<reader_column_schema>> _reader_column_schema;
103 
109  explicit parquet_reader_options(source_info src) : _source{std::move(src)} {}
110 
112 
113  public:
120  explicit parquet_reader_options() = default;
121 
130 
136  [[nodiscard]] source_info const& get_source() const { return _source; }
137 
143  [[nodiscard]] bool is_enabled_convert_strings_to_categories() const
144  {
145  return _convert_strings_to_categories;
146  }
147 
153  [[nodiscard]] bool is_enabled_use_pandas_metadata() const { return _use_pandas_metadata; }
154 
160  [[nodiscard]] bool is_enabled_use_arrow_schema() const { return _use_arrow_schema; }
161 
169  [[nodiscard]] bool is_enabled_allow_mismatched_pq_schemas() const
170  {
171  return _allow_mismatched_pq_schemas;
172  }
173 
181  [[nodiscard]] bool is_enabled_ignore_missing_columns() const { return _ignore_missing_columns; }
182 
188  [[nodiscard]] std::optional<std::vector<reader_column_schema>> get_column_schema() const
189  {
190  return _reader_column_schema;
191  }
192 
198  [[nodiscard]] int64_t get_skip_rows() const { return _skip_rows; }
199 
206  [[nodiscard]] std::optional<int64_t> const& get_num_rows() const { return _num_rows; }
207 
214  [[nodiscard]] size_t get_skip_bytes() const { return _skip_bytes; }
215 
222  [[nodiscard]] std::optional<size_t> const& get_num_bytes() const { return _num_bytes; }
223 
229  [[nodiscard]] auto const& get_columns() const { return _columns; }
230 
236  [[nodiscard]] auto const& get_row_groups() const { return _row_groups; }
237 
243  [[nodiscard]] auto const& get_filter() const { return _filter; }
244 
250  [[nodiscard]] data_type get_timestamp_type() const { return _timestamp_type; }
251 
257  [[nodiscard]] bool is_enabled_use_jit_filter() const { return _use_jit_filter; }
258 
264  void set_source(source_info src) { _source = std::move(src); }
265 
285  void set_columns(std::vector<std::string> col_names) { _columns = std::move(col_names); }
286 
304  void set_row_groups(std::vector<std::vector<size_type>> row_groups);
305 
336  void set_filter(ast::expression const& filter) { _filter = filter; }
337 
343  void enable_convert_strings_to_categories(bool val) { _convert_strings_to_categories = val; }
344 
350  void enable_use_pandas_metadata(bool val) { _use_pandas_metadata = val; }
351 
357  void enable_use_arrow_schema(bool val) { _use_arrow_schema = val; }
358 
366  void enable_allow_mismatched_pq_schemas(bool val) { _allow_mismatched_pq_schemas = val; }
367 
374  void enable_ignore_missing_columns(bool val) { _ignore_missing_columns = val; }
375 
382  void set_column_schema(std::vector<reader_column_schema> val)
383  {
384  _reader_column_schema = std::move(val);
385  }
386 
392  void set_skip_rows(int64_t val);
393 
402  void set_num_rows(int64_t val);
403 
409  void set_skip_bytes(size_t val);
410 
416  void set_num_bytes(size_t val);
417 
423  void set_timestamp_type(data_type type) { _timestamp_type = type; }
424 };
425 
430  parquet_reader_options options;
431 
432  public:
440 
446  explicit parquet_reader_options_builder(source_info src) : options{std::move(src)} {}
447 
454  parquet_reader_options_builder& columns(std::vector<std::string> col_names)
455  {
456  options._columns = std::move(col_names);
457  return *this;
458  }
459 
466  parquet_reader_options_builder& row_groups(std::vector<std::vector<size_type>> row_groups)
467  {
468  options.set_row_groups(std::move(row_groups));
469  return *this;
470  }
471 
477  {
478  options.set_filter(filter);
479  return *this;
480  }
481 
489  {
490  options._convert_strings_to_categories = val;
491  return *this;
492  }
493 
501  {
502  options._use_pandas_metadata = val;
503  return *this;
504  }
505 
513  {
514  options._use_arrow_schema = val;
515  return *this;
516  }
517 
528  {
529  options._allow_mismatched_pq_schemas = val;
530  return *this;
531  }
532 
541  {
542  options._ignore_missing_columns = val;
543  return *this;
544  }
545 
552  parquet_reader_options_builder& set_column_schema(std::vector<reader_column_schema> val)
553  {
554  options._reader_column_schema = std::move(val);
555  return *this;
556  }
557 
565  {
566  options.set_skip_rows(val);
567  return *this;
568  }
569 
580  {
581  options.set_num_rows(val);
582  return *this;
583  }
584 
592  {
593  options.set_skip_bytes(val);
594  return *this;
595  }
596 
604  {
605  options.set_num_bytes(val);
606  return *this;
607  }
608 
616  {
617  options._timestamp_type = type;
618  return *this;
619  }
620 
628  {
629  options._use_jit_filter = use_jit_filter;
630  return *this;
631  }
632 
636  operator parquet_reader_options&&() { return std::move(options); }
637 
645  parquet_reader_options&& build() { return std::move(options); }
646 };
647 
666  parquet_reader_options const& options,
669 
692  std::vector<std::unique_ptr<cudf::io::datasource>>&& sources,
693  std::vector<parquet::FileMetaData>&& parquet_metadatas,
694  parquet_reader_options const& options,
697 
708  public:
716 
731  std::size_t chunk_read_limit,
732  parquet_reader_options const& options,
735 
753  std::size_t chunk_read_limit,
754  std::vector<std::unique_ptr<cudf::io::datasource>>&& sources,
755  std::vector<parquet::FileMetaData>&& parquet_metadatas,
756  parquet_reader_options const& options,
759 
780  std::size_t chunk_read_limit,
781  std::size_t pass_read_limit,
782  parquet_reader_options const& options,
785 
809  std::size_t chunk_read_limit,
810  std::size_t pass_read_limit,
811  std::vector<std::unique_ptr<cudf::io::datasource>>&& sources,
812  std::vector<parquet::FileMetaData>&& parquet_metadatas,
813  parquet_reader_options const& options,
816 
825 
831  [[nodiscard]] bool has_next() const;
832 
844  [[nodiscard]] table_with_metadata read_chunk() const;
845 
846  private:
847  std::unique_ptr<cudf::io::parquet::detail::chunked_reader> reader;
848 };
849  // end of group
861  int column_idx{};
862  bool is_descending{false};
863  bool is_nulls_first{true};
864 };
865 
870  // Specify the sink to use for writer output
871  sink_info _sink;
872  // Specify the compression format to use
873  compression_type _compression = compression_type::SNAPPY;
874  // Specify the level of statistics in the output file
876  // Optional associated metadata
877  std::optional<table_input_metadata> _metadata;
878  // Optional footer key_value_metadata
879  std::vector<std::map<std::string, std::string>> _user_data;
880  // Parquet writer can write INT96 or TIMESTAMP_MICROS. Defaults to TIMESTAMP_MICROS.
881  // If true then overrides any per-column setting in _metadata.
882  bool _write_timestamps_as_int96 = false;
883  // Parquet writer can write timestamps as UTC
884  // Defaults to true because libcudf timestamps are implicitly UTC
885  bool _write_timestamps_as_UTC = true;
886  // Whether to write ARROW schema
887  bool _write_arrow_schema = false;
888  // Maximum size of each row group (unless smaller than a single page)
889  size_t _row_group_size_bytes = default_row_group_size_bytes;
890  // Maximum number of rows in row group (unless smaller than a single page)
891  size_type _row_group_size_rows = default_row_group_size_rows;
892  // Maximum size of each page (uncompressed)
893  size_t _max_page_size_bytes = default_max_page_size_bytes;
894  // Maximum number of rows in a page
895  size_type _max_page_size_rows = default_max_page_size_rows;
896  // Maximum size of min or max values in column index
897  int32_t _column_index_truncate_length = default_column_index_truncate_length;
898  // When to use dictionary encoding for data
899  dictionary_policy _dictionary_policy = dictionary_policy::ADAPTIVE;
900  // Maximum size of column chunk dictionary (in bytes)
901  size_t _max_dictionary_size = default_max_dictionary_size;
902  // Maximum number of rows in a page fragment
903  std::optional<size_type> _max_page_fragment_size;
904  // Optional compression statistics
905  std::shared_ptr<writer_compression_statistics> _compression_stats;
906  // write V2 page headers?
907  bool _v2_page_headers = false;
908  // Which columns in _table are used for sorting
909  std::optional<std::vector<sorting_column>> _sorting_columns;
910 
911  protected:
917  explicit parquet_writer_options_base(sink_info sink) : _sink(std::move(sink)) {}
918 
919  public:
926 
932  [[nodiscard]] sink_info const& get_sink() const { return _sink; }
933 
939  [[nodiscard]] compression_type get_compression() const { return _compression; }
940 
946  [[nodiscard]] statistics_freq get_stats_level() const { return _stats_level; }
947 
953  [[nodiscard]] auto const& get_metadata() const { return _metadata; }
954 
960  [[nodiscard]] std::vector<std::map<std::string, std::string>> const& get_key_value_metadata()
961  const
962  {
963  return _user_data;
964  }
965 
971  [[nodiscard]] bool is_enabled_int96_timestamps() const { return _write_timestamps_as_int96; }
972 
978  [[nodiscard]] auto is_enabled_utc_timestamps() const { return _write_timestamps_as_UTC; }
979 
985  [[nodiscard]] auto is_enabled_write_arrow_schema() const { return _write_arrow_schema; }
986 
992  [[nodiscard]] auto get_row_group_size_bytes() const { return _row_group_size_bytes; }
993 
999  [[nodiscard]] auto get_row_group_size_rows() const { return _row_group_size_rows; }
1000 
1008  [[nodiscard]] auto get_max_page_size_bytes() const
1009  {
1010  return std::min(_max_page_size_bytes, get_row_group_size_bytes());
1011  }
1012 
1020  [[nodiscard]] auto get_max_page_size_rows() const
1021  {
1022  return std::min(_max_page_size_rows, get_row_group_size_rows());
1023  }
1024 
1030  [[nodiscard]] auto get_column_index_truncate_length() const
1031  {
1032  return _column_index_truncate_length;
1033  }
1034 
1040  [[nodiscard]] dictionary_policy get_dictionary_policy() const { return _dictionary_policy; }
1041 
1047  [[nodiscard]] auto get_max_dictionary_size() const { return _max_dictionary_size; }
1048 
1054  [[nodiscard]] auto get_max_page_fragment_size() const { return _max_page_fragment_size; }
1055 
1061  [[nodiscard]] std::shared_ptr<writer_compression_statistics> get_compression_statistics() const
1062  {
1063  return _compression_stats;
1064  }
1065 
1071  [[nodiscard]] auto is_enabled_write_v2_headers() const { return _v2_page_headers; }
1072 
1078  [[nodiscard]] auto const& get_sorting_columns() const { return _sorting_columns; }
1079 
1086 
1092  void set_key_value_metadata(std::vector<std::map<std::string, std::string>> metadata);
1093 
1106 
1113  void enable_int96_timestamps(bool req);
1114 
1120  void enable_utc_timestamps(bool val);
1121 
1128 
1134  void set_row_group_size_bytes(size_t size_bytes);
1135 
1142 
1148  void set_max_page_size_bytes(size_t size_bytes);
1149 
1156 
1162  void set_column_index_truncate_length(int32_t size_bytes);
1163 
1170 
1176  void set_max_dictionary_size(size_t size_bytes);
1177 
1184 
1190  void set_compression_statistics(std::shared_ptr<writer_compression_statistics> comp_stats);
1191 
1197  void enable_write_v2_headers(bool val);
1198 
1204  void set_sorting_columns(std::vector<sorting_column> sorting_columns);
1205 };
1206 
1210 template <class BuilderT, class OptionsT>
1212  OptionsT _options;
1213 
1214  protected:
1220  inline OptionsT& get_options() { return _options; }
1221 
1227  explicit parquet_writer_options_builder_base(OptionsT options);
1228 
1229  public:
1236 
1243  BuilderT& metadata(table_input_metadata metadata);
1244 
1251  BuilderT& key_value_metadata(std::vector<std::map<std::string, std::string>> metadata);
1252 
1260 
1267  BuilderT& compression(compression_type compression);
1268 
1275  BuilderT& row_group_size_bytes(size_t val);
1276 
1284 
1295  BuilderT& max_page_size_bytes(size_t val);
1296 
1305 
1319  BuilderT& column_index_truncate_length(int32_t val);
1320 
1339 
1351  BuilderT& max_dictionary_size(size_t val);
1352 
1364 
1372  std::shared_ptr<writer_compression_statistics> const& comp_stats);
1373 
1380  BuilderT& int96_timestamps(bool enabled);
1381 
1388  BuilderT& utc_timestamps(bool enabled);
1389 
1396  BuilderT& write_arrow_schema(bool enabled);
1397 
1404  BuilderT& write_v2_headers(bool enabled);
1405 
1412  BuilderT& sorting_columns(std::vector<sorting_column> sorting_columns);
1413 
1417  operator OptionsT&&();
1418 
1426  OptionsT&& build();
1427 };
1428 
1430 
1435  // Sets of columns to output
1436  table_view _table;
1437  // Partitions described as {start_row, num_rows} pairs
1438  std::vector<partition_info> _partitions;
1439  // Column chunks file paths to be set in the raw output metadata. One per output file
1440  std::vector<std::string> _column_chunks_file_paths;
1441 
1443 
1450  explicit parquet_writer_options(sink_info const& sink, table_view table);
1451 
1452  public:
1459 
1469 
1476 
1482  [[nodiscard]] table_view get_table() const { return _table; }
1483 
1489  [[nodiscard]] std::vector<partition_info> const& get_partitions() const { return _partitions; }
1490 
1496  [[nodiscard]] std::vector<std::string> const& get_column_chunks_file_paths() const
1497  {
1498  return _column_chunks_file_paths;
1499  }
1500 
1507  void set_partitions(std::vector<partition_info> partitions);
1508 
1515  void set_column_chunks_file_paths(std::vector<std::string> file_paths);
1516 };
1517 
1522  : public parquet_writer_options_builder_base<parquet_writer_options_builder,
1523  parquet_writer_options> {
1524  public:
1530  explicit parquet_writer_options_builder() = default;
1531 
1539 
1547  parquet_writer_options_builder& partitions(std::vector<partition_info> partitions);
1548 
1556  parquet_writer_options_builder& column_chunks_file_paths(std::vector<std::string> file_paths);
1557 };
1558 
1575 std::unique_ptr<std::vector<uint8_t>> write_parquet(
1577 
1587 std::unique_ptr<std::vector<uint8_t>> merge_row_group_metadata(
1588  std::vector<std::unique_ptr<std::vector<uint8_t>>> const& metadata_list);
1589 
1591 
1602 
1604 
1605  public:
1612 
1621 };
1622 
1627  : public parquet_writer_options_builder_base<chunked_parquet_writer_options_builder,
1628  chunked_parquet_writer_options> {
1629  public:
1636 
1643 };
1644 
1665  public:
1672 
1686 
1699  std::vector<partition_info> const& partitions = {});
1700 
1709  std::unique_ptr<std::vector<uint8_t>> close(
1710  std::vector<std::string> const& column_chunks_file_paths = {});
1711 
1713  std::unique_ptr<parquet::detail::writer> writer;
1714 };
1715  // end of group
1717 
1718 } // namespace io
1719 } // namespace CUDF_EXPORT cudf
Indicator for the logical data type of an element in a column.
Definition: types.hpp:269
The chunked parquet reader class to read Parquet file iteratively in to a series of tables,...
Definition: parquet.hpp:707
table_with_metadata read_chunk() const
Read a chunk of rows in the given Parquet file.
bool has_next() const
Check if there is any data in the given file has not yet read.
chunked_parquet_reader(std::size_t chunk_read_limit, std::vector< std::unique_ptr< cudf::io::datasource >> &&sources, std::vector< parquet::FileMetaData > &&parquet_metadatas, parquet_reader_options const &options, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref())
Constructor for chunked reader using pre-existing Parquet datasources and file metadatas.
chunked_parquet_reader(std::size_t chunk_read_limit, std::size_t pass_read_limit, parquet_reader_options const &options, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref())
Constructor for chunked reader.
chunked_parquet_reader(std::size_t chunk_read_limit, parquet_reader_options const &options, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref())
Constructor for chunked reader.
chunked_parquet_reader(std::size_t chunk_read_limit, std::size_t pass_read_limit, std::vector< std::unique_ptr< cudf::io::datasource >> &&sources, std::vector< parquet::FileMetaData > &&parquet_metadatas, parquet_reader_options const &options, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref())
Constructor for chunked reader using pre-existing Parquet datasources and file metadatas.
~chunked_parquet_reader()
Destructor, destroying the internal reader instance.
chunked_parquet_reader()
Default constructor, this should never be used.
Class to build chunked_parquet_writer_options.
Definition: parquet.hpp:1628
chunked_parquet_writer_options_builder()=default
Default constructor.
chunked_parquet_writer_options_builder(sink_info const &sink)
Constructor from sink.
Settings for chunked_parquet_writer.
Definition: parquet.hpp:1595
static chunked_parquet_writer_options_builder builder(sink_info const &sink)
creates builder to build chunked_parquet_writer_options.
chunked_parquet_writer_options()=default
Default constructor.
chunked parquet writer class to handle options and write tables in chunks.
Definition: parquet.hpp:1664
~chunked_parquet_writer()
Default destructor. This is added to not leak detail API.
std::unique_ptr< std::vector< uint8_t > > close(std::vector< std::string > const &column_chunks_file_paths={})
Finishes the chunked/streamed write process.
chunked_parquet_writer(chunked_parquet_writer_options const &options, rmm::cuda_stream_view stream=cudf::get_default_stream())
Constructor with chunked writer options.
std::unique_ptr< parquet::detail::writer > writer
Unique pointer to impl writer class.
Definition: parquet.hpp:1713
chunked_parquet_writer & write(table_view const &table, std::vector< partition_info > const &partitions={})
Writes table to output.
chunked_parquet_writer()
Default constructor, this should never be used. This is added just to satisfy cython....
Builds parquet_reader_options to use for read_parquet().
Definition: parquet.hpp:429
parquet_reader_options_builder & num_bytes(size_t val)
Sets number of bytes after skipping to end reading row groups at.
Definition: parquet.hpp:603
parquet_reader_options_builder & use_arrow_schema(bool val)
Sets to enable/disable use of arrow schema to read.
Definition: parquet.hpp:512
parquet_reader_options_builder(source_info src)
Constructor from source info.
Definition: parquet.hpp:446
parquet_reader_options_builder & skip_rows(int64_t val)
Sets number of rows to skip.
Definition: parquet.hpp:564
parquet_reader_options_builder & allow_mismatched_pq_schemas(bool val)
Sets to enable/disable reading of matching projected and filter columns from mismatched Parquet sourc...
Definition: parquet.hpp:527
parquet_reader_options_builder & columns(std::vector< std::string > col_names)
Sets names of the columns to be read.
Definition: parquet.hpp:454
parquet_reader_options_builder & ignore_missing_columns(bool val)
Sets to enable/disable ignoring of non-existent projected columns while reading.
Definition: parquet.hpp:540
parquet_reader_options_builder & skip_bytes(size_t val)
Sets bytes to skip before starting reading row groups.
Definition: parquet.hpp:591
parquet_reader_options_builder & timestamp_type(data_type type)
timestamp_type used to cast timestamp columns.
Definition: parquet.hpp:615
parquet_reader_options_builder & use_pandas_metadata(bool val)
Sets to enable/disable use of pandas metadata to read.
Definition: parquet.hpp:500
parquet_reader_options_builder()=default
Default constructor.
parquet_reader_options_builder & num_rows(int64_t val)
Sets number of rows to read.
Definition: parquet.hpp:579
parquet_reader_options_builder & row_groups(std::vector< std::vector< size_type >> row_groups)
Sets vector of individual row groups to read.
Definition: parquet.hpp:466
parquet_reader_options_builder & set_column_schema(std::vector< reader_column_schema > val)
Sets reader metadata.
Definition: parquet.hpp:552
parquet_reader_options && build()
move parquet_reader_options member once it's built.
Definition: parquet.hpp:645
parquet_reader_options_builder & filter(ast::expression const &filter)
Sets AST based filter for predicate pushdown.
Definition: parquet.hpp:476
parquet_reader_options_builder & use_jit_filter(bool use_jit_filter)
Enable/disable use of JIT for filter step.
Definition: parquet.hpp:627
parquet_reader_options_builder & convert_strings_to_categories(bool val)
Sets enable/disable conversion of strings to categories.
Definition: parquet.hpp:488
Settings for read_parquet().
Definition: parquet.hpp:66
data_type get_timestamp_type() const
Returns timestamp type used to cast timestamp columns.
Definition: parquet.hpp:250
parquet_reader_options()=default
Default constructor.
void enable_allow_mismatched_pq_schemas(bool val)
Sets to enable/disable reading of matching projected and filter columns from mismatched Parquet sourc...
Definition: parquet.hpp:366
void set_skip_rows(int64_t val)
Sets number of rows to skip.
bool is_enabled_use_jit_filter() const
Returns whether to use JIT compilation for filtering.
Definition: parquet.hpp:257
size_t get_skip_bytes() const
Returns bytes to skip before starting reading row groups.
Definition: parquet.hpp:214
void set_columns(std::vector< std::string > col_names)
Sets the names of columns to be read from all input sources.
Definition: parquet.hpp:285
bool is_enabled_ignore_missing_columns() const
Returns boolean depending on whether to ignore non-existent projected columns while reading.
Definition: parquet.hpp:181
static parquet_reader_options_builder builder(source_info src=source_info{})
Creates a parquet_reader_options_builder to build parquet_reader_options. By default,...
void enable_convert_strings_to_categories(bool val)
Sets to enable/disable conversion of strings to categories.
Definition: parquet.hpp:343
std::optional< std::vector< reader_column_schema > > get_column_schema() const
Returns optional tree of metadata.
Definition: parquet.hpp:188
void set_skip_bytes(size_t val)
Sets bytes to skip before starting reading row groups.
source_info const & get_source() const
Returns source info.
Definition: parquet.hpp:136
auto const & get_row_groups() const
Returns list of individual row groups to be read.
Definition: parquet.hpp:236
void set_row_groups(std::vector< std::vector< size_type >> row_groups)
Specifies which row groups to read from each input source.
void enable_ignore_missing_columns(bool val)
Sets to enable/disable ignoring of non-existent projected columns while reading.
Definition: parquet.hpp:374
void set_source(source_info src)
Set a new source location.
Definition: parquet.hpp:264
auto const & get_columns() const
Returns names of column to be read, if set.
Definition: parquet.hpp:229
void set_timestamp_type(data_type type)
Sets timestamp_type used to cast timestamp columns.
Definition: parquet.hpp:423
std::optional< int64_t > const & get_num_rows() const
Returns number of rows to read.
Definition: parquet.hpp:206
bool is_enabled_convert_strings_to_categories() const
Returns boolean depending on whether strings should be converted to categories.
Definition: parquet.hpp:143
void set_num_rows(int64_t val)
Sets number of rows to read.
void set_num_bytes(size_t val)
Sets number of bytes after skipping to end reading row groups at.
void enable_use_pandas_metadata(bool val)
Sets to enable/disable use of pandas metadata to read.
Definition: parquet.hpp:350
void enable_use_arrow_schema(bool val)
Sets to enable/disable use of arrow schema to read.
Definition: parquet.hpp:357
bool is_enabled_use_pandas_metadata() const
Returns boolean depending on whether to use pandas metadata while reading.
Definition: parquet.hpp:153
bool is_enabled_allow_mismatched_pq_schemas() const
Returns boolean depending on whether to read matching projected and filter columns from mismatched Pa...
Definition: parquet.hpp:169
void set_column_schema(std::vector< reader_column_schema > val)
Sets reader column schema.
Definition: parquet.hpp:382
bool is_enabled_use_arrow_schema() const
Returns boolean depending on whether to use arrow schema while reading.
Definition: parquet.hpp:160
void set_filter(ast::expression const &filter)
Sets AST based filter for predicate pushdown.
Definition: parquet.hpp:336
auto const & get_filter() const
Returns AST based filter for predicate pushdown.
Definition: parquet.hpp:243
std::optional< size_t > const & get_num_bytes() const
Returns number of bytes after skipping to end reading row groups at.
Definition: parquet.hpp:222
int64_t get_skip_rows() const
Returns number of rows to skip from the start.
Definition: parquet.hpp:198
Base settings for write_parquet() and chunked_parquet_writer.
Definition: parquet.hpp:869
void enable_utc_timestamps(bool val)
Sets preference for writing timestamps as UTC. Write timestamps as UTC if set to true.
void enable_write_v2_headers(bool val)
Sets preference for V2 page headers. Write V2 page headers if set to true.
auto const & get_sorting_columns() const
Returns the sorting_columns.
Definition: parquet.hpp:1078
auto get_row_group_size_bytes() const
Returns maximum row group size, in bytes.
Definition: parquet.hpp:992
bool is_enabled_int96_timestamps() const
Returns true if timestamps will be written as INT96.
Definition: parquet.hpp:971
void set_metadata(table_input_metadata metadata)
Sets metadata.
void set_row_group_size_rows(size_type size_rows)
Sets the maximum row group size, in rows.
parquet_writer_options_base(sink_info sink)
Constructor from sink.
Definition: parquet.hpp:917
void set_stats_level(statistics_freq sf)
Sets the level of statistics.
auto get_row_group_size_rows() const
Returns maximum row group size, in rows.
Definition: parquet.hpp:999
parquet_writer_options_base()=default
Default constructor.
void set_max_page_size_bytes(size_t size_bytes)
Sets the maximum uncompressed page size, in bytes.
void set_sorting_columns(std::vector< sorting_column > sorting_columns)
Sets sorting columns.
auto is_enabled_write_arrow_schema() const
Returns true if arrow schema will be written.
Definition: parquet.hpp:985
auto is_enabled_write_v2_headers() const
Returns true if V2 page headers should be written.
Definition: parquet.hpp:1071
void set_dictionary_policy(dictionary_policy policy)
Sets the policy for dictionary use.
auto get_max_page_size_bytes() const
Returns the maximum uncompressed page size, in bytes.
Definition: parquet.hpp:1008
void set_max_dictionary_size(size_t size_bytes)
Sets the maximum dictionary size, in bytes.
compression_type get_compression() const
Returns compression format used.
Definition: parquet.hpp:939
auto get_max_dictionary_size() const
Returns maximum dictionary size, in bytes.
Definition: parquet.hpp:1047
void set_compression(compression_type compression)
Sets compression type.
dictionary_policy get_dictionary_policy() const
Returns policy for dictionary use.
Definition: parquet.hpp:1040
void set_compression_statistics(std::shared_ptr< writer_compression_statistics > comp_stats)
Sets the pointer to the output compression statistics.
std::shared_ptr< writer_compression_statistics > get_compression_statistics() const
Returns a shared pointer to the user-provided compression statistics.
Definition: parquet.hpp:1061
void set_max_page_size_rows(size_type size_rows)
Sets the maximum page size, in rows.
auto get_max_page_fragment_size() const
Returns maximum page fragment size, in rows.
Definition: parquet.hpp:1054
void set_key_value_metadata(std::vector< std::map< std::string, std::string >> metadata)
Sets metadata.
void set_max_page_fragment_size(size_type size_rows)
Sets the maximum page fragment size, in rows.
void enable_write_arrow_schema(bool val)
Sets preference for writing arrow schema. Write arrow schema if set to true.
auto is_enabled_utc_timestamps() const
Returns true if timestamps will be written as UTC.
Definition: parquet.hpp:978
void set_row_group_size_bytes(size_t size_bytes)
Sets the maximum row group size, in bytes.
void enable_int96_timestamps(bool req)
Sets timestamp writing preferences. INT96 timestamps will be written if true and TIMESTAMP_MICROS wil...
statistics_freq get_stats_level() const
Returns level of statistics requested in output file.
Definition: parquet.hpp:946
std::vector< std::map< std::string, std::string > > const & get_key_value_metadata() const
Returns Key-Value footer metadata information.
Definition: parquet.hpp:960
auto const & get_metadata() const
Returns associated metadata.
Definition: parquet.hpp:953
auto get_max_page_size_rows() const
Returns maximum page size, in rows.
Definition: parquet.hpp:1020
auto get_column_index_truncate_length() const
Returns maximum length of min or max values in column index, in bytes.
Definition: parquet.hpp:1030
void set_column_index_truncate_length(int32_t size_bytes)
Sets the maximum length of min or max values in column index, in bytes.
sink_info const & get_sink() const
Returns sink info.
Definition: parquet.hpp:932
Base class for Parquet options builders.
Definition: parquet.hpp:1211
BuilderT & compression(compression_type compression)
Sets compression type.
BuilderT & key_value_metadata(std::vector< std::map< std::string, std::string >> metadata)
Sets Key-Value footer metadata.
OptionsT & get_options()
Return reference to the options object being built.
Definition: parquet.hpp:1220
BuilderT & utc_timestamps(bool enabled)
Set to true if timestamps are to be written as UTC.
BuilderT & max_dictionary_size(size_t val)
Sets the maximum dictionary size, in bytes.
BuilderT & max_page_size_bytes(size_t val)
Sets the maximum uncompressed page size, in bytes.
OptionsT && build()
move options member once it's built.
BuilderT & stats_level(statistics_freq sf)
Sets the level of statistics.
BuilderT & column_index_truncate_length(int32_t val)
Sets the desired maximum size in bytes for min and max values in the column index.
BuilderT & compression_statistics(std::shared_ptr< writer_compression_statistics > const &comp_stats)
Sets the pointer to the output compression statistics.
BuilderT & metadata(table_input_metadata metadata)
Sets metadata.
BuilderT & dictionary_policy(enum dictionary_policy val)
Sets the policy for dictionary use.
parquet_writer_options_builder_base(OptionsT options)
Constructor from options.
BuilderT & int96_timestamps(bool enabled)
Sets whether int96 timestamps are written or not.
BuilderT & row_group_size_bytes(size_t val)
Sets the maximum row group size, in bytes.
BuilderT & sorting_columns(std::vector< sorting_column > sorting_columns)
Sets column sorting metadata.
BuilderT & write_arrow_schema(bool enabled)
Set to true if arrow schema is to be written.
parquet_writer_options_builder_base()=default
Default constructor.
BuilderT & write_v2_headers(bool enabled)
Set to true if V2 page headers are to be written.
BuilderT & max_page_fragment_size(size_type val)
Sets the maximum page fragment size, in rows.
BuilderT & row_group_size_rows(size_type val)
Sets the maximum number of rows in output row groups.
BuilderT & max_page_size_rows(size_type val)
Sets the maximum page size, in rows. Counts only top-level rows, ignoring any nesting....
Class to build parquet_writer_options.
Definition: parquet.hpp:1523
parquet_writer_options_builder(sink_info const &sink, table_view const &table)
Constructor from sink and table.
parquet_writer_options_builder()=default
Default constructor.
parquet_writer_options_builder & partitions(std::vector< partition_info > partitions)
Sets partitions in parquet_writer_options.
parquet_writer_options_builder & column_chunks_file_paths(std::vector< std::string > file_paths)
Sets column chunks file path to be set in the raw output metadata.
Settings for write_parquet().
Definition: parquet.hpp:1434
void set_partitions(std::vector< partition_info > partitions)
Sets partitions.
static parquet_writer_options_builder builder(sink_info const &sink, table_view const &table)
Create builder to create parquet_writer_options.
parquet_writer_options()=default
Default constructor.
std::vector< std::string > const & get_column_chunks_file_paths() const
Returns Column chunks file paths to be set in the raw output metadata.
Definition: parquet.hpp:1496
table_view get_table() const
Returns table_view.
Definition: parquet.hpp:1482
void set_column_chunks_file_paths(std::vector< std::string > file_paths)
Sets column chunks file path to be set in the raw output metadata.
static parquet_writer_options_builder builder()
Create builder to create parquet_writer_options.
std::vector< partition_info > const & get_partitions() const
Returns partitions.
Definition: parquet.hpp:1489
Metadata for a table.
Definition: io/types.hpp:893
A set of cudf::column_view's of the same size.
Definition: table_view.hpp:189
A set of cudf::column's of the same size.
Definition: table.hpp:29
rmm::cuda_stream_view const get_default_stream()
Get the current default stream.
table_with_metadata read_parquet(std::vector< std::unique_ptr< cudf::io::datasource >> &&sources, std::vector< parquet::FileMetaData > &&parquet_metadatas, parquet_reader_options const &options, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref())
Reads a Parquet dataset into a set of columns using pre-existing Parquet datasources and file metadat...
constexpr size_type default_row_group_size_rows
1 million rows per row group
Definition: parquet.hpp:32
constexpr int32_t default_column_index_truncate_length
truncate to 64 bytes
Definition: parquet.hpp:35
constexpr size_t default_row_group_size_bytes
Infinite bytes per row group.
Definition: parquet.hpp:30
bool is_supported_write_parquet(compression_type compression)
Check if the compression type is supported for writing Parquet files.
constexpr size_type default_max_page_fragment_size
5000 rows per page fragment
Definition: parquet.hpp:37
constexpr size_t default_max_dictionary_size
1MB dictionary size
Definition: parquet.hpp:36
bool is_supported_read_parquet(compression_type compression)
Check if the compression type is supported for reading Parquet files.
constexpr size_t default_max_page_size_bytes
512KB per page
Definition: parquet.hpp:33
constexpr size_type default_max_page_size_rows
20k rows per page
Definition: parquet.hpp:34
statistics_freq
Column statistics granularity type for parquet/orc writers.
Definition: io/types.hpp:85
dictionary_policy
Control use of dictionary encoding for parquet writer.
Definition: io/types.hpp:214
compression_type
Compression algorithms.
Definition: io/types.hpp:46
@ STATISTICS_ROWGROUP
Per-Rowgroup column statistics.
Definition: io/types.hpp:87
@ ADAPTIVE
Use dictionary when it will not impact compression.
Definition: io/types.hpp:216
std::unique_ptr< std::vector< uint8_t > > merge_row_group_metadata(std::vector< std::unique_ptr< std::vector< uint8_t >>> const &metadata_list)
Merges multiple raw metadata blobs that were previously created by write_parquet into a single metada...
std::unique_ptr< std::vector< uint8_t > > write_parquet(parquet_writer_options const &options, rmm::cuda_stream_view stream=cudf::get_default_stream())
Writes a set of columns to parquet format.
rmm::device_async_resource_ref get_current_device_resource_ref()
Get the current device memory resource reference.
detail::cccl_async_resource_ref< cuda::mr::resource_ref< cuda::mr::device_accessible > > device_async_resource_ref
std::vector< std::unique_ptr< column > > filter(std::vector< column_view > const &predicate_columns, std::string const &predicate_udf, std::vector< column_view > const &filter_columns, bool is_ptx, std::optional< void * > user_data=std::nullopt, null_aware is_null_aware=null_aware::NO, rmm::cuda_stream_view stream=cudf::get_default_stream(), rmm::device_async_resource_ref mr=cudf::get_current_device_resource_ref())
Creates a new column by applying a filter function against every element of the input columns.
int32_t size_type
Row index type for columns and tables.
Definition: types.hpp:84
cuDF-IO API type definitions
cuDF interfaces
Definition: host_udf.hpp:26
A generic expression that can be evaluated to return a value.
Definition: expressions.hpp:61
Destination information for write interfaces.
Definition: io/types.hpp:471
Struct used to describe column sorting metadata.
Definition: parquet.hpp:860
Source information for read interfaces.
Definition: io/types.hpp:316
Table with table metadata used by io readers to return the metadata by value.
Definition: io/types.hpp:292
Class definitions for (mutable)_table_view
Type declarations for libcudf.