From 8c85d34869e0742b7e9db41a98f0b499f1014830 Mon Sep 17 00:00:00 2001 From: lee <690585471@qq.com> Date: Fri, 28 Jul 2023 18:21:23 +0800 Subject: [PATCH] Write Page Offset Index For All-Nan Pages (#4567) * fix offset index none * add test * add test * Cleanup --------- Co-authored-by: guojie.lgj Co-authored-by: Raphael Taylor-Davies --- parquet/src/arrow/arrow_writer/mod.rs | 21 +++++++++++++++++++++ parquet/src/column/writer/mod.rs | 13 +++++-------- 2 files changed, 26 insertions(+), 8 deletions(-) diff --git a/parquet/src/arrow/arrow_writer/mod.rs b/parquet/src/arrow/arrow_writer/mod.rs index ccec4ffb20c0..d3d4e2626fe3 100644 --- a/parquet/src/arrow/arrow_writer/mod.rs +++ b/parquet/src/arrow/arrow_writer/mod.rs @@ -1650,6 +1650,27 @@ mod tests { writer.close().unwrap(); } + #[test] + fn check_page_offset_index_with_nan() { + let values = Arc::new(Float64Array::from(vec![f64::NAN; 10])); + let schema = Schema::new(vec![Field::new("col", DataType::Float64, true)]); + let batch = RecordBatch::try_new(Arc::new(schema), vec![values]).unwrap(); + + let mut out = Vec::with_capacity(1024); + let mut writer = ArrowWriter::try_new(&mut out, batch.schema(), None) + .expect("Unable to write file"); + writer.write(&batch).unwrap(); + let file_meta_data = writer.close().unwrap(); + for row_group in file_meta_data.row_groups { + for column in row_group.columns { + assert!(column.offset_index_offset.is_some()); + assert!(column.offset_index_length.is_some()); + assert!(column.column_index_offset.is_none()); + assert!(column.column_index_length.is_none()); + } + } + } + #[test] fn i8_single_column() { required_and_optional::(0..SMALL_SIZE as i8); diff --git a/parquet/src/column/writer/mod.rs b/parquet/src/column/writer/mod.rs index 1cacfe793328..3d8ce283ae64 100644 --- a/parquet/src/column/writer/mod.rs +++ b/parquet/src/column/writer/mod.rs @@ -500,14 +500,11 @@ impl<'a, E: ColumnValueEncoder> GenericColumnWriter<'a, E> { let metadata = self.write_column_metadata()?; self.page_writer.close()?; - let (column_index, offset_index) = if self.column_index_builder.valid() { - // build the column and offset index - let column_index = self.column_index_builder.build_to_thrift(); - let offset_index = self.offset_index_builder.build_to_thrift(); - (Some(column_index), Some(offset_index)) - } else { - (None, None) - }; + let column_index = self + .column_index_builder + .valid() + .then(|| self.column_index_builder.build_to_thrift()); + let offset_index = Some(self.offset_index_builder.build_to_thrift()); Ok(ColumnCloseResult { bytes_written: self.column_metrics.total_bytes_written,