Skip to content

Commit 390e404

Browse files
committed
Skip CDC tests on pyarrow < 21, fix stale comment
The CDC tests raised ImportError on pyarrow 18-20, which pyproject still declares as supported.
1 parent 29debee commit 390e404

3 files changed

Lines changed: 18 additions & 11 deletions

File tree

‎mkdocs/docs/configuration.md‎

Lines changed: 3 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -85,9 +85,9 @@ Iceberg tables support table properties to configure table behavior.
8585
| `write.parquet.page-size-bytes` | Size in bytes | 1MB | Set a target threshold for the approximate encoded size of data pages within a column chunk |
8686
| `write.parquet.page-row-limit` | Number of rows | 20000 | Set a target threshold for the maximum number of rows within a column chunk |
8787
| `write.parquet.dict-size-bytes` | Size in bytes | 2MB | Set the dictionary page size limit per row group |
88-
| `write.parquet.content-defined-chunking.enabled` | Boolean | False | Enables content-defined chunking (CDC) for the Parquet writer, which produces stable page boundaries across appends. Requires `pyarrow>=21.0.0`, and raises at write time on older versions. |
89-
| `write.parquet.content-defined-chunking.min-chunk-size` | Size in bytes | 256KiB (262144) | The minimum chunk size used for content-defined chunking |
90-
| `write.parquet.content-defined-chunking.max-chunk-size` | Size in bytes | 1MiB (1048576) | The maximum chunk size used for content-defined chunking |
88+
| `write.parquet.content-defined-chunking.enabled` | Boolean | False | Enables content-defined chunking (CDC) for the Parquet writer, which produces stable page boundaries across appends. Requires `pyarrow>=21.0.0`; raises at write time on older versions. |
89+
| `write.parquet.content-defined-chunking.min-chunk-size` | Size in bytes | 256KiB | The minimum chunk size used for content-defined chunking |
90+
| `write.parquet.content-defined-chunking.max-chunk-size` | Size in bytes | 1MiB | The maximum chunk size used for content-defined chunking |
9191
| `write.parquet.content-defined-chunking.norm-level` | Integer | 0 | The normalization level for content-defined chunking, controlling how tightly chunk sizes cluster around the average |
9292
| `write.metadata.previous-versions-max` | Integer | 100 | The max number of previous version metadata files to keep before deleting after commit. |
9393
| `write.metadata.delete-after-commit.enabled` | Boolean | False | Whether to automatically delete old *tracked* metadata files after each table commit. It will retain a number of the most recent metadata files, which can be set using property `write.metadata.previous-versions-max`. |

‎pyiceberg/io/pyarrow.py‎

Lines changed: 4 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -2939,7 +2939,7 @@ def _get_parquet_writer_kwargs(table_properties: Properties) -> dict[str, Any]:
29392939
if compression_codec == ICEBERG_UNCOMPRESSED_CODEC:
29402940
compression_codec = PYARROW_UNCOMPRESSED_CODEC
29412941

2942-
parquet_writer_kwargs = {
2942+
parquet_writer_kwargs: dict[str, Any] = {
29432943
"compression": compression_codec,
29442944
"compression_level": compression_level,
29452945
"data_page_size": property_as_int(
@@ -2959,17 +2959,15 @@ def _get_parquet_writer_kwargs(table_properties: Properties) -> dict[str, Any]:
29592959
),
29602960
}
29612961

2962-
# Unlike the properties above, which PyArrow's writer never supports and are safe to silently
2963-
# drop, CDC is a version-gated feature: silently ignoring it would produce a table that no longer
2964-
# has the content-defined chunk boundaries the user explicitly asked for, so this raises instead.
2962+
# Unlike the unsupported options warned about above, a CDC request must not be dropped:
2963+
# writing without the requested chunk boundaries silently defeats the point, so raise instead.
29652964
if property_as_bool(
29662965
properties=table_properties,
29672966
property_name=TableProperties.PARQUET_CDC_ENABLED,
29682967
default=TableProperties.PARQUET_CDC_ENABLED_DEFAULT,
29692968
):
29702969
_require_pyarrow_version("21.0.0", "Parquet content-defined chunking")
2971-
# PyArrow itself validates these values (e.g. max-chunk-size > min-chunk-size) and raises a
2972-
# clear OSError, so there's no need to duplicate that validation here.
2970+
# PyArrow validates these values itself (e.g. max-chunk-size > min-chunk-size).
29732971
parquet_writer_kwargs["use_content_defined_chunking"] = {
29742972
"min_chunk_size": property_as_int(
29752973
properties=table_properties,

‎tests/io/test_pyarrow.py‎

Lines changed: 11 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -127,6 +127,11 @@
127127
reason="Requires pyarrow version >= 20.0.0",
128128
)
129129

130+
skip_if_pyarrow_too_old_for_cdc = pytest.mark.skipif(
131+
version.parse(pyarrow.__version__) < version.parse("21.0.0"),
132+
reason="Requires pyarrow version >= 21.0.0",
133+
)
134+
130135

131136
def test_pyarrow_infer_local_fs_from_path() -> None:
132137
"""Test path with `file` scheme and no scheme both use LocalFileSystem"""
@@ -5469,22 +5474,24 @@ def test_dictionary_columns_produces_dict_encoded_output(tmpdir: str) -> None:
54695474
"table_properties,expected",
54705475
[
54715476
({}, None),
5472-
(
5477+
pytest.param(
54735478
{TableProperties.PARQUET_CDC_ENABLED: "true"},
54745479
{
54755480
"min_chunk_size": TableProperties.PARQUET_CDC_MIN_CHUNK_SIZE_DEFAULT,
54765481
"max_chunk_size": TableProperties.PARQUET_CDC_MAX_CHUNK_SIZE_DEFAULT,
54775482
"norm_level": TableProperties.PARQUET_CDC_NORM_LEVEL_DEFAULT,
54785483
},
5484+
marks=skip_if_pyarrow_too_old_for_cdc,
54795485
),
5480-
(
5486+
pytest.param(
54815487
{
54825488
TableProperties.PARQUET_CDC_ENABLED: "true",
54835489
TableProperties.PARQUET_CDC_MIN_CHUNK_SIZE: "4096",
54845490
TableProperties.PARQUET_CDC_MAX_CHUNK_SIZE: "8192",
54855491
TableProperties.PARQUET_CDC_NORM_LEVEL: "2",
54865492
},
54875493
{"min_chunk_size": 4096, "max_chunk_size": 8192, "norm_level": 2},
5494+
marks=skip_if_pyarrow_too_old_for_cdc,
54885495
),
54895496
],
54905497
)
@@ -5499,6 +5506,7 @@ def test_get_parquet_writer_kwargs_cdc_enabled_unsupported_pyarrow_version(monke
54995506
_get_parquet_writer_kwargs({TableProperties.PARQUET_CDC_ENABLED: "true"})
55005507

55015508

5509+
@skip_if_pyarrow_too_old_for_cdc
55025510
def test_get_parquet_writer_kwargs_cdc_invalid_chunk_sizes_raises_from_pyarrow() -> None:
55035511
"""PyArrow validates min/max chunk sizes itself; pyiceberg doesn't duplicate that check."""
55045512
kwargs = _get_parquet_writer_kwargs(
@@ -5514,6 +5522,7 @@ def test_get_parquet_writer_kwargs_cdc_invalid_chunk_sizes_raises_from_pyarrow()
55145522
writer.write_table(table)
55155523

55165524

5525+
@skip_if_pyarrow_too_old_for_cdc
55175526
def test_write_file_with_content_defined_chunking_enabled(tmp_path: Path) -> None:
55185527
"""Writing a table with CDC enabled should forward use_content_defined_chunking to pq.ParquetWriter."""
55195528
from pyiceberg.table import WriteTask

0 commit comments

Comments
 (0)