Skip to content

Commit bf55c10

Browse files
committed
Resolve the write setup once for both write paths
write_file and the streaming branch of _dataframe_to_data_files resolved the file format, the format model, the location provider and the file schema with the same block. Move it into one helper.
1 parent 8f9b510 commit bf55c10

1 file changed

Lines changed: 14 additions & 26 deletions

File tree

‎pyiceberg/io/pyarrow.py‎

Lines changed: 14 additions & 26 deletions
Original file line numberDiff line numberDiff line change
@@ -148,7 +148,7 @@
148148
)
149149
from pyiceberg.table import DOWNCAST_NS_TIMESTAMP_TO_US_ON_WRITE, TableProperties
150150
from pyiceberg.table.deletion_vector import deletion_vectors_from_puffin_file
151-
from pyiceberg.table.locations import load_location_provider
151+
from pyiceberg.table.locations import LocationProvider, load_location_provider
152152
from pyiceberg.table.metadata import TableMetadata
153153
from pyiceberg.table.name_mapping import NameMapping, apply_name_mapping
154154
from pyiceberg.table.puffin import PuffinFile
@@ -2791,25 +2791,24 @@ def _build_data_file(
27912791
)
27922792

27932793

2794-
def write_file(io: FileIO, table_metadata: TableMetadata, tasks: Iterator[WriteTask]) -> Iterator[DataFile]:
2795-
from pyiceberg.table import DOWNCAST_NS_TIMESTAMP_TO_US_ON_WRITE, TableProperties
2796-
2794+
def _resolve_write_setup(table_metadata: TableMetadata) -> tuple[FileFormat, FileFormatModel, LocationProvider, Schema]:
27972795
file_format = FileFormat(
2798-
table_metadata.properties.get(
2799-
TableProperties.WRITE_FILE_FORMAT,
2800-
TableProperties.WRITE_FILE_FORMAT_DEFAULT,
2801-
)
2796+
table_metadata.properties.get(TableProperties.WRITE_FILE_FORMAT, TableProperties.WRITE_FILE_FORMAT_DEFAULT)
28022797
)
28032798
format_model = FileFormatFactory.get(file_format)
28042799
location_provider = load_location_provider(table_location=table_metadata.location, table_properties=table_metadata.properties)
2800+
table_schema = table_metadata.schema()
2801+
if (sanitized_schema := sanitize_column_names(table_schema)) != table_schema:
2802+
file_schema = sanitized_schema
2803+
else:
2804+
file_schema = table_schema
2805+
return file_format, format_model, location_provider, file_schema
28052806

2806-
def write_data_file(task: WriteTask) -> DataFile:
2807-
table_schema = table_metadata.schema()
2808-
if (sanitized_schema := sanitize_column_names(table_schema)) != table_schema:
2809-
file_schema = sanitized_schema
2810-
else:
2811-
file_schema = table_schema
28122807

2808+
def write_file(io: FileIO, table_metadata: TableMetadata, tasks: Iterator[WriteTask]) -> Iterator[DataFile]:
2809+
file_format, format_model, location_provider, file_schema = _resolve_write_setup(table_metadata)
2810+
2811+
def write_data_file(task: WriteTask) -> DataFile:
28132812
downcast_ns_timestamp_to_us = Config().get_bool(DOWNCAST_NS_TIMESTAMP_TO_US_ON_WRITE) or False
28142813
batches = [
28152814
_to_requested_schema(
@@ -3070,18 +3069,7 @@ def _dataframe_to_data_files(
30703069
"Materialise the reader as a pa.Table first, or follow "
30713070
"https://github.com/apache/iceberg-python/issues/2152 for partitioned streaming support."
30723071
)
3073-
file_format = FileFormat(
3074-
table_metadata.properties.get(TableProperties.WRITE_FILE_FORMAT, TableProperties.WRITE_FILE_FORMAT_DEFAULT)
3075-
)
3076-
format_model = FileFormatFactory.get(file_format)
3077-
location_provider = load_location_provider(
3078-
table_location=table_metadata.location, table_properties=table_metadata.properties
3079-
)
3080-
table_schema = table_metadata.schema()
3081-
if (sanitized_schema := sanitize_column_names(table_schema)) != table_schema:
3082-
file_schema = sanitized_schema
3083-
else:
3084-
file_schema = table_schema
3072+
file_format, format_model, location_provider, file_schema = _resolve_write_setup(table_metadata)
30853073

30863074
batches = iter(df)
30873075
for batch in batches:

0 commit comments

Comments
 (0)