Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 2 additions & 15 deletions python/pyspark/eval_handlers/_arrow.py
Original file line number Diff line number Diff line change
Expand Up @@ -251,19 +251,6 @@ def run(self, split_index: int, data: Iterator[GroupedBatch]) -> Iterator[pa.Rec
yield ArrowBatchTransformer.wrap_struct(batch)


def _concat_group_batches(batch_list: list["pa.RecordBatch"]) -> "pa.RecordBatch":
"""Concatenate a group's RecordBatches into a single one, with a fallback for
pyarrow before 19.0.0 (which lacks ``pa.concat_batches``). Remove the fallback
once support for those versions is dropped."""
import pyarrow as pa

if hasattr(pa, "concat_batches"):
return pa.concat_batches(batch_list)
return pa.RecordBatch.from_struct_array(
pa.concat_arrays([b.to_struct_array() for b in batch_list])
)


class ArrowGroupedAggUDFHandler(GroupedEvalTypeHandler["pa.RecordBatch"]):
"""SQL_GROUPED_AGG_ARROW_UDF: each UDF reduces its input columns over the whole
group to a single scalar; emit one row per group with one column per UDF,
Expand All @@ -290,7 +277,7 @@ def run(self, split_index: int, data: Iterator[GroupedBatch]) -> Iterator[pa.Rec
batch_list = list(group)
if not batch_list:
continue
concatenated = _concat_group_batches(batch_list)
concatenated = ArrowBatchTransformer.concat_batches(batch_list)
results = [
udf_func(
*[concatenated.column(o) for o in args_offsets],
Expand Down Expand Up @@ -369,7 +356,7 @@ def run(self, split_index: int, data: Iterator[GroupedBatch]) -> Iterator[pa.Rec
batch_list = list(group)
if not batch_list:
continue
concatenated = _concat_group_batches(batch_list)
concatenated = ArrowBatchTransformer.concat_batches(batch_list)
num_rows = concatenated.num_rows

result_arrays = []
Expand Down
21 changes: 21 additions & 0 deletions python/pyspark/sql/conversion.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@
TYPE_CHECKING,
Any,
Callable,
Iterable,
Iterator,
List,
Optional,
Expand Down Expand Up @@ -180,6 +181,26 @@ def select_columns(cls, batch: "pa.RecordBatch", column_indices: list[int]) -> "
[batch.schema.names[i] for i in column_indices],
)

@classmethod
def concat_batches(cls, batches: Iterable["pa.RecordBatch"]) -> "pa.RecordBatch":

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

the only remaining concern is that if we need to do some validity checks

  • are all batches sharing the same schema
  • will the total size exceeding the upper limit of a record batch (I recall 2GB). otherwise it would need chunked arrays.

But since this is the current behavior we can keep it this way and rely on arrow's error messages for now.

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Agreed. I checked that both pa.concat_batches and the pre-19 pa.concat_arrays fallback already reject incompatible schemas. The size limit depends on the Arrow type and buffer layout, so a generic precheck here would be incomplete. I’ll keep the current behavior and let Arrow report these errors for now.

"""Concatenate same-schema RecordBatches by row.

A single batch is returned unchanged. PyArrow before 19.0.0 has no ``concat_batches``;
the fallback concatenates the equivalent StructArrays and converts the result back to a
RecordBatch.
"""
import pyarrow as pa

batches = tuple(batches)
assert batches
if len(batches) == 1:
return batches[0]
if hasattr(pa, "concat_batches"):
return pa.concat_batches(batches)
return pa.RecordBatch.from_struct_array(
pa.concat_arrays([batch.to_struct_array() for batch in batches])
)

@staticmethod
def wrap_struct(batch: "pa.RecordBatch") -> "pa.RecordBatch":
"""
Expand Down
11 changes: 11 additions & 0 deletions python/pyspark/sql/tests/test_conversion.py
Original file line number Diff line number Diff line change
Expand Up @@ -115,6 +115,17 @@ def test_flatten_struct_empty_batch(self):
self.assertEqual(flattened.num_rows, 0)
self.assertEqual(flattened.num_columns, 2)

def test_concat_batches(self):
import pyarrow as pa

batches = [
pa.RecordBatch.from_arrays([pa.array([1, 2])], ["x"]),
pa.RecordBatch.from_arrays([pa.array([3])], ["x"]),
]
result = ArrowBatchTransformer.concat_batches(iter(batches))
self.assertEqual(result.column(0).to_pylist(), [1, 2, 3])
self.assertIs(ArrowBatchTransformer.concat_batches(iter(batches[:1])), batches[0])

def test_wrap_struct_basic(self):
"""Test wrapping columns into a struct."""
import pyarrow as pa
Expand Down
Loading