This is an automated email from the ASF dual-hosted git repository.
Yicong-Huang pushed a commit to branch branch-4.x
in repository https://gitbox.apache.org/repos/asf/spark.git
The following commit(s) were added to refs/heads/branch-4.x by this push:
new 06607eea4323 [SPARK-57422][PYTHON] Remove unused
CogroupPandasUDFSerializer
06607eea4323 is described below
commit 06607eea43239bf324eb2735719f179882b99a5a
Author: Yicong Huang <[email protected]>
AuthorDate: Wed Jun 17 07:27:26 2026 +0000
[SPARK-57422][PYTHON] Remove unused CogroupPandasUDFSerializer
### What changes were proposed in this pull request?
Delete `CogroupPandasUDFSerializer` from
`python/pyspark/sql/pandas/serializers.py`, along with the now-unused
`from_arrow_schema` import.
### Why are the changes needed?
`CogroupPandasUDFSerializer` is no longer used after SPARK-56718 refactored
`SQL_COGROUPED_MAP_PANDAS_UDF` to use `ArrowStreamCoGroupSerializer` directly.
This class can be safely deleted.
Part of SPARK-55384.
### Does this PR introduce _any_ user-facing change?
No.
### How was this patch tested?
Existing tests: `pyspark.sql.tests.pandas.test_pandas_cogrouped_map`.
### Was this patch authored or co-authored using generative AI tooling?
No.
Closes #56482 from
Yicong-Huang/SPARK-57422/cleanup/cogroup-pandas-udf-serializer.
Authored-by: Yicong Huang <[email protected]>
Signed-off-by: Yicong-Huang <[email protected]>
(cherry picked from commit e262222ad7b2ea8828da901f3a86b8cc37b25c2a)
Signed-off-by: Yicong-Huang <[email protected]>
---
python/pyspark/sql/pandas/serializers.py | 38 +-------------------------------
1 file changed, 1 insertion(+), 37 deletions(-)
diff --git a/python/pyspark/sql/pandas/serializers.py
b/python/pyspark/sql/pandas/serializers.py
index 9e73d7ded12f..d6bf3b7d7416 100644
--- a/python/pyspark/sql/pandas/serializers.py
+++ b/python/pyspark/sql/pandas/serializers.py
@@ -35,10 +35,7 @@ from pyspark.sql.conversion import (
ArrowBatchTransformer,
PandasToArrowConversion,
)
-from pyspark.sql.pandas.types import (
- from_arrow_schema,
- to_arrow_type,
-)
+from pyspark.sql.pandas.types import to_arrow_type
from pyspark.sql.types import (
DataType,
StringType,
@@ -546,39 +543,6 @@ class
ArrowStreamAggPandasUDFSerializer(ArrowStreamPandasUDFSerializer):
return "ArrowStreamAggPandasUDFSerializer"
-class CogroupPandasUDFSerializer(ArrowStreamPandasUDFSerializer):
- def load_stream(self, stream):
- """
- Deserialize Cogrouped ArrowRecordBatches to a tuple of Arrow tables
and yield as two
- lists of pandas.Series.
- """
- import pyarrow as pa
-
- for left_batches, right_batches in
ArrowStreamCoGroupSerializer.load_stream(self, stream):
- left_table = pa.Table.from_batches(left_batches)
- right_table = pa.Table.from_batches(right_batches)
- yield (
- ArrowBatchTransformer.to_pandas(
- left_table,
- timezone=self._timezone,
- schema=from_arrow_schema(left_table.schema),
- struct_in_pandas=self._struct_in_pandas,
- ndarray_as_list=self._ndarray_as_list,
- prefer_int_ext_dtype=self._prefer_int_ext_dtype,
- df_for_struct=self._df_for_struct,
- ),
- ArrowBatchTransformer.to_pandas(
- right_table,
- timezone=self._timezone,
- schema=from_arrow_schema(right_table.schema),
- struct_in_pandas=self._struct_in_pandas,
- ndarray_as_list=self._ndarray_as_list,
- prefer_int_ext_dtype=self._prefer_int_ext_dtype,
- df_for_struct=self._df_for_struct,
- ),
- )
-
-
class ApplyInPandasWithStateSerializer(ArrowStreamPandasUDFSerializer):
"""
Serializer used by Python worker to evaluate UDF for
applyInPandasWithState.
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]