From 3eee3a860773b6ba84e58ba7a763b58ab5c831ba Mon Sep 17 00:00:00 2001 From: Andy Grove Date: Mon, 28 Sep 2026 09:08:49 -0600 Subject: [PATCH] test: fix flaky mixed field id directory test on Spark 3.4 and 3.5 The test wrote each side with parallelize over the five-core test session, which leaves an empty part-00000 per write. The read packs the two empty files into one task, and on Spark 3.x a file without ids read after an empty file in the same task has its error wrapped in one more SparkException. Which of the two failing tasks finished first decided the error the job reported, and the one-level getCause check failed when the wrapped one won. Write each side with repartition(1) so there is one file per side and a single failing task. --- .../scala/org/apache/comet/parquet/ParquetReadSuite.scala | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/spark/src/test/scala/org/apache/comet/parquet/ParquetReadSuite.scala b/spark/src/test/scala/org/apache/comet/parquet/ParquetReadSuite.scala index c5b1d04f54..9746d6553b 100644 --- a/spark/src/test/scala/org/apache/comet/parquet/ParquetReadSuite.scala +++ b/spark/src/test/scala/org/apache/comet/parquet/ParquetReadSuite.scala @@ -2427,7 +2427,10 @@ abstract class ParquetReadSuite extends CometTestBase { // Spark checks each file on its own. A directory holding one file with ids and one without // raises on the second, and with `ignoreMissing` the file without ids reads as nulls because - // no root field of it carries the requested id. + // no root field of it carries the requested id. Each side is written as one file. Spread over + // the session's cores, each write would also leave an empty `part-00000`, and on Spark 3.x a + // file without ids read after an empty one in the same task raises inside one more + // `SparkException`, so the error the job reports would depend on which task failed first. test("a file without ids next to a file with ids is checked on its own") { withSQLConf(SQLConf.PARQUET_FIELD_ID_READ_ENABLED.key -> "true") { withTempPath { dir => @@ -2436,11 +2439,13 @@ abstract class ParquetReadSuite extends CometTestBase { val readSchema = new StructType().add("a", IntegerType, true, withId(1)) spark .createDataFrame(spark.sparkContext.parallelize(Seq(Row(100), Row(200))), idSchema) + .repartition(1) .write .mode("overwrite") .parquet(dir.getCanonicalPath) spark .createDataFrame(spark.sparkContext.parallelize(Seq(Row(1), Row(2))), plainSchema) + .repartition(1) .write .mode("append") .parquet(dir.getCanonicalPath)