From f2415a1cc4294aa245d6e4bbf24a304d5db12850 Mon Sep 17 00:00:00 2001 From: Chengcheng Jin Date: Mon, 5 Jan 2026 10:18:42 +0800 Subject: [PATCH 1/2] [GLUTEN-11088][VL] Enable CSV suite --- .../backendsapi/velox/VeloxRuleApi.scala | 2 - .../utils/velox/VeloxTestSettings.scala | 119 +++++++++--------- .../datasources/csv/GlutenCSVSuite.scala | 4 + 3 files changed, 66 insertions(+), 59 deletions(-) diff --git a/backends-velox/src/main/scala/org/apache/gluten/backendsapi/velox/VeloxRuleApi.scala b/backends-velox/src/main/scala/org/apache/gluten/backendsapi/velox/VeloxRuleApi.scala index 7a78bb84681..02f77a55e06 100644 --- a/backends-velox/src/main/scala/org/apache/gluten/backendsapi/velox/VeloxRuleApi.scala +++ b/backends-velox/src/main/scala/org/apache/gluten/backendsapi/velox/VeloxRuleApi.scala @@ -89,7 +89,6 @@ object VeloxRuleApi { BloomFilterMightContainJointRewriteRule.apply( c.session, c.caller.isBloomFilterStatFunction())) - injector.injectPreTransform(c => ArrowScanReplaceRule.apply(c.session)) injector.injectPreTransform(_ => EliminateRedundantGetTimestamp) // Legacy: The legacy transform rule. @@ -171,7 +170,6 @@ object VeloxRuleApi { BloomFilterMightContainJointRewriteRule.apply( c.session, c.caller.isBloomFilterStatFunction())) - injector.injectPreTransform(c => ArrowScanReplaceRule.apply(c.session)) injector.injectPreTransform(_ => EliminateRedundantGetTimestamp) // Gluten RAS: The RAS rule. diff --git a/gluten-ut/spark40/src/test/scala/org/apache/gluten/utils/velox/VeloxTestSettings.scala b/gluten-ut/spark40/src/test/scala/org/apache/gluten/utils/velox/VeloxTestSettings.scala index c33a4027e00..d1e76212477 100644 --- a/gluten-ut/spark40/src/test/scala/org/apache/gluten/utils/velox/VeloxTestSettings.scala +++ b/gluten-ut/spark40/src/test/scala/org/apache/gluten/utils/velox/VeloxTestSettings.scala @@ -27,6 +27,7 @@ import org.apache.spark.sql.execution._ import org.apache.spark.sql.execution.adaptive.velox.VeloxAdaptiveQueryExecSuite import org.apache.spark.sql.execution.datasources._ import org.apache.spark.sql.execution.datasources.binaryfile.GlutenBinaryFileFormatSuite +import org.apache.spark.sql.execution.datasources.csv.{GlutenCSVLegacyTimeParserSuite, GlutenCSVv1Suite, GlutenCSVv2Suite} import org.apache.spark.sql.execution.datasources.json.{GlutenJsonLegacyTimeParserSuite, GlutenJsonV1Suite, GlutenJsonV2Suite} import org.apache.spark.sql.execution.datasources.orc._ import org.apache.spark.sql.execution.datasources.parquet._ @@ -233,61 +234,66 @@ class VeloxTestSettings extends BackendTestSettings { enableSuite[GlutenBinaryFileFormatSuite] // Exception. .exclude("column pruning - non-readable file") - // TODO: fix in Spark-4.0 - // enableSuite[GlutenCSVv1Suite] - // // file cars.csv include null string, Arrow not support to read - // .exclude("DDL test with schema") - // .exclude("save csv") - // .exclude("save csv with compression codec option") - // .exclude("save csv with empty fields with user defined empty values") - // .exclude("save csv with quote") - // .exclude("SPARK-13543 Write the output as uncompressed via option()") - // .exclude("DDL test with tab separated file") - // .exclude("DDL test parsing decimal type") - // .exclude("test with tab delimiter and double quote") - // // Arrow not support corrupt record - // .exclude("SPARK-27873: disabling enforceSchema should not fail columnNameOfCorruptRecord") - // // varchar - // .exclude("SPARK-48241: CSV parsing failure with char/varchar type columns") - // // Flaky and already excluded in other cases - // .exclude("Gluten - test for FAILFAST parsing mode") + enableSuite[GlutenCSVv1Suite] + // file cars.csv include null string, Arrow not support to read + .exclude("DDL test with schema") + .exclude("save csv") + .exclude("save csv with compression codec option") + .exclude("save csv with empty fields with user defined empty values") + .exclude("save csv with quote") + .exclude("SPARK-13543 Write the output as uncompressed via option()") + .exclude("DDL test with tab separated file") + .exclude("DDL test parsing decimal type") + .exclude("test with tab delimiter and double quote") + .exclude("when mode is null, will fall back to PermissiveMode mode") + .exclude("SPARK-46890: CSV fails on a column with default and without enforcing schema") + // Arrow not support corrupt record + .exclude("SPARK-27873: disabling enforceSchema should not fail columnNameOfCorruptRecord") + // varchar + .exclude("SPARK-48241: CSV parsing failure with char/varchar type columns") + // Flaky and already excluded in other cases + .exclude("Gluten - test for FAILFAST parsing mode") - // enableSuite[GlutenCSVv2Suite] - // .exclude("Gluten - test for FAILFAST parsing mode") - // // Rule org.apache.spark.sql.execution.datasources.v2.V2ScanRelationPushDown in batch - // // Early Filter and Projection Push-Down generated an invalid plan - // .exclude("SPARK-26208: write and read empty data to csv file with headers") - // // file cars.csv include null string, Arrow not support to read - // .exclude("old csv data source name works") - // .exclude("DDL test with schema") - // .exclude("save csv") - // .exclude("save csv with compression codec option") - // .exclude("save csv with empty fields with user defined empty values") - // .exclude("save csv with quote") - // .exclude("SPARK-13543 Write the output as uncompressed via option()") - // .exclude("DDL test with tab separated file") - // .exclude("DDL test parsing decimal type") - // .exclude("test with tab delimiter and double quote") - // // Arrow not support corrupt record - // .exclude("SPARK-27873: disabling enforceSchema should not fail columnNameOfCorruptRecord") - // // varchar - // .exclude("SPARK-48241: CSV parsing failure with char/varchar type columns") + enableSuite[GlutenCSVv2Suite] + .exclude("Gluten - test for FAILFAST parsing mode") + // Rule org.apache.spark.sql.execution.datasources.v2.V2ScanRelationPushDown in batch + // Early Filter and Projection Push-Down generated an invalid plan + .exclude("SPARK-26208: write and read empty data to csv file with headers") + // file cars.csv include null string, Arrow not support to read + .exclude("old csv data source name works") + .exclude("DDL test with schema") + .exclude("save csv") + .exclude("save csv with compression codec option") + .exclude("save csv with empty fields with user defined empty values") + .exclude("save csv with quote") + .exclude("SPARK-13543 Write the output as uncompressed via option()") + .exclude("DDL test with tab separated file") + .exclude("DDL test parsing decimal type") + .exclude("test with tab delimiter and double quote") + .exclude("when mode is null, will fall back to PermissiveMode mode") + .exclude("SPARK-46890: CSV fails on a column with default and without enforcing schema") + // Arrow not support corrupt record + .exclude("SPARK-27873: disabling enforceSchema should not fail columnNameOfCorruptRecord") + // varchar + .exclude("SPARK-48241: CSV parsing failure with char/varchar type columns") - // enableSuite[GlutenCSVLegacyTimeParserSuite] - // // file cars.csv include null string, Arrow not support to read - // .exclude("DDL test with schema") - // .exclude("save csv") - // .exclude("save csv with compression codec option") - // .exclude("save csv with empty fields with user defined empty values") - // .exclude("save csv with quote") - // .exclude("SPARK-13543 Write the output as uncompressed via option()") - // // Arrow not support corrupt record - // .exclude("SPARK-27873: disabling enforceSchema should not fail columnNameOfCorruptRecord") - // .exclude("DDL test with tab separated file") - // .exclude("DDL test parsing decimal type") - // .exclude("test with tab delimiter and double quote") - // // varchar - // .exclude("SPARK-48241: CSV parsing failure with char/varchar type columns") + enableSuite[GlutenCSVLegacyTimeParserSuite] + // file cars.csv include null string, Arrow not support to read + .exclude("DDL test with schema") + .exclude("save csv") + .exclude("save csv with compression codec option") + .exclude("save csv with empty fields with user defined empty values") + .exclude("save csv with quote") + .exclude("SPARK-13543 Write the output as uncompressed via option()") + .exclude("when mode is null, will fall back to PermissiveMode mode") + .exclude("SPARK-46890: CSV fails on a column with default and without enforcing schema") + // Arrow not support corrupt record + .exclude("SPARK-27873: disabling enforceSchema should not fail columnNameOfCorruptRecord") + .exclude("DDL test with tab separated file") + .exclude("DDL test parsing decimal type") + .exclude("test with tab delimiter and double quote") + // varchar + .exclude("SPARK-48241: CSV parsing failure with char/varchar type columns") enableSuite[GlutenJsonV1Suite] // FIXME: Array direct selection fails .exclude("Complex field and type inferring") @@ -556,10 +562,9 @@ class VeloxTestSettings extends BackendTestSettings { enableSuite[GlutenPathFilterStrategySuite] enableSuite[GlutenPathFilterSuite] enableSuite[GlutenPruneFileSourcePartitionsSuite] - // TODO: fix in Spark-4.0 - // enableSuite[GlutenCSVReadSchemaSuite] - // enableSuite[GlutenHeaderCSVReadSchemaSuite] - // .exclude("change column type from int to long") + enableSuite[GlutenCSVReadSchemaSuite] + enableSuite[GlutenHeaderCSVReadSchemaSuite] + .exclude("change column type from int to long") enableSuite[GlutenJsonReadSchemaSuite] enableSuite[GlutenOrcReadSchemaSuite] enableSuite[GlutenVectorizedOrcReadSchemaSuite] diff --git a/gluten-ut/spark40/src/test/scala/org/apache/spark/sql/execution/datasources/csv/GlutenCSVSuite.scala b/gluten-ut/spark40/src/test/scala/org/apache/spark/sql/execution/datasources/csv/GlutenCSVSuite.scala index 6cfa9f2028e..af0f59b9bca 100644 --- a/gluten-ut/spark40/src/test/scala/org/apache/spark/sql/execution/datasources/csv/GlutenCSVSuite.scala +++ b/gluten-ut/spark40/src/test/scala/org/apache/spark/sql/execution/datasources/csv/GlutenCSVSuite.scala @@ -132,4 +132,8 @@ class GlutenCSVLegacyTimeParserSuite extends GlutenCSVSuite { override def sparkConf: SparkConf = super.sparkConf .set(SQLConf.LEGACY_TIME_PARSER_POLICY, "legacy") + + // The source CSVLegacyTimeParserSuite exclude the test + override def excluded: Seq[String] = + Seq("Write timestamps correctly in ISO8601 format by default") } From 69da26c98dcbdde05e82e61a7e4101577a816860 Mon Sep 17 00:00:00 2001 From: Chengcheng Jin Date: Tue, 6 Jan 2026 18:26:55 +0800 Subject: [PATCH 2/2] fix arrow dataset rule --- .../scala/org/apache/gluten/backendsapi/velox/VeloxRuleApi.scala | 1 - 1 file changed, 1 deletion(-) diff --git a/backends-velox/src/main/scala/org/apache/gluten/backendsapi/velox/VeloxRuleApi.scala b/backends-velox/src/main/scala/org/apache/gluten/backendsapi/velox/VeloxRuleApi.scala index 02f77a55e06..e2563ec0fcf 100644 --- a/backends-velox/src/main/scala/org/apache/gluten/backendsapi/velox/VeloxRuleApi.scala +++ b/backends-velox/src/main/scala/org/apache/gluten/backendsapi/velox/VeloxRuleApi.scala @@ -65,7 +65,6 @@ object VeloxRuleApi { injector.injectOptimizerRule(HLLRewriteRule.apply) injector.injectOptimizerRule(CollapseGetJsonObjectExpressionRule.apply) injector.injectOptimizerRule(RewriteCastFromArray.apply) - injector.injectPostHocResolutionRule(ArrowConvertorRule.apply) injector.injectOptimizerRule(RewriteUnboundedWindow.apply) if (BackendsApiManager.getSettings.supportAppendDataExec()) { injector.injectPlannerStrategy(SparkShimLoader.getSparkShims.getRewriteCreateTableAsSelect(_))