Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -65,7 +65,6 @@ object VeloxRuleApi {
injector.injectOptimizerRule(HLLRewriteRule.apply)
injector.injectOptimizerRule(CollapseGetJsonObjectExpressionRule.apply)
injector.injectOptimizerRule(RewriteCastFromArray.apply)
injector.injectPostHocResolutionRule(ArrowConvertorRule.apply)
injector.injectOptimizerRule(RewriteUnboundedWindow.apply)
if (BackendsApiManager.getSettings.supportAppendDataExec()) {
injector.injectPlannerStrategy(SparkShimLoader.getSparkShims.getRewriteCreateTableAsSelect(_))
Expand All @@ -89,7 +88,6 @@ object VeloxRuleApi {
BloomFilterMightContainJointRewriteRule.apply(
c.session,
c.caller.isBloomFilterStatFunction()))
injector.injectPreTransform(c => ArrowScanReplaceRule.apply(c.session))

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Just to confirm: is CSV format no longer supported, or do we only need a fallback for Spark 40 and later versions?

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

CSV format is no longer supported

injector.injectPreTransform(_ => EliminateRedundantGetTimestamp)

// Legacy: The legacy transform rule.
Expand Down Expand Up @@ -171,7 +169,6 @@ object VeloxRuleApi {
BloomFilterMightContainJointRewriteRule.apply(
c.session,
c.caller.isBloomFilterStatFunction()))
injector.injectPreTransform(c => ArrowScanReplaceRule.apply(c.session))
injector.injectPreTransform(_ => EliminateRedundantGetTimestamp)

// Gluten RAS: The RAS rule.
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@ import org.apache.spark.sql.execution._
import org.apache.spark.sql.execution.adaptive.velox.VeloxAdaptiveQueryExecSuite
import org.apache.spark.sql.execution.datasources._
import org.apache.spark.sql.execution.datasources.binaryfile.GlutenBinaryFileFormatSuite
import org.apache.spark.sql.execution.datasources.csv.{GlutenCSVLegacyTimeParserSuite, GlutenCSVv1Suite, GlutenCSVv2Suite}
import org.apache.spark.sql.execution.datasources.json.{GlutenJsonLegacyTimeParserSuite, GlutenJsonV1Suite, GlutenJsonV2Suite}
import org.apache.spark.sql.execution.datasources.orc._
import org.apache.spark.sql.execution.datasources.parquet._
Expand Down Expand Up @@ -233,61 +234,66 @@ class VeloxTestSettings extends BackendTestSettings {
enableSuite[GlutenBinaryFileFormatSuite]
// Exception.
.exclude("column pruning - non-readable file")
// TODO: fix in Spark-4.0
// enableSuite[GlutenCSVv1Suite]
// // file cars.csv include null string, Arrow not support to read
// .exclude("DDL test with schema")
// .exclude("save csv")
// .exclude("save csv with compression codec option")
// .exclude("save csv with empty fields with user defined empty values")
// .exclude("save csv with quote")
// .exclude("SPARK-13543 Write the output as uncompressed via option()")
// .exclude("DDL test with tab separated file")
// .exclude("DDL test parsing decimal type")
// .exclude("test with tab delimiter and double quote")
// // Arrow not support corrupt record
// .exclude("SPARK-27873: disabling enforceSchema should not fail columnNameOfCorruptRecord")
// // varchar
// .exclude("SPARK-48241: CSV parsing failure with char/varchar type columns")
// // Flaky and already excluded in other cases
// .exclude("Gluten - test for FAILFAST parsing mode")
enableSuite[GlutenCSVv1Suite]
// file cars.csv include null string, Arrow not support to read
.exclude("DDL test with schema")
.exclude("save csv")
.exclude("save csv with compression codec option")
.exclude("save csv with empty fields with user defined empty values")
.exclude("save csv with quote")
.exclude("SPARK-13543 Write the output as uncompressed via option()")
.exclude("DDL test with tab separated file")
.exclude("DDL test parsing decimal type")
.exclude("test with tab delimiter and double quote")
.exclude("when mode is null, will fall back to PermissiveMode mode")
.exclude("SPARK-46890: CSV fails on a column with default and without enforcing schema")
// Arrow not support corrupt record
.exclude("SPARK-27873: disabling enforceSchema should not fail columnNameOfCorruptRecord")
// varchar
.exclude("SPARK-48241: CSV parsing failure with char/varchar type columns")
// Flaky and already excluded in other cases
.exclude("Gluten - test for FAILFAST parsing mode")

// enableSuite[GlutenCSVv2Suite]
// .exclude("Gluten - test for FAILFAST parsing mode")
// // Rule org.apache.spark.sql.execution.datasources.v2.V2ScanRelationPushDown in batch
// // Early Filter and Projection Push-Down generated an invalid plan
// .exclude("SPARK-26208: write and read empty data to csv file with headers")
// // file cars.csv include null string, Arrow not support to read
// .exclude("old csv data source name works")
// .exclude("DDL test with schema")
// .exclude("save csv")
// .exclude("save csv with compression codec option")
// .exclude("save csv with empty fields with user defined empty values")
// .exclude("save csv with quote")
// .exclude("SPARK-13543 Write the output as uncompressed via option()")
// .exclude("DDL test with tab separated file")
// .exclude("DDL test parsing decimal type")
// .exclude("test with tab delimiter and double quote")
// // Arrow not support corrupt record
// .exclude("SPARK-27873: disabling enforceSchema should not fail columnNameOfCorruptRecord")
// // varchar
// .exclude("SPARK-48241: CSV parsing failure with char/varchar type columns")
enableSuite[GlutenCSVv2Suite]
.exclude("Gluten - test for FAILFAST parsing mode")
// Rule org.apache.spark.sql.execution.datasources.v2.V2ScanRelationPushDown in batch
// Early Filter and Projection Push-Down generated an invalid plan
.exclude("SPARK-26208: write and read empty data to csv file with headers")
// file cars.csv include null string, Arrow not support to read
.exclude("old csv data source name works")
.exclude("DDL test with schema")
.exclude("save csv")
.exclude("save csv with compression codec option")
.exclude("save csv with empty fields with user defined empty values")
.exclude("save csv with quote")
.exclude("SPARK-13543 Write the output as uncompressed via option()")
.exclude("DDL test with tab separated file")
.exclude("DDL test parsing decimal type")
.exclude("test with tab delimiter and double quote")
.exclude("when mode is null, will fall back to PermissiveMode mode")
.exclude("SPARK-46890: CSV fails on a column with default and without enforcing schema")
// Arrow not support corrupt record
.exclude("SPARK-27873: disabling enforceSchema should not fail columnNameOfCorruptRecord")
// varchar
.exclude("SPARK-48241: CSV parsing failure with char/varchar type columns")

// enableSuite[GlutenCSVLegacyTimeParserSuite]
// // file cars.csv include null string, Arrow not support to read
// .exclude("DDL test with schema")
// .exclude("save csv")
// .exclude("save csv with compression codec option")
// .exclude("save csv with empty fields with user defined empty values")
// .exclude("save csv with quote")
// .exclude("SPARK-13543 Write the output as uncompressed via option()")
// // Arrow not support corrupt record
// .exclude("SPARK-27873: disabling enforceSchema should not fail columnNameOfCorruptRecord")
// .exclude("DDL test with tab separated file")
// .exclude("DDL test parsing decimal type")
// .exclude("test with tab delimiter and double quote")
// // varchar
// .exclude("SPARK-48241: CSV parsing failure with char/varchar type columns")
enableSuite[GlutenCSVLegacyTimeParserSuite]
// file cars.csv include null string, Arrow not support to read
.exclude("DDL test with schema")
.exclude("save csv")
.exclude("save csv with compression codec option")
.exclude("save csv with empty fields with user defined empty values")
.exclude("save csv with quote")
.exclude("SPARK-13543 Write the output as uncompressed via option()")
.exclude("when mode is null, will fall back to PermissiveMode mode")
.exclude("SPARK-46890: CSV fails on a column with default and without enforcing schema")
// Arrow not support corrupt record
.exclude("SPARK-27873: disabling enforceSchema should not fail columnNameOfCorruptRecord")
.exclude("DDL test with tab separated file")
.exclude("DDL test parsing decimal type")
.exclude("test with tab delimiter and double quote")
// varchar
.exclude("SPARK-48241: CSV parsing failure with char/varchar type columns")
enableSuite[GlutenJsonV1Suite]
// FIXME: Array direct selection fails
.exclude("Complex field and type inferring")
Expand Down Expand Up @@ -552,10 +558,9 @@ class VeloxTestSettings extends BackendTestSettings {
enableSuite[GlutenPathFilterStrategySuite]
enableSuite[GlutenPathFilterSuite]
enableSuite[GlutenPruneFileSourcePartitionsSuite]
// TODO: fix in Spark-4.0
// enableSuite[GlutenCSVReadSchemaSuite]
// enableSuite[GlutenHeaderCSVReadSchemaSuite]
// .exclude("change column type from int to long")
enableSuite[GlutenCSVReadSchemaSuite]
enableSuite[GlutenHeaderCSVReadSchemaSuite]
.exclude("change column type from int to long")
enableSuite[GlutenJsonReadSchemaSuite]
enableSuite[GlutenOrcReadSchemaSuite]
enableSuite[GlutenVectorizedOrcReadSchemaSuite]
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -132,4 +132,8 @@ class GlutenCSVLegacyTimeParserSuite extends GlutenCSVSuite {
override def sparkConf: SparkConf =
super.sparkConf
.set(SQLConf.LEGACY_TIME_PARSER_POLICY, "legacy")

// The source CSVLegacyTimeParserSuite exclude the test
override def excluded: Seq[String] =
Seq("Write timestamps correctly in ISO8601 format by default")
}