forked from apache/spark
-
Notifications
You must be signed in to change notification settings - Fork 0
Commit
This commit does not belong to any branch on this repository, and may belong to a fork outside of the repository.
[SPARK-16947][SQL] Support type coercion and foldable expression for …
…inline tables ## What changes were proposed in this pull request? This patch improves inline table support with the following: 1. Support type coercion. 2. Support using foldable expressions. Previously only literals were supported. 3. Improve error message handling. 4. Improve test coverage. ## How was this patch tested? Added a new unit test suite ResolveInlineTablesSuite and a new file-based end-to-end test inline-table.sql. Author: petermaxlee <[email protected]> Closes apache#14676 from petermaxlee/SPARK-16947.
- Loading branch information
1 parent
b72bb62
commit f5472dd
Showing
9 changed files
with
452 additions
and
46 deletions.
There are no files selected for viewing
This file contains bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
112 changes: 112 additions & 0 deletions
112
sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/analysis/ResolveInlineTables.scala
This file contains bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
Original file line number | Diff line number | Diff line change |
---|---|---|
@@ -0,0 +1,112 @@ | ||
/* | ||
* Licensed to the Apache Software Foundation (ASF) under one or more | ||
* contributor license agreements. See the NOTICE file distributed with | ||
* this work for additional information regarding copyright ownership. | ||
* The ASF licenses this file to You under the Apache License, Version 2.0 | ||
* (the "License"); you may not use this file except in compliance with | ||
* the License. You may obtain a copy of the License at | ||
* | ||
* http://www.apache.org/licenses/LICENSE-2.0 | ||
* | ||
* Unless required by applicable law or agreed to in writing, software | ||
* distributed under the License is distributed on an "AS IS" BASIS, | ||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. | ||
* See the License for the specific language governing permissions and | ||
* limitations under the License. | ||
*/ | ||
|
||
package org.apache.spark.sql.catalyst.analysis | ||
|
||
import scala.util.control.NonFatal | ||
|
||
import org.apache.spark.sql.catalyst.InternalRow | ||
import org.apache.spark.sql.catalyst.expressions.Cast | ||
import org.apache.spark.sql.catalyst.plans.logical.{LocalRelation, LogicalPlan} | ||
import org.apache.spark.sql.catalyst.rules.Rule | ||
import org.apache.spark.sql.types.{StructField, StructType} | ||
|
||
/** | ||
* An analyzer rule that replaces [[UnresolvedInlineTable]] with [[LocalRelation]]. | ||
*/ | ||
object ResolveInlineTables extends Rule[LogicalPlan] { | ||
override def apply(plan: LogicalPlan): LogicalPlan = plan transformUp { | ||
case table: UnresolvedInlineTable if table.expressionsResolved => | ||
validateInputDimension(table) | ||
validateInputEvaluable(table) | ||
convert(table) | ||
} | ||
|
||
/** | ||
* Validates the input data dimension: | ||
* 1. All rows have the same cardinality. | ||
* 2. The number of column aliases defined is consistent with the number of columns in data. | ||
* | ||
* This is package visible for unit testing. | ||
*/ | ||
private[analysis] def validateInputDimension(table: UnresolvedInlineTable): Unit = { | ||
if (table.rows.nonEmpty) { | ||
val numCols = table.names.size | ||
table.rows.zipWithIndex.foreach { case (row, ri) => | ||
if (row.size != numCols) { | ||
table.failAnalysis(s"expected $numCols columns but found ${row.size} columns in row $ri") | ||
} | ||
} | ||
} | ||
} | ||
|
||
/** | ||
* Validates that all inline table data are valid expressions that can be evaluated | ||
* (in this they must be foldable). | ||
* | ||
* This is package visible for unit testing. | ||
*/ | ||
private[analysis] def validateInputEvaluable(table: UnresolvedInlineTable): Unit = { | ||
table.rows.foreach { row => | ||
row.foreach { e => | ||
// Note that nondeterministic expressions are not supported since they are not foldable. | ||
if (!e.resolved || !e.foldable) { | ||
e.failAnalysis(s"cannot evaluate expression ${e.sql} in inline table definition") | ||
} | ||
} | ||
} | ||
} | ||
|
||
/** | ||
* Convert a valid (with right shape and foldable inputs) [[UnresolvedInlineTable]] | ||
* into a [[LocalRelation]]. | ||
* | ||
* This function attempts to coerce inputs into consistent types. | ||
* | ||
* This is package visible for unit testing. | ||
*/ | ||
private[analysis] def convert(table: UnresolvedInlineTable): LocalRelation = { | ||
// For each column, traverse all the values and find a common data type and nullability. | ||
val fields = table.rows.transpose.zip(table.names).map { case (column, name) => | ||
val inputTypes = column.map(_.dataType) | ||
val tpe = TypeCoercion.findWiderTypeWithoutStringPromotion(inputTypes).getOrElse { | ||
table.failAnalysis(s"incompatible types found in column $name for inline table") | ||
} | ||
StructField(name, tpe, nullable = column.exists(_.nullable)) | ||
} | ||
val attributes = StructType(fields).toAttributes | ||
assert(fields.size == table.names.size) | ||
|
||
val newRows: Seq[InternalRow] = table.rows.map { row => | ||
InternalRow.fromSeq(row.zipWithIndex.map { case (e, ci) => | ||
val targetType = fields(ci).dataType | ||
try { | ||
if (e.dataType.sameType(targetType)) { | ||
e.eval() | ||
} else { | ||
Cast(e, targetType).eval() | ||
} | ||
} catch { | ||
case NonFatal(ex) => | ||
table.failAnalysis(s"failed to evaluate expression ${e.sql}: ${ex.getMessage}") | ||
} | ||
}) | ||
} | ||
|
||
LocalRelation(attributes, newRows) | ||
} | ||
} |
This file contains bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
This file contains bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
This file contains bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
101 changes: 101 additions & 0 deletions
101
...lyst/src/test/scala/org/apache/spark/sql/catalyst/analysis/ResolveInlineTablesSuite.scala
This file contains bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
Original file line number | Diff line number | Diff line change |
---|---|---|
@@ -0,0 +1,101 @@ | ||
/* | ||
* Licensed to the Apache Software Foundation (ASF) under one or more | ||
* contributor license agreements. See the NOTICE file distributed with | ||
* this work for additional information regarding copyright ownership. | ||
* The ASF licenses this file to You under the Apache License, Version 2.0 | ||
* (the "License"); you may not use this file except in compliance with | ||
* the License. You may obtain a copy of the License at | ||
* | ||
* http://www.apache.org/licenses/LICENSE-2.0 | ||
* | ||
* Unless required by applicable law or agreed to in writing, software | ||
* distributed under the License is distributed on an "AS IS" BASIS, | ||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. | ||
* See the License for the specific language governing permissions and | ||
* limitations under the License. | ||
*/ | ||
|
||
package org.apache.spark.sql.catalyst.analysis | ||
|
||
import org.scalatest.BeforeAndAfter | ||
|
||
import org.apache.spark.sql.AnalysisException | ||
import org.apache.spark.sql.catalyst.expressions.{Literal, Rand} | ||
import org.apache.spark.sql.catalyst.expressions.aggregate.Count | ||
import org.apache.spark.sql.catalyst.plans.PlanTest | ||
import org.apache.spark.sql.types.{LongType, NullType} | ||
|
||
/** | ||
* Unit tests for [[ResolveInlineTables]]. Note that there are also test cases defined in | ||
* end-to-end tests (in sql/core module) for verifying the correct error messages are shown | ||
* in negative cases. | ||
*/ | ||
class ResolveInlineTablesSuite extends PlanTest with BeforeAndAfter { | ||
|
||
private def lit(v: Any): Literal = Literal(v) | ||
|
||
test("validate inputs are foldable") { | ||
ResolveInlineTables.validateInputEvaluable( | ||
UnresolvedInlineTable(Seq("c1", "c2"), Seq(Seq(lit(1))))) | ||
|
||
// nondeterministic (rand) should not work | ||
intercept[AnalysisException] { | ||
ResolveInlineTables.validateInputEvaluable( | ||
UnresolvedInlineTable(Seq("c1"), Seq(Seq(Rand(1))))) | ||
} | ||
|
||
// aggregate should not work | ||
intercept[AnalysisException] { | ||
ResolveInlineTables.validateInputEvaluable( | ||
UnresolvedInlineTable(Seq("c1"), Seq(Seq(Count(lit(1)))))) | ||
} | ||
|
||
// unresolved attribute should not work | ||
intercept[AnalysisException] { | ||
ResolveInlineTables.validateInputEvaluable( | ||
UnresolvedInlineTable(Seq("c1"), Seq(Seq(UnresolvedAttribute("A"))))) | ||
} | ||
} | ||
|
||
test("validate input dimensions") { | ||
ResolveInlineTables.validateInputDimension( | ||
UnresolvedInlineTable(Seq("c1"), Seq(Seq(lit(1)), Seq(lit(2))))) | ||
|
||
// num alias != data dimension | ||
intercept[AnalysisException] { | ||
ResolveInlineTables.validateInputDimension( | ||
UnresolvedInlineTable(Seq("c1", "c2"), Seq(Seq(lit(1)), Seq(lit(2))))) | ||
} | ||
|
||
// num alias == data dimension, but data themselves are inconsistent | ||
intercept[AnalysisException] { | ||
ResolveInlineTables.validateInputDimension( | ||
UnresolvedInlineTable(Seq("c1"), Seq(Seq(lit(1)), Seq(lit(21), lit(22))))) | ||
} | ||
} | ||
|
||
test("do not fire the rule if not all expressions are resolved") { | ||
val table = UnresolvedInlineTable(Seq("c1", "c2"), Seq(Seq(UnresolvedAttribute("A")))) | ||
assert(ResolveInlineTables(table) == table) | ||
} | ||
|
||
test("convert") { | ||
val table = UnresolvedInlineTable(Seq("c1"), Seq(Seq(lit(1)), Seq(lit(2L)))) | ||
val converted = ResolveInlineTables.convert(table) | ||
|
||
assert(converted.output.map(_.dataType) == Seq(LongType)) | ||
assert(converted.data.size == 2) | ||
assert(converted.data(0).getLong(0) == 1L) | ||
assert(converted.data(1).getLong(0) == 2L) | ||
} | ||
|
||
test("nullability inference in convert") { | ||
val table1 = UnresolvedInlineTable(Seq("c1"), Seq(Seq(lit(1)), Seq(lit(2L)))) | ||
val converted1 = ResolveInlineTables.convert(table1) | ||
assert(!converted1.schema.fields(0).nullable) | ||
|
||
val table2 = UnresolvedInlineTable(Seq("c1"), Seq(Seq(lit(1)), Seq(Literal(null, NullType)))) | ||
val converted2 = ResolveInlineTables.convert(table2) | ||
assert(converted2.schema.fields(0).nullable) | ||
} | ||
} |
This file contains bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
Oops, something went wrong.