diff --git a/.github/workflows/publish_snapshot-jdk17.yml b/.github/workflows/publish_snapshot-jdk17.yml
index 5835a4d14b39..27019ee519bf 100644
--- a/.github/workflows/publish_snapshot-jdk17.yml
+++ b/.github/workflows/publish_snapshot-jdk17.yml
@@ -63,8 +63,8 @@ jobs:
echo "$ASF_PASSWORD" >> $tmp_settings
echo "" >> $tmp_settings
- mvn --settings $tmp_settings -ntp clean install -Dgpg.skip -Drat.skip -DskipTests -Papache-release,spark4,flink1 -pl org.apache.paimon:paimon-spark-4.0_2.13,org.apache.paimon:paimon-spark-4.1_2.13 -am
+ mvn --settings $tmp_settings -ntp clean install -Dgpg.skip -Drat.skip -DskipTests -Papache-release,spark4,flink1 -pl org.apache.paimon:paimon-spark-4.0_2.13,org.apache.paimon:paimon-spark-4.1_2.13,org.apache.paimon:paimon-spark-4.2_2.13 -am
# skip deploy paimon-spark-common_2.13 since they are already deployed in publish-snapshot.yml
- mvn --settings $tmp_settings -ntp clean deploy -Dgpg.skip -Drat.skip -DskipTests -Papache-release,spark4,flink1 -pl org.apache.paimon:paimon-spark4-common_2.13,org.apache.paimon:paimon-spark-ut_2.13,org.apache.paimon:paimon-spark-4.0_2.13,org.apache.paimon:paimon-spark-4.1_2.13
+ mvn --settings $tmp_settings -ntp clean deploy -Dgpg.skip -Drat.skip -DskipTests -Papache-release,spark4,flink1 -pl org.apache.paimon:paimon-spark4-common_2.13,org.apache.paimon:paimon-spark-ut_2.13,org.apache.paimon:paimon-spark-4.0_2.13,org.apache.paimon:paimon-spark-4.1_2.13,org.apache.paimon:paimon-spark-4.2_2.13
rm $tmp_settings
diff --git a/.github/workflows/release-java.yml b/.github/workflows/release-java.yml
index 12160e9808d6..f5a11957cbbc 100644
--- a/.github/workflows/release-java.yml
+++ b/.github/workflows/release-java.yml
@@ -121,7 +121,7 @@ jobs:
mvn -ntp -B "${enforcer_goal}" \
-Denforcer.rules="${enforcer_rules}" \
-Papache-release,docs-and-source,spark4 \
- -pl org.apache.paimon:paimon-spark-common_2.13,org.apache.paimon:paimon-spark4-common_2.13,org.apache.paimon:paimon-spark-4.0_2.13,org.apache.paimon:paimon-spark-4.1_2.13 \
+ -pl org.apache.paimon:paimon-spark-common_2.13,org.apache.paimon:paimon-spark4-common_2.13,org.apache.paimon:paimon-spark-4.0_2.13,org.apache.paimon:paimon-spark-4.1_2.13,org.apache.paimon:paimon-spark-4.2_2.13 \
-am
;;
*)
@@ -186,14 +186,14 @@ jobs:
jdk17)
capture_expected_projects \
-Papache-release,docs-and-source,spark4 \
- -pl org.apache.paimon:paimon-spark-common_2.13,org.apache.paimon:paimon-spark4-common_2.13,org.apache.paimon:paimon-spark-4.0_2.13,org.apache.paimon:paimon-spark-4.1_2.13
+ -pl org.apache.paimon:paimon-spark-common_2.13,org.apache.paimon:paimon-spark4-common_2.13,org.apache.paimon:paimon-spark-4.0_2.13,org.apache.paimon:paimon-spark-4.1_2.13,org.apache.paimon:paimon-spark-4.2_2.13
# This install only supplies reactor dependencies for the deploy
# below. Their Javadocs are built by the owning release lane.
mvn clean install -ntp -B \
-Pdocs-and-source,spark4 \
-DskipTests -Dmaven.javadoc.skip=true \
-Dstyle.color=never \
- -pl paimon-spark/paimon-spark-4.0,paimon-spark/paimon-spark-4.1 \
+ -pl paimon-spark/paimon-spark-4.0,paimon-spark/paimon-spark-4.1,paimon-spark/paimon-spark-4.2 \
-am \
2>&1 | tee "${log}"
mvn deploy -ntp -B \
@@ -201,7 +201,7 @@ jobs:
-DskipTests -Dgpg.skip=true -Dstyle.color=never \
-DdeployAtEnd=true \
-DaltDeploymentRepository="${alt_repository}" \
- -pl org.apache.paimon:paimon-spark-common_2.13,org.apache.paimon:paimon-spark4-common_2.13,org.apache.paimon:paimon-spark-4.0_2.13,org.apache.paimon:paimon-spark-4.1_2.13 \
+ -pl org.apache.paimon:paimon-spark-common_2.13,org.apache.paimon:paimon-spark4-common_2.13,org.apache.paimon:paimon-spark-4.0_2.13,org.apache.paimon:paimon-spark-4.1_2.13,org.apache.paimon:paimon-spark-4.2_2.13 \
2>&1 | tee -a "${log}"
;;
*)
diff --git a/.github/workflows/utitcase-spark-4.x.yml b/.github/workflows/utitcase-spark-4.x.yml
index 72c43c018e50..4af6a6254ea8 100644
--- a/.github/workflows/utitcase-spark-4.x.yml
+++ b/.github/workflows/utitcase-spark-4.x.yml
@@ -67,7 +67,7 @@ jobs:
jvm_timezone=$(random_timezone)
echo "JVM timezone is set to $jvm_timezone"
test_modules=""
- for suffix in ut 4.0 4.1; do
+ for suffix in ut 4.0 4.1 4.2; do
test_modules+="org.apache.paimon:paimon-spark-${suffix}_2.13,"
done
test_modules="${test_modules%,}"
diff --git a/docs/docs/ecosystem/index.md b/docs/docs/ecosystem/index.md
index be5017b1f6c2..0091011934fd 100644
--- a/docs/docs/ecosystem/index.md
+++ b/docs/docs/ecosystem/index.md
@@ -29,7 +29,7 @@ under the License.
| Engine | Version | Batch Read | Batch Write | Create Table | Alter Table | Streaming Write | Streaming Read | Batch Overwrite | DELETE & UPDATE | MERGE INTO | Time Travel |
|:-------------------------------------------------------------------------------:|:-------------:|:-----------:|:-----------:|:-------------:|:-------------:|:----------------:|:----------------:|:---------------:|:---------------:|:----------:|:-----------:|
| Flink | 1.16 - 1.20 | ✅ | ✅ | ✅ | ✅(1.17+) | ✅ | ✅ | ✅ | ✅(1.17+) | ❌ | ✅ |
-| Spark | 3.2 - 4.1 | ✅ | ✅ | ✅ | ✅ | ✅(3.3+) | ✅(3.3+) | ✅ | ✅ | ✅ | ✅(3.3+) |
+| Spark | 3.2 - 4.2 | ✅ | ✅ | ✅ | ✅ | ✅(3.3+) | ✅(3.3+) | ✅ | ✅ | ✅ | ✅(3.3+) |
| Hive | 2.1 - 3.1 | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ |
| Trino | 420 - 440 | ✅ | ✅(427+) | ✅(427+) | ✅(427+) | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ |
| Presto | 0.236 - 0.280 | ✅ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
diff --git a/docs/docs/project/download.mdx b/docs/docs/project/download.mdx
index b361f7b84c6e..072bfc6a09b5 100644
--- a/docs/docs/project/download.mdx
+++ b/docs/docs/project/download.mdx
@@ -44,6 +44,7 @@ This documentation is a guide for downloading Paimon Jars.
| Flink 1.17 | [paimon-flink-1.17-@@VERSION@@.jar](https://repository.apache.org/snapshots/org/apache/paimon/paimon-flink-1.17/@@VERSION@@/) |
| Flink 1.16 | [paimon-flink-1.16-@@VERSION@@.jar](https://repository.apache.org/snapshots/org/apache/paimon/paimon-flink-1.16/@@VERSION@@/) |
| Flink Action | [paimon-flink-action-@@VERSION@@.jar](https://repository.apache.org/snapshots/org/apache/paimon/paimon-flink-action/@@VERSION@@/) |
+| Spark 4.2 | [paimon-spark-4.2_2.13-@@VERSION@@.jar](https://repository.apache.org/snapshots/org/apache/paimon/paimon-spark-4.2_2.13/@@VERSION@@/) |
| Spark 4.1 | [paimon-spark-4.1_2.13-@@VERSION@@.jar](https://repository.apache.org/snapshots/org/apache/paimon/paimon-spark-4.1_2.13/@@VERSION@@/) |
| Spark 4.0 | [paimon-spark-4.0_2.13-@@VERSION@@.jar](https://repository.apache.org/snapshots/org/apache/paimon/paimon-spark-4.0_2.13/@@VERSION@@/) |
| Spark 3.5 | [paimon-spark-3.5_2.12-@@VERSION@@.jar](https://repository.apache.org/snapshots/org/apache/paimon/paimon-spark-3.5_2.12/@@VERSION@@/) |
@@ -76,6 +77,7 @@ This documentation is a guide for downloading Paimon Jars.
| Flink 1.17 | [paimon-flink-1.17-@@VERSION@@.jar](https://repo.maven.apache.org/maven2/org/apache/paimon/paimon-flink-1.17/@@VERSION@@/paimon-flink-1.17-@@VERSION@@.jar) |
| Flink 1.16 | [paimon-flink-1.16-@@VERSION@@.jar](https://repo.maven.apache.org/maven2/org/apache/paimon/paimon-flink-1.16/@@VERSION@@/paimon-flink-1.16-@@VERSION@@.jar) |
| Flink Action | [paimon-flink-action-@@VERSION@@.jar](https://repo.maven.apache.org/maven2/org/apache/paimon/paimon-flink-action/@@VERSION@@/paimon-flink-action-@@VERSION@@.jar) |
+| Spark 4.2 | [paimon-spark-4.2_2.13-@@VERSION@@.jar](https://repo.maven.apache.org/maven2/org/apache/paimon/paimon-spark-4.2_2.13/@@VERSION@@/paimon-spark-4.2_2.13-@@VERSION@@.jar) |
| Spark 4.1 | [paimon-spark-4.1_2.13-@@VERSION@@.jar](https://repo.maven.apache.org/maven2/org/apache/paimon/paimon-spark-4.1_2.13/@@VERSION@@/paimon-spark-4.1_2.13-@@VERSION@@.jar) |
| Spark 4.0 | [paimon-spark-4.0_2.13-@@VERSION@@.jar](https://repo.maven.apache.org/maven2/org/apache/paimon/paimon-spark-4.0_2.13/@@VERSION@@/paimon-spark-4.0_2.13-@@VERSION@@.jar) |
| Spark 3.5 | [paimon-spark-3.5_2.12-@@VERSION@@.jar](https://repo.maven.apache.org/maven2/org/apache/paimon/paimon-spark-3.5_2.12/@@VERSION@@/paimon-spark-3.5_2.12-@@VERSION@@.jar) |
diff --git a/docs/docs/project/verifying-a-release-candidate.md b/docs/docs/project/verifying-a-release-candidate.md
index d2ea99b6b36e..7952e0862550 100644
--- a/docs/docs/project/verifying-a-release-candidate.md
+++ b/docs/docs/project/verifying-a-release-candidate.md
@@ -241,7 +241,7 @@ the packaging-equivalent build without reproducing the full JDK 17 test lane:
(
cd "paimon-${PAIMON_VERSION}"
mvn -ntp clean install -DskipTests -Pdocs-and-source,spark4 \
- -pl paimon-spark/paimon-spark-4.0,paimon-spark/paimon-spark-4.1 \
+ -pl paimon-spark/paimon-spark-4.0,paimon-spark/paimon-spark-4.1,paimon-spark/paimon-spark-4.2 \
-am
)
```
diff --git a/docs/docs/spark/quick-start.mdx b/docs/docs/spark/quick-start.mdx
index 7a7432a63f17..71dfb3e88401 100644
--- a/docs/docs/spark/quick-start.mdx
+++ b/docs/docs/spark/quick-start.mdx
@@ -33,7 +33,7 @@ under the License.
Paimon supports the following Spark versions with their respective Java and Scala compatibility. We recommend using the latest Spark version for a better experience.
-- Spark 4.x (including 4.1, 4.0) : Pre-built with Java 17 and Scala 2.13
+- Spark 4.x (including 4.2, 4.1, 4.0) : Pre-built with Java 17 and Scala 2.13
- Spark 3.x (including 3.5, 3.4, 3.3, 3.2) : Pre-built with Java 8 and Scala 2.12/2.13
@@ -45,6 +45,7 @@ Download the jar file with corresponding version.
| Version | Jar (Scala 2.12) | Jar (Scala 2.13) |
|-----------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
+| Spark 4.2 | - | [paimon-spark-4.2_2.13-@@VERSION@@.jar](https://repo.maven.apache.org/maven2/org/apache/paimon/paimon-spark-4.2_2.13/@@VERSION@@/paimon-spark-4.2_2.13-@@VERSION@@.jar) |
| Spark 4.1 | - | [paimon-spark-4.1_2.13-@@VERSION@@.jar](https://repo.maven.apache.org/maven2/org/apache/paimon/paimon-spark-4.1_2.13/@@VERSION@@/paimon-spark-4.1_2.13-@@VERSION@@.jar) |
| Spark 4.0 | - | [paimon-spark-4.0_2.13-@@VERSION@@.jar](https://repo.maven.apache.org/maven2/org/apache/paimon/paimon-spark-4.0_2.13/@@VERSION@@/paimon-spark-4.0_2.13-@@VERSION@@.jar) |
| Spark 3.5 | [paimon-spark-3.5_2.12-@@VERSION@@.jar](https://repo.maven.apache.org/maven2/org/apache/paimon/paimon-spark-3.5_2.12/@@VERSION@@/paimon-spark-3.5_2.12-@@VERSION@@.jar) | [paimon-spark-3.5_2.13-@@VERSION@@.jar](https://repo.maven.apache.org/maven2/org/apache/paimon/paimon-spark-3.5_2.13/@@VERSION@@/paimon-spark-3.5_2.13-@@VERSION@@.jar) |
@@ -62,6 +63,7 @@ Download the jar file with corresponding version.
| Version | Jar (Scala 2.12) | Jar (Scala 2.13) |
|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------|
+| Spark 4.2 | - | [paimon-spark-4.2_2.13-@@VERSION@@.jar](https://repository.apache.org/snapshots/org/apache/paimon/paimon-spark-4.2_2.13/@@VERSION@@/) |
| Spark 4.1 | - | [paimon-spark-4.1_2.13-@@VERSION@@.jar](https://repository.apache.org/snapshots/org/apache/paimon/paimon-spark-4.1_2.13/@@VERSION@@/) |
| Spark 4.0 | - | [paimon-spark-4.0_2.13-@@VERSION@@.jar](https://repository.apache.org/snapshots/org/apache/paimon/paimon-spark-4.0_2.13/@@VERSION@@/) |
| Spark 3.5 | [paimon-spark-3.5_2.12-@@VERSION@@.jar](https://repository.apache.org/snapshots/org/apache/paimon/paimon-spark-3.5_2.12/@@VERSION@@/) | [paimon-spark-3.5_2.13-@@VERSION@@.jar](https://repository.apache.org/snapshots/org/apache/paimon/paimon-spark-3.5_2.13/@@VERSION@@/) |
@@ -89,6 +91,9 @@ mvn clean package -DskipTests -pl paimon-spark/paimon-spark-4.0 -am -Pspark4
# build paimon spark 4.1
mvn clean package -DskipTests -pl paimon-spark/paimon-spark-4.1 -am -Pspark4
+
+# build paimon spark 4.2
+mvn clean package -DskipTests -pl paimon-spark/paimon-spark-4.2 -am -Pspark4
```
For Spark 3.5, you can find the bundled jar in `./paimon-spark/paimon-spark-3.5/target/paimon-spark-3.5_2.12-@@VERSION@@.jar`.
diff --git a/paimon-spark/paimon-spark-4.0/pom.xml b/paimon-spark/paimon-spark-4.0/pom.xml
index a11261a7c9ee..c45574184858 100644
--- a/paimon-spark/paimon-spark-4.0/pom.xml
+++ b/paimon-spark/paimon-spark-4.0/pom.xml
@@ -124,9 +124,13 @@ under the License.
+
org.apache.paimon
- paimon-spark-ut_${scala.binary.version}
+ paimon-spark-ut-4.0_${scala.binary.version}${project.version}teststest
diff --git a/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/catalog/functions/SQLFunctionConverter.scala b/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/catalog/functions/SQLFunctionConverter.scala
new file mode 100644
index 000000000000..4aee17aa2ea9
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/catalog/functions/SQLFunctionConverter.scala
@@ -0,0 +1,167 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.catalog.functions
+
+import org.apache.paimon.function.{Function => PaimonFunction, FunctionDefinition, FunctionImpl}
+import org.apache.paimon.spark.SparkCatalog.FUNCTION_DEFINITION_NAME
+import org.apache.paimon.spark.SparkTypeUtils
+import org.apache.paimon.types.{DataField, RowType}
+
+import org.apache.spark.sql.catalyst.FunctionIdentifier
+import org.apache.spark.sql.catalyst.analysis.SQLFunctionExpression
+import org.apache.spark.sql.catalyst.catalog.{SQLFunction, UserDefinedFunction}
+import org.apache.spark.sql.catalyst.expressions.Expression
+import org.apache.spark.sql.catalyst.parser.ParserInterface
+import org.apache.spark.sql.types.{DataType => SparkDataType, StructType}
+
+import java.util.{Collections, HashMap => JHashMap, List => JList}
+
+import scala.collection.JavaConverters._
+
+/** Converts between Spark SQLFunction and Paimon Function with a SQLFunctionDefinition body. */
+object SQLFunctionConverter {
+
+ // Paimon-specific option keys (prefixed to avoid collision with Spark properties).
+ private val PAIMON_OPTION_PREFIX = "spark.sql-function."
+ private val IS_QUERY = PAIMON_OPTION_PREFIX + "is-query"
+ private val DETERMINISTIC = PAIMON_OPTION_PREFIX + "deterministic"
+ private val CONTAINS_SQL = PAIMON_OPTION_PREFIX + "contains-sql"
+
+ /** Build a Paimon function from a parsed CREATE FUNCTION ... RETURN statement. */
+ def toPaimonFunction(
+ funcIdent: FunctionIdentifier,
+ inputParamText: Option[String],
+ returnTypeText: String,
+ exprText: Option[String],
+ queryText: Option[String],
+ comment: Option[String],
+ isDeterministic: Option[Boolean],
+ containsSQL: Option[Boolean],
+ parser: ParserInterface,
+ properties: Map[String, String] = Map.empty): PaimonFunction = {
+ require(
+ returnTypeText != null && returnTypeText.trim.nonEmpty,
+ s"SQL function $funcIdent must have a return type (explicit or inferred).")
+ val identifier = FunctionIdentifierConverter.toPaimonIdentifier(funcIdent)
+
+ val inputParams: JList[DataField] = inputParamText.filter(_.trim.nonEmpty) match {
+ case Some(text) =>
+ SparkTypeUtils
+ .toPaimonRowType(UserDefinedFunction.parseRoutineParam(text, parser))
+ .getFields
+ case None => Collections.emptyList[DataField]()
+ }
+ val returnSparkType = parseScalarReturnType(funcIdent, returnTypeText, parser)
+ val returnParams: JList[DataField] =
+ Collections.singletonList(
+ new DataField(0, funcIdent.funcName, SparkTypeUtils.toPaimonType(returnSparkType)))
+
+ // Exactly one of exprText / queryText is set by the parser.
+ val isQuery = exprText.isEmpty && queryText.isDefined
+ val body = exprText
+ .orElse(queryText)
+ .getOrElse(throw new IllegalArgumentException(s"SQL function $funcIdent has an empty body."))
+
+ val options = new JHashMap[String, String]()
+ options.put(IS_QUERY, isQuery.toString)
+ isDeterministic.foreach(d => options.put(DETERMINISTIC, d.toString))
+ containsSQL.foreach(c => options.put(CONTAINS_SQL, c.toString))
+ properties.foreach { case (k, v) => options.put(k, v) }
+
+ new FunctionImpl(
+ identifier,
+ inputParams,
+ returnParams,
+ isDeterministic.getOrElse(true), // caller should always pass Some after analysis
+ Collections.singletonMap(FUNCTION_DEFINITION_NAME, FunctionDefinition.sql(body)),
+ comment.orNull,
+ options
+ )
+ }
+
+ /** Resolve a Paimon-stored SQL function into a Spark SQLFunctionExpression. */
+ def toSQLFunctionExpression(
+ funcIdent: FunctionIdentifier,
+ function: PaimonFunction,
+ arguments: Seq[Expression],
+ parser: ParserInterface): Expression = {
+ val options = function.options()
+
+ val body = function.definition(FUNCTION_DEFINITION_NAME) match {
+ case sql: FunctionDefinition.SQLFunctionDefinition => sql.definition()
+ case other =>
+ throw new IllegalStateException(
+ s"Function $funcIdent is not a SQL function, found definition: $other")
+ }
+
+ val inputParam: Option[StructType] = {
+ val ip = function.inputParams()
+ if (ip.isPresent && !ip.get().isEmpty) {
+ Some(SparkTypeUtils.fromPaimonType(new RowType(ip.get())).asInstanceOf[StructType])
+ } else None
+ }
+
+ val rp = function.returnParams()
+ require(
+ rp.isPresent && !rp.get().isEmpty,
+ s"SQL function $funcIdent has no return type in returnParams.")
+ val returnType: SparkDataType = SparkTypeUtils.fromPaimonType(rp.get().get(0).`type`())
+
+ val isQuery = Option(options.get(IS_QUERY))
+ .map(java.lang.Boolean.parseBoolean)
+ .getOrElse {
+ try { parser.parseExpression(body); false }
+ catch { case _: Exception => true }
+ }
+
+ val deterministic = Option(options.get(DETERMINISTIC))
+ .map(_.toBoolean)
+ .orElse(Some(function.isDeterministic))
+
+ val sqlFunction = SQLFunction(
+ name = funcIdent,
+ inputParam = inputParam,
+ returnType = Left(returnType),
+ exprText = if (isQuery) None else Some(body),
+ queryText = if (isQuery) Some(body) else None,
+ comment = Option(function.comment()),
+ deterministic = deterministic,
+ containsSQL = Option(options.get(CONTAINS_SQL)).map(_.toBoolean),
+ isTableFunc = false,
+ properties = options.asScala.filterNot(_._1.startsWith(PAIMON_OPTION_PREFIX)).toMap
+ )
+
+ SQLFunctionExpression(
+ sqlFunction.name.unquotedString,
+ sqlFunction,
+ arguments,
+ Some(sqlFunction.getScalarFuncReturnType))
+ }
+
+ private def parseScalarReturnType(
+ funcIdent: FunctionIdentifier,
+ returnTypeText: String,
+ parser: ParserInterface): SparkDataType =
+ SQLFunction.parseReturnTypeText(returnTypeText, isTableFunc = false, parser) match {
+ case Some(Left(dataType)) => dataType
+ case _ =>
+ throw new UnsupportedOperationException(
+ s"Unsupported return type '$returnTypeText' for scalar SQL function $funcIdent.")
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/MergePaimonScalarSubqueries.scala b/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/MergePaimonScalarSubqueries.scala
new file mode 100644
index 000000000000..dc9fd53d98a3
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/MergePaimonScalarSubqueries.scala
@@ -0,0 +1,90 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.catalyst.optimizer
+
+import org.apache.paimon.spark.PaimonScan
+
+import org.apache.spark.sql.catalyst.expressions.{Attribute, AttributeMap, AttributeReference, ExprId, ScalarSubquery, SortOrder}
+import org.apache.spark.sql.catalyst.plans.logical.LogicalPlan
+import org.apache.spark.sql.execution.datasources.v2.DataSourceV2ScanRelation
+
+object MergePaimonScalarSubqueries extends MergePaimonScalarSubqueriesBase {
+
+ override def tryMergeDataSourceV2ScanRelation(
+ newV2ScanRelation: DataSourceV2ScanRelation,
+ cachedV2ScanRelation: DataSourceV2ScanRelation)
+ : Option[(LogicalPlan, AttributeMap[Attribute])] = {
+ // Match by type and read fields through named accessors: Spark 4.2 (SPARK-56385) added a
+ // sixth `pushedFilters` parameter, which breaks positional patterns.
+ (newV2ScanRelation.scan, cachedV2ScanRelation.scan) match {
+ case (newScan: PaimonScan, cachedScan: PaimonScan) =>
+ val newRelation = newV2ScanRelation.relation
+ val newOutput = newV2ScanRelation.output
+ val newPartitioning = newV2ScanRelation.keyGroupedPartitioning
+ val newOrdering = newV2ScanRelation.ordering
+ val cachedRelation = cachedV2ScanRelation.relation
+ val cachedPartitioning = cachedV2ScanRelation.keyGroupedPartitioning
+ val cacheOrdering = cachedV2ScanRelation.ordering
+
+ checkIdenticalPlans(newRelation, cachedRelation).flatMap {
+ outputMap =>
+ if (
+ samePartitioning(newPartitioning, cachedPartitioning, outputMap) && sameOrdering(
+ newOrdering,
+ cacheOrdering,
+ outputMap)
+ ) {
+ mergePaimonScan(newScan, cachedScan).map {
+ mergedScan =>
+ val mergedAttributes = mergedScan
+ .readSchema()
+ .map(f => AttributeReference(f.name, f.dataType, f.nullable, f.metadata)())
+ val cachedOutputNameMap = cachedRelation.output.map(a => a.name -> a).toMap
+ val mergedOutput =
+ mergedAttributes.map(a => cachedOutputNameMap.getOrElse(a.name, a))
+ val mergedV2ScanRelation =
+ cachedV2ScanRelation.copy(scan = mergedScan, output = mergedOutput)
+
+ val mergedOutputNameMap = mergedOutput.map(a => a.name -> a).toMap
+ val newOutputMap =
+ AttributeMap(newOutput.map(a => a -> mergedOutputNameMap(a.name).toAttribute))
+
+ mergedV2ScanRelation -> newOutputMap
+ }
+ } else {
+ None
+ }
+ }
+
+ case _ => None
+ }
+ }
+
+ private def sameOrdering(
+ newOrdering: Option[Seq[SortOrder]],
+ cachedOrdering: Option[Seq[SortOrder]],
+ outputAttrMap: AttributeMap[Attribute]): Boolean = {
+ val mappedNewOrdering = newOrdering.map(_.map(mapAttributes(_, outputAttrMap)))
+ mappedNewOrdering.map(_.map(_.canonicalized)) == cachedOrdering.map(_.map(_.canonicalized))
+ }
+
+ override protected def createScalarSubquery(plan: LogicalPlan, exprId: ExprId): ScalarSubquery = {
+ ScalarSubquery(plan, exprId = exprId)
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/PushDownMapSelectedKeys.scala b/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/PushDownMapSelectedKeys.scala
new file mode 100644
index 000000000000..88e6347cee2e
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/PushDownMapSelectedKeys.scala
@@ -0,0 +1,39 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.catalyst.optimizer
+
+import org.apache.paimon.spark.PaimonScan
+
+import org.apache.spark.sql.catalyst.expressions.AttributeReference
+import org.apache.spark.sql.execution.datasources.v2.DataSourceV2ScanRelation
+
+object PushDownMapSelectedKeys extends PushDownMapSelectedKeysBase {
+
+ override protected def copyDataSourceV2ScanRelation(
+ relation: DataSourceV2ScanRelation,
+ scan: PaimonScan,
+ output: Seq[AttributeReference]): DataSourceV2ScanRelation = {
+ DataSourceV2ScanRelation(
+ relation.relation,
+ scan,
+ output,
+ relation.keyGroupedPartitioning,
+ relation.ordering)
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/commands/PaimonDynamicPartitionOverwriteCommand.scala b/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/commands/PaimonDynamicPartitionOverwriteCommand.scala
new file mode 100644
index 000000000000..f86593ff2379
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/commands/PaimonDynamicPartitionOverwriteCommand.scala
@@ -0,0 +1,98 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.commands
+
+import org.apache.paimon.options.Options
+import org.apache.paimon.spark.DynamicOverWrite
+import org.apache.paimon.table.FileStoreTable
+
+import org.apache.spark.sql.{Row, SparkSession}
+import org.apache.spark.sql.PaimonUtils.{createDataset, createNewDataFrame}
+import org.apache.spark.sql.catalyst.analysis.NamedRelation
+import org.apache.spark.sql.catalyst.plans.logical.{Command, LogicalPlan, V2WriteCommand}
+import org.apache.spark.sql.connector.catalog.TableWritePrivilege
+import org.apache.spark.sql.execution.command.RunnableCommand
+
+import scala.collection.convert.ImplicitConversions._
+
+/**
+ * Spark 4.0/4.1 flavour of the class with the same FQCN in `paimon-spark-common`, which is compiled
+ * against the profile baseline (Spark 4.2).
+ *
+ * The body is identical; only the compilation target differs. Spark 4.2 mixes the new
+ * `WriteWithSchemaEvolution` trait into `V2WriteCommand`, so a 4.2-compiled subclass calls that
+ * trait's `$init$` from its constructor. The trait does not exist before 4.2, so on a 4.0/4.1
+ * runtime every instantiation dies with `NoClassDefFoundError` — `PaimonAnalysis` builds one for
+ * any dynamic partition overwrite, making it unreachable to write such a table at all.
+ *
+ * This cannot be fixed with a `SparkShim` method: the incompatibility is in the supertype, not in a
+ * call. Dropping `extends V2WriteCommand` is not an option either — Spark matches on that type to
+ * transform the plan, which is the whole reason this class exists.
+ *
+ * A `RunnableCommand` that will execute dynamic partition overwrite using [[WriteIntoPaimonTable]].
+ *
+ * This is a workaround of Spark not supporting V1 fallback for dynamic partition overwrite. Note
+ * the following details:
+ * - Extends [[V2WriteCommand]] so that Spark can transform this plan.
+ * - Exposes the query as a child so that the Spark optimizer can optimize it.
+ */
+case class PaimonDynamicPartitionOverwriteCommand(
+ table: NamedRelation,
+ fileStoreTable: FileStoreTable,
+ query: LogicalPlan,
+ writeOptions: Map[String, String],
+ isByName: Boolean)
+ extends RunnableCommand
+ with V2WriteCommand {
+
+ override def child: LogicalPlan = query
+
+ override def withNewQuery(newQuery: LogicalPlan): PaimonDynamicPartitionOverwriteCommand = {
+ copy(query = newQuery)
+ }
+
+ override def withNewTable(newTable: NamedRelation): PaimonDynamicPartitionOverwriteCommand = {
+ copy(table = newTable)
+ }
+
+ override protected def withNewChildInternal(
+ newChild: LogicalPlan): PaimonDynamicPartitionOverwriteCommand = copy(query = newChild)
+
+ // Declared without `override` so this source stays identical to the `paimon-spark-common` copy,
+ // which must also compile on 4.2 where `WriteWithSchemaEvolution` declares these members.
+ // Dynamic partition overwrite never evolves the target schema.
+ def withSchemaEvolution: Boolean = false
+
+ // Dynamic partition overwrite replaces whole partitions, so it deletes as well as inserts —
+ // the same privilege set Spark's own `OverwritePartitionsDynamic` requests.
+ def writePrivileges: Set[TableWritePrivilege] =
+ Set(TableWritePrivilege.INSERT, TableWritePrivilege.DELETE)
+
+ override def run(sparkSession: SparkSession): Seq[Row] = {
+ WriteIntoPaimonTable(
+ fileStoreTable,
+ DynamicOverWrite,
+ createNewDataFrame(createDataset(sparkSession, query)),
+ Options.fromMap(fileStoreTable.options() ++ writeOptions)
+ ).run(sparkSession)
+ }
+
+ // Do not annotate with override here to maintain compatibility with Spark 3.3-.
+ def storeAnalyzedQuery(): Command = copy(query = query)
+}
diff --git a/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/execution/adaptive/DisableUnnecessaryPaimonBucketedScan.scala b/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/execution/adaptive/DisableUnnecessaryPaimonBucketedScan.scala
new file mode 100644
index 000000000000..b0101ded21bc
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/paimon/spark/execution/adaptive/DisableUnnecessaryPaimonBucketedScan.scala
@@ -0,0 +1,178 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.execution.adaptive
+
+import org.apache.paimon.spark.PaimonScan
+
+import org.apache.spark.sql.catalyst.plans.physical.{AllTuples, ClusteredDistribution}
+import org.apache.spark.sql.catalyst.rules.Rule
+import org.apache.spark.sql.execution._
+import org.apache.spark.sql.execution.aggregate.BaseAggregateExec
+import org.apache.spark.sql.execution.datasources.v2.BatchScanExec
+import org.apache.spark.sql.execution.exchange.{Exchange, ShuffleExchangeLike}
+
+// spotless:off
+/**
+ * This rule is inspired from Spark [[DisableUnnecessaryBucketedScan]] but work for v2 scan.
+ *
+ * Disable unnecessary bucketed table scan based on actual physical query plan.
+ * NOTE: this rule is designed to be applied right after [[EnsureRequirements]],
+ * where all [[ShuffleExchangeLike]] and [[SortExec]] have been added to plan properly.
+ *
+ * When BUCKETING_ENABLED and AUTO_BUCKETED_SCAN_ENABLED are set to true, go through
+ * query plan to check where bucketed table scan is unnecessary, and disable bucketed table
+ * scan if:
+ *
+ * 1. The sub-plan from root to bucketed table scan, does not contain
+ * [[hasInterestingPartitionOrOrder]] operator.
+ *
+ * 2. The sub-plan from the nearest downstream [[hasInterestingPartitionOrOrder]] operator
+ * to the bucketed table scan and at least one [[ShuffleExchangeLike]].
+ *
+ * Examples:
+ * 1. no [[hasInterestingPartitionOrOrder]] operator:
+ * Project
+ * |
+ * Filter
+ * |
+ * Scan(t1: i, j)
+ * (bucketed on column j, DISABLE bucketed scan)
+ *
+ * 2. join:
+ * SortMergeJoin(t1.i = t2.j)
+ * / \
+ * Sort(i) Sort(j)
+ * / \
+ * Shuffle(i) Scan(t2: i, j)
+ * / (bucketed on column j, enable bucketed scan)
+ * Scan(t1: i, j)
+ * (bucketed on column j, DISABLE bucketed scan)
+ *
+ * 3. aggregate:
+ * HashAggregate(i, ..., Final)
+ * |
+ * Shuffle(i)
+ * |
+ * HashAggregate(i, ..., Partial)
+ * |
+ * Filter
+ * |
+ * Scan(t1: i, j)
+ * (bucketed on column j, DISABLE bucketed scan)
+ *
+ * The idea of [[hasInterestingPartitionOrOrder]] is inspired from "interesting order" in
+ * the paper "Access Path Selection in a Relational Database Management System"
+ * (https://dl.acm.org/doi/10.1145/582095.582099).
+ */
+// spotless:on
+object DisableUnnecessaryPaimonBucketedScan extends Rule[SparkPlan] {
+
+ /**
+ * Disable bucketed table scan with pre-order traversal of plan.
+ *
+ * @param hashInterestingPartitionOrOrder
+ * The traversed plan has operator with interesting partition and order.
+ * @param hasExchange
+ * The traversed plan has [[Exchange]] operator.
+ */
+ private def disableBucketScan(
+ plan: SparkPlan,
+ hashInterestingPartitionOrOrder: Boolean,
+ hasExchange: Boolean): SparkPlan = {
+ plan match {
+ case p if hasInterestingPartitionOrOrder(p) =>
+ // Operator with interesting partition, propagates `hashInterestingPartitionOrOrder` as true
+ // to its children, and resets `hasExchange`.
+ p.mapChildren(
+ disableBucketScan(_, hashInterestingPartitionOrOrder = true, hasExchange = false))
+ case exchange: ShuffleExchangeLike =>
+ // Exchange operator propagates `hasExchange` as true to its child.
+ exchange.mapChildren(
+ disableBucketScan(_, hashInterestingPartitionOrOrder, hasExchange = true))
+ case batch: BatchScanExec =>
+ val paimonBucketedScan = extractPaimonBucketedScan(batch)
+ if (paimonBucketedScan.isDefined && (!hashInterestingPartitionOrOrder || hasExchange)) {
+ val (batch, paimonScan) = paimonBucketedScan.get
+ val newBatch = batch.copy(scan = paimonScan.disableBucketedScan())
+ newBatch.copyTagsFrom(batch)
+ newBatch
+ } else {
+ batch
+ }
+ case p if canPassThrough(p) =>
+ p.mapChildren(disableBucketScan(_, hashInterestingPartitionOrOrder, hasExchange))
+ case other =>
+ other.mapChildren(
+ disableBucketScan(_, hashInterestingPartitionOrOrder = false, hasExchange = false))
+ }
+ }
+
+ private def hasInterestingPartitionOrOrder(plan: SparkPlan): Boolean = {
+ val hashPartition = plan.requiredChildDistribution.exists {
+ case _: ClusteredDistribution | AllTuples => true
+ case _ => false
+ }
+ // Some operators may only require local sort without distribution,
+ // so we do not disable bucketed scan for these queries.
+ val hashOrder = plan.requiredChildOrdering.exists(_.nonEmpty)
+ hashPartition || hashOrder
+ }
+
+ /**
+ * Check if the operator is allowed single-child operator. We may revisit this method later as we
+ * probably can remove this restriction to allow arbitrary operator between bucketed table scan
+ * and operator with interesting partition.
+ */
+ private def canPassThrough(plan: SparkPlan): Boolean = {
+ plan match {
+ case _: ProjectExec | _: FilterExec => true
+ case s: SortExec if !s.global => true
+ case partialAgg: BaseAggregateExec =>
+ partialAgg.requiredChildDistributionExpressions.isEmpty
+ case _ => false
+ }
+ }
+
+ def extractPaimonBucketedScan(plan: SparkPlan): Option[(BatchScanExec, PaimonScan)] =
+ plan match {
+ case batch: BatchScanExec =>
+ batch.scan match {
+ case scan: PaimonScan if scan.inputPartitions.forall(_.bucketed) =>
+ Some((batch, scan))
+ case _ => None
+ }
+ case _ => None
+ }
+
+ def apply(plan: SparkPlan): SparkPlan = {
+ lazy val hasBucketedScan = plan.exists {
+ case p if extractPaimonBucketedScan(p).isDefined => true
+ case _ => false
+ }
+
+ // TODO: replace it with `conf.v2BucketingEnabled` after dropping Spark3.1
+ val v2BucketingEnabled =
+ conf.getConfString("spark.sql.sources.v2.bucketing.enabled", "false").toBoolean
+ if (!v2BucketingEnabled || !conf.autoBucketedScanEnabled || !hasBucketedScan) {
+ plan
+ } else {
+ disableBucketScan(plan, hashInterestingPartitionOrOrder = false, hasExchange = false)
+ }
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/spark/sql/paimon/shims/Spark4Shim.scala b/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/spark/sql/paimon/shims/Spark4Shim.scala
index 12dddf4db571..c3dcad75a508 100644
--- a/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/spark/sql/paimon/shims/Spark4Shim.scala
+++ b/paimon-spark/paimon-spark-4.0/src/main/scala/org/apache/spark/sql/paimon/shims/Spark4Shim.scala
@@ -34,10 +34,12 @@ import org.apache.hadoop.fs.Path
import org.apache.spark.sql.SparkSession
import org.apache.spark.sql.catalyst.InternalRow
import org.apache.spark.sql.catalyst.analysis.{CTESubstitution, SubstituteUnresolvedOrdinals}
+import org.apache.spark.sql.catalyst.analysis.NamedRelation
+import org.apache.spark.sql.catalyst.catalog.CatalogStorageFormat
import org.apache.spark.sql.catalyst.expressions.{Attribute, AttributeReference, Expression}
import org.apache.spark.sql.catalyst.expressions.aggregate.AggregateExpression
import org.apache.spark.sql.catalyst.parser.ParserInterface
-import org.apache.spark.sql.catalyst.plans.logical.{Aggregate, Assignment, ColumnDefinition, CTERelationRef, InsertAction, LogicalPlan, MergeAction, MergeIntoTable, MergeRows, SubqueryAlias, TableSpec, UnresolvedWith, UpdateAction}
+import org.apache.spark.sql.catalyst.plans.logical.{Aggregate, Assignment, ColumnDefinition, CTERelationRef, DescribeRelation, InsertAction, LogicalPlan, MergeAction, MergeIntoTable, MergeRows, OverwriteByExpression, OverwritePartitionsDynamic, SubqueryAlias, TableSpec, UnresolvedWith, UpdateAction}
import org.apache.spark.sql.catalyst.plans.logical.MergeRows.Keep
import org.apache.spark.sql.catalyst.plans.physical.{ClusteredDistribution, Distribution}
import org.apache.spark.sql.catalyst.rules.Rule
@@ -48,13 +50,14 @@ import org.apache.spark.sql.connector.read.Scan
import org.apache.spark.sql.connector.write.BatchWrite
import org.apache.spark.sql.execution.{SparkFormatTable, SparkPlan}
import org.apache.spark.sql.execution.datasources.{PartitioningAwareFileIndex, PartitionSpec}
-import org.apache.spark.sql.execution.datasources.v2.{AtomicReplaceTableAsSelectExec, AtomicReplaceTableExec, ReplaceTableAsSelectExec, ReplaceTableExec}
+import org.apache.spark.sql.execution.datasources.v2.{AtomicReplaceTableAsSelectExec, AtomicReplaceTableExec, CreateTableAsSelectExec, DescribeTableExec, ReplaceTableAsSelectExec, ReplaceTableExec}
import org.apache.spark.sql.execution.datasources.v2.{DataSourceV2Relation, DataSourceV2ScanRelation}
import org.apache.spark.sql.execution.streaming.{FileStreamSink, MetadataLogFileIndex}
import org.apache.spark.sql.internal.SQLConf
import org.apache.spark.sql.types.{DataTypes, StructType, VariantType}
import org.apache.spark.unsafe.types.VariantVal
+import java.net.URI
import java.util.{Map => JMap}
/**
@@ -111,6 +114,42 @@ class Spark4Shim extends SparkShim {
tableCatalog.createTable(ident, columns, partitions, properties)
}
+ override def withStorageLocation(
+ storage: CatalogStorageFormat,
+ locationUri: Option[URI]): CatalogStorageFormat =
+ storage.copy(locationUri = locationUri)
+
+ override def overwriteByName(
+ table: NamedRelation,
+ query: LogicalPlan,
+ deleteExpr: Expression,
+ writeOptions: Map[String, String]): OverwriteByExpression =
+ OverwriteByExpression.byName(table, query, deleteExpr, writeOptions)
+
+ override def overwritePartitionsDynamicByName(
+ table: NamedRelation,
+ query: LogicalPlan,
+ writeOptions: Map[String, String]): OverwritePartitionsDynamic =
+ OverwritePartitionsDynamic.byName(table, query, writeOptions)
+
+ override def createCreateTableAsSelectExec(
+ catalog: TableCatalog,
+ ident: Identifier,
+ partitioning: Seq[Transform],
+ query: LogicalPlan,
+ tableSpec: TableSpec,
+ writeOptions: Map[String, String],
+ ifNotExists: Boolean): SparkPlan = {
+ CreateTableAsSelectExec(
+ catalog,
+ ident,
+ partitioning,
+ query,
+ tableSpec,
+ writeOptions,
+ ifNotExists)
+ }
+
override def createReplaceTableAsSelectExec(
catalog: TableCatalog,
ident: Identifier,
@@ -432,6 +471,31 @@ class Spark4Shim extends SparkShim {
parser: org.apache.spark.sql.catalyst.parser.ParserInterface): Expression =
org.apache.paimon.spark.catalog.functions.SQLFunctionConverter
.toSQLFunctionExpression(funcIdent, function, arguments, parser)
+
+ // Spark 4.0/4.1 have no `CreateTableLike` logical plan; `CREATE TABLE LIKE` still arrives as the
+ // V1 `CreateTableLikeCommand`, which `RewriteCreateTableLikeCommand` matches directly.
+ override def createTableLikeParts(plan: LogicalPlan)
+ : Option[(Seq[String], Seq[String], Option[String], Option[String], Map[String, String], Boolean, Boolean)] =
+ None
+
+ // Spark 3.x/4.0/4.1 keep `DESCRIBE ... PARTITION` inside `DescribeRelation`; see
+ // `describeRelationPartitionSpec`.
+ override def describeTablePartition(
+ plan: LogicalPlan): Option[(LogicalPlan, Map[String, String], Boolean, Seq[Attribute])] = None
+
+ override def describeRelationPartitionSpec(plan: DescribeRelation): Map[String, String] =
+ plan.partitionSpec
+
+ override def createDescribeTableExec(
+ output: Seq[Attribute],
+ catalogName: String,
+ identifier: Identifier,
+ table: Table,
+ isExtended: Boolean): SparkPlan =
+ DescribeTableExec(output, table, isExtended)
+
+ // Spark 4.0's MergeIntoTable has neither needSchemaEvolution nor pendingSchemaChanges.
+ override def mergeNeedsSchemaEvolution(merge: MergeIntoTable): Boolean = false
}
object Spark4Shim {
diff --git a/paimon-spark/paimon-spark-4.1/pom.xml b/paimon-spark/paimon-spark-4.1/pom.xml
index 4a7ed59d3305..1213d99135aa 100644
--- a/paimon-spark/paimon-spark-4.1/pom.xml
+++ b/paimon-spark/paimon-spark-4.1/pom.xml
@@ -33,6 +33,11 @@ under the License.
4.1.2
+
+ 2.0.16
@@ -57,6 +62,32 @@ under the License.
${project.version}
+
+
+ org.apache.spark
+ spark-sql-api_${scala.binary.version}
+ ${spark.version}
+
+
+
+ org.apache.spark
+ spark-connect-shims_${scala.binary.version}
+
+
+
+
org.apache.sparkspark-sql_${scala.binary.version}
@@ -83,9 +114,15 @@ under the License.
+
org.apache.paimon
- paimon-spark-ut_${scala.binary.version}
+ paimon-spark-ut-4.1_${scala.binary.version}${project.version}teststest
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/catalog/functions/SQLFunctionConverter.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/catalog/functions/SQLFunctionConverter.scala
new file mode 100644
index 000000000000..4aee17aa2ea9
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/catalog/functions/SQLFunctionConverter.scala
@@ -0,0 +1,167 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.catalog.functions
+
+import org.apache.paimon.function.{Function => PaimonFunction, FunctionDefinition, FunctionImpl}
+import org.apache.paimon.spark.SparkCatalog.FUNCTION_DEFINITION_NAME
+import org.apache.paimon.spark.SparkTypeUtils
+import org.apache.paimon.types.{DataField, RowType}
+
+import org.apache.spark.sql.catalyst.FunctionIdentifier
+import org.apache.spark.sql.catalyst.analysis.SQLFunctionExpression
+import org.apache.spark.sql.catalyst.catalog.{SQLFunction, UserDefinedFunction}
+import org.apache.spark.sql.catalyst.expressions.Expression
+import org.apache.spark.sql.catalyst.parser.ParserInterface
+import org.apache.spark.sql.types.{DataType => SparkDataType, StructType}
+
+import java.util.{Collections, HashMap => JHashMap, List => JList}
+
+import scala.collection.JavaConverters._
+
+/** Converts between Spark SQLFunction and Paimon Function with a SQLFunctionDefinition body. */
+object SQLFunctionConverter {
+
+ // Paimon-specific option keys (prefixed to avoid collision with Spark properties).
+ private val PAIMON_OPTION_PREFIX = "spark.sql-function."
+ private val IS_QUERY = PAIMON_OPTION_PREFIX + "is-query"
+ private val DETERMINISTIC = PAIMON_OPTION_PREFIX + "deterministic"
+ private val CONTAINS_SQL = PAIMON_OPTION_PREFIX + "contains-sql"
+
+ /** Build a Paimon function from a parsed CREATE FUNCTION ... RETURN statement. */
+ def toPaimonFunction(
+ funcIdent: FunctionIdentifier,
+ inputParamText: Option[String],
+ returnTypeText: String,
+ exprText: Option[String],
+ queryText: Option[String],
+ comment: Option[String],
+ isDeterministic: Option[Boolean],
+ containsSQL: Option[Boolean],
+ parser: ParserInterface,
+ properties: Map[String, String] = Map.empty): PaimonFunction = {
+ require(
+ returnTypeText != null && returnTypeText.trim.nonEmpty,
+ s"SQL function $funcIdent must have a return type (explicit or inferred).")
+ val identifier = FunctionIdentifierConverter.toPaimonIdentifier(funcIdent)
+
+ val inputParams: JList[DataField] = inputParamText.filter(_.trim.nonEmpty) match {
+ case Some(text) =>
+ SparkTypeUtils
+ .toPaimonRowType(UserDefinedFunction.parseRoutineParam(text, parser))
+ .getFields
+ case None => Collections.emptyList[DataField]()
+ }
+ val returnSparkType = parseScalarReturnType(funcIdent, returnTypeText, parser)
+ val returnParams: JList[DataField] =
+ Collections.singletonList(
+ new DataField(0, funcIdent.funcName, SparkTypeUtils.toPaimonType(returnSparkType)))
+
+ // Exactly one of exprText / queryText is set by the parser.
+ val isQuery = exprText.isEmpty && queryText.isDefined
+ val body = exprText
+ .orElse(queryText)
+ .getOrElse(throw new IllegalArgumentException(s"SQL function $funcIdent has an empty body."))
+
+ val options = new JHashMap[String, String]()
+ options.put(IS_QUERY, isQuery.toString)
+ isDeterministic.foreach(d => options.put(DETERMINISTIC, d.toString))
+ containsSQL.foreach(c => options.put(CONTAINS_SQL, c.toString))
+ properties.foreach { case (k, v) => options.put(k, v) }
+
+ new FunctionImpl(
+ identifier,
+ inputParams,
+ returnParams,
+ isDeterministic.getOrElse(true), // caller should always pass Some after analysis
+ Collections.singletonMap(FUNCTION_DEFINITION_NAME, FunctionDefinition.sql(body)),
+ comment.orNull,
+ options
+ )
+ }
+
+ /** Resolve a Paimon-stored SQL function into a Spark SQLFunctionExpression. */
+ def toSQLFunctionExpression(
+ funcIdent: FunctionIdentifier,
+ function: PaimonFunction,
+ arguments: Seq[Expression],
+ parser: ParserInterface): Expression = {
+ val options = function.options()
+
+ val body = function.definition(FUNCTION_DEFINITION_NAME) match {
+ case sql: FunctionDefinition.SQLFunctionDefinition => sql.definition()
+ case other =>
+ throw new IllegalStateException(
+ s"Function $funcIdent is not a SQL function, found definition: $other")
+ }
+
+ val inputParam: Option[StructType] = {
+ val ip = function.inputParams()
+ if (ip.isPresent && !ip.get().isEmpty) {
+ Some(SparkTypeUtils.fromPaimonType(new RowType(ip.get())).asInstanceOf[StructType])
+ } else None
+ }
+
+ val rp = function.returnParams()
+ require(
+ rp.isPresent && !rp.get().isEmpty,
+ s"SQL function $funcIdent has no return type in returnParams.")
+ val returnType: SparkDataType = SparkTypeUtils.fromPaimonType(rp.get().get(0).`type`())
+
+ val isQuery = Option(options.get(IS_QUERY))
+ .map(java.lang.Boolean.parseBoolean)
+ .getOrElse {
+ try { parser.parseExpression(body); false }
+ catch { case _: Exception => true }
+ }
+
+ val deterministic = Option(options.get(DETERMINISTIC))
+ .map(_.toBoolean)
+ .orElse(Some(function.isDeterministic))
+
+ val sqlFunction = SQLFunction(
+ name = funcIdent,
+ inputParam = inputParam,
+ returnType = Left(returnType),
+ exprText = if (isQuery) None else Some(body),
+ queryText = if (isQuery) Some(body) else None,
+ comment = Option(function.comment()),
+ deterministic = deterministic,
+ containsSQL = Option(options.get(CONTAINS_SQL)).map(_.toBoolean),
+ isTableFunc = false,
+ properties = options.asScala.filterNot(_._1.startsWith(PAIMON_OPTION_PREFIX)).toMap
+ )
+
+ SQLFunctionExpression(
+ sqlFunction.name.unquotedString,
+ sqlFunction,
+ arguments,
+ Some(sqlFunction.getScalarFuncReturnType))
+ }
+
+ private def parseScalarReturnType(
+ funcIdent: FunctionIdentifier,
+ returnTypeText: String,
+ parser: ParserInterface): SparkDataType =
+ SQLFunction.parseReturnTypeText(returnTypeText, isTableFunc = false, parser) match {
+ case Some(Left(dataType)) => dataType
+ case _ =>
+ throw new UnsupportedOperationException(
+ s"Unsupported return type '$returnTypeText' for scalar SQL function $funcIdent.")
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/MergePaimonScalarSubqueries.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/MergePaimonScalarSubqueries.scala
new file mode 100644
index 000000000000..dc9fd53d98a3
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/MergePaimonScalarSubqueries.scala
@@ -0,0 +1,90 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.catalyst.optimizer
+
+import org.apache.paimon.spark.PaimonScan
+
+import org.apache.spark.sql.catalyst.expressions.{Attribute, AttributeMap, AttributeReference, ExprId, ScalarSubquery, SortOrder}
+import org.apache.spark.sql.catalyst.plans.logical.LogicalPlan
+import org.apache.spark.sql.execution.datasources.v2.DataSourceV2ScanRelation
+
+object MergePaimonScalarSubqueries extends MergePaimonScalarSubqueriesBase {
+
+ override def tryMergeDataSourceV2ScanRelation(
+ newV2ScanRelation: DataSourceV2ScanRelation,
+ cachedV2ScanRelation: DataSourceV2ScanRelation)
+ : Option[(LogicalPlan, AttributeMap[Attribute])] = {
+ // Match by type and read fields through named accessors: Spark 4.2 (SPARK-56385) added a
+ // sixth `pushedFilters` parameter, which breaks positional patterns.
+ (newV2ScanRelation.scan, cachedV2ScanRelation.scan) match {
+ case (newScan: PaimonScan, cachedScan: PaimonScan) =>
+ val newRelation = newV2ScanRelation.relation
+ val newOutput = newV2ScanRelation.output
+ val newPartitioning = newV2ScanRelation.keyGroupedPartitioning
+ val newOrdering = newV2ScanRelation.ordering
+ val cachedRelation = cachedV2ScanRelation.relation
+ val cachedPartitioning = cachedV2ScanRelation.keyGroupedPartitioning
+ val cacheOrdering = cachedV2ScanRelation.ordering
+
+ checkIdenticalPlans(newRelation, cachedRelation).flatMap {
+ outputMap =>
+ if (
+ samePartitioning(newPartitioning, cachedPartitioning, outputMap) && sameOrdering(
+ newOrdering,
+ cacheOrdering,
+ outputMap)
+ ) {
+ mergePaimonScan(newScan, cachedScan).map {
+ mergedScan =>
+ val mergedAttributes = mergedScan
+ .readSchema()
+ .map(f => AttributeReference(f.name, f.dataType, f.nullable, f.metadata)())
+ val cachedOutputNameMap = cachedRelation.output.map(a => a.name -> a).toMap
+ val mergedOutput =
+ mergedAttributes.map(a => cachedOutputNameMap.getOrElse(a.name, a))
+ val mergedV2ScanRelation =
+ cachedV2ScanRelation.copy(scan = mergedScan, output = mergedOutput)
+
+ val mergedOutputNameMap = mergedOutput.map(a => a.name -> a).toMap
+ val newOutputMap =
+ AttributeMap(newOutput.map(a => a -> mergedOutputNameMap(a.name).toAttribute))
+
+ mergedV2ScanRelation -> newOutputMap
+ }
+ } else {
+ None
+ }
+ }
+
+ case _ => None
+ }
+ }
+
+ private def sameOrdering(
+ newOrdering: Option[Seq[SortOrder]],
+ cachedOrdering: Option[Seq[SortOrder]],
+ outputAttrMap: AttributeMap[Attribute]): Boolean = {
+ val mappedNewOrdering = newOrdering.map(_.map(mapAttributes(_, outputAttrMap)))
+ mappedNewOrdering.map(_.map(_.canonicalized)) == cachedOrdering.map(_.map(_.canonicalized))
+ }
+
+ override protected def createScalarSubquery(plan: LogicalPlan, exprId: ExprId): ScalarSubquery = {
+ ScalarSubquery(plan, exprId = exprId)
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/PushDownMapSelectedKeys.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/PushDownMapSelectedKeys.scala
new file mode 100644
index 000000000000..88e6347cee2e
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/PushDownMapSelectedKeys.scala
@@ -0,0 +1,39 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.catalyst.optimizer
+
+import org.apache.paimon.spark.PaimonScan
+
+import org.apache.spark.sql.catalyst.expressions.AttributeReference
+import org.apache.spark.sql.execution.datasources.v2.DataSourceV2ScanRelation
+
+object PushDownMapSelectedKeys extends PushDownMapSelectedKeysBase {
+
+ override protected def copyDataSourceV2ScanRelation(
+ relation: DataSourceV2ScanRelation,
+ scan: PaimonScan,
+ output: Seq[AttributeReference]): DataSourceV2ScanRelation = {
+ DataSourceV2ScanRelation(
+ relation.relation,
+ scan,
+ output,
+ relation.keyGroupedPartitioning,
+ relation.ordering)
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/commands/PaimonDynamicPartitionOverwriteCommand.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/commands/PaimonDynamicPartitionOverwriteCommand.scala
new file mode 100644
index 000000000000..f86593ff2379
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/commands/PaimonDynamicPartitionOverwriteCommand.scala
@@ -0,0 +1,98 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.commands
+
+import org.apache.paimon.options.Options
+import org.apache.paimon.spark.DynamicOverWrite
+import org.apache.paimon.table.FileStoreTable
+
+import org.apache.spark.sql.{Row, SparkSession}
+import org.apache.spark.sql.PaimonUtils.{createDataset, createNewDataFrame}
+import org.apache.spark.sql.catalyst.analysis.NamedRelation
+import org.apache.spark.sql.catalyst.plans.logical.{Command, LogicalPlan, V2WriteCommand}
+import org.apache.spark.sql.connector.catalog.TableWritePrivilege
+import org.apache.spark.sql.execution.command.RunnableCommand
+
+import scala.collection.convert.ImplicitConversions._
+
+/**
+ * Spark 4.0/4.1 flavour of the class with the same FQCN in `paimon-spark-common`, which is compiled
+ * against the profile baseline (Spark 4.2).
+ *
+ * The body is identical; only the compilation target differs. Spark 4.2 mixes the new
+ * `WriteWithSchemaEvolution` trait into `V2WriteCommand`, so a 4.2-compiled subclass calls that
+ * trait's `$init$` from its constructor. The trait does not exist before 4.2, so on a 4.0/4.1
+ * runtime every instantiation dies with `NoClassDefFoundError` — `PaimonAnalysis` builds one for
+ * any dynamic partition overwrite, making it unreachable to write such a table at all.
+ *
+ * This cannot be fixed with a `SparkShim` method: the incompatibility is in the supertype, not in a
+ * call. Dropping `extends V2WriteCommand` is not an option either — Spark matches on that type to
+ * transform the plan, which is the whole reason this class exists.
+ *
+ * A `RunnableCommand` that will execute dynamic partition overwrite using [[WriteIntoPaimonTable]].
+ *
+ * This is a workaround of Spark not supporting V1 fallback for dynamic partition overwrite. Note
+ * the following details:
+ * - Extends [[V2WriteCommand]] so that Spark can transform this plan.
+ * - Exposes the query as a child so that the Spark optimizer can optimize it.
+ */
+case class PaimonDynamicPartitionOverwriteCommand(
+ table: NamedRelation,
+ fileStoreTable: FileStoreTable,
+ query: LogicalPlan,
+ writeOptions: Map[String, String],
+ isByName: Boolean)
+ extends RunnableCommand
+ with V2WriteCommand {
+
+ override def child: LogicalPlan = query
+
+ override def withNewQuery(newQuery: LogicalPlan): PaimonDynamicPartitionOverwriteCommand = {
+ copy(query = newQuery)
+ }
+
+ override def withNewTable(newTable: NamedRelation): PaimonDynamicPartitionOverwriteCommand = {
+ copy(table = newTable)
+ }
+
+ override protected def withNewChildInternal(
+ newChild: LogicalPlan): PaimonDynamicPartitionOverwriteCommand = copy(query = newChild)
+
+ // Declared without `override` so this source stays identical to the `paimon-spark-common` copy,
+ // which must also compile on 4.2 where `WriteWithSchemaEvolution` declares these members.
+ // Dynamic partition overwrite never evolves the target schema.
+ def withSchemaEvolution: Boolean = false
+
+ // Dynamic partition overwrite replaces whole partitions, so it deletes as well as inserts —
+ // the same privilege set Spark's own `OverwritePartitionsDynamic` requests.
+ def writePrivileges: Set[TableWritePrivilege] =
+ Set(TableWritePrivilege.INSERT, TableWritePrivilege.DELETE)
+
+ override def run(sparkSession: SparkSession): Seq[Row] = {
+ WriteIntoPaimonTable(
+ fileStoreTable,
+ DynamicOverWrite,
+ createNewDataFrame(createDataset(sparkSession, query)),
+ Options.fromMap(fileStoreTable.options() ++ writeOptions)
+ ).run(sparkSession)
+ }
+
+ // Do not annotate with override here to maintain compatibility with Spark 3.3-.
+ def storeAnalyzedQuery(): Command = copy(query = query)
+}
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/data/Spark4ArrayData.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/data/Spark4ArrayData.scala
new file mode 100644
index 000000000000..593757ed5bfd
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/data/Spark4ArrayData.scala
@@ -0,0 +1,55 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.data
+
+import org.apache.paimon.types.{DataType, GeographyType, GeometryType}
+
+import org.apache.spark.sql.paimon.shims.SparkShimLoader
+import org.apache.spark.unsafe.types.{GeographyVal, GeometryVal, VariantVal}
+
+/**
+ * Spark 4.1-compatible override of the `paimon-spark4-common` `Spark4ArrayData`. Spark 4.2
+ * (SPARK-57058) replaced `SpecializedGetters`' `getGeography` / `getGeometry` with a single
+ * `getBinaryView` and removed the `GeographyVal` / `GeometryVal` value classes, so the
+ * `paimon-spark4-common` copy implements the 4.2 shape. Spark 4.1 still declares the older pair as
+ * abstract, which this copy implements. Shade writes this module's classes before the ones pulled
+ * in from `paimon-spark4-common`, so this copy wins on Spark 4.1.
+ */
+class Spark4ArrayData(override val elementType: DataType) extends AbstractSparkArrayData {
+
+ override def getVariant(ordinal: Int): VariantVal = {
+ val v = paimonArray.getVariant(ordinal)
+ new VariantVal(v.value(), v.metadata())
+ }
+
+ override def getGeography(ordinal: Int): GeographyVal =
+ SparkShimLoader.shim
+ .toSparkGeography(
+ paimonArray.getBinary(ordinal),
+ elementType.asInstanceOf[GeographyType].getCrs,
+ elementType.asInstanceOf[GeographyType].getAlgorithm.toString)
+ .asInstanceOf[GeographyVal]
+
+ override def getGeometry(ordinal: Int): GeometryVal =
+ SparkShimLoader.shim
+ .toSparkGeometry(
+ paimonArray.getBinary(ordinal),
+ elementType.asInstanceOf[GeometryType].getCrs)
+ .asInstanceOf[GeometryVal]
+}
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/data/Spark4InternalRow.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/data/Spark4InternalRow.scala
new file mode 100644
index 000000000000..05687a68093e
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/data/Spark4InternalRow.scala
@@ -0,0 +1,61 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.data
+
+import org.apache.paimon.spark.AbstractSparkInternalRow
+import org.apache.paimon.types.{GeographyType, GeometryType, RowType}
+
+import org.apache.spark.sql.paimon.shims.SparkShimLoader
+import org.apache.spark.unsafe.types.{GeographyVal, GeometryVal, VariantVal}
+
+/**
+ * Spark 4.1-compatible override of the `paimon-spark4-common` `Spark4InternalRow`. Spark 4.2
+ * (SPARK-57058) replaced `SpecializedGetters`' `getGeography` / `getGeometry` with a single
+ * `getBinaryView` and removed the `GeographyVal` / `GeometryVal` value classes, so the
+ * `paimon-spark4-common` copy implements the 4.2 shape. Spark 4.1 still declares the older pair as
+ * abstract, which this copy implements. Shade writes this module's classes before the ones pulled
+ * in from `paimon-spark4-common`, so this copy wins on Spark 4.1.
+ */
+class Spark4InternalRow(rowType: RowType) extends AbstractSparkInternalRow(rowType) {
+
+ override def getVariant(i: Int): VariantVal = {
+ val v = row.getVariant(i)
+ new VariantVal(v.value(), v.metadata())
+ }
+
+ override def getGeography(ordinal: Int): GeographyVal =
+ SparkShimLoader.shim
+ .toSparkGeography(
+ row.getBinary(ordinal),
+ rowType.getTypeAt(ordinal).asInstanceOf[GeographyType].getCrs,
+ rowType
+ .getTypeAt(ordinal)
+ .asInstanceOf[GeographyType]
+ .getAlgorithm
+ .toString
+ )
+ .asInstanceOf[GeographyVal]
+
+ override def getGeometry(ordinal: Int): GeometryVal =
+ SparkShimLoader.shim
+ .toSparkGeometry(
+ row.getBinary(ordinal),
+ rowType.getTypeAt(ordinal).asInstanceOf[GeometryType].getCrs)
+ .asInstanceOf[GeometryVal]
+}
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/execution/adaptive/DisableUnnecessaryPaimonBucketedScan.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/execution/adaptive/DisableUnnecessaryPaimonBucketedScan.scala
new file mode 100644
index 000000000000..b0101ded21bc
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/paimon/spark/execution/adaptive/DisableUnnecessaryPaimonBucketedScan.scala
@@ -0,0 +1,178 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.execution.adaptive
+
+import org.apache.paimon.spark.PaimonScan
+
+import org.apache.spark.sql.catalyst.plans.physical.{AllTuples, ClusteredDistribution}
+import org.apache.spark.sql.catalyst.rules.Rule
+import org.apache.spark.sql.execution._
+import org.apache.spark.sql.execution.aggregate.BaseAggregateExec
+import org.apache.spark.sql.execution.datasources.v2.BatchScanExec
+import org.apache.spark.sql.execution.exchange.{Exchange, ShuffleExchangeLike}
+
+// spotless:off
+/**
+ * This rule is inspired from Spark [[DisableUnnecessaryBucketedScan]] but work for v2 scan.
+ *
+ * Disable unnecessary bucketed table scan based on actual physical query plan.
+ * NOTE: this rule is designed to be applied right after [[EnsureRequirements]],
+ * where all [[ShuffleExchangeLike]] and [[SortExec]] have been added to plan properly.
+ *
+ * When BUCKETING_ENABLED and AUTO_BUCKETED_SCAN_ENABLED are set to true, go through
+ * query plan to check where bucketed table scan is unnecessary, and disable bucketed table
+ * scan if:
+ *
+ * 1. The sub-plan from root to bucketed table scan, does not contain
+ * [[hasInterestingPartitionOrOrder]] operator.
+ *
+ * 2. The sub-plan from the nearest downstream [[hasInterestingPartitionOrOrder]] operator
+ * to the bucketed table scan and at least one [[ShuffleExchangeLike]].
+ *
+ * Examples:
+ * 1. no [[hasInterestingPartitionOrOrder]] operator:
+ * Project
+ * |
+ * Filter
+ * |
+ * Scan(t1: i, j)
+ * (bucketed on column j, DISABLE bucketed scan)
+ *
+ * 2. join:
+ * SortMergeJoin(t1.i = t2.j)
+ * / \
+ * Sort(i) Sort(j)
+ * / \
+ * Shuffle(i) Scan(t2: i, j)
+ * / (bucketed on column j, enable bucketed scan)
+ * Scan(t1: i, j)
+ * (bucketed on column j, DISABLE bucketed scan)
+ *
+ * 3. aggregate:
+ * HashAggregate(i, ..., Final)
+ * |
+ * Shuffle(i)
+ * |
+ * HashAggregate(i, ..., Partial)
+ * |
+ * Filter
+ * |
+ * Scan(t1: i, j)
+ * (bucketed on column j, DISABLE bucketed scan)
+ *
+ * The idea of [[hasInterestingPartitionOrOrder]] is inspired from "interesting order" in
+ * the paper "Access Path Selection in a Relational Database Management System"
+ * (https://dl.acm.org/doi/10.1145/582095.582099).
+ */
+// spotless:on
+object DisableUnnecessaryPaimonBucketedScan extends Rule[SparkPlan] {
+
+ /**
+ * Disable bucketed table scan with pre-order traversal of plan.
+ *
+ * @param hashInterestingPartitionOrOrder
+ * The traversed plan has operator with interesting partition and order.
+ * @param hasExchange
+ * The traversed plan has [[Exchange]] operator.
+ */
+ private def disableBucketScan(
+ plan: SparkPlan,
+ hashInterestingPartitionOrOrder: Boolean,
+ hasExchange: Boolean): SparkPlan = {
+ plan match {
+ case p if hasInterestingPartitionOrOrder(p) =>
+ // Operator with interesting partition, propagates `hashInterestingPartitionOrOrder` as true
+ // to its children, and resets `hasExchange`.
+ p.mapChildren(
+ disableBucketScan(_, hashInterestingPartitionOrOrder = true, hasExchange = false))
+ case exchange: ShuffleExchangeLike =>
+ // Exchange operator propagates `hasExchange` as true to its child.
+ exchange.mapChildren(
+ disableBucketScan(_, hashInterestingPartitionOrOrder, hasExchange = true))
+ case batch: BatchScanExec =>
+ val paimonBucketedScan = extractPaimonBucketedScan(batch)
+ if (paimonBucketedScan.isDefined && (!hashInterestingPartitionOrOrder || hasExchange)) {
+ val (batch, paimonScan) = paimonBucketedScan.get
+ val newBatch = batch.copy(scan = paimonScan.disableBucketedScan())
+ newBatch.copyTagsFrom(batch)
+ newBatch
+ } else {
+ batch
+ }
+ case p if canPassThrough(p) =>
+ p.mapChildren(disableBucketScan(_, hashInterestingPartitionOrOrder, hasExchange))
+ case other =>
+ other.mapChildren(
+ disableBucketScan(_, hashInterestingPartitionOrOrder = false, hasExchange = false))
+ }
+ }
+
+ private def hasInterestingPartitionOrOrder(plan: SparkPlan): Boolean = {
+ val hashPartition = plan.requiredChildDistribution.exists {
+ case _: ClusteredDistribution | AllTuples => true
+ case _ => false
+ }
+ // Some operators may only require local sort without distribution,
+ // so we do not disable bucketed scan for these queries.
+ val hashOrder = plan.requiredChildOrdering.exists(_.nonEmpty)
+ hashPartition || hashOrder
+ }
+
+ /**
+ * Check if the operator is allowed single-child operator. We may revisit this method later as we
+ * probably can remove this restriction to allow arbitrary operator between bucketed table scan
+ * and operator with interesting partition.
+ */
+ private def canPassThrough(plan: SparkPlan): Boolean = {
+ plan match {
+ case _: ProjectExec | _: FilterExec => true
+ case s: SortExec if !s.global => true
+ case partialAgg: BaseAggregateExec =>
+ partialAgg.requiredChildDistributionExpressions.isEmpty
+ case _ => false
+ }
+ }
+
+ def extractPaimonBucketedScan(plan: SparkPlan): Option[(BatchScanExec, PaimonScan)] =
+ plan match {
+ case batch: BatchScanExec =>
+ batch.scan match {
+ case scan: PaimonScan if scan.inputPartitions.forall(_.bucketed) =>
+ Some((batch, scan))
+ case _ => None
+ }
+ case _ => None
+ }
+
+ def apply(plan: SparkPlan): SparkPlan = {
+ lazy val hasBucketedScan = plan.exists {
+ case p if extractPaimonBucketedScan(p).isDefined => true
+ case _ => false
+ }
+
+ // TODO: replace it with `conf.v2BucketingEnabled` after dropping Spark3.1
+ val v2BucketingEnabled =
+ conf.getConfString("spark.sql.sources.v2.bucketing.enabled", "false").toBoolean
+ if (!v2BucketingEnabled || !conf.autoBucketedScanEnabled || !hasBucketedScan) {
+ plan
+ } else {
+ disableBucketScan(plan, hashInterestingPartitionOrOrder = false, hasExchange = false)
+ }
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/analysis/PureAppendOnlyScope.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/analysis/PureAppendOnlyScope.scala
new file mode 100644
index 000000000000..2f2c3755d43b
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/analysis/PureAppendOnlyScope.scala
@@ -0,0 +1,78 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.spark.sql.catalyst.analysis
+
+import org.apache.paimon.spark.{SparkTable, SparkTypeUtils}
+import org.apache.paimon.table.FileStoreTable
+
+import org.apache.spark.sql.catalyst.plans.logical.LogicalPlan
+import org.apache.spark.sql.execution.datasources.v2.ExtractV2Table
+
+/**
+ * Shared scope predicates for the Spark 4.1 Resolution-batch row-level rewrite rules
+ * ([[Spark41UpdateTableRewrite]] for UPDATE + metadata-only DELETE reverse-optimization,
+ * [[Spark41MergeIntoRewrite]] for MERGE).
+ *
+ * These rules intercept operations against Paimon tables that are valid for Spark's V2
+ * copy-on-write rewrite (no primary key, data evolution, deletion vectors, or fixed-length
+ * `CHAR(n)` columns; row-tracking-only tables are included) or for the delta-based rewrite
+ * (unaware-bucket deletion-vector append tables, see [[targetsV2DeltaTable]]). Tables that violate
+ * these constraints go through Paimon's postHoc V1 commands or Spark's built-in analysis path.
+ *
+ * Kept as a mix-in trait so the two rewrite objects stay single-responsibility (one rule per Spark
+ * row-level command, mirroring Spark's own `RewriteUpdateTable` / `RewriteMergeIntoTable` layout)
+ * while sharing exactly one definition of the scope.
+ */
+trait PureAppendOnlyScope {
+
+ protected def targetsV2CopyOnWriteTable(aliasedTable: LogicalPlan): Boolean = {
+ targetsPaimonFileStoreTable(aliasedTable) {
+ case (sparkTable, fs) =>
+ fs.primaryKeys().isEmpty &&
+ !sparkTable.coreOptions.dataEvolutionEnabled() &&
+ !sparkTable.coreOptions.deletionVectorsEnabled() &&
+ !SparkTypeUtils.containsCharType(fs.rowType())
+ }
+ }
+
+ /**
+ * Whether the target is a Paimon table on the delta-based row-level path. Delegates to
+ * `SparkTable.supportsV2DeltaOps` so the delta gating conditions have a single source of truth
+ * (unlike the copy-on-write scope above, the full capability predicate is also the correct plan
+ * scope: its extra `useV2Write` / Spark version checks are trivially true once the table has
+ * exposed `SupportsRowLevelOperations` on 4.1).
+ */
+ protected def targetsV2DeltaTable(aliasedTable: LogicalPlan): Boolean = {
+ targetsPaimonFileStoreTable(aliasedTable) {
+ case (sparkTable, _) => SparkTable.supportsV2DeltaOps(sparkTable)
+ }
+ }
+
+ private def targetsPaimonFileStoreTable(aliasedTable: LogicalPlan)(
+ predicate: (SparkTable, FileStoreTable) => Boolean): Boolean = {
+ EliminateSubqueryAliases(aliasedTable) match {
+ case ExtractV2Table(sparkTable: SparkTable) =>
+ sparkTable.getTable match {
+ case fs: FileStoreTable => predicate(sparkTable, fs)
+ case _ => false
+ }
+ case _ => false
+ }
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41DeleteMetadataRestore.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41DeleteMetadataRestore.scala
new file mode 100644
index 000000000000..7267b14fc226
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41DeleteMetadataRestore.scala
@@ -0,0 +1,126 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.spark.sql.catalyst.analysis
+
+import org.apache.paimon.spark.SparkTable
+import org.apache.paimon.spark.catalyst.optimizer.OptimizeMetadataOnlyDeleteFromPaimonTable
+import org.apache.paimon.spark.commands.DeleteFromPaimonTableCommand
+import org.apache.paimon.table.FileStoreTable
+
+import org.apache.spark.sql.catalyst.plans.logical.{AnalysisHelper, LogicalPlan, ReplaceData, WriteDelta}
+import org.apache.spark.sql.connector.write.RowLevelOperation.Command.DELETE
+import org.apache.spark.sql.connector.write.RowLevelOperationTable
+import org.apache.spark.sql.execution.datasources.v2.DataSourceV2Relation
+
+/**
+ * Spark 4.1-only Resolution-batch rule that restores Paimon's metadata-only DELETE optimization
+ * after Spark's built-in `RewriteDeleteFromTable` has already rewritten the `DeleteFromTable` node
+ * into a V2 `ReplaceData` plan.
+ *
+ * Unlike [[Spark41UpdateTableRewrite]] / [[Spark41MergeIntoRewrite]], DELETE **does not** hit the
+ * `resolveOperators` short-circuit on 4.1 — Paimon has no Resolution-batch rule that advances the
+ * `DeleteFromTable` subtree to `analyzed=true`, so Spark's own `RewriteDeleteFromTable` fires
+ * normally and produces a correct `ReplaceData`. But that rewrite is unconditional: it also
+ * rewrites metadata-only DELETE (whole-table or partition-only predicate) into `ReplaceData`, which
+ * defeats Paimon's `OptimizeMetadataOnlyDeleteFromPaimonTable` → `TruncatePaimonTableWithFilter`
+ * fast path that a `DeleteFromPaimonTableCommand` would enable.
+ *
+ * This rule pattern-matches the `ReplaceData` Spark produced (tagged with
+ * `RowLevelOperation.Command.DELETE`) and, if the target is a Paimon table eligible for V2
+ * copy-on-write (see [[PureAppendOnlyScope]]) and the predicate is metadata-only, rewrites back to
+ * `DeleteFromPaimonTableCommand`. Non-metadata-only DELETE is left alone (Spark's `ReplaceData` is
+ * correct for data deletes). This is **not** a rewrite of `DeleteFromTable` — it's a restoration
+ * layered on top of Spark's existing rewrite output, hence the `…Restore` naming rather than
+ * `…Rewrite`.
+ */
+object Spark41DeleteMetadataRestore extends RewriteRowLevelCommand with PureAppendOnlyScope {
+
+ override def apply(plan: LogicalPlan): LogicalPlan = {
+ if (org.apache.spark.SPARK_VERSION < "4.1") return plan
+ AnalysisHelper.allowInvokingTransformsInAnalyzer {
+ plan.transformDown {
+ case rd: ReplaceData if isMetadataOnlyDeleteOnAppendOnlyPaimon(rd) =>
+ val origRelation = rd.originalTable.asInstanceOf[DataSourceV2Relation]
+ val fs = origRelation.table.asInstanceOf[SparkTable].getTable.asInstanceOf[FileStoreTable]
+ DeleteFromPaimonTableCommand(origRelation, fs, rd.condition)
+ // The delta-based DELETE form of unaware-bucket deletion-vector append tables: restore
+ // metadata-only DELETE to the V1 command for the same truncate fast path, instead of
+ // marking every row of the dropped partitions in deletion vectors.
+ case wd: WriteDelta if isMetadataOnlyDeleteOnDvPaimon(wd) =>
+ val origRelation = wd.originalTable.asInstanceOf[DataSourceV2Relation]
+ val fs = origRelation.table.asInstanceOf[SparkTable].getTable.asInstanceOf[FileStoreTable]
+ DeleteFromPaimonTableCommand(origRelation, fs, wd.condition)
+ }
+ }
+ }
+
+ /** The [[WriteDelta]] counterpart of [[isMetadataOnlyDeleteOnAppendOnlyPaimon]]. */
+ private def isMetadataOnlyDeleteOnDvPaimon(wd: WriteDelta): Boolean = {
+ val writeIsDelete = wd.table match {
+ case r: DataSourceV2Relation =>
+ r.table match {
+ case op: RowLevelOperationTable => op.operation.command() == DELETE
+ case _ => false
+ }
+ case _ => false
+ }
+ writeIsDelete && (wd.originalTable match {
+ case r: DataSourceV2Relation if targetsV2DeltaTable(r) =>
+ r.table match {
+ case spk: SparkTable =>
+ spk.getTable match {
+ case fs: FileStoreTable =>
+ OptimizeMetadataOnlyDeleteFromPaimonTable.isMetadataOnlyDelete(fs, wd.condition)
+ case _ => false
+ }
+ case _ => false
+ }
+ case _ => false
+ })
+ }
+
+ /**
+ * Whether a `ReplaceData` node (Spark 4.1's post-rewrite DELETE form) targets a Paimon table
+ * eligible for V2 copy-on-write with a metadata-only predicate, such that converting back to
+ * `DeleteFromPaimonTableCommand` would let the optimizer fold to `TruncatePaimonTableWithFilter`.
+ */
+ private def isMetadataOnlyDeleteOnAppendOnlyPaimon(rd: ReplaceData): Boolean = {
+ val writeIsDelete = rd.table match {
+ case r: DataSourceV2Relation =>
+ r.table match {
+ case op: RowLevelOperationTable => op.operation.command() == DELETE
+ case _ => false
+ }
+ case _ => false
+ }
+ writeIsDelete && (rd.originalTable match {
+ case r: DataSourceV2Relation if targetsV2CopyOnWriteTable(r) =>
+ r.table match {
+ case spk: SparkTable =>
+ spk.getTable match {
+ case fs: FileStoreTable =>
+ OptimizeMetadataOnlyDeleteFromPaimonTable.isMetadataOnlyDelete(fs, rd.condition)
+ case _ => false
+ }
+ case _ => false
+ }
+ case _ => false
+ })
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41MergeIntoRewrite.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41MergeIntoRewrite.scala
new file mode 100644
index 000000000000..0d012e18f249
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41MergeIntoRewrite.scala
@@ -0,0 +1,516 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.spark.sql.catalyst.analysis
+
+import org.apache.paimon.spark.SparkTable
+import org.apache.paimon.spark.catalyst.analysis.{MergeSchemaEvolutionHelper, PaimonRelation}
+
+import org.apache.spark.sql.{AnalysisException, SparkSession}
+import org.apache.spark.sql.catalyst.expressions.{Alias, And, Attribute, AttributeReference, Exists, Expression, IsNotNull, Literal, MetadataAttribute, MonotonicallyIncreasingID, OuterReference, PredicateHelper, SubqueryExpression}
+import org.apache.spark.sql.catalyst.expressions.Literal.{FalseLiteral, TrueLiteral}
+import org.apache.spark.sql.catalyst.expressions.aggregate.AggregateExpression
+import org.apache.spark.sql.catalyst.plans.{FullOuter, Inner, JoinType, LeftAnti, LeftOuter, RightOuter}
+import org.apache.spark.sql.catalyst.plans.logical.{AnalysisHelper, AppendData, DeleteAction, Filter, HintInfo, InsertAction, Join, JoinHint, LogicalPlan, MergeAction, MergeIntoTable, MergeRows, NO_BROADCAST_AND_REPLICATION, Project, ReplaceData, UpdateAction, WriteDelta}
+import org.apache.spark.sql.catalyst.plans.logical.MergeRows.{Copy, Delete, Discard, Insert, Instruction, Keep, ROW_ID, Update}
+import org.apache.spark.sql.catalyst.util.RowDeltaUtils.{OPERATION_COLUMN, WRITE_OPERATION, WRITE_WITH_METADATA_OPERATION}
+import org.apache.spark.sql.connector.catalog.SupportsRowLevelOperations
+import org.apache.spark.sql.connector.write.{RowLevelOperationTable, SupportsDelta}
+import org.apache.spark.sql.connector.write.RowLevelOperation.Command.MERGE
+import org.apache.spark.sql.errors.QueryCompilationErrors
+import org.apache.spark.sql.execution.datasources.v2.{DataSourceV2Relation, ExtractV2Table}
+import org.apache.spark.sql.types.IntegerType
+import org.apache.spark.sql.util.CaseInsensitiveStringMap
+
+/**
+ * Spark 4.1-only Resolution-batch rule that rewrites MERGE INTO on Paimon tables eligible for V2
+ * copy-on-write (no PK / DE / DV / CHAR) into V2 `ReplaceData` / `AppendData` plans, mirroring
+ * Spark's built-in `RewriteMergeIntoTable` for non-`SupportsDelta` row-level tables.
+ *
+ * In Spark 4.1, `RewriteMergeIntoTable` runs in the Resolution batch via `resolveOperators`, which
+ * short-circuits on `analyzed=true` plans — by the time it would fire, the `MergeIntoTable` is
+ * already marked analyzed and silently skipped, so the planner rejects it with
+ * `UNSUPPORTED_FEATURE.TABLE_OPERATION`. We intercept via `transformDown` under
+ * `allowInvokingTransformsInAnalyzer` and inline the three `ReplaceData`/`AppendData` branches.
+ * Copy-on-write tables mirror the non-delta rewrite; deletion-vector append tables mirror the
+ * `SupportsDelta` (`WriteDelta`) rewrite.
+ *
+ * We fire before `ResolveAssignments`, so `m.aligned` is `false`. The rule pre-aligns each action
+ * list via `PaimonAssignmentUtils.alignActions` (shared with the postHoc `PaimonMergeInto` rule).
+ *
+ * Row-tracking-only tables use the same V2 copy-on-write rewrite; unaware-bucket deletion-vector
+ * append tables take the transcribed `WriteDelta` branch (delta-based row-level operations). CHAR
+ * columns are excluded — `readSidePadding` races with the rewrite and trips CheckAnalysis; those
+ * plans fall back to the postHoc `PaimonMergeInto` V1 path, which also owns PK / DE tables via
+ * `RowLevelHelper.shouldFallbackToV1MergeInto`.
+ */
+object Spark41MergeIntoRewrite
+ extends RewriteRowLevelCommand
+ with PredicateHelper
+ with MergeSchemaEvolutionHelper
+ with PureAppendOnlyScope {
+
+ final private val ROW_FROM_SOURCE = "__row_from_source"
+ final private val ROW_FROM_TARGET = "__row_from_target"
+
+ override def apply(plan: LogicalPlan): LogicalPlan = {
+ if (org.apache.spark.SPARK_VERSION < "4.1") return plan
+ AnalysisHelper.allowInvokingTransformsInAnalyzer {
+ plan.transformDown {
+ case m: MergeIntoTable
+ if m.resolved && m.rewritable && !m.needSchemaEvolution &&
+ (targetsV2CopyOnWriteTable(m.targetTable) || targetsV2DeltaTable(m.targetTable)) =>
+ // Pure append-only tables skip postHoc `PaimonMergeInto`, so evolve schema here.
+ val evolved = evolveSchemaIfPaimon(m)
+ rewrite(alignAllMergeActions(evolved, evolved.targetTable.output))
+ }
+ }
+ }
+
+ private def evolveSchemaIfPaimon(m: MergeIntoTable): MergeIntoTable = {
+ if (!PaimonRelation.isPaimonTable(m.targetTable)) return m
+ val relation = PaimonRelation.getPaimonRelation(m.targetTable)
+ val v2Table = relation.table.asInstanceOf[SparkTable]
+ evolveTargetIfNeeded(m, relation, v2Table, SparkSession.active, _.notMatchedBySourceActions)
+ .map(_._1)
+ .getOrElse(m)
+ }
+
+ private def rewrite(m: MergeIntoTable): LogicalPlan = {
+ val MergeIntoTable(
+ aliasedTable,
+ source,
+ cond,
+ matchedActions,
+ notMatchedActions,
+ notMatchedBySourceActions,
+ _) = m
+
+ EliminateSubqueryAliases(aliasedTable) match {
+ case r @ ExtractV2Table(tbl: SupportsRowLevelOperations) =>
+ validateMergeIntoConditions(m)
+ val relation = r.asInstanceOf[DataSourceV2Relation]
+ if (
+ matchedActions.isEmpty && notMatchedBySourceActions.isEmpty &&
+ notMatchedActions.size == 1
+ ) {
+ buildSingleInsertAppendPlan(relation, source, cond, notMatchedActions.head)
+ } else if (matchedActions.isEmpty && notMatchedBySourceActions.isEmpty) {
+ buildNotMatchedOnlyAppendPlan(relation, source, cond, notMatchedActions)
+ } else {
+ val operationTable = buildOperationTable(tbl, MERGE, CaseInsensitiveStringMap.empty())
+ operationTable.operation match {
+ case _: SupportsDelta =>
+ buildWriteDeltaPlan(
+ relation,
+ operationTable,
+ source,
+ cond,
+ matchedActions,
+ notMatchedActions,
+ notMatchedBySourceActions)
+ case _ =>
+ buildReplaceDataPlan(
+ relation,
+ operationTable,
+ source,
+ cond,
+ matchedActions,
+ notMatchedActions,
+ notMatchedBySourceActions)
+ }
+ }
+ case _ =>
+ m
+ }
+ }
+
+ // Fast-path #1: single NOT MATCHED InsertAction. Append over a left-anti join.
+ private def buildSingleInsertAppendPlan(
+ r: DataSourceV2Relation,
+ source: LogicalPlan,
+ cond: Expression,
+ notMatchedAction: MergeAction): LogicalPlan = {
+ val insertAction = notMatchedAction.asInstanceOf[InsertAction]
+ val filteredSource = insertAction.condition match {
+ case Some(insertCond) => Filter(insertCond, source)
+ case None => source
+ }
+ val joinPlan = Join(filteredSource, r, LeftAnti, Some(cond), JoinHint.NONE)
+ val output = insertAction.assignments.map(_.value)
+ val outputColNames = r.output.map(_.name)
+ val projectList = output.zip(outputColNames).map { case (expr, name) => Alias(expr, name)() }
+ val project = Project(projectList, joinPlan)
+ AppendData.byPosition(r, project)
+ }
+
+ // Fast-path #2: only NOT MATCHED actions. Append over a left-anti join with `MergeRows`.
+ private def buildNotMatchedOnlyAppendPlan(
+ r: DataSourceV2Relation,
+ source: LogicalPlan,
+ cond: Expression,
+ notMatchedActions: Seq[MergeAction]): LogicalPlan = {
+ val joinPlan = Join(source, r, LeftAnti, Some(cond), JoinHint.NONE)
+ val notMatchedInstructions = notMatchedActions.map {
+ case InsertAction(cond, assignments) =>
+ Keep(Insert, cond.getOrElse(TrueLiteral), assignments.map(_.value))
+ case other =>
+ throw new AnalysisException(
+ errorClass = "_LEGACY_ERROR_TEMP_3053",
+ messageParameters = Map("other" -> other.toString))
+ }
+ val outputs = notMatchedInstructions.flatMap(_.outputs)
+ val mergeRows = MergeRows(
+ isSourceRowPresent = TrueLiteral,
+ isTargetRowPresent = FalseLiteral,
+ matchedInstructions = Nil,
+ notMatchedInstructions = notMatchedInstructions,
+ notMatchedBySourceInstructions = Nil,
+ checkCardinality = false,
+ output = generateExpandOutput(r.output, outputs),
+ joinPlan
+ )
+ AppendData.byPosition(r, mergeRows)
+ }
+
+ // Delta path producing a `WriteDelta` plan for deletion-vector append tables. Mirrors Spark
+ // 4.1's `RewriteMergeIntoTable.{buildWriteDeltaPlan, buildWriteDeltaMergeRowsPlan,
+ // chooseWriteDeltaJoinType, pushDownTargetPredicates}`; only the non-split UPDATE branch is
+ // transcribed because `PaimonSparkDeltaOperation.representUpdateAsDeleteAndInsert` is false.
+ private def buildWriteDeltaPlan(
+ relation: DataSourceV2Relation,
+ operationTable: RowLevelOperationTable,
+ source: LogicalPlan,
+ cond: Expression,
+ matchedActions: Seq[MergeAction],
+ notMatchedActions: Seq[MergeAction],
+ notMatchedBySourceActions: Seq[MergeAction]): WriteDelta = {
+
+ val operation = operationTable.operation.asInstanceOf[SupportsDelta]
+ assert(
+ !operation.representUpdateAsDeleteAndInsert,
+ "Paimon delta operations represent UPDATE as a single operation")
+
+ val rowAttrs = relation.output
+ val rowIdAttrs = resolveRowIdAttrs(relation, operation)
+ val metadataAttrs = resolveRequiredMetadataAttrs(relation, operation)
+
+ val readRelation = buildRelationWithAttrs(relation, operationTable, metadataAttrs, rowIdAttrs)
+
+ // if there is no NOT MATCHED BY SOURCE clause, predicates of the ON condition that
+ // reference only the target table can be pushed down
+ val (filteredReadRelation, joinCond) = if (notMatchedBySourceActions.isEmpty) {
+ pushDownTargetPredicates(readRelation, cond)
+ } else {
+ (readRelation, cond)
+ }
+
+ val checkCardinality = shouldCheckCardinality(matchedActions)
+
+ val joinType = chooseWriteDeltaJoinType(notMatchedActions, notMatchedBySourceActions)
+ val joinPlan = join(filteredReadRelation, source, joinType, joinCond, checkCardinality)
+
+ val mergeRowsPlan = buildWriteDeltaMergeRowsPlan(
+ readRelation,
+ joinPlan,
+ matchedActions,
+ notMatchedActions,
+ notMatchedBySourceActions,
+ rowIdAttrs,
+ checkCardinality)
+
+ val writeRelation = relation.copy(table = operationTable)
+ val projections = buildWriteDeltaProjections(mergeRowsPlan, rowAttrs, rowIdAttrs, metadataAttrs)
+ WriteDelta(writeRelation, cond, mergeRowsPlan, relation, projections)
+ }
+
+ private def chooseWriteDeltaJoinType(
+ notMatchedActions: Seq[MergeAction],
+ notMatchedBySourceActions: Seq[MergeAction]): JoinType = {
+ val unmatchedTargetRowsRequired = notMatchedBySourceActions.nonEmpty
+ val unmatchedSourceRowsRequired = notMatchedActions.nonEmpty
+ if (unmatchedTargetRowsRequired && unmatchedSourceRowsRequired) {
+ FullOuter
+ } else if (unmatchedTargetRowsRequired) {
+ LeftOuter
+ } else if (unmatchedSourceRowsRequired) {
+ RightOuter
+ } else {
+ Inner
+ }
+ }
+
+ private def buildWriteDeltaMergeRowsPlan(
+ targetTable: DataSourceV2Relation,
+ joinPlan: LogicalPlan,
+ matchedActions: Seq[MergeAction],
+ notMatchedActions: Seq[MergeAction],
+ notMatchedBySourceActions: Seq[MergeAction],
+ rowIdAttrs: Seq[Attribute],
+ checkCardinality: Boolean): MergeRows = {
+
+ val (metadataAttrs, rowAttrs) =
+ targetTable.output.partition(attr => MetadataAttribute.isValid(attr.metadata))
+
+ // original row ID values must be preserved and passed back to the table to encode updates
+ // if there are any assignments to row ID attributes, add extra columns for original values
+ val updateAssignments = (matchedActions ++ notMatchedBySourceActions).flatMap {
+ case UpdateAction(_, assignments, _) => assignments
+ case _ => Nil
+ }
+ val originalRowIdValues = buildOriginalRowIdValues(rowIdAttrs, updateAssignments)
+
+ def toDeltaInstruction(action: MergeAction): Instruction = {
+ action match {
+ case UpdateAction(cond, assignments, _) =>
+ val output = deltaUpdateOutput(assignments, metadataAttrs, originalRowIdValues)
+ Keep(Update, cond.getOrElse(TrueLiteral), output)
+ case DeleteAction(cond) =>
+ val output = deltaDeleteOutput(rowAttrs, rowIdAttrs, metadataAttrs, originalRowIdValues)
+ Keep(Delete, cond.getOrElse(TrueLiteral), output)
+ case InsertAction(cond, assignments) =>
+ val output = deltaInsertOutput(assignments, metadataAttrs, originalRowIdValues)
+ Keep(Insert, cond.getOrElse(TrueLiteral), output)
+ case other =>
+ throw new AnalysisException(
+ errorClass = "_LEGACY_ERROR_TEMP_3052",
+ messageParameters = Map("other" -> other.toString))
+ }
+ }
+
+ val matchedInstructions = matchedActions.map(toDeltaInstruction)
+ val notMatchedInstructions = notMatchedActions.map(toDeltaInstruction)
+ val notMatchedBySourceInstructions = notMatchedBySourceActions.map(toDeltaInstruction)
+
+ val rowFromSourceAttr = resolveAttrRef(ROW_FROM_SOURCE, joinPlan)
+ val rowFromTargetAttr = resolveAttrRef(ROW_FROM_TARGET, joinPlan)
+
+ val outputs = matchedInstructions.flatMap(_.outputs) ++
+ notMatchedInstructions.flatMap(_.outputs) ++
+ notMatchedBySourceInstructions.flatMap(_.outputs)
+
+ val operationTypeAttr = AttributeReference(OPERATION_COLUMN, IntegerType, nullable = false)()
+ val originalRowIdAttrs = originalRowIdValues.map(_.toAttribute)
+ val attrs = Seq(operationTypeAttr) ++ targetTable.output ++ originalRowIdAttrs
+
+ MergeRows(
+ isSourceRowPresent = IsNotNull(rowFromSourceAttr),
+ isTargetRowPresent = IsNotNull(rowFromTargetAttr),
+ matchedInstructions = matchedInstructions,
+ notMatchedInstructions = notMatchedInstructions,
+ notMatchedBySourceInstructions = notMatchedBySourceInstructions,
+ checkCardinality = checkCardinality,
+ output = generateExpandOutput(attrs, outputs),
+ joinPlan
+ )
+ }
+
+ private def pushDownTargetPredicates(
+ targetTable: LogicalPlan,
+ cond: Expression): (LogicalPlan, Expression) = {
+ val predicates = splitConjunctivePredicates(cond)
+ val (targetPredicates, joinPredicates) =
+ predicates.partition(predicate => predicate.references.subsetOf(targetTable.outputSet))
+ val targetCond = targetPredicates.reduceOption(And).getOrElse(TrueLiteral)
+ val joinCond = joinPredicates.reduceOption(And).getOrElse(TrueLiteral)
+ (Filter(targetCond, targetTable), joinCond)
+ }
+
+ // General path producing a `ReplaceData` plan. Mirrors Spark 4.1.1's
+ // `RewriteMergeIntoTable.buildReplaceDataPlan` + `buildReplaceDataMergeRowsPlan`.
+ private def buildReplaceDataPlan(
+ relation: DataSourceV2Relation,
+ operationTable: RowLevelOperationTable,
+ source: LogicalPlan,
+ cond: Expression,
+ matchedActions: Seq[MergeAction],
+ notMatchedActions: Seq[MergeAction],
+ notMatchedBySourceActions: Seq[MergeAction]): ReplaceData = {
+
+ val metadataAttrs = resolveRequiredMetadataAttrs(relation, operationTable.operation)
+ val readRelation = buildRelationWithAttrs(relation, operationTable, metadataAttrs)
+
+ val checkCardinality = shouldCheckCardinality(matchedActions)
+
+ val joinType = if (notMatchedActions.isEmpty) LeftOuter else FullOuter
+ val joinPlan = join(readRelation, source, joinType, cond, checkCardinality)
+
+ val mergeRowsPlan = buildReplaceDataMergeRowsPlan(
+ readRelation,
+ joinPlan,
+ matchedActions,
+ notMatchedActions,
+ notMatchedBySourceActions,
+ metadataAttrs,
+ checkCardinality)
+
+ val (pushableCond, groupFilterCond) = if (notMatchedBySourceActions.isEmpty) {
+ (cond, Some(toGroupFilterCondition(relation, source, cond)))
+ } else {
+ (TrueLiteral, None)
+ }
+
+ val writeRelation = relation.copy(table = operationTable)
+ val projections = buildReplaceDataProjections(mergeRowsPlan, relation.output, metadataAttrs)
+ ReplaceData(writeRelation, pushableCond, mergeRowsPlan, relation, projections, groupFilterCond)
+ }
+
+ private def buildReplaceDataMergeRowsPlan(
+ targetTable: LogicalPlan,
+ joinPlan: LogicalPlan,
+ matchedActions: Seq[MergeAction],
+ notMatchedActions: Seq[MergeAction],
+ notMatchedBySourceActions: Seq[MergeAction],
+ metadataAttrs: Seq[Attribute],
+ checkCardinality: Boolean): MergeRows = {
+
+ // Unmatched target rows must be copied through since groups are being replaced wholesale.
+ val carryoverRowsOutput = Literal(WRITE_WITH_METADATA_OPERATION) +: targetTable.output
+ val keepCarryoverRowsInstruction = Keep(Copy, TrueLiteral, carryoverRowsOutput)
+
+ val matchedInstructions = matchedActions.map {
+ action => toInstruction(action, metadataAttrs)
+ } :+ keepCarryoverRowsInstruction
+
+ val notMatchedInstructions =
+ notMatchedActions.map(action => toInstruction(action, metadataAttrs))
+
+ val notMatchedBySourceInstructions = notMatchedBySourceActions.map {
+ action => toInstruction(action, metadataAttrs)
+ } :+ keepCarryoverRowsInstruction
+
+ val rowFromSourceAttr = resolveAttrRef(ROW_FROM_SOURCE, joinPlan)
+ val rowFromTargetAttr = resolveAttrRef(ROW_FROM_TARGET, joinPlan)
+
+ val outputs = matchedInstructions.flatMap(_.outputs) ++
+ notMatchedInstructions.flatMap(_.outputs) ++
+ notMatchedBySourceInstructions.flatMap(_.outputs)
+
+ val operationTypeAttr = AttributeReference(OPERATION_COLUMN, IntegerType, nullable = false)()
+ val attrs = operationTypeAttr +: targetTable.output
+
+ MergeRows(
+ isSourceRowPresent = IsNotNull(rowFromSourceAttr),
+ isTargetRowPresent = IsNotNull(rowFromTargetAttr),
+ matchedInstructions = matchedInstructions,
+ notMatchedInstructions = notMatchedInstructions,
+ notMatchedBySourceInstructions = notMatchedBySourceInstructions,
+ checkCardinality = checkCardinality,
+ output = generateExpandOutput(attrs, outputs),
+ joinPlan
+ )
+ }
+
+ private def toGroupFilterCondition(
+ relation: DataSourceV2Relation,
+ source: LogicalPlan,
+ cond: Expression): Expression = {
+ val condWithOuterRefs = cond.transformUp {
+ case attr: Attribute if relation.outputSet.contains(attr) => OuterReference(attr)
+ case other => other
+ }
+ val outerRefs = condWithOuterRefs.collect { case OuterReference(e) => e }
+ Exists(Filter(condWithOuterRefs, source), outerRefs)
+ }
+
+ private def join(
+ targetTable: LogicalPlan,
+ source: LogicalPlan,
+ joinType: JoinType,
+ joinCond: Expression,
+ checkCardinality: Boolean): LogicalPlan = {
+ val rowFromTarget = Alias(TrueLiteral, ROW_FROM_TARGET)()
+ val targetTableProjExprs = if (checkCardinality) {
+ val rowId = Alias(MonotonicallyIncreasingID(), ROW_ID)()
+ targetTable.output ++ Seq(rowFromTarget, rowId)
+ } else {
+ targetTable.output :+ rowFromTarget
+ }
+ val targetTableProj = Project(targetTableProjExprs, targetTable)
+
+ val rowFromSource = Alias(TrueLiteral, ROW_FROM_SOURCE)()
+ val sourceTableProjExprs = source.output :+ rowFromSource
+ val sourceTableProj = Project(sourceTableProjExprs, source)
+
+ val joinHint = if (checkCardinality) {
+ JoinHint(leftHint = Some(HintInfo(Some(NO_BROADCAST_AND_REPLICATION))), rightHint = None)
+ } else {
+ JoinHint.NONE
+ }
+ Join(targetTableProj, sourceTableProj, joinType, Some(joinCond), joinHint)
+ }
+
+ private def shouldCheckCardinality(matchedActions: Seq[MergeAction]): Boolean = {
+ matchedActions match {
+ case Nil => false
+ case Seq(DeleteAction(None)) => false
+ case _ => true
+ }
+ }
+
+ // Mirrors `RewriteMergeIntoTable.toInstruction`.
+ private def toInstruction(action: MergeAction, metadataAttrs: Seq[Attribute]): Instruction = {
+ action match {
+ case UpdateAction(cond, assignments, _) =>
+ val rowValues = assignments.map(_.value)
+ val metadataValues = nullifyMetadataOnUpdate(metadataAttrs)
+ val output = Seq(Literal(WRITE_WITH_METADATA_OPERATION)) ++ rowValues ++ metadataValues
+ Keep(Update, cond.getOrElse(TrueLiteral), output)
+
+ case DeleteAction(cond) =>
+ Discard(cond.getOrElse(TrueLiteral))
+
+ case InsertAction(cond, assignments) =>
+ val rowValues = assignments.map(_.value)
+ val metadataValues = metadataAttrs.map(attr => Literal(null, attr.dataType))
+ val output = Seq(Literal(WRITE_OPERATION)) ++ rowValues ++ metadataValues
+ Keep(Insert, cond.getOrElse(TrueLiteral), output)
+
+ case other =>
+ throw new AnalysisException(
+ errorClass = "_LEGACY_ERROR_TEMP_3052",
+ messageParameters = Map("other" -> other.toString))
+ }
+ }
+
+ // Mirrors `RewriteMergeIntoTable.validateMergeIntoConditions`.
+ private def validateMergeIntoConditions(merge: MergeIntoTable): Unit = {
+ checkMergeIntoCondition("SEARCH", merge.mergeCondition)
+ val actions = merge.matchedActions ++ merge.notMatchedActions ++ merge.notMatchedBySourceActions
+ actions.foreach {
+ case DeleteAction(Some(cond)) => checkMergeIntoCondition("DELETE", cond)
+ case UpdateAction(Some(cond), _, _) => checkMergeIntoCondition("UPDATE", cond)
+ case InsertAction(Some(cond), _) => checkMergeIntoCondition("INSERT", cond)
+ case _ => // OK
+ }
+ }
+
+ private def checkMergeIntoCondition(condName: String, cond: Expression): Unit = {
+ if (!cond.deterministic) {
+ throw QueryCompilationErrors.nonDeterministicMergeCondition(condName, cond)
+ }
+ if (SubqueryExpression.hasSubquery(cond)) {
+ throw QueryCompilationErrors.subqueryNotAllowedInMergeCondition(condName, cond)
+ }
+ if (cond.exists(_.isInstanceOf[AggregateExpression])) {
+ throw QueryCompilationErrors.aggregationNotAllowedInMergeCondition(condName, cond)
+ }
+ }
+
+ // Scope checks live in `PureAppendOnlyScope`, shared with `Spark41AppendOnlyRowLevelRewrite`.
+}
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41UpdateTableRewrite.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41UpdateTableRewrite.scala
new file mode 100644
index 000000000000..e669f3a1dffd
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41UpdateTableRewrite.scala
@@ -0,0 +1,201 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.spark.sql.catalyst.analysis
+
+import org.apache.paimon.spark.catalyst.analysis.PaimonAssignmentUtils
+
+import org.apache.spark.sql.catalyst.expressions.{Alias, Attribute, EqualNullSafe, Expression, If, Literal, MetadataAttribute, Not, SubqueryExpression}
+import org.apache.spark.sql.catalyst.expressions.Literal.TrueLiteral
+import org.apache.spark.sql.catalyst.plans.logical.{AnalysisHelper, Assignment, Filter, LogicalPlan, Project, ReplaceData, Union, UpdateTable, WriteDelta}
+import org.apache.spark.sql.catalyst.util.RowDeltaUtils.{OPERATION_COLUMN, UPDATE_OPERATION, WRITE_WITH_METADATA_OPERATION}
+import org.apache.spark.sql.connector.catalog.SupportsRowLevelOperations
+import org.apache.spark.sql.connector.write.{RowLevelOperationTable, SupportsDelta}
+import org.apache.spark.sql.connector.write.RowLevelOperation.Command.UPDATE
+import org.apache.spark.sql.execution.datasources.v2.{DataSourceV2Relation, ExtractV2Table}
+import org.apache.spark.sql.util.CaseInsensitiveStringMap
+
+/**
+ * Spark 4.1-only Resolution-batch rule that rewrites UPDATE on pure append-only Paimon tables (see
+ * [[PureAppendOnlyScope]]) into a V2 `ReplaceData` plan, mirroring Spark's built-in
+ * `RewriteUpdateTable`.
+ *
+ * In Spark 4.1, `RewriteUpdateTable` runs in the Resolution batch via `resolveOperators`, which
+ * short-circuits on `analyzed=true` plans — by the time it would fire, the `UpdateTable` is already
+ * marked analyzed and silently skipped, so the planner rejects it with
+ * `UNSUPPORTED_FEATURE.TABLE_OPERATION`. We intercept via `transformDown` under
+ * `allowInvokingTransformsInAnalyzer` and inline `buildReplaceDataPlan` /
+ * `buildReplaceDataWithUnionPlan`. The class sits in `org.apache.spark.sql.catalyst.analysis` to
+ * reach the package-private `RowLevelOperationTable` / `ReplaceData` types and the protected
+ * helpers on `RewriteRowLevelCommand`.
+ *
+ * We fire before `ResolveAssignments`, so `u.aligned` is `false`; the rule pre-aligns via
+ * `PaimonAssignmentUtils.alignUpdateAssignments` before building the plan.
+ *
+ * Row-tracking-only tables use the same V2 copy-on-write rewrite; unaware-bucket deletion-vector
+ * append tables take the transcribed `WriteDelta` branch (delta-based row-level operations). PK /
+ * DE tables go through the postHoc V1 rule because they do not expose `SupportsRowLevelOperations`.
+ * DELETE is handled by [[Spark41DeleteMetadataRestore]]; MERGE by [[Spark41MergeIntoRewrite]].
+ */
+object Spark41UpdateTableRewrite extends RewriteRowLevelCommand with PureAppendOnlyScope {
+
+ override def apply(plan: LogicalPlan): LogicalPlan = {
+ if (org.apache.spark.SPARK_VERSION < "4.1") return plan
+ AnalysisHelper.allowInvokingTransformsInAnalyzer {
+ plan.transformDown {
+ case u @ UpdateTable(aliasedTable, assignments, cond)
+ if u.resolved && u.rewritable &&
+ (targetsV2CopyOnWriteTable(aliasedTable) || targetsV2DeltaTable(aliasedTable)) =>
+ EliminateSubqueryAliases(aliasedTable) match {
+ case r @ ExtractV2Table(tbl: SupportsRowLevelOperations) =>
+ val table = buildOperationTable(tbl, UPDATE, CaseInsensitiveStringMap.empty())
+ val updateCond = cond.getOrElse(TrueLiteral)
+ // `ResolveAssignments` fires later in the batch, so `u.aligned` is still false.
+ // Pre-align via the same utility the postHoc V1 fallback uses.
+ val alignedAssignments = PaimonAssignmentUtils.alignUpdateAssignments(
+ r.output,
+ assignments,
+ fromStar = false,
+ mergeSchemaEnabled = false)
+ table.operation match {
+ case _: SupportsDelta =>
+ buildWriteDeltaPlan(r, table, alignedAssignments, updateCond)
+ case _ if SubqueryExpression.hasSubquery(updateCond) =>
+ buildReplaceDataWithUnionPlan(r, table, alignedAssignments, updateCond)
+ case _ =>
+ buildReplaceDataPlan(r, table, alignedAssignments, updateCond)
+ }
+ case _ =>
+ u
+ }
+ }
+ }
+ }
+
+ // Mirrors Spark 4.1 `RewriteUpdateTable.buildWriteDeltaPlan` for delta tables (Paimon's
+ // `PaimonSparkDeltaOperation` answers `representUpdateAsDeleteAndInsert = false`, so only the
+ // single-row UPDATE projection branch is transcribed).
+ private def buildWriteDeltaPlan(
+ relation: DataSourceV2Relation,
+ operationTable: RowLevelOperationTable,
+ assignments: Seq[Assignment],
+ cond: Expression): WriteDelta = {
+ val operation = operationTable.operation.asInstanceOf[SupportsDelta]
+ val rowAttrs = relation.output
+ val rowIdAttrs = resolveRowIdAttrs(relation, operation)
+ val metadataAttrs = resolveRequiredMetadataAttrs(relation, operation)
+ val readRelation = buildRelationWithAttrs(relation, operationTable, metadataAttrs, rowIdAttrs)
+ val matchedRowsPlan = Filter(cond, readRelation)
+ assert(
+ !operation.representUpdateAsDeleteAndInsert,
+ "Paimon delta operations represent UPDATE as a single operation")
+ val rowDeltaPlan = buildWriteDeltaUpdateProjection(matchedRowsPlan, assignments, rowIdAttrs)
+ val writeRelation = relation.copy(table = operationTable)
+ val projections = buildWriteDeltaProjections(rowDeltaPlan, rowAttrs, rowIdAttrs, metadataAttrs)
+ WriteDelta(writeRelation, cond, rowDeltaPlan, relation, projections)
+ }
+
+ // Mirrors Spark 4.1 `RewriteUpdateTable.buildWriteDeltaUpdateProjection`.
+ private def buildWriteDeltaUpdateProjection(
+ plan: LogicalPlan,
+ assignments: Seq[Assignment],
+ rowIdAttrs: Seq[Attribute]): LogicalPlan = {
+ val assignedValues = assignments.map(_.value)
+ val updatedValues = plan.output.zipWithIndex.map {
+ case (attr, index) =>
+ if (index < assignments.size) {
+ val assignedExpr = assignedValues(index)
+ Alias(assignedExpr, attr.name)()
+ } else {
+ assert(MetadataAttribute.isValid(attr.metadata))
+ if (MetadataAttribute.isPreservedOnUpdate(attr)) {
+ attr
+ } else {
+ Alias(Literal(null, attr.dataType), attr.name)(explicitMetadata = Some(attr.metadata))
+ }
+ }
+ }
+ val originalRowIdValues = buildOriginalRowIdValues(rowIdAttrs, assignments)
+ val operationType = Alias(Literal(UPDATE_OPERATION), OPERATION_COLUMN)()
+ Project(Seq(operationType) ++ updatedValues ++ originalRowIdValues, plan)
+ }
+
+ // Mirrors Spark 4.1.1 `RewriteUpdateTable.{buildReplaceDataPlan, buildReplaceDataWithUnionPlan,
+ // buildReplaceDataUpdateProjection}`.
+ private def buildReplaceDataPlan(
+ relation: DataSourceV2Relation,
+ operationTable: RowLevelOperationTable,
+ assignments: Seq[Assignment],
+ cond: Expression): ReplaceData = {
+ val metadataAttrs = resolveRequiredMetadataAttrs(relation, operationTable.operation)
+ val readRelation = buildRelationWithAttrs(relation, operationTable, metadataAttrs)
+ val updatedAndRemainingRowsPlan =
+ buildReplaceDataUpdateProjection(readRelation, assignments, cond)
+ val writeRelation = relation.copy(table = operationTable)
+ val query = addOperationColumn(WRITE_WITH_METADATA_OPERATION, updatedAndRemainingRowsPlan)
+ val projections = buildReplaceDataProjections(query, relation.output, metadataAttrs)
+ ReplaceData(writeRelation, cond, query, relation, projections, Some(cond))
+ }
+
+ private def buildReplaceDataWithUnionPlan(
+ relation: DataSourceV2Relation,
+ operationTable: RowLevelOperationTable,
+ assignments: Seq[Assignment],
+ cond: Expression): ReplaceData = {
+ val metadataAttrs = resolveRequiredMetadataAttrs(relation, operationTable.operation)
+ val readRelation = buildRelationWithAttrs(relation, operationTable, metadataAttrs)
+
+ val matchedRowsPlan = Filter(cond, readRelation)
+ val updatedRowsPlan = buildReplaceDataUpdateProjection(matchedRowsPlan, assignments)
+
+ val remainingRowFilter = Not(EqualNullSafe(cond, TrueLiteral))
+ val remainingRowsPlan = Filter(remainingRowFilter, readRelation)
+
+ val updatedAndRemainingRowsPlan = Union(updatedRowsPlan, remainingRowsPlan)
+
+ val writeRelation = relation.copy(table = operationTable)
+ val query = addOperationColumn(WRITE_WITH_METADATA_OPERATION, updatedAndRemainingRowsPlan)
+ val projections = buildReplaceDataProjections(query, relation.output, metadataAttrs)
+ ReplaceData(writeRelation, cond, query, relation, projections, Some(cond))
+ }
+
+ /** Assumes assignments are already aligned with the table output. */
+ private def buildReplaceDataUpdateProjection(
+ plan: LogicalPlan,
+ assignments: Seq[Assignment],
+ cond: Expression = TrueLiteral): LogicalPlan = {
+ val assignedValues = assignments.map(_.value)
+ val updatedValues = plan.output.zipWithIndex.map {
+ case (attr, index) =>
+ if (index < assignments.size) {
+ val assignedExpr = assignedValues(index)
+ val updatedValue = If(cond, assignedExpr, attr)
+ Alias(updatedValue, attr.name)()
+ } else {
+ assert(MetadataAttribute.isValid(attr.metadata))
+ if (MetadataAttribute.isPreservedOnUpdate(attr)) {
+ attr
+ } else {
+ val updatedValue = If(cond, Literal(null, attr.dataType), attr)
+ Alias(updatedValue, attr.name)(explicitMetadata = Some(attr.metadata))
+ }
+ }
+ }
+ Project(updatedValues, plan)
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/CreatePaimonSQLFunctionCommand.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/CreatePaimonSQLFunctionCommand.scala
new file mode 100644
index 000000000000..de4195ac3fcd
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/CreatePaimonSQLFunctionCommand.scala
@@ -0,0 +1,514 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.spark.sql.catalyst.parser.extensions
+
+import org.apache.paimon.spark.catalog.SupportV1Function
+import org.apache.paimon.spark.catalog.functions.SQLFunctionConverter
+import org.apache.paimon.spark.leafnode.PaimonLeafRunnableCommand
+
+import org.apache.spark.SparkException
+import org.apache.spark.sql.{AnalysisException, Row, SparkSession}
+import org.apache.spark.sql.catalyst.CapturesConfig
+import org.apache.spark.sql.catalyst.FunctionIdentifier
+import org.apache.spark.sql.catalyst.analysis.{withPosition, Analyzer, SQLFunctionExpression, SQLFunctionNode, SQLScalarFunction, SQLTableFunction, UnresolvedAlias, UnresolvedAttribute, UnresolvedFunction, UnresolvedRelation, UnresolvedTableValuedFunction}
+import org.apache.spark.sql.catalyst.catalog.{SessionCatalog, SQLFunction, UserDefinedFunction, UserDefinedFunctionErrors}
+import org.apache.spark.sql.catalyst.catalog.UserDefinedFunction._
+import org.apache.spark.sql.catalyst.expressions.{Alias, Cast, Expression, Generator, LateralSubquery, Literal, ScalarSubquery, SubqueryExpression, WindowExpression}
+import org.apache.spark.sql.catalyst.expressions.aggregate.AggregateExpression
+import org.apache.spark.sql.catalyst.plans.Inner
+import org.apache.spark.sql.catalyst.plans.logical.{LateralJoin, LocalRelation, LogicalPlan, OneRowRelation, Project, Range, UnresolvedWith, View}
+import org.apache.spark.sql.catalyst.trees.TreePattern.UNRESOLVED_ATTRIBUTE
+import org.apache.spark.sql.connector.catalog.CatalogV2Implicits.MultipartIdentifierHelper
+import org.apache.spark.sql.errors.QueryCompilationErrors
+import org.apache.spark.sql.execution.command.CreateUserDefinedFunctionCommand._
+import org.apache.spark.sql.execution.command.ViewHelper
+import org.apache.spark.sql.types.{DataType, StructField, StructType}
+
+/**
+ * Adapted from Spark's CreateSQLFunctionCommand. Analyzes the function body, validates, derives
+ * deterministic/containsSQL, then persists to the Paimon catalog instead of the session catalog.
+ */
+case class CreatePaimonSQLFunctionCommand(
+ catalog: SupportV1Function,
+ name: FunctionIdentifier,
+ inputParamText: Option[String],
+ returnTypeText: String,
+ exprText: Option[String],
+ queryText: Option[String],
+ comment: Option[String],
+ isDeterministic: Option[Boolean],
+ containsSQL: Option[Boolean],
+ isTableFunc: Boolean,
+ ignoreIfExists: Boolean,
+ replace: Boolean)
+ extends PaimonLeafRunnableCommand
+ with CapturesConfig {
+
+ import SQLFunction._
+
+ override def run(sparkSession: SparkSession): Seq[Row] = {
+ val parser = sparkSession.sessionState.sqlParser
+ val analyzer = sparkSession.sessionState.analyzer
+ val sessionCatalog = sparkSession.sessionState.catalog
+ val conf = sparkSession.sessionState.conf
+
+ val inputParam = inputParamText.map(UserDefinedFunction.parseRoutineParam(_, parser))
+ val returnType = parseReturnTypeText(returnTypeText, isTableFunc, parser)
+
+ val function = SQLFunction(
+ name,
+ inputParam,
+ returnType.getOrElse(if (isTableFunc) Right(null) else Left(null)),
+ exprText,
+ queryText,
+ comment,
+ isDeterministic,
+ containsSQL,
+ isTableFunc,
+ Map.empty
+ )
+
+ val newFunction = {
+ val (expression, query) = function.getExpressionAndQuery(parser, isTableFunc)
+ assert(query.nonEmpty || expression.nonEmpty)
+
+ // Build function input.
+ val inputPlan = if (inputParam.isDefined) {
+ val param = inputParam.get
+ checkParameterNotNull(param, inputParamText.get)
+ checkParameterNameDuplication(param, conf, name)
+ checkDefaultsTrailing(param, name)
+
+ // Qualify the input parameters with the function name so that attributes referencing
+ // the function input parameters can be resolved correctly.
+ val qualifier = Seq(name.funcName)
+ val input = param.map(
+ p =>
+ Alias(
+ {
+ val defaultExpr = p.getDefault()
+ if (defaultExpr.isEmpty) {
+ Literal.create(null, p.dataType)
+ } else {
+ val defaultPlan = parseDefault(defaultExpr.get, parser)
+ if (SubqueryExpression.hasSubquery(defaultPlan)) {
+ throw new AnalysisException(
+ errorClass = "USER_DEFINED_FUNCTIONS.NOT_A_VALID_DEFAULT_EXPRESSION",
+ messageParameters =
+ Map("functionName" -> name.funcName, "parameterName" -> p.name))
+ } else if (defaultPlan.containsPattern(UNRESOLVED_ATTRIBUTE)) {
+ // TODO(SPARK-50698): use parsed expression instead of expression string.
+ defaultPlan.collect {
+ case a: UnresolvedAttribute =>
+ throw QueryCompilationErrors.unresolvedAttributeError(
+ "UNRESOLVED_COLUMN",
+ a.sql,
+ Seq.empty,
+ a.origin)
+ }
+ }
+ Cast(defaultPlan, p.dataType)
+ }
+ },
+ p.name
+ )(qualifier = qualifier))
+ Project(input, OneRowRelation())
+ } else {
+ OneRowRelation()
+ }
+
+ // Build the function body and check if the function body can be analyzed successfully.
+ val (unresolvedPlan, analyzedPlan, inferredReturnType) = if (!isTableFunc) {
+ // Build SQL scalar function plan.
+ val outputExpr = if (query.isDefined) ScalarSubquery(query.get) else expression.get
+ val plan: LogicalPlan = returnType
+ .map {
+ t =>
+ val retType: DataType = t match {
+ case Left(t) => t
+ case _ =>
+ throw SparkException.internalError("Unexpected return type for a scalar SQL UDF.")
+ }
+ val outputCast = Seq(Alias(Cast(outputExpr, retType), name.funcName)())
+ Project(outputCast, inputPlan)
+ }
+ .getOrElse {
+ // If no explicit RETURNS clause is present, infer the result type from the function body.
+ val outputAlias = Seq(Alias(outputExpr, name.funcName)())
+ Project(outputAlias, inputPlan)
+ }
+
+ // Check cyclic function reference before running the analyzer.
+ checkCyclicFunctionReference(sessionCatalog, name, plan)
+
+ // Check the function body can be analyzed correctly.
+ val analyzed = analyzer.execute(plan)
+ val (resolved, resolvedReturnType) = analyzed match {
+ case p @ Project(expr :: Nil, _) if expr.resolved =>
+ (p, Left(expr.dataType))
+ case other =>
+ (other, function.returnType)
+ }
+
+ // Check if the SQL function body contains aggregate/window functions.
+ // This check needs to be performed before checkAnalysis to provide better error messages.
+ checkAggOrWindowOrGeneratorExpr(resolved)
+
+ // Check if the SQL function body can be analyzed.
+ checkFunctionBodyAnalysis(analyzer, function, resolved)
+
+ (plan, resolved, resolvedReturnType)
+ } else {
+ // Build SQL table function plan.
+ if (query.isEmpty) {
+ throw UserDefinedFunctionErrors.bodyIsNotAQueryForSqlTableUdf(name.funcName)
+ }
+ // Check cyclic function reference before running the analyzer.
+ checkCyclicFunctionReference(sessionCatalog, name, query.get)
+
+ // Construct a lateral join to analyze the function body.
+ val plan = LateralJoin(inputPlan, LateralSubquery(query.get), Inner, None)
+ val analyzed = analyzer.execute(plan)
+ val newPlan = analyzed match {
+ case Project(_, j: LateralJoin) => j
+ case j: LateralJoin => j
+ case _ =>
+ throw SparkException.internalError(
+ "Unexpected plan returned when " +
+ s"creating a SQL TVF: ${analyzed.getClass.getSimpleName}.")
+ }
+ val maybeResolved = newPlan.asInstanceOf[LateralJoin].right.plan
+
+ // Check if the function body can be analyzed.
+ checkFunctionBodyAnalysis(analyzer, function, maybeResolved)
+
+ // Get the function's return schema.
+ val returnParam: StructType = returnType
+ .map {
+ case Right(t) => t
+ case Left(_) =>
+ throw SparkException.internalError(
+ "Unexpected return schema for a SQL table function.")
+ }
+ .getOrElse {
+ query.get match {
+ case Project(projectList, _) if projectList.exists(_.isInstanceOf[UnresolvedAlias]) =>
+ throw UserDefinedFunctionErrors.missingColumnNamesForSqlTableUdf(name.funcName)
+ case _ =>
+ StructType(analyzed.asInstanceOf[LateralJoin].right.plan.output.map {
+ col => StructField(col.name, col.dataType)
+ })
+ }
+ }
+
+ // Check the return columns cannot have NOT NULL specified.
+ checkParameterNotNull(returnParam, returnTypeText)
+
+ // Check duplicated return column names.
+ checkReturnsColumnDuplication(returnParam, conf, name)
+
+ // Check if the actual output size equals to the number of return parameters.
+ val outputSize = maybeResolved.output.size
+ if (outputSize != returnParam.size) {
+ throw new AnalysisException(
+ errorClass = "USER_DEFINED_FUNCTIONS.RETURN_COLUMN_COUNT_MISMATCH",
+ messageParameters = Map(
+ "outputSize" -> s"$outputSize",
+ "returnParamSize" -> s"${returnParam.size}",
+ "name" -> s"$name"
+ )
+ )
+ }
+
+ (plan, analyzed, Right(returnParam))
+ }
+
+ // A permanent function is not allowed to reference temporary objects.
+ verifyTemporaryObjectsNotExists(sessionCatalog, name, unresolvedPlan, analyzedPlan)
+
+ // Generate function properties.
+ val properties = generateFunctionProperties(sparkSession, unresolvedPlan, analyzedPlan)
+
+ // Derive determinism of the SQL function.
+ val deterministic = analyzedPlan.deterministic
+
+ // Derive and check a SQL function with CONTAINS SQL data access should not reads SQL data.
+ val readsSQLData = deriveSQLDataAccess(analyzedPlan)
+
+ function.copy(
+ // Assign the return type, inferring from the function body if needed.
+ returnType = inferredReturnType,
+ deterministic = Some(function.deterministic.getOrElse(deterministic)),
+ containsSQL = Some(function.containsSQL.getOrElse(!readsSQLData)),
+ properties = properties
+ )
+ }
+
+ // ---- Paimon-specific: persist to Paimon catalog ----
+ val resolvedReturnTypeText = newFunction.returnType match {
+ case Left(dt) if dt != null => dt.sql
+ case _ =>
+ throw new UnsupportedOperationException(
+ s"Cannot infer return type for SQL function ${name.funcName}. " +
+ "Please add an explicit RETURNS clause.")
+ }
+
+ val paimonFunction = SQLFunctionConverter.toPaimonFunction(
+ name,
+ inputParamText,
+ if (returnTypeText != null && returnTypeText.trim.nonEmpty) returnTypeText
+ else resolvedReturnTypeText,
+ exprText,
+ queryText,
+ comment,
+ newFunction.deterministic,
+ newFunction.containsSQL,
+ parser,
+ newFunction.properties
+ )
+
+ if (replace) {
+ catalog.dropV1Function(name, true)
+ }
+ catalog.createV1Function(paimonFunction, ignoreIfExists)
+ Nil
+ }
+
+ /** Check if the function body can be analyzed. */
+ private def checkFunctionBodyAnalysis(
+ analyzer: Analyzer,
+ function: SQLFunction,
+ body: LogicalPlan): Unit = {
+ analyzer.checkAnalysis(SQLFunctionNode(function, body))
+ }
+
+ /** Collect all temporary views and functions and return the identifiers separately */
+ private def collectTemporaryObjectsInUnresolvedPlan(
+ catalog: SessionCatalog,
+ child: LogicalPlan): (Seq[Seq[String]], Seq[String]) = {
+ import org.apache.spark.sql.connector.catalog.CatalogV2Implicits._
+ def collectTempViews(child: LogicalPlan): Seq[Seq[String]] = {
+ child.flatMap {
+ case UnresolvedRelation(nameParts, _, _) if catalog.isTempView(nameParts) =>
+ Seq(nameParts)
+ case w: UnresolvedWith if !w.resolved => w.innerChildren.flatMap(collectTempViews)
+ case plan if !plan.resolved =>
+ plan.expressions.flatMap(_.flatMap {
+ case e: SubqueryExpression => collectTempViews(e.plan)
+ case _ => Seq.empty
+ })
+ case _ => Seq.empty
+ }.distinct
+ }
+
+ def collectTempFunctions(child: LogicalPlan): Seq[String] = {
+ child.flatMap {
+ case w: UnresolvedWith if !w.resolved => w.innerChildren.flatMap(collectTempFunctions)
+ case plan if !plan.resolved =>
+ plan.expressions.flatMap(_.flatMap {
+ case e: SubqueryExpression => collectTempFunctions(e.plan)
+ case e: UnresolvedFunction
+ if catalog.isTemporaryFunction(e.nameParts.asFunctionIdentifier) =>
+ Seq(e.nameParts.asFunctionIdentifier.funcName)
+ case _ => Seq.empty
+ })
+ case _ => Seq.empty
+ }.distinct
+ }
+ (collectTempViews(child), collectTempFunctions(child))
+ }
+
+ /**
+ * Permanent functions are not allowed to reference temp objects, including temp functions and
+ * temp views.
+ */
+ private def verifyTemporaryObjectsNotExists(
+ catalog: SessionCatalog,
+ name: FunctionIdentifier,
+ child: LogicalPlan,
+ analyzed: LogicalPlan): Unit = {
+ import org.apache.spark.sql.connector.catalog.CatalogV2Implicits._
+ val (tempViews, tempFunctions) = collectTemporaryObjectsInUnresolvedPlan(catalog, child)
+ tempViews.foreach {
+ nameParts =>
+ throw UserDefinedFunctionErrors.invalidTempViewReference(
+ routineName = name.asMultipart,
+ tempViewName = nameParts)
+ }
+ tempFunctions.foreach {
+ funcName =>
+ throw UserDefinedFunctionErrors.invalidTempFuncReference(
+ routineName = name.asMultipart,
+ tempFuncName = funcName)
+ }
+ val tempVars = ViewHelper.collectTemporaryVariables(analyzed)
+ tempVars.foreach {
+ varName =>
+ throw UserDefinedFunctionErrors.invalidTempVarReference(
+ routineName = name.asMultipart,
+ varName = varName)
+ }
+ }
+
+ /** Check if the given plan contains cyclic function references. */
+ private def checkCyclicFunctionReference(
+ catalog: SessionCatalog,
+ identifier: FunctionIdentifier,
+ plan: LogicalPlan): Unit = {
+ import org.apache.spark.sql.connector.catalog.CatalogV2Implicits._
+
+ def checkPlan(plan: LogicalPlan, path: Seq[FunctionIdentifier]): Unit = {
+ plan.foreach {
+ case u @ UnresolvedTableValuedFunction(nameParts, arguments, _) =>
+ try {
+ val funcId = nameParts.asFunctionIdentifier
+ val info = catalog.lookupFunctionInfo(funcId)
+ if (isSQLFunction(info.getClassName)) {
+ val f = withPosition(u) {
+ catalog.lookupTableFunction(funcId, arguments).asInstanceOf[SQLTableFunction]
+ }
+ val newPath = path :+ f.function.name
+ if (f.function.name == name) {
+ throw UserDefinedFunctionErrors.cyclicFunctionReference(newPath.mkString(" -> "))
+ }
+ val plan = catalog.makeSQLTableFunctionPlan(f.name, f.function, f.inputs, f.output)
+ checkPlan(plan, newPath)
+ }
+ } catch {
+ case _: AnalysisException =>
+ }
+ case p: LogicalPlan =>
+ p.expressions.foreach(checkExpression(_, path))
+ }
+ }
+
+ def checkExpression(expression: Expression, path: Seq[FunctionIdentifier]): Unit = {
+ expression.foreach {
+ case s: SubqueryExpression => checkPlan(s.plan, path)
+ case u @ UnresolvedFunction(nameParts, arguments, _, _, _, _, _) =>
+ try {
+ val funcId = nameParts.asFunctionIdentifier
+ val info = catalog.lookupFunctionInfo(funcId)
+ if (isSQLFunction(info.getClassName)) {
+ val f = withPosition(u) {
+ catalog.lookupFunction(funcId, arguments).asInstanceOf[SQLFunctionExpression]
+ }
+ val newPath = path :+ f.function.name
+ if (f.function.name == name) {
+ throw UserDefinedFunctionErrors.cyclicFunctionReference(newPath.mkString(" -> "))
+ }
+ val plan = catalog.makeSQLFunctionPlan(f.name, f.function, f.inputs)
+ checkPlan(plan, newPath)
+ }
+ } catch {
+ case _: AnalysisException =>
+ }
+ case _ =>
+ }
+ }
+
+ checkPlan(plan, Seq(identifier))
+ }
+
+ /**
+ * Check if the SQL function body contains aggregate/window/generate functions. Note subqueries
+ * inside the SQL function body can contain aggregate/window/generate functions.
+ */
+ private def checkAggOrWindowOrGeneratorExpr(plan: LogicalPlan): Unit = {
+ if (plan.resolved) {
+ plan.transformAllExpressions {
+ case e
+ if e.isInstanceOf[WindowExpression] || e.isInstanceOf[Generator] ||
+ e.isInstanceOf[AggregateExpression] =>
+ throw new AnalysisException(
+ errorClass = "USER_DEFINED_FUNCTIONS.CANNOT_CONTAIN_COMPLEX_FUNCTIONS",
+ messageParameters = Map("queryText" -> s"${exprText.orElse(queryText).get}")
+ )
+ }
+ }
+ }
+
+ /**
+ * Derive the SQL data access routine of the function and check if the SQL function matches its
+ * data access routine. If the data access is CONTAINS SQL, the expression should not access
+ * operators and expressions that read SQL data.
+ *
+ * Returns true is SQL data access routine is READS SQL DATA, otherwise returns false.
+ */
+ private def deriveSQLDataAccess(plan: LogicalPlan): Boolean = {
+ // Find logical plan nodes that read SQL data.
+ val readsSQLData = plan.find {
+ case _: View => true
+ case p if p.children.isEmpty =>
+ p match {
+ case _: OneRowRelation | _: LocalRelation | _: Range => false
+ case _ => true
+ }
+ case f: SQLFunctionNode => f.function.containsSQL.contains(false)
+ case p: LogicalPlan =>
+ lazy val sub = p.subqueries.exists(deriveSQLDataAccess)
+ // If the SQL function contains another SQL function that has SQL data access routine
+ // to be READS SQL DATA, then this SQL function will also be READS SQL DATA.
+ p.expressions.exists(
+ expr =>
+ expr.find {
+ case f: SQLScalarFunction => f.function.containsSQL.contains(false)
+ case sub: SubqueryExpression => deriveSQLDataAccess(sub.plan)
+ case _ => false
+ }.isDefined)
+ }.isDefined
+
+ if (containsSQL.contains(true) && readsSQLData) {
+ throw new AnalysisException(
+ errorClass = "INVALID_SQL_FUNCTION_DATA_ACCESS",
+ messageParameters = Map.empty
+ )
+ }
+
+ readsSQLData
+ }
+
+ /**
+ * Generate the function properties, including:
+ * 1. the SQL configs when creating the function.
+ * 2. the catalog and database name when creating the function. This will be used to provide
+ * context during nested function resolution.
+ * 3. referred temporary object names if the function is a temp function.
+ */
+ private def generateFunctionProperties(
+ session: SparkSession,
+ plan: LogicalPlan,
+ analyzed: LogicalPlan): Map[String, String] = {
+ val catalog = session.sessionState.catalog
+ val conf = session.sessionState.conf
+ val manager = session.sessionState.catalogManager
+
+ val tempVars = ViewHelper.collectTemporaryVariables(analyzed)
+
+ sqlConfigsToProps(conf, SQL_CONFIG_PREFIX) ++
+ catalogAndNamespaceToProps(
+ manager.currentCatalog.name,
+ manager.currentNamespace.toIndexedSeq) ++
+ referredTempNamesToProps(Nil, Nil, tempVars)
+ }
+
+ override def simpleString(maxFields: Int): String = {
+ s"CreatePaimonSQLFunctionCommand: $name"
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/paimon/shims/Spark4Shim.scala b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/paimon/shims/Spark4Shim.scala
new file mode 100644
index 000000000000..03e1311d78b9
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.1/src/main/scala/org/apache/spark/sql/paimon/shims/Spark4Shim.scala
@@ -0,0 +1,527 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.spark.sql.paimon.shims
+
+import org.apache.paimon.Snapshot
+import org.apache.paimon.data.variant.{GenericVariant, Variant}
+import org.apache.paimon.spark.catalyst.analysis.Spark4ResolutionRules
+import org.apache.paimon.spark.catalyst.parser.extensions.PaimonSpark4SqlExtensionsParser
+import org.apache.paimon.spark.data.{Spark4ArrayData, Spark4InternalRow, Spark4InternalRowWithBlob, SparkArrayData, SparkInternalRow}
+import org.apache.paimon.spark.format.FormatTableBatchWrite
+import org.apache.paimon.spark.rowops.PaimonCopyOnWriteScan
+import org.apache.paimon.spark.write.{PaimonBatchWrite, PaimonDeltaBatchWrite}
+import org.apache.paimon.table.{FileStoreTable, FormatTable}
+import org.apache.paimon.types.{DataType, RowType}
+
+import org.apache.hadoop.conf.Configuration
+import org.apache.hadoop.fs.Path
+import org.apache.spark.sql.SparkSession
+import org.apache.spark.sql.catalyst.InternalRow
+import org.apache.spark.sql.catalyst.analysis.CTESubstitution
+import org.apache.spark.sql.catalyst.analysis.NamedRelation
+import org.apache.spark.sql.catalyst.catalog.CatalogStorageFormat
+import org.apache.spark.sql.catalyst.expressions.{Attribute, AttributeReference, Expression}
+import org.apache.spark.sql.catalyst.expressions.aggregate.AggregateExpression
+import org.apache.spark.sql.catalyst.parser.ParserInterface
+import org.apache.spark.sql.catalyst.plans.logical.{Aggregate, Assignment, ColumnDefinition, CTERelationRef, DescribeRelation, InsertAction, LogicalPlan, MergeAction, MergeIntoTable, MergeRows, OverwriteByExpression, OverwritePartitionsDynamic, SubqueryAlias, TableSpec, UnresolvedWith, UpdateAction}
+import org.apache.spark.sql.catalyst.plans.logical.MergeRows.{Copy, Insert, Keep, Update}
+import org.apache.spark.sql.catalyst.plans.physical.{ClusteredDistribution, Distribution}
+import org.apache.spark.sql.catalyst.rules.Rule
+import org.apache.spark.sql.catalyst.util.{ArrayData, GeneratedColumn, IdentityColumn, ResolveDefaultColumns, STUtils}
+import org.apache.spark.sql.connector.catalog.{CatalogV2Util, Column, Identifier, StagingTableCatalog, Table, TableCatalog}
+import org.apache.spark.sql.connector.expressions.Transform
+import org.apache.spark.sql.connector.read.Scan
+import org.apache.spark.sql.connector.write.BatchWrite
+import org.apache.spark.sql.execution.{SparkFormatTable, SparkPlan}
+import org.apache.spark.sql.execution.datasources.{PartitioningAwareFileIndex, PartitionSpec}
+import org.apache.spark.sql.execution.datasources.v2.{AtomicReplaceTableAsSelectExec, AtomicReplaceTableExec, CreateTableAsSelectExec, DescribeTableExec, ReplaceTableAsSelectExec, ReplaceTableExec}
+import org.apache.spark.sql.execution.datasources.v2.{DataSourceV2Relation, DataSourceV2ScanRelation}
+import org.apache.spark.sql.execution.streaming.runtime.MetadataLogFileIndex
+import org.apache.spark.sql.execution.streaming.sinks.FileStreamSink
+import org.apache.spark.sql.internal.SQLConf
+import org.apache.spark.sql.types.{DataTypes, Geography, GeographyType, Geometry, GeometryType, StructType, VariantType}
+import org.apache.spark.unsafe.types.VariantVal
+
+import java.net.URI
+import java.util.{Map => JMap}
+
+class Spark4Shim extends SparkShim {
+
+ override def classicApi: ClassicApi = new Classic4Api
+
+ override def createSparkParser(delegate: ParserInterface): ParserInterface = {
+ new PaimonSpark4SqlExtensionsParser(delegate)
+ }
+
+ override def createCustomResolution(spark: SparkSession): Rule[LogicalPlan] = {
+ Spark4ResolutionRules(spark)
+ }
+
+ override def createSparkInternalRow(rowType: RowType): SparkInternalRow = {
+ new Spark4InternalRow(rowType)
+ }
+
+ override def createSparkInternalRowWithBlob(
+ rowType: RowType,
+ blobFields: Set[Int],
+ blobAsDescriptor: Boolean): SparkInternalRow = {
+ new Spark4InternalRowWithBlob(rowType, blobFields, blobAsDescriptor)
+ }
+
+ override def createSparkArrayData(elementType: DataType): SparkArrayData = {
+ new Spark4ArrayData(elementType)
+ }
+
+ override def createTable(
+ tableCatalog: TableCatalog,
+ ident: Identifier,
+ schema: StructType,
+ partitions: Array[Transform],
+ properties: JMap[String, String]): Table = {
+ val columns = CatalogV2Util.structTypeToV2Columns(schema)
+ tableCatalog.createTable(ident, columns, partitions, properties)
+ }
+
+ // Spark 4.0/4.1 have no `CreateTableLike` logical plan; `CREATE TABLE LIKE` still arrives as the
+ // V1 `CreateTableLikeCommand`, which `RewriteCreateTableLikeCommand` matches directly.
+ override def createTableLikeParts(plan: LogicalPlan)
+ : Option[(Seq[String], Seq[String], Option[String], Option[String], Map[String, String], Boolean, Boolean)] =
+ None
+
+ // Spark 3.x/4.0/4.1 keep `DESCRIBE ... PARTITION` inside `DescribeRelation`; see
+ // `describeRelationPartitionSpec`.
+ override def describeTablePartition(
+ plan: LogicalPlan): Option[(LogicalPlan, Map[String, String], Boolean, Seq[Attribute])] = None
+
+ override def describeRelationPartitionSpec(plan: DescribeRelation): Map[String, String] =
+ plan.partitionSpec
+
+ override def createDescribeTableExec(
+ output: Seq[Attribute],
+ catalogName: String,
+ identifier: Identifier,
+ table: Table,
+ isExtended: Boolean): SparkPlan =
+ DescribeTableExec(output, table, isExtended)
+
+ // Spark 4.1 exposes this as `needSchemaEvolution`; 4.2 replaced it with `pendingSchemaChanges`
+ // and 4.0 has neither.
+ override def mergeNeedsSchemaEvolution(merge: MergeIntoTable): Boolean =
+ merge.needSchemaEvolution
+
+ override def withStorageLocation(
+ storage: CatalogStorageFormat,
+ locationUri: Option[URI]): CatalogStorageFormat =
+ storage.copy(locationUri = locationUri)
+
+ override def overwriteByName(
+ table: NamedRelation,
+ query: LogicalPlan,
+ deleteExpr: Expression,
+ writeOptions: Map[String, String]): OverwriteByExpression =
+ OverwriteByExpression.byName(table, query, deleteExpr, writeOptions)
+
+ override def overwritePartitionsDynamicByName(
+ table: NamedRelation,
+ query: LogicalPlan,
+ writeOptions: Map[String, String]): OverwritePartitionsDynamic =
+ OverwritePartitionsDynamic.byName(table, query, writeOptions)
+
+ override def createCreateTableAsSelectExec(
+ catalog: TableCatalog,
+ ident: Identifier,
+ partitioning: Seq[Transform],
+ query: LogicalPlan,
+ tableSpec: TableSpec,
+ writeOptions: Map[String, String],
+ ifNotExists: Boolean): SparkPlan = {
+ CreateTableAsSelectExec(
+ catalog,
+ ident,
+ partitioning,
+ query,
+ tableSpec,
+ writeOptions,
+ ifNotExists)
+ }
+
+ override def createReplaceTableAsSelectExec(
+ catalog: TableCatalog,
+ ident: Identifier,
+ partitioning: Seq[Transform],
+ query: LogicalPlan,
+ tableSpec: TableSpec,
+ writeOptions: Map[String, String],
+ orCreate: Boolean): SparkPlan = {
+ ReplaceTableAsSelectExec(
+ catalog,
+ ident,
+ partitioning,
+ query,
+ tableSpec,
+ writeOptions,
+ orCreate = orCreate,
+ invalidateCache)
+ }
+
+ override def createAtomicReplaceTableAsSelectExec(
+ catalog: StagingTableCatalog,
+ ident: Identifier,
+ partitioning: Seq[Transform],
+ query: LogicalPlan,
+ tableSpec: TableSpec,
+ writeOptions: Map[String, String],
+ orCreate: Boolean): SparkPlan = {
+ AtomicReplaceTableAsSelectExec(
+ catalog,
+ ident,
+ partitioning,
+ query,
+ tableSpec,
+ writeOptions,
+ orCreate = orCreate,
+ invalidateCache)
+ }
+
+ override def createReplaceTableExec(
+ catalog: TableCatalog,
+ ident: Identifier,
+ columns: Array[Column],
+ partitioning: Seq[Transform],
+ tableSpec: TableSpec,
+ orCreate: Boolean): SparkPlan = {
+ ReplaceTableExec(
+ catalog,
+ ident,
+ columns,
+ partitioning,
+ tableSpec,
+ orCreate = orCreate,
+ invalidateCache)
+ }
+
+ override def createAtomicReplaceTableExec(
+ catalog: StagingTableCatalog,
+ ident: Identifier,
+ columns: Array[Column],
+ partitioning: Seq[Transform],
+ tableSpec: TableSpec,
+ orCreate: Boolean): SparkPlan = {
+ AtomicReplaceTableExec(
+ catalog,
+ ident,
+ columns,
+ partitioning,
+ tableSpec,
+ orCreate = orCreate,
+ invalidateCache)
+ }
+
+ override def toReplaceTableColumns(
+ tableSchema: StructType,
+ schemaOrColumns: Any,
+ catalog: TableCatalog,
+ ident: Identifier): Array[Column] = {
+ val statementType = "REPLACE TABLE"
+ val columns = schemaOrColumns.asInstanceOf[Seq[ColumnDefinition]]
+ ResolveDefaultColumns.validateCatalogForDefaultValue(columns, catalog, ident)
+ GeneratedColumn.validateGeneratedColumns(tableSchema, catalog, ident, statementType)
+ IdentityColumn.validateIdentityColumn(tableSchema, catalog, ident)
+ columns.map(_.toV2Column(statementType)).toArray
+ }
+
+ override def copyTableSpec(
+ tableSpec: TableSpec,
+ additionalProperties: Map[String, String],
+ location: Option[String]): TableSpec = {
+ tableSpec.copy(properties = tableSpec.properties ++ additionalProperties, location = location)
+ }
+
+ private def invalidateCache(tableCatalog: TableCatalog, ident: Identifier): Unit = {
+ tableCatalog.invalidateTable(ident)
+ }
+
+ override def createPaimonBatchWrite(
+ table: FileStoreTable,
+ writeSchema: StructType,
+ dataSchema: StructType,
+ overwritePartitions: Option[Map[String, String]],
+ copyOnWriteScan: Option[PaimonCopyOnWriteScan],
+ operationType: Option[Snapshot.Operation]): BatchWrite =
+ new PaimonBatchWrite(
+ table,
+ writeSchema,
+ dataSchema,
+ overwritePartitions,
+ copyOnWriteScan,
+ operationType)
+
+ override def createPaimonDeltaBatchWrite(
+ table: FileStoreTable,
+ rowSchema: StructType,
+ rowIdSchema: StructType,
+ operationType: Snapshot.Operation,
+ readSnapshotId: Option[Long]): BatchWrite =
+ new PaimonDeltaBatchWrite(table, rowSchema, rowIdSchema, operationType, readSnapshotId)
+
+ override def createFormatTableBatchWrite(
+ table: FormatTable,
+ overwriteDynamic: Option[Boolean],
+ overwritePartitions: Option[Map[String, String]],
+ writeSchema: StructType): BatchWrite =
+ new FormatTableBatchWrite(table, overwriteDynamic, overwritePartitions, writeSchema)
+
+ override def createCTERelationRef(
+ cteId: Long,
+ resolved: Boolean,
+ output: Seq[Attribute],
+ isStreaming: Boolean): CTERelationRef = {
+ CTERelationRef(cteId, resolved, output.toSeq, isStreaming)
+ }
+
+ override def createClusteredDistribution(
+ expressions: Seq[Expression],
+ numPartitions: Int): Distribution =
+ ClusteredDistribution(
+ expressions,
+ requireAllClusterKeys = false,
+ requiredNumPartitions = Some(numPartitions))
+
+ override def supportsHashAggregate(
+ aggregateBufferAttributes: Seq[Attribute],
+ groupingExpression: Seq[Expression]): Boolean = {
+ Aggregate.supportsHashAggregate(aggregateBufferAttributes.toSeq, groupingExpression.toSeq)
+ }
+
+ override def supportsObjectHashAggregate(
+ aggregateExpressions: Seq[AggregateExpression],
+ groupByExpressions: Seq[Expression]): Boolean =
+ Aggregate.supportsObjectHashAggregate(aggregateExpressions.toSeq, groupByExpressions.toSeq)
+
+ override def createMergeIntoTable(
+ targetTable: LogicalPlan,
+ sourceTable: LogicalPlan,
+ mergeCondition: Expression,
+ matchedActions: Seq[MergeAction],
+ notMatchedActions: Seq[MergeAction],
+ notMatchedBySourceActions: Seq[MergeAction],
+ withSchemaEvolution: Boolean): MergeIntoTable = {
+ MergeIntoTable(
+ targetTable,
+ sourceTable,
+ mergeCondition,
+ matchedActions,
+ notMatchedActions,
+ notMatchedBySourceActions,
+ withSchemaEvolution)
+ }
+
+ override def notMatchedBySourceActions(merge: MergeIntoTable): Seq[MergeAction] =
+ merge.notMatchedBySourceActions
+
+ override def createUpdateAction(
+ condition: Option[Expression],
+ assignments: Seq[Assignment]): UpdateAction =
+ UpdateAction(condition, assignments)
+
+ override def createInsertAction(
+ condition: Option[Expression],
+ assignments: Seq[Assignment]): InsertAction =
+ InsertAction(condition, assignments)
+
+ override def copyDataSourceV2Relation(
+ relation: DataSourceV2Relation,
+ table: Table,
+ output: Seq[AttributeReference]): DataSourceV2Relation = {
+ relation.copy(table = table, output = output)
+ }
+
+ override def createDataSourceV2ScanRelation(
+ relation: DataSourceV2ScanRelation,
+ scan: Scan,
+ output: Seq[AttributeReference]): DataSourceV2ScanRelation = {
+ DataSourceV2ScanRelation(relation.relation, scan, output, None, None)
+ }
+
+ override def createClusteredDistribution(
+ expressions: Seq[Expression],
+ requiredNumPartitions: Option[Int]): Distribution = {
+ ClusteredDistribution(expressions, requiredNumPartitions = requiredNumPartitions)
+ }
+
+ override def earlyBatchRules(): Seq[Rule[LogicalPlan]] = Seq(CTESubstitution)
+
+ override def mergeRowsKeepCopy(condition: Expression, output: Seq[Expression]): AnyRef =
+ Keep(Copy, condition, output)
+
+ override def mergeRowsKeepUpdate(condition: Expression, output: Seq[Expression]): AnyRef =
+ Keep(Update, condition, output)
+
+ override def mergeRowsKeepInsert(condition: Expression, output: Seq[Expression]): AnyRef =
+ Keep(Insert, condition, output)
+
+ override def transformUnresolvedWithCteRelations(
+ u: UnresolvedWith,
+ transform: SubqueryAlias => SubqueryAlias): UnresolvedWith = {
+ u.copy(cteRelations = u.cteRelations.map {
+ case (name, alias, depth) => (name, transform(alias), depth)
+ })
+ }
+
+ override def hasFileStreamSinkMetadata(
+ paths: Seq[String],
+ hadoopConf: Configuration,
+ sqlConf: SQLConf): Boolean = {
+ FileStreamSink.hasMetadata(paths, hadoopConf, sqlConf)
+ }
+
+ override def createPartitionedMetadataLogFileIndex(
+ sparkSession: SparkSession,
+ path: Path,
+ parameters: Map[String, String],
+ userSpecifiedSchema: Option[StructType],
+ partitionSchema: StructType): PartitioningAwareFileIndex = {
+ new Spark4Shim.PartitionedMetadataLogFileIndex(
+ sparkSession,
+ path,
+ parameters,
+ userSpecifiedSchema,
+ partitionSchema)
+ }
+
+ override def toPaimonVariant(o: Object): Variant = {
+ val v = o.asInstanceOf[VariantVal]
+ new GenericVariant(v.getValue, v.getMetadata)
+ }
+
+ override def toPaimonVariant(row: InternalRow, pos: Int): Variant = {
+ val v = row.getVariant(pos)
+ new GenericVariant(v.getValue, v.getMetadata)
+ }
+
+ override def toPaimonVariant(array: ArrayData, pos: Int): Variant = {
+ val v = array.getVariant(pos)
+ new GenericVariant(v.getValue, v.getMetadata)
+ }
+
+ override def isSparkVariantType(dataType: org.apache.spark.sql.types.DataType): Boolean =
+ dataType.isInstanceOf[VariantType]
+
+ override def SparkVariantType(): org.apache.spark.sql.types.DataType = DataTypes.VariantType
+
+ override def toPaimonGeometry(o: Object): Array[Byte] =
+ o.asInstanceOf[Geometry].getBytes
+
+ // Spark 4.1 shape. 4.2 (SPARK-57058) replaced `SpecializedGetters`' `getGeometry` /
+ // `getGeography` with a single `getBinaryView` and split `STUtils.stAsBinary` into
+ // `stGeomAsBinary` / `stGeogAsBinary`; `paimon-spark4-common` carries that version.
+ override def toPaimonGeometry(row: InternalRow, pos: Int): Array[Byte] =
+ STUtils.stAsBinary(row.getGeometry(pos))
+
+ override def toPaimonGeometry(array: ArrayData, pos: Int): Array[Byte] =
+ STUtils.stAsBinary(array.getGeometry(pos))
+
+ override def toPaimonGeography(o: Object): Array[Byte] =
+ o.asInstanceOf[Geography].getBytes
+
+ override def toPaimonGeography(row: InternalRow, pos: Int): Array[Byte] =
+ STUtils.stAsBinary(row.getGeography(pos))
+
+ override def toPaimonGeography(array: ArrayData, pos: Int): Array[Byte] =
+ STUtils.stAsBinary(array.getGeography(pos))
+
+ override def toSparkGeometry(wkb: Array[Byte], crs: String): Object = {
+ val geometryType = sparkGeometryType(crs)
+ STUtils.stGeomFromWKB(wkb, geometryType.srid)
+ }
+
+ override def toSparkGeography(wkb: Array[Byte], crs: String, algorithm: String): Object = {
+ val geographyType = sparkGeographyType(crs, algorithm)
+ STUtils.stSetSrid(STUtils.stGeogFromWKB(wkb), geographyType.srid)
+ }
+
+ override def isSparkGeometryType(dataType: org.apache.spark.sql.types.DataType): Boolean =
+ dataType.isInstanceOf[GeometryType]
+
+ override def isSparkGeographyType(dataType: org.apache.spark.sql.types.DataType): Boolean =
+ dataType.isInstanceOf[GeographyType]
+
+ override def SparkGeometryType(crs: String): org.apache.spark.sql.types.DataType =
+ sparkGeometryType(crs)
+
+ override def SparkGeographyType(
+ crs: String,
+ algorithm: String): org.apache.spark.sql.types.DataType = sparkGeographyType(crs, algorithm)
+
+ override def sparkGeometryCrs(dataType: org.apache.spark.sql.types.DataType): String = {
+ val geometryType = dataType.asInstanceOf[GeometryType]
+ require(!geometryType.isMixedSrid, "Paimon does not support mixed-SRID geometry values")
+ geometryType.crs
+ }
+
+ override def sparkGeographyCrs(dataType: org.apache.spark.sql.types.DataType): String = {
+ val geographyType = dataType.asInstanceOf[GeographyType]
+ require(!geographyType.isMixedSrid, "Paimon does not support mixed-SRID geography values")
+ geographyType.crs
+ }
+
+ override def sparkGeographyAlgorithm(dataType: org.apache.spark.sql.types.DataType): String =
+ dataType.asInstanceOf[GeographyType].algorithm.toString
+
+ private def sparkGeometryType(crs: String): GeometryType = {
+ val geometryType = GeometryType(crs)
+ require(!geometryType.isMixedSrid, "Paimon does not support mixed-SRID geometry values")
+ geometryType
+ }
+
+ private def sparkGeographyType(crs: String, algorithm: String): GeographyType = {
+ val geographyType = GeographyType(crs, algorithm)
+ require(!geographyType.isMixedSrid, "Paimon does not support mixed-SRID geography values")
+ geographyType
+ }
+
+ // SQL UDFs (CREATE FUNCTION ... RETURN ...).
+ override def rewritePaimonSQLFunctionCommands(spark: SparkSession): Rule[LogicalPlan] =
+ org.apache.spark.sql.catalyst.parser.extensions.RewritePaimonSQLFunctionCommands(spark)
+
+ override def resolvePaimonSQLFunction(
+ funcIdent: org.apache.spark.sql.catalyst.FunctionIdentifier,
+ function: org.apache.paimon.function.Function,
+ arguments: Seq[Expression],
+ parser: org.apache.spark.sql.catalyst.parser.ParserInterface): Expression =
+ org.apache.paimon.spark.catalog.functions.SQLFunctionConverter
+ .toSQLFunctionExpression(funcIdent, function, arguments, parser)
+}
+
+object Spark4Shim {
+
+ /** Paimon's partition-aware wrapper over Spark's `MetadataLogFileIndex`. */
+ private[shims] class PartitionedMetadataLogFileIndex(
+ sparkSession: SparkSession,
+ path: Path,
+ parameters: Map[String, String],
+ userSpecifiedSchema: Option[StructType],
+ override val partitionSchema: StructType)
+ extends MetadataLogFileIndex(sparkSession, path, parameters, userSpecifiedSchema) {
+
+ override def partitionSpec(): PartitionSpec = {
+ SparkFormatTable.alignPartitionSpec(super.partitionSpec(), partitionSchema)
+ }
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.2/pom.xml b/paimon-spark/paimon-spark-4.2/pom.xml
new file mode 100644
index 000000000000..cc1c1ae04c07
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/pom.xml
@@ -0,0 +1,186 @@
+
+
+
+ 4.0.0
+
+
+ org.apache.paimon
+ paimon-spark
+ 2.1-SNAPSHOT
+
+
+ paimon-spark-4.2_2.13
+ Paimon : Spark : 4.2 : 2.13
+
+
+ 4.2.0
+
+ 2.13
+
+
+
+
+ org.apache.paimon
+ paimon-format
+
+
+
+ org.apache.paimon
+ paimon-spark4-common_${scala.binary.version}
+ ${project.version}
+
+
+
+ org.apache.paimon
+ paimon-spark-common_${scala.binary.version}
+ ${project.version}
+
+
+
+
+
+ org.apache.spark
+ spark-sql_${scala.binary.version}
+ ${spark.version}
+
+
+
+ org.apache.spark
+ spark-core_${scala.binary.version}
+ ${spark.version}
+
+
+
+ org.apache.spark
+ spark-catalyst_${scala.binary.version}
+ ${spark.version}
+
+
+
+ org.apache.spark
+ spark-hive_${scala.binary.version}
+ ${spark.version}
+
+
+
+
+
+
+ org.apache.paimon
+ paimon-spark-ut_${scala.binary.version}
+ ${project.version}
+ tests
+ test
+
+
+ *
+ *
+
+
+
+
+
+ org.apache.paimon
+ paimon-spark4-common_${scala.binary.version}
+ ${project.version}
+ tests
+ test
+
+
+ *
+ *
+
+
+
+
+
+ org.apache.spark
+ spark-sql_${scala.binary.version}
+ ${spark.version}
+ tests
+ test
+
+
+ org.apache.spark
+ spark-connect-shims_${scala.binary.version}
+
+
+
+
+
+ org.apache.spark
+ spark-catalyst_${scala.binary.version}
+ ${spark.version}
+ tests
+ test
+
+
+
+ org.apache.spark
+ spark-core_${scala.binary.version}
+ ${spark.version}
+ tests
+ test
+
+
+
+
+
+
+ org.apache.maven.plugins
+ maven-shade-plugin
+
+
+ shade-paimon
+ package
+
+ shade
+
+
+
+
+ *
+
+ com/github/luben/zstd/**
+ **/*libzstd-jni-*.so
+ **/*libzstd-jni-*.dll
+
+
+
+
+
+ org.apache.paimon:paimon-spark4-common_${scala.binary.version}
+
+
+
+
+
+
+
+
+
diff --git a/paimon-spark/paimon-spark-4.2/src/main/scala/org/apache/paimon/spark/read/PaimonSupportsPushDownVariantExtractions.scala b/paimon-spark/paimon-spark-4.2/src/main/scala/org/apache/paimon/spark/read/PaimonSupportsPushDownVariantExtractions.scala
new file mode 100644
index 000000000000..6499bbf043a5
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/main/scala/org/apache/paimon/spark/read/PaimonSupportsPushDownVariantExtractions.scala
@@ -0,0 +1,52 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.read
+
+import org.apache.paimon.spark.SparkTypeUtils
+
+import org.apache.spark.sql.connector.read.{SupportsPushDownVariantExtractions, VariantExtraction}
+import org.apache.spark.sql.execution.datasources.VariantMetadata
+import org.apache.spark.sql.types.VariantType
+
+/**
+ * Spark 4.1+ binding; shadows the no-op `paimon-spark-common` trait by FQN.
+ *
+ * Duplicated in `paimon-spark-4.1` and `paimon-spark-4.2` rather than living in
+ * `paimon-spark4-common`, because `paimon-spark-4.0` must keep the no-op: Spark 4.0 has no
+ * `SupportsPushDownVariantExtractions`, so a module-level binding would fail to link there.
+ */
+trait PaimonSupportsPushDownVariantExtractions extends SupportsPushDownVariantExtractions {
+ protected var acceptedVariantExtractions: Map[Seq[String], Seq[VariantExtractionInfo]] = Map.empty
+
+ override def pushVariantExtractions(extractions: Array[VariantExtraction]): Array[Boolean] = {
+ val decoded = extractions.iterator.map {
+ ex =>
+ val vm = VariantMetadata.fromMetadata(ex.metadata())
+ val info = VariantExtractionInfo(
+ paimonType = SparkTypeUtils.toPaimonType(ex.expectedDataType()),
+ path = vm.path,
+ failOnError = vm.failOnError,
+ timeZoneId = vm.timeZoneId)
+ (ex.columnName().toSeq, info, ex.expectedDataType() == VariantType)
+ }.toIndexedSeq
+ val (newMap, accepted) = VariantPushDownUtils.acceptByPath(decoded)
+ acceptedVariantExtractions = newMap
+ accepted
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/procedure/CompactProcedureTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/procedure/CompactProcedureTest.scala
new file mode 100644
index 000000000000..322d50a62127
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/procedure/CompactProcedureTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.procedure
+
+class CompactProcedureTest extends CompactProcedureTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/procedure/ProcedureTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/procedure/ProcedureTest.scala
new file mode 100644
index 000000000000..d57846709877
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/procedure/ProcedureTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.procedure
+
+class ProcedureTest extends ProcedureTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/AnalyzeTableTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/AnalyzeTableTest.scala
new file mode 100644
index 000000000000..255906d04bf2
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/AnalyzeTableTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class AnalyzeTableTest extends AnalyzeTableTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/BlobUpdateTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/BlobUpdateTest.scala
new file mode 100644
index 000000000000..b190abbc912c
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/BlobUpdateTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class BlobUpdateTest extends BlobUpdateTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/CopyIntoTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/CopyIntoTest.scala
new file mode 100644
index 000000000000..e7eb9a0d9516
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/CopyIntoTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class CopyIntoTest extends CopyIntoTestBase with CopyIntoOnErrorTest {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DDLTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DDLTest.scala
new file mode 100644
index 000000000000..b729f57b33e7
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DDLTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class DDLTest extends DDLTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DDLWithHiveCatalogTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DDLWithHiveCatalogTest.scala
new file mode 100644
index 000000000000..cb139d2a57be
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DDLWithHiveCatalogTest.scala
@@ -0,0 +1,23 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class DDLWithHiveCatalogTest extends DDLWithHiveCatalogTestBase {}
+
+class DefaultDatabaseTest extends DefaultDatabaseTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DataEvolutionDeletionTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DataEvolutionDeletionTest.scala
new file mode 100644
index 000000000000..43aeb915dbe4
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DataEvolutionDeletionTest.scala
@@ -0,0 +1,33 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+import org.apache.spark.SparkConf
+
+class DataEvolutionDeletionTest extends DataEvolutionDeletionTestBase {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "false")
+ }
+}
+
+class V2DataEvolutionDeletionTest extends DataEvolutionDeletionTestBase {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "true")
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DataFrameWriteTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DataFrameWriteTest.scala
new file mode 100644
index 000000000000..6170e2fd6c5c
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DataFrameWriteTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class DataFrameWriteTest extends DataFrameWriteTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DeleteFromTableTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DeleteFromTableTest.scala
new file mode 100644
index 000000000000..8d620ece8245
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DeleteFromTableTest.scala
@@ -0,0 +1,33 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+import org.apache.spark.SparkConf
+
+class DeleteFromTableTest extends DeleteFromTableTestBase {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "false")
+ }
+}
+
+class V2DeleteFromTableTest extends DeleteFromTableTestBase {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "true")
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DescribeTableTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DescribeTableTest.scala
new file mode 100644
index 000000000000..c6aa77419241
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/DescribeTableTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class DescribeTableTest extends DescribeTableTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/FormatTableTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/FormatTableTest.scala
new file mode 100644
index 000000000000..ba49976ab6c0
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/FormatTableTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class FormatTableTest extends FormatTableTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/InsertOverwriteTableTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/InsertOverwriteTableTest.scala
new file mode 100644
index 000000000000..4f66584c303b
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/InsertOverwriteTableTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class InsertOverwriteTableTest extends InsertOverwriteTableTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/MapSelectedKeysSharedShreddingE2ETest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/MapSelectedKeysSharedShreddingE2ETest.scala
new file mode 100644
index 000000000000..1bdf43ce20c2
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/MapSelectedKeysSharedShreddingE2ETest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class MapSelectedKeysSharedShreddingE2ETest extends MapSelectedKeysSharedShreddingE2ETestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/MergeIntoTableTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/MergeIntoTableTest.scala
new file mode 100644
index 000000000000..c8ae09a26be4
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/MergeIntoTableTest.scala
@@ -0,0 +1,103 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+import org.apache.paimon.spark.{PaimonAppendBucketedTableTest, PaimonAppendNonBucketTableTest, PaimonPrimaryKeyBucketedTableTest, PaimonPrimaryKeyNonBucketTableTest}
+
+import org.apache.spark.SparkConf
+
+class MergeIntoPrimaryKeyBucketedTableTest
+ extends MergeIntoTableTestBase
+ with MergeIntoPrimaryKeyTableTest
+ with MergeIntoNotMatchedBySourceTest
+ with PaimonPrimaryKeyBucketedTableTest {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "false")
+ }
+}
+
+class MergeIntoPrimaryKeyNonBucketTableTest
+ extends MergeIntoTableTestBase
+ with MergeIntoPrimaryKeyTableTest
+ with MergeIntoNotMatchedBySourceTest
+ with PaimonPrimaryKeyNonBucketTableTest {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "false")
+ }
+}
+
+class MergeIntoAppendBucketedTableTest
+ extends MergeIntoTableTestBase
+ with MergeIntoAppendTableTest
+ with MergeIntoNotMatchedBySourceTest
+ with PaimonAppendBucketedTableTest {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "false")
+ }
+}
+
+class MergeIntoAppendNonBucketedTableTest
+ extends MergeIntoTableTestBase
+ with MergeIntoAppendTableTest
+ with MergeIntoNotMatchedBySourceTest
+ with PaimonAppendNonBucketTableTest {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "false")
+ }
+}
+
+class V2MergeIntoPrimaryKeyBucketedTableTest
+ extends MergeIntoTableTestBase
+ with MergeIntoPrimaryKeyTableTest
+ with MergeIntoNotMatchedBySourceTest
+ with PaimonPrimaryKeyBucketedTableTest {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "true")
+ }
+}
+
+class V2MergeIntoPrimaryKeyNonBucketTableTest
+ extends MergeIntoTableTestBase
+ with MergeIntoPrimaryKeyTableTest
+ with MergeIntoNotMatchedBySourceTest
+ with PaimonPrimaryKeyNonBucketTableTest {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "true")
+ }
+}
+
+class V2MergeIntoAppendBucketedTableTest
+ extends MergeIntoTableTestBase
+ with MergeIntoAppendTableTest
+ with MergeIntoNotMatchedBySourceTest
+ with PaimonAppendBucketedTableTest {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "true")
+ }
+}
+
+class V2MergeIntoAppendNonBucketedTableTest
+ extends MergeIntoTableTestBase
+ with MergeIntoAppendTableTest
+ with MergeIntoNotMatchedBySourceTest
+ with PaimonAppendNonBucketTableTest {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "true")
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonCompositePartitionKeyTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonCompositePartitionKeyTest.scala
new file mode 100644
index 000000000000..635185a9ed0e
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonCompositePartitionKeyTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class PaimonCompositePartitionKeyTest extends PaimonCompositePartitionKeyTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonOptimizationTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonOptimizationTest.scala
new file mode 100644
index 000000000000..ec140a89bbd3
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonOptimizationTest.scala
@@ -0,0 +1,39 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+import org.apache.spark.sql.catalyst.dsl.expressions._
+import org.apache.spark.sql.catalyst.expressions.{Attribute, GetStructField, NamedExpression, ScalarSubquery}
+import org.apache.spark.sql.paimon.shims.SparkShimLoader
+
+class PaimonOptimizationTest extends PaimonOptimizationTestBase {
+
+ override def extractorExpression(
+ cteIndex: Int,
+ output: Seq[Attribute],
+ fieldIndex: Int): NamedExpression = {
+ GetStructField(
+ ScalarSubquery(
+ SparkShimLoader.shim
+ .createCTERelationRef(cteIndex, resolved = true, output.toSeq, isStreaming = false)),
+ fieldIndex,
+ None)
+ .as("scalarsubquery()")
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonPushDownTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonPushDownTest.scala
new file mode 100644
index 000000000000..26677d85c71a
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonPushDownTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class PaimonPushDownTest extends PaimonPushDownTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonSQLFunctionTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonSQLFunctionTest.scala
new file mode 100644
index 000000000000..967795d4393e
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonSQLFunctionTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class PaimonSQLFunctionTest extends PaimonSQLFunctionTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonV1FunctionTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonV1FunctionTest.scala
new file mode 100644
index 000000000000..f37fbad27033
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonV1FunctionTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class PaimonV1FunctionTest extends PaimonV1FunctionTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonViewTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonViewTest.scala
new file mode 100644
index 000000000000..6ab8a2671b51
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/PaimonViewTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class PaimonViewTest extends PaimonViewTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/RowIdPushDownTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/RowIdPushDownTest.scala
new file mode 100644
index 000000000000..da4c9b854df3
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/RowIdPushDownTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class RowIdPushDownTest extends RowIdPushDownTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/RowTrackingTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/RowTrackingTest.scala
new file mode 100644
index 000000000000..382aa1e77880
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/RowTrackingTest.scala
@@ -0,0 +1,128 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+import org.apache.paimon.spark.SparkTable
+import org.apache.paimon.spark.schema.PaimonMetadataColumn
+
+import org.apache.spark.sql.Row
+import org.apache.spark.sql.connector.catalog.SupportsRowLevelOperations
+import org.apache.spark.sql.types.Metadata
+
+class RowTrackingTest extends RowTrackingTestBase {
+
+ test("Row Tracking: metadata columns expose Spark preserve flags") {
+ val rowIdMetadata = Metadata.fromJson(PaimonMetadataColumn.ROW_ID.metadataInJSON())
+ assert(rowIdMetadata.getBoolean("__preserve_on_delete"))
+ assert(rowIdMetadata.getBoolean("__preserve_on_update"))
+ assert(!rowIdMetadata.getBoolean("__preserve_on_reinsert"))
+
+ val sequenceNumberMetadata =
+ Metadata.fromJson(PaimonMetadataColumn.SEQUENCE_NUMBER.metadataInJSON())
+ assert(sequenceNumberMetadata.getBoolean("__preserve_on_delete"))
+ assert(!sequenceNumberMetadata.getBoolean("__preserve_on_update"))
+ assert(!sequenceNumberMetadata.getBoolean("__preserve_on_reinsert"))
+ }
+
+ test("Row Tracking: Spark 4.1 uses V2 copy-on-write for DML") {
+ withSparkSQLConf("spark.paimon.write.use-v2-write" -> "true") {
+ withTable("s", "t") {
+ sql("CREATE TABLE t (id INT, data INT) TBLPROPERTIES ('row-tracking.enabled' = 'true')")
+ sql("INSERT INTO t VALUES (1, 1), (2, 2)")
+ sql("INSERT INTO t VALUES (3, 3), (4, 4)")
+
+ assertPlanContains("DELETE FROM t WHERE id = 2", "ReplaceData")
+ sql("DELETE FROM t WHERE id = 2")
+
+ assertPlanContains("UPDATE t SET data = 30 WHERE id = 3", "ReplaceData")
+ sql("UPDATE t SET data = 30 WHERE id = 3")
+
+ sql("CREATE TABLE s (id INT, data INT)")
+ sql("INSERT INTO s VALUES (3, 300), (5, 500)")
+ assertPlanContains(
+ """
+ |MERGE INTO t
+ |USING s
+ |ON t.id = s.id
+ |WHEN MATCHED THEN UPDATE SET data = s.data
+ |WHEN NOT MATCHED THEN INSERT *
+ |""".stripMargin,
+ "ReplaceData"
+ )
+ sql("""
+ |MERGE INTO t
+ |USING s
+ |ON t.id = s.id
+ |WHEN MATCHED THEN UPDATE SET data = s.data
+ |WHEN NOT MATCHED THEN INSERT *
+ |""".stripMargin)
+
+ checkAnswer(
+ sql("SELECT *, _ROW_ID, _SEQUENCE_NUMBER FROM t ORDER BY id"),
+ Seq(Row(1, 1, 0, 1), Row(3, 300, 2, 5), Row(4, 4, 3, 2), Row(5, 500, 4, 5))
+ )
+ }
+ }
+ }
+
+ test("Row Tracking: nested CHAR columns do not expose V2 row-level capability") {
+ withSparkSQLConf("spark.paimon.write.use-v2-write" -> "true") {
+ withTable("t") {
+ sql("""
+ |CREATE TABLE t (
+ | id INT,
+ | info STRUCT
+ |) TBLPROPERTIES ('row-tracking.enabled' = 'true')
+ |""".stripMargin)
+
+ assert(!SparkTable.of(loadTable("t")).isInstanceOf[SupportsRowLevelOperations])
+ }
+ }
+ }
+
+ test("Row Tracking: Spark 4.1 restores metadata-only delete fast path") {
+ withSparkSQLConf("spark.paimon.write.use-v2-write" -> "true") {
+ withTable("t") {
+ sql("""
+ |CREATE TABLE t (id INT, data INT, dt STRING)
+ |PARTITIONED BY (dt)
+ |TBLPROPERTIES ('row-tracking.enabled' = 'true')
+ |""".stripMargin)
+ sql("INSERT INTO t VALUES (1, 1, 'p1'), (2, 2, 'p1'), (3, 3, 'p2')")
+
+ assertPlanContains("DELETE FROM t WHERE dt = 'p1'", "DeleteFromPaimonTableCommand")
+ sql("DELETE FROM t WHERE dt = 'p1'")
+
+ checkAnswer(
+ sql("SELECT *, _ROW_ID, _SEQUENCE_NUMBER FROM t ORDER BY id"),
+ Seq(Row(3, 3, "p2", 0, 1))
+ )
+ }
+ }
+ }
+
+ private def assertPlanContains(sqlText: String, fragment: String): Unit = {
+ val plan = explain(sqlText)
+ assert(plan.contains(fragment), plan)
+ }
+
+ private def explain(sqlText: String): String = {
+ sql(s"EXPLAIN EXTENDED $sqlText").collect().map(_.getString(0)).mkString("\n")
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/ShowColumnsTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/ShowColumnsTest.scala
new file mode 100644
index 000000000000..6601dc2fca37
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/ShowColumnsTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class ShowColumnsTest extends PaimonShowColumnsTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/SparkV2FilterConverterTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/SparkV2FilterConverterTest.scala
new file mode 100644
index 000000000000..21c4c8a495ed
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/SparkV2FilterConverterTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class SparkV2FilterConverterTest extends SparkV2FilterConverterTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/TagDdlTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/TagDdlTest.scala
new file mode 100644
index 000000000000..92309d54167b
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/TagDdlTest.scala
@@ -0,0 +1,21 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+class TagDdlTest extends PaimonTagDdlTestBase {}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/UpdateTableTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/UpdateTableTest.scala
new file mode 100644
index 000000000000..3a0f56cd4820
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/UpdateTableTest.scala
@@ -0,0 +1,33 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+import org.apache.spark.SparkConf
+
+class UpdateTableTest extends UpdateTableTestBase {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "false")
+ }
+}
+
+class V2UpdateTableTest extends UpdateTableTestBase {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf.set("spark.paimon.write.use-v2-write", "true")
+ }
+}
diff --git a/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/VariantTest.scala b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/VariantTest.scala
new file mode 100644
index 000000000000..368020422cf6
--- /dev/null
+++ b/paimon-spark/paimon-spark-4.2/src/test/scala/org/apache/paimon/spark/sql/VariantTest.scala
@@ -0,0 +1,53 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark.sql
+
+import org.apache.spark.SparkConf
+
+class VariantTest extends VariantTestBase {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf
+ .set("spark.paimon.variant.inferShreddingSchema", "false")
+ .set("spark.sql.variant.pushVariantIntoScan", "false")
+ }
+}
+
+class VariantInferShreddingTest extends VariantTestBase {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf
+ .set("spark.paimon.variant.inferShreddingSchema", "true")
+ .set("spark.sql.variant.pushVariantIntoScan", "false")
+ }
+}
+
+class VariantWithPushDownTest extends VariantTestBase {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf
+ .set("spark.paimon.variant.inferShreddingSchema", "false")
+ .set("spark.sql.variant.pushVariantIntoScan", "true")
+ }
+}
+
+class VariantInferShreddingWithPushDownTest extends VariantTestBase {
+ override protected def sparkConf: SparkConf = {
+ super.sparkConf
+ .set("spark.paimon.variant.inferShreddingSchema", "true")
+ .set("spark.sql.variant.pushVariantIntoScan", "true")
+ }
+}
diff --git a/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/RollbackStagedTable.java b/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/RollbackStagedTable.java
index 2f90545b6030..0a5f79642a53 100644
--- a/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/RollbackStagedTable.java
+++ b/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/RollbackStagedTable.java
@@ -25,6 +25,7 @@
import org.apache.spark.sql.connector.catalog.Table;
import org.apache.spark.sql.connector.catalog.TableCapability;
import org.apache.spark.sql.connector.expressions.Transform;
+import org.apache.spark.sql.connector.metric.CustomTaskMetric;
import org.apache.spark.sql.connector.read.ScanBuilder;
import org.apache.spark.sql.connector.write.LogicalWriteInfo;
import org.apache.spark.sql.connector.write.WriteBuilder;
@@ -98,6 +99,14 @@ public Set capabilities() {
return table.capabilities();
}
+ // Spark 4.2 (SPARK-56598) added a default reportDriverMetrics() to TruncatableTable, which
+ // collides with the one StagedTable has carried since 4.0 (SPARK-50285). Java requires an
+ // explicit override to disambiguate. No @Override annotation: neither interface declares this
+ // method on Spark 3.5, where the annotation would fail to compile.
+ public CustomTaskMetric[] reportDriverMetrics() {
+ return new CustomTaskMetric[] {};
+ }
+
@Override
public void deleteWhere(Filter[] filters) {
call(SupportsDelete.class, t -> t.deleteWhere(filters));
diff --git a/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/SparkUtils.java b/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/SparkUtils.java
index 89e20dd95877..834caeba4318 100644
--- a/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/SparkUtils.java
+++ b/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/SparkUtils.java
@@ -30,6 +30,7 @@
import org.apache.spark.sql.connector.catalog.CatalogManager;
import org.apache.spark.sql.connector.catalog.CatalogPlugin;
import org.apache.spark.sql.connector.catalog.Identifier;
+import org.apache.spark.sql.paimon.shims.SparkVersionCompat;
import java.util.List;
import java.util.function.BiFunction;
@@ -115,8 +116,8 @@ public static CatalogAndIdentifier catalogAndIdentifier(
CatalogManager catalogManager = spark.sessionState().catalogManager();
String[] currentNamespace;
- if (defaultCatalog.equals(catalogManager.currentCatalog())) {
- currentNamespace = catalogManager.currentNamespace();
+ if (defaultCatalog.equals(SparkVersionCompat.currentCatalog(catalogManager))) {
+ currentNamespace = SparkVersionCompat.currentNamespace(catalogManager);
} else {
currentNamespace = defaultCatalog.defaultNamespace();
}
@@ -126,7 +127,7 @@ public static CatalogAndIdentifier catalogAndIdentifier(
nameParts,
catalogName -> {
try {
- return catalogManager.catalog(catalogName);
+ return SparkVersionCompat.catalog(catalogManager, catalogName);
} catch (Exception e) {
return null;
}
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/PaimonSparkMicroBatchMetadata.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/PaimonSparkMicroBatchMetadata.scala
index cc5b94fe6b4c..0dd750875978 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/PaimonSparkMicroBatchMetadata.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/PaimonSparkMicroBatchMetadata.scala
@@ -27,6 +27,7 @@ import org.apache.spark.sql.Dataset
import org.apache.spark.sql.connector.read.InputPartition
import org.apache.spark.sql.execution.datasources.v2.DataSourceRDD
+import java.lang.reflect.Method
import java.util.{IdentityHashMap, List => JList, Map => JMap, Optional, UUID}
import scala.util.control.NonFatal
@@ -66,30 +67,25 @@ object PaimonSparkMicroBatchMetadata {
var occurrenceOnly: PaimonMicroBatchMetadata = null
var inputCount = 0
var valid = true
- val partitions = dataSourceRDD.partitions
- var partitionIndex = 0
-
- while (valid && partitionIndex < partitions.length) {
- val inputs = dataSourceInputPartitions(partitions(partitionIndex)).iterator
- while (valid && inputs.hasNext) {
- inputs.next() match {
- case input: PaimonMicroBatchInputPartition =>
- val current = input.metadata
- if (current eq null) {
- valid = false
- } else if (occurrenceOnly eq null) {
- occurrenceOnly = current
- inputCount += 1
- } else if ((occurrenceOnly eq current) || occurrenceOnly == current) {
- inputCount += 1
- } else {
- valid = false
- }
- case _: PaimonInputPartition => valid = false
- case _ =>
- }
+ val inputs = dataSourceInputPartitions(dataSourceRDD)
+
+ while (valid && inputs.hasNext) {
+ inputs.next() match {
+ case input: PaimonMicroBatchInputPartition =>
+ val current = input.metadata
+ if (current eq null) {
+ valid = false
+ } else if (occurrenceOnly eq null) {
+ occurrenceOnly = current
+ inputCount += 1
+ } else if ((occurrenceOnly eq current) || occurrenceOnly == current) {
+ inputCount += 1
+ } else {
+ valid = false
+ }
+ case _: PaimonInputPartition => valid = false
+ case _ =>
}
- partitionIndex += 1
}
if (!valid || ((occurrenceOnly ne null) && inputCount != occurrenceOnly.splitCount)) {
@@ -134,39 +130,61 @@ object PaimonSparkMicroBatchMetadata {
}
}
- private def dataSourceInputPartitions(partition: Partition): Seq[InputPartition] = {
- if (partition == null) {
- throw new IllegalArgumentException("Data source RDD partition must not be null.")
+ /**
+ * The input partitions of every `DataSourceRDDPartition` of `rdd`, lazily. All partitions of one
+ * RDD are the same class, so the accessor is resolved once for the whole RDD.
+ */
+ private def dataSourceInputPartitions(rdd: DataSourceRDD): Iterator[InputPartition] = {
+ val partitions = rdd.partitions
+ if (partitions.isEmpty) {
+ Iterator.empty
+ } else {
+ val accessor = inputPartitionAccessor(partitions.head)
+ partitions.iterator.flatMap {
+ partition =>
+ requireNonNullPartition(partition)
+ normalizeInputPartitions(accessor.invoke(partition))
+ }
}
+ }
- val pluralMethod =
- try {
- Some(partition.getClass.getMethod("inputPartitions"))
- } catch {
- case _: NoSuchMethodException => None
- }
+ private def inputPartitionAccessor(partition: Partition): Method = {
+ requireNonNullPartition(partition)
+ methodOf(partition, "inputPartitions")
+ .orElse(methodOf(partition, "inputPartition"))
+ .getOrElse(throw new IllegalArgumentException(
+ s"No input partition accessor on ${partition.getClass.getName}."))
+ }
- pluralMethod match {
- case Some(method) => requireInputPartitions(method.invoke(partition))
- case None =>
- Seq(requireInputPartition(partition.getClass.getMethod("inputPartition").invoke(partition)))
+ private def requireNonNullPartition(partition: Partition): Unit = {
+ if (partition == null) {
+ throw new IllegalArgumentException("Data source RDD partition must not be null.")
}
}
- private def requireInputPartitions(value: Any): Seq[InputPartition] =
- value match {
- case null => throw new IllegalArgumentException("Input partitions must not be null.")
- case values: scala.collection.Seq[_] =>
- values.iterator.map(requireInputPartition).toVector
- case other =>
- throw new IllegalArgumentException(
- s"Unexpected input partitions type ${other.getClass.getName}.")
+ private def methodOf(partition: Partition, name: String): Option[Method] =
+ try {
+ Some(partition.getClass.getMethod(name))
+ } catch {
+ case _: NoSuchMethodException => None
}
- private def requireInputPartition(value: Any): InputPartition =
+ /**
+ * `DataSourceRDDPartition` has held its input partition(s) in three shapes across the supported
+ * Spark versions: a bare `InputPartition` (3.2), a `Seq[InputPartition]` from 3.3 to 4.1, where
+ * the RDD itself grouped storage-partitioned-join partitions, and an `Option[InputPartition]`
+ * since 4.2 (SPARK-55535), which moved that grouping out into `GroupPartitionsExec`. A `None`
+ * there is a padded empty partition and contributes no input, the same as the empty `Seq` the
+ * middle shape used for it.
+ */
+ private def normalizeInputPartitions(value: Any): Seq[InputPartition] =
value match {
- case input: InputPartition => input
- case null => throw new IllegalArgumentException("Input partition must not be null.")
+ case null => throw new IllegalArgumentException("Input partitions must not be null.")
+ case input: InputPartition => Seq(input)
+ case values: scala.collection.Seq[_] =>
+ values.iterator.flatMap(normalizeInputPartitions).toVector
+ case option: Option[_] =>
+ option.iterator.flatMap(normalizeInputPartitions).toVector
case other =>
throw new IllegalArgumentException(
s"Unexpected input partition type ${other.getClass.getName}.")
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/SparkSource.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/SparkSource.scala
index 0e7c1b8b6059..ee4df8882e4f 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/SparkSource.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/SparkSource.scala
@@ -33,6 +33,7 @@ import org.apache.spark.sql.{DataFrame, PaimonSparkSession, SaveMode => SparkSav
import org.apache.spark.sql.connector.catalog.{Identifier => SparkIdentifier, SessionConfigSupport, Table, TableCatalog}
import org.apache.spark.sql.connector.expressions.Transform
import org.apache.spark.sql.execution.streaming.Sink
+import org.apache.spark.sql.paimon.shims.SparkVersionCompat
import org.apache.spark.sql.sources.{BaseRelation, CreatableRelationProvider, DataSourceRegister, StreamSinkProvider}
import org.apache.spark.sql.streaming.OutputMode
import org.apache.spark.sql.types.StructType
@@ -105,7 +106,9 @@ class SparkSource
val catalogName = options.get(CATALOG)
val dataBaseName = Option(options.get(DATABASE)).getOrElse(CatalogUtils.database(path))
val tableName = Option(options.get(TABLE)).getOrElse(CatalogUtils.table(path))
- val sparkCatalog = sessionState.catalogManager.catalog(catalogName).asInstanceOf[TableCatalog]
+ val sparkCatalog = SparkVersionCompat
+ .catalog(sessionState.catalogManager, catalogName)
+ .asInstanceOf[TableCatalog]
sparkCatalog
.loadTable(SparkIdentifier.of(Array(dataBaseName), tableName))
.asInstanceOf[SparkTable]
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/SparkTypeUtils.java b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/SparkTypeUtils.java
index 7ced9f29ffee..1a44b0a8bdca 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/SparkTypeUtils.java
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/SparkTypeUtils.java
@@ -167,7 +167,9 @@ private static class PaimonToSparkTypeVisitor extends DataTypeDefaultVisitor
u.funcIdent.catalog match {
case Some(catalog) =>
- catalogManager.catalog(catalog) match {
+ SparkVersionCompat.catalog(catalogManager, catalog) match {
case v1FunctionCatalog: SupportV1Function =>
v1FunctionCatalog.registerAndResolveV1Function(u)
case _ =>
@@ -129,8 +130,8 @@ case class PaimonFunctionResolver(spark: SparkSession) extends Rule[LogicalPlan]
private def functionCatalog(nameParts: Seq[String]): CatalogPlugin = {
nameParts.length match {
- case 2 => catalogManager.currentCatalog
- case 3 => catalogManager.catalog(nameParts.head)
+ case 2 => SparkVersionCompat.currentCatalog(catalogManager)
+ case 3 => SparkVersionCompat.catalog(catalogManager, nameParts.head)
case _ =>
throw new UnsupportedOperationException(
s"Invalid function identifier: ${nameParts.mkString(".")}")
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/catalyst/analysis/PaimonViewResolver.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/catalyst/analysis/PaimonViewResolver.scala
index d60e9ab8f3a4..ff44393618ff 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/catalyst/analysis/PaimonViewResolver.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/catalyst/analysis/PaimonViewResolver.scala
@@ -50,13 +50,18 @@ case class PaimonViewResolver(spark: SparkSession)
u
}
- case u @ UnresolvedTableOrView(CatalogAndIdentifier(catalog: SupportView, ident), _, _) =>
- try {
- catalog.loadView(ident)
- ResolvedPaimonView(catalog, ident)
- } catch {
- case _: ViewNotExistException =>
- u
+ // Match by type and use the named accessor instead of a positional pattern, because the
+ // number of `UnresolvedTableOrView` parameters differs across supported Spark versions.
+ case u: UnresolvedTableOrView =>
+ u.multipartIdentifier match {
+ case CatalogAndIdentifier(catalog: SupportView, ident) =>
+ try {
+ catalog.loadView(ident)
+ ResolvedPaimonView(catalog, ident)
+ } catch {
+ case _: ViewNotExistException => u
+ }
+ case _ => u
}
}
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/catalyst/analysis/ReplacePaimonFunctions.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/catalyst/analysis/ReplacePaimonFunctions.scala
index d81560ffe631..b016bb910edf 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/catalyst/analysis/ReplacePaimonFunctions.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/catalyst/analysis/ReplacePaimonFunctions.scala
@@ -37,6 +37,7 @@ import org.apache.spark.sql.catalyst.plans.logical.{AnalysisHelper, LogicalPlan}
import org.apache.spark.sql.catalyst.rules.Rule
import org.apache.spark.sql.connector.catalog.{CatalogPlugin, Identifier}
import org.apache.spark.sql.connector.catalog.PaimonCatalogImplicits._
+import org.apache.spark.sql.paimon.shims.SparkVersionCompat
import org.apache.spark.sql.types.{BinaryType, DataType, DayTimeIntervalType, NullType, StringType}
import org.apache.spark.unsafe.types.UTF8String
@@ -96,7 +97,10 @@ object ReplacePaimonFunctions {
Literal(null, BinaryType)
} else {
val catalogAndIdentifier = SparkUtils
- .catalogAndIdentifier(spark, tableName, spark.sessionState.catalogManager.currentCatalog)
+ .catalogAndIdentifier(
+ spark,
+ tableName,
+ SparkVersionCompat.currentCatalog(spark.sessionState.catalogManager))
if (!catalogAndIdentifier.catalog().isInstanceOf[SparkBaseCatalog]) {
throw new UnsupportedOperationException(
s"${catalogAndIdentifier.catalog()} is not a Paimon catalog")
@@ -146,7 +150,7 @@ case class ReplacePaimonFunctions(spark: SparkSession) extends Rule[LogicalPlan]
.catalogAndIdentifier(
spark,
tableName.toString,
- spark.sessionState.catalogManager.currentCatalog)
+ SparkVersionCompat.currentCatalog(spark.sessionState.catalogManager))
if (!catalogAndIdentifier.catalog().isInstanceOf[SparkBaseCatalog]) {
throw new UnsupportedOperationException(
s"${catalogAndIdentifier.catalog()} is not a Paimon catalog")
@@ -248,8 +252,8 @@ case class ReplacePaimonFunctions(spark: SparkSession) extends Rule[LogicalPlan]
s"validity must be INTERVAL DAY TO SECOND type, but found ${other.simpleString}")
}
val functionCatalog = Option(function.catalogName())
- .map(catalogManager.catalog)
- .getOrElse(catalogManager.currentCatalog)
+ .map(SparkVersionCompat.catalog(catalogManager, _))
+ .getOrElse(SparkVersionCompat.currentCatalog(catalogManager))
ReplacePaimonFunctions.resolveDescriptorToPresignedUrl(
spark,
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/catalyst/plans/logical/PaimonTableValuedFunctions.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/catalyst/plans/logical/PaimonTableValuedFunctions.scala
index 4c0123824f95..15eccd817ac1 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/catalyst/plans/logical/PaimonTableValuedFunctions.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/catalyst/plans/logical/PaimonTableValuedFunctions.scala
@@ -38,6 +38,7 @@ import org.apache.spark.sql.catalyst.plans.logical.{Filter, LeafNode, LogicalPla
import org.apache.spark.sql.catalyst.util.MapData
import org.apache.spark.sql.connector.catalog.{Identifier, Table, TableCatalog}
import org.apache.spark.sql.execution.datasources.v2.DataSourceV2Relation
+import org.apache.spark.sql.paimon.shims.SparkVersionCompat
import org.apache.spark.sql.util.CaseInsensitiveStringMap
import org.apache.spark.unsafe.types.UTF8String
@@ -121,14 +122,19 @@ object PaimonTableValuedFunctions {
val (catalogName, dbName, tableName) = {
sessionState.sqlParser.parseMultipartIdentifier(identifier) match {
case Seq(table) =>
- (catalogManager.currentCatalog.name(), catalogManager.currentNamespace.head, table)
- case Seq(db, table) => (catalogManager.currentCatalog.name(), db, table)
+ (
+ SparkVersionCompat.currentCatalog(catalogManager).name(),
+ SparkVersionCompat.currentNamespace(catalogManager).head,
+ table)
+ case Seq(db, table) =>
+ (SparkVersionCompat.currentCatalog(catalogManager).name(), db, table)
case Seq(catalog, db, table) => (catalog, db, table)
case _ => throw new RuntimeException(s"Invalid table identifier: $identifier")
}
}
- val sparkCatalog = catalogManager.catalog(catalogName).asInstanceOf[TableCatalog]
+ val sparkCatalog =
+ SparkVersionCompat.catalog(catalogManager, catalogName).asInstanceOf[TableCatalog]
val ident: Identifier = Identifier.of(Array(dbName), tableName)
val sparkTable = sparkCatalog.loadTable(ident)
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/commands/PaimonDynamicPartitionOverwriteCommand.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/commands/PaimonDynamicPartitionOverwriteCommand.scala
index 0f722d4c8066..65d2cf00950c 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/commands/PaimonDynamicPartitionOverwriteCommand.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/commands/PaimonDynamicPartitionOverwriteCommand.scala
@@ -26,6 +26,7 @@ import org.apache.spark.sql.{Row, SparkSession}
import org.apache.spark.sql.PaimonUtils.{createDataset, createNewDataFrame}
import org.apache.spark.sql.catalyst.analysis.NamedRelation
import org.apache.spark.sql.catalyst.plans.logical.{Command, LogicalPlan, V2WriteCommand}
+import org.apache.spark.sql.connector.catalog.TableWritePrivilege
import org.apache.spark.sql.execution.command.RunnableCommand
import scala.collection.convert.ImplicitConversions._
@@ -60,6 +61,17 @@ case class PaimonDynamicPartitionOverwriteCommand(
override protected def withNewChildInternal(
newChild: LogicalPlan): PaimonDynamicPartitionOverwriteCommand = copy(query = newChild)
+ // Spark 4.2 mixes `WriteWithSchemaEvolution` into `V2WriteCommand`, which declares these two
+ // members. Declared without `override` so the same source compiles on Spark 3.5/4.0/4.1,
+ // where the trait — and therefore the members being overridden — does not exist.
+ // Dynamic partition overwrite never evolves the target schema.
+ def withSchemaEvolution: Boolean = false
+
+ // Dynamic partition overwrite replaces whole partitions, so it deletes as well as inserts —
+ // the same privilege set Spark's own `OverwritePartitionsDynamic` requests.
+ def writePrivileges: Set[TableWritePrivilege] =
+ Set(TableWritePrivilege.INSERT, TableWritePrivilege.DELETE)
+
override def run(sparkSession: SparkSession): Seq[Row] = {
WriteIntoPaimonTable(
fileStoreTable,
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/execution/PaimonStrategy.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/execution/PaimonStrategy.scala
index 770b9eec66cd..3c694694d515 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/execution/PaimonStrategy.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/execution/PaimonStrategy.scala
@@ -56,7 +56,7 @@ import org.apache.spark.sql.execution.datasources.v2.{DataSourceV2Implicits, Dat
import org.apache.spark.sql.execution.exchange.ShuffleExchangeLike
import org.apache.spark.sql.execution.joins.{BroadcastHashJoinExec, BroadcastNestedLoopJoinExec}
import org.apache.spark.sql.execution.shim.{PaimonCreateTableAsSelectStrategy, PaimonReplaceTableAsSelectStrategy, PaimonReplaceTableStrategy}
-import org.apache.spark.sql.paimon.shims.SparkShimLoader
+import org.apache.spark.sql.paimon.shims.{SparkShimLoader, SparkVersionCompat}
import scala.collection.JavaConverters._
import scala.collection.mutable.ArrayBuffer
@@ -197,19 +197,51 @@ case class PaimonStrategy(spark: SparkSession)
case ShowCreateTable(ResolvedPaimonView(viewCatalog, ident), _, output) =>
ShowCreatePaimonViewExec(output, viewCatalog, ident) :: Nil
- case DescribeRelation(ResolvedPaimonView(viewCatalog, ident), _, isExtended, output) =>
- DescribePaimonViewExec(output, viewCatalog, ident, isExtended) :: Nil
-
- case DescribeRelation(r: ResolvedTable, partitionSpec, isExtended, output) =>
- (r.table, r.catalog) match {
- case (sparkTable: SparkTable, sparkCatalog: SparkBaseCatalog) =>
- PaimonDescribeTableExec(
- output,
- sparkCatalog,
- r.identifier,
- sparkTable,
- partitionSpec,
- isExtended) :: Nil
+ // Spark 4.2 (SPARK-39660) routes `DESCRIBE ... PARTITION` through its own
+ // `DescribeTablePartition` plan. Intercept it so Paimon tables keep emitting the same rows they
+ // do on 3.x/4.0/4.1; the upstream exec has a different row shape. The shim returns `None` on
+ // every version that lacks the node, where the spec arrives inside `DescribeRelation` below.
+ case DescribeTablePartitionPlan(relation, partitionSpec, isExtended, output) =>
+ relation match {
+ case r: ResolvedTable =>
+ (r.table, r.catalog) match {
+ case (sparkTable: SparkTable, sparkCatalog: SparkBaseCatalog) =>
+ PaimonDescribeTableExec(
+ output,
+ sparkCatalog,
+ r.identifier,
+ sparkTable,
+ partitionSpec,
+ isExtended) :: Nil
+ case _ => Nil
+ }
+ case _ => Nil
+ }
+
+ // `DescribeRelation` has 3 leading fields on Spark <= 4.1 but only 2 on 4.2 (SPARK-39660 moved
+ // `partitionSpec` out into a separate `DescribeTablePartition` plan), so match by type and read
+ // the members through named accessors instead of a positional pattern.
+ case d: DescribeRelation if d.relation.isInstanceOf[ResolvedPaimonView] =>
+ val v = d.relation.asInstanceOf[ResolvedPaimonView]
+ DescribePaimonViewExec(d.output, v.catalog, v.identifier, d.isExtended) :: Nil
+
+ case d: DescribeRelation =>
+ d.relation match {
+ case r: ResolvedTable =>
+ (r.table, r.catalog) match {
+ case (sparkTable: SparkTable, sparkCatalog: SparkBaseCatalog) =>
+ PaimonDescribeTableExec(
+ d.output,
+ sparkCatalog,
+ r.identifier,
+ sparkTable,
+ // Spark 4.2 moved DESCRIBE ... PARTITION out of DescribeRelation; the shim
+ // returns an empty spec there.
+ SparkShimLoader.shim.describeRelationPartitionSpec(d),
+ d.isExtended
+ ) :: Nil
+ case _ => Nil
+ }
case _ => Nil
}
@@ -321,10 +353,24 @@ case class PaimonStrategy(spark: SparkSession)
new GenericInternalRow(values)
}
+ /**
+ * Matches Spark 4.2's `DescribeTablePartition` through the shim, which returns `None` on every
+ * version that lacks the node. An extractor rather than a `case p if shim...isDefined` guard so a
+ * matching node runs the shim once instead of twice; non-matching nodes cost the same as before.
+ */
+ private object DescribeTablePartitionPlan {
+ def unapply(
+ plan: LogicalPlan): Option[(LogicalPlan, Map[String, String], Boolean, Seq[Attribute])] =
+ SparkShimLoader.shim.describeTablePartition(plan)
+ }
+
private object PaimonCatalogAndIdentifier {
def unapply(identifier: Seq[String]): Option[(TableCatalog, Identifier)] = {
val catalogAndIdentifier =
- SparkUtils.catalogAndIdentifier(spark, identifier.asJava, catalogManager.currentCatalog)
+ SparkUtils.catalogAndIdentifier(
+ spark,
+ identifier.asJava,
+ SparkVersionCompat.currentCatalog(catalogManager))
catalogAndIdentifier.catalog match {
case paimonCatalog: SparkCatalog =>
Some((paimonCatalog, catalogAndIdentifier.identifier()))
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/execution/adaptive/DisableUnnecessaryPaimonBucketedScan.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/execution/adaptive/DisableUnnecessaryPaimonBucketedScan.scala
index b0101ded21bc..6bada4fb9542 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/execution/adaptive/DisableUnnecessaryPaimonBucketedScan.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/paimon/spark/execution/adaptive/DisableUnnecessaryPaimonBucketedScan.scala
@@ -96,6 +96,10 @@ object DisableUnnecessaryPaimonBucketedScan extends Rule[SparkPlan] {
hashInterestingPartitionOrOrder: Boolean,
hasExchange: Boolean): SparkPlan = {
plan match {
+ case p if consumesBucketPartitioning(p) =>
+ // Stop here: this operator was added because its child reports the bucket partitioning, so
+ // every bucketed scan below it is in use by definition.
+ p
case p if hasInterestingPartitionOrOrder(p) =>
// Operator with interesting partition, propagates `hashInterestingPartitionOrOrder` as true
// to its children, and resets `hasExchange`.
@@ -123,6 +127,37 @@ object DisableUnnecessaryPaimonBucketedScan extends Rule[SparkPlan] {
}
}
+ /**
+ * Whether `plan` was inserted specifically to consume its child's bucket (key-grouped)
+ * partitioning, so that disabling the bucketed scan underneath it would leave the plan
+ * inconsistent.
+ *
+ * Spark 4.2 (SPARK-55535) reworked storage-partitioned joins: instead of letting the join read
+ * the scan's `KeyGroupedPartitioning` directly, `EnsureRequirements` now wraps the child in a
+ * `GroupPartitionsExec`, which coalesces the scan's input partitions by partition key. That node
+ * casts `child.outputPartitioning` to `Partitioning with Expression` unconditionally, because it
+ * is only ever added when the child reports `KeyedPartitioning`. If we then disable the bucketed
+ * scan below it, `PaimonScan.outputPartitioning` degrades to `UnknownPartitioning`, which is not
+ * an `Expression`, and the cast fails with a `ClassCastException` at execution time.
+ *
+ * The whole subtree is left alone, not just a scan directly beneath the node:
+ * `EnsureRequirements` wraps whatever satisfied the distribution, which can be a join or an
+ * aggregate with several scans below it, and disabling only some of them would instead trip
+ * `PartitioningCollection`'s equal-`numPartitions` requirement. Bucketing therefore stays on for
+ * scans in that subtree that do not feed the keyed partitioning; that costs some parallelism and
+ * changes no result.
+ *
+ * Matched by class name because the class does not exist before 4.2, while this file is compiled
+ * once (against the newest supported Spark) and has to load on every supported version — an
+ * `isInstanceOf` would resolve the class and fail with `NoClassDefFoundError` on the older ones.
+ * A rename in a later Spark would silently stop the match, which is why the crash itself is
+ * pinned by a query rather than by this string: see `BucketedTableQueryTest`'s "join - negative
+ * case", whose `t1 JOIN t5` (equal bucket counts, different input partition counts) is the shape
+ * that reaches `GroupPartitionsExec`.
+ */
+ private def consumesBucketPartitioning(plan: SparkPlan): Boolean =
+ plan.getClass.getName == "org.apache.spark.sql.execution.datasources.v2.GroupPartitionsExec"
+
private def hasInterestingPartitionOrOrder(plan: SparkPlan): Boolean = {
val hashPartition = plan.requiredChildDistribution.exists {
case _: ClusteredDistribution | AllTuples => true
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/catalog/PaimonV1FunctionRegistry.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/catalog/PaimonV1FunctionRegistry.scala
index ed69abd8bb5d..6d9c51e2116d 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/catalog/PaimonV1FunctionRegistry.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/catalog/PaimonV1FunctionRegistry.scala
@@ -159,12 +159,33 @@ case class PaimonV1FunctionRegistry(session: SparkSession) extends SQLConfHelper
throw new IllegalArgumentException(s"Cannot load class: $className")
}
val clazz = PaimonUtils.classForName(className)
- val name = func.identifier.unquotedString
+ // Drop the catalog for the *name* handed to the expression builder, even though the registry
+ // key keeps it (see `qualifyIdentifier`). This string becomes the built expression's
+ // user-facing function name in every case — the Hive wrappers set `prettyName = name`,
+ // `ScalaUDAF` takes it as `udafName` and exposes it as `name`/`nodeName` — and for a scalar
+ // or aggregate result an unaliased projection derives its column name from that name, via
+ // `Alias(e, toPrettySQL(e))`. So keeping the catalog here would silently rename the default
+ // column name of `SELECT udf(...)` from `db.udf(...)` to `catalog.db.udf(...)`.
+ val name = func.identifier.copy(catalog = None).unquotedString
(input) => functionExpressionBuilder.makeExpression(name, clazz, input)
}
private def qualifyIdentifier(ident: FunctionIdentifier): FunctionIdentifier = {
- FunctionIdentifier(funcName = format(ident.funcName), database = ident.database)
+ // Carry the catalog through. Spark 4.2's `SimpleFunctionRegistryBase.normalizeFuncName` asserts
+ // the identifier is fully qualified (3-part), so dropping it fails with
+ // "Function identifier must be fully qualified". Callers already supply all three parts (see
+ // `PaimonFunctionLookup.CatalogAndFunctionIdentifier`), and keeping the catalog is also more
+ // correct on older versions: two identically named functions in different catalogs would
+ // otherwise collide in the registry. Upstream's own `SessionCatalog.qualifyIdentifier` does
+ // the same.
+ //
+ // Note this identifier is also stamped into the `CatalogFunction` that builds the expression,
+ // so `makeFunctionBuilder` strips the catalog back off for the builder name — that string is
+ // user-visible as a column name.
+ FunctionIdentifier(
+ funcName = format(ident.funcName),
+ database = ident.database,
+ catalog = ident.catalog)
}
protected def format(name: String): String = {
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/PaimonFunctionLookup.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/PaimonFunctionLookup.scala
index 4a4b179ff645..3dd8bb50e220 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/PaimonFunctionLookup.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/PaimonFunctionLookup.scala
@@ -29,6 +29,7 @@ import org.apache.spark.sql.catalyst.FunctionIdentifier
import org.apache.spark.sql.catalyst.analysis.{UnresolvedFunctionName, UnresolvedIdentifier}
import org.apache.spark.sql.catalyst.plans.logical.LogicalPlan
import org.apache.spark.sql.connector.catalog.{CatalogManager, CatalogPlugin, LookupCatalog}
+import org.apache.spark.sql.paimon.shims.SparkVersionCompat
/** Resolves Paimon function identifiers in parser-stage plans. */
case class PaimonFunctionLookup(catalogManager: CatalogManager) extends LookupCatalog {
@@ -48,7 +49,7 @@ case class PaimonFunctionLookup(catalogManager: CatalogManager) extends LookupCa
def unapply(nameParts: Seq[String]): Option[(CatalogPlugin, FunctionIdentifier, Boolean)] = {
nameParts match {
// Spark's built-in or tmp functions is without database name or catalog name.
- case Seq(funName) if isSparkBuiltInFunction(FunctionIdentifier(funName)) =>
+ case Seq(funName) if isSparkBuiltInFunction(funName) =>
None
case Seq(funName) if isSparkTmpFunc(FunctionIdentifier(funName)) =>
Some(null, FunctionIdentifier(funName), true)
@@ -76,12 +77,16 @@ case class PaimonFunctionLookup(catalogManager: CatalogManager) extends LookupCa
}
}
- def isSparkBuiltInFunction(funcIdent: FunctionIdentifier): Boolean = {
- catalogManager.v1SessionCatalog.isBuiltinFunction(funcIdent)
+ def isSparkBuiltInFunction(funcName: String): Boolean = {
+ // Takes a bare name, not a FunctionIdentifier: see SparkVersionCompat.isBuiltinFunction —
+ // the qualifier is not part of the question on any version.
+ SparkVersionCompat.isBuiltinFunction(
+ SparkVersionCompat.v1SessionCatalog(catalogManager),
+ funcName)
}
def isSparkTmpFunc(funcIdent: FunctionIdentifier): Boolean = {
- catalogManager.v1SessionCatalog.isTemporaryFunction(funcIdent)
+ SparkVersionCompat.v1SessionCatalog(catalogManager).isTemporaryFunction(funcIdent)
}
}
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/RewriteCreateTableLikeCommand.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/RewriteCreateTableLikeCommand.scala
index 018644e9f389..025cf00bb242 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/RewriteCreateTableLikeCommand.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/RewriteCreateTableLikeCommand.scala
@@ -28,6 +28,7 @@ import org.apache.spark.sql.catalyst.plans.logical.LogicalPlan
import org.apache.spark.sql.catalyst.rules.Rule
import org.apache.spark.sql.connector.catalog.{CatalogManager, Identifier, LookupCatalog, TableCatalog}
import org.apache.spark.sql.execution.command.{CreateTableLikeCommand => SparkCreateTableLikeCommand}
+import org.apache.spark.sql.paimon.shims.SparkShimLoader
case class RewriteCreateTableLikeCommand(spark: SparkSession)
extends Rule[LogicalPlan]
@@ -42,6 +43,45 @@ case class RewriteCreateTableLikeCommand(spark: SparkSession)
}
plan.resolveOperatorsUp {
+ // Spark 4.2 parses `CREATE TABLE LIKE` for v2 catalogs into its own logical plan, so the V1
+ // command below never appears. The shim hands back unresolved name parts, which get resolved
+ // here exactly as the V1 path resolves its `TableIdentifier`s.
+ case p if SparkShimLoader.shim.createTableLikeParts(p).isDefined =>
+ val (
+ targetParts,
+ sourceParts,
+ provider,
+ location,
+ properties,
+ ifNotExists,
+ hasHiveStorageSyntax) = SparkShimLoader.shim.createTableLikeParts(p).get
+ (targetParts, sourceParts) match {
+ // `SparkCatalog` always creates Paimon tables, so it always takes over. A
+ // `SparkGenericCatalog` target only does when the statement asks for Paimon: otherwise
+ // this is e.g. `CREATE TABLE csv_like LIKE csv_source`, which must keep Spark's own
+ // semantics (Spark does not copy the source comment, Paimon's command does). Same
+ // condition the V1 branch below applies.
+ case (
+ CatalogAndIdentifier(targetCatalog: SparkBaseCatalog, targetIdent),
+ CatalogAndIdentifier(sourceCatalog: TableCatalog, sourceIdent))
+ if targetCatalog.isInstanceOf[SparkCatalog] ||
+ provider.exists(SparkBaseCatalog.usePaimon) =>
+ if (hasHiveStorageSyntax) {
+ throw new UnsupportedOperationException(
+ "CREATE TABLE LIKE ... STORED AS is not supported for SparkCatalog.")
+ }
+ PaimonCreateTableLikeCommand(
+ targetCatalog,
+ targetIdent,
+ sourceCatalog,
+ sourceIdent,
+ provider,
+ location,
+ properties,
+ ifNotExists)
+ case _ => p
+ }
+
case c: SparkCreateTableLikeCommand =>
val targetParts = toMultipartIdentifier(c.targetTable)
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/RewritePaimonFunctionCommands.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/RewritePaimonFunctionCommands.scala
index 6be680cd60ff..7e5502698c42 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/RewritePaimonFunctionCommands.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/RewritePaimonFunctionCommands.scala
@@ -33,7 +33,7 @@ import org.apache.spark.sql.catalyst.plans.logical.{CreateFunction, DescribeFunc
import org.apache.spark.sql.catalyst.rules.Rule
import org.apache.spark.sql.catalyst.trees.TreePattern.{TreePattern, UNRESOLVED_FUNCTION}
import org.apache.spark.sql.connector.catalog.CatalogManager
-import org.apache.spark.sql.paimon.shims.SparkShimLoader
+import org.apache.spark.sql.paimon.shims.{SparkShimLoader, SparkVersionCompat}
import org.apache.spark.sql.types.DataType
case class RewritePaimonFunctionCommands(spark: SparkSession) extends Rule[LogicalPlan] {
@@ -191,6 +191,13 @@ object UnResolvedPaimonV1Function {
funcIdent: FunctionIdentifier,
u: UnresolvedFunction,
fun: Option[PaimonFunction]): UnResolvedPaimonV1Function = {
- UnResolvedPaimonV1Function(funcIdent, u.arguments, u.isDistinct, u.filter, u.ignoreNulls, fun)
+ // Spark 4.2 widened `ignoreNulls` from Boolean to Option[Boolean].
+ UnResolvedPaimonV1Function(
+ funcIdent = funcIdent,
+ arguments = u.arguments,
+ isDistinct = u.isDistinct,
+ filter = u.filter,
+ ignoreNulls = SparkVersionCompat.ignoreNulls(u),
+ func = fun)
}
}
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/RewritePaimonViewCommands.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/RewritePaimonViewCommands.scala
index 4b4187b2d8c4..5e9c3f92bf89 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/RewritePaimonViewCommands.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/RewritePaimonViewCommands.scala
@@ -26,6 +26,7 @@ import org.apache.spark.sql.catalyst.analysis.{CTESubstitution, ResolveCatalogs,
import org.apache.spark.sql.catalyst.plans.logical._
import org.apache.spark.sql.catalyst.rules.Rule
import org.apache.spark.sql.connector.catalog.{CatalogManager, LookupCatalog}
+import org.apache.spark.sql.paimon.shims.SparkVersionCompat
case class RewritePaimonViewCommands(spark: SparkSession)
extends Rule[LogicalPlan]
@@ -60,16 +61,16 @@ case class RewritePaimonViewCommands(spark: SparkSession)
DropPaimonView(resolved, ifExists)
case ShowViews(namespace, pattern, output)
- if catalogManager.currentCatalog.isInstanceOf[SupportView] =>
+ if SparkVersionCompat.currentCatalog(catalogManager).isInstanceOf[SupportView] =>
val resolvedNamespace = new ResolveCatalogs(catalogManager)(namespace).transform {
case r: ResolvedNamespace if r.namespace.isEmpty =>
- r.copy(namespace = catalogManager.currentNamespace)
+ r.copy(namespace = SparkVersionCompat.currentNamespace(catalogManager))
}
ShowPaimonViews(resolvedNamespace, pattern, output)
}
private def isTempView(nameParts: Seq[String]): Boolean = {
- catalogManager.v1SessionCatalog.isTempView(nameParts)
+ SparkVersionCompat.v1SessionCatalog(catalogManager).isTempView(nameParts)
}
private object ResolvedIdent {
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/connector/catalog/SparkV1PartitionManagement.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/connector/catalog/SparkV1PartitionManagement.scala
index d0a1ed890e98..785269cae00f 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/connector/catalog/SparkV1PartitionManagement.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/connector/catalog/SparkV1PartitionManagement.scala
@@ -25,6 +25,7 @@ import org.apache.spark.sql.catalyst.catalog.{CatalogTable, CatalogTablePartitio
import org.apache.spark.sql.catalyst.util.CharVarcharUtils
import org.apache.spark.sql.execution.datasources.v2.V2SessionCatalog
import org.apache.spark.sql.internal.SQLConf
+import org.apache.spark.sql.paimon.shims.SparkShimLoader
import org.apache.spark.sql.types.StructType
import java.util.{Map => JMap}
@@ -78,8 +79,10 @@ class SparkV1PartitionManagement(catalogTable: CatalogTable, catalog: SessionCat
val location = scalaProperties.get("location")
CatalogTablePartition(
toPartitionSpec(ident),
- catalogTable.storage.copy(locationUri = location.map(CatalogUtils.stringToURI)),
- parameters = scalaProperties - "location")
+ SparkShimLoader.shim
+ .withStorageLocation(catalogTable.storage, location.map(CatalogUtils.stringToURI)),
+ parameters = scalaProperties - "location"
+ )
}
catalog.createPartitions(catalogTable.identifier, partitions, ignoreIfExists = false)
}
@@ -119,7 +122,9 @@ class SparkV1PartitionManagement(catalogTable: CatalogTable, catalog: SessionCat
val scalaProperties = properties.asScala.toMap
val storage = scalaProperties.get("location") match {
case Some(location) =>
- partition.storage.copy(locationUri = Some(CatalogUtils.stringToURI(location)))
+ SparkShimLoader.shim.withStorageLocation(
+ partition.storage,
+ Some(CatalogUtils.stringToURI(location)))
case None => partition.storage
}
catalog.alterPartitions(
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/execution/PaimonDescribeTableExec.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/execution/PaimonDescribeTableExec.scala
index 52f96d58f5b5..67e6be5c081a 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/execution/PaimonDescribeTableExec.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/execution/PaimonDescribeTableExec.scala
@@ -30,7 +30,7 @@ import org.apache.spark.sql.catalyst.catalog.{CatalogStatistics, CatalogStorageF
import org.apache.spark.sql.catalyst.catalog.CatalogTypes.TablePartitionSpec
import org.apache.spark.sql.catalyst.expressions.Attribute
import org.apache.spark.sql.connector.catalog.Identifier
-import org.apache.spark.sql.execution.datasources.v2.DescribeTableExec
+import org.apache.spark.sql.paimon.shims.SparkShimLoader
import org.apache.spark.sql.types.StructType
import scala.collection.JavaConverters._
@@ -47,7 +47,9 @@ case class PaimonDescribeTableExec(
override protected def run(): Seq[InternalRow] = {
val rows =
- ArrayBuffer.empty ++= DescribeTableExec(output, table, isExtended).executeCollect()
+ ArrayBuffer.empty ++= SparkShimLoader.shim
+ .createDescribeTableExec(output, catalog.name(), identifier, table, isExtended)
+ .executeCollect()
if (partitionSpec.nonEmpty) {
describeDetailedPartitionInfo(rows)
@@ -90,8 +92,11 @@ case class PaimonDescribeTableExec(
s"Found ${partition.size} matching partitions. " +
s"Expected exactly one partition to match the partition spec.")
}
- val dummyStorageFormat =
- CatalogStorageFormat(None, None, None, None, compressed = false, Map.empty)
+ // `CatalogStorageFormat.empty` over the case-class constructor: Spark 4.2 added a 7th
+ // `serdeName` field, so a 6-arg call compiled against 4.2 emits `apply$default$7`, which does
+ // not exist on 3.5/4.0/4.1. The factory is present on every supported version and yields the
+ // same all-empty value.
+ val dummyStorageFormat = CatalogStorageFormat.empty
val statistics = partition.head
// Include only reported values. Spark omits the "Partition Parameters" row for an empty map.
val partParameters: Map[String, String] = Seq(
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/execution/PaimonTableAsSelectHelper.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/execution/PaimonTableAsSelectHelper.scala
index 0e6411dcfab5..a823068273a0 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/execution/PaimonTableAsSelectHelper.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/execution/PaimonTableAsSelectHelper.scala
@@ -31,7 +31,7 @@ import org.apache.spark.sql.SparkSession
import org.apache.spark.sql.catalyst.analysis.NoSuchTableException
import org.apache.spark.sql.catalyst.catalog.CatalogUtils
import org.apache.spark.sql.catalyst.expressions.Literal
-import org.apache.spark.sql.catalyst.plans.logical.{LogicalPlan, OverwriteByExpression, OverwritePartitionsDynamic, TableSpec}
+import org.apache.spark.sql.catalyst.plans.logical.{LogicalPlan, TableSpec}
import org.apache.spark.sql.connector.catalog.{CatalogPlugin, Identifier, TableCatalog}
import org.apache.spark.sql.connector.expressions.Transform
import org.apache.spark.sql.execution.datasources.v2.DataSourceV2Relation
@@ -154,10 +154,11 @@ object PaimonTableAsSelectHelper {
DataSourceV2Relation.create(existing, Some(catalog), Some(ident))
val dynamicOverwrite = existing.partitioning().nonEmpty &&
spark.sessionState.conf.partitionOverwriteMode == PartitionOverwriteMode.DYNAMIC
+ val shim = SparkShimLoader.shim
if (dynamicOverwrite) {
- Some(OverwritePartitionsDynamic.byName(relation, query, writeOptions))
+ Some(shim.overwritePartitionsDynamicByName(relation, query, writeOptions))
} else {
- Some(OverwriteByExpression.byName(relation, query, Literal(true), writeOptions))
+ Some(shim.overwriteByName(relation, query, Literal(true), writeOptions))
}
} catch {
case _: Exception => None
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/execution/shim/PaimonCreateTableAsSelectStrategy.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/execution/shim/PaimonCreateTableAsSelectStrategy.scala
index 1a8e3ffe4b75..3270743eb345 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/execution/shim/PaimonCreateTableAsSelectStrategy.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/execution/shim/PaimonCreateTableAsSelectStrategy.scala
@@ -28,7 +28,7 @@ import org.apache.spark.sql.catalyst.analysis.ResolvedIdentifier
import org.apache.spark.sql.catalyst.plans.logical.{CreateTableAsSelect, LogicalPlan, TableSpec}
import org.apache.spark.sql.execution.{PaimonTableAsSelectHelper, SparkPlan, SparkStrategy}
import org.apache.spark.sql.execution.PaimonTableAsSelectHelper._
-import org.apache.spark.sql.execution.datasources.v2.CreateTableAsSelectExec
+import org.apache.spark.sql.paimon.shims.SparkShimLoader
case class PaimonCreateTableAsSelectStrategy(spark: SparkSession)
extends SparkStrategy
@@ -62,7 +62,7 @@ case class PaimonCreateTableAsSelectStrategy(spark: SparkSession)
"Using CTAS with partitioned format table is not supported yet.")
}
- CreateTableAsSelectExec(
+ SparkShimLoader.shim.createCreateTableAsSelectExec(
catalog.asTableCatalog,
ident,
parts,
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/paimon/shims/SparkShim.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/paimon/shims/SparkShim.scala
index 690c20b85b74..f529075b1d27 100644
--- a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/paimon/shims/SparkShim.scala
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/paimon/shims/SparkShim.scala
@@ -29,10 +29,12 @@ import org.apache.paimon.types.{DataType, RowType}
import org.apache.spark.sql.SparkSession
import org.apache.spark.sql.catalyst.FunctionIdentifier
import org.apache.spark.sql.catalyst.InternalRow
+import org.apache.spark.sql.catalyst.analysis.NamedRelation
+import org.apache.spark.sql.catalyst.catalog.CatalogStorageFormat
import org.apache.spark.sql.catalyst.expressions.{Attribute, AttributeReference, Expression}
import org.apache.spark.sql.catalyst.expressions.aggregate.AggregateExpression
import org.apache.spark.sql.catalyst.parser.ParserInterface
-import org.apache.spark.sql.catalyst.plans.logical.{Assignment, CTERelationRef, InsertAction, LogicalPlan, MergeAction, MergeIntoTable, SubqueryAlias, TableSpec, UnresolvedWith, UpdateAction}
+import org.apache.spark.sql.catalyst.plans.logical.{Assignment, CTERelationRef, DescribeRelation, InsertAction, LogicalPlan, MergeAction, MergeIntoTable, OverwriteByExpression, OverwritePartitionsDynamic, SubqueryAlias, TableSpec, UnresolvedWith, UpdateAction}
import org.apache.spark.sql.catalyst.plans.physical.Distribution
import org.apache.spark.sql.catalyst.rules.Rule
import org.apache.spark.sql.catalyst.util.ArrayData
@@ -44,6 +46,7 @@ import org.apache.spark.sql.execution.SparkPlan
import org.apache.spark.sql.execution.datasources.v2.{DataSourceV2Relation, DataSourceV2ScanRelation}
import org.apache.spark.sql.types.StructType
+import java.net.URI
import java.util.{Map => JMap}
/**
@@ -75,6 +78,57 @@ trait SparkShim {
partitions: Array[Transform],
properties: JMap[String, String]): Table
+ /**
+ * Returns `storage` with its `locationUri` replaced.
+ *
+ * Spark 4.2 added a 7th `serdeName` field to `CatalogStorageFormat`, so a named-argument `copy`
+ * emits `copy$default$7`, which older runtimes lack. Every version can express the replacement,
+ * but only against its own field count, so the call belongs in a per-version module.
+ */
+ def withStorageLocation(
+ storage: CatalogStorageFormat,
+ locationUri: Option[URI]): CatalogStorageFormat
+
+ /**
+ * Builds `OverwriteByExpression.byName` / `OverwritePartitionsDynamic.byName`.
+ *
+ * Both gained a trailing `withSchemaEvolution: Boolean` in Spark 4.2. Omitting it makes the
+ * compiler emit `byName$default$N`, absent on 3.x/4.0/4.1; naming it does not compile there. As
+ * with `createCreateTableAsSelectExec`, the construction has to live in a per-version module.
+ *
+ * Paimon never evolves the target schema on an overwrite, so on 4.2 these pass `false`.
+ */
+ def overwriteByName(
+ table: NamedRelation,
+ query: LogicalPlan,
+ deleteExpr: Expression,
+ writeOptions: Map[String, String]): OverwriteByExpression
+
+ /** Companion of [[overwriteByName]]; same reason for being a shim method. */
+ def overwritePartitionsDynamicByName(
+ table: NamedRelation,
+ query: LogicalPlan,
+ writeOptions: Map[String, String]): OverwritePartitionsDynamic
+
+ /**
+ * Builds a Spark `CreateTableAsSelectExec`.
+ *
+ * Goes through the shim because Spark 4.2 added an 8th `transaction` parameter. Its type,
+ * `Option[connector.catalog.transactions.Transaction]`, only exists on 4.2, so
+ * `paimon-spark-common` cannot name it; omitting it instead makes the compiler emit
+ * `CreateTableAsSelectExec$.apply$default$8`, which older runtimes do not have. Neither shape
+ * links everywhere, so the construction has to happen in a per-version module — the same reason
+ * `createReplaceTableAsSelectExec` below is a shim method.
+ */
+ def createCreateTableAsSelectExec(
+ catalog: TableCatalog,
+ ident: Identifier,
+ partitioning: Seq[Transform],
+ query: LogicalPlan,
+ tableSpec: TableSpec,
+ writeOptions: Map[String, String],
+ ifNotExists: Boolean): SparkPlan
+
def createReplaceTableAsSelectExec(
catalog: TableCatalog,
ident: Identifier,
@@ -315,4 +369,81 @@ trait SparkShim {
function: PaimonFunction,
arguments: Seq[Expression],
parser: ParserInterface): Expression
+
+ /**
+ * Destructures Spark 4.2's `CreateTableLike` logical plan, or returns `None` on 3.x/4.0/4.1 where
+ * the node does not exist.
+ *
+ * Spark 4.2 (SPARK-51350) taught the parser `CREATE TABLE LIKE` for v2 catalogs and added
+ * `TableCatalog.createTableLike`, whose default implementation throws. Paimon used to intercept
+ * the syntax by catching the `ParseException` Spark raised for a catalog-qualified target; now
+ * that Spark parses it, the fallback never fires and the plan reaches the upstream exec. The node
+ * has to be matched instead, and it cannot be named here because `paimon-spark-common` also
+ * compiles against Spark 3.5.
+ *
+ * Returns the target and source name parts *unresolved*: this runs as a parser rule, before
+ * analysis, so the children are still `UnresolvedIdentifier` / `UnresolvedRelation`. The caller
+ * resolves them the same way it does for the V1 command.
+ *
+ * Returns (targetNameParts, sourceNameParts, provider, location, properties, ifNotExists,
+ * hasHiveStorageSyntax).
+ */
+ def createTableLikeParts(plan: LogicalPlan)
+ : Option[(Seq[String], Seq[String], Option[String], Option[String], Map[String, String], Boolean, Boolean)]
+
+ /**
+ * If `plan` is Spark 4.2's `DescribeTablePartition`, returns its relation, resolved partition
+ * spec, `isExtended` flag and output; `None` on 3.x/4.0/4.1 where the node does not exist.
+ *
+ * Spark 4.2 (SPARK-39660) split `DESCRIBE ... PARTITION` out of `DescribeRelation` into this
+ * separate plan. Without intercepting it Paimon tables fall through to the upstream
+ * `DescribeTablePartitionExec`, whose row shape differs from what Paimon emits (one row per
+ * metadata key, versus Paimon's single `Partition Parameters` row plus its `Database` / `Table`
+ * rows and `# Column Not Null` section). The node cannot be named here because
+ * `paimon-spark-common` also compiles against Spark 3.5.
+ *
+ * The spec values are returned already rendered to strings, keyed by partition column name, so
+ * they can be compared against Paimon's own `Partition.spec()` map. `ResolvedPartitionSpec`
+ * stores them as an `InternalRow`, so each field is read by the type `table.partitionSchema()`
+ * declares for it (the same schema `ResolvePartitionSpec` used to build the row) and rendered
+ * with `Literal.toString`.
+ *
+ * Note that this is NOT what upstream's own `DescribeTablePartitionExec` does: it renders via
+ * `ToPrettyString(...).eval(null)` plus `escapePathName`, which differs for null (`NULL` vs
+ * `null`), binary, and decimal. Paimon needs a string it can compare against `Partition.spec()`,
+ * not a display string. This rendering and `Partition.spec()`'s own do not agree for every type
+ * either; the 4.2 implementation's comment works DATE through as the example.
+ */
+ def describeTablePartition(
+ plan: LogicalPlan): Option[(LogicalPlan, Map[String, String], Boolean, Seq[Attribute])]
+
+ /**
+ * Extracts the partition spec from a `DescribeRelation` node.
+ *
+ * Spark 4.2 (SPARK-39660) removed `partitionSpec` from `DescribeRelation` and introduced a
+ * separate `DescribeTablePartition` plan for `DESCRIBE ... PARTITION`, so on 4.2 this always
+ * returns an empty map and [[describeTablePartition]] carries the spec instead.
+ */
+ def describeRelationPartitionSpec(plan: DescribeRelation): Map[String, String]
+
+ /**
+ * Builds a Spark `DescribeTableExec`.
+ *
+ * Spark 4.2 (SPARK-56678) changed the constructor from `(output, table, isExtended)` to
+ * `(output, catalogName, identifier, table, isExtended)`.
+ */
+ def createDescribeTableExec(
+ output: Seq[Attribute],
+ catalogName: String,
+ identifier: Identifier,
+ table: Table,
+ isExtended: Boolean): SparkPlan
+
+ /**
+ * Whether the given `MergeIntoTable` requires schema evolution before rewrite.
+ *
+ * Spark 4.1 exposes this as `needSchemaEvolution`; Spark 4.2 replaced it with
+ * `pendingSchemaChanges`. Spark 3.x and 4.0 have neither.
+ */
+ def mergeNeedsSchemaEvolution(merge: MergeIntoTable): Boolean
}
diff --git a/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/paimon/shims/SparkVersionCompat.scala b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/paimon/shims/SparkVersionCompat.scala
new file mode 100644
index 000000000000..a51bc7ff4939
--- /dev/null
+++ b/paimon-spark/paimon-spark-common/src/main/scala/org/apache/spark/sql/paimon/shims/SparkVersionCompat.scala
@@ -0,0 +1,140 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.spark.sql.paimon.shims
+
+import org.apache.spark.sql.catalyst.FunctionIdentifier
+import org.apache.spark.sql.catalyst.analysis.UnresolvedFunction
+import org.apache.spark.sql.catalyst.catalog.SessionCatalog
+import org.apache.spark.sql.connector.catalog.{CatalogManager, CatalogPlugin}
+
+/**
+ * Reflective accessors for a handful of Spark internals whose *signatures* (not just arity) changed
+ * across supported versions, in ways that named accessors cannot paper over.
+ *
+ * Everything here is deliberately narrow: prefer a named accessor, or a per-version
+ * `MinorVersionShim` method, over adding to this object. Reflection is the last resort, for the
+ * cases where the same logical operation has an incompatible static type on different Spark
+ * versions. It cannot be delegated to `MinorVersionShim` the way arity changes are, because that
+ * object lives in the per-version modules (`paimon-spark3-common`, `paimon-spark-3.2`,
+ * `paimon-spark-3.3`, ...) and those depend on `paimon-spark-common` — the dependency cannot be
+ * inverted, so `paimon-spark-common` has to resolve such differences itself at runtime.
+ */
+object SparkVersionCompat {
+
+ /**
+ * Spark 4.2 turned `CatalogManager` from a class into an interface. Source-compatible, binary
+ * incompatible in *both* directions: the compiler picks `invokevirtual` or `invokeinterface` from
+ * the owner's kind, and the JVM raises `IncompatibleClassChangeError` when the two disagree.
+ * Since `paimon-spark-common` is compiled once against the newest supported Spark and shipped to
+ * every older 4.x runtime, a direct call would break all of them.
+ *
+ * Reflection is immune: only invoke opcodes carry the class/interface distinction, so
+ * `Class.getMethod` resolves the same either way. These four are every `CatalogManager` member
+ * `paimon-spark-common` reaches today; a fifth belongs here too. Nothing enforces that
+ * automatically — `tools/spark-binary-compat/check_linkage.py` finds a direct call, but it is a
+ * manual script, not wired into the build.
+ */
+ private lazy val currentCatalogMethod = catalogManagerMethod("currentCatalog")
+ private lazy val catalogByNameMethod = catalogManagerMethod("catalog", classOf[String])
+ private lazy val currentNamespaceMethod = catalogManagerMethod("currentNamespace")
+ private lazy val v1SessionCatalogMethod = catalogManagerMethod("v1SessionCatalog")
+
+ private def catalogManagerMethod(name: String, paramTypes: Class[_]*): java.lang.reflect.Method =
+ classOf[CatalogManager].getMethod(name, paramTypes: _*)
+
+ /**
+ * Invokes a method reflectively, unwrapping the reflection layer so callers see exactly what a
+ * direct call would have thrown. `CatalogManager.catalog` raises `CatalogNotFoundException` for
+ * an unknown name and callers depend on catching it, so letting an `InvocationTargetException`
+ * escape would silently change control flow. Every reflective call in this object goes through
+ * here for that reason.
+ */
+ private def invoke[T](method: java.lang.reflect.Method, receiver: AnyRef, args: Any*): T =
+ try {
+ method.invoke(receiver, args.map(_.asInstanceOf[AnyRef]): _*).asInstanceOf[T]
+ } catch {
+ case e: java.lang.reflect.InvocationTargetException => throw e.getCause
+ }
+
+ def currentCatalog(catalogManager: CatalogManager): CatalogPlugin =
+ invoke[CatalogPlugin](currentCatalogMethod, catalogManager)
+
+ def catalog(catalogManager: CatalogManager, name: String): CatalogPlugin =
+ invoke[CatalogPlugin](catalogByNameMethod, catalogManager, name)
+
+ def currentNamespace(catalogManager: CatalogManager): Array[String] =
+ invoke[Array[String]](currentNamespaceMethod, catalogManager)
+
+ def v1SessionCatalog(catalogManager: CatalogManager): SessionCatalog =
+ invoke[SessionCatalog](v1SessionCatalogMethod, catalogManager)
+
+ // Spark 4.2 narrowed `SessionCatalog.isBuiltinFunction` from `FunctionIdentifier` to `String`,
+ // dropping the database/catalog qualifier from the lookup. This accessor therefore takes a bare
+ // function name: it is the only input both overloads can answer identically. Passing a qualified
+ // identifier would yield `true` on 4.2 and `false` on <= 4.1 for e.g. `mydb.upper`.
+ private lazy val byNameMethod: Option[java.lang.reflect.Method] =
+ try {
+ Some(classOf[SessionCatalog].getMethod("isBuiltinFunction", classOf[String]))
+ } catch {
+ case _: NoSuchMethodException => None
+ }
+
+ private lazy val byIdentMethod: Option[java.lang.reflect.Method] =
+ try {
+ Some(classOf[SessionCatalog].getMethod("isBuiltinFunction", classOf[FunctionIdentifier]))
+ } catch {
+ case _: NoSuchMethodException => None
+ }
+
+ def isBuiltinFunction(catalog: SessionCatalog, name: String): Boolean = {
+ byNameMethod
+ .map(invoke[java.lang.Boolean](_, catalog, name).booleanValue())
+ .orElse(byIdentMethod.map(invoke[java.lang.Boolean](_, catalog, FunctionIdentifier(name))
+ .booleanValue()))
+ .getOrElse(throw new NoSuchMethodError(
+ "SessionCatalog.isBuiltinFunction was added in Spark 3.3; found neither the String nor " +
+ "the FunctionIdentifier overload"))
+ }
+
+ // Spark 4.2 widened `UnresolvedFunction.ignoreNulls` from `Boolean` to `Option[Boolean]`. Name
+ // and (empty) parameter list are unchanged, and `getMethod` ignores the return type, so a single
+ // lookup against the declared class covers every version.
+ private lazy val ignoreNullsMethod: java.lang.reflect.Method =
+ classOf[UnresolvedFunction].getMethod("ignoreNulls")
+
+ def ignoreNulls(u: UnresolvedFunction): Boolean =
+ toBoolean(invoke[AnyRef](ignoreNullsMethod, u))
+
+ /**
+ * Normalizes a reflectively read `ignoreNulls` value. Absent (`None`, Spark 4.2+) means "not
+ * specified", which is `false` — the same reading Spark itself applies in
+ * `FunctionResolution.resolveIgnoreNulls`.
+ *
+ * Any other shape is rejected rather than quietly treated as `false`: silently dropping an
+ * `IGNORE NULLS` clause would return wrong query results instead of failing, which is the worst
+ * way for a compat layer to break.
+ */
+ private[shims] def toBoolean(raw: Any): Boolean = raw match {
+ case b: java.lang.Boolean => b.booleanValue()
+ case None => false
+ case Some(b: java.lang.Boolean) => b.booleanValue()
+ case other =>
+ throw new IllegalStateException(s"Unexpected UnresolvedFunction.ignoreNulls value: $other")
+ }
+}
diff --git a/paimon-spark/paimon-spark-ut-4.0/pom.xml b/paimon-spark/paimon-spark-ut-4.0/pom.xml
new file mode 100644
index 000000000000..2f599f82887b
--- /dev/null
+++ b/paimon-spark/paimon-spark-ut-4.0/pom.xml
@@ -0,0 +1,231 @@
+
+
+
+ 4.0.0
+
+
+ org.apache.paimon
+ paimon-spark
+ 2.1-SNAPSHOT
+
+
+ paimon-spark-ut-4.0_${scala.binary.version}
+ Paimon : Spark : UT : 4.0 : ${scala.binary.version}
+
+
+
+ 4.0.3
+ ${project.basedir}/../paimon-spark-ut
+
+ 2.13
+
+
+
+
+
+ org.apache.paimon
+ paimon-format
+
+
+
+ org.apache.paimon
+ ${paimon-sparkx-common}
+ ${project.version}
+
+
+
+
+ org.apache.spark
+ spark-sql-api_${scala.binary.version}
+ ${spark.version}
+
+
+ org.apache.spark
+ spark-connect-shims_${scala.binary.version}
+
+
+
+
+
+ org.apache.spark
+ spark-sql_${scala.binary.version}
+ ${spark.version}
+
+
+ org.apache.spark
+ spark-connect-shims_${scala.binary.version}
+
+
+
+
+
+ org.apache.spark
+ spark-catalyst_${scala.binary.version}
+ ${spark.version}
+
+
+
+ org.apache.spark
+ spark-core_${scala.binary.version}
+ ${spark.version}
+
+
+
+
+
+ org.apache.spark
+ spark-sql_${scala.binary.version}
+ ${spark.version}
+ tests
+ test
+
+
+
+ org.apache.spark
+ spark-catalyst_${scala.binary.version}
+ ${spark.version}
+ tests
+ test
+
+
+
+ org.apache.spark
+ spark-core_${scala.binary.version}
+ ${spark.version}
+ tests
+ test
+
+
+
+ org.apache.spark
+ spark-hive_${scala.binary.version}
+ ${spark.version}
+ test
+
+
+
+ org.apache.spark
+ spark-avro_${scala.binary.version}
+ ${spark.version}
+ test
+
+
+
+ org.apache.paimon
+ paimon-lumina
+ ${project.version}
+ test
+
+
+
+ org.mockito
+ mockito-core
+ ${mockito.version}
+ test
+
+
+
+
+
+
+ ${ut.module.dir}/src/test/resources
+
+
+
+
+
+ org.codehaus.mojo
+ build-helper-maven-plugin
+
+
+ add-ut-test-sources
+ generate-test-sources
+
+ add-test-source
+
+
+
+ ${ut.module.dir}/src/test/scala
+ ${ut.module.dir}/src/test/java
+
+
+
+
+
+
+
+ org.apache.maven.plugins
+ maven-jar-plugin
+
+
+ prepare-test-jar
+ test-compile
+
+ test-jar
+
+
+
+
+
+
+
+ org.scalatest
+ scalatest-maven-plugin
+
+ true
+
+
+
+
+
diff --git a/paimon-spark/paimon-spark-ut-4.1/pom.xml b/paimon-spark/paimon-spark-ut-4.1/pom.xml
new file mode 100644
index 000000000000..e6a4509c9080
--- /dev/null
+++ b/paimon-spark/paimon-spark-ut-4.1/pom.xml
@@ -0,0 +1,231 @@
+
+
+
+ 4.0.0
+
+
+ org.apache.paimon
+ paimon-spark
+ 2.1-SNAPSHOT
+
+
+ paimon-spark-ut-4.1_${scala.binary.version}
+ Paimon : Spark : UT : 4.1 : ${scala.binary.version}
+
+
+
+ 4.1.2
+ ${project.basedir}/../paimon-spark-ut
+
+ 2.13
+
+
+
+
+
+ org.apache.paimon
+ paimon-format
+
+
+
+ org.apache.paimon
+ ${paimon-sparkx-common}
+ ${project.version}
+
+
+
+
+ org.apache.spark
+ spark-sql-api_${scala.binary.version}
+ ${spark.version}
+
+
+ org.apache.spark
+ spark-connect-shims_${scala.binary.version}
+
+
+
+
+
+ org.apache.spark
+ spark-sql_${scala.binary.version}
+ ${spark.version}
+
+
+ org.apache.spark
+ spark-connect-shims_${scala.binary.version}
+
+
+
+
+
+ org.apache.spark
+ spark-catalyst_${scala.binary.version}
+ ${spark.version}
+
+
+
+ org.apache.spark
+ spark-core_${scala.binary.version}
+ ${spark.version}
+
+
+
+
+
+ org.apache.spark
+ spark-sql_${scala.binary.version}
+ ${spark.version}
+ tests
+ test
+
+
+
+ org.apache.spark
+ spark-catalyst_${scala.binary.version}
+ ${spark.version}
+ tests
+ test
+
+
+
+ org.apache.spark
+ spark-core_${scala.binary.version}
+ ${spark.version}
+ tests
+ test
+
+
+
+ org.apache.spark
+ spark-hive_${scala.binary.version}
+ ${spark.version}
+ test
+
+
+
+ org.apache.spark
+ spark-avro_${scala.binary.version}
+ ${spark.version}
+ test
+
+
+
+ org.apache.paimon
+ paimon-lumina
+ ${project.version}
+ test
+
+
+
+ org.mockito
+ mockito-core
+ ${mockito.version}
+ test
+
+
+
+
+
+
+ ${ut.module.dir}/src/test/resources
+
+
+
+
+
+ org.codehaus.mojo
+ build-helper-maven-plugin
+
+
+ add-ut-test-sources
+ generate-test-sources
+
+ add-test-source
+
+
+
+ ${ut.module.dir}/src/test/scala
+ ${ut.module.dir}/src/test/java
+
+
+
+
+
+
+
+ org.apache.maven.plugins
+ maven-jar-plugin
+
+
+ prepare-test-jar
+ test-compile
+
+ test-jar
+
+
+
+
+
+
+
+ org.scalatest
+ scalatest-maven-plugin
+
+ true
+
+
+
+
+
diff --git a/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/ShowCreateTableTestUtils.java b/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/ShowCreateTableTestUtils.java
new file mode 100644
index 000000000000..59153333829e
--- /dev/null
+++ b/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/ShowCreateTableTestUtils.java
@@ -0,0 +1,54 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.paimon.spark;
+
+import org.apache.spark.sql.SparkSession;
+
+/** Helpers for asserting on {@code SHOW CREATE TABLE} output across Spark versions. */
+public class ShowCreateTableTestUtils {
+
+ private static final String EXPLICIT_BINARY_COLLATION = " COLLATE UTF8_BINARY";
+
+ /** Runs {@code SHOW CREATE TABLE} and returns its output via {@link #stripBinaryCollation}. */
+ public static String showCreateTable(SparkSession spark, String table) {
+ return stripBinaryCollation(
+ spark.sql("SHOW CREATE TABLE " + table).collectAsList().toString());
+ }
+
+ /**
+ * Drops the explicit {@code COLLATE UTF8_BINARY} markers Spark 4.2 adds to every string-ish
+ * column of {@code SHOW CREATE TABLE}.
+ *
+ *
SPARK-55372 made the command print the collation even for a column that has none, so that
+ * replaying the emitted DDL cannot silently pick up a table- or schema-level {@code DEFAULT
+ * COLLATION} instead. Paimon's string columns map to the {@code StringType} case object, which
+ * Spark reads as "no explicit collation" and so renders this way; a built-in source such as
+ * parquet gets the same treatment, so there is nothing Paimon-specific to fix. STRING, VARCHAR,
+ * CHAR and the string leaves of ARRAY / MAP / STRUCT are all affected. The tests in this module
+ * run against every supported Spark version from one copy of the source, so they assert on the
+ * output with the marker removed rather than branching on the version.
+ *
+ *
Only the binary collation is stripped. A column carrying a real collation, say {@code
+ * STRING COLLATE UTF8_LCASE}, still shows up, so an assertion cannot be fooled into accepting
+ * the wrong collation.
+ */
+ public static String stripBinaryCollation(String showCreateTableOutput) {
+ return showCreateTableOutput.replace(EXPLICIT_BINARY_COLLATION, "");
+ }
+}
diff --git a/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/SparkReadITCase.java b/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/SparkReadITCase.java
index fbc2f660d42f..7d54803198b1 100644
--- a/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/SparkReadITCase.java
+++ b/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/SparkReadITCase.java
@@ -45,6 +45,7 @@
import java.util.stream.Collectors;
import java.util.stream.IntStream;
+import static org.apache.paimon.spark.ShowCreateTableTestUtils.showCreateTable;
import static org.assertj.core.api.Assertions.assertThat;
import static org.assertj.core.api.Assertions.assertThatThrownBy;
@@ -194,7 +195,7 @@ public void testCreateTableAs() {
spark.sql(
"CREATE TABLE partitionedTableAs PARTITIONED BY (a) AS SELECT * FROM partitionedTable");
Path tablePath = new Path(warehousePath, "default.db/partitionedTableAs");
- assertThat(spark.sql("SHOW CREATE TABLE partitionedTableAs").collectAsList().toString())
+ assertThat(showCreateTable(spark, "partitionedTableAs"))
.isEqualTo(
String.format(
"[[%s"
@@ -224,7 +225,7 @@ public void testCreateTableAs() {
spark.sql(
"CREATE TABLE testTableAs TBLPROPERTIES ('file.format' = 'parquet') AS SELECT * FROM testTable");
tablePath = new Path(warehousePath, "default.db/testTableAs");
- assertThat(spark.sql("SHOW CREATE TABLE testTableAs").collectAsList().toString())
+ assertThat(showCreateTable(spark, "testTableAs"))
.isEqualTo(
String.format(
"[[%s"
@@ -255,7 +256,7 @@ public void testCreateTableAs() {
spark.sql("INSERT INTO t_pk VALUES(1,'aaa','bbb')");
spark.sql("CREATE TABLE t_pk_as TBLPROPERTIES ('primary-key' = 'a') AS SELECT * FROM t_pk");
tablePath = new Path(warehousePath, "default.db/t_pk_as");
- assertThat(spark.sql("SHOW CREATE TABLE t_pk_as").collectAsList().toString())
+ assertThat(showCreateTable(spark, "t_pk_as"))
.isEqualTo(
String.format(
"[[%s"
@@ -284,7 +285,7 @@ public void testCreateTableAs() {
spark.sql(
"CREATE TABLE t_all_as PARTITIONED BY (dt) TBLPROPERTIES ('primary-key' = 'dt,hh') AS SELECT * FROM t_all");
tablePath = new Path(warehousePath, "default.db/t_all_as");
- assertThat(spark.sql("SHOW CREATE TABLE t_all_as").collectAsList().toString())
+ assertThat(showCreateTable(spark, "t_all_as"))
.isEqualTo(
String.format(
"[[%s"
@@ -388,7 +389,7 @@ public void testShowCreateTable() {
+ ")");
Path tablePath = new Path(warehousePath, "default.db/tbl");
- assertThat(spark.sql("SHOW CREATE TABLE tbl").collectAsList().toString())
+ assertThat(showCreateTable(spark, "tbl"))
.isEqualTo(
String.format(
"[[%s"
@@ -729,7 +730,7 @@ private void innerTestNestedTypeFilterPushDown(Dataset dataset) {
public void testCreateNestedField() {
spark.sql(
"CREATE TABLE nested_table ( a INT, b STRUCT, b2 BIGINT>)");
- assertThat(spark.sql("SHOW CREATE TABLE nested_table").collectAsList().toString())
+ assertThat(showCreateTable(spark, "nested_table"))
.contains(
showCreateString(
"nested_table",
diff --git a/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/SparkSchemaEvolutionITCase.java b/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/SparkSchemaEvolutionITCase.java
index 4e15b0880bde..bc21975e3870 100644
--- a/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/SparkSchemaEvolutionITCase.java
+++ b/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/SparkSchemaEvolutionITCase.java
@@ -33,6 +33,7 @@
import java.util.Map;
import java.util.stream.Collectors;
+import static org.apache.paimon.spark.ShowCreateTableTestUtils.showCreateTable;
import static org.apache.paimon.testutils.assertj.PaimonAssertions.anyCauseMatches;
import static org.assertj.core.api.Assertions.assertThat;
import static org.assertj.core.api.Assertions.assertThatThrownBy;
@@ -76,13 +77,13 @@ public void testAddColumn() {
createTable("testAddColumn");
writeTable("testAddColumn", "(1, 2L, '1')", "(5, 6L, '3')");
- List beforeAdd = spark.sql("SHOW CREATE TABLE testAddColumn").collectAsList();
- assertThat(beforeAdd.toString()).contains(defaultShowCreateString("testAddColumn"));
+ String beforeAdd = showCreateTable(spark, "testAddColumn");
+ assertThat(beforeAdd).contains(defaultShowCreateString("testAddColumn"));
spark.sql("ALTER TABLE testAddColumn ADD COLUMN d STRING");
- List afterAdd = spark.sql("SHOW CREATE TABLE testAddColumn").collectAsList();
- assertThat(afterAdd.toString())
+ String afterAdd = showCreateTable(spark, "testAddColumn");
+ assertThat(afterAdd)
.contains(
showCreateString(
"testAddColumn",
@@ -99,8 +100,8 @@ public void testAddColumn() {
public void testAddNotNullColumn() {
createTable("testAddNotNullColumn");
- List beforeAdd = spark.sql("SHOW CREATE TABLE testAddNotNullColumn").collectAsList();
- assertThat(beforeAdd.toString()).contains(defaultShowCreateString("testAddNotNullColumn"));
+ String beforeAdd = showCreateTable(spark, "testAddNotNullColumn");
+ assertThat(beforeAdd).contains(defaultShowCreateString("testAddNotNullColumn"));
assertThatThrownBy(
() ->
@@ -116,9 +117,8 @@ public void testAddNotNullColumn() {
public void testAddColumnPosition() {
createTable("testAddColumnPositionFirst");
spark.sql("ALTER TABLE testAddColumnPositionFirst ADD COLUMN d INT FIRST");
- List result =
- spark.sql("SHOW CREATE TABLE testAddColumnPositionFirst").collectAsList();
- assertThat(result.toString())
+ String result = showCreateTable(spark, "testAddColumnPositionFirst");
+ assertThat(result)
.contains(
showCreateString(
"testAddColumnPositionFirst",
@@ -129,8 +129,8 @@ public void testAddColumnPosition() {
createTable("testAddColumnPositionAfter");
spark.sql("ALTER TABLE testAddColumnPositionAfter ADD COLUMN d INT AFTER b");
- result = spark.sql("SHOW CREATE TABLE testAddColumnPositionAfter").collectAsList();
- assertThat(result.toString())
+ result = showCreateTable(spark, "testAddColumnPositionAfter");
+ assertThat(result)
.contains(
showCreateString(
"testAddColumnPositionAfter",
@@ -157,8 +157,8 @@ public void testRenameTable() {
assertThat(tables.stream().map(Row::toString))
.containsExactlyInAnyOrder("[default,t2,false]", "[default,t3,false]");
- List afterRename = spark.sql("SHOW CREATE TABLE t3").collectAsList();
- assertThat(afterRename.toString()).contains(defaultShowCreateString("t3"));
+ String afterRename = showCreateTable(spark, "t3");
+ assertThat(afterRename).contains(defaultShowCreateString("t3"));
List data = spark.sql("SELECT * FROM t3").collectAsList();
assertThat(data.toString()).isEqualTo("[[1,2,1], [5,6,3]]");
@@ -169,15 +169,15 @@ public void testRenameColumn() {
createTable("testRenameColumn");
writeTable("testRenameColumn", "(1, 2L, '1')", "(5, 6L, '3')");
- List beforeRename = spark.sql("SHOW CREATE TABLE testRenameColumn").collectAsList();
- assertThat(beforeRename.toString()).contains(defaultShowCreateString("testRenameColumn"));
+ String beforeRename = showCreateTable(spark, "testRenameColumn");
+ assertThat(beforeRename).contains(defaultShowCreateString("testRenameColumn"));
List results = spark.table("testRenameColumn").select("a", "c").collectAsList();
assertThat(results.toString()).isEqualTo("[[1,1], [5,3]]");
// Rename "b" to "bb"
spark.sql("ALTER TABLE testRenameColumn RENAME COLUMN b to bb");
- List afterRename = spark.sql("SHOW CREATE TABLE testRenameColumn").collectAsList();
- assertThat(afterRename.toString())
+ String afterRename = showCreateTable(spark, "testRenameColumn");
+ assertThat(afterRename)
.contains(
showCreateString(
"testRenameColumn", "a INT NOT NULL", "bb BIGINT", "c STRING"));
@@ -203,9 +203,8 @@ public void testRenamePartitionKey() {
+ "a BIGINT,\n"
+ "b STRING)\n"
+ "PARTITIONED BY (a)\n");
- List beforeRename =
- spark.sql("SHOW CREATE TABLE testRenamePartitionKey").collectAsList();
- assertThat(beforeRename.toString())
+ String beforeRename = showCreateTable(spark, "testRenamePartitionKey");
+ assertThat(beforeRename)
.contains(showCreateString("testRenamePartitionKey", "a BIGINT", "b STRING"));
assertThatThrownBy(
@@ -221,13 +220,13 @@ public void testDropSingleColumn() {
createTable("testDropSingleColumn");
writeTable("testDropSingleColumn", "(1, 2L, '1')", "(5, 6L, '3')");
- List beforeDrop = spark.sql("SHOW CREATE TABLE testDropSingleColumn").collectAsList();
- assertThat(beforeDrop.toString()).contains(defaultShowCreateString("testDropSingleColumn"));
+ String beforeDrop = showCreateTable(spark, "testDropSingleColumn");
+ assertThat(beforeDrop).contains(defaultShowCreateString("testDropSingleColumn"));
spark.sql("ALTER TABLE testDropSingleColumn DROP COLUMN b");
- List afterDrop = spark.sql("SHOW CREATE TABLE testDropSingleColumn").collectAsList();
- assertThat(afterDrop.toString())
+ String afterDrop = showCreateTable(spark, "testDropSingleColumn");
+ assertThat(afterDrop)
.contains(showCreateString("testDropSingleColumn", "a INT NOT NULL", "c STRING"));
List results = spark.table("testDropSingleColumn").collectAsList();
@@ -238,14 +237,13 @@ public void testDropSingleColumn() {
public void testDropColumns() {
createTable("testDropColumns");
- List beforeDrop = spark.sql("SHOW CREATE TABLE testDropColumns").collectAsList();
- assertThat(beforeDrop.toString()).contains(defaultShowCreateString("testDropColumns"));
+ String beforeDrop = showCreateTable(spark, "testDropColumns");
+ assertThat(beforeDrop).contains(defaultShowCreateString("testDropColumns"));
spark.sql("ALTER TABLE testDropColumns DROP COLUMNS b,c");
- List afterDrop = spark.sql("SHOW CREATE TABLE testDropColumns").collectAsList();
- assertThat(afterDrop.toString())
- .contains(showCreateString("testDropColumns", "a INT NOT NULL"));
+ String afterDrop = showCreateTable(spark, "testDropColumns");
+ assertThat(afterDrop).contains(showCreateString("testDropColumns", "a INT NOT NULL"));
}
@Test
@@ -273,8 +271,8 @@ public void testDropPartitionKey() {
+ "b STRING) \n"
+ "PARTITIONED BY (a)");
- List beforeDrop = spark.sql("SHOW CREATE TABLE testDropPartitionKey").collectAsList();
- assertThat(beforeDrop.toString())
+ String beforeDrop = showCreateTable(spark, "testDropPartitionKey");
+ assertThat(beforeDrop)
.contains(showCreateString("testDropPartitionKey", "a BIGINT", "b STRING"));
assertThatThrownBy(() -> spark.sql("ALTER TABLE testDropPartitionKey DROP COLUMN a"))
@@ -293,8 +291,8 @@ public void testDropPrimaryKey() {
+ "PARTITIONED BY (a)\n"
+ "TBLPROPERTIES ('primary-key' = 'a, b')");
- List beforeDrop = spark.sql("SHOW CREATE TABLE testDropPrimaryKey").collectAsList();
- assertThat(beforeDrop.toString())
+ String beforeDrop = showCreateTable(spark, "testDropPrimaryKey");
+ assertThat(beforeDrop)
.contains(
showCreateString(
"testDropPrimaryKey", "a BIGINT NOT NULL", "b STRING NOT NULL"));
@@ -318,9 +316,8 @@ public void testRenamePrimaryKey() {
spark.sql("ALTER TABLE test_rename_primary_key_table RENAME COLUMN a to a_");
- List result =
- spark.sql("SHOW CREATE TABLE test_rename_primary_key_table").collectAsList();
- assertThat(result.toString())
+ String result = showCreateTable(spark, "test_rename_primary_key_table");
+ assertThat(result)
.contains(
showCreateString(
"test_rename_primary_key_table", "a_ BIGINT NOT NULL", "b STRING"))
@@ -354,9 +351,8 @@ public void testRenameBucketKey() {
spark.sql("ALTER TABLE test_rename_bucket_key_table RENAME COLUMN b to b_");
- List result =
- spark.sql("SHOW CREATE TABLE test_rename_bucket_key_table").collectAsList();
- assertThat(result.toString())
+ String result = showCreateTable(spark, "test_rename_bucket_key_table");
+ assertThat(result)
.contains(
showCreateString(
"test_rename_bucket_key_table", "a BIGINT NOT NULL", "b_ STRING"))
@@ -376,21 +372,21 @@ public void testUpdateColumnPosition() {
// move first
createTable("tableFirst");
spark.sql("ALTER TABLE tableFirst ALTER COLUMN b FIRST");
- List result = spark.sql("SHOW CREATE TABLE tableFirst").collectAsList();
- assertThat(result.toString())
+ String result = showCreateTable(spark, "tableFirst");
+ assertThat(result)
.contains(showCreateString("tableFirst", "b BIGINT", "a INT NOT NULL", "c STRING"));
// move after
createTable("tableAfter");
spark.sql("ALTER TABLE tableAfter ALTER COLUMN c AFTER a");
- result = spark.sql("SHOW CREATE TABLE tableAfter").collectAsList();
- assertThat(result.toString())
+ result = showCreateTable(spark, "tableAfter");
+ assertThat(result)
.contains(showCreateString("tableAfter", "a INT NOT NULL", "c STRING", "b BIGINT"));
spark.sql("CREATE TABLE tableAfter1 (a INT, b BIGINT, c STRING, d DOUBLE)");
spark.sql("ALTER TABLE tableAfter1 ALTER COLUMN b AFTER c");
- result = spark.sql("SHOW CREATE TABLE tableAfter1").collectAsList();
- assertThat(result.toString())
+ result = showCreateTable(spark, "tableAfter1");
+ assertThat(result)
.contains(
showCreateString(
"tableAfter1", "a INT", "c STRING", "b BIGINT", "d DOUBLE"));
@@ -434,16 +430,16 @@ public void testAlterColumnType() {
})
.hasStackTraceContaining("value appeared in non-nullable field");
- List beforeAlter = spark.sql("SHOW CREATE TABLE testAlterColumnType").collectAsList();
- assertThat(beforeAlter.toString())
+ String beforeAlter = showCreateTable(spark, "testAlterColumnType");
+ assertThat(beforeAlter)
.contains(defaultShowCreateStringWithNonNullColumn("testAlterColumnType"));
spark.sql("ALTER TABLE testAlterColumnType ALTER COLUMN b TYPE DOUBLE");
assertThat(spark.table("testAlterColumnType").collectAsList().toString())
.isEqualTo("[[1,2.0,1], [5,6.0,3]]");
- List afterAlter = spark.sql("SHOW CREATE TABLE testAlterColumnType").collectAsList();
- assertThat(afterAlter.toString())
+ String afterAlter = showCreateTable(spark, "testAlterColumnType");
+ assertThat(afterAlter)
.contains(
showCreateString(
"testAlterColumnType",
@@ -1122,8 +1118,7 @@ public void testAddBlobColumnViaCommentDirective() {
+ table
+ " ADD COLUMN picture BINARY COMMENT '__BLOB_FIELD; profile picture'");
- String createSql =
- spark.sql("SHOW CREATE TABLE " + table).collectAsList().get(0).toString();
+ String createSql = showCreateTable(spark, table);
assertThat(createSql).doesNotContain("__BLOB");
assertThat(createSql).contains("desc_col");
assertThat(createSql).contains("picture");
diff --git a/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/SparkWriteITCase.java b/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/SparkWriteITCase.java
index d040116310a9..990973cfe2ae 100644
--- a/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/SparkWriteITCase.java
+++ b/paimon-spark/paimon-spark-ut/src/test/java/org/apache/paimon/spark/SparkWriteITCase.java
@@ -50,6 +50,7 @@
import scala.collection.JavaConverters;
+import static org.apache.paimon.spark.ShowCreateTableTestUtils.showCreateTable;
import static org.assertj.core.api.Assertions.assertThat;
import static org.assertj.core.api.Assertions.assertThatThrownBy;
@@ -97,8 +98,8 @@ public void testWriteWithDefaultValue() {
+ " ('file.format'='avro')");
// test show create table
- List show = spark.sql("SHOW CREATE TABLE T").collectAsList();
- assertThat(show.toString())
+ String show = showCreateTable(spark, "T");
+ assertThat(show)
.contains("a INT,\n" + " b INT DEFAULT 2,\n" + " c STRING DEFAULT 'my_value'");
// test partial write
@@ -140,8 +141,8 @@ public void testWriteWithArrayDefaultValue() {
+ " ('file.format'='avro')");
// test show create table for array
- List show = spark.sql("SHOW CREATE TABLE T").collectAsList();
- assertThat(show.toString())
+ String show = showCreateTable(spark, "T");
+ assertThat(show)
.contains("tags ARRAY DEFAULT ARRAY('tag1', 'tag2')")
.contains("numbers ARRAY DEFAULT ARRAY(1, 2, 3)");
@@ -199,8 +200,8 @@ public void testWriteWithMapDefaultValue() {
+ " ('file.format'='avro')");
// test show create table for map
- List show = spark.sql("SHOW CREATE TABLE T").collectAsList();
- assertThat(show.toString())
+ String show = showCreateTable(spark, "T");
+ assertThat(show)
.contains(
"properties MAP DEFAULT MAP('key1', 'value1', 'key2', 'value2')");
@@ -243,8 +244,8 @@ public void testWriteWithStructDefaultValue() {
+ " ('file.format'='avro')");
// test show create table for struct
- List show = spark.sql("SHOW CREATE TABLE T").collectAsList();
- assertThat(show.toString())
+ String show = showCreateTable(spark, "T");
+ assertThat(show)
.contains("nested STRUCT DEFAULT STRUCT(42, 'default_value')");
// test partial write with struct defaults
@@ -281,8 +282,8 @@ public void testWriteWithNestedComplexDefaultValue() {
+ " ('file.format'='avro')");
// test show create table for nested complex types
- List show = spark.sql("SHOW CREATE TABLE T").collectAsList();
- assertThat(show.toString())
+ String show = showCreateTable(spark, "T");
+ assertThat(show)
.contains(
"nested_array ARRAY> DEFAULT ARRAY(STRUCT('item1', 10), STRUCT('item2', 20))")
.contains(
diff --git a/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/sql/CatalogQualifiedCreateTableLikeTest.scala b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/sql/CatalogQualifiedCreateTableLikeTest.scala
index c4eb2cd6443a..2f09f61ba42b 100644
--- a/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/sql/CatalogQualifiedCreateTableLikeTest.scala
+++ b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/sql/CatalogQualifiedCreateTableLikeTest.scala
@@ -143,13 +143,24 @@ class CatalogQualifiedCreateTableLikeTest extends PaimonSparkTestBase {
parseCreateTableLikeCommand(
"CREATE TABLE paimon.test.extra.target_tbl LIKE paimon.test.extra.source_tbl")
Assertions.assertEquals("target_tbl", nestedIdentifierCommand.targetIdent.name())
- Assertions.assertEquals(
- Seq("test.extra"),
- nestedIdentifierCommand.targetIdent.namespace().toSeq)
Assertions.assertEquals("source_tbl", nestedIdentifierCommand.sourceIdent.name())
+ // Assert only that the middle parts survive, in order: how they are *split* differs by version
+ // and is not what this test is about. Up to 4.1 Spark's own parser rejects a catalog-qualified
+ // CREATE TABLE LIKE, so Paimon's extension parser takes over and flattens everything between
+ // the catalog and the table into the single `database` slot of a V1 `TableIdentifier`; 4.2
+ // parses the statement itself into `CreateTableLike` and the raw name parts survive as-is.
+ //
+ // Neither shape is a table Paimon can create, but for different reasons: on 4.2 the two-part
+ // namespace violates the single-namespace rule (`CatalogUtils.checkNamespace`), while on <= 4.1
+ // the flattened `test.extra` is a legal namespace that simply does not exist. This test stops
+ // at the parser -- what it pins is that a name with an extra part is accepted and lands on
+ // `PaimonCreateTableLikeCommand` carrying the right table names.
+ Assertions.assertEquals(
+ "test.extra",
+ nestedIdentifierCommand.targetIdent.namespace().mkString("."))
Assertions.assertEquals(
- Seq("test.extra"),
- nestedIdentifierCommand.sourceIdent.namespace().toSeq)
+ "test.extra",
+ nestedIdentifierCommand.sourceIdent.namespace().mkString("."))
}
private def createSourceTable(): Unit = {
diff --git a/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/sql/PaimonV1FunctionTestBase.scala b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/sql/PaimonV1FunctionTestBase.scala
index 367327ed47fb..ee5686d36cfd 100644
--- a/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/sql/PaimonV1FunctionTestBase.scala
+++ b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/sql/PaimonV1FunctionTestBase.scala
@@ -97,6 +97,22 @@ abstract class PaimonV1FunctionTestBase extends PaimonSparkTestWithRestCatalogBa
}
}
+ test("Paimon V1 Function: default output column name excludes the catalog") {
+ // The registry key is the fully qualified 3-part identifier, but the name handed to the
+ // expression builder must stay 2-part, or an unaliased scalar/aggregate `SELECT udf(...)`
+ // column gets renamed from `db.udf(...)` to `catalog.db.udf(...)`. See
+ // `PaimonV1FunctionRegistry`'s `makeFunctionBuilder` for the mechanism.
+ withUserDefinedFunction("udf_add2" -> false) {
+ sql(s"""
+ |CREATE FUNCTION udf_add2 AS '$UDFExampleAdd2Class'
+ |USING JAR '$testUDFJarPath'
+ |""".stripMargin)
+ // `dbName0`, not a literal: the expected name follows the session's current database, which
+ // the test base sets. A literal would keep passing if that ever changed.
+ assert(sql("SELECT udf_add2(3, 4)").columns.toSeq === Seq(s"$dbName0.udf_add2(3, 4)"))
+ }
+ }
+
test("Paimon V1 Function: select with built-in function") {
withUserDefinedFunction("udf_add2" -> false) {
sql(s"""
@@ -261,11 +277,19 @@ abstract class PaimonV1FunctionTestBase extends PaimonSparkTestWithRestCatalogBa
|USING JAR '$testUDFJarPath'
|""".stripMargin)
- assert(intercept[Exception] {
+ // Spark reworded this in 4.2: `error-conditions.json` dropped the "built-in/temporary"
+ // phrasing in favour of pointing at DROP TEMPORARY FUNCTION. Both mean the same refusal, and
+ // this suite runs on every supported version, so accept either.
+ val dropTempMessage = intercept[Exception] {
sql(s"""
|DROP FUNCTION udf_add2
|""".stripMargin)
- }.getMessage.contains("udf_add2 is a built-in/temporary function"))
+ }.getMessage
+ assert(
+ dropTempMessage.contains("udf_add2 is a built-in/temporary function") ||
+ dropTempMessage.contains("'DROP FUNCTION' expects a persistent function"),
+ s"unexpected DROP FUNCTION refusal: $dropTempMessage"
+ )
}
}
diff --git a/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/sql/VariantTestBase.scala b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/sql/VariantTestBase.scala
index 4cb3837071c9..cddb258eceea 100644
--- a/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/sql/VariantTestBase.scala
+++ b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/sql/VariantTestBase.scala
@@ -31,7 +31,7 @@ abstract class VariantTestBase extends PaimonSparkTestBase {
private def scanReadSchemaOf(df: org.apache.spark.sql.DataFrame): StructType = {
df.queryExecution.optimizedPlan
- .collectFirst { case DataSourceV2ScanRelation(_, scan, _, _, _) => scan.readSchema() }
+ .collectFirst { case r: DataSourceV2ScanRelation => r.scan.readSchema() }
.getOrElse(fail("expected a DataSourceV2ScanRelation in the optimized plan"))
}
@@ -1107,7 +1107,7 @@ abstract class VariantTestBase extends PaimonSparkTestBase {
val df = sql("SELECT variant_get(v, '$.age', 'int') FROM T")
val desc = df.queryExecution.optimizedPlan
- .collectFirst { case DataSourceV2ScanRelation(_, scan, _, _, _) => scan.description() }
+ .collectFirst { case r: DataSourceV2ScanRelation => r.scan.description() }
.getOrElse(fail("expected a DataSourceV2ScanRelation in the plan"))
if (variantPushDownEnabled) {
diff --git a/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/spark/sql/paimon/shims/SparkVersionCompatTest.scala b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/spark/sql/paimon/shims/SparkVersionCompatTest.scala
new file mode 100644
index 000000000000..62a76fb4a03f
--- /dev/null
+++ b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/spark/sql/paimon/shims/SparkVersionCompatTest.scala
@@ -0,0 +1,113 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.spark.sql.paimon.shims
+
+import org.apache.paimon.spark.PaimonSparkTestBase
+
+import org.apache.spark.sql.catalyst.analysis.UnresolvedFunction
+
+class SparkVersionCompatTest extends PaimonSparkTestBase {
+
+ private def catalogManager = spark.sessionState.catalogManager
+
+ test("isBuiltinFunction recognizes a Spark builtin across versions") {
+ val catalog = SparkVersionCompat.v1SessionCatalog(catalogManager)
+ assert(SparkVersionCompat.isBuiltinFunction(catalog, "upper"))
+ }
+
+ test("isBuiltinFunction rejects a non-existent function") {
+ val catalog = SparkVersionCompat.v1SessionCatalog(catalogManager)
+ assert(!SparkVersionCompat.isBuiltinFunction(catalog, "paimon_no_such_fn"))
+ }
+
+ // Spark 4.2 turned `CatalogManager` from a class into an interface, which changes the invoke
+ // opcode a direct call compiles to and makes a 4.2-built classfile unusable on 4.0/4.1. These
+ // four accessors go through reflection to stay version-neutral; the tests assert they return
+ // what a direct call would, so a wrong `Method` lookup cannot pass unnoticed.
+
+ test("currentCatalog matches the session's configured catalog") {
+ // `PaimonSparkTestBase.beforeEach` leaves the session on the `paimon` catalog.
+ assert(SparkVersionCompat.currentCatalog(catalogManager).name() === "paimon")
+ }
+
+ test("catalog looks up a catalog by name") {
+ val byName = SparkVersionCompat.catalog(catalogManager, "paimon")
+ assert(byName === SparkVersionCompat.currentCatalog(catalogManager))
+ }
+
+ test("catalog propagates the original failure for an unknown name") {
+ // `SparkUtils.catalogAndIdentifier` relies on catching this to fall back to the default
+ // catalog, so the reflective layer must not wrap it in an InvocationTargetException.
+ val e = intercept[Exception](SparkVersionCompat.catalog(catalogManager, "no_such_catalog"))
+ assert(!e.isInstanceOf[java.lang.reflect.InvocationTargetException])
+ assert(e.getMessage.contains("no_such_catalog"))
+ }
+
+ test("currentNamespace returns the session's namespace") {
+ assert(SparkVersionCompat.currentNamespace(catalogManager).toSeq === Seq("test"))
+ }
+
+ test("v1SessionCatalog returns a usable SessionCatalog") {
+ assert(SparkVersionCompat.v1SessionCatalog(catalogManager).databaseExists("default"))
+ }
+
+ test("ignoreNulls reads false for a plain UnresolvedFunction") {
+ // `parseExpression` yields an unresolved node on every supported Spark version; building
+ // `UnresolvedFunction` directly is not portable (the `Seq[String]`-first 3-arg apply does
+ // not exist on Spark 3.5).
+ val expr = spark.sessionState.sqlParser.parseExpression("upper('a')")
+ val u = expr.collectFirst { case u: UnresolvedFunction => u }.get
+ assert(!SparkVersionCompat.ignoreNulls(u))
+ }
+
+ test("ignoreNulls reads true when IGNORE NULLS is specified") {
+ val expr = spark.sessionState.sqlParser
+ .parseExpression("first(a) IGNORE NULLS")
+ val u = expr.collectFirst { case u: UnresolvedFunction => u }.get
+ assert(SparkVersionCompat.ignoreNulls(u))
+ }
+
+ // The `Option` shapes below only occur on Spark 4.2+, where `UnresolvedFunction.ignoreNulls`
+ // returns `Option[Boolean]`. Testing the normalization directly keeps that branch covered on
+ // every profile, instead of leaving it unexercised until the Spark baseline moves.
+
+ test("toBoolean reads a boxed Boolean (Spark <= 4.1)") {
+ assert(SparkVersionCompat.toBoolean(java.lang.Boolean.TRUE))
+ assert(!SparkVersionCompat.toBoolean(java.lang.Boolean.FALSE))
+ }
+
+ test("toBoolean reads Some(true) (Spark 4.2+)") {
+ assert(SparkVersionCompat.toBoolean(Some(true)))
+ }
+
+ test("toBoolean reads Some(false) (Spark 4.2+)") {
+ assert(!SparkVersionCompat.toBoolean(Some(false)))
+ }
+
+ test("toBoolean treats None as not specified (Spark 4.2+)") {
+ assert(!SparkVersionCompat.toBoolean(None))
+ }
+
+ test("toBoolean rejects an unrecognized shape instead of defaulting to false") {
+ // Guards against a future Spark widening `ignoreNulls` again: silently reading such a value
+ // as `false` would drop an IGNORE NULLS clause and return wrong results.
+ val e = intercept[IllegalStateException](SparkVersionCompat.toBoolean(Some("yes")))
+ assert(e.getMessage.contains("Unexpected UnresolvedFunction.ignoreNulls value"))
+ }
+}
diff --git a/paimon-spark/paimon-spark3-common/src/main/scala/org/apache/spark/sql/paimon/shims/Spark3Shim.scala b/paimon-spark/paimon-spark3-common/src/main/scala/org/apache/spark/sql/paimon/shims/Spark3Shim.scala
index b568ce16c3c4..6fd5de63d80e 100644
--- a/paimon-spark/paimon-spark3-common/src/main/scala/org/apache/spark/sql/paimon/shims/Spark3Shim.scala
+++ b/paimon-spark/paimon-spark3-common/src/main/scala/org/apache/spark/sql/paimon/shims/Spark3Shim.scala
@@ -34,10 +34,12 @@ import org.apache.hadoop.fs.Path
import org.apache.spark.sql.SparkSession
import org.apache.spark.sql.catalyst.InternalRow
import org.apache.spark.sql.catalyst.analysis.{CTESubstitution, SubstituteUnresolvedOrdinals}
+import org.apache.spark.sql.catalyst.analysis.NamedRelation
+import org.apache.spark.sql.catalyst.catalog.CatalogStorageFormat
import org.apache.spark.sql.catalyst.expressions.{Attribute, AttributeReference, Expression}
import org.apache.spark.sql.catalyst.expressions.aggregate.AggregateExpression
import org.apache.spark.sql.catalyst.parser.ParserInterface
-import org.apache.spark.sql.catalyst.plans.logical.{Aggregate, Assignment, CTERelationRef, InsertAction, LogicalPlan, MergeAction, MergeIntoTable, SubqueryAlias, TableSpec, UnresolvedWith, UpdateAction}
+import org.apache.spark.sql.catalyst.plans.logical.{Aggregate, Assignment, CTERelationRef, DescribeRelation, InsertAction, LogicalPlan, MergeAction, MergeIntoTable, OverwriteByExpression, OverwritePartitionsDynamic, SubqueryAlias, TableSpec, UnresolvedWith, UpdateAction}
import org.apache.spark.sql.catalyst.plans.physical.Distribution
// NOTE: `MergeRows` / `MergeRows.Keep` were introduced in Spark 3.4. We access them only via
// reflection inside the `mergeRowsKeep*` method bodies so that loading `Spark3Shim` does not fail
@@ -52,12 +54,13 @@ import org.apache.spark.sql.connector.read.Scan
import org.apache.spark.sql.connector.write.BatchWrite
import org.apache.spark.sql.execution.{SparkFormatTable, SparkPlan}
import org.apache.spark.sql.execution.datasources.{PartitioningAwareFileIndex, PartitionSpec}
-import org.apache.spark.sql.execution.datasources.v2.{AtomicReplaceTableAsSelectExec, AtomicReplaceTableExec, ReplaceTableAsSelectExec, ReplaceTableExec}
+import org.apache.spark.sql.execution.datasources.v2.{AtomicReplaceTableAsSelectExec, AtomicReplaceTableExec, CreateTableAsSelectExec, DescribeTableExec, ReplaceTableAsSelectExec, ReplaceTableExec}
import org.apache.spark.sql.execution.datasources.v2.{DataSourceV2Relation, DataSourceV2ScanRelation}
import org.apache.spark.sql.execution.streaming.{FileStreamSink, MetadataLogFileIndex}
import org.apache.spark.sql.internal.SQLConf
import org.apache.spark.sql.types.StructType
+import java.net.URI
import java.util.{Map => JMap}
class Spark3Shim extends SparkShim {
@@ -96,6 +99,42 @@ class Spark3Shim extends SparkShim {
tableCatalog.createTable(ident, schema, partitions, properties)
}
+ override def withStorageLocation(
+ storage: CatalogStorageFormat,
+ locationUri: Option[URI]): CatalogStorageFormat =
+ storage.copy(locationUri = locationUri)
+
+ override def overwriteByName(
+ table: NamedRelation,
+ query: LogicalPlan,
+ deleteExpr: Expression,
+ writeOptions: Map[String, String]): OverwriteByExpression =
+ OverwriteByExpression.byName(table, query, deleteExpr, writeOptions)
+
+ override def overwritePartitionsDynamicByName(
+ table: NamedRelation,
+ query: LogicalPlan,
+ writeOptions: Map[String, String]): OverwritePartitionsDynamic =
+ OverwritePartitionsDynamic.byName(table, query, writeOptions)
+
+ override def createCreateTableAsSelectExec(
+ catalog: TableCatalog,
+ ident: Identifier,
+ partitioning: Seq[Transform],
+ query: LogicalPlan,
+ tableSpec: TableSpec,
+ writeOptions: Map[String, String],
+ ifNotExists: Boolean): SparkPlan = {
+ CreateTableAsSelectExec(
+ catalog,
+ ident,
+ partitioning,
+ query,
+ tableSpec,
+ writeOptions,
+ ifNotExists)
+ }
+
override def createReplaceTableAsSelectExec(
catalog: TableCatalog,
ident: Identifier,
@@ -414,6 +453,31 @@ class Spark3Shim extends SparkShim {
parser: org.apache.spark.sql.catalyst.parser.ParserInterface): Expression =
throw new UnsupportedOperationException(
"SQL user-defined functions (CREATE FUNCTION ... RETURN) require Spark 4.0 or later.")
+
+ // Spark 4.0/4.1 have no `CreateTableLike` logical plan; `CREATE TABLE LIKE` still arrives as the
+ // V1 `CreateTableLikeCommand`, which `RewriteCreateTableLikeCommand` matches directly.
+ override def createTableLikeParts(plan: LogicalPlan)
+ : Option[(Seq[String], Seq[String], Option[String], Option[String], Map[String, String], Boolean, Boolean)] =
+ None
+
+ // Spark 3.x/4.0/4.1 keep `DESCRIBE ... PARTITION` inside `DescribeRelation`; see
+ // `describeRelationPartitionSpec`.
+ override def describeTablePartition(
+ plan: LogicalPlan): Option[(LogicalPlan, Map[String, String], Boolean, Seq[Attribute])] = None
+
+ override def describeRelationPartitionSpec(plan: DescribeRelation): Map[String, String] =
+ plan.partitionSpec
+
+ override def createDescribeTableExec(
+ output: Seq[Attribute],
+ catalogName: String,
+ identifier: Identifier,
+ table: Table,
+ isExtended: Boolean): SparkPlan =
+ DescribeTableExec(output, table, isExtended)
+
+ // Spark 3.x has no schema evolution for MERGE INTO.
+ override def mergeNeedsSchemaEvolution(merge: MergeIntoTable): Boolean = false
}
object Spark3Shim {
diff --git a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/catalog/functions/SQLFunctionConverter.scala b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/catalog/functions/SQLFunctionConverter.scala
index 4aee17aa2ea9..5f05ac654751 100644
--- a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/catalog/functions/SQLFunctionConverter.scala
+++ b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/catalog/functions/SQLFunctionConverter.scala
@@ -63,7 +63,8 @@ object SQLFunctionConverter {
val inputParams: JList[DataField] = inputParamText.filter(_.trim.nonEmpty) match {
case Some(text) =>
SparkTypeUtils
- .toPaimonRowType(UserDefinedFunction.parseRoutineParam(text, parser))
+ // Spark 4.2 added a trailing `collation` parameter with no default.
+ .toPaimonRowType(UserDefinedFunction.parseRoutineParam(text, parser, None))
.getFields
case None => Collections.emptyList[DataField]()
}
@@ -141,6 +142,7 @@ object SQLFunctionConverter {
exprText = if (isQuery) None else Some(body),
queryText = if (isQuery) Some(body) else None,
comment = Option(function.comment()),
+ collation = None,
deterministic = deterministic,
containsSQL = Option(options.get(CONTAINS_SQL)).map(_.toBoolean),
isTableFunc = false,
@@ -158,7 +160,7 @@ object SQLFunctionConverter {
funcIdent: FunctionIdentifier,
returnTypeText: String,
parser: ParserInterface): SparkDataType =
- SQLFunction.parseReturnTypeText(returnTypeText, isTableFunc = false, parser) match {
+ SQLFunction.parseReturnTypeText(returnTypeText, isTableFunc = false, parser, None) match {
case Some(Left(dataType)) => dataType
case _ =>
throw new UnsupportedOperationException(
diff --git a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/MergePaimonScalarSubqueries.scala b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/MergePaimonScalarSubqueries.scala
index e86195f1af0b..dc9fd53d98a3 100644
--- a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/MergePaimonScalarSubqueries.scala
+++ b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/catalyst/optimizer/MergePaimonScalarSubqueries.scala
@@ -30,20 +30,18 @@ object MergePaimonScalarSubqueries extends MergePaimonScalarSubqueriesBase {
newV2ScanRelation: DataSourceV2ScanRelation,
cachedV2ScanRelation: DataSourceV2ScanRelation)
: Option[(LogicalPlan, AttributeMap[Attribute])] = {
- (newV2ScanRelation, cachedV2ScanRelation) match {
- case (
- DataSourceV2ScanRelation(
- newRelation,
- newScan: PaimonScan,
- newOutput,
- newPartitioning,
- newOrdering),
- DataSourceV2ScanRelation(
- cachedRelation,
- cachedScan: PaimonScan,
- _,
- cachedPartitioning,
- cacheOrdering)) =>
+ // Match by type and read fields through named accessors: Spark 4.2 (SPARK-56385) added a
+ // sixth `pushedFilters` parameter, which breaks positional patterns.
+ (newV2ScanRelation.scan, cachedV2ScanRelation.scan) match {
+ case (newScan: PaimonScan, cachedScan: PaimonScan) =>
+ val newRelation = newV2ScanRelation.relation
+ val newOutput = newV2ScanRelation.output
+ val newPartitioning = newV2ScanRelation.keyGroupedPartitioning
+ val newOrdering = newV2ScanRelation.ordering
+ val cachedRelation = cachedV2ScanRelation.relation
+ val cachedPartitioning = cachedV2ScanRelation.keyGroupedPartitioning
+ val cacheOrdering = cachedV2ScanRelation.ordering
+
checkIdenticalPlans(newRelation, cachedRelation).flatMap {
outputMap =>
if (
@@ -60,14 +58,14 @@ object MergePaimonScalarSubqueries extends MergePaimonScalarSubqueriesBase {
val cachedOutputNameMap = cachedRelation.output.map(a => a.name -> a).toMap
val mergedOutput =
mergedAttributes.map(a => cachedOutputNameMap.getOrElse(a.name, a))
- val newV2ScanRelation =
+ val mergedV2ScanRelation =
cachedV2ScanRelation.copy(scan = mergedScan, output = mergedOutput)
val mergedOutputNameMap = mergedOutput.map(a => a.name -> a).toMap
val newOutputMap =
AttributeMap(newOutput.map(a => a -> mergedOutputNameMap(a.name).toAttribute))
- newV2ScanRelation -> newOutputMap
+ mergedV2ScanRelation -> newOutputMap
}
} else {
None
diff --git a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/data/Spark4ArrayData.scala b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/data/Spark4ArrayData.scala
index 80e0456568e5..7cf9e186755b 100644
--- a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/data/Spark4ArrayData.scala
+++ b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/data/Spark4ArrayData.scala
@@ -21,7 +21,7 @@ package org.apache.paimon.spark.data
import org.apache.paimon.types.{DataType, GeographyType, GeometryType}
import org.apache.spark.sql.paimon.shims.SparkShimLoader
-import org.apache.spark.unsafe.types.{GeographyVal, GeometryVal, VariantVal}
+import org.apache.spark.unsafe.types.{BinaryView, VariantVal}
class Spark4ArrayData(override val elementType: DataType) extends AbstractSparkArrayData {
@@ -30,18 +30,20 @@ class Spark4ArrayData(override val elementType: DataType) extends AbstractSparkA
new VariantVal(v.value(), v.metadata())
}
- override def getGeography(ordinal: Int): GeographyVal =
- SparkShimLoader.shim
- .toSparkGeography(
- paimonArray.getBinary(ordinal),
- elementType.asInstanceOf[GeographyType].getCrs,
- elementType.asInstanceOf[GeographyType].getAlgorithm.toString)
- .asInstanceOf[GeographyVal]
-
- override def getGeometry(ordinal: Int): GeometryVal =
- SparkShimLoader.shim
- .toSparkGeometry(
- paimonArray.getBinary(ordinal),
- elementType.asInstanceOf[GeometryType].getCrs)
- .asInstanceOf[GeometryVal]
+ // Spark 4.2 (SPARK-57058) replaced `getGeography` / `getGeometry` on `SpecializedGetters` with a
+ // single `getBinaryView`; the geo value classes `GeographyVal` / `GeometryVal` were removed with
+ // them. Dispatch on the Paimon element type, which is what the pre-4.2 pair of overrides did
+ // implicitly. `paimon-spark-4.1` forks this class to keep the older two overrides.
+ override def getBinaryView(ordinal: Int): BinaryView = elementType match {
+ case g: GeographyType =>
+ SparkShimLoader.shim
+ .toSparkGeography(paimonArray.getBinary(ordinal), g.getCrs, g.getAlgorithm.toString)
+ .asInstanceOf[BinaryView]
+ case g: GeometryType =>
+ SparkShimLoader.shim
+ .toSparkGeometry(paimonArray.getBinary(ordinal), g.getCrs)
+ .asInstanceOf[BinaryView]
+ case other =>
+ throw new UnsupportedOperationException(s"Not a BinaryView-backed Paimon type: $other")
+ }
}
diff --git a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/data/Spark4InternalRow.scala b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/data/Spark4InternalRow.scala
index dc54eb4c6094..5ad8855f7bf9 100644
--- a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/data/Spark4InternalRow.scala
+++ b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/paimon/spark/data/Spark4InternalRow.scala
@@ -22,7 +22,7 @@ import org.apache.paimon.spark.AbstractSparkInternalRow
import org.apache.paimon.types.{GeographyType, GeometryType, RowType}
import org.apache.spark.sql.paimon.shims.SparkShimLoader
-import org.apache.spark.unsafe.types.{GeographyVal, GeometryVal, VariantVal}
+import org.apache.spark.unsafe.types.{BinaryView, VariantVal}
class Spark4InternalRow(rowType: RowType) extends AbstractSparkInternalRow(rowType) {
@@ -31,23 +31,20 @@ class Spark4InternalRow(rowType: RowType) extends AbstractSparkInternalRow(rowTy
new VariantVal(v.value(), v.metadata())
}
- override def getGeography(ordinal: Int): GeographyVal =
- SparkShimLoader.shim
- .toSparkGeography(
- row.getBinary(ordinal),
- rowType.getTypeAt(ordinal).asInstanceOf[GeographyType].getCrs,
- rowType
- .getTypeAt(ordinal)
- .asInstanceOf[GeographyType]
- .getAlgorithm
- .toString
- )
- .asInstanceOf[GeographyVal]
-
- override def getGeometry(ordinal: Int): GeometryVal =
- SparkShimLoader.shim
- .toSparkGeometry(
- row.getBinary(ordinal),
- rowType.getTypeAt(ordinal).asInstanceOf[GeometryType].getCrs)
- .asInstanceOf[GeometryVal]
+ // Spark 4.2 (SPARK-57058) replaced `getGeography` / `getGeometry` on `SpecializedGetters` with a
+ // single `getBinaryView`; the geo value classes `GeographyVal` / `GeometryVal` were removed with
+ // them. Dispatch on the Paimon field type, which is what the pre-4.2 pair of overrides did
+ // implicitly. `paimon-spark-4.1` forks this class to keep the older two overrides.
+ override def getBinaryView(ordinal: Int): BinaryView = rowType.getTypeAt(ordinal) match {
+ case g: GeographyType =>
+ SparkShimLoader.shim
+ .toSparkGeography(row.getBinary(ordinal), g.getCrs, g.getAlgorithm.toString)
+ .asInstanceOf[BinaryView]
+ case g: GeometryType =>
+ SparkShimLoader.shim
+ .toSparkGeometry(row.getBinary(ordinal), g.getCrs)
+ .asInstanceOf[BinaryView]
+ case other =>
+ throw new UnsupportedOperationException(s"Not a BinaryView-backed Paimon type: $other")
+ }
}
diff --git a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41MergeIntoRewrite.scala b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41MergeIntoRewrite.scala
index 0d012e18f249..64b72b7f21a6 100644
--- a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41MergeIntoRewrite.scala
+++ b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41MergeIntoRewrite.scala
@@ -28,12 +28,13 @@ import org.apache.spark.sql.catalyst.expressions.aggregate.AggregateExpression
import org.apache.spark.sql.catalyst.plans.{FullOuter, Inner, JoinType, LeftAnti, LeftOuter, RightOuter}
import org.apache.spark.sql.catalyst.plans.logical.{AnalysisHelper, AppendData, DeleteAction, Filter, HintInfo, InsertAction, Join, JoinHint, LogicalPlan, MergeAction, MergeIntoTable, MergeRows, NO_BROADCAST_AND_REPLICATION, Project, ReplaceData, UpdateAction, WriteDelta}
import org.apache.spark.sql.catalyst.plans.logical.MergeRows.{Copy, Delete, Discard, Insert, Instruction, Keep, ROW_ID, Update}
-import org.apache.spark.sql.catalyst.util.RowDeltaUtils.{OPERATION_COLUMN, WRITE_OPERATION, WRITE_WITH_METADATA_OPERATION}
+import org.apache.spark.sql.catalyst.util.RowDeltaUtils.{COPY_OPERATION, INSERT_OPERATION, OPERATION_COLUMN, UPDATE_OPERATION}
import org.apache.spark.sql.connector.catalog.SupportsRowLevelOperations
import org.apache.spark.sql.connector.write.{RowLevelOperationTable, SupportsDelta}
import org.apache.spark.sql.connector.write.RowLevelOperation.Command.MERGE
import org.apache.spark.sql.errors.QueryCompilationErrors
import org.apache.spark.sql.execution.datasources.v2.{DataSourceV2Relation, ExtractV2Table}
+import org.apache.spark.sql.paimon.shims.SparkShimLoader
import org.apache.spark.sql.types.IntegerType
import org.apache.spark.sql.util.CaseInsensitiveStringMap
@@ -73,7 +74,8 @@ object Spark41MergeIntoRewrite
AnalysisHelper.allowInvokingTransformsInAnalyzer {
plan.transformDown {
case m: MergeIntoTable
- if m.resolved && m.rewritable && !m.needSchemaEvolution &&
+ if m.resolved && m.rewritable &&
+ !SparkShimLoader.shim.mergeNeedsSchemaEvolution(m) &&
(targetsV2CopyOnWriteTable(m.targetTable) || targetsV2DeltaTable(m.targetTable)) =>
// Pure append-only tables skip postHoc `PaimonMergeInto`, so evolve schema here.
val evolved = evolveSchemaIfPaimon(m)
@@ -380,7 +382,7 @@ object Spark41MergeIntoRewrite
checkCardinality: Boolean): MergeRows = {
// Unmatched target rows must be copied through since groups are being replaced wholesale.
- val carryoverRowsOutput = Literal(WRITE_WITH_METADATA_OPERATION) +: targetTable.output
+ val carryoverRowsOutput = Literal(COPY_OPERATION) +: targetTable.output
val keepCarryoverRowsInstruction = Keep(Copy, TrueLiteral, carryoverRowsOutput)
val matchedInstructions = matchedActions.map {
@@ -463,13 +465,18 @@ object Spark41MergeIntoRewrite
}
}
- // Mirrors `RewriteMergeIntoTable.toInstruction`.
+ // Mirrors `RewriteMergeIntoTable.toInstruction`. Spark 4.2 (SPARK-56510) dropped
+ // WRITE_WITH_METADATA_OPERATION / WRITE_OPERATION: an updated row now carries UPDATE_OPERATION
+ // and a newly inserted row INSERT_OPERATION, while COPY_OPERATION is reserved for carried-over
+ // rows. The distinction matters on the write side — `DataAndMetadataWritingSparkTask` routes
+ // UPDATE/COPY through `write(metadata, data)` and INSERT through `write(data)`, so an inserted
+ // row must not be labelled COPY or it would be written with an all-null metadata row.
private def toInstruction(action: MergeAction, metadataAttrs: Seq[Attribute]): Instruction = {
action match {
case UpdateAction(cond, assignments, _) =>
val rowValues = assignments.map(_.value)
val metadataValues = nullifyMetadataOnUpdate(metadataAttrs)
- val output = Seq(Literal(WRITE_WITH_METADATA_OPERATION)) ++ rowValues ++ metadataValues
+ val output = Seq(Literal(UPDATE_OPERATION)) ++ rowValues ++ metadataValues
Keep(Update, cond.getOrElse(TrueLiteral), output)
case DeleteAction(cond) =>
@@ -478,7 +485,7 @@ object Spark41MergeIntoRewrite
case InsertAction(cond, assignments) =>
val rowValues = assignments.map(_.value)
val metadataValues = metadataAttrs.map(attr => Literal(null, attr.dataType))
- val output = Seq(Literal(WRITE_OPERATION)) ++ rowValues ++ metadataValues
+ val output = Seq(Literal(INSERT_OPERATION)) ++ rowValues ++ metadataValues
Keep(Insert, cond.getOrElse(TrueLiteral), output)
case other =>
diff --git a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41UpdateTableRewrite.scala b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41UpdateTableRewrite.scala
index e669f3a1dffd..5fa8faec2193 100644
--- a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41UpdateTableRewrite.scala
+++ b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/catalyst/analysis/Spark41UpdateTableRewrite.scala
@@ -23,7 +23,7 @@ import org.apache.paimon.spark.catalyst.analysis.PaimonAssignmentUtils
import org.apache.spark.sql.catalyst.expressions.{Alias, Attribute, EqualNullSafe, Expression, If, Literal, MetadataAttribute, Not, SubqueryExpression}
import org.apache.spark.sql.catalyst.expressions.Literal.TrueLiteral
import org.apache.spark.sql.catalyst.plans.logical.{AnalysisHelper, Assignment, Filter, LogicalPlan, Project, ReplaceData, Union, UpdateTable, WriteDelta}
-import org.apache.spark.sql.catalyst.util.RowDeltaUtils.{OPERATION_COLUMN, UPDATE_OPERATION, WRITE_WITH_METADATA_OPERATION}
+import org.apache.spark.sql.catalyst.util.RowDeltaUtils.{COPY_OPERATION, OPERATION_COLUMN, UPDATE_OPERATION}
import org.apache.spark.sql.connector.catalog.SupportsRowLevelOperations
import org.apache.spark.sql.connector.write.{RowLevelOperationTable, SupportsDelta}
import org.apache.spark.sql.connector.write.RowLevelOperation.Command.UPDATE
@@ -136,7 +136,13 @@ object Spark41UpdateTableRewrite extends RewriteRowLevelCommand with PureAppendO
}
// Mirrors Spark 4.1.1 `RewriteUpdateTable.{buildReplaceDataPlan, buildReplaceDataWithUnionPlan,
- // buildReplaceDataUpdateProjection}`.
+ // buildReplaceDataUpdateProjection}`. Spark 4.2 (SPARK-56510) replaced
+ // WRITE_WITH_METADATA_OPERATION with COPY_OPERATION; updated and carried-over rows are both
+ // written through `write(metadata, data)`, which is exactly what COPY_OPERATION selects, so a
+ // single flat label keeps the written rows identical. Upstream 4.2 additionally labels the rows
+ // matching the condition UPDATE_OPERATION so its `numUpdatedRows` / `numCopiedRows` UI metrics
+ // split the two; those metrics are display-only here (Paimon does not implement
+ // `BatchWrite.commit(.., WriteSummary)`), so we keep the simpler shape.
private def buildReplaceDataPlan(
relation: DataSourceV2Relation,
operationTable: RowLevelOperationTable,
@@ -147,7 +153,7 @@ object Spark41UpdateTableRewrite extends RewriteRowLevelCommand with PureAppendO
val updatedAndRemainingRowsPlan =
buildReplaceDataUpdateProjection(readRelation, assignments, cond)
val writeRelation = relation.copy(table = operationTable)
- val query = addOperationColumn(WRITE_WITH_METADATA_OPERATION, updatedAndRemainingRowsPlan)
+ val query = addOperationColumn(COPY_OPERATION, updatedAndRemainingRowsPlan)
val projections = buildReplaceDataProjections(query, relation.output, metadataAttrs)
ReplaceData(writeRelation, cond, query, relation, projections, Some(cond))
}
@@ -169,7 +175,7 @@ object Spark41UpdateTableRewrite extends RewriteRowLevelCommand with PureAppendO
val updatedAndRemainingRowsPlan = Union(updatedRowsPlan, remainingRowsPlan)
val writeRelation = relation.copy(table = operationTable)
- val query = addOperationColumn(WRITE_WITH_METADATA_OPERATION, updatedAndRemainingRowsPlan)
+ val query = addOperationColumn(COPY_OPERATION, updatedAndRemainingRowsPlan)
val projections = buildReplaceDataProjections(query, relation.output, metadataAttrs)
ReplaceData(writeRelation, cond, query, relation, projections, Some(cond))
}
diff --git a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/CreatePaimonSQLFunctionCommand.scala b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/CreatePaimonSQLFunctionCommand.scala
index de4195ac3fcd..9c890b178737 100644
--- a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/CreatePaimonSQLFunctionCommand.scala
+++ b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/catalyst/parser/extensions/CreatePaimonSQLFunctionCommand.scala
@@ -68,8 +68,9 @@ case class CreatePaimonSQLFunctionCommand(
val sessionCatalog = sparkSession.sessionState.catalog
val conf = sparkSession.sessionState.conf
- val inputParam = inputParamText.map(UserDefinedFunction.parseRoutineParam(_, parser))
- val returnType = parseReturnTypeText(returnTypeText, isTableFunc, parser)
+ // Spark 4.2 added a trailing `collation` parameter with no default to both helpers.
+ val inputParam = inputParamText.map(UserDefinedFunction.parseRoutineParam(_, parser, None))
+ val returnType = parseReturnTypeText(returnTypeText, isTableFunc, parser, None)
val function = SQLFunction(
name,
@@ -78,6 +79,7 @@ case class CreatePaimonSQLFunctionCommand(
exprText,
queryText,
comment,
+ None, // collation, added in Spark 4.2
isDeterministic,
containsSQL,
isTableFunc,
diff --git a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/paimon/shims/Spark4Shim.scala b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/paimon/shims/Spark4Shim.scala
index eea503d7d5ce..dda8290012b1 100644
--- a/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/paimon/shims/Spark4Shim.scala
+++ b/paimon-spark/paimon-spark4-common/src/main/scala/org/apache/spark/sql/paimon/shims/Spark4Shim.scala
@@ -33,22 +33,26 @@ import org.apache.hadoop.conf.Configuration
import org.apache.hadoop.fs.Path
import org.apache.spark.sql.SparkSession
import org.apache.spark.sql.catalyst.InternalRow
+import org.apache.spark.sql.catalyst.analysis.{ResolvedPartitionSpec, ResolvedTable}
+import org.apache.spark.sql.catalyst.analysis.{UnresolvedIdentifier, UnresolvedTableOrView}
import org.apache.spark.sql.catalyst.analysis.CTESubstitution
-import org.apache.spark.sql.catalyst.expressions.{Attribute, AttributeReference, Expression}
+import org.apache.spark.sql.catalyst.analysis.NamedRelation
+import org.apache.spark.sql.catalyst.catalog.CatalogStorageFormat
+import org.apache.spark.sql.catalyst.expressions.{Attribute, AttributeReference, Expression, Literal}
import org.apache.spark.sql.catalyst.expressions.aggregate.AggregateExpression
import org.apache.spark.sql.catalyst.parser.ParserInterface
-import org.apache.spark.sql.catalyst.plans.logical.{Aggregate, Assignment, ColumnDefinition, CTERelationRef, InsertAction, LogicalPlan, MergeAction, MergeIntoTable, MergeRows, SubqueryAlias, TableSpec, UnresolvedWith, UpdateAction}
+import org.apache.spark.sql.catalyst.plans.logical.{Aggregate, Assignment, ColumnDefinition, CreateTableLike, CTERelationRef, DescribeRelation, DescribeTablePartition, InsertAction, LogicalPlan, MergeAction, MergeIntoTable, MergeRows, OverwriteByExpression, OverwritePartitionsDynamic, SubqueryAlias, TableSpec, UnresolvedWith, UpdateAction}
import org.apache.spark.sql.catalyst.plans.logical.MergeRows.{Copy, Insert, Keep, Update}
import org.apache.spark.sql.catalyst.plans.physical.{ClusteredDistribution, Distribution}
import org.apache.spark.sql.catalyst.rules.Rule
import org.apache.spark.sql.catalyst.util.{ArrayData, GeneratedColumn, IdentityColumn, ResolveDefaultColumns, STUtils}
-import org.apache.spark.sql.connector.catalog.{CatalogV2Util, Column, Identifier, StagingTableCatalog, Table, TableCatalog}
+import org.apache.spark.sql.connector.catalog.{CatalogV2Util, Column, Identifier, StagingTableCatalog, SupportsPartitionManagement, Table, TableCatalog}
import org.apache.spark.sql.connector.expressions.Transform
import org.apache.spark.sql.connector.read.Scan
import org.apache.spark.sql.connector.write.BatchWrite
import org.apache.spark.sql.execution.{SparkFormatTable, SparkPlan}
import org.apache.spark.sql.execution.datasources.{PartitioningAwareFileIndex, PartitionSpec}
-import org.apache.spark.sql.execution.datasources.v2.{AtomicReplaceTableAsSelectExec, AtomicReplaceTableExec, ReplaceTableAsSelectExec, ReplaceTableExec}
+import org.apache.spark.sql.execution.datasources.v2.{AtomicReplaceTableAsSelectExec, AtomicReplaceTableExec, CreateTableAsSelectExec, DescribeTableExec, ReplaceTableAsSelectExec, ReplaceTableExec}
import org.apache.spark.sql.execution.datasources.v2.{DataSourceV2Relation, DataSourceV2ScanRelation}
import org.apache.spark.sql.execution.streaming.runtime.MetadataLogFileIndex
import org.apache.spark.sql.execution.streaming.sinks.FileStreamSink
@@ -56,6 +60,7 @@ import org.apache.spark.sql.internal.SQLConf
import org.apache.spark.sql.types.{DataTypes, Geography, GeographyType, Geometry, GeometryType, StructType, VariantType}
import org.apache.spark.unsafe.types.VariantVal
+import java.net.URI
import java.util.{Map => JMap}
class Spark4Shim extends SparkShim {
@@ -95,6 +100,47 @@ class Spark4Shim extends SparkShim {
tableCatalog.createTable(ident, columns, partitions, properties)
}
+ override def withStorageLocation(
+ storage: CatalogStorageFormat,
+ locationUri: Option[URI]): CatalogStorageFormat =
+ storage.copy(locationUri = locationUri)
+
+ override def overwriteByName(
+ table: NamedRelation,
+ query: LogicalPlan,
+ deleteExpr: Expression,
+ writeOptions: Map[String, String]): OverwriteByExpression =
+ OverwriteByExpression.byName(
+ table,
+ query,
+ deleteExpr,
+ writeOptions,
+ withSchemaEvolution = false)
+
+ override def overwritePartitionsDynamicByName(
+ table: NamedRelation,
+ query: LogicalPlan,
+ writeOptions: Map[String, String]): OverwritePartitionsDynamic =
+ OverwritePartitionsDynamic.byName(table, query, writeOptions, withSchemaEvolution = false)
+
+ override def createCreateTableAsSelectExec(
+ catalog: TableCatalog,
+ ident: Identifier,
+ partitioning: Seq[Transform],
+ query: LogicalPlan,
+ tableSpec: TableSpec,
+ writeOptions: Map[String, String],
+ ifNotExists: Boolean): SparkPlan = {
+ CreateTableAsSelectExec(
+ catalog,
+ ident,
+ partitioning,
+ query,
+ tableSpec,
+ writeOptions,
+ ifNotExists)
+ }
+
override def createReplaceTableAsSelectExec(
catalog: TableCatalog,
ident: Identifier,
@@ -362,20 +408,24 @@ class Spark4Shim extends SparkShim {
override def toPaimonGeometry(o: Object): Array[Byte] =
o.asInstanceOf[Geometry].getBytes
+ // Spark 4.2 (SPARK-57058) folded the geo value classes into `BinaryView`: `SpecializedGetters`
+ // lost `getGeometry` / `getGeography` in favour of `getBinaryView`, and `STUtils.stAsBinary` was
+ // split into `stGeomAsBinary` / `stGeogAsBinary`. `paimon-spark-4.1` forks this file to keep the
+ // pre-4.2 calls.
override def toPaimonGeometry(row: InternalRow, pos: Int): Array[Byte] =
- STUtils.stAsBinary(row.getGeometry(pos))
+ STUtils.stGeomAsBinary(row.getBinaryView(pos))
override def toPaimonGeometry(array: ArrayData, pos: Int): Array[Byte] =
- STUtils.stAsBinary(array.getGeometry(pos))
+ STUtils.stGeomAsBinary(array.getBinaryView(pos))
override def toPaimonGeography(o: Object): Array[Byte] =
o.asInstanceOf[Geography].getBytes
override def toPaimonGeography(row: InternalRow, pos: Int): Array[Byte] =
- STUtils.stAsBinary(row.getGeography(pos))
+ STUtils.stGeogAsBinary(row.getBinaryView(pos))
override def toPaimonGeography(array: ArrayData, pos: Int): Array[Byte] =
- STUtils.stAsBinary(array.getGeography(pos))
+ STUtils.stGeogAsBinary(array.getBinaryView(pos))
override def toSparkGeometry(wkb: Array[Byte], crs: String): Object = {
val geometryType = sparkGeometryType(crs)
@@ -384,7 +434,8 @@ class Spark4Shim extends SparkShim {
override def toSparkGeography(wkb: Array[Byte], crs: String, algorithm: String): Object = {
val geographyType = sparkGeographyType(crs, algorithm)
- STUtils.stSetSrid(STUtils.stGeogFromWKB(wkb), geographyType.srid)
+ // 4.2 renamed the single `stSetSrid` overload pair to `stGeogSetSrid` / `stGeomSetSrid`.
+ STUtils.stGeogSetSrid(STUtils.stGeogFromWKB(wkb), geographyType.srid)
}
override def isSparkGeometryType(dataType: org.apache.spark.sql.types.DataType): Boolean =
@@ -438,6 +489,84 @@ class Spark4Shim extends SparkShim {
parser: org.apache.spark.sql.catalyst.parser.ParserInterface): Expression =
org.apache.paimon.spark.catalog.functions.SQLFunctionConverter
.toSQLFunctionExpression(funcIdent, function, arguments, parser)
+
+ // Spark 4.2 (SPARK-39660) removed partitionSpec from DescribeRelation; DESCRIBE ... PARTITION is
+ // a separate DescribeTablePartition plan there.
+ override def createTableLikeParts(plan: LogicalPlan)
+ : Option[(Seq[String], Seq[String], Option[String], Option[String], Map[String, String], Boolean, Boolean)] =
+ plan match {
+ case c: CreateTableLike =>
+ // Parser-stage rule: the children have not been analyzed yet.
+ (c.name, c.source) match {
+ case (target: UnresolvedIdentifier, source: UnresolvedTableOrView) =>
+ Some(
+ (
+ target.nameParts,
+ source.multipartIdentifier,
+ c.provider,
+ c.location,
+ c.properties,
+ c.ifNotExists,
+ // `STORED AS` lands in `serdeInfo`; Paimon tables cannot honour Hive storage syntax.
+ c.serdeInfo.isDefined))
+ case _ => None
+ }
+ case _ => None
+ }
+
+ override def describeTablePartition(
+ plan: LogicalPlan): Option[(LogicalPlan, Map[String, String], Boolean, Seq[Attribute])] =
+ plan match {
+ case d: DescribeTablePartition =>
+ (d.table, d.partitionSpec) match {
+ case (
+ r @ ResolvedTable(_, _, table: SupportsPartitionManagement, _),
+ spec: ResolvedPartitionSpec) =>
+ // `ResolvedPartitionSpec` holds the values as an `InternalRow`, so read each field by
+ // its declared type and render it. Read the types from `partitionSchema()`, the same
+ // schema `ResolvePartitionSpec` used to build `names` and `ident`, so a name can never
+ // be missing. (Char/varchar is the one place the declared type and the stored value
+ // differ: `convertToPartIdent` casts through `replaceCharVarcharWithString`, leaving a
+ // plain `UTF8String` under a `CharType(n)` field. Harmless here — `InternalRow.get`
+ // ignores the type argument for a `GenericInternalRow`, `Literal`'s validation
+ // dispatches on the physical type, where `CharType` maps to `PhysicalStringType` and
+ // accepts a `UTF8String`, and `Literal.toString` has no char/varchar/string branch at
+ // all, so the value falls through to `other.toString`.)
+ //
+ // The rendering is `Literal.toString`, NOT what upstream's own
+ // `DescribeTablePartitionExec` uses (`ToPrettyString(...)` + `escapePathName`), because
+ // the result is compared for equality against Paimon's `Partition.spec()` rather than
+ // displayed. Note this rendering and `Partition.spec()`'s own do not agree for every
+ // type: with `partition.legacy-name`
+ // (the default) Paimon stores `field.toString()`, so a DATE column holds the epoch day
+ // while this renders `2021-01-01`. That mismatch predates Spark 4.2 — on <= 4.1 the
+ // parser produced the same `2021-01-01` via `Cast(literal, StringType)`.
+ val partSchema = table.partitionSchema()
+ val values = spec.names.zipWithIndex.map {
+ case (name, i) =>
+ val field = partSchema(name)
+ val value = spec.ident.get(i, field.dataType)
+ name -> Literal(value, field.dataType).toString
+ }
+ Some((r, values.toMap, d.isExtended, d.output))
+ case _ => None
+ }
+ case _ => None
+ }
+
+ override def describeRelationPartitionSpec(plan: DescribeRelation): Map[String, String] =
+ Map.empty
+
+ override def createDescribeTableExec(
+ output: Seq[Attribute],
+ catalogName: String,
+ identifier: Identifier,
+ table: Table,
+ isExtended: Boolean): SparkPlan =
+ DescribeTableExec(output, catalogName, identifier, table, isExtended)
+
+ override def mergeNeedsSchemaEvolution(merge: MergeIntoTable): Boolean =
+ merge.pendingSchemaChanges.nonEmpty
}
object Spark4Shim {
diff --git a/pom.xml b/pom.xml
index c84036cbddf2..b61996e5efb6 100644
--- a/pom.xml
+++ b/pom.xml
@@ -429,21 +429,24 @@ under the License.
spark4paimon-spark/paimon-spark4-common
+ paimon-spark/paimon-spark-ut-4.0
+ paimon-spark/paimon-spark-ut-4.1paimon-spark/paimon-spark-4.0paimon-spark/paimon-spark-4.1
+ paimon-spark/paimon-spark-4.2174.13.12.13
-
- 2.13.17
- 4.1.2
+ 2.13.18
+ 4.2.0paimon-spark4-common_2.1318.1.0
- 4.1
- 4.1.2
+ 4.2
+ 4.2.0