Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
60 changes: 60 additions & 0 deletions common/utils/src/main/resources/error/error-conditions.json
Original file line number Diff line number Diff line change
Expand Up @@ -1248,6 +1248,60 @@
],
"sqlState" : "22003"
},
"COLUMN_UPDATE_DUPLICATE_REQUIRED_DATA_ATTRIBUTE" : {
"message" : [
"Connector <connector> mixes in `SupportsColumnUpdates` but declared duplicate column(s) <duplicateAttributes> in `requiredDataAttributes()`. Each column must be declared at most once."
],
"sqlState" : "42000"
},
"COLUMN_UPDATE_EMPTY_REQUIRED_DATA_ATTRIBUTES" : {
"message" : [
"Connector <connector> mixes in `SupportsColumnUpdates` but returned an empty array from `requiredDataAttributes()`. Connectors that opt into column-level updates must declare at least one required data attribute."
],
"sqlState" : "42000"
},
"COLUMN_UPDATE_METADATA_REQUIRED_DATA_ATTRIBUTE" : {
"message" : [
"Connector <connector> mixes in `SupportsColumnUpdates` but declared metadata column(s) <metadataAttributes> in `requiredDataAttributes()`. Declare metadata columns in `requiredMetadataAttributes()` instead."
],
"sqlState" : "42000"
},
"COLUMN_UPDATE_NESTED_REQUIRED_DATA_ATTRIBUTE" : {
"message" : [
"Connector <connector> mixes in `SupportsColumnUpdates` but declared nested field(s) <nestedAttributes> in `requiredDataAttributes()`. Declare the root struct column instead of a nested field."
],
"sqlState" : "42000"
},
"COLUMN_UPDATE_REQUIRED_DATA_ATTRIBUTES_MISSING_UPDATED_COLUMNS" : {
"message" : [
"Connector <connector> mixes in `SupportsColumnUpdates` but its `requiredDataAttributes()` does not cover every column being updated. Missing columns: <missingColumns>. Connectors must include every column reported by `RowLevelOperationInfo.updatedColumns()` in `requiredDataAttributes()`."
],
"sqlState" : "42000"
},
"COLUMN_UPDATE_SPLIT_ROW_ID_NOT_DECLARED" : {
"message" : [
"Connector <connector> mixes in `SupportsColumnUpdates` and represents UPDATE as delete and insert, but row ID column(s) <rowIds> do not reach the reinserted row, which then has no identity for the connector to place it by. Declare each data row ID column in `requiredDataAttributes()`, and return each metadata row ID column from `requiredMetadataAttributes()` with `MetadataColumn.PRESERVE_ON_REINSERT` set."
],
"sqlState" : "42000"
},
"COLUMN_UPDATE_SPLIT_ROW_ID_REASSIGNMENT" : {
"message" : [
"Connector <connector> mixes in `SupportsColumnUpdates` and represents UPDATE as delete and insert, so UPDATE cannot assign row ID column(s) <rowIds>. The reinserted row carries only the columns in `requiredDataAttributes()`, and with a new row ID it cannot be matched to the row whose other columns the connector must preserve. Do not assign row ID columns, or include every table column in `requiredDataAttributes()`."
],
"sqlState" : "42000"
},
"COLUMN_UPDATE_UNDECLARED_WRITE_REQUIREMENT_COLUMNS" : {
"message" : [
"Connector <connector> mixes in `SupportsColumnUpdates`, but its write requires a distribution or ordering by column(s) <columns> that the column-level UPDATE does not read. Declare each data column in `requiredDataAttributes()` and each metadata column in `requiredMetadataAttributes()`."
],
"sqlState" : "42000"
},
"COLUMN_UPDATE_UNKNOWN_REQUIRED_DATA_ATTRIBUTE" : {
"message" : [
"Connector <connector> mixes in `SupportsColumnUpdates` but declared column(s) <unknownAttributes> in `requiredDataAttributes()` that do not exist in the table."
],
"sqlState" : "42703"
},
"COMPARATOR_RETURNS_NULL" : {
"message" : [
"The comparator has returned a NULL for a comparison between <firstValue> and <secondValue>.",
Expand Down Expand Up @@ -2397,6 +2451,12 @@
],
"sqlState" : "42K03"
},
"DATA_SOURCE_WRITE_COLUMN_UPDATE_NOT_IMPLEMENTED" : {
"message" : [
"<class> does not override `writeColumnUpdate(record)`. A data writer that receives rows in the `LogicalWriteInfo.columnUpdateSchema()` layout must override `writeColumnUpdate(record)`, or `writeColumnUpdate(metadata, record)` if the operation returns metadata columns from `requiredMetadataAttributes()`."
],
"sqlState" : "0A000"
},
"DATETIME_FIELD_OUT_OF_BOUNDS" : {
"message" : [
"<rangeMessage>."
Expand Down
5 changes: 4 additions & 1 deletion project/MimaExcludes.scala
Original file line number Diff line number Diff line change
Expand Up @@ -46,7 +46,10 @@ object MimaExcludes {
"org.apache.spark.ml.regression.DecisionTreeRegressionModel.numLeave"),
// [SPARK-59154] Remove unused prediction variance helper after inlining its implementation.
ProblemFilters.exclude[DirectMissingMethodProblem](
"org.apache.spark.ml.regression.DecisionTreeRegressionModel.predictVariance")
"org.apache.spark.ml.regression.DecisionTreeRegressionModel.predictVariance"),
// [SPARK-58111] Write schema narrowing for column-level UPDATE in DSv2
ProblemFilters.exclude[ReversedMissingMethodProblem](
"org.apache.spark.sql.connector.write.RowLevelOperationInfo.updatedColumns")
)

// Exclude rules for 4.3.x from 4.2.0 (add 4.3-specific filters below as needed).
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,9 @@
import java.io.Closeable;
import java.io.IOException;
import java.util.Iterator;
import java.util.Map;

import org.apache.spark.SparkUnsupportedOperationException;
import org.apache.spark.annotation.Evolving;
import org.apache.spark.sql.connector.metric.CustomTaskMetric;

Expand Down Expand Up @@ -82,6 +84,60 @@ default void write(T metadata, T record) throws IOException {
write(record);
}

/**
* Writes one updated or copied record with metadata in the column update layout.
* <p>
* When {@link LogicalWriteInfo#columnUpdateSchema()} is present for a row-level operation that
* does not mix in {@link SupportsDelta}, Spark passes updated and copied records to this method
* instead of {@link #write(Object, Object)} if the operation returns a non-empty
* {@link RowLevelOperation#requiredMetadataAttributes()}, and to
* {@link #writeColumnUpdate(Object)} otherwise. The record follows
* {@link LogicalWriteInfo#columnUpdateSchema()} and the metadata follows
* {@link LogicalWriteInfo#metadataSchema()}. Operations that mix in {@link SupportsDelta} receive
* such rows through {@link DeltaWriter#update} and {@link DeltaWriter#reinsert} instead.
* <p>
* By default, delegates to {@link #writeColumnUpdate(Object)} and drops the metadata.
* <p>
* If this method fails (by throwing an exception), {@link #abort()} will be called and this
* data writer is considered to have been failed.
*
* @throws IOException if failure happens during disk/network IO like writing files.
* @throws SparkUnsupportedOperationException if neither this method nor
* {@link #writeColumnUpdate(Object)} is overridden.
*
* @since 4.4.0
*/
default void writeColumnUpdate(T metadata, T record) throws IOException {
writeColumnUpdate(record);
}

/**
* Writes one updated or copied record without metadata in the column update layout.
* <p>
* When {@link LogicalWriteInfo#columnUpdateSchema()} is present for a row-level operation that
* does not mix in {@link SupportsDelta}, Spark passes updated and copied records to this method
* instead of {@link #write(Object)} if the operation returns no
* {@link RowLevelOperation#requiredMetadataAttributes()}. The record follows
* {@link LogicalWriteInfo#columnUpdateSchema()}.
* <p>
* A writer for such an operation must override this method, unless the operation returns a
* non-empty {@link RowLevelOperation#requiredMetadataAttributes()} and the writer overrides
* {@link #writeColumnUpdate(Object, Object)}.
* <p>
* If this method fails (by throwing an exception), {@link #abort()} will be called and this
* data writer is considered to have been failed.
*
* @throws IOException if failure happens during disk/network IO like writing files.
* @throws SparkUnsupportedOperationException if this method is not overridden.
*
* @since 4.4.0
*/
default void writeColumnUpdate(T record) throws IOException {
throw new SparkUnsupportedOperationException(
"DATA_SOURCE_WRITE_COLUMN_UPDATE_NOT_IMPLEMENTED",
Map.of("class", getClass().getName()));
}

/**
* Writes one record.
* <p>
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,9 @@ public interface DeltaWriter<T> extends DataWriter<T> {

/**
* Updates a row.
* <p>
* When {@link LogicalWriteInfo#columnUpdateSchema()} is present, the {@code row} follows it;
* otherwise it follows {@link LogicalWriteInfo#schema()}.
*
* @param metadata values for metadata columns that were projected but are not part of the row ID
* @param id a row ID to update
Expand All @@ -52,6 +55,9 @@ public interface DeltaWriter<T> extends DataWriter<T> {
* Reinserts a row with metadata.
* <p>
* This method handles the insert portion of updated rows split into deletes and inserts.
* <p>
* When {@link LogicalWriteInfo#columnUpdateSchema()} is present, the {@code row} follows it;
* otherwise it follows {@link LogicalWriteInfo#schema()}.
*
* @param metadata values for metadata columns
* @param row a row to reinsert
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,10 @@ public interface LogicalWriteInfo {

/**
* the schema of the input data from Spark to data source.
* <p>
* When {@link #columnUpdateSchema()} is present, this schema covers only newly inserted rows, as
* updated, copied, and reinserted rows follow {@link #columnUpdateSchema()}. It is then empty
* when the command inserts no new rows, such as UPDATE.
*/
StructType schema();

Expand All @@ -65,4 +69,18 @@ default Optional<StructType> metadataSchema() {
throw new SparkUnsupportedOperationException(
"DATA_SOURCE_METADATA_SCHEMA_NOT_IMPLEMENTED", Map.of("class", getClass().getName()));
}

/**
* the schema of updated, copied, and reinserted rows from Spark to data source in a
* column-level update. Present when the operation mixes in {@link SupportsColumnUpdates} and
* Spark delivers these rows with the columns of
* {@link SupportsColumnUpdates#requiredDataAttributes()}, which currently happens only for
* UPDATE. It covers every table column if every column is declared. When present,
* {@link #schema()} covers only newly inserted rows.
*
* @since 4.4.0
*/
default Optional<StructType> columnUpdateSchema() {
return Optional.empty();
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@
package org.apache.spark.sql.connector.write;

import org.apache.spark.annotation.Experimental;
import org.apache.spark.sql.connector.expressions.NamedReference;
import org.apache.spark.sql.connector.write.RowLevelOperation.Command;
import org.apache.spark.sql.util.CaseInsensitiveStringMap;

Expand All @@ -37,4 +38,16 @@ public interface RowLevelOperationInfo {
* Returns the row-level SQL command (e.g. DELETE, UPDATE, MERGE).
*/
Command command();

/**
* Returns the columns being updated by this operation. Currently only UPDATE populates it;
* other commands report an empty array.
* <p>
* A column is reported only if it is assigned a new value, so identity assignments such as
* {@code SET a = a} are excluded. Nested struct field updates are reported at root-column
* granularity (e.g. {@code SET s.c1 = -1} returns {@code s}).
*
* @since 4.4.0
*/
NamedReference[] updatedColumns();
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,94 @@
/*
* Licensed to the Apache Software Foundation (ASF) under one or more
* contributor license agreements. See the NOTICE file distributed with
* this work for additional information regarding copyright ownership.
* The ASF licenses this file to You under the Apache License, Version 2.0
* (the "License"); you may not use this file except in compliance with
* the License. You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/

package org.apache.spark.sql.connector.write;

import org.apache.spark.annotation.Experimental;
import org.apache.spark.sql.connector.expressions.NamedReference;

/**
* A mix-in interface for {@link RowLevelOperation}. Data sources can implement this interface to
* receive a narrow row containing only the columns declared via {@link #requiredDataAttributes()}
* for updated, copied, and reinserted records, instead of the full table row.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Finding 11. The interface doc reads as if the mix-in applies to any RowLevelOperation, but only RewriteUpdateTable honours it: RewriteDeleteFromTable and RewriteMergeIntoTable call the three-argument buildReplaceDataProjections / four-argument buildWriteDeltaProjections, so updateRowProjection stays None, updateSchema() stays absent, and requiredDataAttributes() is never read. I confirmed a MERGE against a SupportsColumnUpdates connector runs correctly with full-width rows -- so this is not a correctness problem, but a connector reading the doc would reasonably expect narrow rows there.

Finding 2 fixed the equivalent note on updatedColumns(); this is the same point on the mix-in itself, which is what @dongjoon-hyun originally asked for ("shall we mention that DELETE and MERGE ignores this method?" on the old RowLevelOperation.java).

Suggested change
* for updated, copied, and reinserted records, instead of the full table row.
* for updated, copied, and reinserted records, instead of the full table row.
* <p>
* Currently honored only for UPDATE. DELETE and MERGE ignore this interface: those operations
* receive full-width rows and {@link LogicalWriteInfo#updateSchema()} is absent for them.

Worth a cross-reference from DeltaWriter#update and DeltaWriter#reinsert too -- their row parameter is documented as "a row with updated values" with no hint that it may follow updateSchema() rather than schema().

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Added the lines on all mentioned API docs.

* <p>
* When {@link LogicalWriteInfo#columnUpdateSchema()} is present, updated, copied, and reinserted
* rows follow it: group-based operations receive them through
* {@link DataWriter#writeColumnUpdate(Object, Object)} or
* {@link DataWriter#writeColumnUpdate(Object)}, and operations that mix in {@link SupportsDelta}
* through {@link DeltaWriter#update} and {@link DeltaWriter#reinsert}. Inserted rows always follow
* {@link LogicalWriteInfo#schema()} and arrive through {@link DataWriter#write(Object)} and
* {@link DeltaWriter#insert}. When {@link LogicalWriteInfo#columnUpdateSchema()} is absent, every
* row follows {@link LogicalWriteInfo#schema()}, as for any other operation. Connectors must
* decide which layout to expect by whether {@link LogicalWriteInfo#columnUpdateSchema()} is
* present, not by the command. A builder that wants full-width rows for some command can check
* {@link RowLevelOperationInfo#command()} and build an operation without this mix-in. Currently
* Spark narrows only UPDATE.
* <p>
* The scan builder returned by {@link #newScanBuilder} should implement
* {@link org.apache.spark.sql.connector.read.SupportsPushDownRequiredColumns}, so that the scan
* does not read columns the command neither references nor declares. Otherwise the scan reads
* every column.
*
* @since 4.4.0
*/
@Experimental
public interface SupportsColumnUpdates extends RowLevelOperation {
/**
* Returns the data column references required to perform this row-level operation.
* <p>
* When Spark narrows the write, the returned columns become
* {@link LogicalWriteInfo#columnUpdateSchema()}, in declared order. Implementations must include
* every column they want to receive: every column reported by
* {@link RowLevelOperationInfo#updatedColumns()}, plus any columns needed for row lookup or
* routing, e.g. a primary key. Columns that are not declared are absent from the rows the
* connector receives, so the connector must preserve their values itself.
* <p>
* Each entry must name a top-level data column of the table. For updates on nested fields such
* as {@code SET s.c1 = -1}, the connector must declare the root struct column {@code s}. Spark
* rejects with an analysis exception an empty array, a column that does not exist, a column
* declared more than once, a nested field, a metadata column (declare those through
* {@link #requiredMetadataAttributes()} instead), and an array that misses a column reported by
* {@link RowLevelOperationInfo#updatedColumns()}.
* <p>
* If this operation also mixes in {@link SupportsDelta} and represents updates as deletes and
* inserts ({@link SupportsDelta#representUpdateAsDeleteAndInsert()} returns {@code true}),
* every row-ID column ({@link SupportsDelta#rowId()}) must reach the reinserted row: a data
* row-ID column must be declared here, and a metadata row-ID column must be returned by
* {@link #requiredMetadataAttributes()} with
* {@link org.apache.spark.sql.connector.catalog.MetadataColumn#PRESERVE_ON_REINSERT} set. In
* this mode, an UPDATE that assigns a new value to a row-ID column is also rejected, unless
* every table column is declared here. Both cases are rejected with an analysis exception.
* <p>
* Data columns the write needs must be declared here too, even if the command does not
* reference them, such as the source columns of partition transforms and any columns used by
* {@link RequiresDistributionAndOrdering#requiredDistribution()} or
* {@link RequiresDistributionAndOrdering#requiredOrdering()}. Such columns appear in the rows
* the connector receives; a connector that does not want to persist them should project them
* away inside the connector before writing. A write whose distribution or ordering references
* any other column is rejected with an analysis exception, except for metadata columns returned
* by {@link #requiredMetadataAttributes()} and, for operations that mix in
* {@link SupportsDelta}, row-ID columns.
* <p>
* Spark does not read an undeclared column that neither the command nor a table constraint
* references, so such a column is not available for runtime filtering. A scan that reports
* runtime filter attributes based on the columns it reads loses runtime group filtering on it,
* and a scan that still reports it as a runtime filter attribute fails with
* {@code DATA_SOURCE_INVALID_RUNTIME_FILTER_ATTRIBUTE}. Declare partition source columns used
* for runtime filtering to keep it.
*/
NamedReference[] requiredDataAttributes();
}
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,14 @@ case class ProjectingInternalRow(schema: StructType,
this.row = row
}

/**
* Returns a projection with the same schema whose field `i` reads input ordinal
* `ordinalMap(colOrdinals(i))` instead of `colOrdinals(i)`.
*/
def remapOrdinals(ordinalMap: Int => Int): ProjectingInternalRow = {
ProjectingInternalRow(schema, colOrdinals.map(ordinalMap))
}

override def setNullAt(i: Int): Unit = throw SparkUnsupportedOperationException()

override def update(i: Int, value: Any): Unit = throw SparkUnsupportedOperationException()
Expand Down
Loading
Loading