diff --git a/common/utils/src/main/resources/error/error-conditions.json b/common/utils/src/main/resources/error/error-conditions.json index 0acf778aa6719..1276c28bdfe10 100644 --- a/common/utils/src/main/resources/error/error-conditions.json +++ b/common/utils/src/main/resources/error/error-conditions.json @@ -9605,6 +9605,13 @@ ], "sqlState" : "0A000" }, + "UNSUPPORTED_JSON_CHAR_VARCHAR_MAP_KEY" : { + "message" : [ + "The JSON object name is not a valid key for the map key type .", + "A CHAR key must match the declared length exactly and a VARCHAR key must not exceed it; keys are never padded or trimmed. Use a STRING map key to accept any name." + ], + "sqlState" : "0A000" + }, "UNSUPPORTED_MERGE_CONDITION" : { "message" : [ "MERGE operation contains unsupported condition." diff --git a/docs/sql-migration-guide.md b/docs/sql-migration-guide.md index f9c7bf4771227..e1dce2b8abcce 100644 --- a/docs/sql-migration-guide.md +++ b/docs/sql-migration-guide.md @@ -28,6 +28,7 @@ license: | ## Upgrading from Spark SQL 4.3 to 4.4 +- Since Spark 4.4, when `spark.sql.charVarchar.standardSemantics.enabled` is true, `from_json` length-checks JSON object names used as `MAP` or `MAP` keys without padding or trimming them. A `CHAR(n)` key must already be exactly `n` characters, and a `VARCHAR(n)` key must already be at most `n` characters; duplicate names are kept, exactly as for STRING keys, and `spark.sql.mapKeyDedupPolicy` is not applied. A key that fails the check makes the enclosing map a bad record: in `PERMISSIVE` mode the map is set to `null` while sibling fields are preserved (the whole `from_json` result is `null` only when the map is the top-level type), and in `FAILFAST` mode parsing fails with `UNSUPPORTED_JSON_CHAR_VARCHAR_MAP_KEY` (SQLSTATE `0A000`) as the cause of `MALFORMED_RECORD_IN_PARSING`. The same parser check runs when reading JSON files with a user-specified CHAR/VARCHAR reader schema. Reads whose CHAR/VARCHAR keys are handled read-side instead (for example catalog tables) are unchanged: there keys are padded, `spark.sql.mapKeyDedupPolicy` is applied, and an over-long key raises `EXCEED_LIMIT_LENGTH`. - Since Spark 4.4, when `spark.sql.preserveCharVarcharTypeInfo` is true and `spark.sql.charVarchar.standardSemantics.enabled` is false, ORC reads that apply a CHAR/VARCHAR schema over STRING storage return the stored values without ORC truncation, matching Parquet. Previously the ORC reader requested `char(n)`/`varchar(n)` and truncated STRING-stored values to `n`. Read-side length checks (`EXCEED_LIMIT_LENGTH`) apply only when `spark.sql.charVarchar.standardSemantics.enabled` is true. - Since Spark 4.4, the options maps passed to `from_csv`, `to_csv`, `schema_of_csv`, `from_json`, `to_json`, `schema_of_json`, `from_xml`, `to_xml`, and `schema_of_xml` must be foldable after replacing `RuntimeReplaceable` expressions. Previously, Spark evaluated non-foldable options during analysis, which allowed some constant expressions but could fail with an internal error or incorrectly evaluate row-dependent expressions. To allow deterministic and row-independent non-foldable options, set `spark.sql.legacy.allowNonFoldableOptions` to `true`. Row-dependent, unevaluable, and nondeterministic options are always rejected. - Since Spark 4.4, when an already-analyzed Data Source V2 query is refreshed after a compatible schema change, connectors can return more data columns from the current table schema in `Scan.readSchema()` than requested by `SupportsPushDownRequiredColumns.pruneColumns`. Previously, this partial pruning could fail planning because the scan reported columns absent from the analyzed relation output. diff --git a/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/expressions/json/JsonExpressionEvalUtils.scala b/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/expressions/json/JsonExpressionEvalUtils.scala index 8174a8911c274..c02bd3ac874fe 100644 --- a/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/expressions/json/JsonExpressionEvalUtils.scala +++ b/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/expressions/json/JsonExpressionEvalUtils.scala @@ -119,7 +119,8 @@ case class JsonToStructsEvaluator( nullableSchema: DataType, nameOfCorruptRecord: String, timeZoneId: Option[String], - variantAllowDuplicateKeys: Boolean) { + variantAllowDuplicateKeys: Boolean, + charVarcharStandardSemantics: Boolean) { // This converts parsed rows to the desired output by the given schema. @transient @@ -149,7 +150,8 @@ case class JsonToStructsEvaluator( (StructType(Array(StructField("value", other))), other) } - val rawParser = new JacksonParser(actualSchema, parsedOptions, allowArrayAsStructs = false) + val rawParser = new JacksonParser(actualSchema, parsedOptions, allowArrayAsStructs = false, + charVarcharStandardSemanticsOverride = Some(charVarcharStandardSemantics)) val createParser = CreateJacksonParser.utf8String _ new FailureSafeParser[UTF8String]( diff --git a/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/expressions/jsonExpressions.scala b/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/expressions/jsonExpressions.scala index 91ddcec4473b8..324440d3fb02b 100644 --- a/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/expressions/jsonExpressions.scala +++ b/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/expressions/jsonExpressions.scala @@ -1873,7 +1873,14 @@ case class JsonToStructs( options: Map[String, String], child: Expression, timeZoneId: Option[String] = None, - variantAllowDuplicateKeys: Boolean = SQLConf.get.getConf(SQLConf.VARIANT_ALLOW_DUPLICATE_KEYS)) + variantAllowDuplicateKeys: Boolean = SQLConf.get.getConf(SQLConf.VARIANT_ALLOW_DUPLICATE_KEYS), + // `spark.sql.charVarchar.standardSemantics.enabled` has PERSISTED binding, so it is captured + // here at analysis time (as the default arg, like `variantAllowDuplicateKeys`) and then + // threaded all the way into the parser via `charVarcharStandardSemanticsOverride`. This keeps + // a view's CHAR/VARCHAR map-key semantics tied to its creation-time flag rather than the + // caller's session setting. (`variantAllowDuplicateKeys` is only captured, not threaded: the + // parser still reads that one live from `SQLConf.get`.) + charVarcharStandardSemantics: Boolean = SQLConf.get.charVarcharStandardSemantics) extends UnaryExpression with TimeZoneAwareExpression with CodegenFallback @@ -1935,7 +1942,8 @@ case class JsonToStructs( @transient private lazy val evaluator = new JsonToStructsEvaluator( - options, nullableSchema, nameOfCorruptRecord, timeZoneId, variantAllowDuplicateKeys) + options, nullableSchema, nameOfCorruptRecord, timeZoneId, variantAllowDuplicateKeys, + charVarcharStandardSemantics) override def stateful: Boolean = true override def nullSafeEval(json: Any): Any = { diff --git a/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/json/JacksonParser.scala b/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/json/JacksonParser.scala index e87b1d0ca2a8e..c6f4f74a7975b 100644 --- a/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/json/JacksonParser.scala +++ b/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/json/JacksonParser.scala @@ -51,7 +51,13 @@ class JacksonParser( schema: DataType, val options: JSONOptions, allowArrayAsStructs: Boolean, - filters: Seq[Filter] = Seq.empty) extends Logging { + filters: Seq[Filter] = Seq.empty, + // `spark.sql.charVarchar.standardSemantics.enabled` has PERSISTED binding, so for `from_json` + // it must be captured when the expression is analyzed (see `JsonToStructs`) rather than read + // live here, otherwise a view created under the flag would skip the CHAR/VARCHAR map-key check + // when queried from a session with the flag off. `None` falls back to the live conf, which is + // correct for the file-based JSON data source where there is no view to capture. + charVarcharStandardSemanticsOverride: Option[Boolean] = None) extends Logging { import JacksonUtils._ import com.fasterxml.jackson.core.JsonToken._ @@ -60,6 +66,13 @@ class JacksonParser( // to a value in a field for `InternalRow`. private type ValueConverter = JsonParser => AnyRef + // CHAR/VARCHAR JSON map keys are length-checked without pad or trim under this flag. Lazy so + // its value is independent of field declaration order: `makeMapKeyChecker` reads it while the + // converters below are built, and an eager val read before its own initializer would silently + // see `false` and disable the check. + private lazy val charVarcharStandardSemantics = + charVarcharStandardSemanticsOverride.getOrElse(SQLConf.get.charVarcharStandardSemantics) + // `ValueConverter`s for the root schema for all fields in the schema private val rootConverter = makeRootConverter(schema) @@ -191,8 +204,9 @@ class JacksonParser( private def makeMapRootConverter(mt: MapType): JsonParser => Iterable[InternalRow] = { val fieldConverter = makeConverter(mt.valueType) + val keyChecker = makeMapKeyChecker(mt.keyType) (parser: JsonParser) => parseJsonToken[Iterable[InternalRow]](parser, mt) { - case START_OBJECT => Some(InternalRow(convertMap(parser, fieldConverter))) + case START_OBJECT => Some(InternalRow(convertMap(parser, fieldConverter, keyChecker))) } } @@ -484,8 +498,9 @@ class JacksonParser( case mt: MapType => val valueConverter = makeConverter(mt.valueType) + val keyChecker = makeMapKeyChecker(mt.keyType) (parser: JsonParser) => parseJsonToken[MapData](parser, dataType) { - case START_OBJECT => convertMap(parser, valueConverter) + case START_OBJECT => convertMap(parser, valueConverter, keyChecker) } case udt: UserDefinedType[_] => @@ -618,38 +633,65 @@ class JacksonParser( } /** - * Parse an object as a Map, preserving all fields. + * Parse an object as a Map. + * + * JSON object names used as CHAR/VARCHAR keys are length-checked without rewriting (see + * `makeMapKeyChecker`): CHAR keys must already be exactly n characters, and VARCHAR keys + * must already be at most n characters. Padding, trimming, and `mapKeyDedupPolicy` are not + * applied, and duplicate names are kept, exactly as for STRING keys. This differs from XML, + * which pads keys and then applies `mapKeyDedupPolicy`. + * + * A rejected key is treated like a failed value, regardless of `enablePartialResults`: the + * cause is recorded and the parser steps past the value so the loop still consumes the map's + * END_OBJECT, leaving the key unpaired. The check must not surface its result while the parser + * is still on the FIELD_NAME, or an enclosing struct would misread the map's remaining entries + * as its own sibling fields (SPARK-60108). STRING keys have no checker and take the fast path. */ private def convertMap( parser: JsonParser, - fieldConverter: ValueConverter): MapData = { + fieldConverter: ValueConverter, + keyChecker: Option[UTF8String => Option[Throwable]]): MapData = { val keys = ArrayBuffer.empty[UTF8String] val values = ArrayBuffer.empty[Any] var badRecordException: Option[Throwable] = None while (nextUntil(parser, JsonToken.END_OBJECT)) { - keys += UTF8String.fromString(parser.currentName) - try { - values += fieldConverter.apply(parser) - } catch { - case err: PartialValueException if enablePartialResults => - badRecordException = badRecordException.orElse(Some(err.cause)) - values += err.partialResult - case NonFatal(e) if enablePartialResults => + val key = UTF8String.fromString(parser.currentName) + keys += key + // Avoid allocating a closure per entry on the common no-checker path. + val keyRejection = keyChecker match { + case Some(check) => check(key) + case None => None + } + keyRejection match { + case Some(e) => badRecordException = badRecordException.orElse(Some(e)) + parser.nextToken() // step from FIELD_NAME onto the value parser.skipChildren() + case None => + try { + values += fieldConverter.apply(parser) + } catch { + case err: PartialValueException if enablePartialResults => + badRecordException = badRecordException.orElse(Some(err.cause)) + values += err.partialResult + case NonFatal(e) if enablePartialResults => + badRecordException = badRecordException.orElse(Some(e)) + parser.skipChildren() + } } } - // Value conversion can fail after the key is recorded. Do not build MapData from - // unpaired buffers: ArrayBasedMapData would throw a cardinality error and hide - // the original conversion failure (for example EXCEED_LIMIT_LENGTH). + // A rejected key or a failed value leaves the key recorded without a value. Rethrow the + // real cause instead of building an unbalanced ArrayBasedMapData, whose cardinality + // `require` would otherwise mask it. The row then becomes a bad record (null in PERMISSIVE, + // surfaced in FAILFAST). Balanced partial results (a trimmed value) still flow through + // PartialMapDataResultException below. if (keys.length != values.length) { throw badRecordException.get } - // Preserve every parsed JSON key/value pair, including exact duplicate names. - // ArrayBasedMapData is used directly to retain this historical behavior. + // The JSON map keeps every parsed pair, including exact duplicate names. val mapData = ArrayBasedMapData(keys.toArray, values.toArray) if (badRecordException.isEmpty) { @@ -659,6 +701,44 @@ class JacksonParser( } } + /** + * Builds the length check applied to this map's CHAR/VARCHAR keys, or `None` when the keys + * need no check (STRING keys, or standard semantics are off). Built once per map converter so + * the per-entry loop in `convertMap` does not re-inspect the key type. The check never pads or + * trims; it returns the rejection cause (with the actual key type, collation included, so the + * message is accurate) rather than throwing, so `convertMap` keeps full control of the parser + * position when a key is rejected. + */ + private def makeMapKeyChecker(keyType: DataType): Option[UTF8String => Option[Throwable]] = { + if (!charVarcharStandardSemantics) { + None + } else { + keyType match { + case c: CharType => Some(checkCharJsonMapKey(_, c)) + case v: VarcharType => Some(checkVarcharJsonMapKey(_, v)) + case _ => None + } + } + } + + // A CHAR(n) JSON object name is never padded: it must already be exactly n characters. + private def checkCharJsonMapKey(key: UTF8String, keyType: CharType): Option[Throwable] = { + if (key.numChars() != keyType.length) { + Some(QueryExecutionErrors.unsupportedJsonCharVarcharMapKey(key, keyType)) + } else { + None + } + } + + // A VARCHAR(n) JSON object name is never trimmed: it must already be at most n characters. + private def checkVarcharJsonMapKey(key: UTF8String, keyType: VarcharType): Option[Throwable] = { + if (key.numChars() > keyType.length) { + Some(QueryExecutionErrors.unsupportedJsonCharVarcharMapKey(key, keyType)) + } else { + None + } + } + /** * Parse an object as a Array. */ diff --git a/sql/catalyst/src/main/scala/org/apache/spark/sql/errors/QueryExecutionErrors.scala b/sql/catalyst/src/main/scala/org/apache/spark/sql/errors/QueryExecutionErrors.scala index b8bfd00cb2c84..d9db984f200b0 100644 --- a/sql/catalyst/src/main/scala/org/apache/spark/sql/errors/QueryExecutionErrors.scala +++ b/sql/catalyst/src/main/scala/org/apache/spark/sql/errors/QueryExecutionErrors.scala @@ -2739,6 +2739,30 @@ private[sql] object QueryExecutionErrors extends QueryErrorsBase with ExecutionE ) } + // SQLSTATE is 0A000 (feature-not-supported), not the 54006 of the value-side + // EXCEED_LIMIT_LENGTH: a too-long CHAR/VARCHAR value is truncatable overflow, but a JSON object + // name is a key that is never padded or trimmed, so an off-width name has no representation in + // a CHAR(n)/VARCHAR(n) map at all. It is a SparkRuntimeException (not + // SparkSQLFeatureNotSupportedException) because it is raised while parsing a row and must flow + // through Jackson's throw/catch parse-mode model (PERMISSIVE wraps it as a bad record, FAILFAST + // surfaces it). + def unsupportedJsonCharVarcharMapKey( + key: UTF8String, dataType: DataType): SparkRuntimeException = { + // The key is a raw JSON object name, which in map-shaped data is often user data (ids, + // emails) and can run up to Jackson's 50,000-character name limit, so cap what we echo. + val maxKeyChars = 128 + val displayKey = if (key.numChars() > maxKeyChars) { + UTF8String.concat(key.substring(0, maxKeyChars), UTF8String.fromString("...")) + } else { + key + } + new SparkRuntimeException( + errorClass = "UNSUPPORTED_JSON_CHAR_VARCHAR_MAP_KEY", + messageParameters = Map( + "key" -> toSQLValue(displayKey, StringType), + "dataType" -> toSQLType(dataType))) + } + def timestampAddOverflowError(micros: Long, amount: Long, unit: String): ArithmeticException = { new SparkArithmeticException( errorClass = "DATETIME_OVERFLOW", diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/date.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/date.sql.out index 0e4d2d4e99e26..5c53637eff6b6 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/date.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/date.sql.out @@ -736,7 +736,7 @@ Project [to_date(26/October/2015, Some(dd/MMMMM/yyyy), Some(America/Los_Angeles) -- !query select from_json('{"d":"26/October/2015"}', 'd Date', map('dateFormat', 'dd/MMMMM/yyyy')) -- !query analysis -Project [from_json(StructField(d,DateType,true), (dateFormat,dd/MMMMM/yyyy), {"d":"26/October/2015"}, Some(America/Los_Angeles), false) AS from_json({"d":"26/October/2015"})#x] +Project [from_json(StructField(d,DateType,true), (dateFormat,dd/MMMMM/yyyy), {"d":"26/October/2015"}, Some(America/Los_Angeles), false, false) AS from_json({"d":"26/October/2015"})#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/datetime-legacy.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/datetime-legacy.sql.out index ee7f132edac96..17a71df4101f6 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/datetime-legacy.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/datetime-legacy.sql.out @@ -736,7 +736,7 @@ Project [to_date(26/October/2015, Some(dd/MMMMM/yyyy), Some(America/Los_Angeles) -- !query select from_json('{"d":"26/October/2015"}', 'd Date', map('dateFormat', 'dd/MMMMM/yyyy')) -- !query analysis -Project [from_json(StructField(d,DateType,true), (dateFormat,dd/MMMMM/yyyy), {"d":"26/October/2015"}, Some(America/Los_Angeles), false) AS from_json({"d":"26/October/2015"})#x] +Project [from_json(StructField(d,DateType,true), (dateFormat,dd/MMMMM/yyyy), {"d":"26/October/2015"}, Some(America/Los_Angeles), false, false) AS from_json({"d":"26/October/2015"})#x] +- OneRowRelation @@ -1970,7 +1970,7 @@ Project [unix_timestamp(22 05 2020 Friday, dd MM yyyy EEEEE, Some(America/Los_An -- !query select from_json('{"t":"26/October/2015"}', 't Timestamp', map('timestampFormat', 'dd/MMMMM/yyyy')) -- !query analysis -Project [from_json(StructField(t,TimestampType,true), (timestampFormat,dd/MMMMM/yyyy), {"t":"26/October/2015"}, Some(America/Los_Angeles), false) AS from_json({"t":"26/October/2015"})#x] +Project [from_json(StructField(t,TimestampType,true), (timestampFormat,dd/MMMMM/yyyy), {"t":"26/October/2015"}, Some(America/Los_Angeles), false, false) AS from_json({"t":"26/October/2015"})#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/interval.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/interval.sql.out index eb9bc4d913de7..154c2565d7a89 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/interval.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/interval.sql.out @@ -2116,7 +2116,7 @@ SELECT to_csv(named_struct('a', interval 32 year, 'b', interval 10 month)), from_csv(to_csv(named_struct('a', interval 32 year, 'b', interval 10 month)), 'a interval year, b interval month') -- !query analysis -Project [from_json(StructField(a,CalendarIntervalType,true), {"a":"1 days"}, Some(America/Los_Angeles), false) AS from_json({"a":"1 days"})#x, from_csv(StructField(a,IntegerType,true), StructField(b,YearMonthIntervalType(0,0),true), 1, 1, Some(America/Los_Angeles), None) AS from_csv(1, 1)#x, to_json(from_json(StructField(a,CalendarIntervalType,true), {"a":"1 days"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"a":"1 days"}))#x, to_csv(from_csv(StructField(a,IntegerType,true), StructField(b,YearMonthIntervalType(0,0),true), 1, 1, Some(America/Los_Angeles), None), Some(America/Los_Angeles)) AS to_csv(from_csv(1, 1))#x, to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH), Some(America/Los_Angeles)) AS to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH))#x, from_csv(StructField(a,YearMonthIntervalType(0,0),true), StructField(b,YearMonthIntervalType(1,1),true), to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH), Some(America/Los_Angeles)), Some(America/Los_Angeles), None) AS from_csv(to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH)))#x] +Project [from_json(StructField(a,CalendarIntervalType,true), {"a":"1 days"}, Some(America/Los_Angeles), false, false) AS from_json({"a":"1 days"})#x, from_csv(StructField(a,IntegerType,true), StructField(b,YearMonthIntervalType(0,0),true), 1, 1, Some(America/Los_Angeles), None) AS from_csv(1, 1)#x, to_json(from_json(StructField(a,CalendarIntervalType,true), {"a":"1 days"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"a":"1 days"}))#x, to_csv(from_csv(StructField(a,IntegerType,true), StructField(b,YearMonthIntervalType(0,0),true), 1, 1, Some(America/Los_Angeles), None), Some(America/Los_Angeles)) AS to_csv(from_csv(1, 1))#x, to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH), Some(America/Los_Angeles)) AS to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH))#x, from_csv(StructField(a,YearMonthIntervalType(0,0),true), StructField(b,YearMonthIntervalType(1,1),true), to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH), Some(America/Los_Angeles)), Some(America/Los_Angeles), None) AS from_csv(to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH)))#x] +- OneRowRelation @@ -2127,7 +2127,7 @@ SELECT to_json(map('a', interval 100 day 130 minute)), from_json(to_json(map('a', interval 100 day 130 minute)), 'a interval day to minute') -- !query analysis -Project [from_json(StructField(a,DayTimeIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false) AS from_json({"a":"1"})#x, to_json(from_json(StructField(a,DayTimeIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"a":"1"}))#x, to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE), Some(America/Los_Angeles)) AS to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE))#x, from_json(StructField(a,DayTimeIntervalType(0,2),true), to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE)))#x] +Project [from_json(StructField(a,DayTimeIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false, false) AS from_json({"a":"1"})#x, to_json(from_json(StructField(a,DayTimeIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"a":"1"}))#x, to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE), Some(America/Los_Angeles)) AS to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE))#x, from_json(StructField(a,DayTimeIntervalType(0,2),true), to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE)))#x] +- OneRowRelation @@ -2138,7 +2138,7 @@ SELECT to_json(map('a', interval 32 year 10 month)), from_json(to_json(map('a', interval 32 year 10 month)), 'a interval year to month') -- !query analysis -Project [from_json(StructField(a,YearMonthIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false) AS from_json({"a":"1"})#x, to_json(from_json(StructField(a,YearMonthIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"a":"1"}))#x, to_json(map(a, INTERVAL '32-10' YEAR TO MONTH), Some(America/Los_Angeles)) AS to_json(map(a, INTERVAL '32-10' YEAR TO MONTH))#x, from_json(StructField(a,YearMonthIntervalType(0,1),true), to_json(map(a, INTERVAL '32-10' YEAR TO MONTH), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(map(a, INTERVAL '32-10' YEAR TO MONTH)))#x] +Project [from_json(StructField(a,YearMonthIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false, false) AS from_json({"a":"1"})#x, to_json(from_json(StructField(a,YearMonthIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"a":"1"}))#x, to_json(map(a, INTERVAL '32-10' YEAR TO MONTH), Some(America/Los_Angeles)) AS to_json(map(a, INTERVAL '32-10' YEAR TO MONTH))#x, from_json(StructField(a,YearMonthIntervalType(0,1),true), to_json(map(a, INTERVAL '32-10' YEAR TO MONTH), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(map(a, INTERVAL '32-10' YEAR TO MONTH)))#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/json-functions.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/json-functions.sql.out index d21974b86640a..7f7f3dab9cafb 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/json-functions.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/json-functions.sql.out @@ -118,14 +118,14 @@ org.apache.spark.sql.AnalysisException -- !query select from_json('{"a":1}', 'a INT') -- !query analysis -Project [from_json(StructField(a,IntegerType,true), {"a":1}, Some(America/Los_Angeles), false) AS from_json({"a":1})#x] +Project [from_json(StructField(a,IntegerType,true), {"a":1}, Some(America/Los_Angeles), false, false) AS from_json({"a":1})#x] +- OneRowRelation -- !query select from_json('{"time":"26/08/2015"}', 'time Timestamp', map('timestampFormat', 'dd/MM/yyyy')) -- !query analysis -Project [from_json(StructField(time,TimestampType,true), (timestampFormat,dd/MM/yyyy), {"time":"26/08/2015"}, Some(America/Los_Angeles), false) AS from_json({"time":"26/08/2015"})#x] +Project [from_json(StructField(time,TimestampType,true), (timestampFormat,dd/MM/yyyy), {"time":"26/08/2015"}, Some(America/Los_Angeles), false, false) AS from_json({"time":"26/08/2015"})#x] +- OneRowRelation @@ -279,14 +279,14 @@ DropTempViewCommand jsonTable, true -- !query select from_json('{"a":1, "b":2}', 'map') -- !query analysis -Project [from_json(MapType(StringType,IntegerType,true), {"a":1, "b":2}, Some(America/Los_Angeles), false) AS entries#x] +Project [from_json(MapType(StringType,IntegerType,true), {"a":1, "b":2}, Some(America/Los_Angeles), false, false) AS entries#x] +- OneRowRelation -- !query select from_json('{"a":1, "b":"2"}', 'struct') -- !query analysis -Project [from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), {"a":1, "b":"2"}, Some(America/Los_Angeles), false) AS from_json({"a":1, "b":"2"})#x] +Project [from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), {"a":1, "b":"2"}, Some(America/Los_Angeles), false, false) AS from_json({"a":1, "b":"2"})#x] +- OneRowRelation @@ -300,70 +300,70 @@ Project [schema_of_json({"c1":0, "c2":[1]}) AS schema_of_json({"c1":0, "c2":[1]} -- !query select from_json('{"c1":[1, 2, 3]}', schema_of_json('{"c1":[0]}')) -- !query analysis -Project [from_json(StructField(c1,ArrayType(LongType,true),true), {"c1":[1, 2, 3]}, Some(America/Los_Angeles), false) AS from_json({"c1":[1, 2, 3]})#x] +Project [from_json(StructField(c1,ArrayType(LongType,true),true), {"c1":[1, 2, 3]}, Some(America/Los_Angeles), false, false) AS from_json({"c1":[1, 2, 3]})#x] +- OneRowRelation -- !query select from_json('[1, 2, 3]', 'array') -- !query analysis -Project [from_json(ArrayType(IntegerType,true), [1, 2, 3], Some(America/Los_Angeles), false) AS from_json([1, 2, 3])#x] +Project [from_json(ArrayType(IntegerType,true), [1, 2, 3], Some(America/Los_Angeles), false, false) AS from_json([1, 2, 3])#x] +- OneRowRelation -- !query select from_json('[1, "2", 3]', 'array') -- !query analysis -Project [from_json(ArrayType(IntegerType,true), [1, "2", 3], Some(America/Los_Angeles), false) AS from_json([1, "2", 3])#x] +Project [from_json(ArrayType(IntegerType,true), [1, "2", 3], Some(America/Los_Angeles), false, false) AS from_json([1, "2", 3])#x] +- OneRowRelation -- !query select from_json('[1, 2, null]', 'array') -- !query analysis -Project [from_json(ArrayType(IntegerType,true), [1, 2, null], Some(America/Los_Angeles), false) AS from_json([1, 2, null])#x] +Project [from_json(ArrayType(IntegerType,true), [1, 2, null], Some(America/Los_Angeles), false, false) AS from_json([1, 2, null])#x] +- OneRowRelation -- !query select from_json('[{"a": 1}, {"a":2}]', 'array>') -- !query analysis -Project [from_json(ArrayType(StructType(StructField(a,IntegerType,true)),true), [{"a": 1}, {"a":2}], Some(America/Los_Angeles), false) AS from_json([{"a": 1}, {"a":2}])#x] +Project [from_json(ArrayType(StructType(StructField(a,IntegerType,true)),true), [{"a": 1}, {"a":2}], Some(America/Los_Angeles), false, false) AS from_json([{"a": 1}, {"a":2}])#x] +- OneRowRelation -- !query select from_json('{"a": 1}', 'array>') -- !query analysis -Project [from_json(ArrayType(StructType(StructField(a,IntegerType,true)),true), {"a": 1}, Some(America/Los_Angeles), false) AS from_json({"a": 1})#x] +Project [from_json(ArrayType(StructType(StructField(a,IntegerType,true)),true), {"a": 1}, Some(America/Los_Angeles), false, false) AS from_json({"a": 1})#x] +- OneRowRelation -- !query select from_json('[null, {"a":2}]', 'array>') -- !query analysis -Project [from_json(ArrayType(StructType(StructField(a,IntegerType,true)),true), [null, {"a":2}], Some(America/Los_Angeles), false) AS from_json([null, {"a":2}])#x] +Project [from_json(ArrayType(StructType(StructField(a,IntegerType,true)),true), [null, {"a":2}], Some(America/Los_Angeles), false, false) AS from_json([null, {"a":2}])#x] +- OneRowRelation -- !query select from_json('[{"a": 1}, {"b":2}]', 'array>') -- !query analysis -Project [from_json(ArrayType(MapType(StringType,IntegerType,true),true), [{"a": 1}, {"b":2}], Some(America/Los_Angeles), false) AS from_json([{"a": 1}, {"b":2}])#x] +Project [from_json(ArrayType(MapType(StringType,IntegerType,true),true), [{"a": 1}, {"b":2}], Some(America/Los_Angeles), false, false) AS from_json([{"a": 1}, {"b":2}])#x] +- OneRowRelation -- !query select from_json('[{"a": 1}, 2]', 'array>') -- !query analysis -Project [from_json(ArrayType(MapType(StringType,IntegerType,true),true), [{"a": 1}, 2], Some(America/Los_Angeles), false) AS from_json([{"a": 1}, 2])#x] +Project [from_json(ArrayType(MapType(StringType,IntegerType,true),true), [{"a": 1}, 2], Some(America/Los_Angeles), false, false) AS from_json([{"a": 1}, 2])#x] +- OneRowRelation -- !query select from_json('{"d": "2012-12-15", "t": "2012-12-15 15:15:15"}', 'd date, t timestamp') -- !query analysis -Project [from_json(StructField(d,DateType,true), StructField(t,TimestampType,true), {"d": "2012-12-15", "t": "2012-12-15 15:15:15"}, Some(America/Los_Angeles), false) AS from_json({"d": "2012-12-15", "t": "2012-12-15 15:15:15"})#x] +Project [from_json(StructField(d,DateType,true), StructField(t,TimestampType,true), {"d": "2012-12-15", "t": "2012-12-15 15:15:15"}, Some(America/Los_Angeles), false, false) AS from_json({"d": "2012-12-15", "t": "2012-12-15 15:15:15"})#x] +- OneRowRelation @@ -373,7 +373,7 @@ select from_json( 'd date, t timestamp', map('dateFormat', 'MM/dd yyyy', 'timestampFormat', 'MM/dd yyyy HH:mm:ss')) -- !query analysis -Project [from_json(StructField(d,DateType,true), StructField(t,TimestampType,true), (dateFormat,MM/dd yyyy), (timestampFormat,MM/dd yyyy HH:mm:ss), {"d": "12/15 2012", "t": "12/15 2012 15:15:15"}, Some(America/Los_Angeles), false) AS from_json({"d": "12/15 2012", "t": "12/15 2012 15:15:15"})#x] +Project [from_json(StructField(d,DateType,true), StructField(t,TimestampType,true), (dateFormat,MM/dd yyyy), (timestampFormat,MM/dd yyyy HH:mm:ss), {"d": "12/15 2012", "t": "12/15 2012 15:15:15"}, Some(America/Los_Angeles), false, false) AS from_json({"d": "12/15 2012", "t": "12/15 2012 15:15:15"})#x] +- OneRowRelation @@ -383,7 +383,7 @@ select from_json( 'd date', map('dateFormat', 'MM-dd')) -- !query analysis -Project [from_json(StructField(d,DateType,true), (dateFormat,MM-dd), {"d": "02-29"}, Some(America/Los_Angeles), false) AS from_json({"d": "02-29"})#x] +Project [from_json(StructField(d,DateType,true), (dateFormat,MM-dd), {"d": "02-29"}, Some(America/Los_Angeles), false, false) AS from_json({"d": "02-29"})#x] +- OneRowRelation @@ -393,7 +393,7 @@ select from_json( 't timestamp', map('timestampFormat', 'MM-dd')) -- !query analysis -Project [from_json(StructField(t,TimestampType,true), (timestampFormat,MM-dd), {"t": "02-29"}, Some(America/Los_Angeles), false) AS from_json({"t": "02-29"})#x] +Project [from_json(StructField(t,TimestampType,true), (timestampFormat,MM-dd), {"t": "02-29"}, Some(America/Los_Angeles), false, false) AS from_json({"t": "02-29"})#x] +- OneRowRelation @@ -917,56 +917,56 @@ DropTempViewCommand jsonTable, true -- !query select from_json('{"time": "14:30:45"}', 'time TIME(0)') -- !query analysis -Project [from_json(StructField(time,TimeType(0),true), {"time": "14:30:45"}, Some(America/Los_Angeles), false) AS from_json({"time": "14:30:45"})#x] +Project [from_json(StructField(time,TimeType(0),true), {"time": "14:30:45"}, Some(America/Los_Angeles), false, false) AS from_json({"time": "14:30:45"})#x] +- OneRowRelation -- !query select from_json('{"time": "14:30:45.123"}', 'time TIME(3)') -- !query analysis -Project [from_json(StructField(time,TimeType(3),true), {"time": "14:30:45.123"}, Some(America/Los_Angeles), false) AS from_json({"time": "14:30:45.123"})#x] +Project [from_json(StructField(time,TimeType(3),true), {"time": "14:30:45.123"}, Some(America/Los_Angeles), false, false) AS from_json({"time": "14:30:45.123"})#x] +- OneRowRelation -- !query select from_json('{"time": "14:30:45.123456"}', 'time TIME(6)') -- !query analysis -Project [from_json(StructField(time,TimeType(6),true), {"time": "14:30:45.123456"}, Some(America/Los_Angeles), false) AS from_json({"time": "14:30:45.123456"})#x] +Project [from_json(StructField(time,TimeType(6),true), {"time": "14:30:45.123456"}, Some(America/Los_Angeles), false, false) AS from_json({"time": "14:30:45.123456"})#x] +- OneRowRelation -- !query select from_json('{"time": "14-30-45.123456"}', 'time TIME(6)', map('timeFormat', 'HH-mm-ss.SSSSSS')) -- !query analysis -Project [from_json(StructField(time,TimeType(6),true), (timeFormat,HH-mm-ss.SSSSSS), {"time": "14-30-45.123456"}, Some(America/Los_Angeles), false) AS from_json({"time": "14-30-45.123456"})#x] +Project [from_json(StructField(time,TimeType(6),true), (timeFormat,HH-mm-ss.SSSSSS), {"time": "14-30-45.123456"}, Some(America/Los_Angeles), false, false) AS from_json({"time": "14-30-45.123456"})#x] +- OneRowRelation -- !query select from_json('{"t1": "09:00:00", "t2": "17:30:00"}', 't1 TIME, t2 TIME') -- !query analysis -Project [from_json(StructField(t1,TimeType(6),true), StructField(t2,TimeType(6),true), {"t1": "09:00:00", "t2": "17:30:00"}, Some(America/Los_Angeles), false) AS from_json({"t1": "09:00:00", "t2": "17:30:00"})#x] +Project [from_json(StructField(t1,TimeType(6),true), StructField(t2,TimeType(6),true), {"t1": "09:00:00", "t2": "17:30:00"}, Some(America/Los_Angeles), false, false) AS from_json({"t1": "09:00:00", "t2": "17:30:00"})#x] +- OneRowRelation -- !query select from_json('{"time": "25:00:00"}', 'time TIME') -- !query analysis -Project [from_json(StructField(time,TimeType(6),true), {"time": "25:00:00"}, Some(America/Los_Angeles), false) AS from_json({"time": "25:00:00"})#x] +Project [from_json(StructField(time,TimeType(6),true), {"time": "25:00:00"}, Some(America/Los_Angeles), false, false) AS from_json({"time": "25:00:00"})#x] +- OneRowRelation -- !query select from_json('{"time": "invalid"}', 'time TIME') -- !query analysis -Project [from_json(StructField(time,TimeType(6),true), {"time": "invalid"}, Some(America/Los_Angeles), false) AS from_json({"time": "invalid"})#x] +Project [from_json(StructField(time,TimeType(6),true), {"time": "invalid"}, Some(America/Los_Angeles), false, false) AS from_json({"time": "invalid"})#x] +- OneRowRelation -- !query select from_json('{"time": null}', 'time TIME') -- !query analysis -Project [from_json(StructField(time,TimeType(6),true), {"time": null}, Some(America/Los_Angeles), false) AS from_json({"time": null})#x] +Project [from_json(StructField(time,TimeType(6),true), {"time": null}, Some(America/Los_Angeles), false, false) AS from_json({"time": null})#x] +- OneRowRelation @@ -1001,126 +1001,126 @@ Project [to_json(array(09:00:00, 17:45:30), Some(America/Los_Angeles)) AS to_jso -- !query select from_json(to_json(named_struct('time', TIME'14:30:45')), 'time TIME(0)') -- !query analysis -Project [from_json(StructField(time,TimeType(0),true), to_json(named_struct(time, 14:30:45), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(named_struct(time, TIME '14:30:45')))#x] +Project [from_json(StructField(time,TimeType(0),true), to_json(named_struct(time, 14:30:45), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(named_struct(time, TIME '14:30:45')))#x] +- OneRowRelation -- !query select from_json(to_json(named_struct('time', TIME'14:30:45.1')), 'time TIME(1)') -- !query analysis -Project [from_json(StructField(time,TimeType(1),true), to_json(named_struct(time, 14:30:45.1), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(named_struct(time, TIME '14:30:45.1')))#x] +Project [from_json(StructField(time,TimeType(1),true), to_json(named_struct(time, 14:30:45.1), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(named_struct(time, TIME '14:30:45.1')))#x] +- OneRowRelation -- !query select from_json(to_json(named_struct('time', TIME'14:30:45.12')), 'time TIME(2)') -- !query analysis -Project [from_json(StructField(time,TimeType(2),true), to_json(named_struct(time, 14:30:45.12), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(named_struct(time, TIME '14:30:45.12')))#x] +Project [from_json(StructField(time,TimeType(2),true), to_json(named_struct(time, 14:30:45.12), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(named_struct(time, TIME '14:30:45.12')))#x] +- OneRowRelation -- !query select from_json(to_json(named_struct('time', TIME'14:30:45.123')), 'time TIME(3)') -- !query analysis -Project [from_json(StructField(time,TimeType(3),true), to_json(named_struct(time, 14:30:45.123), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(named_struct(time, TIME '14:30:45.123')))#x] +Project [from_json(StructField(time,TimeType(3),true), to_json(named_struct(time, 14:30:45.123), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(named_struct(time, TIME '14:30:45.123')))#x] +- OneRowRelation -- !query select from_json(to_json(named_struct('time', TIME'14:30:45.1234')), 'time TIME(4)') -- !query analysis -Project [from_json(StructField(time,TimeType(4),true), to_json(named_struct(time, 14:30:45.1234), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(named_struct(time, TIME '14:30:45.1234')))#x] +Project [from_json(StructField(time,TimeType(4),true), to_json(named_struct(time, 14:30:45.1234), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(named_struct(time, TIME '14:30:45.1234')))#x] +- OneRowRelation -- !query select from_json(to_json(named_struct('time', TIME'14:30:45.12345')), 'time TIME(5)') -- !query analysis -Project [from_json(StructField(time,TimeType(5),true), to_json(named_struct(time, 14:30:45.12345), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(named_struct(time, TIME '14:30:45.12345')))#x] +Project [from_json(StructField(time,TimeType(5),true), to_json(named_struct(time, 14:30:45.12345), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(named_struct(time, TIME '14:30:45.12345')))#x] +- OneRowRelation -- !query select from_json(to_json(named_struct('time', TIME'14:30:45.123456')), 'time TIME(6)') -- !query analysis -Project [from_json(StructField(time,TimeType(6),true), to_json(named_struct(time, 14:30:45.123456), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(named_struct(time, TIME '14:30:45.123456')))#x] +Project [from_json(StructField(time,TimeType(6),true), to_json(named_struct(time, 14:30:45.123456), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(named_struct(time, TIME '14:30:45.123456')))#x] +- OneRowRelation -- !query select from_json(to_json(named_struct('time', TIME'00:00:00')), 'time TIME(0)') -- !query analysis -Project [from_json(StructField(time,TimeType(0),true), to_json(named_struct(time, 00:00:00), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(named_struct(time, TIME '00:00:00')))#x] +Project [from_json(StructField(time,TimeType(0),true), to_json(named_struct(time, 00:00:00), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(named_struct(time, TIME '00:00:00')))#x] +- OneRowRelation -- !query select from_json(to_json(named_struct('time', TIME'23:59:59.999999')), 'time TIME(6)') -- !query analysis -Project [from_json(StructField(time,TimeType(6),true), to_json(named_struct(time, 23:59:59.999999), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(named_struct(time, TIME '23:59:59.999999')))#x] +Project [from_json(StructField(time,TimeType(6),true), to_json(named_struct(time, 23:59:59.999999), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(named_struct(time, TIME '23:59:59.999999')))#x] +- OneRowRelation -- !query select to_json(from_json('{"time":"14:30:45"}', 'time TIME(0)')) -- !query analysis -Project [to_json(from_json(StructField(time,TimeType(0),true), {"time":"14:30:45"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45"}))#x] +Project [to_json(from_json(StructField(time,TimeType(0),true), {"time":"14:30:45"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45"}))#x] +- OneRowRelation -- !query select to_json(from_json('{"time":"14:30:45.1"}', 'time TIME(1)')) -- !query analysis -Project [to_json(from_json(StructField(time,TimeType(1),true), {"time":"14:30:45.1"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45.1"}))#x] +Project [to_json(from_json(StructField(time,TimeType(1),true), {"time":"14:30:45.1"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45.1"}))#x] +- OneRowRelation -- !query select to_json(from_json('{"time":"14:30:45.12"}', 'time TIME(2)')) -- !query analysis -Project [to_json(from_json(StructField(time,TimeType(2),true), {"time":"14:30:45.12"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45.12"}))#x] +Project [to_json(from_json(StructField(time,TimeType(2),true), {"time":"14:30:45.12"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45.12"}))#x] +- OneRowRelation -- !query select to_json(from_json('{"time":"14:30:45.123"}', 'time TIME(3)')) -- !query analysis -Project [to_json(from_json(StructField(time,TimeType(3),true), {"time":"14:30:45.123"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45.123"}))#x] +Project [to_json(from_json(StructField(time,TimeType(3),true), {"time":"14:30:45.123"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45.123"}))#x] +- OneRowRelation -- !query select to_json(from_json('{"time":"14:30:45.1234"}', 'time TIME(4)')) -- !query analysis -Project [to_json(from_json(StructField(time,TimeType(4),true), {"time":"14:30:45.1234"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45.1234"}))#x] +Project [to_json(from_json(StructField(time,TimeType(4),true), {"time":"14:30:45.1234"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45.1234"}))#x] +- OneRowRelation -- !query select to_json(from_json('{"time":"14:30:45.12345"}', 'time TIME(5)')) -- !query analysis -Project [to_json(from_json(StructField(time,TimeType(5),true), {"time":"14:30:45.12345"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45.12345"}))#x] +Project [to_json(from_json(StructField(time,TimeType(5),true), {"time":"14:30:45.12345"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45.12345"}))#x] +- OneRowRelation -- !query select to_json(from_json('{"time":"14:30:45.123456"}', 'time TIME(6)')) -- !query analysis -Project [to_json(from_json(StructField(time,TimeType(6),true), {"time":"14:30:45.123456"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45.123456"}))#x] +Project [to_json(from_json(StructField(time,TimeType(6),true), {"time":"14:30:45.123456"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"14:30:45.123456"}))#x] +- OneRowRelation -- !query select to_json(from_json('{"time":"00:00:00"}', 'time TIME(0)')) -- !query analysis -Project [to_json(from_json(StructField(time,TimeType(0),true), {"time":"00:00:00"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"00:00:00"}))#x] +Project [to_json(from_json(StructField(time,TimeType(0),true), {"time":"00:00:00"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"00:00:00"}))#x] +- OneRowRelation -- !query select to_json(from_json('{"time":"23:59:59.999999"}', 'time TIME(6)')) -- !query analysis -Project [to_json(from_json(StructField(time,TimeType(6),true), {"time":"23:59:59.999999"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"23:59:59.999999"}))#x] +Project [to_json(from_json(StructField(time,TimeType(6),true), {"time":"23:59:59.999999"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"time":"23:59:59.999999"}))#x] +- OneRowRelation @@ -1143,7 +1143,7 @@ select from_json('{"time": "14:30:45"}', 'time TIME') LIMIT 1 -- !query analysis GlobalLimit 1 +- LocalLimit 1 - +- Project [from_json(StructField(time,TimeType(6),true), {"time": "14:30:45"}, Some(America/Los_Angeles), false) AS from_json({"time": "14:30:45"})#x] + +- Project [from_json(StructField(time,TimeType(6),true), {"time": "14:30:45"}, Some(America/Los_Angeles), false, false) AS from_json({"time": "14:30:45"})#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/date.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/date.sql.out index 88c7d7b4e7d72..9420c7670c4a6 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/date.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/date.sql.out @@ -811,7 +811,7 @@ Project [to_date(26/October/2015, Some(dd/MMMMM/yyyy), Some(America/Los_Angeles) -- !query select from_json('{"d":"26/October/2015"}', 'd Date', map('dateFormat', 'dd/MMMMM/yyyy')) -- !query analysis -Project [from_json(StructField(d,DateType,true), (dateFormat,dd/MMMMM/yyyy), {"d":"26/October/2015"}, Some(America/Los_Angeles), false) AS from_json({"d":"26/October/2015"})#x] +Project [from_json(StructField(d,DateType,true), (dateFormat,dd/MMMMM/yyyy), {"d":"26/October/2015"}, Some(America/Los_Angeles), false, false) AS from_json({"d":"26/October/2015"})#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/interval.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/interval.sql.out index 52b02dae19100..cc0ad2de6e434 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/interval.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/interval.sql.out @@ -2116,7 +2116,7 @@ SELECT to_csv(named_struct('a', interval 32 year, 'b', interval 10 month)), from_csv(to_csv(named_struct('a', interval 32 year, 'b', interval 10 month)), 'a interval year, b interval month') -- !query analysis -Project [from_json(StructField(a,CalendarIntervalType,true), {"a":"1 days"}, Some(America/Los_Angeles), false) AS from_json({"a":"1 days"})#x, from_csv(StructField(a,IntegerType,true), StructField(b,YearMonthIntervalType(0,0),true), 1, 1, Some(America/Los_Angeles), None) AS from_csv(1, 1)#x, to_json(from_json(StructField(a,CalendarIntervalType,true), {"a":"1 days"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"a":"1 days"}))#x, to_csv(from_csv(StructField(a,IntegerType,true), StructField(b,YearMonthIntervalType(0,0),true), 1, 1, Some(America/Los_Angeles), None), Some(America/Los_Angeles)) AS to_csv(from_csv(1, 1))#x, to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH), Some(America/Los_Angeles)) AS to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH))#x, from_csv(StructField(a,YearMonthIntervalType(0,0),true), StructField(b,YearMonthIntervalType(1,1),true), to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH), Some(America/Los_Angeles)), Some(America/Los_Angeles), None) AS from_csv(to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH)))#x] +Project [from_json(StructField(a,CalendarIntervalType,true), {"a":"1 days"}, Some(America/Los_Angeles), false, false) AS from_json({"a":"1 days"})#x, from_csv(StructField(a,IntegerType,true), StructField(b,YearMonthIntervalType(0,0),true), 1, 1, Some(America/Los_Angeles), None) AS from_csv(1, 1)#x, to_json(from_json(StructField(a,CalendarIntervalType,true), {"a":"1 days"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"a":"1 days"}))#x, to_csv(from_csv(StructField(a,IntegerType,true), StructField(b,YearMonthIntervalType(0,0),true), 1, 1, Some(America/Los_Angeles), None), Some(America/Los_Angeles)) AS to_csv(from_csv(1, 1))#x, to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH), Some(America/Los_Angeles)) AS to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH))#x, from_csv(StructField(a,YearMonthIntervalType(0,0),true), StructField(b,YearMonthIntervalType(1,1),true), to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH), Some(America/Los_Angeles)), Some(America/Los_Angeles), None) AS from_csv(to_csv(named_struct(a, INTERVAL '32' YEAR, b, INTERVAL '10' MONTH)))#x] +- OneRowRelation @@ -2127,7 +2127,7 @@ SELECT to_json(map('a', interval 100 day 130 minute)), from_json(to_json(map('a', interval 100 day 130 minute)), 'a interval day to minute') -- !query analysis -Project [from_json(StructField(a,DayTimeIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false) AS from_json({"a":"1"})#x, to_json(from_json(StructField(a,DayTimeIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"a":"1"}))#x, to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE), Some(America/Los_Angeles)) AS to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE))#x, from_json(StructField(a,DayTimeIntervalType(0,2),true), to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE)))#x] +Project [from_json(StructField(a,DayTimeIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false, false) AS from_json({"a":"1"})#x, to_json(from_json(StructField(a,DayTimeIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"a":"1"}))#x, to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE), Some(America/Los_Angeles)) AS to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE))#x, from_json(StructField(a,DayTimeIntervalType(0,2),true), to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(map(a, INTERVAL '100 02:10' DAY TO MINUTE)))#x] +- OneRowRelation @@ -2138,7 +2138,7 @@ SELECT to_json(map('a', interval 32 year 10 month)), from_json(to_json(map('a', interval 32 year 10 month)), 'a interval year to month') -- !query analysis -Project [from_json(StructField(a,YearMonthIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false) AS from_json({"a":"1"})#x, to_json(from_json(StructField(a,YearMonthIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false), Some(America/Los_Angeles)) AS to_json(from_json({"a":"1"}))#x, to_json(map(a, INTERVAL '32-10' YEAR TO MONTH), Some(America/Los_Angeles)) AS to_json(map(a, INTERVAL '32-10' YEAR TO MONTH))#x, from_json(StructField(a,YearMonthIntervalType(0,1),true), to_json(map(a, INTERVAL '32-10' YEAR TO MONTH), Some(America/Los_Angeles)), Some(America/Los_Angeles), false) AS from_json(to_json(map(a, INTERVAL '32-10' YEAR TO MONTH)))#x] +Project [from_json(StructField(a,YearMonthIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false, false) AS from_json({"a":"1"})#x, to_json(from_json(StructField(a,YearMonthIntervalType(0,0),true), {"a":"1"}, Some(America/Los_Angeles), false, false), Some(America/Los_Angeles)) AS to_json(from_json({"a":"1"}))#x, to_json(map(a, INTERVAL '32-10' YEAR TO MONTH), Some(America/Los_Angeles)) AS to_json(map(a, INTERVAL '32-10' YEAR TO MONTH))#x, from_json(StructField(a,YearMonthIntervalType(0,1),true), to_json(map(a, INTERVAL '32-10' YEAR TO MONTH), Some(America/Los_Angeles)), Some(America/Los_Angeles), false, false) AS from_json(to_json(map(a, INTERVAL '32-10' YEAR TO MONTH)))#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/parse-schema-string.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/parse-schema-string.sql.out index ae8e47ed3665c..f85bfed72879c 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/parse-schema-string.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/parse-schema-string.sql.out @@ -16,12 +16,12 @@ Project [from_csv(StructField(cube,IntegerType,true), 1, Some(America/Los_Angele -- !query select from_json('{"create":1}', 'create INT') -- !query analysis -Project [from_json(StructField(create,IntegerType,true), {"create":1}, Some(America/Los_Angeles), false) AS from_json({"create":1})#x] +Project [from_json(StructField(create,IntegerType,true), {"create":1}, Some(America/Los_Angeles), false, false) AS from_json({"create":1})#x] +- OneRowRelation -- !query select from_json('{"cube":1}', 'cube INT') -- !query analysis -Project [from_json(StructField(cube,IntegerType,true), {"cube":1}, Some(America/Los_Angeles), false) AS from_json({"cube":1})#x] +Project [from_json(StructField(cube,IntegerType,true), {"cube":1}, Some(America/Los_Angeles), false, false) AS from_json({"cube":1})#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/timestamp.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/timestamp.sql.out index 86dc87b07ceef..2d7e8c0f5bbb5 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/timestamp.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/nonansi/timestamp.sql.out @@ -1086,7 +1086,7 @@ Project [unix_timestamp(22 05 2020 Friday, dd MM yyyy EEEEE, Some(America/Los_An -- !query select from_json('{"t":"26/October/2015"}', 't Timestamp', map('timestampFormat', 'dd/MMMMM/yyyy')) -- !query analysis -Project [from_json(StructField(t,TimestampType,true), (timestampFormat,dd/MMMMM/yyyy), {"t":"26/October/2015"}, Some(America/Los_Angeles), false) AS from_json({"t":"26/October/2015"})#x] +Project [from_json(StructField(t,TimestampType,true), (timestampFormat,dd/MMMMM/yyyy), {"t":"26/October/2015"}, Some(America/Los_Angeles), false, false) AS from_json({"t":"26/October/2015"})#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/parse-schema-string.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/parse-schema-string.sql.out index ae8e47ed3665c..f85bfed72879c 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/parse-schema-string.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/parse-schema-string.sql.out @@ -16,12 +16,12 @@ Project [from_csv(StructField(cube,IntegerType,true), 1, Some(America/Los_Angele -- !query select from_json('{"create":1}', 'create INT') -- !query analysis -Project [from_json(StructField(create,IntegerType,true), {"create":1}, Some(America/Los_Angeles), false) AS from_json({"create":1})#x] +Project [from_json(StructField(create,IntegerType,true), {"create":1}, Some(America/Los_Angeles), false, false) AS from_json({"create":1})#x] +- OneRowRelation -- !query select from_json('{"cube":1}', 'cube INT') -- !query analysis -Project [from_json(StructField(cube,IntegerType,true), {"cube":1}, Some(America/Los_Angeles), false) AS from_json({"cube":1})#x] +Project [from_json(StructField(cube,IntegerType,true), {"cube":1}, Some(America/Los_Angeles), false, false) AS from_json({"cube":1})#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/sql-session-variables.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/sql-session-variables.sql.out index f22d06642b739..787c70a1dd7c7 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/sql-session-variables.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/sql-session-variables.sql.out @@ -2677,7 +2677,7 @@ CreateVariable default(cast(a INT as string), sql=''a INT''), true -- !query SELECT from_json('{"a": 1}', var1) -- !query analysis -Project [from_json(StructField(a,IntegerType,true), {"a": 1}, Some(America/Los_Angeles), false) AS from_json({"a": 1})#x] +Project [from_json(StructField(a,IntegerType,true), {"a": 1}, Some(America/Los_Angeles), false, false) AS from_json({"a": 1})#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/sql-udf.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/sql-udf.sql.out index e6e933bdeb33b..d387c3979a014 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/sql-udf.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/sql-udf.sql.out @@ -1297,7 +1297,7 @@ Project [spark_catalog.default.foo9n(a#x) AS spark_catalog.default.foo9n(array(1 SELECT foo9n(from_json('[1, 2, 3]', 'array')) -- !query analysis Project [spark_catalog.default.foo9n(a#x) AS spark_catalog.default.foo9n(from_json([1, 2, 3]))#x] -+- Project [cast(from_json(ArrayType(IntegerType,true), [1, 2, 3], Some(America/Los_Angeles), false) as array) AS a#x] ++- Project [cast(from_json(ArrayType(IntegerType,true), [1, 2, 3], Some(America/Los_Angeles), false, false) as array) AS a#x] +- OneRowRelation @@ -1319,7 +1319,7 @@ Project [spark_catalog.default.foo9o(a#x) AS spark_catalog.default.foo9o(map(hel SELECT foo9o(from_json('{"hello":1, "world":2}', 'map')) -- !query analysis Project [spark_catalog.default.foo9o(a#x) AS spark_catalog.default.foo9o(entries)#x] -+- Project [cast(from_json(MapType(StringType,IntegerType,true), {"hello":1, "world":2}, Some(America/Los_Angeles), false) as map) AS a#x] ++- Project [cast(from_json(MapType(StringType,IntegerType,true), {"hello":1, "world":2}, Some(America/Los_Angeles), false, false) as map) AS a#x] +- OneRowRelation @@ -1341,7 +1341,7 @@ Project [spark_catalog.default.foo9p(a#x) AS spark_catalog.default.foo9p(struct( SELECT foo9p(from_json('{1:"hello"}', 'struct')) -- !query analysis Project [spark_catalog.default.foo9p(a#x) AS spark_catalog.default.foo9p(from_json({1:"hello"}))#x] -+- Project [cast(from_json(StructField(a1,IntegerType,true), StructField(a2,StringType,true), {1:"hello"}, Some(America/Los_Angeles), false) as struct) AS a#x] ++- Project [cast(from_json(StructField(a1,IntegerType,true), StructField(a2,StringType,true), {1:"hello"}, Some(America/Los_Angeles), false, false) as struct) AS a#x] +- OneRowRelation @@ -1371,7 +1371,7 @@ Project [spark_catalog.default.foo9q(a#x) AS spark_catalog.default.foo9q(array(n SELECT foo9q(from_json('[{1:"hello"}, {2:"world"}]', 'array>')) -- !query analysis Project [spark_catalog.default.foo9q(a#x) AS spark_catalog.default.foo9q(from_json([{1:"hello"}, {2:"world"}]))#x] -+- Project [cast(from_json(ArrayType(StructType(StructField(a1,IntegerType,true),StructField(a2,StringType,true)),true), [{1:"hello"}, {2:"world"}], Some(America/Los_Angeles), false) as array>) AS a#x] ++- Project [cast(from_json(ArrayType(StructType(StructField(a1,IntegerType,true),StructField(a2,StringType,true)),true), [{1:"hello"}, {2:"world"}], Some(America/Los_Angeles), false, false) as array>) AS a#x] +- OneRowRelation @@ -1393,7 +1393,7 @@ Project [spark_catalog.default.foo9r(a#x) AS spark_catalog.default.foo9r(array(m SELECT foo9r(from_json('[{"hello":1}, {"world":2}]', 'array>')) -- !query analysis Project [spark_catalog.default.foo9r(a#x) AS spark_catalog.default.foo9r(from_json([{"hello":1}, {"world":2}]))#x] -+- Project [cast(from_json(ArrayType(MapType(StringType,IntegerType,true),true), [{"hello":1}, {"world":2}], Some(America/Los_Angeles), false) as array>) AS a#x] ++- Project [cast(from_json(ArrayType(MapType(StringType,IntegerType,true),true), [{"hello":1}, {"world":2}], Some(America/Los_Angeles), false, false) as array>) AS a#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/subexp-elimination.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/subexp-elimination.sql.out index d3da23ecca109..fcac55613194d 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/subexp-elimination.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/subexp-elimination.sql.out @@ -15,7 +15,7 @@ AS testData(a, b), false, true, LocalTempView, UNSUPPORTED, true -- !query SELECT from_json(a, 'struct').a, from_json(a, 'struct').b, from_json(b, 'array>')[0].a, from_json(b, 'array>')[0].b FROM testData -- !query analysis -Project [from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false).a AS from_json(a).a#x, from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false).b AS from_json(a).b#x, from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false)[0].a AS from_json(b)[0].a#x, from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false)[0].b AS from_json(b)[0].b#x] +Project [from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false, false).a AS from_json(a).a#x, from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false, false).b AS from_json(a).b#x, from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false, false)[0].a AS from_json(b)[0].a#x, from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false, false)[0].b AS from_json(b)[0].b#x] +- SubqueryAlias testdata +- View (`testData`, [a#x, b#x]) +- Project [cast(a#x as string) AS a#x, cast(b#x as string) AS b#x] @@ -27,7 +27,7 @@ Project [from_json(StructField(a,IntegerType,true), StructField(b,StringType,tru -- !query SELECT if(from_json(a, 'struct').a > 1, from_json(b, 'array>')[0].a, from_json(b, 'array>')[0].a + 1) FROM testData -- !query analysis -Project [if ((from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false).a > 1)) from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false)[0].a else (from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false)[0].a + 1) AS (IF((from_json(a).a > 1), from_json(b)[0].a, (from_json(b)[0].a + 1)))#x] +Project [if ((from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false, false).a > 1)) from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false, false)[0].a else (from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false, false)[0].a + 1) AS (IF((from_json(a).a > 1), from_json(b)[0].a, (from_json(b)[0].a + 1)))#x] +- SubqueryAlias testdata +- View (`testData`, [a#x, b#x]) +- Project [cast(a#x as string) AS a#x, cast(b#x as string) AS b#x] @@ -39,7 +39,7 @@ Project [if ((from_json(StructField(a,IntegerType,true), StructField(b,StringTyp -- !query SELECT if(isnull(from_json(a, 'struct').a), from_json(b, 'array>')[0].b + 1, from_json(b, 'array>')[0].b) FROM testData -- !query analysis -Project [if (isnull(from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false).a)) (from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false)[0].b + 1) else from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false)[0].b AS (IF((from_json(a).a IS NULL), (from_json(b)[0].b + 1), from_json(b)[0].b))#x] +Project [if (isnull(from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false, false).a)) (from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false, false)[0].b + 1) else from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false, false)[0].b AS (IF((from_json(a).a IS NULL), (from_json(b)[0].b + 1), from_json(b)[0].b))#x] +- SubqueryAlias testdata +- View (`testData`, [a#x, b#x]) +- Project [cast(a#x as string) AS a#x, cast(b#x as string) AS b#x] @@ -51,7 +51,7 @@ Project [if (isnull(from_json(StructField(a,IntegerType,true), StructField(b,Str -- !query SELECT case when from_json(a, 'struct').a > 5 then from_json(a, 'struct').b when from_json(a, 'struct').a > 4 then from_json(a, 'struct').b + 1 else from_json(a, 'struct').b + 2 end FROM testData -- !query analysis -Project [CASE WHEN (from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false).a > 5) THEN cast(from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false).b as bigint) WHEN (from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false).a > 4) THEN (cast(from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false).b as bigint) + cast(1 as bigint)) ELSE (cast(from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false).b as bigint) + cast(2 as bigint)) END AS CASE WHEN (from_json(a).a > 5) THEN from_json(a).b WHEN (from_json(a).a > 4) THEN (from_json(a).b + 1) ELSE (from_json(a).b + 2) END#xL] +Project [CASE WHEN (from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false, false).a > 5) THEN cast(from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false, false).b as bigint) WHEN (from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false, false).a > 4) THEN (cast(from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false, false).b as bigint) + cast(1 as bigint)) ELSE (cast(from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false, false).b as bigint) + cast(2 as bigint)) END AS CASE WHEN (from_json(a).a > 5) THEN from_json(a).b WHEN (from_json(a).a > 4) THEN (from_json(a).b + 1) ELSE (from_json(a).b + 2) END#xL] +- SubqueryAlias testdata +- View (`testData`, [a#x, b#x]) +- Project [cast(a#x as string) AS a#x, cast(b#x as string) AS b#x] @@ -63,7 +63,7 @@ Project [CASE WHEN (from_json(StructField(a,IntegerType,true), StructField(b,Str -- !query SELECT case when from_json(a, 'struct').a > 5 then from_json(b, 'array>')[0].b when from_json(a, 'struct').a > 4 then from_json(b, 'array>')[0].b + 1 else from_json(b, 'array>')[0].b + 2 end FROM testData -- !query analysis -Project [CASE WHEN (from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false).a > 5) THEN from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false)[0].b WHEN (from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false).a > 4) THEN (from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false)[0].b + 1) ELSE (from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false)[0].b + 2) END AS CASE WHEN (from_json(a).a > 5) THEN from_json(b)[0].b WHEN (from_json(a).a > 4) THEN (from_json(b)[0].b + 1) ELSE (from_json(b)[0].b + 2) END#x] +Project [CASE WHEN (from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false, false).a > 5) THEN from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false, false)[0].b WHEN (from_json(StructField(a,IntegerType,true), StructField(b,StringType,true), a#x, Some(America/Los_Angeles), false, false).a > 4) THEN (from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false, false)[0].b + 1) ELSE (from_json(ArrayType(StructType(StructField(a,IntegerType,true),StructField(b,IntegerType,true)),true), b#x, Some(America/Los_Angeles), false, false)[0].b + 2) END AS CASE WHEN (from_json(a).a > 5) THEN from_json(b)[0].b WHEN (from_json(a).a > 4) THEN (from_json(b)[0].b + 1) ELSE (from_json(b)[0].b + 2) END#x] +- SubqueryAlias testdata +- View (`testData`, [a#x, b#x]) +- Project [cast(a#x as string) AS a#x, cast(b#x as string) AS b#x] diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/timestamp.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/timestamp.sql.out index 95f861a0f5a8c..f6d88593a4488 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/timestamp.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/timestamp.sql.out @@ -1014,7 +1014,7 @@ Project [unix_timestamp(22 05 2020 Friday, dd MM yyyy EEEEE, Some(America/Los_An -- !query select from_json('{"t":"26/October/2015"}', 't Timestamp', map('timestampFormat', 'dd/MMMMM/yyyy')) -- !query analysis -Project [from_json(StructField(t,TimestampType,true), (timestampFormat,dd/MMMMM/yyyy), {"t":"26/October/2015"}, Some(America/Los_Angeles), false) AS from_json({"t":"26/October/2015"})#x] +Project [from_json(StructField(t,TimestampType,true), (timestampFormat,dd/MMMMM/yyyy), {"t":"26/October/2015"}, Some(America/Los_Angeles), false, false) AS from_json({"t":"26/October/2015"})#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/timestampNTZ/timestamp-ansi.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/timestampNTZ/timestamp-ansi.sql.out index 12f7e3a3b8d8e..6db51872317c1 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/timestampNTZ/timestamp-ansi.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/timestampNTZ/timestamp-ansi.sql.out @@ -1033,7 +1033,7 @@ Project [unix_timestamp(22 05 2020 Friday, dd MM yyyy EEEEE, Some(America/Los_An -- !query select from_json('{"t":"26/October/2015"}', 't Timestamp', map('timestampFormat', 'dd/MMMMM/yyyy')) -- !query analysis -Project [from_json(StructField(t,TimestampNTZType,true), (timestampFormat,dd/MMMMM/yyyy), {"t":"26/October/2015"}, Some(America/Los_Angeles), false) AS from_json({"t":"26/October/2015"})#x] +Project [from_json(StructField(t,TimestampNTZType,true), (timestampFormat,dd/MMMMM/yyyy), {"t":"26/October/2015"}, Some(America/Los_Angeles), false, false) AS from_json({"t":"26/October/2015"})#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/timestampNTZ/timestamp.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/timestampNTZ/timestamp.sql.out index 69c25c81dea00..afe3152de0801 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/timestampNTZ/timestamp.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/timestampNTZ/timestamp.sql.out @@ -1093,7 +1093,7 @@ Project [unix_timestamp(22 05 2020 Friday, dd MM yyyy EEEEE, Some(America/Los_An -- !query select from_json('{"t":"26/October/2015"}', 't Timestamp', map('timestampFormat', 'dd/MMMMM/yyyy')) -- !query analysis -Project [from_json(StructField(t,TimestampNTZType,true), (timestampFormat,dd/MMMMM/yyyy), {"t":"26/October/2015"}, Some(America/Los_Angeles), false) AS from_json({"t":"26/October/2015"})#x] +Project [from_json(StructField(t,TimestampNTZType,true), (timestampFormat,dd/MMMMM/yyyy), {"t":"26/October/2015"}, Some(America/Los_Angeles), false, false) AS from_json({"t":"26/October/2015"})#x] +- OneRowRelation diff --git a/sql/core/src/test/resources/sql-tests/analyzer-results/typeCoercion/native/stringCastAndExpressions.sql.out b/sql/core/src/test/resources/sql-tests/analyzer-results/typeCoercion/native/stringCastAndExpressions.sql.out index e57f803124ee3..a1e95e9c20b2b 100644 --- a/sql/core/src/test/resources/sql-tests/analyzer-results/typeCoercion/native/stringCastAndExpressions.sql.out +++ b/sql/core/src/test/resources/sql-tests/analyzer-results/typeCoercion/native/stringCastAndExpressions.sql.out @@ -370,7 +370,7 @@ Project [c0#x] -- !query select from_json(a, 'a INT') from t -- !query analysis -Project [from_json(StructField(a,IntegerType,true), a#x, Some(America/Los_Angeles), false) AS from_json(a)#x] +Project [from_json(StructField(a,IntegerType,true), a#x, Some(America/Los_Angeles), false, false) AS from_json(a)#x] +- SubqueryAlias t +- View (`t`, [a#x]) +- Project [cast(a#x as string) AS a#x] diff --git a/sql/core/src/test/scala/org/apache/spark/sql/CharVarcharTestSuite.scala b/sql/core/src/test/scala/org/apache/spark/sql/CharVarcharTestSuite.scala index c203fa1d8c344..f5491f8bacee5 100644 --- a/sql/core/src/test/scala/org/apache/spark/sql/CharVarcharTestSuite.scala +++ b/sql/core/src/test/scala/org/apache/spark/sql/CharVarcharTestSuite.scala @@ -1044,6 +1044,28 @@ class BasicCharVarcharTestSuite extends SharedSparkSession { parameters = Map("limit" -> expectedLimit)) } + private def assertUnsupportedJsonMapKey( + query: String, key: String, dataTypeSql: String): Unit = { + assertUnsupportedJsonMapKeyError(sql(query).collect(), key, dataTypeSql) + } + + private def assertUnsupportedJsonMapKeyError( + body: => Any, key: String, dataTypeSql: String): Unit = { + val error = intercept[Exception] { body } + val cause = Iterator.iterate[Throwable](error)(_.getCause) + .takeWhile(_ != null) + .collectFirst { + case e: SparkRuntimeException + if e.getCondition == "UNSUPPORTED_JSON_CHAR_VARCHAR_MAP_KEY" => e + } + .getOrElse(fail("expected UNSUPPORTED_JSON_CHAR_VARCHAR_MAP_KEY cause", error)) + checkError( + exception = cause, + condition = "UNSUPPORTED_JSON_CHAR_VARCHAR_MAP_KEY", + sqlState = "0A000", + parameters = Map("key" -> s"'$key'", "dataType" -> s""""$dataTypeSql"""")) + } + private def assertDuplicateMapKey(query: String, expectedKey: String = "a "): Unit = { assertDuplicateMapKeyError(sql(query).collect(), expectedKey) } @@ -3072,6 +3094,255 @@ class BasicCharVarcharTestSuite extends SharedSparkSession { } } + test("SPARK-60108: JSON CHAR/VARCHAR map keys are not padded or trimmed") { + withSQLConf(SQLConf.CHAR_VARCHAR_STANDARD_SEMANTICS.key -> "true") { + // Exact CHAR and within-limit VARCHAR names are accepted as-is: the trailing space in + // "ab " is preserved for CHAR(3) and not trimmed for VARCHAR(3). + checkAnswer( + sql("""SELECT from_json('{"ab ": 1}', 'MAP')"""), + Row(Map("ab " -> 1))) + checkAnswer( + sql("""SELECT from_json('{"ab": 1}', 'MAP')"""), + Row(Map("ab" -> 1))) + // Binary-distinct names ("ab" vs "ab ") are both kept. + checkAnswer( + sql("""SELECT map_entries(from_json('{"ab": 1, "ab ": 2}', 'MAP'))"""), + Row(Seq(Row("ab", 1), Row("ab ", 2)))) + // Duplicate names are kept, exactly as for STRING keys; mapKeyDedupPolicy is not applied. + // checkAnswer alone would collapse the duplicates when it builds a Scala Map, so pin the + // behavior with size and map_entries. + checkAnswer( + sql("""SELECT size(from_json('{"abc": 1, "abc": 2}', 'MAP'))"""), + Row(2)) + checkAnswer( + sql("""SELECT map_entries(from_json('{"abc": 1, "abc": 2}', 'MAP'))"""), + Row(Seq(Row("abc", 1), Row("abc", 2)))) + // Length counts characters (code points), not UTF-16 units or bytes: one non-BMP code + // point is exactly CHAR(1) and is kept verbatim, and the same name is rejected by CHAR(2). + // scalastyle:off nonascii (the \u escape is decoded before the nonascii check runs) + val nonBmp = "\uD83D\uDE00" // U+1F600, one code point stored as a UTF-16 surrogate pair + // scalastyle:on nonascii + checkAnswer( + sql(s"""SELECT map_keys(from_json('{"$nonBmp": 1}', 'MAP'))[0]"""), + Row(nonBmp)) + checkAnswer(sql(s"""SELECT from_json('{"$nonBmp": 1}', 'MAP')"""), Row(null)) + + // Length-0 boundary: CHAR(0)/VARCHAR(0) accept only the empty-string name, and any + // non-empty name is rejected (the empty key is still kept as a real map entry). + checkAnswer( + sql("""SELECT from_json('{"": 1}', 'MAP')"""), + Row(Map("" -> 1))) + checkAnswer( + sql("""SELECT from_json('{"": 1}', 'MAP')"""), + Row(Map("" -> 1))) + checkAnswer(sql("""SELECT from_json('{"a": 1}', 'MAP')"""), Row(null)) + checkAnswer(sql("""SELECT from_json('{"a": 1}', 'MAP')"""), Row(null)) + assertUnsupportedJsonMapKey( + """SELECT from_json('{"a": 1}', 'MAP', map('mode', 'FAILFAST'))""", + key = "a", + dataTypeSql = "CHAR(0)") + assertUnsupportedJsonMapKey( + """SELECT from_json('{"a": 1}', 'MAP', map('mode', 'FAILFAST'))""", + key = "a", + dataTypeSql = "VARCHAR(0)") + + // A key longer than the 128-character echo cap is truncated with a trailing "..." in the + // error so a huge JSON object name does not bloat the message. + val longKey = "a" * 200 + assertUnsupportedJsonMapKey( + s"""SELECT from_json('{"$longKey": 1}', 'MAP', map('mode', 'FAILFAST'))""", + key = "a" * 128 + "...", + dataTypeSql = "CHAR(3)") + + // Valid keys inside nested containers. + checkAnswer( + sql("""SELECT from_json('{"outer": {"xy ": 1}}', + | 'MAP>')""".stripMargin), + Row(Map("outer" -> Map("xy " -> 1)))) + checkAnswer( + sql("""SELECT from_json('[{"ab ": 1}]', 'ARRAY>')"""), + Row(Seq(Map("ab " -> 1)))) + + // A rejected key makes the row a bad record: null in PERMISSIVE (not EXCEED_LIMIT_LENGTH), + // and in FAILFAST the UNSUPPORTED_JSON_CHAR_VARCHAR_MAP_KEY cause is surfaced. + checkAnswer(sql("""SELECT from_json('{"a": 1}', 'MAP')"""), Row(null)) + checkAnswer(sql("""SELECT from_json('{"abcd": 1}', 'MAP')"""), Row(null)) + checkAnswer(sql("""SELECT from_json('{"abcd": 1}', 'MAP')"""), Row(null)) + checkAnswer(sql("""SELECT from_json('{"ab ": 1}', 'MAP')"""), Row(null)) + assertUnsupportedJsonMapKey( + """SELECT from_json('{"a": 1}', 'MAP', map('mode', 'FAILFAST'))""", + key = "a", + dataTypeSql = "CHAR(3)") + assertUnsupportedJsonMapKey( + """SELECT from_json('{"abcd": 1}', 'MAP', map('mode', 'FAILFAST'))""", + key = "abcd", + dataTypeSql = "CHAR(3)") + assertUnsupportedJsonMapKey( + """SELECT from_json('{"abcd": 1}', 'MAP', map('mode', 'FAILFAST'))""", + key = "abcd", + dataTypeSql = "VARCHAR(3)") + // A rejected key inside a nested container is surfaced too (not swallowed). + assertUnsupportedJsonMapKey( + """SELECT from_json('{"outer": {"a": 1}}', + | 'MAP>', map('mode', 'FAILFAST'))""".stripMargin, + key = "a", + dataTypeSql = "CHAR(3)") + assertUnsupportedJsonMapKey( + """SELECT from_json('[{"a": 1}]', + | 'ARRAY>', map('mode', 'FAILFAST'))""".stripMargin, + key = "a", + dataTypeSql = "CHAR(3)") + + // The DataFrame functions.from_json path constructs JsonToStructs the same way and must + // behave identically: accepted key kept, rejected key null in PERMISSIVE and surfaced in + // FAILFAST. + checkAnswer( + spark.range(1).select( + functions.from_json( + functions.lit("""{"ab ": 1}"""), "MAP", Map.empty[String, String])), + Row(Map("ab " -> 1))) + checkAnswer( + spark.range(1).select( + functions.from_json( + functions.lit("""{"a": 1}"""), "MAP", Map.empty[String, String])), + Row(null)) + assertUnsupportedJsonMapKeyError( + spark.range(1).select( + functions.from_json( + functions.lit("""{"a": 1}"""), + "MAP", + Map("mode" -> "FAILFAST"))).collect(), + key = "a", + dataTypeSql = "CHAR(3)") + + // SPARK-60108: a rejected key must not corrupt a sibling field of an enclosing struct. + // The map becomes null but the following `tail` field is still parsed, and an inner name + // ("tail") is not mistaken for the outer field. This holds in both partial-results modes. + Seq(true, false).foreach { partial => + withSQLConf(SQLConf.JSON_ENABLE_PARTIAL_RESULTS.key -> partial.toString) { + val schema = "m MAP, tail INT" + checkAnswer( + sql(s"""SELECT from_json('{"m":{"a":1},"tail":1}', '$schema')"""), + Row(Row(null, 1))) + checkAnswer( + sql(s"""SELECT from_json('{"m":{"a":1,"tail":5},"tail":1}', '$schema')"""), + Row(Row(null, 1))) + assertUnsupportedJsonMapKey( + s"""SELECT from_json('{"m":{"a":1},"tail":1}', '$schema', + | map('mode', 'FAILFAST'))""".stripMargin, + key = "a", + dataTypeSql = "CHAR(3)") + } + } + + // A rejected key whose value is an object or array exercises skipChildren(): the entire + // nested value is consumed so the map is nulled and the sibling `tail` field survives, and + // in FAILFAST the key is still surfaced. This holds in both partial-results modes. + Seq(true, false).foreach { partial => + withSQLConf(SQLConf.JSON_ENABLE_PARTIAL_RESULTS.key -> partial.toString) { + checkAnswer( + sql("""SELECT from_json('{"m":{"a":{"x":1}},"tail":2}', + | 'm MAP>, tail INT')""".stripMargin), + Row(Row(null, 2))) + checkAnswer( + sql("""SELECT from_json('{"m":{"a":[1,2,3]},"tail":2}', + | 'm MAP>, tail INT')""".stripMargin), + Row(Row(null, 2))) + assertUnsupportedJsonMapKey( + """SELECT from_json('{"a":{"x":1}}', + | 'MAP>', map('mode', 'FAILFAST'))""".stripMargin, + key = "a", + dataTypeSql = "CHAR(3)") + assertUnsupportedJsonMapKey( + """SELECT from_json('{"a":[1,2,3]}', + | 'MAP>', map('mode', 'FAILFAST'))""".stripMargin, + key = "a", + dataTypeSql = "CHAR(3)") + } + } + + // multiLine + streaming top-level array: a bad key in one element must not add spurious + // rows. Two elements in, two rows out, with the bad element reported in _corrupt_record. + withTempPath { file => + val document = """[{"m":{"a":1}},{"m":{"abc":2}}]""" + java.nio.file.Files.write(file.toPath, + document.getBytes(java.nio.charset.StandardCharsets.UTF_8)) + val df = spark.read + .schema("m MAP, _corrupt_record STRING") + .option("multiLine", true) + .option("enableStreamingTopLevelArray", true) + .json(file.getCanonicalPath) + checkAnswer(df, Seq(Row(null, document), Row(Map("abc" -> 2), null))) + } + + withTempPath { path => + Seq("""{"m":{"ab ":1}}""", """{"m":{"a":1}}""").toDS() + .repartition(1) + .write.text(path.getCanonicalPath) + val schema = "m MAP" + val permissive = spark.read.schema(schema).json(path.getCanonicalPath) + checkAnswer(permissive, Seq(Row(Map("ab " -> 1)), Row(null))) + val failFast = spark.read.option("mode", "FAILFAST").schema(schema) + .json(path.getCanonicalPath) + assertUnsupportedJsonMapKeyError(failFast.collect(), key = "a", dataTypeSql = "CHAR(3)") + } + + // mapKeyDedupPolicy is not applied: EXCEPTION does not raise DUPLICATED_MAP_KEY, the + // duplicate names are simply kept (contrast XML, which pads keys and applies the policy). + withSQLConf( + SQLConf.MAP_KEY_DEDUP_POLICY.key -> SQLConf.MapKeyDedupPolicy.EXCEPTION.toString) { + checkAnswer( + sql("""SELECT map_entries(from_json('{"abc": 1, "abc": 2}', 'MAP'))"""), + Row(Seq(Row("abc", 1), Row("abc", 2)))) + } + + // Collation is not consulted: binary-distinct names are both kept, and the collation is + // reported in the error message rather than dropped. + withSQLConf(SQLConf.ALLOW_COLLATIONS_IN_MAP_KEYS.key -> "true") { + checkAnswer( + sql("""SELECT map_entries(from_json('{"ab": 1, "AB": 2}', + | 'MAP'))""".stripMargin), + Row(Seq(Row("ab", 1), Row("AB", 2)))) + assertUnsupportedJsonMapKey( + """SELECT from_json('{"abc": 1}', + | 'MAP', map('mode', 'FAILFAST'))""".stripMargin, + key = "abc", + dataTypeSql = "VARCHAR(2) COLLATE UTF8_LCASE") + } + } + + // The flag has PERSISTED binding, so a view created under standard semantics keeps the + // key check when queried from a session with the flag off (SPARK-60108, review follow-up). + withView("v_spark_60108") { + withSQLConf(SQLConf.CHAR_VARCHAR_STANDARD_SEMANTICS.key -> "true") { + sql("""CREATE VIEW v_spark_60108 AS + |SELECT from_json('{"a": 1}', 'MAP', map('mode', 'FAILFAST')) AS m""" + .stripMargin) + } + withSQLConf(SQLConf.CHAR_VARCHAR_STANDARD_SEMANTICS.key -> "false") { + assertUnsupportedJsonMapKeyError( + sql("SELECT * FROM v_spark_60108").collect(), key = "a", dataTypeSql = "CHAR(3)") + } + } + + // With the flag off the new key check is not applied: a name that FAILFAST rejects under + // standard semantics is accepted, and from_json leaves the raw object name in place. (Any + // CHAR padding seen when the result is collected comes from the CHAR encoder, not from_json, + // so to_json is used here to observe exactly what from_json produced.) + withSQLConf( + SQLConf.CHAR_VARCHAR_STANDARD_SEMANTICS.key -> "false", + SQLConf.PRESERVE_CHAR_VARCHAR_TYPE_INFO.key -> "true") { + checkAnswer( + sql("""SELECT to_json(from_json('{"a": 1}', + | 'MAP', map('mode', 'FAILFAST')))""".stripMargin), + Row("""{"a":1}""")) + checkAnswer( + sql("""SELECT to_json(from_json('{"abcd": 1}', + | 'MAP', map('mode', 'FAILFAST')))""".stripMargin), + Row("""{"abcd":1}""")) + } + } + test("SPARK-59274: JSON map value overflow keeps EXCEED_LIMIT_LENGTH") { withSQLConf(SQLConf.CHAR_VARCHAR_STANDARD_SEMANTICS.key -> "true") { val json = """{"m":{"k":"abcdef"},"tail":1}"""