ganeshashree commented on code in PR #58450:
URL: https://github.com/apache/spark/pull/58450#discussion_r4032829073
##########
sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/analysis/FunctionResolution.scala:
##########
@@ -404,6 +388,110 @@ class FunctionResolution(
}
}
+ /**
+ * Returns whether an unqualified function name reaches `system.builtin`
before any temp or
+ * persistent function in the effective SQL PATH. When a temp or persistent
function shadows the
+ * builtin, special-syntax handling that only applies to Spark's builtins
must not fire, since the
+ * name no longer refers to the builtin -- e.g. rejecting a direct star
(bare `*` or qualified
+ * `t.*`) in a routed SQL/JSON function or the `count(tbl.*)` guard.
Parser-built `count(*)` is
+ * normalized to `count(1)` in `AstBuilder` so it skips this probe, but a
DataFrame `count("*")`
+ * keeps its star and does reach the probe during analyzer normalization.
+ *
+ * Precondition: `functionName` must already be known to be a stock built-in
name (as
+ * `functionNameResolvesToBuiltin` ensures by checking
`FunctionRegistry.functionSet` first). This
+ * returns true as soon as the PATH reaches `system.builtin`, without
verifying that
+ * `system.builtin` actually defines a function of this name, so calling it
for a non-builtin name
+ * would wrongly report builtin ownership.
+ */
+ def unqualifiedFunctionResolvesToBuiltinBeforeAnyShadow(functionName:
String): Boolean = {
+ // Walk the PATH in order and stop at the first entry that owns the name.
The default order puts
+ // system.builtin first, so the common case returns on the first entry
with no catalog lookup;
+ // only a custom PATH that lists a persistent catalog ahead of
system.builtin reaches the probe
+ // below (one lookup per such preceding entry, recomputed on each call --
not cached).
+ sqlResolutionPathEntriesForAnalysis.foreach { pathEntry =>
+ val candidate = pathEntry :+ functionName
+ FunctionResolution.sessionNamespaceKind(candidate) match {
+ case
Some(org.apache.spark.sql.catalyst.catalog.SessionCatalog.Builtin) =>
+ return true
+ case Some(org.apache.spark.sql.catalyst.catalog.SessionCatalog.Temp) =>
+ // A visible temp scalar function shadows the builtin; a visible
temp *table* function
+ // makes scalar resolution terminal at this PATH entry
(NOT_A_SCALAR_FUNCTION). Either way
+ // the name never reaches system.builtin, mirroring
`resolveFunctionCandidate`.
+ val ident = FunctionIdentifier(functionName)
+ if (v1SessionCatalog.isTemporaryScalarFunctionVisible(ident) ||
+ v1SessionCatalog.isTemporaryTableFunctionVisible(ident)) {
+ return false
+ }
+ case None =>
+ if (persistentFunctionExists(candidate)) {
+ return false
+ }
+ }
+ }
+ false
+ }
+
+ /**
+ * Returns true when a function reference resolves to the system built-in
with the requested name.
+ * This mirrors [[resolveFunction]] for special parser/analyzer rewrites
that must run only for
+ * Spark's built-ins. In particular, two-part `builtin.name` is not always a
system built-in:
+ * with `spark.sql.legacy.persistentCatalogFirst=true`, an existing
persistent
+ * `current_catalog.builtin.name` takes precedence.
+ */
+ def functionNameResolvesToBuiltin(nameParts: Seq[String], expectedName:
String): Boolean = {
+ if (!FunctionRegistry.functionSet.contains(
+ FunctionRegistry.builtinFunctionIdentifier(expectedName)) ||
+ !FunctionResolution.isUnqualifiedOrBuiltinFunctionName(nameParts,
expectedName)) {
+ return false
+ }
+ nameParts.length match {
+ case 1 =>
+ unqualifiedFunctionResolvesToBuiltinBeforeAnyShadow(nameParts.head)
+ case 2 =>
+ conf.prioritizeSystemCatalog || !persistentFunctionExists(nameParts)
Review Comment:
Filed SPARK-59600 for the owner-binding follow-up.
##########
sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/parser/AstBuilder.scala:
##########
@@ -4289,38 +4321,42 @@ class AstBuilder extends DataTypeAstBuilder
*/
override def visitJsonArray(ctx: JsonArrayContext): Expression =
withOrigin(ctx) {
val arrayValues = ctx.values.asScala.map(v => expression(v.value)).toSeq
- // Freeze the FORMAT JSON decisions here, from the lexical argument, so a
later
- // analyzer/optimizer rewrite that wraps or swaps the child cannot change
them (see
- // [[ImplicitlyFormattedAsJson]]). For each element:
- // - `formatJson`: whether it is already-JSON text spliced raw. True when
it carries an
- // explicit `FORMAT JSON` clause, or is a (lexically) nested JSON
constructor -- seen through
- // a value-preserving `COLLATE` via `JsonArray.isImplicitlyJson`.
- // - `needsValidation`: whether its raw text is arbitrary user input to
JSON-validate at eval.
- // True only for an explicit `FORMAT JSON` on something that is NOT a
JSON constructor; a
- // nested constructor emits well-formed JSON by construction and is
trusted.
- val formatArgs = ctx.values.asScala.zip(arrayValues).map { case (v, expr)
=>
- val explicit = v.FORMAT() != null
- val implicitlyJson = JsonArray.isImplicitlyJson(expr)
- (explicit || implicitlyJson, explicit && !implicitlyJson)
- }.toSeq
- val formatJson = formatArgs.map(_._1)
- val needsValidation = formatArgs.map(_._2)
- // Default RETURNING type is STRING; the result is JSON text. A
CHAR/VARCHAR RETURNING is
- // normalized to STRING unconditionally: JSON_ARRAY serializes the
fragment itself and never
- // advertises a CHAR/VARCHAR length it does not enforce. The
CharVarcharUtils helpers cannot be
- // used here -- they honor spark.sql.preserveCharVarcharTypeInfo and would
leave a VARCHAR(n)
- // length in the output type when that flag is set. A non-string RETURNING
is left intact for
- // checkInputDataTypes to fail.
- val returning = Option(ctx.returning).map(typedVisit[DataType]).map {
- case c: CharType => c.toStringType
- case v: VarcharType => v.toStringType
- case other => other
- }.getOrElse(StringType)
- // Default ON NULL behavior is ABSENT ON NULL (drop NULL elements).
- val nullBehavior = Option(ctx.nullBehavior)
- .map(buildJsonConstructorNullBehavior)
- .getOrElse(JsonConstructorNullBehavior.Absent)
- JsonArray(arrayValues, formatJson, needsValidation, nullBehavior,
returning)
+ val hasExplicitFormat = ctx.values.asScala.exists(_.FORMAT() != null)
+ val hasImplicitJson = arrayValues.exists(JsonArray.isImplicitlyJson)
Review Comment:
Done.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]