gene-db commented on code in PR #57827:
URL: https://github.com/apache/spark/pull/57827#discussion_r3906927324


##########
sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/expressions/variant/variantExpressions.scala:
##########
@@ -1603,6 +1659,74 @@ object VariantExplode {
       case _ => Nil
     }
   }
+
+  private case class ExplodeEntry(path: String, pos: Int, key: UTF8String, 
value: Variant)
+
+  def variantExplodeRecursive(
+      input: VariantVal,
+      isNull: Boolean): Iterable[InternalRow] = {
+    if (isNull) {
+      return Iterable.empty
+    }
+
+    new Iterable[InternalRow] {
+      override def iterator: Iterator[InternalRow] = {
+        val stack = new ArrayDeque[ExplodeEntry]()
+        pushChildren(new Variant(input.getValue, input.getMetadata), "$", 
stack)
+        new Iterator[InternalRow] {
+          override def hasNext: Boolean = !stack.isEmpty
+
+          override def next(): InternalRow = {
+            val entry = stack.pop()
+            pushChildren(entry.value, entry.path, stack)
+            InternalRow(
+              UTF8String.fromString(entry.path),
+              entry.pos,
+              entry.key,
+              new VariantVal(entry.value.getValue, entry.value.getMetadata))
+          }
+        }
+      }
+    }
+  }
+
+  private def pushChildren(
+      v: Variant,
+      parentPath: String,
+      stack: ArrayDeque[ExplodeEntry]): Unit = {
+    v.getType match {
+      case Type.OBJECT =>
+        for (i <- v.objectSize() - 1 to 0 by -1) {
+          val field = v.getFieldAtIndex(i)
+          stack.push(ExplodeEntry(
+            appendObjectPath(parentPath, field.key),
+            i,
+            UTF8String.fromString(field.key),
+            field.value))
+        }
+      case Type.ARRAY =>
+        for (i <- v.arraySize() - 1 to 0 by -1) {
+          stack.push(ExplodeEntry(
+            s"$parentPath[$i]",
+            i,
+            null,
+            v.getElementAtIndex(i)))
+        }
+      case _ =>
+    }
+  }
+
+  private def appendObjectPath(parentPath: String, key: String): String = {
+    if (key.nonEmpty && !key.contains('.') && !key.contains('[')) {
+      s"$parentPath.$key"
+    } else if (!key.contains('"')) {
+      s"""$parentPath["$key"]"""
+    } else if (!key.contains('\'')) {
+      s"$parentPath['$key']"
+    } else {
+      s"""$parentPath["${StringEscapeUtils.escapeJson(key)}"]"""

Review Comment:
   Is the escaped version the correct canonicalized json path? If so, maybe we 
need to update the json parser to correctly parse all valid canonicalized json 
paths.



##########
sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/expressions/variant/variantExpressions.scala:
##########
@@ -1851,6 +1907,74 @@ object VariantExplode {
       case _ => Nil
     }
   }
+
+  private case class ExplodeEntry(path: String, pos: Int, key: UTF8String, 
value: Variant)
+
+  def variantExplodeRecursive(
+      input: VariantVal,
+      isNull: Boolean): Iterable[InternalRow] = {
+    if (isNull) {
+      return Iterable.empty
+    }
+
+    new Iterable[InternalRow] {
+      override def iterator: Iterator[InternalRow] = {
+        val stack = new ArrayDeque[ExplodeEntry]()
+        pushChildren(new Variant(input.getValue, input.getMetadata), "$", 
stack)
+        new Iterator[InternalRow] {
+          override def hasNext: Boolean = !stack.isEmpty
+
+          override def next(): InternalRow = {
+            val entry = stack.pop()
+            pushChildren(entry.value, entry.path, stack)
+            InternalRow(
+              UTF8String.fromString(entry.path),
+              entry.pos,
+              entry.key,
+              new VariantVal(entry.value.getValue, entry.value.getMetadata))
+          }
+        }
+      }
+    }
+  }
+
+  private def pushChildren(
+      v: Variant,
+      parentPath: String,
+      stack: ArrayDeque[ExplodeEntry]): Unit = {
+    v.getType match {
+      case Type.OBJECT =>
+        for (i <- v.objectSize() - 1 to 0 by -1) {
+          val field = v.getFieldAtIndex(i)
+          stack.push(ExplodeEntry(
+            appendObjectPath(parentPath, field.key),

Review Comment:
   Does this mean different parts of the paths might be in dot notation or 
canonical notation? Do we have tests for all these strange cases with 
interesting characters in the paths?



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to