voonhous commented on code in PR #19657:
URL: https://github.com/apache/hudi/pull/19657#discussion_r3940321894


##########
hudi-spark-datasource/hudi-spark/src/test/scala/org/apache/hudi/functional/TestCOWDataSource.scala:
##########
@@ -1450,6 +1450,73 @@ class TestCOWDataSource extends 
HoodieSparkClientTestBase with ScalaAssertionSup
     assertTrue(recordsReadDF.filter(col("_hoodie_partition_path") =!= 
udf_date_format(col("current_ts"))).count() == 0)
   }
 
+  @ParameterizedTest
+  @EnumSource(value = classOf[HoodieRecordType], names = Array("AVRO", 
"SPARK"))
+  def testTimestampBasedKeyGeneratorWithVariousConfigurations(recordType: 
HoodieRecordType) {
+    val (writeOpts, readOpts) = 
getWriterReaderOptsLessPartitionPath(recordType)
+
+    val records = recordsToStrings(dataGen.generateInserts("000", 
100)).asScala.toList
+    val inputDF = spark.read.json(spark.sparkContext.parallelize(records, 2))
+      .withColumn("current_ts_micros", col("current_ts") * 1000)
+      .withColumn("current_date_string",
+        date_format((col("current_ts") / 1000).cast("timestamp"), "yyyy-MM-dd 
HH:mm:ss"))
+      .withColumn("current_ts_hours", (col("current_ts") / 
3600000).cast("long"))
+      .withColumn("current_ts_seconds", (col("current_ts") / 
1000).cast("long"))
+
+    case class TestCase(partitionCol: String, tsType: String, outFmt: String,
+                        extraOpts: Map[String, String] = Map.empty,
+                        expectedPartitionUdf: 
org.apache.spark.sql.expressions.UserDefinedFunction)
+
+    def runTestCase(tc: TestCase): Unit = {
+      val writer = tc.extraOpts.foldLeft(
+        inputDF.write.format("hudi")
+          .options(writeOpts)
+          .option(KEYGENERATOR_CLASS_NAME.key(), 
classOf[TimestampBasedKeyGenerator].getName)
+          .mode(SaveMode.Overwrite)
+      ) { case (w, (k, v)) => w.option(k, v) }
+      writer.partitionBy(tc.partitionCol)
+        .option(TIMESTAMP_TYPE_FIELD.key, tc.tsType)
+        .option(TIMESTAMP_OUTPUT_DATE_FORMAT.key, tc.outFmt)
+        .save(basePath)
+      val readDF = 
spark.read.format("org.apache.hudi").options(readOpts).load(basePath)
+      assertTrue(readDF.filter(col("_hoodie_partition_path") =!= 
tc.expectedPartitionUdf(col(tc.partitionCol))).count() == 0)
+    }
+
+    // Test 1: EPOCHMILLISECONDS with timezone GMT+08:00
+    val tzMillisOutFmt = "yyyy-MM-dd HH"
+    val udfMillisTz = udf((millis: Long) =>
+      new 
DateTime(millis).withZone(org.joda.time.DateTimeZone.forID("GMT+08:00"))
+        
.toString(DateTimeFormat.forPattern(tzMillisOutFmt).withZone(org.joda.time.DateTimeZone.forID("GMT+08:00"))))
+    runTestCase(TestCase("current_ts", "EPOCHMILLISECONDS", tzMillisOutFmt,
+      Map(TIMESTAMP_TIMEZONE_FORMAT.key -> "GMT+08:00"), udfMillisTz))
+
+    // Test 2: EPOCHMICROSECONDS
+    val microsOutFmt = "yyyy-MM-dd HH"
+    val udfMicros = udf((micros: Long) =>
+      new DateTime(micros / 
1000).toString(DateTimeFormat.forPattern(microsOutFmt)))
+    runTestCase(TestCase("current_ts_micros", "EPOCHMICROSECONDS", 
microsOutFmt,
+      expectedPartitionUdf = udfMicros))
+
+    // Test 3: DATE_STRING with timezone
+    val dateStrOutFmt = "yyyy-MM-dd HH"
+    val dateStrInFmt = "yyyy-MM-dd HH:mm:ss"
+    val udfDateStrTz = udf((s: String) =>

Review Comment:
   The parse zone did move onto the input formatter in 45131da, which is what 
this asks for.
   
   It never ran, though: `DateTimeZone.forID("GMT+08:00")` throws 
`IllegalArgumentException: The datetime zone id 'GMT+08:00' is not recognised`. 
Joda's `forID` takes tz-db ids or bare offsets like `+08:00`, not `GMT+hh:mm`, 
so this UDF and the EPOCHMILLISECONDS one both blew up before the assertion. 
That was the Azure failure on 45131da.
   
   Fixed in 302bf99c by resolving the id the way `HoodieDateTimeParser` does, 
`DateTimeZone.forTimeZone(TimeZone.getTimeZone(id))`.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]

Reply via email to