snuyanzin commented on code in PR #29313:
URL: https://github.com/apache/flink/pull/29313#discussion_r4152617722


##########
flink-table/flink-table-runtime/src/main/java/org/apache/flink/table/runtime/util/StringUtf8Utils.java:
##########
@@ -35,100 +33,14 @@
  */
 public class StringUtf8Utils {
 
-    private static final int MAX_BYTES_PER_CHAR = 3;
-
     /** This method must have the same result with JDK's String.getBytes. */
     public static byte[] encodeUTF8(String str) {
-        byte[] bytes = allocateReuseBytes(str.length() * MAX_BYTES_PER_CHAR);
-        int len = encodeUTF8(str, bytes);
-        return Arrays.copyOf(bytes, len);
-    }
-
-    public static int encodeUTF8(String str, byte[] bytes) {
-        int offset = 0;
-        int len = str.length();
-        int sl = offset + len;
-        int dp = 0;
-        int dlASCII = dp + Math.min(len, bytes.length);
-
-        // ASCII only optimized loop
-        while (dp < dlASCII && str.charAt(offset) < '\u0080') {
-            bytes[dp++] = (byte) str.charAt(offset++);
-        }
-
-        while (offset < sl) {
-            char c = str.charAt(offset++);
-            if (c < 0x80) {
-                // Have at most seven bits
-                bytes[dp++] = (byte) c;
-            } else if (c < 0x800) {
-                // 2 bytes, 11 bits
-                bytes[dp++] = (byte) (0xc0 | (c >> 6));
-                bytes[dp++] = (byte) (0x80 | (c & 0x3f));
-            } else if (Character.isSurrogate(c)) {
-                final int uc;
-                int ip = offset - 1;
-                if (Character.isHighSurrogate(c)) {
-                    if (sl - ip < 2) {
-                        uc = -1;
-                    } else {
-                        char d = str.charAt(ip + 1);
-                        if (Character.isLowSurrogate(d)) {
-                            uc = Character.toCodePoint(c, d);
-                        } else {
-                            // for some illegal character
-                            // the jdk will ignore the origin character and 
cast it to '?'
-                            // this acts the same with jdk
-                            return defaultEncodeUTF8(str, bytes);
-                        }
-                    }
-                } else {
-                    if (Character.isLowSurrogate(c)) {
-                        // for some illegal character
-                        // the jdk will ignore the origin character and cast 
it to '?'
-                        // this acts the same with jdk
-                        return defaultEncodeUTF8(str, bytes);
-                    } else {
-                        uc = c;
-                    }
-                }
-
-                if (uc < 0) {
-                    bytes[dp++] = (byte) '?';
-                } else {
-                    bytes[dp++] = (byte) (0xf0 | ((uc >> 18)));
-                    bytes[dp++] = (byte) (0x80 | ((uc >> 12) & 0x3f));
-                    bytes[dp++] = (byte) (0x80 | ((uc >> 6) & 0x3f));
-                    bytes[dp++] = (byte) (0x80 | (uc & 0x3f));
-                    offset++; // 2 chars
-                }
-            } else {
-                // 3 bytes, 16 bits
-                bytes[dp++] = (byte) (0xe0 | ((c >> 12)));
-                bytes[dp++] = (byte) (0x80 | ((c >> 6) & 0x3f));
-                bytes[dp++] = (byte) (0x80 | (c & 0x3f));
-            }
-        }
-        return dp;
-    }
-
-    public static int defaultEncodeUTF8(String str, byte[] bytes) {
-        try {
-            byte[] buffer = str.getBytes("UTF-8");
-            System.arraycopy(buffer, 0, bytes, 0, buffer.length);
-            return buffer.length;
-        } catch (UnsupportedEncodingException e) {
-            throw new RuntimeException("encodeUTF8 error", e);
-        }
+        return str.getBytes(StandardCharsets.UTF_8);
     }
 
     public static String decodeUTF8(byte[] input, int offset, int byteLen) {
-        char[] chars = allocateReuseChars(byteLen);
-        int len = decodeUTF8Strict(input, offset, byteLen, chars);
-        if (len < 0) {
-            return defaultDecodeUTF8(input, offset, byteLen);
-        }
-        return new String(chars, 0, len);
+        // JDK-intrinsified decode; maps malformed input to U+FFFD (matches 
the previous fallback).
+        return new String(input, offset, byteLen, StandardCharsets.UTF_8);

Review Comment:
   yes, reuse makes sense...
   
   however in long term probably need to merge these 2 classes, no idea why we 
need both 🤷 



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]

Reply via email to