MartijnVisser commented on code in PR #29313:
URL: https://github.com/apache/flink/pull/29313#discussion_r4129863303
##########
flink-table/flink-table-common/src/main/java/org/apache/flink/table/data/binary/StringUtf8Utils.java:
##########
@@ -18,117 +18,25 @@
package org.apache.flink.table.data.binary;
import org.apache.flink.annotation.Internal;
-import org.apache.flink.core.memory.MemorySegment;
-import java.io.UnsupportedEncodingException;
import java.nio.charset.StandardCharsets;
-import java.util.Arrays;
-
-import static
org.apache.flink.table.data.binary.BinarySegmentUtils.allocateReuseBytes;
-import static
org.apache.flink.table.data.binary.BinarySegmentUtils.allocateReuseChars;
/** Utilities for String UTF-8. */
@Internal
public final class StringUtf8Utils {
- private static final int MAX_BYTES_PER_CHAR = 3;
-
private StringUtf8Utils() {
// do not instantiate
}
/** This method must have the same result with JDK's String.getBytes. */
public static byte[] encodeUTF8(String str) {
- byte[] bytes = allocateReuseBytes(str.length() * MAX_BYTES_PER_CHAR);
- int len = encodeUTF8(str, bytes);
- return Arrays.copyOf(bytes, len);
- }
-
- public static int encodeUTF8(String str, byte[] bytes) {
- int offset = 0;
- int len = str.length();
- int sl = offset + len;
- int dp = 0;
- int dlASCII = dp + Math.min(len, bytes.length);
-
- // ASCII only optimized loop
- while (dp < dlASCII && str.charAt(offset) < '\u0080') {
- bytes[dp++] = (byte) str.charAt(offset++);
- }
-
- while (offset < sl) {
- char c = str.charAt(offset++);
- if (c < 0x80) {
- // Have at most seven bits
- bytes[dp++] = (byte) c;
- } else if (c < 0x800) {
- // 2 bytes, 11 bits
- bytes[dp++] = (byte) (0xc0 | (c >> 6));
- bytes[dp++] = (byte) (0x80 | (c & 0x3f));
- } else if (Character.isSurrogate(c)) {
- final int uc;
- int ip = offset - 1;
- if (Character.isHighSurrogate(c)) {
- if (sl - ip < 2) {
- uc = -1;
- } else {
- char d = str.charAt(ip + 1);
- if (Character.isLowSurrogate(d)) {
- uc = Character.toCodePoint(c, d);
- } else {
- // for some illegal character
- // the jdk will ignore the origin character and
cast it to '?'
- // this acts the same with jdk
- return defaultEncodeUTF8(str, bytes);
- }
- }
- } else {
- if (Character.isLowSurrogate(c)) {
- // for some illegal character
- // the jdk will ignore the origin character and cast
it to '?'
- // this acts the same with jdk
- return defaultEncodeUTF8(str, bytes);
- } else {
- uc = c;
- }
- }
-
- if (uc < 0) {
- bytes[dp++] = (byte) '?';
- } else {
- bytes[dp++] = (byte) (0xf0 | ((uc >> 18)));
- bytes[dp++] = (byte) (0x80 | ((uc >> 12) & 0x3f));
- bytes[dp++] = (byte) (0x80 | ((uc >> 6) & 0x3f));
- bytes[dp++] = (byte) (0x80 | (uc & 0x3f));
- offset++; // 2 chars
- }
- } else {
- // 3 bytes, 16 bits
- bytes[dp++] = (byte) (0xe0 | ((c >> 12)));
- bytes[dp++] = (byte) (0x80 | ((c >> 6) & 0x3f));
- bytes[dp++] = (byte) (0x80 | (c & 0x3f));
- }
- }
- return dp;
- }
-
- public static int defaultEncodeUTF8(String str, byte[] bytes) {
- try {
- byte[] buffer = str.getBytes("UTF-8");
- System.arraycopy(buffer, 0, bytes, 0, buffer.length);
- return buffer.length;
- } catch (UnsupportedEncodingException e) {
- throw new RuntimeException("encodeUTF8 error", e);
- }
+ return str.getBytes(StandardCharsets.UTF_8);
}
public static String decodeUTF8(byte[] input, int offset, int byteLen) {
- char[] chars = allocateReuseChars(byteLen);
- int len = decodeUTF8Strict(input, offset, byteLen, chars);
- if (len < 0) {
- return defaultDecodeUTF8(input, offset, byteLen);
- }
- return new String(chars, 0, len);
+ // JDK-intrinsified decode; maps malformed input to U+FFFD (matches
the previous fallback).
+ return new String(input, offset, byteLen, StandardCharsets.UTF_8);
Review Comment:
JMH on the codec alone: Latin-1 decodes at 0.63x of today on JDK 11 and 17,
short Latin-1 and CJK strings at 0.3x to 0.6x on 21 too, ASCII 1.5x to 8x
faster.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]