snuyanzin commented on code in PR #29313:
URL: https://github.com/apache/flink/pull/29313#discussion_r4133862283
##########
flink-table/flink-table-common/src/main/java/org/apache/flink/table/data/binary/StringUtf8Utils.java:
##########
@@ -18,117 +18,25 @@
package org.apache.flink.table.data.binary;
import org.apache.flink.annotation.Internal;
-import org.apache.flink.core.memory.MemorySegment;
-import java.io.UnsupportedEncodingException;
import java.nio.charset.StandardCharsets;
-import java.util.Arrays;
-
-import static
org.apache.flink.table.data.binary.BinarySegmentUtils.allocateReuseBytes;
-import static
org.apache.flink.table.data.binary.BinarySegmentUtils.allocateReuseChars;
/** Utilities for String UTF-8. */
@Internal
public final class StringUtf8Utils {
- private static final int MAX_BYTES_PER_CHAR = 3;
-
private StringUtf8Utils() {
// do not instantiate
}
/** This method must have the same result with JDK's String.getBytes. */
public static byte[] encodeUTF8(String str) {
- byte[] bytes = allocateReuseBytes(str.length() * MAX_BYTES_PER_CHAR);
- int len = encodeUTF8(str, bytes);
- return Arrays.copyOf(bytes, len);
- }
-
- public static int encodeUTF8(String str, byte[] bytes) {
- int offset = 0;
- int len = str.length();
- int sl = offset + len;
- int dp = 0;
- int dlASCII = dp + Math.min(len, bytes.length);
-
- // ASCII only optimized loop
- while (dp < dlASCII && str.charAt(offset) < '\u0080') {
- bytes[dp++] = (byte) str.charAt(offset++);
- }
-
- while (offset < sl) {
- char c = str.charAt(offset++);
- if (c < 0x80) {
- // Have at most seven bits
- bytes[dp++] = (byte) c;
- } else if (c < 0x800) {
- // 2 bytes, 11 bits
- bytes[dp++] = (byte) (0xc0 | (c >> 6));
- bytes[dp++] = (byte) (0x80 | (c & 0x3f));
- } else if (Character.isSurrogate(c)) {
- final int uc;
- int ip = offset - 1;
- if (Character.isHighSurrogate(c)) {
- if (sl - ip < 2) {
- uc = -1;
- } else {
- char d = str.charAt(ip + 1);
- if (Character.isLowSurrogate(d)) {
- uc = Character.toCodePoint(c, d);
- } else {
- // for some illegal character
- // the jdk will ignore the origin character and
cast it to '?'
- // this acts the same with jdk
- return defaultEncodeUTF8(str, bytes);
- }
- }
- } else {
- if (Character.isLowSurrogate(c)) {
- // for some illegal character
- // the jdk will ignore the origin character and cast
it to '?'
- // this acts the same with jdk
- return defaultEncodeUTF8(str, bytes);
- } else {
- uc = c;
- }
- }
-
- if (uc < 0) {
- bytes[dp++] = (byte) '?';
- } else {
- bytes[dp++] = (byte) (0xf0 | ((uc >> 18)));
- bytes[dp++] = (byte) (0x80 | ((uc >> 12) & 0x3f));
- bytes[dp++] = (byte) (0x80 | ((uc >> 6) & 0x3f));
- bytes[dp++] = (byte) (0x80 | (uc & 0x3f));
- offset++; // 2 chars
- }
- } else {
- // 3 bytes, 16 bits
- bytes[dp++] = (byte) (0xe0 | ((c >> 12)));
- bytes[dp++] = (byte) (0x80 | ((c >> 6) & 0x3f));
- bytes[dp++] = (byte) (0x80 | (c & 0x3f));
- }
- }
- return dp;
- }
-
- public static int defaultEncodeUTF8(String str, byte[] bytes) {
- try {
- byte[] buffer = str.getBytes("UTF-8");
- System.arraycopy(buffer, 0, bytes, 0, buffer.length);
- return buffer.length;
- } catch (UnsupportedEncodingException e) {
- throw new RuntimeException("encodeUTF8 error", e);
- }
+ return str.getBytes(StandardCharsets.UTF_8);
}
public static String decodeUTF8(byte[] input, int offset, int byteLen) {
- char[] chars = allocateReuseChars(byteLen);
- int len = decodeUTF8Strict(input, offset, byteLen, chars);
- if (len < 0) {
- return defaultDecodeUTF8(input, offset, byteLen);
- }
- return new String(chars, 0, len);
+ // JDK-intrinsified decode; maps malformed input to U+FFFD (matches
the previous fallback).
+ return new String(input, offset, byteLen, StandardCharsets.UTF_8);
Review Comment:
thanks for the feedback
that's true
however I found another way: I added SWAR check and depending on result it
goes ascii or non ascii way.
I this case there is regression for ascii for very short strings (1-5
chars), then similar for ~ 10 and after 50 there is noticeable improvement
for non ascii timings are similar to before changes
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]