snuyanzin commented on code in PR #29313:
URL: https://github.com/apache/flink/pull/29313#discussion_r4152612255
##########
flink-table/flink-table-common/src/main/java/org/apache/flink/table/data/binary/StringUtf8Utils.java:
##########
@@ -18,119 +18,162 @@
package org.apache.flink.table.data.binary;
import org.apache.flink.annotation.Internal;
-import org.apache.flink.core.memory.MemorySegment;
-import java.io.UnsupportedEncodingException;
+import java.lang.invoke.MethodHandles;
+import java.lang.invoke.VarHandle;
+import java.nio.ByteOrder;
import java.nio.charset.StandardCharsets;
-import java.util.Arrays;
-import static
org.apache.flink.table.data.binary.BinarySegmentUtils.allocateReuseBytes;
import static
org.apache.flink.table.data.binary.BinarySegmentUtils.allocateReuseChars;
/** Utilities for String UTF-8. */
@Internal
public final class StringUtf8Utils {
- private static final int MAX_BYTES_PER_CHAR = 3;
+ /** Reads 8 bytes at a time from a {@code byte[]} for the SWAR ASCII scan.
*/
+ private static final VarHandle LONG_VIEW =
+ MethodHandles.byteArrayViewVarHandle(long[].class,
ByteOrder.nativeOrder());
+
+ /** High bit of each byte in a 64-bit word; a set bit marks a non-ASCII
byte. */
+ private static final long ASCII_MASK = 0x8080808080808080L;
private StringUtf8Utils() {
// do not instantiate
}
/** This method must have the same result with JDK's String.getBytes. */
public static byte[] encodeUTF8(String str) {
- byte[] bytes = allocateReuseBytes(str.length() * MAX_BYTES_PER_CHAR);
- int len = encodeUTF8(str, bytes);
- return Arrays.copyOf(bytes, len);
+ return str.getBytes(StandardCharsets.UTF_8);
+ }
+
+ public static String decodeUTF8(byte[] input, int offset, int byteLen) {
+ // Most real text is ASCII: route it to the JDK's compact-string path,
which is a large win
+ // for longer strings. Anything with a multibyte sequence goes through
the hand-rolled
+ // decoder; it beats the CharsetDecoder path for non-ASCII input.
+ if (isAscii(input, offset, byteLen)) {
+ // Pure ASCII: bytes map 1:1 to chars; ISO-8859-1 is an intrinsic
LATIN1 copy.
+ return new String(input, offset, byteLen,
StandardCharsets.ISO_8859_1);
+ }
+ char[] chars = allocateReuseChars(byteLen);
+ int len = decodeUTF8Strict(input, offset, byteLen, chars);
+ if (len < 0) {
+ // Malformed input; map to U+FFFD via the JDK decoder (matches the
previous fallback).
+ return new String(input, offset, byteLen, StandardCharsets.UTF_8);
+ }
+ return new String(chars, 0, len);
+ }
+
+ /**
+ * SWAR ASCII test: reads 8 bytes per step and checks the high bit of each
via {@link
+ * #ASCII_MASK}, so a fully ASCII range is scanned ~8x faster than
byte-by-byte and non-ASCII
+ * input bails at the first eight-byte block that contains a set high bit.
+ */
+ private static boolean isAscii(byte[] bytes, int offset, int len) {
+ int i = offset;
+ final int end = offset + len;
+ final int swarEnd = offset + (len & ~7);
+ while (i < swarEnd) {
+ if (((long) LONG_VIEW.get(bytes, i) & ASCII_MASK) != 0) {
+ return false;
Review Comment:
added some tests here
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]