airborne12 commented on code in PR #67918:
URL: https://github.com/apache/doris/pull/67918#discussion_r4089472428
##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -226,26 +1049,410 @@ private static String resolveTokenFilterIdentity(String
filterList) {
* IMPORTANT: Order is preserved because filter order is semantically
significant.
*/
private static String resolveCharFilterIdentity(String filterList) {
+ return resolveCharFilterIdentity(filterList, null);
+ }
+
+ private static String resolveCharFilterIdentity(String filterList,
FoldContext downstreamFold) {
+ ArrayDeque<String> identities = new ArrayDeque<>();
+ walkCharFilters(filterList, downstreamFold, identities);
+ return String.join(",", identities);
+ }
+
+ /**
+ * Resolve the chain from its last filter to its first, collecting
identities, and return the
+ * case-folding context that a filter placed in front of the chain would
run in.
+ */
+ private static FoldContext walkCharFilters(
+ String filterList, FoldContext downstreamFold, Deque<String>
identities) {
+ FoldContext fold = downstreamFold;
if (Strings.isNullOrEmpty(filterList)) {
- return "";
+ return fold;
}
- StringBuilder sb = new StringBuilder();
String[] filters = filterList.split(",\\s*");
// DO NOT sort - filter order is semantically significant
- for (int i = 0; i < filters.length; i++) {
- String filter = filters[i].trim();
- if (i > 0) {
- sb.append(",");
+ for (int i = filters.length - 1; i >= 0; --i) {
+ String filterName = filters[i].trim();
+ String filter = resolveComponentIdentity(filterName,
IndexPolicyTypeEnum.CHAR_FILTER, fold);
+ if (Strings.isNullOrEmpty(filter)) {
+ continue;
+ }
+ // Repeating a char_replace filter rewrites the same bytes to the
same byte again.
+ if (!filter.equals(identities.peekFirst()) ||
!isIdempotentCharFilter(filterName)) {
+ identities.addFirst(filter);
}
+ fold = foldContextBefore(filterName, fold);
+ }
+ return fold;
+ }
- if (IndexPolicy.BUILTIN_CHAR_FILTERS.contains(filter)) {
- sb.append(filter);
- } else {
- sb.append(resolveComponentIdentity(filter,
IndexPolicyTypeEnum.CHAR_FILTER));
+ /**
+ * Context for the filter that runs before this one: a case fold starts a
fresh context, a
+ * char_replace filter adds the bytes it rewrites, and any other filter
ends the context.
+ */
+ private static FoldContext foldContextBefore(String filterName,
FoldContext fold) {
+ FoldContext caseFold = caseFoldingCharFilterContext(filterName);
+ if (caseFold != null) {
+ return caseFold;
+ }
+ if (fold == null) {
+ return null;
+ }
+ boolean[] sourceBytes = charReplaceSourceBytes(filterName);
+ if (sourceBytes == null) {
+ return null;
+ }
+ fold.block(sourceBytes);
+ return fold;
+ }
+
+ /**
+ * Whether the filter is a usable char_replace, which replaces each
pattern byte with the same
+ * single byte and so leaves the stream unchanged when it runs again.
+ */
+ private static boolean isIdempotentCharFilter(String filterName) {
+ return charReplaceSourceBytes(filterName) != null;
+ }
+
+ /**
+ * Bytes a char_replace filter rewrites, or null for any other filter. A
bare built-in reference
+ * is instantiated with the factory defaults.
+ */
+ private static boolean[] charReplaceSourceBytes(String filterName) {
+ String pattern = CHAR_REPLACE_DEFAULT_PATTERN;
+ String replacement = CHAR_REPLACE_DEFAULT_REPLACEMENT;
+ IndexPolicy policy = findPolicy(filterName,
IndexPolicyTypeEnum.CHAR_FILTER);
+ if (policy != null) {
+ if (policy.isInvalid() || policy.getProperties() == null) {
+ return null;
+ }
+ Map<String, String> properties = policy.getProperties();
+ String type = normalizeBuiltinComponentName(
+ properties.get(IndexPolicy.PROP_TYPE),
IndexPolicyTypeEnum.CHAR_FILTER);
+ if (!CHAR_REPLACE_FILTER.equals(type)) {
+ return null;
}
+ pattern = properties.getOrDefault(PROP_PATTERN,
CHAR_REPLACE_DEFAULT_PATTERN);
+ replacement = properties.getOrDefault(PROP_REPLACEMENT,
CHAR_REPLACE_DEFAULT_REPLACEMENT);
+ } else if (!CHAR_REPLACE_FILTER.equals(
+ normalizeBuiltinComponentName(filterName,
IndexPolicyTypeEnum.CHAR_FILTER))) {
+ return null;
}
- return sb.toString();
+ // Replacing the single replacement byte with itself leaves the stream
unchanged.
+ int replacementByte = replacement.length() == 1 &&
replacement.charAt(0) < 128 ? replacement.charAt(0) : -1;
+ boolean[] sourceBytes = new boolean[256];
+ for (int i = 0; i < pattern.length(); ++i) {
+ char patternByte = pattern.charAt(i);
+ if (patternByte < sourceBytes.length && patternByte !=
replacementByte) {
+ sourceBytes[patternByte] = true;
+ }
+ }
+ return sourceBytes;
+ }
+
+ /** The named policy when one exists with the expected type, or null. */
+ private static IndexPolicy findPolicy(String name, IndexPolicyTypeEnum
expectedType) {
+ if (Strings.isNullOrEmpty(name)) {
+ return null;
+ }
+ try {
+ Env env = Env.getCurrentEnv();
+ if (env != null && env.getIndexPolicyMgr() != null) {
+ IndexPolicy policy =
env.getIndexPolicyMgr().getPolicyByName(name);
+ if (policy != null && policy.getType() == expectedType) {
+ return policy;
+ }
+ }
+ } catch (RuntimeException e) {
+ // Treat lookup failures as an unknown policy.
+ }
+ return null;
+ }
+
+ /** Fold context started by a named or built-in case-folding char filter,
or null for any other filter. */
+ private static FoldContext caseFoldingCharFilterContext(String name) {
+ if (Strings.isNullOrEmpty(name)) {
+ return null;
+ }
+
+ try {
+ Env env = Env.getCurrentEnv();
+ if (env != null && env.getIndexPolicyMgr() != null) {
+ IndexPolicy policy =
env.getIndexPolicyMgr().getPolicyByName(name);
+ if (policy != null && policy.getType() ==
IndexPolicyTypeEnum.CHAR_FILTER) {
+ if (policy.isInvalid()) {
+ return null;
+ }
+ Map<String, String> properties = policy.getProperties();
+ if (properties != null && !properties.isEmpty()) {
+ String type = normalizeBuiltinComponentName(
+ properties.get(IndexPolicy.PROP_TYPE),
IndexPolicyTypeEnum.CHAR_FILTER);
+ return "icu_normalizer".equals(type) ?
icuNormalizerFoldContext(properties) : null;
+ }
+ }
+ }
+ } catch (RuntimeException e) {
+ // Fall through to built-in resolution.
+ }
+
+ return "icu_normalizer".equals(normalizeBuiltinComponentName(name,
IndexPolicyTypeEnum.CHAR_FILTER))
+ ? FoldContext.unfiltered() : null;
+ }
+
+ /**
+ * Fold context of an icu_normalizer component: the default nfkc_cf form
folds case over every
+ * code point, or only inside a parsable non-empty unicode_set_filter.
Null for other forms.
+ */
+ private static FoldContext icuNormalizerFoldContext(Map<String, String>
properties) {
+ if (!"nfkc_cf".equals(icuNormalizerName(properties))) {
+ return null;
+ }
+ String filter = properties.get("unicode_set_filter");
+ if (filter == null || filter.isEmpty()) {
+ return FoldContext.unfiltered();
+ }
+ try {
+ UnicodeSet unicodeSet = new UnicodeSet(filter);
+ return unicodeSet.isEmpty() ? FoldContext.unfiltered() : new
FoldContext(unicodeSet.freeze());
+ } catch (IllegalArgumentException e) {
+ return null;
+ }
+ }
+
+ /** Whether an icu_normalizer component leaves ASCII letters as they are.
*/
+ private static boolean isAsciiCaseTransparentIcuNormalizer(Map<String,
String> properties) {
+ String name = icuNormalizerName(properties);
+ return "nfc".equals(name) || "nfd".equals(name) || "nfkc".equals(name)
|| "nfkd".equals(name);
+ }
+
+ private static String icuNormalizerName(Map<String, String> properties) {
+ return properties.getOrDefault("name",
"nfkc_cf").trim().toLowerCase(Locale.ROOT);
+ }
+
+ /** The outer char filter runs before everything else, so it takes the
analyzer's fold context. */
+ private static String appendOuterCharFilterIdentity(
+ String analyzerIdentity, Map<String, String> properties,
FoldContext fold) {
+ String type =
properties.get(InvertedIndexProperties.INVERTED_INDEX_PARSER_CHAR_FILTER_TYPE);
+ String pattern =
properties.get(InvertedIndexProperties.INVERTED_INDEX_PARSER_CHAR_FILTER_PATTERN);
+ if (!"char_replace".equals(type) || Strings.isNullOrEmpty(pattern)) {
+ return analyzerIdentity;
+ }
+ String replacement = properties.getOrDefault(
+
InvertedIndexProperties.INVERTED_INDEX_PARSER_CHAR_FILTER_REPLACEMENT, " ");
+ String canonicalPattern = canonicalizeCharReplacePattern(pattern,
replacement, fold);
+ if (canonicalPattern.isEmpty()) {
+ return analyzerIdentity;
+ }
+ return analyzerIdentity + "|outer_char_filter=char_replace:"
+ + canonicalPattern.length() + ":" + canonicalPattern + ":"
+ + replacement.length() + ":" + replacement + ";";
+ }
+
+ /**
+ * Canonicalize the ASCII pattern to the BE filter's byte set.
+ * Order, duplicate bytes, and replacements of a byte with itself do not
change the stream.
+ */
+ private static String canonicalizeCharReplacePattern(
+ String pattern, String replacement, FoldContext fold) {
+ if (replacement.length() != 1) {
+ return pattern;
+ }
+ char replacementByte = replacement.charAt(0);
+ boolean[] replacedBytes = new boolean[256];
+ for (int i = 0; i < pattern.length(); ++i) {
+ char patternByte = pattern.charAt(i);
+ if (patternByte < replacedBytes.length && patternByte !=
replacementByte) {
+ replacedBytes[patternByte] = true;
+ }
+ }
+ if (fold != null && replacementByte >= 'a' && replacementByte <= 'z') {
+ // The downstream fold maps the upper-case byte to the replacement
anyway.
+ int upperByte = replacementByte - ('a' - 'A');
+ if (fold.foldsByte(upperByte, replacementByte)) {
+ replacedBytes[upperByte] = false;
+ }
+ } else if (fold != null && replacementByte >= 'A' && replacementByte
<= 'Z') {
+ // The downstream fold maps the replacement back to the lower-case
byte it replaced.
+ int lowerByte = replacementByte + ('a' - 'A');
+ if (fold.foldsByte(replacementByte, lowerByte)) {
+ replacedBytes[lowerByte] = false;
+ }
+ }
+
+ StringBuilder canonical = new StringBuilder();
+ for (int i = 0; i < replacedBytes.length; ++i) {
+ if (replacedBytes[i]) {
+ canonical.append((char) i);
+ }
+ }
+ return canonical.toString();
+ }
+
+ /**
+ * Fold context of a built-in IK analyzer. IK lower-cases single-byte
ASCII in the buffer its
+ * lexeme text is copied from, which lower_case=false does not reach.
+ */
+ private static FoldContext builtinIkFoldContext() {
+ return FoldContext.unfiltered();
+ }
+
+ /**
+ * Fold context for the outer char filter of a custom analyzer or
normalizer, which BE applies
+ * before the policy's own char filters. Unknown or unresolvable policies
get no context.
+ */
+ private static FoldContext customAnalyzerFoldContext(String analyzerName) {
+ if (IndexPolicy.BUILTIN_ANALYZERS.contains(analyzerName)) {
+ return null;
+ }
+ if (isBuiltinNormalizerBinding(analyzerName)) {
+ // The built-in normalizer lowercases keyword tokens without char
filters of its own.
+ return FoldContext.unfiltered();
+ }
+ IndexPolicy policy = findPolicy(analyzerName,
IndexPolicyTypeEnum.ANALYZER);
+ if (policy == null) {
+ policy = findPolicy(analyzerName, IndexPolicyTypeEnum.NORMALIZER);
+ }
+ if (policy == null || policy.isInvalid() || policy.getProperties() ==
null
+ || policy.getProperties().isEmpty()) {
+ return null;
+ }
+ Map<String, String> properties = policy.getProperties();
+ try {
+ String tokenizerIdentity = resolveComponentIdentity(
+ properties.get(IndexPolicy.PROP_TOKENIZER),
IndexPolicyTypeEnum.TOKENIZER);
+ return
walkCharFilters(properties.get(IndexPolicy.PROP_CHAR_FILTER),
+ foldsAsciiCaseAfterCharFilters(policy.getType(),
properties, tokenizerIdentity),
+ new ArrayDeque<>());
+ } catch (RuntimeException e) {
+ return null;
+ }
+ }
+
+ /**
+ * The fold the tokenizer and token filters apply to ASCII letters, so a
char filter that only
+ * lowercases such a letter cannot change the output, or null when they
keep case.
+ */
+ private static FoldContext foldsAsciiCaseAfterCharFilters(
+ IndexPolicyTypeEnum type, Map<String, String> properties, String
tokenizerIdentity) {
+ if (type == IndexPolicyTypeEnum.NORMALIZER) {
+ // A normalizer always tokenizes with keyword, which is case
transparent.
+ return
tokenFiltersFoldAsciiCase(properties.get(IndexPolicy.PROP_TOKEN_FILTER));
+ }
+ if ("ik_smart".equals(tokenizerIdentity) ||
"ik_max_word".equals(tokenizerIdentity)) {
+ return FoldContext.unfiltered();
+ }
+ return
isCaseTransparentTokenizer(properties.get(IndexPolicy.PROP_TOKENIZER))
+ ?
tokenFiltersFoldAsciiCase(properties.get(IndexPolicy.PROP_TOKEN_FILTER)) : null;
+ }
+
+ /** Whether the tokenizer splits and emits ASCII letters the same way
regardless of their case. */
+ private static boolean isCaseTransparentTokenizer(String name) {
+ TreeMap<String, String> settings = resolveComponentSettings(name,
IndexPolicyTypeEnum.TOKENIZER);
+ if (settings == null) {
+ return false;
+ }
+ String type = settings.get(IndexPolicy.PROP_TYPE);
+ // Judge the same canonical settings the tokenizer identity is built
from.
+ canonicalizeEffectiveComponentProperties(settings, type,
IndexPolicyTypeEnum.TOKENIZER);
+ switch (type) {
+ case "standard":
+ case "keyword":
+ case "icu":
+ case "basic":
+ return true;
+ case "ngram":
+ case "edge_ngram":
+ return !settings.containsKey("custom_token_chars");
+ case "char_group":
+ return
tokenizeOnCharsIgnoreAsciiLetters(settings.get("tokenize_on_chars"));
+ default:
+ return false;
+ }
+ }
+
+ /** Settings of a named or built-in component with a canonical type, or
null when unknown. */
+ private static TreeMap<String, String> resolveComponentSettings(String
name, IndexPolicyTypeEnum expectedType) {
+ if (Strings.isNullOrEmpty(name)) {
+ return null;
+ }
+ TreeMap<String, String> settings = new TreeMap<>();
+ IndexPolicy policy = findPolicy(name, expectedType);
+ if (policy != null) {
+ if (policy.isInvalid()) {
+ return null;
+ }
+ if (policy.getProperties() != null) {
+ settings.putAll(policy.getProperties());
+ }
+ }
+ String type = normalizeBuiltinComponentName(
+ settings.isEmpty() ? name :
settings.get(IndexPolicy.PROP_TYPE), expectedType);
+ if (type == null) {
+ return null;
+ }
+ settings.put(IndexPolicy.PROP_TYPE, type);
+ return settings;
+ }
+
+ // Escaped entries keep the conservative answer rather than reproducing BE
unescaping.
+ private static boolean tokenizeOnCharsIgnoreAsciiLetters(String value) {
+ if (value == null) {
+ return true;
+ }
+ List<String> entries = parseEntryList(value);
+ if (entries == null) {
+ return false;
+ }
+ for (String entry : entries) {
+ if (CHAR_GROUP_TYPES.contains(entry)) {
+ continue;
+ }
+ if (entry.indexOf('\\') >= 0 || entry.codePointCount(0,
entry.length()) != 1) {
+ return false;
+ }
+ int codePoint = entry.codePointAt(0);
+ if ((codePoint >= 'A' && codePoint <= 'Z') || (codePoint >= 'a' &&
codePoint <= 'z')) {
+ return false;
+ }
+ }
+ return true;
+ }
+
+ /**
+ * The first token filter that folds ASCII case, reached before any filter
that could tell an
+ * upper-case letter from its lower-case form, or null when there is none.
+ */
+ private static FoldContext tokenFiltersFoldAsciiCase(String filterList) {
+ if (Strings.isNullOrEmpty(filterList)) {
+ return null;
+ }
+ for (String filterName : filterList.split(",\\s*")) {
+ TreeMap<String, String> settings = resolveComponentSettings(
+ filterName.trim(), IndexPolicyTypeEnum.TOKEN_FILTER);
+ if (settings == null) {
+ return null;
+ }
+ switch (settings.get(IndexPolicy.PROP_TYPE)) {
+ case "lowercase":
+ return FoldContext.unfiltered();
+ case "empty":
+ case "asciifolding":
+ // ASCII bytes pass through ASCII folding unchanged.
+ continue;
+ case "icu_normalizer":
+ FoldContext fold = icuNormalizerFoldContext(settings);
+ if (fold != null) {
+ return fold;
+ }
+ if (isAsciiCaseTransparentIcuNormalizer(settings)) {
+ continue;
+ }
+ return null;
+ default:
Review Comment:
Thanks - the two chains you describe do fold ASCII case, so an outer `A ->
a` cannot change what they emit.
We are drawing a line here, and we would rather say where and why than close
this quietly.
Every finding in this area so far has fallen into one of two directions. One
direction blocks users: an identity that merges two configurations BE actually
treats differently makes CREATE and ALTER reject a pair of indexes that emit
different terms. We have fixed every one of those, including the one the
previous round found in our own commit, where the standard analyzer lost
`lower_case` and `stopwords`.
The other direction is the one this thread is in. The identity keeps two
configurations apart that BE treats identically, so the duplicate fence admits
a redundant index. The cost is a second index holding the same postings -
storage and write amplification - and nothing a user is prevented from doing,
nothing queried incorrectly.
The configuration space here is large enough that completeness in that
direction is not a finish line we can reach: each of the last four rounds has
found a deeper case in the same family, and we expect that to continue. So we
are keeping the fence sound in the blocking direction and accepting that it
stays conservative in this one.
This is a judgement about scope, not a dispute about the mechanism - if a
maintainer wants the remaining canonicalization completed, it is well worth a
follow-up change of its own, and this thread describes what it would need to
cover.
Worth recording for whoever picks this up: the pinyin case has to be derived
from the effective settings rather than the filter name, since
`lowercase=false` and the outputs that can carry source case have to stay
distinct, and the WordDelimiter case needs `split_on_case_change=false`
together with no protected words and no case-distinguishing type table before
it may continue to the later fold.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]