airborne12 commented on code in PR #67918:
URL: https://github.com/apache/doris/pull/67918#discussion_r4070126488
##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -226,26 +704,374 @@ private static String resolveTokenFilterIdentity(String
filterList) {
* IMPORTANT: Order is preserved because filter order is semantically
significant.
*/
private static String resolveCharFilterIdentity(String filterList) {
+ return resolveCharFilterIdentity(filterList, false);
+ }
+
+ private static String resolveCharFilterIdentity(String filterList, boolean
lowercaseDownstream) {
+ ArrayDeque<String> identities = new ArrayDeque<>();
+ walkCharFilters(filterList, lowercaseDownstream, identities);
+ return String.join(",", identities);
+ }
+
+ /**
+ * Resolve the chain from its last filter to its first, collecting
identities, and return the
+ * case-folding context that a filter placed in front of the chain would
run in.
+ */
+ private static boolean[] walkCharFilters(
+ String filterList, boolean lowercaseDownstream, Deque<String>
identities) {
+ boolean[] foldBlockedBytes = lowercaseDownstream ? new boolean[256] :
null;
if (Strings.isNullOrEmpty(filterList)) {
- return "";
+ return foldBlockedBytes;
}
- StringBuilder sb = new StringBuilder();
String[] filters = filterList.split(",\\s*");
// DO NOT sort - filter order is semantically significant
- for (int i = 0; i < filters.length; i++) {
- String filter = filters[i].trim();
- if (i > 0) {
- sb.append(",");
+ for (int i = filters.length - 1; i >= 0; --i) {
+ String filterName = filters[i].trim();
+ String filter = resolveComponentIdentity(
+ filterName, IndexPolicyTypeEnum.CHAR_FILTER,
foldBlockedBytes);
+ if (Strings.isNullOrEmpty(filter)) {
+ continue;
}
+ identities.addFirst(filter);
+ foldBlockedBytes = foldBlockedBytesBefore(filterName,
foldBlockedBytes);
+ }
+ return foldBlockedBytes;
+ }
- if (IndexPolicy.BUILTIN_CHAR_FILTERS.contains(filter)) {
- sb.append(filter);
- } else {
- sb.append(resolveComponentIdentity(filter,
IndexPolicyTypeEnum.CHAR_FILTER));
+ /**
+ * Context for the filter that runs before this one: a case fold starts a
fresh context, a
+ * char_replace filter adds the bytes it rewrites, and any other filter
ends the context.
+ */
+ private static boolean[] foldBlockedBytesBefore(String filterName,
boolean[] foldBlockedBytes) {
+ if (isCaseFoldingCharFilter(filterName)) {
+ return new boolean[256];
+ }
+ if (foldBlockedBytes == null) {
+ return null;
+ }
+ boolean[] sourceBytes = charReplaceSourceBytes(filterName);
+ if (sourceBytes == null) {
+ return null;
+ }
+ for (int i = 0; i < foldBlockedBytes.length; ++i) {
+ foldBlockedBytes[i] |= sourceBytes[i];
+ }
+ return foldBlockedBytes;
+ }
+
+ /** Bytes a named char_replace filter rewrites, or null for any other
filter. */
+ private static boolean[] charReplaceSourceBytes(String filterName) {
+ IndexPolicy policy = findPolicy(filterName,
IndexPolicyTypeEnum.CHAR_FILTER);
+ if (policy == null || policy.isInvalid() || policy.getProperties() ==
null) {
+ return null;
+ }
+ Map<String, String> properties = policy.getProperties();
+ String type = normalizeBuiltinComponentName(
+ properties.get(IndexPolicy.PROP_TYPE),
IndexPolicyTypeEnum.CHAR_FILTER);
+ String pattern = properties.get("pattern");
+ if (!"char_replace".equals(type) || pattern == null) {
+ return null;
+ }
+ boolean[] sourceBytes = new boolean[256];
+ for (int i = 0; i < pattern.length(); ++i) {
+ char patternByte = pattern.charAt(i);
+ if (patternByte < sourceBytes.length) {
+ sourceBytes[patternByte] = true;
}
}
- return sb.toString();
+ return sourceBytes;
+ }
+
+ /** The named policy when one exists with the expected type, or null. */
+ private static IndexPolicy findPolicy(String name, IndexPolicyTypeEnum
expectedType) {
+ if (Strings.isNullOrEmpty(name)) {
+ return null;
+ }
+ try {
+ Env env = Env.getCurrentEnv();
+ if (env != null && env.getIndexPolicyMgr() != null) {
+ IndexPolicy policy =
env.getIndexPolicyMgr().getPolicyByName(name);
+ if (policy != null && policy.getType() == expectedType) {
+ return policy;
+ }
+ }
+ } catch (RuntimeException e) {
+ // Treat lookup failures as an unknown policy.
+ }
+ return null;
+ }
+
+ private static boolean isCaseFoldingCharFilter(String name) {
+ if (Strings.isNullOrEmpty(name)) {
+ return false;
+ }
+
+ try {
+ Env env = Env.getCurrentEnv();
+ if (env != null && env.getIndexPolicyMgr() != null) {
+ IndexPolicy policy =
env.getIndexPolicyMgr().getPolicyByName(name);
+ if (policy != null && policy.getType() ==
IndexPolicyTypeEnum.CHAR_FILTER) {
+ if (policy.isInvalid()) {
+ return false;
+ }
+ Map<String, String> properties = policy.getProperties();
+ if (properties != null && !properties.isEmpty()) {
+ String type = normalizeBuiltinComponentName(
+ properties.get(IndexPolicy.PROP_TYPE),
IndexPolicyTypeEnum.CHAR_FILTER);
+ return "icu_normalizer".equals(type) &&
isCaseFoldingIcuNormalizer(properties);
+ }
+ }
+ }
+ } catch (RuntimeException e) {
+ // Fall through to built-in resolution.
+ }
+
+ return "icu_normalizer".equals(
+ normalizeBuiltinComponentName(name,
IndexPolicyTypeEnum.CHAR_FILTER));
+ }
+
+ /** Whether an icu_normalizer component folds case: the default nfkc_cf
form over every code point. */
+ private static boolean isCaseFoldingIcuNormalizer(Map<String, String>
properties) {
Review Comment:
Fixed in 77fa36419ba, with a stricter condition than "contains the
upper-case byte". BE's `FilteredNormalizer2` normalizes each in-set span on its
own, so `[A]` does absorb `A -> a`, but `contains(X)` alone is not sufficient:
with set `[Á]` and input `Á`, folding without the replacement composes `á`
inside one span while replacing first yields `á` as well only because the mark
follows; in general a combining mark in the set can change what the span
composes to.
The fold context now carries the restricting `UnicodeSet` (`FoldContext`),
and an upper-case byte is treated as folded only when the set contains it and
either also contains its lower-case form or contains no non-starter
(`[:^ccc=0:]`). Char-filter and token-filter forms share
`icuNormalizerFoldContext`.
`testFilteredCaseFoldAbsorbsReplacementOfCodePointInsideSet` covers `[A]` and
`[Aa]` as positives and `[B]`, `[a]`, `[Á]` as negatives; ALTER is in
`testAddInvertedIndexRejectsReplacementByteFilteredFoldAndBuiltinNormalizerAliases`.
##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -150,49 +221,458 @@ private static String
buildIdentityFromPolicyProperties(IndexPolicyTypeEnum type
* Resolve a component (tokenizer) to its identity.
*/
private static String resolveComponentIdentity(String name,
IndexPolicyTypeEnum expectedType) {
+ return resolveComponentIdentity(name, expectedType, null);
+ }
+
+ /**
+ * {@code foldBlockedBytes} is the case-folding context of a char filter:
null without a
+ * downstream fold, otherwise the bytes that filters between this one and
the fold rewrite.
+ */
+ private static String resolveComponentIdentity(
+ String name, IndexPolicyTypeEnum expectedType, boolean[]
foldBlockedBytes) {
if (Strings.isNullOrEmpty(name)) {
return "";
}
- // Check if it's a built-in component
- if (expectedType == IndexPolicyTypeEnum.TOKENIZER
- && IndexPolicy.BUILTIN_TOKENIZERS.contains(name)) {
- return name;
+ // Existing named policies take precedence over built-ins for upgrade
compatibility.
+ try {
+ Env env = Env.getCurrentEnv();
+ if (env != null && env.getIndexPolicyMgr() != null) {
+ IndexPolicy policy =
env.getIndexPolicyMgr().getPolicyByName(name);
+ if (policy != null && policy.getType() == expectedType) {
+ if (policy.isInvalid()) {
+ return "invalid-policy:" + policy.getId() + ":" +
policy.getName();
+ }
+ Map<String, String> props = policy.getProperties();
+ if (props != null && !props.isEmpty()) {
+ TreeMap<String, String> sortedProps = new
TreeMap<>(props);
+ String type = sortedProps.get(IndexPolicy.PROP_TYPE);
+ String normalizedType =
normalizeBuiltinComponentName(type, expectedType);
+ if (normalizedType != null) {
+ if ("empty".equals(normalizedType)) {
+ return "";
+ }
+ sortedProps.put(IndexPolicy.PROP_TYPE,
normalizedType);
+ canonicalizeEffectiveComponentProperties(
+ sortedProps, normalizedType, expectedType);
+ if (sortedProps.size() == 1) {
+ return normalizedType;
+ }
+ }
+ if (expectedType == IndexPolicyTypeEnum.TOKENIZER
+ &&
"ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+ // This setting only limits policy creation; it
does not change emitted tokens.
+ sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+ }
+ if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER
+ &&
"char_replace".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+ String replacement =
sortedProps.getOrDefault("replacement", " ");
+ String pattern = canonicalizeCharReplacePattern(
+ sortedProps.get("pattern"), replacement,
foldBlockedBytes);
+ if (pattern.isEmpty()) {
+ return "";
+ }
+ sortedProps.put("pattern", pattern);
+ sortedProps.put("replacement", replacement);
+ }
+ if (normalizedType != null && sortedProps.size() == 1)
{
+ return normalizedType;
+ }
+ return sortedProps.toString();
+ }
+ }
+ }
+ } catch (RuntimeException e) {
+ // Fall through to built-in resolution or the original name.
+ }
+
+ String normalizedName = normalizeBuiltinComponentName(name,
expectedType);
+ return "empty".equals(normalizedName) ? "" : normalizedName == null ?
name : normalizedName;
+ }
+
+ private static void canonicalizeEffectiveComponentProperties(
+ TreeMap<String, String> properties, String type,
IndexPolicyTypeEnum expectedType) {
+ if ("pinyin".equals(type)) {
+ removeBooleanDefaults(properties, true,
+ "keep_first_letter", "keep_full_pinyin",
"keep_none_chinese",
+ "keep_none_chinese_together",
"keep_none_chinese_in_first_letter",
+ "lowercase", "trim_whitespace", "ignore_pinyin_offset",
+ "none_chinese_pinyin_tokenize");
+ removeBooleanDefaults(properties, false,
+ "keep_separate_first_letter", "keep_joined_full_pinyin",
"keep_original",
+ "keep_none_chinese_in_joined_full_pinyin",
"remove_duplicated_term",
+ "fixed_pinyin_offset", "keep_separate_chinese");
+ removeIntegerDefault(properties, "limit_first_letter_length", 16);
+ canonicalizePinyinDependencies(properties, expectedType);
+ return;
+ }
+
+ if (expectedType == IndexPolicyTypeEnum.TOKEN_FILTER) {
+ if ("asciifolding".equals(type)) {
+ removeBooleanDefaults(properties, false, "preserve_original");
+ } else if ("word_delimiter".equals(type)) {
+ removeBooleanDefaults(properties, true, "generate_word_parts",
"generate_number_parts",
+ "split_on_case_change", "split_on_numerics",
"stem_english_possessive");
+ removeBooleanDefaults(properties, false, "catenate_words",
"catenate_numbers",
+ "catenate_all", "preserve_original");
+ canonicalizeWordSet(properties, "protected_words");
+ canonicalizeTypeTable(properties);
+ } else if ("icu_normalizer".equals(type)) {
+ canonicalizeIcuNormalizerDefaults(properties, false);
+ }
+ return;
}
- // For custom component, get its properties
+ if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER) {
+ if ("icu_normalizer".equals(type)) {
+ canonicalizeIcuNormalizerDefaults(properties, true);
+ }
+ return;
+ }
+
+ if (expectedType != IndexPolicyTypeEnum.TOKENIZER) {
+ return;
+ }
+ switch (type) {
+ case "ngram":
+ case "edge_ngram":
+ removeIntegerDefault(properties, "min_gram", 1);
+ removeIntegerDefault(properties, "max_gram", 2);
+ canonicalizeWordSet(properties, "token_chars");
+ canonicalizeCustomTokenChars(properties);
+ break;
+ case "standard":
+ removeIntegerDefault(properties, "max_token_length", 255);
+ break;
+ case "char_group":
+ removeIntegerDefault(properties, "max_token_length", 255);
+ canonicalizeTokenizeOnChars(properties);
+ break;
+ case "keyword":
+ // BE only range-checks buffer_size; the emitted term is
always capped by a constant.
+ properties.remove("buffer_size");
+ break;
+ case "basic":
+ canonicalizeBasicExtraChars(properties);
+ break;
+ default:
+ break;
+ }
+ }
+
+ private static void removeBooleanDefaults(
+ TreeMap<String, String> properties, boolean defaultValue,
String... keys) {
+ for (String key : keys) {
+ String value = properties.get(key);
+ if (value == null || !("true".equalsIgnoreCase(value) ||
"false".equalsIgnoreCase(value))) {
+ continue;
+ }
+ boolean parsed = Boolean.parseBoolean(value);
+ if (parsed == defaultValue) {
+ properties.remove(key);
+ } else {
+ properties.put(key, Boolean.toString(parsed));
+ }
+ }
+ }
+
+ private static void removeIntegerDefault(
+ TreeMap<String, String> properties, String key, int defaultValue) {
+ String value = properties.get(key);
+ if (value == null) {
+ return;
+ }
try {
- Env env = Env.getCurrentEnv();
- if (env == null || env.getIndexPolicyMgr() == null) {
- return name;
+ int parsed = Integer.parseInt(value);
+ if (parsed == defaultValue) {
+ properties.remove(key);
+ } else {
+ properties.put(key, Integer.toString(parsed));
}
+ } catch (NumberFormatException e) {
+ // Invalid policies keep their original identity.
+ }
+ }
- IndexPolicy policy = env.getIndexPolicyMgr().getPolicyByName(name);
- if (policy == null || policy.getType() != expectedType) {
- return name;
+ private static void canonicalizeIcuNormalizerDefaults(
+ TreeMap<String, String> properties, boolean hasMode) {
+ String name = properties.get("name");
+ if (name != null) {
+ String normalizedName = name.trim().toLowerCase(Locale.ROOT);
+ if ("nfkc_cf".equals(normalizedName)) {
+ properties.remove("name");
+ } else {
+ properties.put("name", normalizedName);
}
- if (policy.isInvalid()) {
- return "invalid-policy:" + policy.getId() + ":" +
policy.getName();
+ }
+ String filter = properties.get("unicode_set_filter");
+ if (filter != null && filter.isEmpty()) {
+ // BE treats an explicit empty string like an absent filter.
+ properties.remove("unicode_set_filter");
+ } else if (filter != null) {
+ try {
+ UnicodeSet unicodeSet = new UnicodeSet(filter);
+ if (unicodeSet.isEmpty()) {
+ properties.remove("unicode_set_filter");
+ } else {
+ properties.put("unicode_set_filter",
unicodeSet.toPattern(false));
+ }
+ } catch (IllegalArgumentException e) {
+ // Invalid policies keep their original identity.
}
+ }
+ if (hasMode) {
+ canonicalizeIcuNormalizerMode(properties);
+ }
+ }
- Map<String, String> props = policy.getProperties();
- if (props == null || props.isEmpty()) {
- return name;
+ private static void canonicalizeIcuNormalizerMode(TreeMap<String, String>
properties) {
+ removeStringDefault(properties, "mode", "compose");
+ if (!"decompose".equals(properties.get("mode"))) {
+ return;
+ }
+ // BE ignores mode for nfd/nfkd, and nfc/nfkc in decompose mode are
the same ICU instances.
+ String name = properties.get("name");
+ if ("nfc".equals(name) || "nfd".equals(name)) {
+ properties.put("name", "nfd");
+ properties.remove("mode");
+ } else if ("nfkc".equals(name) || "nfkd".equals(name)) {
+ properties.put("name", "nfkd");
+ properties.remove("mode");
+ }
+ }
+
+ // BE reads these settings as unordered sets of trimmed, non-empty words.
+ private static void canonicalizeWordSet(TreeMap<String, String>
properties, String key) {
+ String value = properties.get(key);
+ if (value == null) {
+ return;
+ }
+ TreeSet<String> words = new TreeSet<>();
+ for (String word : value.split(",")) {
+ String trimmed = trimAsciiWhitespace(word);
+ if (!trimmed.isEmpty()) {
+ words.add(trimmed);
}
+ }
+ if (words.isEmpty()) {
+ properties.remove(key);
+ } else {
+ properties.put(key, String.join(",", words));
+ }
+ }
+
+ // BE matches custom token characters as a code point set.
+ private static void canonicalizeCustomTokenChars(TreeMap<String, String>
properties) {
+ String value = properties.get("custom_token_chars");
+ if (value == null) {
+ return;
+ }
+ StringBuilder canonical = new StringBuilder();
+
value.codePoints().distinct().sorted().forEach(canonical::appendCodePoint);
+ properties.put("custom_token_chars", canonical.toString());
+ }
- // Build identity from sorted properties
- TreeMap<String, String> sortedProps = new TreeMap<>(props);
- if (expectedType == IndexPolicyTypeEnum.TOKENIZER
- && "ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE)))
{
- // This setting only limits policy creation; it does not
change emitted tokens.
- sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+ // BE collects tokenize_on_chars entries into sets, so order and repeats
do not matter.
+ private static void canonicalizeTokenizeOnChars(TreeMap<String, String>
properties) {
+ List<String> entries =
parseEntryList(properties.get("tokenize_on_chars"));
+ if (entries == null) {
+ return;
+ }
+ putEntryList(properties, "tokenize_on_chars", new TreeSet<>(entries));
+ }
+
+ // BE builds a per-character type map where a later rule for the same
character wins.
+ private static void canonicalizeTypeTable(TreeMap<String, String>
properties) {
+ List<String> rules = parseEntryList(properties.get("type_table"));
+ if (rules == null) {
+ return;
+ }
+ TreeMap<Integer, String> types = new TreeMap<>();
+ for (String rule : rules) {
+ int arrow = rule.lastIndexOf("=>");
+ if (arrow < 0 || rule.indexOf('\n') >= 0 || rule.indexOf('\r') >=
0) {
+ return;
}
- return sortedProps.toString();
- } catch (RuntimeException e) {
- return name;
+ String character = trimAsciiWhitespace(rule.substring(0, arrow));
+ String type = trimAsciiWhitespace(rule.substring(arrow + 2));
+ // Escaped characters keep the original identity rather than
reproducing BE unescaping.
+ if (character.indexOf('\\') >= 0 || character.codePointCount(0,
character.length()) != 1
+ || !WORD_DELIMITER_TYPES.contains(type)) {
+ return;
+ }
+ types.put(character.codePointAt(0), type);
+ }
+ List<String> canonicalRules = new ArrayList<>();
+ for (Map.Entry<Integer, String> entry : types.entrySet()) {
+ canonicalRules.add(new String(Character.toChars(entry.getKey())) +
"=>" + entry.getValue());
Review Comment:
Partially fixed in 77fa36419ba; the example in this thread is not equivalent
as stated. BE fills a custom table through `WordDelimiterIterator::get_type()`
(`u_charType`), but with no `type_table` it uses `DEFAULT_WORD_DELIM_TABLE`,
which is built from `u_isULowercase`/`u_isUUppercase`/`u_isdigit`; the two
disagree on Latin-1 code points such as `ª º ² ³ ¹ ¼ ½ ¾`, so `[a => LOWER]`
and an absent table split `ªA` differently and must keep distinct identities.
What is a no-op is a rule that restates `get_type()` inside a table that
also carries an effective rule. The identity now drops such rules for ASCII
code points (`[a => LOWER],[b => DIGIT]` is `[b => DIGIT]`), keeps the table
when every rule is a restatement (because it still replaces the default table),
and leaves non-ASCII rules alone.
`testWordDelimiterTypeTableDropsRulesRestatingBeClassification` covers the drop
plus `[a => LOWER]`, `[a => DIGIT]`, `[a => ALPHA]` and the absent table
staying pairwise distinct; CREATE/ALTER are in the shared cases.
##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -150,49 +221,458 @@ private static String
buildIdentityFromPolicyProperties(IndexPolicyTypeEnum type
* Resolve a component (tokenizer) to its identity.
*/
private static String resolveComponentIdentity(String name,
IndexPolicyTypeEnum expectedType) {
+ return resolveComponentIdentity(name, expectedType, null);
+ }
+
+ /**
+ * {@code foldBlockedBytes} is the case-folding context of a char filter:
null without a
+ * downstream fold, otherwise the bytes that filters between this one and
the fold rewrite.
+ */
+ private static String resolveComponentIdentity(
+ String name, IndexPolicyTypeEnum expectedType, boolean[]
foldBlockedBytes) {
if (Strings.isNullOrEmpty(name)) {
return "";
}
- // Check if it's a built-in component
- if (expectedType == IndexPolicyTypeEnum.TOKENIZER
- && IndexPolicy.BUILTIN_TOKENIZERS.contains(name)) {
- return name;
+ // Existing named policies take precedence over built-ins for upgrade
compatibility.
+ try {
+ Env env = Env.getCurrentEnv();
+ if (env != null && env.getIndexPolicyMgr() != null) {
+ IndexPolicy policy =
env.getIndexPolicyMgr().getPolicyByName(name);
+ if (policy != null && policy.getType() == expectedType) {
+ if (policy.isInvalid()) {
+ return "invalid-policy:" + policy.getId() + ":" +
policy.getName();
+ }
+ Map<String, String> props = policy.getProperties();
+ if (props != null && !props.isEmpty()) {
+ TreeMap<String, String> sortedProps = new
TreeMap<>(props);
+ String type = sortedProps.get(IndexPolicy.PROP_TYPE);
+ String normalizedType =
normalizeBuiltinComponentName(type, expectedType);
+ if (normalizedType != null) {
+ if ("empty".equals(normalizedType)) {
+ return "";
+ }
+ sortedProps.put(IndexPolicy.PROP_TYPE,
normalizedType);
+ canonicalizeEffectiveComponentProperties(
+ sortedProps, normalizedType, expectedType);
+ if (sortedProps.size() == 1) {
+ return normalizedType;
+ }
+ }
+ if (expectedType == IndexPolicyTypeEnum.TOKENIZER
+ &&
"ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+ // This setting only limits policy creation; it
does not change emitted tokens.
+ sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+ }
+ if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER
+ &&
"char_replace".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+ String replacement =
sortedProps.getOrDefault("replacement", " ");
+ String pattern = canonicalizeCharReplacePattern(
+ sortedProps.get("pattern"), replacement,
foldBlockedBytes);
+ if (pattern.isEmpty()) {
+ return "";
+ }
+ sortedProps.put("pattern", pattern);
+ sortedProps.put("replacement", replacement);
+ }
+ if (normalizedType != null && sortedProps.size() == 1)
{
+ return normalizedType;
+ }
+ return sortedProps.toString();
+ }
+ }
+ }
+ } catch (RuntimeException e) {
+ // Fall through to built-in resolution or the original name.
+ }
+
+ String normalizedName = normalizeBuiltinComponentName(name,
expectedType);
+ return "empty".equals(normalizedName) ? "" : normalizedName == null ?
name : normalizedName;
+ }
+
+ private static void canonicalizeEffectiveComponentProperties(
+ TreeMap<String, String> properties, String type,
IndexPolicyTypeEnum expectedType) {
+ if ("pinyin".equals(type)) {
+ removeBooleanDefaults(properties, true,
+ "keep_first_letter", "keep_full_pinyin",
"keep_none_chinese",
+ "keep_none_chinese_together",
"keep_none_chinese_in_first_letter",
+ "lowercase", "trim_whitespace", "ignore_pinyin_offset",
+ "none_chinese_pinyin_tokenize");
+ removeBooleanDefaults(properties, false,
+ "keep_separate_first_letter", "keep_joined_full_pinyin",
"keep_original",
+ "keep_none_chinese_in_joined_full_pinyin",
"remove_duplicated_term",
+ "fixed_pinyin_offset", "keep_separate_chinese");
+ removeIntegerDefault(properties, "limit_first_letter_length", 16);
+ canonicalizePinyinDependencies(properties, expectedType);
+ return;
+ }
+
+ if (expectedType == IndexPolicyTypeEnum.TOKEN_FILTER) {
+ if ("asciifolding".equals(type)) {
+ removeBooleanDefaults(properties, false, "preserve_original");
+ } else if ("word_delimiter".equals(type)) {
+ removeBooleanDefaults(properties, true, "generate_word_parts",
"generate_number_parts",
+ "split_on_case_change", "split_on_numerics",
"stem_english_possessive");
+ removeBooleanDefaults(properties, false, "catenate_words",
"catenate_numbers",
+ "catenate_all", "preserve_original");
+ canonicalizeWordSet(properties, "protected_words");
+ canonicalizeTypeTable(properties);
+ } else if ("icu_normalizer".equals(type)) {
+ canonicalizeIcuNormalizerDefaults(properties, false);
+ }
+ return;
}
- // For custom component, get its properties
+ if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER) {
+ if ("icu_normalizer".equals(type)) {
+ canonicalizeIcuNormalizerDefaults(properties, true);
+ }
+ return;
+ }
+
+ if (expectedType != IndexPolicyTypeEnum.TOKENIZER) {
+ return;
+ }
+ switch (type) {
+ case "ngram":
+ case "edge_ngram":
+ removeIntegerDefault(properties, "min_gram", 1);
+ removeIntegerDefault(properties, "max_gram", 2);
+ canonicalizeWordSet(properties, "token_chars");
+ canonicalizeCustomTokenChars(properties);
+ break;
+ case "standard":
+ removeIntegerDefault(properties, "max_token_length", 255);
+ break;
+ case "char_group":
+ removeIntegerDefault(properties, "max_token_length", 255);
+ canonicalizeTokenizeOnChars(properties);
+ break;
+ case "keyword":
+ // BE only range-checks buffer_size; the emitted term is
always capped by a constant.
+ properties.remove("buffer_size");
+ break;
+ case "basic":
+ canonicalizeBasicExtraChars(properties);
+ break;
+ default:
+ break;
+ }
+ }
+
+ private static void removeBooleanDefaults(
+ TreeMap<String, String> properties, boolean defaultValue,
String... keys) {
+ for (String key : keys) {
+ String value = properties.get(key);
+ if (value == null || !("true".equalsIgnoreCase(value) ||
"false".equalsIgnoreCase(value))) {
+ continue;
+ }
+ boolean parsed = Boolean.parseBoolean(value);
+ if (parsed == defaultValue) {
+ properties.remove(key);
+ } else {
+ properties.put(key, Boolean.toString(parsed));
+ }
+ }
+ }
+
+ private static void removeIntegerDefault(
+ TreeMap<String, String> properties, String key, int defaultValue) {
+ String value = properties.get(key);
+ if (value == null) {
+ return;
+ }
try {
- Env env = Env.getCurrentEnv();
- if (env == null || env.getIndexPolicyMgr() == null) {
- return name;
+ int parsed = Integer.parseInt(value);
+ if (parsed == defaultValue) {
+ properties.remove(key);
+ } else {
+ properties.put(key, Integer.toString(parsed));
}
+ } catch (NumberFormatException e) {
+ // Invalid policies keep their original identity.
+ }
+ }
- IndexPolicy policy = env.getIndexPolicyMgr().getPolicyByName(name);
- if (policy == null || policy.getType() != expectedType) {
- return name;
+ private static void canonicalizeIcuNormalizerDefaults(
+ TreeMap<String, String> properties, boolean hasMode) {
+ String name = properties.get("name");
+ if (name != null) {
+ String normalizedName = name.trim().toLowerCase(Locale.ROOT);
+ if ("nfkc_cf".equals(normalizedName)) {
+ properties.remove("name");
+ } else {
+ properties.put("name", normalizedName);
}
- if (policy.isInvalid()) {
- return "invalid-policy:" + policy.getId() + ":" +
policy.getName();
+ }
+ String filter = properties.get("unicode_set_filter");
+ if (filter != null && filter.isEmpty()) {
+ // BE treats an explicit empty string like an absent filter.
+ properties.remove("unicode_set_filter");
+ } else if (filter != null) {
+ try {
+ UnicodeSet unicodeSet = new UnicodeSet(filter);
+ if (unicodeSet.isEmpty()) {
+ properties.remove("unicode_set_filter");
+ } else {
+ properties.put("unicode_set_filter",
unicodeSet.toPattern(false));
+ }
+ } catch (IllegalArgumentException e) {
+ // Invalid policies keep their original identity.
}
+ }
+ if (hasMode) {
+ canonicalizeIcuNormalizerMode(properties);
+ }
+ }
- Map<String, String> props = policy.getProperties();
- if (props == null || props.isEmpty()) {
- return name;
+ private static void canonicalizeIcuNormalizerMode(TreeMap<String, String>
properties) {
+ removeStringDefault(properties, "mode", "compose");
+ if (!"decompose".equals(properties.get("mode"))) {
+ return;
+ }
+ // BE ignores mode for nfd/nfkd, and nfc/nfkc in decompose mode are
the same ICU instances.
+ String name = properties.get("name");
+ if ("nfc".equals(name) || "nfd".equals(name)) {
+ properties.put("name", "nfd");
+ properties.remove("mode");
+ } else if ("nfkc".equals(name) || "nfkd".equals(name)) {
+ properties.put("name", "nfkd");
+ properties.remove("mode");
+ }
+ }
+
+ // BE reads these settings as unordered sets of trimmed, non-empty words.
+ private static void canonicalizeWordSet(TreeMap<String, String>
properties, String key) {
+ String value = properties.get(key);
+ if (value == null) {
+ return;
+ }
+ TreeSet<String> words = new TreeSet<>();
+ for (String word : value.split(",")) {
+ String trimmed = trimAsciiWhitespace(word);
+ if (!trimmed.isEmpty()) {
+ words.add(trimmed);
}
+ }
+ if (words.isEmpty()) {
+ properties.remove(key);
+ } else {
+ properties.put(key, String.join(",", words));
+ }
+ }
+
+ // BE matches custom token characters as a code point set.
+ private static void canonicalizeCustomTokenChars(TreeMap<String, String>
properties) {
+ String value = properties.get("custom_token_chars");
+ if (value == null) {
+ return;
+ }
+ StringBuilder canonical = new StringBuilder();
+
value.codePoints().distinct().sorted().forEach(canonical::appendCodePoint);
+ properties.put("custom_token_chars", canonical.toString());
+ }
- // Build identity from sorted properties
- TreeMap<String, String> sortedProps = new TreeMap<>(props);
- if (expectedType == IndexPolicyTypeEnum.TOKENIZER
- && "ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE)))
{
- // This setting only limits policy creation; it does not
change emitted tokens.
- sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+ // BE collects tokenize_on_chars entries into sets, so order and repeats
do not matter.
+ private static void canonicalizeTokenizeOnChars(TreeMap<String, String>
properties) {
+ List<String> entries =
parseEntryList(properties.get("tokenize_on_chars"));
+ if (entries == null) {
+ return;
+ }
+ putEntryList(properties, "tokenize_on_chars", new TreeSet<>(entries));
+ }
+
+ // BE builds a per-character type map where a later rule for the same
character wins.
+ private static void canonicalizeTypeTable(TreeMap<String, String>
properties) {
+ List<String> rules = parseEntryList(properties.get("type_table"));
+ if (rules == null) {
+ return;
+ }
+ TreeMap<Integer, String> types = new TreeMap<>();
+ for (String rule : rules) {
+ int arrow = rule.lastIndexOf("=>");
+ if (arrow < 0 || rule.indexOf('\n') >= 0 || rule.indexOf('\r') >=
0) {
+ return;
}
- return sortedProps.toString();
- } catch (RuntimeException e) {
- return name;
+ String character = trimAsciiWhitespace(rule.substring(0, arrow));
+ String type = trimAsciiWhitespace(rule.substring(arrow + 2));
+ // Escaped characters keep the original identity rather than
reproducing BE unescaping.
+ if (character.indexOf('\\') >= 0 || character.codePointCount(0,
character.length()) != 1
+ || !WORD_DELIMITER_TYPES.contains(type)) {
+ return;
+ }
+ types.put(character.codePointAt(0), type);
+ }
+ List<String> canonicalRules = new ArrayList<>();
+ for (Map.Entry<Integer, String> entry : types.entrySet()) {
+ canonicalRules.add(new String(Character.toChars(entry.getKey())) +
"=>" + entry.getValue());
+ }
+ putEntryList(properties, "type_table", canonicalRules);
+ }
+
+ /** Parse a bracketed entry list as BE does, or return null for a
malformed list. */
+ private static List<String> parseEntryList(String value) {
+ if (value == null) {
+ return null;
+ }
+ List<String> entries = new ArrayList<>();
+ String trimmed = trimAsciiWhitespace(value);
+ if (trimmed.isEmpty()) {
+ return entries;
+ }
+ for (String item : ENTRY_SEPARATOR.split(trimmed)) {
+ String entry = trimAsciiWhitespace(item);
+ if (entry.length() < 2 || entry.charAt(0) != '[' ||
entry.charAt(entry.length() - 1) != ']') {
+ return null;
+ }
+ String content = entry.substring(1, entry.length() - 1);
+ if (!content.isEmpty()) {
+ entries.add(content);
+ }
+ }
+ return entries;
+ }
+
+ private static void putEntryList(TreeMap<String, String> properties,
String key, Collection<String> entries) {
+ if (entries.isEmpty()) {
+ properties.remove(key);
+ return;
+ }
+ StringBuilder canonical = new StringBuilder();
+ for (String entry : entries) {
+ if (canonical.length() > 0) {
+ canonical.append(",");
+ }
+ canonical.append("[").append(entry).append("]");
+ }
+ properties.put(key, canonical.toString());
+ }
+
+ // Trim the same ASCII whitespace that BE trims.
+ private static String trimAsciiWhitespace(String value) {
+ int begin = 0;
+ int end = value.length();
+ while (begin < end && isAsciiWhitespace(value.charAt(begin))) {
+ ++begin;
+ }
+ while (end > begin && isAsciiWhitespace(value.charAt(end - 1))) {
+ --end;
+ }
+ return value.substring(begin, end);
+ }
+
+ private static boolean isAsciiWhitespace(char value) {
+ return value == ' ' || (value >= '\t' && value <= '\r');
+ }
+
+ private static void canonicalizeBasicExtraChars(TreeMap<String, String>
properties) {
Review Comment:
Fixed in 77fa36419ba. Confirmed on BE: `BasicTokenizer::cut()` consumes an
`is_alnum` run (CLucene's `[0-9A-Za-z]` table) before the branch that consults
`_extra_char_set`, so alphanumerics in `extra_chars` are unreachable.
Reproduced first: `{type=basic}` and `{type=basic, extra_chars=A0}` had
different identities.
`canonicalizeBasicExtraChars` now drops ASCII letters and digits and removes
the property when nothing remains. `testBasicExtraCharsIgnoreAlphanumerics`
covers an alphanumeric-only value and a punctuation negative; CREATE/ALTER are
in the shared cases.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]