airborne12 commented on code in PR #67918:
URL: https://github.com/apache/doris/pull/67918#discussion_r4067854471


##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -150,49 +205,287 @@ private static String 
buildIdentityFromPolicyProperties(IndexPolicyTypeEnum type
      * Resolve a component (tokenizer) to its identity.
      */
     private static String resolveComponentIdentity(String name, 
IndexPolicyTypeEnum expectedType) {
+        return resolveComponentIdentity(name, expectedType, false);
+    }
+
+    private static String resolveComponentIdentity(
+            String name, IndexPolicyTypeEnum expectedType, boolean 
lowercaseDownstream) {
         if (Strings.isNullOrEmpty(name)) {
             return "";
         }
 
-        // Check if it's a built-in component
-        if (expectedType == IndexPolicyTypeEnum.TOKENIZER
-                && IndexPolicy.BUILTIN_TOKENIZERS.contains(name)) {
-            return name;
-        }
-
-        // For custom component, get its properties
+        // Existing named policies take precedence over built-ins for upgrade 
compatibility.
         try {
             Env env = Env.getCurrentEnv();
-            if (env == null || env.getIndexPolicyMgr() == null) {
-                return name;
+            if (env != null && env.getIndexPolicyMgr() != null) {
+                IndexPolicy policy = 
env.getIndexPolicyMgr().getPolicyByName(name);
+                if (policy != null && policy.getType() == expectedType) {
+                    if (policy.isInvalid()) {
+                        return "invalid-policy:" + policy.getId() + ":" + 
policy.getName();
+                    }
+                    Map<String, String> props = policy.getProperties();
+                    if (props != null && !props.isEmpty()) {
+                        TreeMap<String, String> sortedProps = new 
TreeMap<>(props);
+                        String type = sortedProps.get(IndexPolicy.PROP_TYPE);
+                        String normalizedType = 
normalizeBuiltinComponentName(type, expectedType);
+                        if (normalizedType != null) {
+                            if ("empty".equals(normalizedType)) {
+                                return "";
+                            }
+                            sortedProps.put(IndexPolicy.PROP_TYPE, 
normalizedType);
+                            canonicalizeEffectiveComponentProperties(
+                                    sortedProps, normalizedType, expectedType);
+                            if (sortedProps.size() == 1) {
+                                return normalizedType;
+                            }
+                        }
+                        if (expectedType == IndexPolicyTypeEnum.TOKENIZER
+                                && 
"ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+                            // This setting only limits policy creation; it 
does not change emitted tokens.
+                            sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+                        }
+                        if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER
+                                && 
"char_replace".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+                            String replacement = 
sortedProps.getOrDefault("replacement", " ");
+                            String pattern = canonicalizeCharReplacePattern(
+                                    sortedProps.get("pattern"), replacement, 
lowercaseDownstream);
+                            if (pattern.isEmpty()) {
+                                return "";
+                            }
+                            sortedProps.put("pattern", pattern);
+                            sortedProps.put("replacement", replacement);
+                        }
+                        if (normalizedType != null && sortedProps.size() == 1) 
{
+                            return normalizedType;
+                        }
+                        return sortedProps.toString();
+                    }
+                }
+            }
+        } catch (RuntimeException e) {
+            // Fall through to built-in resolution or the original name.
+        }
+
+        String normalizedName = normalizeBuiltinComponentName(name, 
expectedType);
+        return "empty".equals(normalizedName) ? "" : normalizedName == null ? 
name : normalizedName;
+    }
+
+    private static void canonicalizeEffectiveComponentProperties(
+            TreeMap<String, String> properties, String type, 
IndexPolicyTypeEnum expectedType) {
+        if ("pinyin".equals(type)) {
+            removeBooleanDefaults(properties, true,
+                    "keep_first_letter", "keep_full_pinyin", 
"keep_none_chinese",
+                    "keep_none_chinese_together", 
"keep_none_chinese_in_first_letter",
+                    "lowercase", "trim_whitespace", "ignore_pinyin_offset",
+                    "none_chinese_pinyin_tokenize");
+            removeBooleanDefaults(properties, false,
+                    "keep_separate_first_letter", "keep_joined_full_pinyin", 
"keep_original",
+                    "keep_none_chinese_in_joined_full_pinyin", 
"remove_duplicated_term",
+                    "fixed_pinyin_offset", "keep_separate_chinese");
+            removeIntegerDefault(properties, "limit_first_letter_length", 16);
+            canonicalizePinyinDependencies(properties);
+            return;
+        }
+
+        if (expectedType == IndexPolicyTypeEnum.TOKEN_FILTER) {
+            if ("asciifolding".equals(type)) {
+                removeBooleanDefaults(properties, false, "preserve_original");
+            } else if ("word_delimiter".equals(type)) {
+                removeBooleanDefaults(properties, true, "generate_word_parts", 
"generate_number_parts",
+                        "split_on_case_change", "split_on_numerics", 
"stem_english_possessive");
+                removeBooleanDefaults(properties, false, "catenate_words", 
"catenate_numbers",
+                        "catenate_all", "preserve_original");
+            } else if ("icu_normalizer".equals(type)) {
+                canonicalizeIcuNormalizerDefaults(properties, false);
+            }
+            return;
+        }
+
+        if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER) {
+            if ("icu_normalizer".equals(type)) {
+                canonicalizeIcuNormalizerDefaults(properties, true);
             }
+            return;
+        }
+
+        if (expectedType != IndexPolicyTypeEnum.TOKENIZER) {
+            return;
+        }
+        switch (type) {
+            case "ngram":
+                removeIntegerDefault(properties, "min_gram", 1);
+                removeIntegerDefault(properties, "max_gram", 2);
+                break;
+            case "edge_ngram":
+                removeIntegerDefault(properties, "min_gram", 1);
+                removeIntegerDefault(properties, "max_gram", 2);
+                break;
+            case "standard":
+            case "char_group":
+                removeIntegerDefault(properties, "max_token_length", 255);
+                break;
+            case "keyword":
+                removeIntegerDefault(properties, "buffer_size", 256);
+                break;
+            case "basic":
+                canonicalizeBasicExtraChars(properties);
+                break;
+            default:
+                break;
+        }
+    }
 
-            IndexPolicy policy = env.getIndexPolicyMgr().getPolicyByName(name);
-            if (policy == null || policy.getType() != expectedType) {
-                return name;
+    private static void removeBooleanDefaults(
+            TreeMap<String, String> properties, boolean defaultValue, 
String... keys) {
+        for (String key : keys) {
+            String value = properties.get(key);
+            if (value == null || !("true".equalsIgnoreCase(value) || 
"false".equalsIgnoreCase(value))) {
+                continue;
             }
-            if (policy.isInvalid()) {
-                return "invalid-policy:" + policy.getId() + ":" + 
policy.getName();
+            boolean parsed = Boolean.parseBoolean(value);
+            if (parsed == defaultValue) {
+                properties.remove(key);
+            } else {
+                properties.put(key, Boolean.toString(parsed));
             }
+        }
+    }
 
-            Map<String, String> props = policy.getProperties();
-            if (props == null || props.isEmpty()) {
-                return name;
+    private static void removeIntegerDefault(
+            TreeMap<String, String> properties, String key, int defaultValue) {
+        String value = properties.get(key);
+        if (value == null) {
+            return;
+        }
+        try {
+            int parsed = Integer.parseInt(value);
+            if (parsed == defaultValue) {
+                properties.remove(key);
+            } else {
+                properties.put(key, Integer.toString(parsed));
             }
+        } catch (NumberFormatException e) {
+            // Invalid policies keep their original identity.
+        }
+    }
 
-            // Build identity from sorted properties
-            TreeMap<String, String> sortedProps = new TreeMap<>(props);
-            if (expectedType == IndexPolicyTypeEnum.TOKENIZER
-                    && "ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) 
{
-                // This setting only limits policy creation; it does not 
change emitted tokens.
-                sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+    private static void canonicalizeIcuNormalizerDefaults(
+            TreeMap<String, String> properties, boolean hasMode) {
+        String name = properties.get("name");
+        if (name != null) {
+            String normalizedName = name.trim().toLowerCase(Locale.ROOT);
+            if ("nfkc_cf".equals(normalizedName)) {
+                properties.remove("name");
+            } else {
+                properties.put("name", normalizedName);
             }
-            return sortedProps.toString();
-        } catch (RuntimeException e) {
-            return name;
+        }
+        String filter = properties.get("unicode_set_filter");
+        if (filter != null) {
+            try {
+                UnicodeSet unicodeSet = new UnicodeSet(filter);
+                if (unicodeSet.isEmpty()) {
+                    properties.remove("unicode_set_filter");
+                } else {
+                    properties.put("unicode_set_filter", 
unicodeSet.toPattern(false));
+                }
+            } catch (IllegalArgumentException e) {
+                // Invalid policies keep their original identity.
+            }
+        }
+        if (hasMode) {
+            removeStringDefault(properties, "mode", "compose");

Review Comment:
   Fixed in 31d36ebc317. Confirmed on BE: 
`ICUNormalizerCharFilterFactory::get_normalizer` returns 
`getNFDInstance`/`getNFKDInstance` for `nfd`/`nfkd` regardless of `mode`, and 
`nfc`/`nfkc` with `UNORM2_DECOMPOSE` are the same ICU instances as NFD/NFKD.
   
   The identity now drops `mode` for `nfd`/`nfkd` and folds `nfc`+`decompose` 
to `nfd` and `nfkc`+`decompose` to `nfkd`; `nfkc_cf`+`decompose` stays a 
distinct identity because it is a different ICU instance. Added 
`testIcuNormalizerModeFollowsSelectedNormalizer` plus CREATE 
(`InvertedIndexPropertiesTest`) and ALTER (`SchemaChangeHandlerTest`) 
duplicate-rejection cases.



##########
be/src/storage/index/inverted/tokenizer/tokenizer.h:
##########
@@ -39,12 +45,131 @@ class DorisTokenizer : public Tokenizer, public 
DorisTokenStream {
 
     using Tokenizer::reset;
     // Only use the parameterless reset method
-    void reset() override { _in = _in_pending; };
+    void reset() override {
+        _in = _in_pending;
+        _source_byte_offsets.clear();

Review Comment:
   Fixed in f4f79a496e0. Reproduced first: after one 96 KiB ASCII value through 
`empty -> word_delimiter -> pinyin(ignore_pinyin_offset=false)`, `reset()` left 
512 KiB of capacity in the tokenizer's 
`_source_byte_offsets`/`_source_byte_end_offsets`.
   
   The 64 KiB high-water helper moved from `pinyin_filter.cpp` into 
`token_stream.h`, and `DorisTokenizer::reset()` now releases both vectors when 
they exceed it while keeping ordinary capacities. 
`TestResetReleasesOversizedUpstreamProvenanceScratch` covers ordinary reuse 
(capacity retained), an oversized value, then an empty reset and a small-token 
reset that emits the same tokens.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to