github-actions[bot] commented on code in PR #67918:
URL: https://github.com/apache/doris/pull/67918#discussion_r4068997985


##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -226,26 +681,321 @@ private static String resolveTokenFilterIdentity(String 
filterList) {
      * IMPORTANT: Order is preserved because filter order is semantically 
significant.
      */
     private static String resolveCharFilterIdentity(String filterList) {
+        return resolveCharFilterIdentity(filterList, false);
+    }
+
+    private static String resolveCharFilterIdentity(String filterList, boolean 
lowercaseDownstream) {
+        ArrayDeque<String> identities = new ArrayDeque<>();
+        walkCharFilters(filterList, lowercaseDownstream, identities);
+        return String.join(",", identities);
+    }
+
+    /**
+     * Resolve the chain from its last filter to its first, collecting 
identities, and return the
+     * case-folding context that a filter placed in front of the chain would 
run in.
+     */
+    private static boolean[] walkCharFilters(
+            String filterList, boolean lowercaseDownstream, Deque<String> 
identities) {
+        boolean[] foldBlockedBytes = lowercaseDownstream ? new boolean[256] : 
null;
         if (Strings.isNullOrEmpty(filterList)) {
-            return "";
+            return foldBlockedBytes;
         }
 
-        StringBuilder sb = new StringBuilder();
         String[] filters = filterList.split(",\\s*");
         // DO NOT sort - filter order is semantically significant
 
-        for (int i = 0; i < filters.length; i++) {
-            String filter = filters[i].trim();
-            if (i > 0) {
-                sb.append(",");
+        for (int i = filters.length - 1; i >= 0; --i) {
+            String filterName = filters[i].trim();
+            String filter = resolveComponentIdentity(
+                    filterName, IndexPolicyTypeEnum.CHAR_FILTER, 
foldBlockedBytes);
+            if (Strings.isNullOrEmpty(filter)) {
+                continue;
             }
+            identities.addFirst(filter);
+            foldBlockedBytes = foldBlockedBytesBefore(filterName, 
foldBlockedBytes);
+        }
+        return foldBlockedBytes;
+    }
 
-            if (IndexPolicy.BUILTIN_CHAR_FILTERS.contains(filter)) {
-                sb.append(filter);
-            } else {
-                sb.append(resolveComponentIdentity(filter, 
IndexPolicyTypeEnum.CHAR_FILTER));
+    /**
+     * Context for the filter that runs before this one: a case fold starts a 
fresh context, a
+     * char_replace filter adds the bytes it rewrites, and any other filter 
ends the context.
+     */
+    private static boolean[] foldBlockedBytesBefore(String filterName, 
boolean[] foldBlockedBytes) {
+        if (isCaseFoldingCharFilter(filterName)) {
+            return new boolean[256];
+        }
+        if (foldBlockedBytes == null) {
+            return null;
+        }
+        boolean[] sourceBytes = charReplaceSourceBytes(filterName);
+        if (sourceBytes == null) {
+            return null;
+        }
+        for (int i = 0; i < foldBlockedBytes.length; ++i) {
+            foldBlockedBytes[i] |= sourceBytes[i];
+        }
+        return foldBlockedBytes;
+    }
+
+    /** Bytes a named char_replace filter rewrites, or null for any other 
filter. */
+    private static boolean[] charReplaceSourceBytes(String filterName) {
+        IndexPolicy policy = findPolicy(filterName, 
IndexPolicyTypeEnum.CHAR_FILTER);
+        if (policy == null || policy.isInvalid() || policy.getProperties() == 
null) {
+            return null;
+        }
+        Map<String, String> properties = policy.getProperties();
+        String type = normalizeBuiltinComponentName(
+                properties.get(IndexPolicy.PROP_TYPE), 
IndexPolicyTypeEnum.CHAR_FILTER);
+        String pattern = properties.get("pattern");
+        if (!"char_replace".equals(type) || pattern == null) {
+            return null;
+        }
+        boolean[] sourceBytes = new boolean[256];
+        for (int i = 0; i < pattern.length(); ++i) {
+            char patternByte = pattern.charAt(i);
+            if (patternByte < sourceBytes.length) {
+                sourceBytes[patternByte] = true;
             }
         }
-        return sb.toString();
+        return sourceBytes;
+    }
+
+    /** The named policy when one exists with the expected type, or null. */
+    private static IndexPolicy findPolicy(String name, IndexPolicyTypeEnum 
expectedType) {
+        if (Strings.isNullOrEmpty(name)) {
+            return null;
+        }
+        try {
+            Env env = Env.getCurrentEnv();
+            if (env != null && env.getIndexPolicyMgr() != null) {
+                IndexPolicy policy = 
env.getIndexPolicyMgr().getPolicyByName(name);
+                if (policy != null && policy.getType() == expectedType) {
+                    return policy;
+                }
+            }
+        } catch (RuntimeException e) {
+            // Treat lookup failures as an unknown policy.
+        }
+        return null;
+    }
+
+    private static boolean isCaseFoldingCharFilter(String name) {
+        if (Strings.isNullOrEmpty(name)) {
+            return false;
+        }
+
+        try {
+            Env env = Env.getCurrentEnv();
+            if (env != null && env.getIndexPolicyMgr() != null) {
+                IndexPolicy policy = 
env.getIndexPolicyMgr().getPolicyByName(name);
+                if (policy != null && policy.getType() == 
IndexPolicyTypeEnum.CHAR_FILTER) {
+                    if (policy.isInvalid()) {
+                        return false;
+                    }
+                    Map<String, String> properties = policy.getProperties();
+                    if (properties != null && !properties.isEmpty()) {
+                        String type = normalizeBuiltinComponentName(
+                                properties.get(IndexPolicy.PROP_TYPE), 
IndexPolicyTypeEnum.CHAR_FILTER);

Review Comment:
   [P1] `isCaseFoldingCharFilter` treats a named `icu_normalizer` with 
`unicode_set_filter=[]` as non-folding because it checks the raw text, but BE 
parses `[]` as an empty set and uses the same unfiltered `nfkc_cf` normalizer. 
A preceding `char_replace A->a` is therefore absorbed at runtime yet retained 
in this identity, so aliases such as `lower_a,fold_empty` and `fold_empty` can 
pass CREATE/ALTER duplicate checks. Reuse the parsed/canonicalized set here and 
add the identity plus both DDL-path cases.



##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -226,26 +681,321 @@ private static String resolveTokenFilterIdentity(String 
filterList) {
      * IMPORTANT: Order is preserved because filter order is semantically 
significant.
      */
     private static String resolveCharFilterIdentity(String filterList) {
+        return resolveCharFilterIdentity(filterList, false);
+    }
+
+    private static String resolveCharFilterIdentity(String filterList, boolean 
lowercaseDownstream) {
+        ArrayDeque<String> identities = new ArrayDeque<>();
+        walkCharFilters(filterList, lowercaseDownstream, identities);
+        return String.join(",", identities);
+    }
+
+    /**
+     * Resolve the chain from its last filter to its first, collecting 
identities, and return the
+     * case-folding context that a filter placed in front of the chain would 
run in.
+     */
+    private static boolean[] walkCharFilters(
+            String filterList, boolean lowercaseDownstream, Deque<String> 
identities) {
+        boolean[] foldBlockedBytes = lowercaseDownstream ? new boolean[256] : 
null;
         if (Strings.isNullOrEmpty(filterList)) {
-            return "";
+            return foldBlockedBytes;
         }
 
-        StringBuilder sb = new StringBuilder();
         String[] filters = filterList.split(",\\s*");
         // DO NOT sort - filter order is semantically significant
 
-        for (int i = 0; i < filters.length; i++) {
-            String filter = filters[i].trim();
-            if (i > 0) {
-                sb.append(",");
+        for (int i = filters.length - 1; i >= 0; --i) {
+            String filterName = filters[i].trim();
+            String filter = resolveComponentIdentity(
+                    filterName, IndexPolicyTypeEnum.CHAR_FILTER, 
foldBlockedBytes);
+            if (Strings.isNullOrEmpty(filter)) {
+                continue;
             }
+            identities.addFirst(filter);
+            foldBlockedBytes = foldBlockedBytesBefore(filterName, 
foldBlockedBytes);
+        }
+        return foldBlockedBytes;
+    }
 
-            if (IndexPolicy.BUILTIN_CHAR_FILTERS.contains(filter)) {
-                sb.append(filter);
-            } else {
-                sb.append(resolveComponentIdentity(filter, 
IndexPolicyTypeEnum.CHAR_FILTER));
+    /**
+     * Context for the filter that runs before this one: a case fold starts a 
fresh context, a
+     * char_replace filter adds the bytes it rewrites, and any other filter 
ends the context.
+     */
+    private static boolean[] foldBlockedBytesBefore(String filterName, 
boolean[] foldBlockedBytes) {
+        if (isCaseFoldingCharFilter(filterName)) {
+            return new boolean[256];
+        }
+        if (foldBlockedBytes == null) {
+            return null;
+        }
+        boolean[] sourceBytes = charReplaceSourceBytes(filterName);
+        if (sourceBytes == null) {
+            return null;
+        }
+        for (int i = 0; i < foldBlockedBytes.length; ++i) {
+            foldBlockedBytes[i] |= sourceBytes[i];
+        }
+        return foldBlockedBytes;
+    }
+
+    /** Bytes a named char_replace filter rewrites, or null for any other 
filter. */
+    private static boolean[] charReplaceSourceBytes(String filterName) {
+        IndexPolicy policy = findPolicy(filterName, 
IndexPolicyTypeEnum.CHAR_FILTER);
+        if (policy == null || policy.isInvalid() || policy.getProperties() == 
null) {
+            return null;
+        }
+        Map<String, String> properties = policy.getProperties();
+        String type = normalizeBuiltinComponentName(
+                properties.get(IndexPolicy.PROP_TYPE), 
IndexPolicyTypeEnum.CHAR_FILTER);
+        String pattern = properties.get("pattern");
+        if (!"char_replace".equals(type) || pattern == null) {
+            return null;
+        }
+        boolean[] sourceBytes = new boolean[256];
+        for (int i = 0; i < pattern.length(); ++i) {
+            char patternByte = pattern.charAt(i);
+            if (patternByte < sourceBytes.length) {
+                sourceBytes[patternByte] = true;
             }
         }
-        return sb.toString();
+        return sourceBytes;
+    }
+
+    /** The named policy when one exists with the expected type, or null. */
+    private static IndexPolicy findPolicy(String name, IndexPolicyTypeEnum 
expectedType) {
+        if (Strings.isNullOrEmpty(name)) {
+            return null;
+        }
+        try {
+            Env env = Env.getCurrentEnv();
+            if (env != null && env.getIndexPolicyMgr() != null) {
+                IndexPolicy policy = 
env.getIndexPolicyMgr().getPolicyByName(name);
+                if (policy != null && policy.getType() == expectedType) {
+                    return policy;
+                }
+            }
+        } catch (RuntimeException e) {
+            // Treat lookup failures as an unknown policy.
+        }
+        return null;
+    }
+
+    private static boolean isCaseFoldingCharFilter(String name) {
+        if (Strings.isNullOrEmpty(name)) {
+            return false;
+        }
+
+        try {
+            Env env = Env.getCurrentEnv();
+            if (env != null && env.getIndexPolicyMgr() != null) {
+                IndexPolicy policy = 
env.getIndexPolicyMgr().getPolicyByName(name);
+                if (policy != null && policy.getType() == 
IndexPolicyTypeEnum.CHAR_FILTER) {
+                    if (policy.isInvalid()) {
+                        return false;
+                    }
+                    Map<String, String> properties = policy.getProperties();
+                    if (properties != null && !properties.isEmpty()) {
+                        String type = normalizeBuiltinComponentName(
+                                properties.get(IndexPolicy.PROP_TYPE), 
IndexPolicyTypeEnum.CHAR_FILTER);
+                        String normalizer = properties.getOrDefault("name", 
"nfkc_cf").trim();
+                        String unicodeSet = 
properties.getOrDefault("unicode_set_filter", "").trim();
+                        return "icu_normalizer".equals(type)
+                                && "nfkc_cf".equalsIgnoreCase(normalizer)
+                                && unicodeSet.isEmpty();
+                    }
+                }
+            }
+        } catch (RuntimeException e) {
+            // Fall through to built-in resolution.
+        }
+
+        return "icu_normalizer".equals(
+                normalizeBuiltinComponentName(name, 
IndexPolicyTypeEnum.CHAR_FILTER));
+    }
+
+    /** The outer char filter runs before everything else, so it takes the 
analyzer's fold context. */
+    private static String appendOuterCharFilterIdentity(
+            String analyzerIdentity, Map<String, String> properties, boolean[] 
foldBlockedBytes) {
+        String type = 
properties.get(InvertedIndexProperties.INVERTED_INDEX_PARSER_CHAR_FILTER_TYPE);
+        String pattern = 
properties.get(InvertedIndexProperties.INVERTED_INDEX_PARSER_CHAR_FILTER_PATTERN);
+        if (!"char_replace".equals(type) || Strings.isNullOrEmpty(pattern)) {
+            return analyzerIdentity;
+        }
+        String replacement = properties.getOrDefault(
+                
InvertedIndexProperties.INVERTED_INDEX_PARSER_CHAR_FILTER_REPLACEMENT, " ");
+        String canonicalPattern = canonicalizeCharReplacePattern(pattern, 
replacement, foldBlockedBytes);
+        if (canonicalPattern.isEmpty()) {
+            return analyzerIdentity;
+        }
+        return analyzerIdentity + "|outer_char_filter=char_replace:"
+                + canonicalPattern.length() + ":" + canonicalPattern + ":"
+                + replacement.length() + ":" + replacement + ";";
+    }
+
+    /**
+     * Canonicalize the ASCII pattern to the BE filter's byte set.
+     * Order, duplicate bytes, and replacements of a byte with itself do not 
change the stream.
+     */
+    private static String canonicalizeCharReplacePattern(
+            String pattern, String replacement, boolean[] foldBlockedBytes) {
+        if (replacement.length() != 1) {
+            return pattern;
+        }
+        char replacementByte = replacement.charAt(0);
+        boolean[] replacedBytes = new boolean[256];
+        for (int i = 0; i < pattern.length(); ++i) {
+            char patternByte = pattern.charAt(i);
+            if (patternByte < replacedBytes.length && patternByte != 
replacementByte) {
+                replacedBytes[patternByte] = true;
+            }
+        }
+        if (foldBlockedBytes != null && replacementByte >= 'a' && 
replacementByte <= 'z') {
+            // The downstream fold maps the upper-case byte to the replacement 
anyway, unless a
+            // filter in between rewrites either byte.
+            int upperByte = replacementByte - ('a' - 'A');
+            if (!foldBlockedBytes[upperByte] && 
!foldBlockedBytes[replacementByte]) {
+                replacedBytes[upperByte] = false;
+            }
+        }
+
+        StringBuilder canonical = new StringBuilder();
+        for (int i = 0; i < replacedBytes.length; ++i) {
+            if (replacedBytes[i]) {
+                canonical.append((char) i);
+            }
+        }
+        return canonical.toString();
+    }
+
+    private static boolean[] builtinIkFoldContext(String analyzerIdentity) {
+        return isDefaultLowercaseBuiltinIkIdentity(analyzerIdentity) ? new 
boolean[256] : null;
+    }
+
+    private static boolean isDefaultLowercaseBuiltinIkIdentity(String 
analyzerIdentity) {
+        return (IndexPolicyTypeEnum.ANALYZER.name() + 
":tokenizer=ik_smart;").equals(analyzerIdentity)
+                || (IndexPolicyTypeEnum.ANALYZER.name() + 
":tokenizer=ik_max_word;").equals(analyzerIdentity);
+    }
+
+    /**
+     * Fold context for the outer char filter of a custom analyzer, which BE 
applies before the
+     * analyzer's own char filters. Unknown or unresolvable analyzers get no 
context.

Review Comment:
   [P1] `buildAnalyzerIdentity` sends both analyzer and normalizer names 
through `customAnalyzerFoldContext`, but that helper calls `findPolicy(..., 
ANALYZER)` and returns null for NORMALIZER policies. BE applies the external 
reader before a custom normalizer's keyword-plus-lowercase chain, so two 
normalizer aliases with the same pipeline, with and without outer `char_replace 
A->a`, emit the same terms/offsets while this code appends different 
identities. Derive fold context for normalizers (or conservatively 
retain/validate the suffix) and cover CREATE/ALTER duplicate rejection.



##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -150,47 +221,433 @@ private static String 
buildIdentityFromPolicyProperties(IndexPolicyTypeEnum type
      * Resolve a component (tokenizer) to its identity.
      */
     private static String resolveComponentIdentity(String name, 
IndexPolicyTypeEnum expectedType) {
+        return resolveComponentIdentity(name, expectedType, null);
+    }
+
+    /**
+     * {@code foldBlockedBytes} is the case-folding context of a char filter: 
null without a
+     * downstream fold, otherwise the bytes that filters between this one and 
the fold rewrite.
+     */
+    private static String resolveComponentIdentity(
+            String name, IndexPolicyTypeEnum expectedType, boolean[] 
foldBlockedBytes) {
         if (Strings.isNullOrEmpty(name)) {
             return "";
         }
 
-        // Check if it's a built-in component
-        if (expectedType == IndexPolicyTypeEnum.TOKENIZER
-                && IndexPolicy.BUILTIN_TOKENIZERS.contains(name)) {
-            return name;
+        // Existing named policies take precedence over built-ins for upgrade 
compatibility.
+        try {
+            Env env = Env.getCurrentEnv();
+            if (env != null && env.getIndexPolicyMgr() != null) {
+                IndexPolicy policy = 
env.getIndexPolicyMgr().getPolicyByName(name);
+                if (policy != null && policy.getType() == expectedType) {
+                    if (policy.isInvalid()) {
+                        return "invalid-policy:" + policy.getId() + ":" + 
policy.getName();
+                    }
+                    Map<String, String> props = policy.getProperties();
+                    if (props != null && !props.isEmpty()) {
+                        TreeMap<String, String> sortedProps = new 
TreeMap<>(props);
+                        String type = sortedProps.get(IndexPolicy.PROP_TYPE);
+                        String normalizedType = 
normalizeBuiltinComponentName(type, expectedType);
+                        if (normalizedType != null) {
+                            if ("empty".equals(normalizedType)) {
+                                return "";
+                            }
+                            sortedProps.put(IndexPolicy.PROP_TYPE, 
normalizedType);
+                            canonicalizeEffectiveComponentProperties(
+                                    sortedProps, normalizedType, expectedType);
+                            if (sortedProps.size() == 1) {
+                                return normalizedType;
+                            }
+                        }
+                        if (expectedType == IndexPolicyTypeEnum.TOKENIZER
+                                && 
"ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+                            // This setting only limits policy creation; it 
does not change emitted tokens.
+                            sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+                        }
+                        if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER
+                                && 
"char_replace".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+                            String replacement = 
sortedProps.getOrDefault("replacement", " ");
+                            String pattern = canonicalizeCharReplacePattern(
+                                    sortedProps.get("pattern"), replacement, 
foldBlockedBytes);
+                            if (pattern.isEmpty()) {
+                                return "";
+                            }
+                            sortedProps.put("pattern", pattern);
+                            sortedProps.put("replacement", replacement);
+                        }
+                        if (normalizedType != null && sortedProps.size() == 1) 
{
+                            return normalizedType;
+                        }
+                        return sortedProps.toString();
+                    }
+                }
+            }
+        } catch (RuntimeException e) {
+            // Fall through to built-in resolution or the original name.
+        }
+
+        String normalizedName = normalizeBuiltinComponentName(name, 
expectedType);
+        return "empty".equals(normalizedName) ? "" : normalizedName == null ? 
name : normalizedName;
+    }
+
+    private static void canonicalizeEffectiveComponentProperties(
+            TreeMap<String, String> properties, String type, 
IndexPolicyTypeEnum expectedType) {
+        if ("pinyin".equals(type)) {
+            removeBooleanDefaults(properties, true,
+                    "keep_first_letter", "keep_full_pinyin", 
"keep_none_chinese",
+                    "keep_none_chinese_together", 
"keep_none_chinese_in_first_letter",
+                    "lowercase", "trim_whitespace", "ignore_pinyin_offset",
+                    "none_chinese_pinyin_tokenize");
+            removeBooleanDefaults(properties, false,
+                    "keep_separate_first_letter", "keep_joined_full_pinyin", 
"keep_original",
+                    "keep_none_chinese_in_joined_full_pinyin", 
"remove_duplicated_term",

Review Comment:
   [P1] `keep_none_chinese_in_joined_full_pinyin` remains in the Pinyin 
identity even when `keep_joined_full_pinyin=false`. BE only uses the former to 
append `full_pinyin_buffer`, and emits that buffer only under the latter gate; 
with joined output disabled this setting cannot affect terms, positions, or 
offsets. A type-only Pinyin policy and the same policy with this flag enabled 
therefore get different identities and can admit equivalent duplicate indexes. 
Remove this gated property when joined output is disabled and add 
identity/CREATE/ALTER coverage.



##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -226,26 +681,321 @@ private static String resolveTokenFilterIdentity(String 
filterList) {
      * IMPORTANT: Order is preserved because filter order is semantically 
significant.
      */
     private static String resolveCharFilterIdentity(String filterList) {
+        return resolveCharFilterIdentity(filterList, false);
+    }
+
+    private static String resolveCharFilterIdentity(String filterList, boolean 
lowercaseDownstream) {
+        ArrayDeque<String> identities = new ArrayDeque<>();
+        walkCharFilters(filterList, lowercaseDownstream, identities);
+        return String.join(",", identities);
+    }
+
+    /**
+     * Resolve the chain from its last filter to its first, collecting 
identities, and return the
+     * case-folding context that a filter placed in front of the chain would 
run in.
+     */
+    private static boolean[] walkCharFilters(
+            String filterList, boolean lowercaseDownstream, Deque<String> 
identities) {
+        boolean[] foldBlockedBytes = lowercaseDownstream ? new boolean[256] : 
null;
         if (Strings.isNullOrEmpty(filterList)) {
-            return "";
+            return foldBlockedBytes;
         }
 
-        StringBuilder sb = new StringBuilder();
         String[] filters = filterList.split(",\\s*");
         // DO NOT sort - filter order is semantically significant
 
-        for (int i = 0; i < filters.length; i++) {
-            String filter = filters[i].trim();
-            if (i > 0) {
-                sb.append(",");
+        for (int i = filters.length - 1; i >= 0; --i) {
+            String filterName = filters[i].trim();
+            String filter = resolveComponentIdentity(
+                    filterName, IndexPolicyTypeEnum.CHAR_FILTER, 
foldBlockedBytes);
+            if (Strings.isNullOrEmpty(filter)) {
+                continue;
             }
+            identities.addFirst(filter);
+            foldBlockedBytes = foldBlockedBytesBefore(filterName, 
foldBlockedBytes);
+        }
+        return foldBlockedBytes;
+    }
 
-            if (IndexPolicy.BUILTIN_CHAR_FILTERS.contains(filter)) {
-                sb.append(filter);
-            } else {
-                sb.append(resolveComponentIdentity(filter, 
IndexPolicyTypeEnum.CHAR_FILTER));
+    /**
+     * Context for the filter that runs before this one: a case fold starts a 
fresh context, a
+     * char_replace filter adds the bytes it rewrites, and any other filter 
ends the context.
+     */
+    private static boolean[] foldBlockedBytesBefore(String filterName, 
boolean[] foldBlockedBytes) {
+        if (isCaseFoldingCharFilter(filterName)) {
+            return new boolean[256];
+        }
+        if (foldBlockedBytes == null) {
+            return null;
+        }
+        boolean[] sourceBytes = charReplaceSourceBytes(filterName);
+        if (sourceBytes == null) {
+            return null;
+        }
+        for (int i = 0; i < foldBlockedBytes.length; ++i) {
+            foldBlockedBytes[i] |= sourceBytes[i];
+        }
+        return foldBlockedBytes;
+    }
+
+    /** Bytes a named char_replace filter rewrites, or null for any other 
filter. */
+    private static boolean[] charReplaceSourceBytes(String filterName) {
+        IndexPolicy policy = findPolicy(filterName, 
IndexPolicyTypeEnum.CHAR_FILTER);
+        if (policy == null || policy.isInvalid() || policy.getProperties() == 
null) {
+            return null;
+        }
+        Map<String, String> properties = policy.getProperties();
+        String type = normalizeBuiltinComponentName(
+                properties.get(IndexPolicy.PROP_TYPE), 
IndexPolicyTypeEnum.CHAR_FILTER);
+        String pattern = properties.get("pattern");
+        if (!"char_replace".equals(type) || pattern == null) {
+            return null;
+        }
+        boolean[] sourceBytes = new boolean[256];
+        for (int i = 0; i < pattern.length(); ++i) {
+            char patternByte = pattern.charAt(i);
+            if (patternByte < sourceBytes.length) {
+                sourceBytes[patternByte] = true;
             }
         }
-        return sb.toString();
+        return sourceBytes;
+    }
+
+    /** The named policy when one exists with the expected type, or null. */
+    private static IndexPolicy findPolicy(String name, IndexPolicyTypeEnum 
expectedType) {
+        if (Strings.isNullOrEmpty(name)) {
+            return null;
+        }
+        try {
+            Env env = Env.getCurrentEnv();
+            if (env != null && env.getIndexPolicyMgr() != null) {
+                IndexPolicy policy = 
env.getIndexPolicyMgr().getPolicyByName(name);
+                if (policy != null && policy.getType() == expectedType) {
+                    return policy;
+                }
+            }
+        } catch (RuntimeException e) {
+            // Treat lookup failures as an unknown policy.
+        }
+        return null;
+    }
+
+    private static boolean isCaseFoldingCharFilter(String name) {
+        if (Strings.isNullOrEmpty(name)) {
+            return false;
+        }
+
+        try {
+            Env env = Env.getCurrentEnv();
+            if (env != null && env.getIndexPolicyMgr() != null) {
+                IndexPolicy policy = 
env.getIndexPolicyMgr().getPolicyByName(name);
+                if (policy != null && policy.getType() == 
IndexPolicyTypeEnum.CHAR_FILTER) {
+                    if (policy.isInvalid()) {
+                        return false;
+                    }
+                    Map<String, String> properties = policy.getProperties();
+                    if (properties != null && !properties.isEmpty()) {
+                        String type = normalizeBuiltinComponentName(
+                                properties.get(IndexPolicy.PROP_TYPE), 
IndexPolicyTypeEnum.CHAR_FILTER);
+                        String normalizer = properties.getOrDefault("name", 
"nfkc_cf").trim();
+                        String unicodeSet = 
properties.getOrDefault("unicode_set_filter", "").trim();
+                        return "icu_normalizer".equals(type)
+                                && "nfkc_cf".equalsIgnoreCase(normalizer)
+                                && unicodeSet.isEmpty();
+                    }
+                }
+            }
+        } catch (RuntimeException e) {
+            // Fall through to built-in resolution.
+        }
+
+        return "icu_normalizer".equals(
+                normalizeBuiltinComponentName(name, 
IndexPolicyTypeEnum.CHAR_FILTER));
+    }
+
+    /** The outer char filter runs before everything else, so it takes the 
analyzer's fold context. */
+    private static String appendOuterCharFilterIdentity(
+            String analyzerIdentity, Map<String, String> properties, boolean[] 
foldBlockedBytes) {
+        String type = 
properties.get(InvertedIndexProperties.INVERTED_INDEX_PARSER_CHAR_FILTER_TYPE);
+        String pattern = 
properties.get(InvertedIndexProperties.INVERTED_INDEX_PARSER_CHAR_FILTER_PATTERN);
+        if (!"char_replace".equals(type) || Strings.isNullOrEmpty(pattern)) {
+            return analyzerIdentity;
+        }
+        String replacement = properties.getOrDefault(
+                
InvertedIndexProperties.INVERTED_INDEX_PARSER_CHAR_FILTER_REPLACEMENT, " ");
+        String canonicalPattern = canonicalizeCharReplacePattern(pattern, 
replacement, foldBlockedBytes);
+        if (canonicalPattern.isEmpty()) {
+            return analyzerIdentity;
+        }
+        return analyzerIdentity + "|outer_char_filter=char_replace:"
+                + canonicalPattern.length() + ":" + canonicalPattern + ":"
+                + replacement.length() + ":" + replacement + ";";
+    }
+
+    /**
+     * Canonicalize the ASCII pattern to the BE filter's byte set.
+     * Order, duplicate bytes, and replacements of a byte with itself do not 
change the stream.
+     */
+    private static String canonicalizeCharReplacePattern(
+            String pattern, String replacement, boolean[] foldBlockedBytes) {
+        if (replacement.length() != 1) {
+            return pattern;
+        }
+        char replacementByte = replacement.charAt(0);
+        boolean[] replacedBytes = new boolean[256];
+        for (int i = 0; i < pattern.length(); ++i) {
+            char patternByte = pattern.charAt(i);
+            if (patternByte < replacedBytes.length && patternByte != 
replacementByte) {
+                replacedBytes[patternByte] = true;
+            }
+        }
+        if (foldBlockedBytes != null && replacementByte >= 'a' && 
replacementByte <= 'z') {
+            // The downstream fold maps the upper-case byte to the replacement 
anyway, unless a
+            // filter in between rewrites either byte.
+            int upperByte = replacementByte - ('a' - 'A');
+            if (!foldBlockedBytes[upperByte] && 
!foldBlockedBytes[replacementByte]) {
+                replacedBytes[upperByte] = false;
+            }
+        }
+
+        StringBuilder canonical = new StringBuilder();
+        for (int i = 0; i < replacedBytes.length; ++i) {
+            if (replacedBytes[i]) {
+                canonical.append((char) i);
+            }
+        }
+        return canonical.toString();
+    }
+
+    private static boolean[] builtinIkFoldContext(String analyzerIdentity) {
+        return isDefaultLowercaseBuiltinIkIdentity(analyzerIdentity) ? new 
boolean[256] : null;
+    }
+
+    private static boolean isDefaultLowercaseBuiltinIkIdentity(String 
analyzerIdentity) {
+        return (IndexPolicyTypeEnum.ANALYZER.name() + 
":tokenizer=ik_smart;").equals(analyzerIdentity)
+                || (IndexPolicyTypeEnum.ANALYZER.name() + 
":tokenizer=ik_max_word;").equals(analyzerIdentity);
+    }
+
+    /**
+     * Fold context for the outer char filter of a custom analyzer, which BE 
applies before the
+     * analyzer's own char filters. Unknown or unresolvable analyzers get no 
context.
+     */
+    private static boolean[] customAnalyzerFoldContext(String analyzerName) {
+        if (IndexPolicy.BUILTIN_ANALYZERS.contains(analyzerName)
+                || IndexPolicy.BUILTIN_NORMALIZERS.contains(analyzerName)) {
+            return null;
+        }
+        IndexPolicy policy = findPolicy(analyzerName, 
IndexPolicyTypeEnum.ANALYZER);
+        if (policy == null || policy.isInvalid() || policy.getProperties() == 
null
+                || policy.getProperties().isEmpty()) {
+            return null;
+        }
+        Map<String, String> properties = policy.getProperties();
+        try {
+            String tokenizerIdentity = resolveComponentIdentity(
+                    properties.get(IndexPolicy.PROP_TOKENIZER), 
IndexPolicyTypeEnum.TOKENIZER);
+            return 
walkCharFilters(properties.get(IndexPolicy.PROP_CHAR_FILTER),
+                    foldsAsciiCaseAfterCharFilters(properties, 
tokenizerIdentity), new ArrayDeque<>());
+        } catch (RuntimeException e) {
+            return null;
+        }
+    }
+
+    /**
+     * Whether the tokenizer and token filters emit the same tokens for an 
ASCII letter of either
+     * case, so a char filter that only lowercases such a letter cannot change 
the output.
+     */
+    private static boolean foldsAsciiCaseAfterCharFilters(
+            Map<String, String> properties, String tokenizerIdentity) {
+        if ("ik_smart".equals(tokenizerIdentity) || 
"ik_max_word".equals(tokenizerIdentity)) {
+            return true;

Review Comment:
   [P1] `foldsAsciiCaseAfterCharFilters` requires the first effective token 
filter to be `lowercase`, but a valid `keyword -> asciifolding -> lowercase` 
chain is still ASCII-case transparent: BE's ASCII folding leaves A/a unchanged 
and the following lowercase produces the same term after or without outer 
`char_replace A->a`. The helper returns false for this ordering, so equivalent 
analyzer aliases get different identities and can pass CREATE/ALTER duplicate 
checks. Carry the proven fold context through ASCII-transparent filters (and 
cover this ordering, including default ICU case-folding where applicable).



##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -150,47 +221,433 @@ private static String 
buildIdentityFromPolicyProperties(IndexPolicyTypeEnum type
      * Resolve a component (tokenizer) to its identity.
      */
     private static String resolveComponentIdentity(String name, 
IndexPolicyTypeEnum expectedType) {
+        return resolveComponentIdentity(name, expectedType, null);
+    }
+
+    /**
+     * {@code foldBlockedBytes} is the case-folding context of a char filter: 
null without a
+     * downstream fold, otherwise the bytes that filters between this one and 
the fold rewrite.
+     */
+    private static String resolveComponentIdentity(
+            String name, IndexPolicyTypeEnum expectedType, boolean[] 
foldBlockedBytes) {
         if (Strings.isNullOrEmpty(name)) {
             return "";
         }
 
-        // Check if it's a built-in component
-        if (expectedType == IndexPolicyTypeEnum.TOKENIZER
-                && IndexPolicy.BUILTIN_TOKENIZERS.contains(name)) {
-            return name;
+        // Existing named policies take precedence over built-ins for upgrade 
compatibility.
+        try {
+            Env env = Env.getCurrentEnv();
+            if (env != null && env.getIndexPolicyMgr() != null) {
+                IndexPolicy policy = 
env.getIndexPolicyMgr().getPolicyByName(name);
+                if (policy != null && policy.getType() == expectedType) {
+                    if (policy.isInvalid()) {
+                        return "invalid-policy:" + policy.getId() + ":" + 
policy.getName();
+                    }
+                    Map<String, String> props = policy.getProperties();
+                    if (props != null && !props.isEmpty()) {
+                        TreeMap<String, String> sortedProps = new 
TreeMap<>(props);
+                        String type = sortedProps.get(IndexPolicy.PROP_TYPE);
+                        String normalizedType = 
normalizeBuiltinComponentName(type, expectedType);
+                        if (normalizedType != null) {
+                            if ("empty".equals(normalizedType)) {
+                                return "";
+                            }
+                            sortedProps.put(IndexPolicy.PROP_TYPE, 
normalizedType);
+                            canonicalizeEffectiveComponentProperties(
+                                    sortedProps, normalizedType, expectedType);
+                            if (sortedProps.size() == 1) {
+                                return normalizedType;
+                            }
+                        }
+                        if (expectedType == IndexPolicyTypeEnum.TOKENIZER
+                                && 
"ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+                            // This setting only limits policy creation; it 
does not change emitted tokens.
+                            sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+                        }
+                        if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER
+                                && 
"char_replace".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+                            String replacement = 
sortedProps.getOrDefault("replacement", " ");
+                            String pattern = canonicalizeCharReplacePattern(
+                                    sortedProps.get("pattern"), replacement, 
foldBlockedBytes);
+                            if (pattern.isEmpty()) {
+                                return "";
+                            }
+                            sortedProps.put("pattern", pattern);
+                            sortedProps.put("replacement", replacement);
+                        }
+                        if (normalizedType != null && sortedProps.size() == 1) 
{
+                            return normalizedType;
+                        }
+                        return sortedProps.toString();
+                    }
+                }
+            }
+        } catch (RuntimeException e) {
+            // Fall through to built-in resolution or the original name.
+        }
+
+        String normalizedName = normalizeBuiltinComponentName(name, 
expectedType);
+        return "empty".equals(normalizedName) ? "" : normalizedName == null ? 
name : normalizedName;
+    }
+
+    private static void canonicalizeEffectiveComponentProperties(
+            TreeMap<String, String> properties, String type, 
IndexPolicyTypeEnum expectedType) {
+        if ("pinyin".equals(type)) {
+            removeBooleanDefaults(properties, true,
+                    "keep_first_letter", "keep_full_pinyin", 
"keep_none_chinese",
+                    "keep_none_chinese_together", 
"keep_none_chinese_in_first_letter",
+                    "lowercase", "trim_whitespace", "ignore_pinyin_offset",
+                    "none_chinese_pinyin_tokenize");
+            removeBooleanDefaults(properties, false,
+                    "keep_separate_first_letter", "keep_joined_full_pinyin", 
"keep_original",
+                    "keep_none_chinese_in_joined_full_pinyin", 
"remove_duplicated_term",
+                    "fixed_pinyin_offset", "keep_separate_chinese");
+            removeIntegerDefault(properties, "limit_first_letter_length", 16);
+            canonicalizePinyinDependencies(properties);
+            return;
+        }
+
+        if (expectedType == IndexPolicyTypeEnum.TOKEN_FILTER) {
+            if ("asciifolding".equals(type)) {
+                removeBooleanDefaults(properties, false, "preserve_original");
+            } else if ("word_delimiter".equals(type)) {
+                removeBooleanDefaults(properties, true, "generate_word_parts", 
"generate_number_parts",
+                        "split_on_case_change", "split_on_numerics", 
"stem_english_possessive");
+                removeBooleanDefaults(properties, false, "catenate_words", 
"catenate_numbers",
+                        "catenate_all", "preserve_original");
+                canonicalizeWordSet(properties, "protected_words");
+                canonicalizeTypeTable(properties);
+            } else if ("icu_normalizer".equals(type)) {
+                canonicalizeIcuNormalizerDefaults(properties, false);
+            }
+            return;
+        }
+
+        if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER) {
+            if ("icu_normalizer".equals(type)) {
+                canonicalizeIcuNormalizerDefaults(properties, true);
+            }
+            return;
         }
 
-        // For custom component, get its properties
+        if (expectedType != IndexPolicyTypeEnum.TOKENIZER) {
+            return;
+        }
+        switch (type) {
+            case "ngram":
+            case "edge_ngram":
+                removeIntegerDefault(properties, "min_gram", 1);
+                removeIntegerDefault(properties, "max_gram", 2);
+                canonicalizeWordSet(properties, "token_chars");
+                canonicalizeCustomTokenChars(properties);
+                break;
+            case "standard":
+                removeIntegerDefault(properties, "max_token_length", 255);
+                break;
+            case "char_group":
+                removeIntegerDefault(properties, "max_token_length", 255);
+                canonicalizeTokenizeOnChars(properties);
+                break;
+            case "keyword":
+                // BE only range-checks buffer_size; the emitted term is 
always capped by a constant.
+                properties.remove("buffer_size");
+                break;
+            case "basic":
+                canonicalizeBasicExtraChars(properties);
+                break;
+            default:
+                break;
+        }
+    }
+
+    private static void removeBooleanDefaults(
+            TreeMap<String, String> properties, boolean defaultValue, 
String... keys) {
+        for (String key : keys) {
+            String value = properties.get(key);
+            if (value == null || !("true".equalsIgnoreCase(value) || 
"false".equalsIgnoreCase(value))) {
+                continue;
+            }
+            boolean parsed = Boolean.parseBoolean(value);
+            if (parsed == defaultValue) {
+                properties.remove(key);
+            } else {
+                properties.put(key, Boolean.toString(parsed));
+            }
+        }
+    }
+
+    private static void removeIntegerDefault(
+            TreeMap<String, String> properties, String key, int defaultValue) {
+        String value = properties.get(key);
+        if (value == null) {
+            return;
+        }
         try {
-            Env env = Env.getCurrentEnv();
-            if (env == null || env.getIndexPolicyMgr() == null) {
-                return name;
+            int parsed = Integer.parseInt(value);
+            if (parsed == defaultValue) {
+                properties.remove(key);
+            } else {
+                properties.put(key, Integer.toString(parsed));
             }
+        } catch (NumberFormatException e) {
+            // Invalid policies keep their original identity.
+        }
+    }
 
-            IndexPolicy policy = env.getIndexPolicyMgr().getPolicyByName(name);
-            if (policy == null || policy.getType() != expectedType) {
-                return name;
+    private static void canonicalizeIcuNormalizerDefaults(
+            TreeMap<String, String> properties, boolean hasMode) {
+        String name = properties.get("name");
+        if (name != null) {
+            String normalizedName = name.trim().toLowerCase(Locale.ROOT);
+            if ("nfkc_cf".equals(normalizedName)) {
+                properties.remove("name");
+            } else {
+                properties.put("name", normalizedName);
             }
-            if (policy.isInvalid()) {
-                return "invalid-policy:" + policy.getId() + ":" + 
policy.getName();
+        }
+        String filter = properties.get("unicode_set_filter");
+        if (filter != null) {
+            try {
+                UnicodeSet unicodeSet = new UnicodeSet(filter);
+                if (unicodeSet.isEmpty()) {
+                    properties.remove("unicode_set_filter");
+                } else {
+                    properties.put("unicode_set_filter", 
unicodeSet.toPattern(false));
+                }
+            } catch (IllegalArgumentException e) {
+                // Invalid policies keep their original identity.
+            }
+        }
+        if (hasMode) {
+            canonicalizeIcuNormalizerMode(properties);
+        }
+    }
+
+    private static void canonicalizeIcuNormalizerMode(TreeMap<String, String> 
properties) {
+        removeStringDefault(properties, "mode", "compose");
+        if (!"decompose".equals(properties.get("mode"))) {
+            return;
+        }
+        // BE ignores mode for nfd/nfkd, and nfc/nfkc in decompose mode are 
the same ICU instances.
+        String name = properties.get("name");
+        if ("nfc".equals(name) || "nfd".equals(name)) {
+            properties.put("name", "nfd");
+            properties.remove("mode");
+        } else if ("nfkc".equals(name) || "nfkd".equals(name)) {
+            properties.put("name", "nfkd");
+            properties.remove("mode");
+        }
+    }
+
+    // BE reads these settings as unordered sets of trimmed, non-empty words.
+    private static void canonicalizeWordSet(TreeMap<String, String> 
properties, String key) {
+        String value = properties.get(key);
+        if (value == null) {
+            return;
+        }
+        TreeSet<String> words = new TreeSet<>();
+        for (String word : value.split(",")) {
+            String trimmed = trimAsciiWhitespace(word);
+            if (!trimmed.isEmpty()) {
+                words.add(trimmed);
             }
+        }
+        if (words.isEmpty()) {
+            properties.remove(key);
+        } else {
+            properties.put(key, String.join(",", words));
+        }
+    }
 
-            Map<String, String> props = policy.getProperties();
-            if (props == null || props.isEmpty()) {
-                return name;
+    // BE matches custom token characters as a code point set.
+    private static void canonicalizeCustomTokenChars(TreeMap<String, String> 
properties) {
+        String value = properties.get("custom_token_chars");
+        if (value == null) {
+            return;
+        }
+        StringBuilder canonical = new StringBuilder();
+        
value.codePoints().distinct().sorted().forEach(canonical::appendCodePoint);
+        properties.put("custom_token_chars", canonical.toString());
+    }
+
+    // BE collects tokenize_on_chars entries into sets, so order and repeats 
do not matter.
+    private static void canonicalizeTokenizeOnChars(TreeMap<String, String> 
properties) {
+        List<String> entries = 
parseEntryList(properties.get("tokenize_on_chars"));
+        if (entries == null) {
+            return;
+        }
+        putEntryList(properties, "tokenize_on_chars", new TreeSet<>(entries));
+    }
+
+    // BE builds a per-character type map where a later rule for the same 
character wins.
+    private static void canonicalizeTypeTable(TreeMap<String, String> 
properties) {
+        List<String> rules = parseEntryList(properties.get("type_table"));
+        if (rules == null) {
+            return;
+        }
+        TreeMap<Integer, String> types = new TreeMap<>();
+        for (String rule : rules) {
+            int arrow = rule.lastIndexOf("=>");
+            if (arrow < 0 || rule.indexOf('\n') >= 0 || rule.indexOf('\r') >= 
0) {
+                return;
             }
+            String character = trimAsciiWhitespace(rule.substring(0, arrow));
+            String type = trimAsciiWhitespace(rule.substring(arrow + 2));
+            // Escaped characters keep the original identity rather than 
reproducing BE unescaping.
+            if (character.indexOf('\\') >= 0 || character.codePointCount(0, 
character.length()) != 1
+                    || !WORD_DELIMITER_TYPES.contains(type)) {
+                return;
+            }
+            types.put(character.codePointAt(0), type);
+        }
+        List<String> canonicalRules = new ArrayList<>();
+        for (Map.Entry<Integer, String> entry : types.entrySet()) {
+            canonicalRules.add(new String(Character.toChars(entry.getKey())) + 
"=>" + entry.getValue());
+        }
+        putEntryList(properties, "type_table", canonicalRules);
+    }
 
-            // Build identity from sorted properties
-            TreeMap<String, String> sortedProps = new TreeMap<>(props);
-            if (expectedType == IndexPolicyTypeEnum.TOKENIZER
-                    && "ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) 
{
-                // This setting only limits policy creation; it does not 
change emitted tokens.
-                sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+    /** Parse a bracketed entry list as BE does, or return null for a 
malformed list. */
+    private static List<String> parseEntryList(String value) {
+        if (value == null) {
+            return null;
+        }
+        List<String> entries = new ArrayList<>();
+        String trimmed = trimAsciiWhitespace(value);
+        if (trimmed.isEmpty()) {
+            return entries;
+        }
+        for (String item : ENTRY_SEPARATOR.split(trimmed)) {
+            String entry = trimAsciiWhitespace(item);
+            if (entry.length() < 2 || entry.charAt(0) != '[' || 
entry.charAt(entry.length() - 1) != ']') {
+                return null;
             }
-            return sortedProps.toString();
-        } catch (RuntimeException e) {
-            return name;
+            String content = entry.substring(1, entry.length() - 1);
+            if (!content.isEmpty()) {
+                entries.add(content);
+            }
+        }
+        return entries;
+    }
+
+    private static void putEntryList(TreeMap<String, String> properties, 
String key, Collection<String> entries) {
+        if (entries.isEmpty()) {
+            properties.remove(key);
+            return;
+        }
+        StringBuilder canonical = new StringBuilder();
+        for (String entry : entries) {
+            if (canonical.length() > 0) {
+                canonical.append(",");
+            }
+            canonical.append("[").append(entry).append("]");
+        }
+        properties.put(key, canonical.toString());
+    }
+
+    // Trim the same ASCII whitespace that BE trims.
+    private static String trimAsciiWhitespace(String value) {
+        int begin = 0;
+        int end = value.length();
+        while (begin < end && isAsciiWhitespace(value.charAt(begin))) {
+            ++begin;
         }
+        while (end > begin && isAsciiWhitespace(value.charAt(end - 1))) {
+            --end;
+        }
+        return value.substring(begin, end);
+    }
+
+    private static boolean isAsciiWhitespace(char value) {
+        return value == ' ' || (value >= '\t' && value <= '\r');
+    }
+
+    private static void canonicalizeBasicExtraChars(TreeMap<String, String> 
properties) {
+        String extraChars = properties.get("extra_chars");
+        if (extraChars == null) {
+            return;
+        }
+        boolean[] present = new boolean[128];
+        for (int i = 0; i < extraChars.length(); ++i) {
+            char value = extraChars.charAt(i);
+            if (value >= present.length) {
+                return;
+            }
+            present[value] = true;
+        }
+        StringBuilder canonical = new StringBuilder();
+        for (int i = 0; i < present.length; ++i) {
+            if (present[i]) {
+                canonical.append((char) i);
+            }
+        }
+        if (canonical.length() == 0) {
+            properties.remove("extra_chars");
+        } else {
+            properties.put("extra_chars", canonical.toString());
+        }
+    }
+
+    private static void canonicalizePinyinDependencies(TreeMap<String, String> 
properties) {
+        Boolean keepFirstLetter = effectiveBoolean(properties, 
"keep_first_letter", true);
+        if (Boolean.FALSE.equals(keepFirstLetter)) {
+            properties.remove("limit_first_letter_length");
+            properties.remove("keep_none_chinese_in_first_letter");
+        }
+
+        Boolean keepNoneChinese = effectiveBoolean(properties, 
"keep_none_chinese", true);
+        Boolean keepNoneChineseTogether = effectiveBoolean(properties, 
"keep_none_chinese_together", true);
+        Boolean noneChinesePinyinTokenize = effectiveBoolean(properties, 
"none_chinese_pinyin_tokenize", true);
+        if (Boolean.FALSE.equals(keepNoneChinese)) {

Review Comment:
   [P1] `none_chinese_pinyin_tokenize` remains significant when 
`keep_none_chinese_together=false`, but BE's non-together ASCII path emits each 
alphanumeric immediately and never calls the `processAsciiBuffer`/`parseBuff` 
branch that reads this setting. Two Pinyin aliases differing only in this 
property therefore produce the same terms/positions/offsets yet get different 
identities and can pass CREATE/ALTER duplicate checks. Remove it whenever the 
non-together path bypasses that consumer and add tokenizer/token-filter 
identity plus both DDL-path coverage.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to