github-actions[bot] commented on code in PR #67918:
URL: https://github.com/apache/doris/pull/67918#discussion_r4079155560
##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -150,47 +302,589 @@ private static String
buildIdentityFromPolicyProperties(IndexPolicyTypeEnum type
* Resolve a component (tokenizer) to its identity.
*/
private static String resolveComponentIdentity(String name,
IndexPolicyTypeEnum expectedType) {
+ return resolveComponentIdentity(name, expectedType, null);
+ }
+
+ /** {@code fold} is the case-folding context of a char filter, or null
without a downstream fold. */
+ private static String resolveComponentIdentity(
+ String name, IndexPolicyTypeEnum expectedType, FoldContext fold) {
if (Strings.isNullOrEmpty(name)) {
return "";
}
- // Check if it's a built-in component
- if (expectedType == IndexPolicyTypeEnum.TOKENIZER
- && IndexPolicy.BUILTIN_TOKENIZERS.contains(name)) {
- return name;
+ // Existing named policies take precedence over built-ins for upgrade
compatibility.
+ try {
+ Env env = Env.getCurrentEnv();
+ if (env != null && env.getIndexPolicyMgr() != null) {
+ IndexPolicy policy =
env.getIndexPolicyMgr().getPolicyByName(name);
+ if (policy != null && policy.getType() == expectedType) {
+ if (policy.isInvalid()) {
+ return "invalid-policy:" + policy.getId() + ":" +
policy.getName();
+ }
+ Map<String, String> props = policy.getProperties();
+ if (props != null && !props.isEmpty()) {
+ TreeMap<String, String> sortedProps = new
TreeMap<>(props);
+ String type = sortedProps.get(IndexPolicy.PROP_TYPE);
+ String normalizedType =
normalizeBuiltinComponentName(type, expectedType);
+ if (normalizedType != null) {
+ if ("empty".equals(normalizedType)) {
+ return "";
+ }
+ sortedProps.put(IndexPolicy.PROP_TYPE,
normalizedType);
+ canonicalizeEffectiveComponentProperties(
+ sortedProps, normalizedType, expectedType);
+ if (sortedProps.size() == 1) {
+ return normalizedType;
+ }
+ }
+ if (expectedType == IndexPolicyTypeEnum.TOKENIZER
+ &&
"ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+ // This setting only limits policy creation; it
does not change emitted tokens.
+ sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+ }
+ if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER
+ &&
CHAR_REPLACE_FILTER.equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+ String replacement = sortedProps.getOrDefault(
+ PROP_REPLACEMENT,
CHAR_REPLACE_DEFAULT_REPLACEMENT);
+ String pattern = canonicalizeCharReplacePattern(
+ sortedProps.getOrDefault(PROP_PATTERN,
CHAR_REPLACE_DEFAULT_PATTERN),
+ replacement, fold);
+ if (pattern.isEmpty()) {
+ return "";
+ }
+ if (isCharReplaceDefault(pattern, replacement,
fold)) {
+ // Restating the factory defaults is the bare
built-in reference.
+ sortedProps.remove(PROP_PATTERN);
+ sortedProps.remove(PROP_REPLACEMENT);
+ } else {
+ sortedProps.put(PROP_PATTERN, pattern);
+ sortedProps.put(PROP_REPLACEMENT, replacement);
+ }
+ }
+ if (normalizedType != null && sortedProps.size() == 1)
{
+ return normalizedType;
+ }
+ return sortedProps.toString();
+ }
+ }
+ }
+ } catch (RuntimeException e) {
+ // Fall through to built-in resolution or the original name.
+ }
+
+ String normalizedName = normalizeBuiltinComponentName(name,
expectedType);
+ return "empty".equals(normalizedName) ? "" : normalizedName == null ?
name : normalizedName;
+ }
+
+ /** Whether this canonical char_replace configuration is what a bare
built-in reference gets. */
+ private static boolean isCharReplaceDefault(String pattern, String
replacement, FoldContext fold) {
+ return CHAR_REPLACE_DEFAULT_REPLACEMENT.equals(replacement)
+ && canonicalizeCharReplacePattern(
+ CHAR_REPLACE_DEFAULT_PATTERN, replacement,
fold).equals(pattern);
+ }
+
+ private static void canonicalizeEffectiveComponentProperties(
+ TreeMap<String, String> properties, String type,
IndexPolicyTypeEnum expectedType) {
+ if ("pinyin".equals(type)) {
+ removeBooleanDefaults(properties, true,
+ "keep_first_letter", "keep_full_pinyin",
"keep_none_chinese",
+ "keep_none_chinese_together",
"keep_none_chinese_in_first_letter",
+ "lowercase", "trim_whitespace", "ignore_pinyin_offset",
+ "none_chinese_pinyin_tokenize");
+ removeBooleanDefaults(properties, false,
+ "keep_separate_first_letter", "keep_joined_full_pinyin",
"keep_original",
+ "keep_none_chinese_in_joined_full_pinyin",
"remove_duplicated_term",
+ "fixed_pinyin_offset", "keep_separate_chinese");
+ removeIntegerDefault(properties, "limit_first_letter_length", 16);
+ canonicalizePinyinDependencies(properties, expectedType);
+ return;
+ }
+
+ if (expectedType == IndexPolicyTypeEnum.TOKEN_FILTER) {
+ if ("asciifolding".equals(type)) {
+ removeBooleanDefaults(properties, false, "preserve_original");
+ } else if ("word_delimiter".equals(type)) {
+ removeBooleanDefaults(properties, true, "generate_word_parts",
"generate_number_parts",
+ "split_on_case_change", "split_on_numerics",
"stem_english_possessive");
+ removeBooleanDefaults(properties, false, "catenate_words",
"catenate_numbers",
+ "catenate_all", "preserve_original");
+ canonicalizeWordSet(properties, "protected_words");
+ canonicalizeTypeTable(properties);
+ } else if ("icu_normalizer".equals(type)) {
+ canonicalizeIcuNormalizerDefaults(properties, false);
+ }
+ return;
+ }
+
+ if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER) {
+ if ("icu_normalizer".equals(type)) {
+ canonicalizeIcuNormalizerDefaults(properties, true);
+ }
+ return;
}
- // For custom component, get its properties
+ if (expectedType != IndexPolicyTypeEnum.TOKENIZER) {
+ return;
+ }
+ switch (type) {
+ case "ngram":
+ case "edge_ngram":
+ removeIntegerDefault(properties, "min_gram", 1);
+ removeIntegerDefault(properties, "max_gram", 2);
+ canonicalizeWordSet(properties, "token_chars");
+ canonicalizeCustomTokenChars(properties);
+ break;
+ case "standard":
+ removeIntegerDefault(properties, "max_token_length", 255);
+ break;
+ case "char_group":
+ removeIntegerDefault(properties, "max_token_length", 255);
+ canonicalizeTokenizeOnChars(properties);
+ break;
+ case "keyword":
+ // BE only range-checks buffer_size; the emitted term is
always capped by a constant.
+ properties.remove("buffer_size");
+ break;
+ case "basic":
+ canonicalizeBasicExtraChars(properties);
+ break;
+ default:
+ break;
+ }
+ }
+
+ private static void removeBooleanDefaults(
+ TreeMap<String, String> properties, boolean defaultValue,
String... keys) {
+ for (String key : keys) {
+ String value = properties.get(key);
+ if (value == null || !("true".equalsIgnoreCase(value) ||
"false".equalsIgnoreCase(value))) {
+ continue;
+ }
+ boolean parsed = Boolean.parseBoolean(value);
+ if (parsed == defaultValue) {
+ properties.remove(key);
+ } else {
+ properties.put(key, Boolean.toString(parsed));
+ }
+ }
+ }
+
+ private static void removeIntegerDefault(
+ TreeMap<String, String> properties, String key, int defaultValue) {
+ String value = properties.get(key);
+ if (value == null) {
+ return;
+ }
try {
- Env env = Env.getCurrentEnv();
- if (env == null || env.getIndexPolicyMgr() == null) {
- return name;
+ int parsed = Integer.parseInt(value);
+ if (parsed == defaultValue) {
+ properties.remove(key);
+ } else {
+ properties.put(key, Integer.toString(parsed));
}
+ } catch (NumberFormatException e) {
+ // Invalid policies keep their original identity.
+ }
+ }
- IndexPolicy policy = env.getIndexPolicyMgr().getPolicyByName(name);
- if (policy == null || policy.getType() != expectedType) {
- return name;
+ private static void canonicalizeIcuNormalizerDefaults(
+ TreeMap<String, String> properties, boolean hasMode) {
+ String name = properties.get("name");
+ if (name != null) {
+ String normalizedName = name.trim().toLowerCase(Locale.ROOT);
+ if ("nfkc_cf".equals(normalizedName)) {
+ properties.remove("name");
+ } else {
+ properties.put("name", normalizedName);
}
- if (policy.isInvalid()) {
- return "invalid-policy:" + policy.getId() + ":" +
policy.getName();
+ }
+ String filter = properties.get("unicode_set_filter");
+ if (filter != null && filter.isEmpty()) {
+ // BE treats an explicit empty string like an absent filter.
+ properties.remove("unicode_set_filter");
+ } else if (filter != null) {
+ try {
+ UnicodeSet unicodeSet = new UnicodeSet(filter);
+ if (unicodeSet.isEmpty()) {
+ properties.remove("unicode_set_filter");
+ } else {
+ properties.put("unicode_set_filter",
unicodeSet.toPattern(false));
+ }
+ } catch (IllegalArgumentException e) {
+ // Invalid policies keep their original identity.
}
+ }
+ if (hasMode) {
+ canonicalizeIcuNormalizerMode(properties);
+ }
+ }
+
+ private static void canonicalizeIcuNormalizerMode(TreeMap<String, String>
properties) {
+ removeStringDefault(properties, "mode", "compose");
+ if (!"decompose".equals(properties.get("mode"))) {
+ return;
+ }
+ // BE ignores mode for nfd/nfkd, and nfc/nfkc in decompose mode are
the same ICU instances.
+ String name = properties.get("name");
+ if ("nfc".equals(name) || "nfd".equals(name)) {
+ properties.put("name", "nfd");
+ properties.remove("mode");
+ } else if ("nfkc".equals(name) || "nfkd".equals(name)) {
+ properties.put("name", "nfkd");
+ properties.remove("mode");
+ }
+ }
+
+ // BE reads these settings as unordered sets of trimmed, non-empty words.
+ private static void canonicalizeWordSet(TreeMap<String, String>
properties, String key) {
+ String value = properties.get(key);
+ if (value == null) {
+ return;
+ }
+ TreeSet<String> words = new TreeSet<>();
+ for (String word : value.split(",")) {
+ String trimmed = trimAsciiWhitespace(word);
+ if (!trimmed.isEmpty()) {
+ words.add(trimmed);
+ }
+ }
+ if (words.isEmpty()) {
+ properties.remove(key);
+ } else {
+ properties.put(key, String.join(",", words));
+ }
+ }
- Map<String, String> props = policy.getProperties();
- if (props == null || props.isEmpty()) {
- return name;
+ // BE matches custom token characters as a code point set ORed with the
named classes.
+ private static void canonicalizeCustomTokenChars(TreeMap<String, String>
properties) {
+ String value = properties.get("custom_token_chars");
+ if (value == null) {
+ return;
+ }
+ String tokenChars = properties.getOrDefault("token_chars", "");
+ Set<String> classes = new TreeSet<>(List.of(tokenChars.split(",")));
+ StringBuilder canonical = new StringBuilder();
+ value.codePoints().distinct().sorted()
+ .filter(codePoint -> !isCoveredByAsciiClass(codePoint,
classes))
+ .forEach(canonical::appendCodePoint);
+ if (canonical.length() == 0 && !value.isEmpty() &&
classes.remove("custom")) {
+ properties.remove("custom_token_chars");
+ properties.put("token_chars", String.join(",", classes));
+ return;
+ }
+ properties.put("custom_token_chars", canonical.toString());
+ }
+
+ // BE collects tokenize_on_chars entries into sets and checks the
categories before the literals.
+ private static void canonicalizeTokenizeOnChars(TreeMap<String, String>
properties) {
+ List<String> entries =
parseEntryList(properties.get("tokenize_on_chars"));
+ if (entries == null) {
+ return;
+ }
+ TreeSet<String> canonical = new TreeSet<>(entries);
+ Set<String> classes = new TreeSet<>(canonical);
+ classes.retainAll(CHAR_GROUP_TYPES);
+ canonical.removeIf(entry -> entry.indexOf('\\') < 0
+ && entry.codePointCount(0, entry.length()) == 1
+ && isCoveredByAsciiClass(entry.codePointAt(0), classes));
+ putEntryList(properties, "tokenize_on_chars", canonical);
+ }
+
+ /**
+ * Whether a named character class of the ngram or char_group tokenizer
already matches this
+ * code point. Only ASCII is judged: its categories never change between
the ICU versions FE
+ * and BE link against, and the class predicates agree there.
+ */
+ private static boolean isCoveredByAsciiClass(int codePoint, Set<String>
classes) {
+ if (codePoint >= 128) {
+ return false;
+ }
+ int type = UCharacter.getType(codePoint);
+ for (String name : classes) {
+ switch (name) {
+ case "letter":
+ if (UCharacter.isLetter(codePoint)) {
+ return true;
+ }
+ break;
+ case "digit":
+ if (UCharacter.isDigit(codePoint)) {
+ return true;
+ }
+ break;
+ case "whitespace":
+ if (UCharacter.isWhitespace(codePoint)) {
+ return true;
+ }
+ break;
+ case "punctuation":
+ if (type == UCharacter.START_PUNCTUATION || type ==
UCharacter.END_PUNCTUATION
+ || type == UCharacter.OTHER_PUNCTUATION || type ==
UCharacter.CONNECTOR_PUNCTUATION
+ || type == UCharacter.DASH_PUNCTUATION || type ==
UCharacter.INITIAL_PUNCTUATION
+ || type == UCharacter.FINAL_PUNCTUATION) {
+ return true;
+ }
+ break;
+ case "symbol":
+ if (type == UCharacter.CURRENCY_SYMBOL || type ==
UCharacter.MATH_SYMBOL
+ || type == UCharacter.OTHER_SYMBOL || type ==
UCharacter.MODIFIER_SYMBOL) {
+ return true;
+ }
+ break;
+ default:
+ break;
}
+ }
+ return false;
+ }
- // Build identity from sorted properties
- TreeMap<String, String> sortedProps = new TreeMap<>(props);
- if (expectedType == IndexPolicyTypeEnum.TOKENIZER
- && "ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE)))
{
- // This setting only limits policy creation; it does not
change emitted tokens.
- sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+ // BE builds a per-character type map where a later rule for the same
character wins.
+ private static void canonicalizeTypeTable(TreeMap<String, String>
properties) {
+ List<String> rules = parseEntryList(properties.get("type_table"));
+ if (rules == null) {
+ return;
+ }
+ TreeMap<Integer, String> types = new TreeMap<>();
+ for (String rule : rules) {
+ int arrow = rule.lastIndexOf("=>");
+ if (arrow < 0 || rule.indexOf('\n') >= 0 || rule.indexOf('\r') >=
0) {
+ return;
}
- return sortedProps.toString();
- } catch (RuntimeException e) {
- return name;
+ String character = trimAsciiWhitespace(rule.substring(0, arrow));
+ String type = trimAsciiWhitespace(rule.substring(arrow + 2));
+ // Escaped characters keep the original identity rather than
reproducing BE unescaping.
+ if (character.indexOf('\\') >= 0 || character.codePointCount(0,
character.length()) != 1
+ || !WORD_DELIMITER_TYPES.contains(type)) {
+ return;
+ }
+ types.put(character.codePointAt(0), type);
+ }
+ // A rule that restates BE's own classification changes nothing, but a
table made only of
+ // such rules still replaces BE's default table, which classifies
Latin-1 differently.
+ TreeMap<Integer, String> effectiveTypes = new TreeMap<>(types);
+ effectiveTypes.entrySet().removeIf(
+ entry ->
entry.getValue().equals(defaultWordDelimiterType(entry.getKey())));
+ if (effectiveTypes.isEmpty()) {
+ effectiveTypes = types;
+ }
+ List<String> canonicalRules = new ArrayList<>();
+ for (Map.Entry<Integer, String> entry : effectiveTypes.entrySet()) {
+ canonicalRules.add(new String(Character.toChars(entry.getKey())) +
"=>" + entry.getValue());
+ }
+ putEntryList(properties, "type_table", canonicalRules);
+ }
+
+ /** BE's u_charType classification of an ASCII code point, or null for
anything else. */
+ private static String defaultWordDelimiterType(int codePoint) {
+ if (codePoint >= 128) {
+ return null;
+ }
+ switch (UCharacter.getType(codePoint)) {
+ case UCharacter.UPPERCASE_LETTER:
+ return "UPPER";
+ case UCharacter.LOWERCASE_LETTER:
+ return "LOWER";
+ case UCharacter.DECIMAL_DIGIT_NUMBER:
+ return "DIGIT";
+ default:
+ return "SUBWORD_DELIM";
+ }
+ }
+
+ /** Parse a bracketed entry list as BE does, or return null for a
malformed list. */
+ private static List<String> parseEntryList(String value) {
+ if (value == null) {
+ return null;
+ }
+ List<String> entries = new ArrayList<>();
+ String trimmed = trimAsciiWhitespace(value);
+ if (trimmed.isEmpty()) {
+ return entries;
+ }
+ for (String item : ENTRY_SEPARATOR.split(trimmed)) {
+ String entry = trimAsciiWhitespace(item);
+ if (entry.length() < 2 || entry.charAt(0) != '[' ||
entry.charAt(entry.length() - 1) != ']') {
+ return null;
+ }
+ String content = entry.substring(1, entry.length() - 1);
+ if (!content.isEmpty()) {
+ entries.add(content);
+ }
+ }
+ return entries;
+ }
+
+ private static void putEntryList(TreeMap<String, String> properties,
String key, Collection<String> entries) {
+ if (entries.isEmpty()) {
+ properties.remove(key);
+ return;
+ }
+ StringBuilder canonical = new StringBuilder();
+ for (String entry : entries) {
+ if (canonical.length() > 0) {
+ canonical.append(",");
+ }
+ canonical.append("[").append(entry).append("]");
+ }
+ properties.put(key, canonical.toString());
+ }
+
+ // Trim the same ASCII whitespace that BE trims.
+ private static String trimAsciiWhitespace(String value) {
+ int begin = 0;
+ int end = value.length();
+ while (begin < end && isAsciiWhitespace(value.charAt(begin))) {
+ ++begin;
+ }
+ while (end > begin && isAsciiWhitespace(value.charAt(end - 1))) {
+ --end;
+ }
+ return value.substring(begin, end);
+ }
+
+ private static boolean isAsciiWhitespace(char value) {
+ return value == ' ' || (value >= '\t' && value <= '\r');
+ }
+
+ // BE consumes an ASCII alphanumeric run before it consults extra_chars.
+ private static void canonicalizeBasicExtraChars(TreeMap<String, String>
properties) {
+ String extraChars = properties.get("extra_chars");
+ if (extraChars == null) {
+ return;
+ }
+ boolean[] present = new boolean[128];
+ for (int i = 0; i < extraChars.length(); ++i) {
+ char value = extraChars.charAt(i);
+ if (value >= present.length) {
+ return;
+ }
+ boolean alphanumeric = (value >= '0' && value <= '9') || (value >=
'A' && value <= 'Z')
+ || (value >= 'a' && value <= 'z');
+ present[value] = !alphanumeric;
+ }
+ StringBuilder canonical = new StringBuilder();
+ for (int i = 0; i < present.length; ++i) {
+ if (present[i]) {
+ canonical.append((char) i);
+ }
+ }
+ if (canonical.length() == 0) {
+ properties.remove("extra_chars");
+ } else {
+ properties.put("extra_chars", canonical.toString());
+ }
+ }
+
+ private static void canonicalizePinyinDependencies(
+ TreeMap<String, String> properties, IndexPolicyTypeEnum
expectedType) {
+ Boolean keepFirstLetter = effectiveBoolean(properties,
"keep_first_letter", true);
+ Boolean keepFullPinyin = effectiveBoolean(properties,
"keep_full_pinyin", true);
+ Boolean keepSeparateFirstLetter = effectiveBoolean(properties,
"keep_separate_first_letter", false);
+ Boolean keepOriginal = effectiveBoolean(properties, "keep_original",
false);
+ Boolean keepNoneChinese = effectiveBoolean(properties,
"keep_none_chinese", true);
+ Boolean keepNoneChineseTogether = effectiveBoolean(properties,
"keep_none_chinese_together", true);
+ Boolean noneChinesePinyinTokenize = effectiveBoolean(properties,
"none_chinese_pinyin_tokenize", true);
+ Boolean ignorePinyinOffset = effectiveBoolean(properties,
"ignore_pinyin_offset", true);
+ Boolean keepJoinedFullPinyin = effectiveBoolean(properties,
"keep_joined_full_pinyin", false);
+ // Only the pinyin tokenizer also consults
keep_none_chinese_in_joined_full_pinyin, when no
+ // other setting settles whether it emits an untokenized ASCII buffer.
+ boolean tokenizerReadsJoinedSetting = expectedType ==
IndexPolicyTypeEnum.TOKENIZER
+ && !Boolean.FALSE.equals(keepNoneChinese)
+ && !Boolean.FALSE.equals(keepNoneChineseTogether)
+ && !Boolean.TRUE.equals(noneChinesePinyinTokenize)
+ && !Boolean.TRUE.equals(keepFirstLetter)
+ && !Boolean.TRUE.equals(keepSeparateFirstLetter)
+ && !Boolean.TRUE.equals(keepFullPinyin);
+
+ // The tokenizer only trims its candidates, and without the original
every candidate is
+ // pinyin or ASCII alphanumerics; the token filter also trims the
incoming token.
+ if (expectedType == IndexPolicyTypeEnum.TOKENIZER &&
Boolean.FALSE.equals(keepOriginal)) {
+ properties.remove("trim_whitespace");
+ }
+
+ // With every per-character candidate disabled at most one term is
emitted per input, the
+ // joined pinyin or the token filter's fallback original, so nothing
can be a duplicate.
+ if (Boolean.FALSE.equals(keepFirstLetter) &&
Boolean.FALSE.equals(keepFullPinyin)
+ && Boolean.FALSE.equals(keepSeparateFirstLetter) &&
Boolean.FALSE.equals(keepOriginal)
+ && Boolean.FALSE.equals(keepNoneChinese)
+ && Boolean.FALSE.equals(effectiveBoolean(properties,
"keep_separate_chinese", false))) {
+ properties.remove("remove_duplicated_term");
+ }
+
+ if (Boolean.FALSE.equals(keepFirstLetter)) {
+ properties.remove("limit_first_letter_length");
+ properties.remove("keep_none_chinese_in_first_letter");
+ }
+
+ if (Boolean.FALSE.equals(keepNoneChinese)) {
+ properties.remove("keep_none_chinese_together");
+ properties.remove("none_chinese_pinyin_tokenize");
+ } else if (Boolean.TRUE.equals(keepNoneChinese) &&
Boolean.FALSE.equals(keepNoneChineseTogether)) {
+ // BE emits each ASCII letter on its own here and never tokenizes
an ASCII buffer.
+ properties.remove("none_chinese_pinyin_tokenize");
}
+
+ if (Boolean.TRUE.equals(ignorePinyinOffset)
+ || Boolean.FALSE.equals(keepNoneChinese)
+ || Boolean.FALSE.equals(keepNoneChineseTogether)
+ || Boolean.FALSE.equals(noneChinesePinyinTokenize)) {
+ properties.remove("fixed_pinyin_offset");
+ }
+
+ // The joined full pinyin buffer is only emitted behind
keep_joined_full_pinyin.
+ if (Boolean.FALSE.equals(keepJoinedFullPinyin) &&
!tokenizerReadsJoinedSetting) {
+ properties.remove("keep_none_chinese_in_joined_full_pinyin");
+ }
+
+ // With the original, ASCII, first-letter and joined outputs all
disabled, every candidate
+ // the tokenizer emits comes from the pinyin dictionary, which is
already lower case. The
+ // token filter keeps the setting: it falls back to the original token
when nothing else
+ // would be emitted.
+ if (expectedType == IndexPolicyTypeEnum.TOKENIZER
+ && Boolean.FALSE.equals(keepFirstLetter)
+ && Boolean.FALSE.equals(keepNoneChinese)
Review Comment:
[P1] Also drop `lowercase` when enabled ASCII is forced through
`PinyinAlphabetTokenizer`. For a tokenizer with `keep_first_letter=false`,
`keep_original=false`, and `keep_joined_full_pinyin=false` (other relevant
settings at their defaults), `keep_none_chinese` remains true but `parseBuff`
sends that buffer to `PinyinAlphabetTokenizer::walk`, whose `segPinyinStr`
lowercases it unconditionally. Chinese full-pinyin is already lowercase and no
original/first/joined/separate output remains, so `lowercase=true` and `false`
emit the same terms, positions, offsets, and provenance; the different
identities let aliases pass CREATE/ALTER. This is distinct from the earlier
no-ASCII thread because ASCII stays enabled here. Extend the effective
case-preserving predicate and cover this configuration, retaining separate or
untokenized ASCII as negative cases.
##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -150,47 +302,589 @@ private static String
buildIdentityFromPolicyProperties(IndexPolicyTypeEnum type
* Resolve a component (tokenizer) to its identity.
*/
private static String resolveComponentIdentity(String name,
IndexPolicyTypeEnum expectedType) {
+ return resolveComponentIdentity(name, expectedType, null);
+ }
+
+ /** {@code fold} is the case-folding context of a char filter, or null
without a downstream fold. */
+ private static String resolveComponentIdentity(
+ String name, IndexPolicyTypeEnum expectedType, FoldContext fold) {
if (Strings.isNullOrEmpty(name)) {
return "";
}
- // Check if it's a built-in component
- if (expectedType == IndexPolicyTypeEnum.TOKENIZER
- && IndexPolicy.BUILTIN_TOKENIZERS.contains(name)) {
- return name;
+ // Existing named policies take precedence over built-ins for upgrade
compatibility.
+ try {
+ Env env = Env.getCurrentEnv();
+ if (env != null && env.getIndexPolicyMgr() != null) {
+ IndexPolicy policy =
env.getIndexPolicyMgr().getPolicyByName(name);
+ if (policy != null && policy.getType() == expectedType) {
+ if (policy.isInvalid()) {
+ return "invalid-policy:" + policy.getId() + ":" +
policy.getName();
+ }
+ Map<String, String> props = policy.getProperties();
+ if (props != null && !props.isEmpty()) {
+ TreeMap<String, String> sortedProps = new
TreeMap<>(props);
+ String type = sortedProps.get(IndexPolicy.PROP_TYPE);
+ String normalizedType =
normalizeBuiltinComponentName(type, expectedType);
+ if (normalizedType != null) {
+ if ("empty".equals(normalizedType)) {
+ return "";
+ }
+ sortedProps.put(IndexPolicy.PROP_TYPE,
normalizedType);
+ canonicalizeEffectiveComponentProperties(
+ sortedProps, normalizedType, expectedType);
+ if (sortedProps.size() == 1) {
+ return normalizedType;
+ }
+ }
+ if (expectedType == IndexPolicyTypeEnum.TOKENIZER
+ &&
"ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+ // This setting only limits policy creation; it
does not change emitted tokens.
+ sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+ }
+ if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER
+ &&
CHAR_REPLACE_FILTER.equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+ String replacement = sortedProps.getOrDefault(
+ PROP_REPLACEMENT,
CHAR_REPLACE_DEFAULT_REPLACEMENT);
+ String pattern = canonicalizeCharReplacePattern(
+ sortedProps.getOrDefault(PROP_PATTERN,
CHAR_REPLACE_DEFAULT_PATTERN),
+ replacement, fold);
+ if (pattern.isEmpty()) {
+ return "";
+ }
+ if (isCharReplaceDefault(pattern, replacement,
fold)) {
+ // Restating the factory defaults is the bare
built-in reference.
+ sortedProps.remove(PROP_PATTERN);
+ sortedProps.remove(PROP_REPLACEMENT);
+ } else {
+ sortedProps.put(PROP_PATTERN, pattern);
+ sortedProps.put(PROP_REPLACEMENT, replacement);
+ }
+ }
+ if (normalizedType != null && sortedProps.size() == 1)
{
+ return normalizedType;
+ }
+ return sortedProps.toString();
+ }
+ }
+ }
+ } catch (RuntimeException e) {
+ // Fall through to built-in resolution or the original name.
+ }
+
+ String normalizedName = normalizeBuiltinComponentName(name,
expectedType);
+ return "empty".equals(normalizedName) ? "" : normalizedName == null ?
name : normalizedName;
+ }
+
+ /** Whether this canonical char_replace configuration is what a bare
built-in reference gets. */
+ private static boolean isCharReplaceDefault(String pattern, String
replacement, FoldContext fold) {
+ return CHAR_REPLACE_DEFAULT_REPLACEMENT.equals(replacement)
+ && canonicalizeCharReplacePattern(
+ CHAR_REPLACE_DEFAULT_PATTERN, replacement,
fold).equals(pattern);
+ }
+
+ private static void canonicalizeEffectiveComponentProperties(
+ TreeMap<String, String> properties, String type,
IndexPolicyTypeEnum expectedType) {
+ if ("pinyin".equals(type)) {
+ removeBooleanDefaults(properties, true,
+ "keep_first_letter", "keep_full_pinyin",
"keep_none_chinese",
+ "keep_none_chinese_together",
"keep_none_chinese_in_first_letter",
+ "lowercase", "trim_whitespace", "ignore_pinyin_offset",
+ "none_chinese_pinyin_tokenize");
+ removeBooleanDefaults(properties, false,
+ "keep_separate_first_letter", "keep_joined_full_pinyin",
"keep_original",
+ "keep_none_chinese_in_joined_full_pinyin",
"remove_duplicated_term",
+ "fixed_pinyin_offset", "keep_separate_chinese");
+ removeIntegerDefault(properties, "limit_first_letter_length", 16);
+ canonicalizePinyinDependencies(properties, expectedType);
+ return;
+ }
+
+ if (expectedType == IndexPolicyTypeEnum.TOKEN_FILTER) {
+ if ("asciifolding".equals(type)) {
+ removeBooleanDefaults(properties, false, "preserve_original");
+ } else if ("word_delimiter".equals(type)) {
+ removeBooleanDefaults(properties, true, "generate_word_parts",
"generate_number_parts",
+ "split_on_case_change", "split_on_numerics",
"stem_english_possessive");
+ removeBooleanDefaults(properties, false, "catenate_words",
"catenate_numbers",
+ "catenate_all", "preserve_original");
+ canonicalizeWordSet(properties, "protected_words");
+ canonicalizeTypeTable(properties);
+ } else if ("icu_normalizer".equals(type)) {
+ canonicalizeIcuNormalizerDefaults(properties, false);
+ }
+ return;
+ }
+
+ if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER) {
+ if ("icu_normalizer".equals(type)) {
+ canonicalizeIcuNormalizerDefaults(properties, true);
+ }
+ return;
}
- // For custom component, get its properties
+ if (expectedType != IndexPolicyTypeEnum.TOKENIZER) {
+ return;
+ }
+ switch (type) {
+ case "ngram":
+ case "edge_ngram":
+ removeIntegerDefault(properties, "min_gram", 1);
+ removeIntegerDefault(properties, "max_gram", 2);
+ canonicalizeWordSet(properties, "token_chars");
+ canonicalizeCustomTokenChars(properties);
+ break;
+ case "standard":
+ removeIntegerDefault(properties, "max_token_length", 255);
+ break;
+ case "char_group":
+ removeIntegerDefault(properties, "max_token_length", 255);
+ canonicalizeTokenizeOnChars(properties);
+ break;
+ case "keyword":
+ // BE only range-checks buffer_size; the emitted term is
always capped by a constant.
+ properties.remove("buffer_size");
+ break;
+ case "basic":
+ canonicalizeBasicExtraChars(properties);
+ break;
+ default:
+ break;
+ }
+ }
+
+ private static void removeBooleanDefaults(
+ TreeMap<String, String> properties, boolean defaultValue,
String... keys) {
+ for (String key : keys) {
+ String value = properties.get(key);
+ if (value == null || !("true".equalsIgnoreCase(value) ||
"false".equalsIgnoreCase(value))) {
+ continue;
+ }
+ boolean parsed = Boolean.parseBoolean(value);
+ if (parsed == defaultValue) {
+ properties.remove(key);
+ } else {
+ properties.put(key, Boolean.toString(parsed));
+ }
+ }
+ }
+
+ private static void removeIntegerDefault(
+ TreeMap<String, String> properties, String key, int defaultValue) {
+ String value = properties.get(key);
+ if (value == null) {
+ return;
+ }
try {
- Env env = Env.getCurrentEnv();
- if (env == null || env.getIndexPolicyMgr() == null) {
- return name;
+ int parsed = Integer.parseInt(value);
+ if (parsed == defaultValue) {
+ properties.remove(key);
+ } else {
+ properties.put(key, Integer.toString(parsed));
}
+ } catch (NumberFormatException e) {
+ // Invalid policies keep their original identity.
+ }
+ }
- IndexPolicy policy = env.getIndexPolicyMgr().getPolicyByName(name);
- if (policy == null || policy.getType() != expectedType) {
- return name;
+ private static void canonicalizeIcuNormalizerDefaults(
+ TreeMap<String, String> properties, boolean hasMode) {
+ String name = properties.get("name");
+ if (name != null) {
+ String normalizedName = name.trim().toLowerCase(Locale.ROOT);
+ if ("nfkc_cf".equals(normalizedName)) {
+ properties.remove("name");
+ } else {
+ properties.put("name", normalizedName);
}
- if (policy.isInvalid()) {
- return "invalid-policy:" + policy.getId() + ":" +
policy.getName();
+ }
+ String filter = properties.get("unicode_set_filter");
+ if (filter != null && filter.isEmpty()) {
+ // BE treats an explicit empty string like an absent filter.
+ properties.remove("unicode_set_filter");
+ } else if (filter != null) {
+ try {
+ UnicodeSet unicodeSet = new UnicodeSet(filter);
+ if (unicodeSet.isEmpty()) {
+ properties.remove("unicode_set_filter");
+ } else {
+ properties.put("unicode_set_filter",
unicodeSet.toPattern(false));
+ }
+ } catch (IllegalArgumentException e) {
+ // Invalid policies keep their original identity.
}
+ }
+ if (hasMode) {
+ canonicalizeIcuNormalizerMode(properties);
+ }
+ }
+
+ private static void canonicalizeIcuNormalizerMode(TreeMap<String, String>
properties) {
+ removeStringDefault(properties, "mode", "compose");
+ if (!"decompose".equals(properties.get("mode"))) {
+ return;
+ }
+ // BE ignores mode for nfd/nfkd, and nfc/nfkc in decompose mode are
the same ICU instances.
+ String name = properties.get("name");
+ if ("nfc".equals(name) || "nfd".equals(name)) {
+ properties.put("name", "nfd");
+ properties.remove("mode");
+ } else if ("nfkc".equals(name) || "nfkd".equals(name)) {
+ properties.put("name", "nfkd");
+ properties.remove("mode");
+ }
+ }
+
+ // BE reads these settings as unordered sets of trimmed, non-empty words.
+ private static void canonicalizeWordSet(TreeMap<String, String>
properties, String key) {
+ String value = properties.get(key);
+ if (value == null) {
+ return;
+ }
+ TreeSet<String> words = new TreeSet<>();
+ for (String word : value.split(",")) {
+ String trimmed = trimAsciiWhitespace(word);
+ if (!trimmed.isEmpty()) {
+ words.add(trimmed);
+ }
+ }
+ if (words.isEmpty()) {
+ properties.remove(key);
+ } else {
+ properties.put(key, String.join(",", words));
+ }
+ }
- Map<String, String> props = policy.getProperties();
- if (props == null || props.isEmpty()) {
- return name;
+ // BE matches custom token characters as a code point set ORed with the
named classes.
+ private static void canonicalizeCustomTokenChars(TreeMap<String, String>
properties) {
+ String value = properties.get("custom_token_chars");
+ if (value == null) {
+ return;
+ }
+ String tokenChars = properties.getOrDefault("token_chars", "");
+ Set<String> classes = new TreeSet<>(List.of(tokenChars.split(",")));
+ StringBuilder canonical = new StringBuilder();
+ value.codePoints().distinct().sorted()
+ .filter(codePoint -> !isCoveredByAsciiClass(codePoint,
classes))
+ .forEach(canonical::appendCodePoint);
+ if (canonical.length() == 0 && !value.isEmpty() &&
classes.remove("custom")) {
+ properties.remove("custom_token_chars");
+ properties.put("token_chars", String.join(",", classes));
+ return;
+ }
+ properties.put("custom_token_chars", canonical.toString());
+ }
+
+ // BE collects tokenize_on_chars entries into sets and checks the
categories before the literals.
+ private static void canonicalizeTokenizeOnChars(TreeMap<String, String>
properties) {
+ List<String> entries =
parseEntryList(properties.get("tokenize_on_chars"));
+ if (entries == null) {
+ return;
+ }
+ TreeSet<String> canonical = new TreeSet<>(entries);
+ Set<String> classes = new TreeSet<>(canonical);
+ classes.retainAll(CHAR_GROUP_TYPES);
+ canonical.removeIf(entry -> entry.indexOf('\\') < 0
+ && entry.codePointCount(0, entry.length()) == 1
+ && isCoveredByAsciiClass(entry.codePointAt(0), classes));
+ putEntryList(properties, "tokenize_on_chars", canonical);
+ }
+
+ /**
+ * Whether a named character class of the ngram or char_group tokenizer
already matches this
+ * code point. Only ASCII is judged: its categories never change between
the ICU versions FE
+ * and BE link against, and the class predicates agree there.
+ */
+ private static boolean isCoveredByAsciiClass(int codePoint, Set<String>
classes) {
+ if (codePoint >= 128) {
+ return false;
+ }
+ int type = UCharacter.getType(codePoint);
+ for (String name : classes) {
+ switch (name) {
+ case "letter":
+ if (UCharacter.isLetter(codePoint)) {
+ return true;
+ }
+ break;
+ case "digit":
+ if (UCharacter.isDigit(codePoint)) {
+ return true;
+ }
+ break;
+ case "whitespace":
+ if (UCharacter.isWhitespace(codePoint)) {
+ return true;
+ }
+ break;
+ case "punctuation":
+ if (type == UCharacter.START_PUNCTUATION || type ==
UCharacter.END_PUNCTUATION
+ || type == UCharacter.OTHER_PUNCTUATION || type ==
UCharacter.CONNECTOR_PUNCTUATION
+ || type == UCharacter.DASH_PUNCTUATION || type ==
UCharacter.INITIAL_PUNCTUATION
+ || type == UCharacter.FINAL_PUNCTUATION) {
+ return true;
+ }
+ break;
+ case "symbol":
+ if (type == UCharacter.CURRENCY_SYMBOL || type ==
UCharacter.MATH_SYMBOL
+ || type == UCharacter.OTHER_SYMBOL || type ==
UCharacter.MODIFIER_SYMBOL) {
+ return true;
+ }
+ break;
+ default:
+ break;
}
+ }
+ return false;
+ }
- // Build identity from sorted properties
- TreeMap<String, String> sortedProps = new TreeMap<>(props);
- if (expectedType == IndexPolicyTypeEnum.TOKENIZER
- && "ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE)))
{
- // This setting only limits policy creation; it does not
change emitted tokens.
- sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+ // BE builds a per-character type map where a later rule for the same
character wins.
+ private static void canonicalizeTypeTable(TreeMap<String, String>
properties) {
+ List<String> rules = parseEntryList(properties.get("type_table"));
+ if (rules == null) {
+ return;
+ }
+ TreeMap<Integer, String> types = new TreeMap<>();
+ for (String rule : rules) {
+ int arrow = rule.lastIndexOf("=>");
+ if (arrow < 0 || rule.indexOf('\n') >= 0 || rule.indexOf('\r') >=
0) {
+ return;
}
- return sortedProps.toString();
- } catch (RuntimeException e) {
- return name;
+ String character = trimAsciiWhitespace(rule.substring(0, arrow));
+ String type = trimAsciiWhitespace(rule.substring(arrow + 2));
+ // Escaped characters keep the original identity rather than
reproducing BE unescaping.
+ if (character.indexOf('\\') >= 0 || character.codePointCount(0,
character.length()) != 1
+ || !WORD_DELIMITER_TYPES.contains(type)) {
+ return;
+ }
+ types.put(character.codePointAt(0), type);
+ }
+ // A rule that restates BE's own classification changes nothing, but a
table made only of
+ // such rules still replaces BE's default table, which classifies
Latin-1 differently.
+ TreeMap<Integer, String> effectiveTypes = new TreeMap<>(types);
+ effectiveTypes.entrySet().removeIf(
+ entry ->
entry.getValue().equals(defaultWordDelimiterType(entry.getKey())));
+ if (effectiveTypes.isEmpty()) {
+ effectiveTypes = types;
Review Comment:
[P1] Give every all-no-op explicit type table one canonical explicit-table
identity. After this removal, the fallback restores the original rules, so `[a
=> LOWER]` and `[b => LOWER]` get different identities. BE builds both explicit
tables by filling the same 256 entries with `WordDelimiterIterator::get_type()`
and then writing those already-present values, so they emit identical terms,
positions, offsets, and provenance; differently named aliases can therefore
pass the CREATE/ALTER duplicate fences. This is distinct from the earlier
absent-vs-explicit case: both sides here are explicit tables. Serialize a
deterministic explicit-default sentinel and cover these two aliases while
retaining absent-vs-explicit as a negative.
##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -150,47 +302,589 @@ private static String
buildIdentityFromPolicyProperties(IndexPolicyTypeEnum type
* Resolve a component (tokenizer) to its identity.
*/
private static String resolveComponentIdentity(String name,
IndexPolicyTypeEnum expectedType) {
+ return resolveComponentIdentity(name, expectedType, null);
+ }
+
+ /** {@code fold} is the case-folding context of a char filter, or null
without a downstream fold. */
+ private static String resolveComponentIdentity(
+ String name, IndexPolicyTypeEnum expectedType, FoldContext fold) {
if (Strings.isNullOrEmpty(name)) {
return "";
}
- // Check if it's a built-in component
- if (expectedType == IndexPolicyTypeEnum.TOKENIZER
- && IndexPolicy.BUILTIN_TOKENIZERS.contains(name)) {
- return name;
+ // Existing named policies take precedence over built-ins for upgrade
compatibility.
+ try {
+ Env env = Env.getCurrentEnv();
+ if (env != null && env.getIndexPolicyMgr() != null) {
+ IndexPolicy policy =
env.getIndexPolicyMgr().getPolicyByName(name);
+ if (policy != null && policy.getType() == expectedType) {
+ if (policy.isInvalid()) {
+ return "invalid-policy:" + policy.getId() + ":" +
policy.getName();
+ }
+ Map<String, String> props = policy.getProperties();
+ if (props != null && !props.isEmpty()) {
+ TreeMap<String, String> sortedProps = new
TreeMap<>(props);
+ String type = sortedProps.get(IndexPolicy.PROP_TYPE);
+ String normalizedType =
normalizeBuiltinComponentName(type, expectedType);
+ if (normalizedType != null) {
+ if ("empty".equals(normalizedType)) {
+ return "";
+ }
+ sortedProps.put(IndexPolicy.PROP_TYPE,
normalizedType);
+ canonicalizeEffectiveComponentProperties(
+ sortedProps, normalizedType, expectedType);
+ if (sortedProps.size() == 1) {
+ return normalizedType;
+ }
+ }
+ if (expectedType == IndexPolicyTypeEnum.TOKENIZER
+ &&
"ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+ // This setting only limits policy creation; it
does not change emitted tokens.
+ sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+ }
+ if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER
+ &&
CHAR_REPLACE_FILTER.equals(sortedProps.get(IndexPolicy.PROP_TYPE))) {
+ String replacement = sortedProps.getOrDefault(
+ PROP_REPLACEMENT,
CHAR_REPLACE_DEFAULT_REPLACEMENT);
+ String pattern = canonicalizeCharReplacePattern(
+ sortedProps.getOrDefault(PROP_PATTERN,
CHAR_REPLACE_DEFAULT_PATTERN),
+ replacement, fold);
+ if (pattern.isEmpty()) {
+ return "";
+ }
+ if (isCharReplaceDefault(pattern, replacement,
fold)) {
+ // Restating the factory defaults is the bare
built-in reference.
+ sortedProps.remove(PROP_PATTERN);
+ sortedProps.remove(PROP_REPLACEMENT);
+ } else {
+ sortedProps.put(PROP_PATTERN, pattern);
+ sortedProps.put(PROP_REPLACEMENT, replacement);
+ }
+ }
+ if (normalizedType != null && sortedProps.size() == 1)
{
+ return normalizedType;
+ }
+ return sortedProps.toString();
+ }
+ }
+ }
+ } catch (RuntimeException e) {
+ // Fall through to built-in resolution or the original name.
+ }
+
+ String normalizedName = normalizeBuiltinComponentName(name,
expectedType);
+ return "empty".equals(normalizedName) ? "" : normalizedName == null ?
name : normalizedName;
+ }
+
+ /** Whether this canonical char_replace configuration is what a bare
built-in reference gets. */
+ private static boolean isCharReplaceDefault(String pattern, String
replacement, FoldContext fold) {
+ return CHAR_REPLACE_DEFAULT_REPLACEMENT.equals(replacement)
+ && canonicalizeCharReplacePattern(
+ CHAR_REPLACE_DEFAULT_PATTERN, replacement,
fold).equals(pattern);
+ }
+
+ private static void canonicalizeEffectiveComponentProperties(
+ TreeMap<String, String> properties, String type,
IndexPolicyTypeEnum expectedType) {
+ if ("pinyin".equals(type)) {
+ removeBooleanDefaults(properties, true,
+ "keep_first_letter", "keep_full_pinyin",
"keep_none_chinese",
+ "keep_none_chinese_together",
"keep_none_chinese_in_first_letter",
+ "lowercase", "trim_whitespace", "ignore_pinyin_offset",
+ "none_chinese_pinyin_tokenize");
+ removeBooleanDefaults(properties, false,
+ "keep_separate_first_letter", "keep_joined_full_pinyin",
"keep_original",
+ "keep_none_chinese_in_joined_full_pinyin",
"remove_duplicated_term",
+ "fixed_pinyin_offset", "keep_separate_chinese");
+ removeIntegerDefault(properties, "limit_first_letter_length", 16);
+ canonicalizePinyinDependencies(properties, expectedType);
+ return;
+ }
+
+ if (expectedType == IndexPolicyTypeEnum.TOKEN_FILTER) {
+ if ("asciifolding".equals(type)) {
+ removeBooleanDefaults(properties, false, "preserve_original");
+ } else if ("word_delimiter".equals(type)) {
+ removeBooleanDefaults(properties, true, "generate_word_parts",
"generate_number_parts",
+ "split_on_case_change", "split_on_numerics",
"stem_english_possessive");
+ removeBooleanDefaults(properties, false, "catenate_words",
"catenate_numbers",
+ "catenate_all", "preserve_original");
+ canonicalizeWordSet(properties, "protected_words");
+ canonicalizeTypeTable(properties);
+ } else if ("icu_normalizer".equals(type)) {
+ canonicalizeIcuNormalizerDefaults(properties, false);
+ }
+ return;
+ }
+
+ if (expectedType == IndexPolicyTypeEnum.CHAR_FILTER) {
+ if ("icu_normalizer".equals(type)) {
+ canonicalizeIcuNormalizerDefaults(properties, true);
+ }
+ return;
}
- // For custom component, get its properties
+ if (expectedType != IndexPolicyTypeEnum.TOKENIZER) {
+ return;
+ }
+ switch (type) {
+ case "ngram":
+ case "edge_ngram":
+ removeIntegerDefault(properties, "min_gram", 1);
+ removeIntegerDefault(properties, "max_gram", 2);
+ canonicalizeWordSet(properties, "token_chars");
+ canonicalizeCustomTokenChars(properties);
+ break;
+ case "standard":
+ removeIntegerDefault(properties, "max_token_length", 255);
+ break;
+ case "char_group":
+ removeIntegerDefault(properties, "max_token_length", 255);
+ canonicalizeTokenizeOnChars(properties);
+ break;
+ case "keyword":
+ // BE only range-checks buffer_size; the emitted term is
always capped by a constant.
+ properties.remove("buffer_size");
+ break;
+ case "basic":
+ canonicalizeBasicExtraChars(properties);
+ break;
+ default:
+ break;
+ }
+ }
+
+ private static void removeBooleanDefaults(
+ TreeMap<String, String> properties, boolean defaultValue,
String... keys) {
+ for (String key : keys) {
+ String value = properties.get(key);
+ if (value == null || !("true".equalsIgnoreCase(value) ||
"false".equalsIgnoreCase(value))) {
+ continue;
+ }
+ boolean parsed = Boolean.parseBoolean(value);
+ if (parsed == defaultValue) {
+ properties.remove(key);
+ } else {
+ properties.put(key, Boolean.toString(parsed));
+ }
+ }
+ }
+
+ private static void removeIntegerDefault(
+ TreeMap<String, String> properties, String key, int defaultValue) {
+ String value = properties.get(key);
+ if (value == null) {
+ return;
+ }
try {
- Env env = Env.getCurrentEnv();
- if (env == null || env.getIndexPolicyMgr() == null) {
- return name;
+ int parsed = Integer.parseInt(value);
+ if (parsed == defaultValue) {
+ properties.remove(key);
+ } else {
+ properties.put(key, Integer.toString(parsed));
}
+ } catch (NumberFormatException e) {
+ // Invalid policies keep their original identity.
+ }
+ }
- IndexPolicy policy = env.getIndexPolicyMgr().getPolicyByName(name);
- if (policy == null || policy.getType() != expectedType) {
- return name;
+ private static void canonicalizeIcuNormalizerDefaults(
+ TreeMap<String, String> properties, boolean hasMode) {
+ String name = properties.get("name");
+ if (name != null) {
+ String normalizedName = name.trim().toLowerCase(Locale.ROOT);
+ if ("nfkc_cf".equals(normalizedName)) {
+ properties.remove("name");
+ } else {
+ properties.put("name", normalizedName);
}
- if (policy.isInvalid()) {
- return "invalid-policy:" + policy.getId() + ":" +
policy.getName();
+ }
+ String filter = properties.get("unicode_set_filter");
+ if (filter != null && filter.isEmpty()) {
+ // BE treats an explicit empty string like an absent filter.
+ properties.remove("unicode_set_filter");
+ } else if (filter != null) {
+ try {
+ UnicodeSet unicodeSet = new UnicodeSet(filter);
+ if (unicodeSet.isEmpty()) {
+ properties.remove("unicode_set_filter");
+ } else {
+ properties.put("unicode_set_filter",
unicodeSet.toPattern(false));
+ }
+ } catch (IllegalArgumentException e) {
+ // Invalid policies keep their original identity.
}
+ }
+ if (hasMode) {
+ canonicalizeIcuNormalizerMode(properties);
+ }
+ }
+
+ private static void canonicalizeIcuNormalizerMode(TreeMap<String, String>
properties) {
+ removeStringDefault(properties, "mode", "compose");
+ if (!"decompose".equals(properties.get("mode"))) {
+ return;
+ }
+ // BE ignores mode for nfd/nfkd, and nfc/nfkc in decompose mode are
the same ICU instances.
+ String name = properties.get("name");
+ if ("nfc".equals(name) || "nfd".equals(name)) {
+ properties.put("name", "nfd");
+ properties.remove("mode");
+ } else if ("nfkc".equals(name) || "nfkd".equals(name)) {
+ properties.put("name", "nfkd");
+ properties.remove("mode");
+ }
+ }
+
+ // BE reads these settings as unordered sets of trimmed, non-empty words.
+ private static void canonicalizeWordSet(TreeMap<String, String>
properties, String key) {
+ String value = properties.get(key);
+ if (value == null) {
+ return;
+ }
+ TreeSet<String> words = new TreeSet<>();
+ for (String word : value.split(",")) {
+ String trimmed = trimAsciiWhitespace(word);
+ if (!trimmed.isEmpty()) {
+ words.add(trimmed);
+ }
+ }
+ if (words.isEmpty()) {
+ properties.remove(key);
+ } else {
+ properties.put(key, String.join(",", words));
+ }
+ }
- Map<String, String> props = policy.getProperties();
- if (props == null || props.isEmpty()) {
- return name;
+ // BE matches custom token characters as a code point set ORed with the
named classes.
+ private static void canonicalizeCustomTokenChars(TreeMap<String, String>
properties) {
+ String value = properties.get("custom_token_chars");
+ if (value == null) {
+ return;
+ }
+ String tokenChars = properties.getOrDefault("token_chars", "");
+ Set<String> classes = new TreeSet<>(List.of(tokenChars.split(",")));
+ StringBuilder canonical = new StringBuilder();
+ value.codePoints().distinct().sorted()
+ .filter(codePoint -> !isCoveredByAsciiClass(codePoint,
classes))
+ .forEach(canonical::appendCodePoint);
+ if (canonical.length() == 0 && !value.isEmpty() &&
classes.remove("custom")) {
+ properties.remove("custom_token_chars");
+ properties.put("token_chars", String.join(",", classes));
+ return;
+ }
+ properties.put("custom_token_chars", canonical.toString());
+ }
+
+ // BE collects tokenize_on_chars entries into sets and checks the
categories before the literals.
+ private static void canonicalizeTokenizeOnChars(TreeMap<String, String>
properties) {
+ List<String> entries =
parseEntryList(properties.get("tokenize_on_chars"));
+ if (entries == null) {
+ return;
+ }
+ TreeSet<String> canonical = new TreeSet<>(entries);
+ Set<String> classes = new TreeSet<>(canonical);
+ classes.retainAll(CHAR_GROUP_TYPES);
+ canonical.removeIf(entry -> entry.indexOf('\\') < 0
+ && entry.codePointCount(0, entry.length()) == 1
+ && isCoveredByAsciiClass(entry.codePointAt(0), classes));
+ putEntryList(properties, "tokenize_on_chars", canonical);
+ }
+
+ /**
+ * Whether a named character class of the ngram or char_group tokenizer
already matches this
+ * code point. Only ASCII is judged: its categories never change between
the ICU versions FE
+ * and BE link against, and the class predicates agree there.
+ */
+ private static boolean isCoveredByAsciiClass(int codePoint, Set<String>
classes) {
+ if (codePoint >= 128) {
+ return false;
+ }
+ int type = UCharacter.getType(codePoint);
+ for (String name : classes) {
+ switch (name) {
+ case "letter":
+ if (UCharacter.isLetter(codePoint)) {
+ return true;
+ }
+ break;
+ case "digit":
+ if (UCharacter.isDigit(codePoint)) {
+ return true;
+ }
+ break;
+ case "whitespace":
+ if (UCharacter.isWhitespace(codePoint)) {
+ return true;
+ }
+ break;
+ case "punctuation":
+ if (type == UCharacter.START_PUNCTUATION || type ==
UCharacter.END_PUNCTUATION
+ || type == UCharacter.OTHER_PUNCTUATION || type ==
UCharacter.CONNECTOR_PUNCTUATION
+ || type == UCharacter.DASH_PUNCTUATION || type ==
UCharacter.INITIAL_PUNCTUATION
+ || type == UCharacter.FINAL_PUNCTUATION) {
+ return true;
+ }
+ break;
+ case "symbol":
+ if (type == UCharacter.CURRENCY_SYMBOL || type ==
UCharacter.MATH_SYMBOL
+ || type == UCharacter.OTHER_SYMBOL || type ==
UCharacter.MODIFIER_SYMBOL) {
+ return true;
+ }
+ break;
+ default:
+ break;
}
+ }
+ return false;
+ }
- // Build identity from sorted properties
- TreeMap<String, String> sortedProps = new TreeMap<>(props);
- if (expectedType == IndexPolicyTypeEnum.TOKENIZER
- && "ngram".equals(sortedProps.get(IndexPolicy.PROP_TYPE)))
{
- // This setting only limits policy creation; it does not
change emitted tokens.
- sortedProps.remove(PROP_MAX_NGRAM_DIFF);
+ // BE builds a per-character type map where a later rule for the same
character wins.
+ private static void canonicalizeTypeTable(TreeMap<String, String>
properties) {
+ List<String> rules = parseEntryList(properties.get("type_table"));
+ if (rules == null) {
+ return;
+ }
+ TreeMap<Integer, String> types = new TreeMap<>();
+ for (String rule : rules) {
+ int arrow = rule.lastIndexOf("=>");
+ if (arrow < 0 || rule.indexOf('\n') >= 0 || rule.indexOf('\r') >=
0) {
+ return;
}
- return sortedProps.toString();
- } catch (RuntimeException e) {
- return name;
+ String character = trimAsciiWhitespace(rule.substring(0, arrow));
+ String type = trimAsciiWhitespace(rule.substring(arrow + 2));
+ // Escaped characters keep the original identity rather than
reproducing BE unescaping.
+ if (character.indexOf('\\') >= 0 || character.codePointCount(0,
character.length()) != 1
+ || !WORD_DELIMITER_TYPES.contains(type)) {
+ return;
+ }
+ types.put(character.codePointAt(0), type);
+ }
+ // A rule that restates BE's own classification changes nothing, but a
table made only of
+ // such rules still replaces BE's default table, which classifies
Latin-1 differently.
+ TreeMap<Integer, String> effectiveTypes = new TreeMap<>(types);
+ effectiveTypes.entrySet().removeIf(
+ entry ->
entry.getValue().equals(defaultWordDelimiterType(entry.getKey())));
+ if (effectiveTypes.isEmpty()) {
+ effectiveTypes = types;
+ }
+ List<String> canonicalRules = new ArrayList<>();
+ for (Map.Entry<Integer, String> entry : effectiveTypes.entrySet()) {
+ canonicalRules.add(new String(Character.toChars(entry.getKey())) +
"=>" + entry.getValue());
+ }
+ putEntryList(properties, "type_table", canonicalRules);
+ }
+
+ /** BE's u_charType classification of an ASCII code point, or null for
anything else. */
+ private static String defaultWordDelimiterType(int codePoint) {
+ if (codePoint >= 128) {
+ return null;
+ }
+ switch (UCharacter.getType(codePoint)) {
+ case UCharacter.UPPERCASE_LETTER:
+ return "UPPER";
+ case UCharacter.LOWERCASE_LETTER:
+ return "LOWER";
+ case UCharacter.DECIMAL_DIGIT_NUMBER:
+ return "DIGIT";
+ default:
+ return "SUBWORD_DELIM";
+ }
+ }
+
+ /** Parse a bracketed entry list as BE does, or return null for a
malformed list. */
+ private static List<String> parseEntryList(String value) {
+ if (value == null) {
+ return null;
+ }
+ List<String> entries = new ArrayList<>();
+ String trimmed = trimAsciiWhitespace(value);
+ if (trimmed.isEmpty()) {
+ return entries;
+ }
+ for (String item : ENTRY_SEPARATOR.split(trimmed)) {
+ String entry = trimAsciiWhitespace(item);
+ if (entry.length() < 2 || entry.charAt(0) != '[' ||
entry.charAt(entry.length() - 1) != ']') {
+ return null;
+ }
+ String content = entry.substring(1, entry.length() - 1);
+ if (!content.isEmpty()) {
+ entries.add(content);
+ }
+ }
+ return entries;
+ }
+
+ private static void putEntryList(TreeMap<String, String> properties,
String key, Collection<String> entries) {
+ if (entries.isEmpty()) {
+ properties.remove(key);
+ return;
+ }
+ StringBuilder canonical = new StringBuilder();
+ for (String entry : entries) {
+ if (canonical.length() > 0) {
+ canonical.append(",");
+ }
+ canonical.append("[").append(entry).append("]");
+ }
+ properties.put(key, canonical.toString());
+ }
+
+ // Trim the same ASCII whitespace that BE trims.
+ private static String trimAsciiWhitespace(String value) {
+ int begin = 0;
+ int end = value.length();
+ while (begin < end && isAsciiWhitespace(value.charAt(begin))) {
+ ++begin;
+ }
+ while (end > begin && isAsciiWhitespace(value.charAt(end - 1))) {
+ --end;
+ }
+ return value.substring(begin, end);
+ }
+
+ private static boolean isAsciiWhitespace(char value) {
+ return value == ' ' || (value >= '\t' && value <= '\r');
+ }
+
+ // BE consumes an ASCII alphanumeric run before it consults extra_chars.
+ private static void canonicalizeBasicExtraChars(TreeMap<String, String>
properties) {
+ String extraChars = properties.get("extra_chars");
+ if (extraChars == null) {
+ return;
+ }
+ boolean[] present = new boolean[128];
+ for (int i = 0; i < extraChars.length(); ++i) {
+ char value = extraChars.charAt(i);
+ if (value >= present.length) {
+ return;
+ }
+ boolean alphanumeric = (value >= '0' && value <= '9') || (value >=
'A' && value <= 'Z')
+ || (value >= 'a' && value <= 'z');
+ present[value] = !alphanumeric;
+ }
+ StringBuilder canonical = new StringBuilder();
+ for (int i = 0; i < present.length; ++i) {
+ if (present[i]) {
+ canonical.append((char) i);
+ }
+ }
+ if (canonical.length() == 0) {
+ properties.remove("extra_chars");
+ } else {
+ properties.put("extra_chars", canonical.toString());
+ }
+ }
+
+ private static void canonicalizePinyinDependencies(
+ TreeMap<String, String> properties, IndexPolicyTypeEnum
expectedType) {
+ Boolean keepFirstLetter = effectiveBoolean(properties,
"keep_first_letter", true);
+ Boolean keepFullPinyin = effectiveBoolean(properties,
"keep_full_pinyin", true);
+ Boolean keepSeparateFirstLetter = effectiveBoolean(properties,
"keep_separate_first_letter", false);
+ Boolean keepOriginal = effectiveBoolean(properties, "keep_original",
false);
+ Boolean keepNoneChinese = effectiveBoolean(properties,
"keep_none_chinese", true);
+ Boolean keepNoneChineseTogether = effectiveBoolean(properties,
"keep_none_chinese_together", true);
+ Boolean noneChinesePinyinTokenize = effectiveBoolean(properties,
"none_chinese_pinyin_tokenize", true);
+ Boolean ignorePinyinOffset = effectiveBoolean(properties,
"ignore_pinyin_offset", true);
+ Boolean keepJoinedFullPinyin = effectiveBoolean(properties,
"keep_joined_full_pinyin", false);
+ // Only the pinyin tokenizer also consults
keep_none_chinese_in_joined_full_pinyin, when no
+ // other setting settles whether it emits an untokenized ASCII buffer.
+ boolean tokenizerReadsJoinedSetting = expectedType ==
IndexPolicyTypeEnum.TOKENIZER
+ && !Boolean.FALSE.equals(keepNoneChinese)
+ && !Boolean.FALSE.equals(keepNoneChineseTogether)
+ && !Boolean.TRUE.equals(noneChinesePinyinTokenize)
+ && !Boolean.TRUE.equals(keepFirstLetter)
+ && !Boolean.TRUE.equals(keepSeparateFirstLetter)
+ && !Boolean.TRUE.equals(keepFullPinyin);
+
+ // The tokenizer only trims its candidates, and without the original
every candidate is
+ // pinyin or ASCII alphanumerics; the token filter also trims the
incoming token.
+ if (expectedType == IndexPolicyTypeEnum.TOKENIZER &&
Boolean.FALSE.equals(keepOriginal)) {
+ properties.remove("trim_whitespace");
+ }
+
+ // With every per-character candidate disabled at most one term is
emitted per input, the
+ // joined pinyin or the token filter's fallback original, so nothing
can be a duplicate.
+ if (Boolean.FALSE.equals(keepFirstLetter) &&
Boolean.FALSE.equals(keepFullPinyin)
Review Comment:
[P1] Drop `remove_duplicated_term` for first-letter-only Pinyin too. With
`keep_first_letter=true` and full, joined, separate-first-letter, original,
non-Chinese, and separate-Chinese outputs disabled, each tokenizer input can
emit only the one aggregate, and the token filter emits either that aggregate
or one fallback original; its dedup set also resets for every upstream token.
The flag therefore cannot affect terms, positions, offsets, or provenance, but
this predicate requires `keep_first_letter=false` and gives equivalent aliases
different identities through CREATE/ALTER. This is a residual of the earlier
at-most-one-output fix: derive cardinality from all effective gates and cover
the first-letter-only shape, retaining any multi-output configuration as a
negative.
##########
fe/fe-core/src/main/java/org/apache/doris/analysis/invertedindex/AnalyzerIdentityBuilder.java:
##########
@@ -206,17 +900,21 @@ private static String resolveTokenFilterIdentity(String
filterList) {
String[] filters = filterList.split(",\\s*");
// DO NOT sort - filter order is semantically significant
- for (int i = 0; i < filters.length; i++) {
- String filter = filters[i].trim();
- if (i > 0) {
- sb.append(",");
+ String previous = null;
+ for (String filterName : filters) {
+ String filter = resolveComponentIdentity(filterName.trim(),
IndexPolicyTypeEnum.TOKEN_FILTER);
+ if (Strings.isNullOrEmpty(filter)) {
+ continue;
}
-
- if (IndexPolicy.BUILTIN_TOKEN_FILTERS.contains(filter)) {
- sb.append(filter);
- } else {
- sb.append(resolveComponentIdentity(filter,
IndexPolicyTypeEnum.TOKEN_FILTER));
+ // Repeating an idempotent filter leaves the terms, offsets and
provenance unchanged.
+ if (filter.equals(previous) &&
IDEMPOTENT_TOKEN_FILTERS.contains(filter)) {
Review Comment:
[P1] Treat the canonical default `asciifolding` filter as idempotent here
too. With `preserve_original=false` (the bare identity), the first pass either
leaves a code point unchanged or maps it to ASCII, so a second pass cannot
change the term or position. Exact source maps are delegated on both passes;
when the first pass changes rune count or removes malformed bytes, it publishes
a conservative span that the unchanged second pass also delegates. Keeping both
entries therefore gives equivalent aliases different identities through
CREATE/ALTER. This is a residual beyond the earlier lowercase case: add bare
`asciifolding` to the proven-idempotent set, while retaining
`preserve_original=true` and non-idempotent filters as negatives.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]