This is an automated email from the ASF dual-hosted git repository.
eldenmoon pushed a commit to branch branch-4.1
in repository https://gitbox.apache.org/repos/asf/doris.git
The following commit(s) were added to refs/heads/branch-4.1 by this push:
new 851858d5702 branch-4.1: [fix](inverted index) Add a norms index
property and a BE config to skip norms on variant paths (#68140)
851858d5702 is described below
commit 851858d5702cb9ab48c566c0959f08267b1d7883
Author: lihangyu <[email protected]>
AuthorDate: Sun Sep 20 00:30:14 2026 +0800
branch-4.1: [fix](inverted index) Add a norms index property and a BE
config to skip norms on variant paths (#68140)
---
be/src/common/config.cpp | 5 +
be/src/common/config.h | 5 +
.../storage/compaction/collection_statistics.cpp | 20 +-
.../index/inverted/inverted_index_parser.cpp | 7 +
.../storage/index/inverted/inverted_index_parser.h | 7 +
.../index/inverted/inverted_index_writer.cpp | 17 +-
.../index/inverted/similarity/bm25_similarity.cpp | 2 +-
.../storage/segment/inverted_index_writer_test.cpp | 106 +++++++-
.../apache/doris/analysis/InvertedIndexUtil.java | 9 +
.../analysis/InvertedIndexNormsPropertyTest.java | 56 ++++
.../test_variant_subcolumn_index_norms.out | 16 ++
.../test_variant_subcolumn_index_norms.groovy | 285 +++++++++++++++++++++
12 files changed, 527 insertions(+), 8 deletions(-)
diff --git a/be/src/common/config.cpp b/be/src/common/config.cpp
index 5c2e0ebf8d6..4b0ead5a7a8 100644
--- a/be/src/common/config.cpp
+++ b/be/src/common/config.cpp
@@ -1368,6 +1368,11 @@ DEFINE_mBool(debug_inverted_index_compaction, "false");
DEFINE_mBool(inverted_index_ram_dir_enable, "true");
// wheather index by RAM directory when base compaction
DEFINE_mBool(inverted_index_ram_dir_enable_when_base_compaction, "true");
+// Norms cost one byte per segment row, including rows that hold no value for
the field. A segment
+// holds one index per variant path, so writing norms for them costs rows *
paths bytes. Turn this on
+// to leave norms out of every index on a variant path, whatever its "norms"
property says; BM25
+// scoring (score()) on those indexes then fails.
+DEFINE_mBool(inverted_index_skip_norms_for_variant, "false");
// use num_broadcast_buffer blocks as buffer to do broadcast
DEFINE_Int32(num_broadcast_buffer, "32");
diff --git a/be/src/common/config.h b/be/src/common/config.h
index fca3d1c6534..94a9fd88f86 100644
--- a/be/src/common/config.h
+++ b/be/src/common/config.h
@@ -1388,6 +1388,11 @@ DECLARE_mBool(debug_inverted_index_compaction);
DECLARE_mBool(inverted_index_ram_dir_enable);
// wheather index by RAM directory when base compaction
DECLARE_mBool(inverted_index_ram_dir_enable_when_base_compaction);
+// Norms cost one byte per segment row, including rows that hold no value for
the field. A segment
+// holds one index per variant path, so writing norms for them costs rows *
paths bytes. Turn this on
+// to leave norms out of every index on a variant path, whatever its "norms"
property says; BM25
+// scoring (score()) on those indexes then fails.
+DECLARE_mBool(inverted_index_skip_norms_for_variant);
// use num_broadcast_buffer blocks as buffer to do broadcast
DECLARE_Int32(num_broadcast_buffer);
diff --git a/be/src/storage/compaction/collection_statistics.cpp
b/be/src/storage/compaction/collection_statistics.cpp
index 6a680ad036c..8f21b18e34f 100644
--- a/be/src/storage/compaction/collection_statistics.cpp
+++ b/be/src/storage/compaction/collection_statistics.cpp
@@ -205,8 +205,22 @@ Status CollectionStatistics::process_segment(const
RowsetSharedPtr& rowset, int3
#endif
total_seg_num_docs = std::max(total_seg_num_docs,
index_reader->maxDoc());
- _total_num_tokens[ws_field_name] +=
-
index_reader->sumTotalTermFreq(ws_field_name.c_str()).value_or(0);
+ // BM25 on an analyzed index needs the record length of every row, and
CLucene keeps
+ // them, together with the field's token count, in the norms. A
segment written without
+ // norms would feed a zero avgdl, or rank its rows as zero-length
documents next to the
+ // segments that have norms, so refuse to score the collection. An
index that is not
+ // analyzed never writes norms and is left as it is.
+ const auto token_count =
index_reader->sumTotalTermFreq(ws_field_name.c_str());
+ if (!token_count.has_value() &&
+ segment_v2::inverted_index::InvertedIndexAnalyzer::should_analyzer(
+ collect_info.index_meta->properties())) {
+ return Status::Error<ErrorCode::INVERTED_INDEX_NOT_SUPPORTED>(
+ "BM25 scoring requires norms, but segment {} was written
without norms for "
+ "field {}. Norms are left out when the index sets
\"norms\" = \"false\" or, "
+ "for a variant path, when
inverted_index_skip_norms_for_variant is on",
+ seg_path, StringHelper::to_string(ws_field_name));
+ }
+ _total_num_tokens[ws_field_name] += token_count.value_or(0);
for (const auto& term_info : collect_info.term_infos) {
auto iter = TermIterator::create(io_ctx, false, index_reader,
ws_field_name,
@@ -289,4 +303,4 @@ float CollectionStatistics::get_or_calculate_idf(const
std::wstring& lucene_col_
}
#include "common/compile_check_end.h"
-} // namespace doris
\ No newline at end of file
+} // namespace doris
diff --git a/be/src/storage/index/inverted/inverted_index_parser.cpp
b/be/src/storage/index/inverted/inverted_index_parser.cpp
index 47819cc62f6..e9573ce2739 100644
--- a/be/src/storage/index/inverted/inverted_index_parser.cpp
+++ b/be/src/storage/index/inverted/inverted_index_parser.cpp
@@ -104,6 +104,13 @@ std::string
get_parser_phrase_support_string_from_properties(
return INVERTED_INDEX_PARSER_PHRASE_SUPPORT_NO;
}
+bool get_index_norms_from_properties(const std::map<std::string, std::string>&
properties) {
+ if (auto it = properties.find(INVERTED_INDEX_NORMS_KEY); it !=
properties.end()) {
+ return it->second == INVERTED_INDEX_PARSER_TRUE;
+ }
+ return true;
+}
+
CharFilterMap get_parser_char_filter_map_from_properties(
const std::map<std::string, std::string>& properties) {
if (!properties.contains(INVERTED_INDEX_PARSER_CHAR_FILTER_TYPE)) {
diff --git a/be/src/storage/index/inverted/inverted_index_parser.h
b/be/src/storage/index/inverted/inverted_index_parser.h
index d2d3df47abd..8aee17b16de 100644
--- a/be/src/storage/index/inverted/inverted_index_parser.h
+++ b/be/src/storage/index/inverted/inverted_index_parser.h
@@ -82,6 +82,9 @@ const std::string INVERTED_INDEX_PARSER_PHRASE_SUPPORT_KEY =
"support_phrase";
const std::string INVERTED_INDEX_PARSER_PHRASE_SUPPORT_YES = "true";
const std::string INVERTED_INDEX_PARSER_PHRASE_SUPPORT_NO = "false";
+// Whether an analyzed index stores BM25 norms, which take one byte per row of
the segment.
+const std::string INVERTED_INDEX_NORMS_KEY = "norms";
+
const std::string INVERTED_INDEX_PARSER_CHAR_FILTER_TYPE = "char_filter_type";
const std::string INVERTED_INDEX_PARSER_CHAR_FILTER_PATTERN =
"char_filter_pattern";
const std::string INVERTED_INDEX_PARSER_CHAR_FILTER_REPLACEMENT =
"char_filter_replacement";
@@ -138,6 +141,10 @@ std::string get_parser_mode_string_from_properties(
std::string get_parser_phrase_support_string_from_properties(
const std::map<std::string, std::string>& properties);
+// Whether this index writes BM25 norms, which it does unless "norms" =
"false" says otherwise.
+// Norms cost one byte per row of the segment, including rows that have no
value for the field.
+bool get_index_norms_from_properties(const std::map<std::string, std::string>&
properties);
+
CharFilterMap get_parser_char_filter_map_from_properties(
const std::map<std::string, std::string>& properties);
diff --git a/be/src/storage/index/inverted/inverted_index_writer.cpp
b/be/src/storage/index/inverted/inverted_index_writer.cpp
index 8e4730cc063..01d48afb285 100644
--- a/be/src/storage/index/inverted/inverted_index_writer.cpp
+++ b/be/src/storage/index/inverted/inverted_index_writer.cpp
@@ -162,7 +162,20 @@ Status
InvertedIndexColumnWriter<field_type>::create_field(lucene::document::Fie
(*field)->setOmitTermFreqAndPositions(
!(get_parser_phrase_support_string_from_properties(_index_meta->properties()) ==
INVERTED_INDEX_PARSER_PHRASE_SUPPORT_YES));
- (*field)->setOmitNorms(false);
+ // An analyzed index writes norms unless its "norms" property says
otherwise. Norms cost one byte
+ // per segment row, including rows without a value, and a variant path
index (a field_pattern
+ // index, or the copy inherited by one extracted subcolumn, which carries
the path as its index
+ // suffix) is one of possibly thousands in a segment, so their norms can
dwarf the data.
+ // inverted_index_skip_norms_for_variant drops norms for those indexes
whatever their property
+ // says, so that a cluster can reclaim that space without rewriting its
index definitions.
+ const bool variant_path_index =
+ !_index_meta->get_index_suffix().empty() ||
!_index_meta->field_pattern().empty();
+ const bool skipped_by_config =
+ variant_path_index &&
config::inverted_index_skip_norms_for_variant;
+ if (_should_analyzer && !skipped_by_config &&
+ get_index_norms_from_properties(_index_meta->properties())) {
+ (*field)->setOmitNorms(false);
+ }
DBUG_EXECUTE_IF("InvertedIndexColumnWriter::create_field_v3", {
if (_index_file_writer->get_storage_format() !=
InvertedIndexStorageFormatPB::V3) {
return
Status::Error<doris::ErrorCode::INVERTED_INDEX_CLUCENE_ERROR>(
@@ -677,4 +690,4 @@ template class
InvertedIndexColumnWriter<FieldType::OLAP_FIELD_TYPE_IPV6>;
template class InvertedIndexColumnWriter<FieldType::OLAP_FIELD_TYPE_FLOAT>;
template class InvertedIndexColumnWriter<FieldType::OLAP_FIELD_TYPE_DOUBLE>;
-} // namespace doris::segment_v2
\ No newline at end of file
+} // namespace doris::segment_v2
diff --git a/be/src/storage/index/inverted/similarity/bm25_similarity.cpp
b/be/src/storage/index/inverted/similarity/bm25_similarity.cpp
index d3e1ba5ac63..e02242985b3 100644
--- a/be/src/storage/index/inverted/similarity/bm25_similarity.cpp
+++ b/be/src/storage/index/inverted/similarity/bm25_similarity.cpp
@@ -148,4 +148,4 @@ int32_t BM25Similarity::byte4_to_int(uint8_t b) {
}
#include "common/compile_check_end.h"
-} // namespace doris::segment_v2
\ No newline at end of file
+} // namespace doris::segment_v2
diff --git a/be/test/storage/segment/inverted_index_writer_test.cpp
b/be/test/storage/segment/inverted_index_writer_test.cpp
index c5bdf4c5547..6af028c13d1 100644
--- a/be/test/storage/segment/inverted_index_writer_test.cpp
+++ b/be/test/storage/segment/inverted_index_writer_test.cpp
@@ -362,7 +362,9 @@ public:
}
// Helper method to create an inverted index with tokenization enabled
- void create_tokenized_index(std::string_view rowset_id, int seg_id, bool
enable_analyzer) {
+ void create_tokenized_index(std::string_view rowset_id, int seg_id, bool
enable_analyzer,
+ const std::string& index_suffix = "",
+ const std::map<std::string, std::string>&
extra_properties = {}) {
auto tablet_schema = create_schema();
// Create index meta with tokenization setting
@@ -383,9 +385,15 @@ public:
// This will make should_analyzer() return true
(*properties)["parser"] = "standard";
}
+ for (const auto& [key, value] : extra_properties) {
+ (*properties)[key] = value;
+ }
TabletIndex idx_meta;
idx_meta.init_from_pb(*index_meta_pb.get());
+ if (!index_suffix.empty()) {
+ idx_meta.set_escaped_escaped_index_suffix_path(index_suffix);
+ }
std::string index_path_prefix
{InvertedIndexDescriptor::get_index_file_path_prefix(
local_segment_path(kTestDir, rowset_id, seg_id))};
@@ -1478,4 +1486,98 @@ TEST_F(InvertedIndexWriterTest,
FileCreationAndOutputErrorHandling) {
// but it should not crash
}
-} // namespace doris::segment_v2
\ No newline at end of file
+// Norms take one byte per segment row for every indexed path, so an index on
a variant path (a
+// field_pattern index, or the copy inherited by one extracted subcolumn,
which carries the path as
+// its index suffix) writes none by default. The "norms" property overrides
that per index.
+TEST_F(InvertedIndexWriterTest, NormsFollowIndexNormsProperty) {
+ auto make_index_meta = [](const std::string& index_suffix,
+ const std::map<std::string, std::string>&
extra_properties) {
+ TabletIndexPB index_meta_pb;
+ index_meta_pb.set_index_type(IndexType::INVERTED);
+ index_meta_pb.set_index_id(1);
+ index_meta_pb.set_index_name("test");
+ index_meta_pb.add_col_unique_id(1); // c2 column id
+ (*index_meta_pb.mutable_properties())["parser"] = "standard";
+ for (const auto& [key, value] : extra_properties) {
+ (*index_meta_pb.mutable_properties())[key] = value;
+ }
+ TabletIndex index_meta;
+ index_meta.init_from_pb(index_meta_pb);
+ if (!index_suffix.empty()) {
+ index_meta.set_escaped_escaped_index_suffix_path(index_suffix);
+ }
+ return index_meta;
+ };
+ auto path_prefix = [this](const std::string& rowset_id, int seg_id) {
+ return std::string
{InvertedIndexDescriptor::get_index_file_path_prefix(
+ local_segment_path(kTestDir, rowset_id, seg_id))};
+ };
+
+ bool original_skip_norms_for_variant =
config::inverted_index_skip_norms_for_variant;
+
+ // an analyzed index writes norms wherever it sits, and only "norms" =
"false" drops them
+ config::inverted_index_skip_norms_for_variant = false;
+
+ create_tokenized_index("plain_column_default", 0, true, "");
+ TabletIndex plain_default = make_index_meta("", {});
+ EXPECT_TRUE(check_norms_file_exists(path_prefix("plain_column_default",
0), &plain_default))
+ << "an analyzed index must write .nrm by default";
+
+ create_tokenized_index("plain_column_norms_off", 1, true, "", {{"norms",
"false"}});
+ TabletIndex plain_norms_off = make_index_meta("", {{"norms", "false"}});
+ EXPECT_FALSE(
+ check_norms_file_exists(path_prefix("plain_column_norms_off", 1),
&plain_norms_off))
+ << "norms = false must drop .nrm for an ordinary column index";
+
+ create_tokenized_index("variant_subcolumn_default", 2, true, "v.s_host");
+ TabletIndex subcolumn_default = make_index_meta("v.s_host", {});
+
EXPECT_TRUE(check_norms_file_exists(path_prefix("variant_subcolumn_default", 2),
+ &subcolumn_default))
+ << "a variant subcolumn index must write .nrm by default too";
+
+ create_tokenized_index("variant_subcolumn_norms_off", 3, true, "v.s_host",
+ {{"norms", "false"}});
+ TabletIndex subcolumn_norms_off = make_index_meta("v.s_host", {{"norms",
"false"}});
+
EXPECT_FALSE(check_norms_file_exists(path_prefix("variant_subcolumn_norms_off",
3),
+ &subcolumn_norms_off))
+ << "norms = false must drop .nrm for a variant subcolumn index";
+
+ create_tokenized_index("field_pattern_default", 4, true, "",
{{"field_pattern", "s_*"}});
+ TabletIndex field_pattern_default = make_index_meta("", {{"field_pattern",
"s_*"}});
+ EXPECT_TRUE(check_norms_file_exists(path_prefix("field_pattern_default",
4),
+ &field_pattern_default))
+ << "a field_pattern index must write .nrm by default too";
+
+ // the config drops norms for a variant path index whatever its property
says, and leaves every
+ // other index alone
+ config::inverted_index_skip_norms_for_variant = true;
+
+ create_tokenized_index("variant_subcolumn_skipped", 5, true, "v.s_host");
+ TabletIndex subcolumn_skipped = make_index_meta("v.s_host", {});
+
EXPECT_FALSE(check_norms_file_exists(path_prefix("variant_subcolumn_skipped",
5),
+ &subcolumn_skipped))
+ << "the config must drop .nrm for a variant subcolumn index";
+
+ create_tokenized_index("variant_subcolumn_norms_on_skipped", 6, true,
"v.s_host",
+ {{"norms", "true"}});
+ TabletIndex subcolumn_norms_on_skipped = make_index_meta("v.s_host",
{{"norms", "true"}});
+
EXPECT_FALSE(check_norms_file_exists(path_prefix("variant_subcolumn_norms_on_skipped",
6),
+ &subcolumn_norms_on_skipped))
+ << "the config must win over norms = true on a variant subcolumn
index";
+
+ create_tokenized_index("field_pattern_skipped", 7, true, "",
{{"field_pattern", "s_*"}});
+ TabletIndex field_pattern_skipped = make_index_meta("", {{"field_pattern",
"s_*"}});
+ EXPECT_FALSE(check_norms_file_exists(path_prefix("field_pattern_skipped",
7),
+ &field_pattern_skipped))
+ << "the config must drop .nrm for a field_pattern index";
+
+ create_tokenized_index("plain_column_not_skipped", 8, true, "");
+ TabletIndex plain_not_skipped = make_index_meta("", {});
+ EXPECT_TRUE(
+ check_norms_file_exists(path_prefix("plain_column_not_skipped",
8), &plain_not_skipped))
+ << "the config must leave an ordinary column index alone";
+
+ config::inverted_index_skip_norms_for_variant =
original_skip_norms_for_variant;
+}
+
+} // namespace doris::segment_v2
diff --git
a/fe/fe-core/src/main/java/org/apache/doris/analysis/InvertedIndexUtil.java
b/fe/fe-core/src/main/java/org/apache/doris/analysis/InvertedIndexUtil.java
index b7c31c33b7d..146a0da30c1 100644
--- a/fe/fe-core/src/main/java/org/apache/doris/analysis/InvertedIndexUtil.java
+++ b/fe/fe-core/src/main/java/org/apache/doris/analysis/InvertedIndexUtil.java
@@ -68,6 +68,8 @@ public class InvertedIndexUtil {
public static String INVERTED_INDEX_SUPPORT_PHRASE_KEY = "support_phrase";
+ public static String INVERTED_INDEX_NORMS_KEY = "norms";
+
public static String INVERTED_INDEX_PARSER_IGNORE_ABOVE_KEY =
"ignore_above";
public static String INVERTED_INDEX_PARSER_LOWERCASE_KEY = "lower_case";
@@ -251,6 +253,7 @@ public class InvertedIndexUtil {
INVERTED_INDEX_PARSER_KEY_ALIAS,
INVERTED_INDEX_PARSER_MODE_KEY,
INVERTED_INDEX_SUPPORT_PHRASE_KEY,
+ INVERTED_INDEX_NORMS_KEY,
INVERTED_INDEX_PARSER_CHAR_FILTER_TYPE,
INVERTED_INDEX_PARSER_CHAR_FILTER_PATTERN,
INVERTED_INDEX_PARSER_CHAR_FILTER_REPLACEMENT,
@@ -334,6 +337,12 @@ public class InvertedIndexUtil {
+ ", support_phrase must be true or false");
}
+ String norms = properties.get(INVERTED_INDEX_NORMS_KEY);
+ if (norms != null && !norms.matches("true|false")) {
+ throw new AnalysisException("Invalid inverted index 'norms' value:
" + norms
+ + ", norms must be true or false");
+ }
+
if (charFilterType != null) {
if
(!INVERTED_INDEX_CHAR_FILTER_CHAR_REPLACE.equals(charFilterType)) {
throw new AnalysisException("Invalid 'char_filter_type', only
'"
diff --git
a/fe/fe-core/src/test/java/org/apache/doris/analysis/InvertedIndexNormsPropertyTest.java
b/fe/fe-core/src/test/java/org/apache/doris/analysis/InvertedIndexNormsPropertyTest.java
new file mode 100644
index 00000000000..9e44b7c631a
--- /dev/null
+++
b/fe/fe-core/src/test/java/org/apache/doris/analysis/InvertedIndexNormsPropertyTest.java
@@ -0,0 +1,56 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements. See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership. The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License. You may obtain a copy of the License at
+//
+// http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied. See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+package org.apache.doris.analysis;
+
+import org.apache.doris.catalog.PrimitiveType;
+import org.apache.doris.common.AnalysisException;
+import org.apache.doris.thrift.TInvertedIndexFileStorageFormat;
+
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+import java.util.HashMap;
+import java.util.Map;
+
+public class InvertedIndexNormsPropertyTest {
+
+ @Test
+ public void testNormsPropertyAccepted() throws AnalysisException {
+ for (String value : new String[] {"true", "false"}) {
+ Map<String, String> properties = new HashMap<>();
+ properties.put("parser", "english");
+ properties.put("norms", value);
+
+ InvertedIndexUtil.checkInvertedIndexParser("col1",
PrimitiveType.STRING, properties,
+ TInvertedIndexFileStorageFormat.V2);
+ }
+ }
+
+ @Test
+ public void testNormsPropertyRejectsOtherValues() {
+ Map<String, String> properties = new HashMap<>();
+ properties.put("parser", "english");
+ properties.put("norms", "yes");
+
+ AnalysisException exception =
Assertions.assertThrows(AnalysisException.class,
+ () -> InvertedIndexUtil.checkInvertedIndexParser("col1",
PrimitiveType.STRING,
+ properties, TInvertedIndexFileStorageFormat.V2));
+ Assertions.assertTrue(exception.getMessage().contains("norms must be
true or false"),
+ exception.getMessage());
+ }
+}
diff --git
a/regression-test/data/inverted_index_p0/test_variant_subcolumn_index_norms.out
b/regression-test/data/inverted_index_p0/test_variant_subcolumn_index_norms.out
new file mode 100644
index 00000000000..a22a1151ed2
--- /dev/null
+++
b/regression-test/data/inverted_index_p0/test_variant_subcolumn_index_norms.out
@@ -0,0 +1,16 @@
+-- This file is automatically generated. You should know what you did if you
want to edit this
+-- !variant_subcolumn_score --
+2 0.6931
+3 0.61
+
+-- !variant_subcolumn_match_no_norms --
+2
+3
+
+-- !plain_column_score --
+1 0.5754
+3 0.8714
+
+-- !mixed_match --
+1
+2
diff --git
a/regression-test/suites/inverted_index_p0/test_variant_subcolumn_index_norms.groovy
b/regression-test/suites/inverted_index_p0/test_variant_subcolumn_index_norms.groovy
new file mode 100644
index 00000000000..1a82e6bbba1
--- /dev/null
+++
b/regression-test/suites/inverted_index_p0/test_variant_subcolumn_index_norms.groovy
@@ -0,0 +1,285 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements. See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership. The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License. You may obtain a copy of the License at
+//
+// http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied. See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+// An analyzed index writes dense BM25 norms (.nrm, one byte per segment row),
on a variant path as
+// on any other column, and "norms" = "false" drops them. Norms on a variant
path cost rows * paths
+// bytes, so inverted_index_skip_norms_for_variant leaves them out there
whatever the property says.
+// This covers both an index declared with a field_pattern and a whole-column
+// index on a VARIANT column, whose per-subcolumn copies inherit the
properties of the index they
+// come from. BM25 scoring needs norms: score() on an index without them
fails, also while only
+// some segments lack them, and MATCH filtering keeps working.
+// It flips a BE config, so it must not share the cluster with other suites.
+suite("test_variant_subcolumn_index_norms", "p0,nonConcurrent") {
+ if (isCloudMode()) {
+ return
+ }
+
+ sql """ set enable_match_without_inverted_index = false """
+ sql """ set default_variant_enable_typed_paths_to_sparse = false """
+ sql """ set default_variant_enable_doc_mode = false """
+
+ sql "DROP TABLE IF EXISTS test_variant_subcolumn_index_norms"
+ sql """
+ CREATE TABLE test_variant_subcolumn_index_norms (
+ id INT,
+ content TEXT,
+ v variant<
+ 's_*' : text,
+ 't_*' : text,
+ PROPERTIES("variant_max_subcolumns_count"="0")
+ >,
+ vd variant<
+ 'a_*' : text,
+ PROPERTIES("variant_max_subcolumns_count"="0")
+ >,
+ vn variant<
+ 'b_*' : text,
+ PROPERTIES("variant_max_subcolumns_count"="0")
+ >,
+ INDEX idx_content (content) USING INVERTED PROPERTIES(
+ "parser"="english",
+ "support_phrase"="true"
+ ),
+ INDEX idx_v_s (v) USING INVERTED PROPERTIES(
+ "parser"="english",
+ "support_phrase"="true",
+ "field_pattern"="s_*"
+ ),
+ INDEX idx_v_t (v) USING INVERTED PROPERTIES(
+ "parser"="english",
+ "support_phrase"="true",
+ "field_pattern"="t_*",
+ "norms"="false"
+ ),
+ INDEX idx_vd (vd) USING INVERTED PROPERTIES(
+ "parser"="english",
+ "support_phrase"="true"
+ ),
+ INDEX idx_vn (vn) USING INVERTED PROPERTIES(
+ "parser"="english",
+ "support_phrase"="true",
+ "norms"="false"
+ )
+ ) ENGINE=OLAP DUPLICATE KEY(id)
+ DISTRIBUTED BY HASH(id) BUCKETS 1
+ PROPERTIES (
+ "replication_allocation" = "tag.location.default: 1",
+ "disable_auto_compaction" = "true",
+ "inverted_index_storage_format" = "V2"
+ )
+ """
+ sql """ insert into test_variant_subcolumn_index_norms values
+ (1, 'alpha database server', parse_to_variant('{"s_host":"alpha
database server"}'),
+ parse_to_variant('{"a_host":"alpha database server"}'),
+ parse_to_variant('{"b_host":"alpha database server"}')),
+ (2, 'beta server cluster', parse_to_variant('{"s_host":"beta
server cluster", "s_note":"alpha", "t_note":"alpha"}'),
+ parse_to_variant('{"a_host":"beta server cluster"}'),
+ parse_to_variant('{"b_host":"beta server cluster"}')),
+ (3, 'alpha', parse_to_variant('{"s_note":"alpha alpha beta",
"t_note":"alpha beta"}'),
+ parse_to_variant('{"a_host":"alpha"}'),
+ parse_to_variant('{"b_host":"alpha"}')),
+ (4, 'gamma', parse_to_variant('{"other":"alpha"}'),
+ parse_to_variant('{"other":"alpha"}'),
+ parse_to_variant('{"other":"alpha"}'))
+ """
+ sql " sync "
+
+ // scores are printed so that a NaN (which compares greater than 0) cannot
slip through
+ order_qt_variant_subcolumn_score """
+ select id, round(score(), 4)
+ from test_variant_subcolumn_index_norms
+ where cast(v["s_note"] as string) match_phrase "alpha"
+ order by score() desc
+ limit 10
+ """
+ // without norms MATCH still filters, and score() is refused
+ order_qt_variant_subcolumn_match_no_norms """
+ select id
+ from test_variant_subcolumn_index_norms
+ where cast(v["t_note"] as string) match_phrase "alpha"
+ """
+ test {
+ sql """
+ select id, score()
+ from test_variant_subcolumn_index_norms
+ where cast(v["t_note"] as string) match_phrase "alpha"
+ order by score() desc
+ limit 10
+ """
+ exception "written without norms"
+ }
+ order_qt_plain_column_score """
+ select id, round(score(), 4)
+ from test_variant_subcolumn_index_norms
+ where content match_phrase "alpha"
+ order by score() desc
+ limit 10
+ """
+
+ def backendIdToIp = [:]
+ def backendIdToHttpPort = [:]
+ getBackendIpHttpPort(backendIdToIp, backendIdToHttpPort)
+ def normsBySuffixOf = { tableName ->
+ def tablet = sql_return_maparray("show tablets from ${tableName}")[0]
+ def (code, out, err) = http_client("GET", String.format(
+ "http://%s:%s/api/show_nested_index_file?tablet_id=%s",
+ backendIdToIp.get(tablet.BackendId),
backendIdToHttpPort.get(tablet.BackendId),
+ tablet.TabletId))
+ logger.info("show_nested_index_file of ${tableName} code=${code},
out=${out}, err=${err}")
+ assertEquals(0, code)
+ def norms = [:]
+ for (def rowset in parseJson(out.trim()).rowsets) {
+ for (def segment in rowset.segments) {
+ for (def index in segment.indices) {
+ norms[index.index_suffix] = index.files.any { file ->
file.name.endsWith(".nrm") }
+ }
+ }
+ }
+ logger.info("norms by index suffix of ${tableName}: ${norms}")
+ return norms
+ }
+ // the suffix is the escaped variant path, e.g. v%2Es%5Fhost for v.s_host
+ def normsOf = { norms, path ->
+ norms.find { suffix, hasNorms ->
+ suffix.replace("%2E", ".").replace("%5F", "_").contains(path)
+ }?.value
+ }
+
+ def normsBySuffix = normsBySuffixOf("test_variant_subcolumn_index_norms")
+ // an analyzed index writes norms wherever it sits, and "norms" = "false"
drops them
+ assertEquals(true, normsBySuffix[""])
+ assertEquals(true, normsOf(normsBySuffix, "s_host"))
+ assertEquals(true, normsOf(normsBySuffix, "s_note"))
+ assertEquals(false, normsOf(normsBySuffix, "t_note"))
+ // a whole-column index on a VARIANT column has no suffix of its own, but
every subcolumn copy
+ // inherits its properties: idx_vd keeps norms, idx_vn drops them because
it asks to
+ // keep them off.
+ assertEquals(true, normsOf(normsBySuffix, "a_host"))
+ assertEquals(false, normsOf(normsBySuffix, "b_host"))
+
+ // the skip is a dynamic BE config: with it turned on, every index on a
variant path leaves norms
+ // out, even one that asks for them, while an ordinary column index is
untouched
+ setBeConfigTemporary([inverted_index_skip_norms_for_variant: true]) {
+ sql "DROP TABLE IF EXISTS test_variant_subcolumn_index_norms_config"
+ sql """
+ CREATE TABLE test_variant_subcolumn_index_norms_config (
+ id INT,
+ content TEXT,
+ v variant<
+ 's_*' : text,
+ PROPERTIES("variant_max_subcolumns_count"="0")
+ >,
+ vf variant<
+ 'c_*' : text,
+ PROPERTIES("variant_max_subcolumns_count"="0")
+ >,
+ INDEX idx_content (content) USING INVERTED PROPERTIES(
+ "parser"="english",
+ "support_phrase"="true"
+ ),
+ INDEX idx_v_s (v) USING INVERTED PROPERTIES(
+ "parser"="english",
+ "support_phrase"="true",
+ "field_pattern"="s_*"
+ ),
+ INDEX idx_vf (vf) USING INVERTED PROPERTIES(
+ "parser"="english",
+ "support_phrase"="true",
+ "norms"="true"
+ )
+ ) ENGINE=OLAP DUPLICATE KEY(id)
+ DISTRIBUTED BY HASH(id) BUCKETS 1
+ PROPERTIES (
+ "replication_allocation" = "tag.location.default: 1",
+ "disable_auto_compaction" = "true",
+ "inverted_index_storage_format" = "V2"
+ )
+ """
+ sql """ insert into test_variant_subcolumn_index_norms_config values
+ (1, 'alpha database server',
parse_to_variant('{"s_host":"alpha database server"}'),
+ parse_to_variant('{"c_host":"alpha database server"}')),
+ (2, 'beta server cluster', parse_to_variant('{"s_host":"beta
server cluster"}'),
+ parse_to_variant('{"c_host":"beta server cluster"}'))
+ """
+ sql " sync "
+
+ def configNorms =
normsBySuffixOf("test_variant_subcolumn_index_norms_config")
+ assertEquals(false, normsOf(configNorms, "s_host"))
+ assertEquals(false, normsOf(configNorms, "c_host"))
+ assertEquals(true, configNorms[""])
+ }
+ test {
+ sql """
+ select id, score()
+ from test_variant_subcolumn_index_norms_config
+ where cast(vf["c_host"] as string) match_phrase "alpha"
+ order by score() desc
+ limit 10
+ """
+ exception "written without norms"
+ }
+
+ // segments with and without norms side by side, as while the config is
being turned on:
+ // MATCH still filters, and score() is refused rather than ranking the
rows of the newer
+ // segment as zero-length documents
+ sql "DROP TABLE IF EXISTS test_variant_subcolumn_index_norms_mixed"
+ sql """
+ CREATE TABLE test_variant_subcolumn_index_norms_mixed (
+ id INT,
+ v variant<
+ 's_*' : text,
+ PROPERTIES("variant_max_subcolumns_count"="0")
+ >,
+ INDEX idx_v_s (v) USING INVERTED PROPERTIES(
+ "parser"="english",
+ "support_phrase"="true",
+ "field_pattern"="s_*"
+ )
+ ) ENGINE=OLAP DUPLICATE KEY(id)
+ DISTRIBUTED BY HASH(id) BUCKETS 1
+ PROPERTIES (
+ "replication_allocation" = "tag.location.default: 1",
+ "disable_auto_compaction" = "true",
+ "inverted_index_storage_format" = "V2"
+ )
+ """
+ sql """ insert into test_variant_subcolumn_index_norms_mixed values
+ (1, parse_to_variant('{"s_note":"alpha database server"}'))
+ """
+ setBeConfigTemporary([inverted_index_skip_norms_for_variant: true]) {
+ sql """ insert into test_variant_subcolumn_index_norms_mixed values
+ (2, parse_to_variant('{"s_note":"alpha"}'))
+ """
+ }
+ sql " sync "
+
+ order_qt_mixed_match """
+ select id
+ from test_variant_subcolumn_index_norms_mixed
+ where cast(v["s_note"] as string) match_phrase "alpha"
+ """
+ test {
+ sql """
+ select id, score()
+ from test_variant_subcolumn_index_norms_mixed
+ where cast(v["s_note"] as string) match_phrase "alpha"
+ order by score() desc
+ limit 10
+ """
+ exception "written without norms"
+ }
+}
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]