github-actions[bot] commented on code in PR #68661:
URL: https://github.com/apache/doris/pull/68661#discussion_r4198039056


##########
be/src/exprs/function/match.cpp:
##########
@@ -44,6 +44,165 @@ const InvertedIndexAnalyzerCtx* 
get_match_analyzer_ctx(FunctionContext* context)
     return analyzer_ctx;
 }
 
+bool match_phrase_tokens(const std::vector<segment_v2::TermInfo>& data_tokens,
+                         const std::vector<segment_v2::TermInfo>& 
query_tokens) {
+    bool matched = false;
+    auto data_it = data_tokens.begin();
+    while (data_it != data_tokens.end()) {
+        data_it = std::find_if(data_it, data_tokens.end(), [&](const 
segment_v2::TermInfo& info) {
+            return info.get_single_term() == query_tokens[0].get_single_term();
+        });
+        if (data_it != data_tokens.end()) {
+            matched = true;
+            auto data_it_next = ++data_it;
+            auto query_it = query_tokens.begin() + 1;
+            while (query_it != query_tokens.end()) {
+                if (data_it_next == data_tokens.end() ||
+                    data_it_next->get_single_term() != 
query_it->get_single_term()) {
+                    matched = false;
+                    break;
+                }
+                query_it++;
+                data_it_next++;
+            }
+
+            if (matched) {
+                break;
+            }
+        }
+    }
+
+    return matched;
+}
+
+bool match_phrase_prefix_tokens(const std::vector<segment_v2::TermInfo>& 
data_tokens,
+                                const std::vector<segment_v2::TermInfo>& 
query_tokens) {
+    if (data_tokens.size() < query_tokens.size()) {
+        return false;
+    }
+    const auto dis_count = data_tokens.size() - query_tokens.size();
+
+    for (size_t j = 0; j < dis_count + 1; j++) {
+        if (data_tokens[j].get_single_term() == 
query_tokens[0].get_single_term() ||
+            query_tokens.size() == 1) {
+            bool match = true;
+            for (size_t k = 0; k < query_tokens.size(); k++) {
+                const std::string& data_token = data_tokens[j + 
k].get_single_term();
+                const std::string& query_token = 
query_tokens[k].get_single_term();
+                if (k == query_tokens.size() - 1) {
+                    if (!data_token.starts_with(query_token)) {
+                        match = false;
+                        break;
+                    }
+                } else {
+                    if (data_token != query_token) {
+                        match = false;
+                        break;
+                    }
+                }
+            }
+            if (match) {
+                return true;
+            }
+        }
+    }
+    return false;
+}
+
+bool match_phrase_edge_tokens(const std::vector<segment_v2::TermInfo>& 
data_tokens,
+                              const std::vector<segment_v2::TermInfo>& 
query_tokens) {
+    if (data_tokens.size() < query_tokens.size()) {
+        return false;
+    }
+    const auto dis_count = data_tokens.size() - query_tokens.size();
+
+    for (size_t j = 0; j < dis_count + 1; j++) {
+        bool match = true;
+        if (query_tokens.size() == 1) {
+            if 
(data_tokens[j].get_single_term().find(query_tokens[0].get_single_term()) ==
+                std::string::npos) {
+                match = false;
+            }
+        } else {
+            for (size_t k = 0; k < query_tokens.size(); k++) {
+                const std::string& data_token = data_tokens[j + 
k].get_single_term();
+                const std::string& query_token = 
query_tokens[k].get_single_term();
+                if (k == 0) {
+                    if (!data_token.ends_with(query_token)) {
+                        match = false;
+                        break;
+                    }
+                } else if (k == query_tokens.size() - 1) {
+                    if (!data_token.starts_with(query_token)) {
+                        match = false;
+                        break;
+                    }
+                } else {
+                    if (data_token != query_token) {
+                        match = false;
+                        break;
+                    }
+                }
+            }
+        }
+        if (match) {
+            return true;
+        }
+    }
+    return false;
+}
+
+template <typename Callback>
+bool for_each_data_element_tokens(const FunctionMatchBase& function, const 
std::string& column_name,
+                                  const InvertedIndexAnalyzerCtx* analyzer_ctx,
+                                  const ColumnString* string_col, size_t row,
+                                  const ColumnArray::Offsets64* array_offsets,
+                                  const ColumnUInt8::Container* 
array_element_null_map,
+                                  Callback&& callback) {
+    const size_t begin = array_offsets ? (row == 0 ? 0 : (*array_offsets)[row 
- 1]) : row;
+    const size_t end = array_offsets ? (*array_offsets)[row] : row + 1;
+    int32_t unused_array_offset = 0;
+    for (size_t element = begin; element < end; ++element) {
+        if (array_element_null_map && (*array_element_null_map)[element]) {
+            continue;
+        }
+        auto tokens = function.analyse_data_token(column_name, analyzer_ctx, 
string_col, element,
+                                                  nullptr, 
unused_array_offset);
+        if (callback(tokens)) {
+            return true;
+        }
+    }
+    return false;
+}
+
+using PhraseMatcher = bool (*)(const std::vector<segment_v2::TermInfo>&,
+                               const std::vector<segment_v2::TermInfo>&);
+
+bool match_phrase_data_tokens(const FunctionMatchBase& function, const 
std::string& column_name,
+                              const InvertedIndexAnalyzerCtx* analyzer_ctx,
+                              const ColumnString* string_col, size_t row,
+                              const ColumnArray::Offsets64* array_offsets,
+                              const ColumnUInt8::Container* 
array_element_null_map,
+                              const std::vector<segment_v2::TermInfo>& 
query_tokens,
+                              PhraseMatcher matcher) {
+    std::vector<segment_v2::TermInfo> window;
+    window.reserve(query_tokens.size());
+    return for_each_data_element_tokens(
+            function, column_name, analyzer_ctx, string_col, row, 
array_offsets,
+            array_element_null_map, [&](std::vector<segment_v2::TermInfo>& 
tokens) {
+                for (auto& token : tokens) {
+                    if (window.size() == query_tokens.size()) {
+                        window.erase(window.begin());
+                    }

Review Comment:
   [P2] Advance the phrase window without shifting every token. For a 
10,000-token English-analyzed STRING of `alpha` and a 1,000-token 
MATCH_PHRASE_PREFIX or MATCH_PHRASE_EDGE query starting with `beta`, the 
previous loop rejected each start on its first term. This `erase(begin())` 
instead moves 999 TermInfo values for each of 9,000 advances, about 9 million 
moves before those same quick rejections. The extra O(N Q) work also affects 
ordinary misses. Use a circular window or indices so advancing one token stays 
O(1).



##########
be/src/exprs/function/match.cpp:
##########
@@ -44,6 +44,165 @@ const InvertedIndexAnalyzerCtx* 
get_match_analyzer_ctx(FunctionContext* context)
     return analyzer_ctx;
 }
 
+bool match_phrase_tokens(const std::vector<segment_v2::TermInfo>& data_tokens,
+                         const std::vector<segment_v2::TermInfo>& 
query_tokens) {
+    bool matched = false;
+    auto data_it = data_tokens.begin();
+    while (data_it != data_tokens.end()) {
+        data_it = std::find_if(data_it, data_tokens.end(), [&](const 
segment_v2::TermInfo& info) {
+            return info.get_single_term() == query_tokens[0].get_single_term();
+        });
+        if (data_it != data_tokens.end()) {
+            matched = true;
+            auto data_it_next = ++data_it;
+            auto query_it = query_tokens.begin() + 1;
+            while (query_it != query_tokens.end()) {
+                if (data_it_next == data_tokens.end() ||
+                    data_it_next->get_single_term() != 
query_it->get_single_term()) {
+                    matched = false;
+                    break;
+                }
+                query_it++;
+                data_it_next++;
+            }
+
+            if (matched) {
+                break;
+            }
+        }
+    }
+
+    return matched;
+}
+
+bool match_phrase_prefix_tokens(const std::vector<segment_v2::TermInfo>& 
data_tokens,
+                                const std::vector<segment_v2::TermInfo>& 
query_tokens) {
+    if (data_tokens.size() < query_tokens.size()) {
+        return false;
+    }
+    const auto dis_count = data_tokens.size() - query_tokens.size();
+
+    for (size_t j = 0; j < dis_count + 1; j++) {
+        if (data_tokens[j].get_single_term() == 
query_tokens[0].get_single_term() ||
+            query_tokens.size() == 1) {
+            bool match = true;
+            for (size_t k = 0; k < query_tokens.size(); k++) {
+                const std::string& data_token = data_tokens[j + 
k].get_single_term();
+                const std::string& query_token = 
query_tokens[k].get_single_term();
+                if (k == query_tokens.size() - 1) {
+                    if (!data_token.starts_with(query_token)) {
+                        match = false;
+                        break;
+                    }
+                } else {
+                    if (data_token != query_token) {
+                        match = false;
+                        break;
+                    }
+                }
+            }
+            if (match) {
+                return true;
+            }
+        }
+    }
+    return false;
+}
+
+bool match_phrase_edge_tokens(const std::vector<segment_v2::TermInfo>& 
data_tokens,
+                              const std::vector<segment_v2::TermInfo>& 
query_tokens) {
+    if (data_tokens.size() < query_tokens.size()) {
+        return false;
+    }
+    const auto dis_count = data_tokens.size() - query_tokens.size();
+
+    for (size_t j = 0; j < dis_count + 1; j++) {
+        bool match = true;
+        if (query_tokens.size() == 1) {
+            if 
(data_tokens[j].get_single_term().find(query_tokens[0].get_single_term()) ==
+                std::string::npos) {
+                match = false;
+            }
+        } else {
+            for (size_t k = 0; k < query_tokens.size(); k++) {
+                const std::string& data_token = data_tokens[j + 
k].get_single_term();
+                const std::string& query_token = 
query_tokens[k].get_single_term();
+                if (k == 0) {
+                    if (!data_token.ends_with(query_token)) {
+                        match = false;
+                        break;
+                    }
+                } else if (k == query_tokens.size() - 1) {
+                    if (!data_token.starts_with(query_token)) {
+                        match = false;
+                        break;
+                    }
+                } else {
+                    if (data_token != query_token) {
+                        match = false;
+                        break;
+                    }
+                }
+            }
+        }
+        if (match) {
+            return true;
+        }
+    }
+    return false;
+}
+
+template <typename Callback>
+bool for_each_data_element_tokens(const FunctionMatchBase& function, const 
std::string& column_name,
+                                  const InvertedIndexAnalyzerCtx* analyzer_ctx,
+                                  const ColumnString* string_col, size_t row,
+                                  const ColumnArray::Offsets64* array_offsets,
+                                  const ColumnUInt8::Container* 
array_element_null_map,
+                                  Callback&& callback) {
+    const size_t begin = array_offsets ? (row == 0 ? 0 : (*array_offsets)[row 
- 1]) : row;
+    const size_t end = array_offsets ? (*array_offsets)[row] : row + 1;
+    int32_t unused_array_offset = 0;
+    for (size_t element = begin; element < end; ++element) {
+        if (array_element_null_map && (*array_element_null_map)[element]) {
+            continue;
+        }
+        auto tokens = function.analyse_data_token(column_name, analyzer_ctx, 
string_col, element,
+                                                  nullptr, 
unused_array_offset);
+        if (callback(tokens)) {
+            return true;
+        }
+    }
+    return false;
+}
+
+using PhraseMatcher = bool (*)(const std::vector<segment_v2::TermInfo>&,
+                               const std::vector<segment_v2::TermInfo>&);
+
+bool match_phrase_data_tokens(const FunctionMatchBase& function, const 
std::string& column_name,
+                              const InvertedIndexAnalyzerCtx* analyzer_ctx,
+                              const ColumnString* string_col, size_t row,
+                              const ColumnArray::Offsets64* array_offsets,
+                              const ColumnUInt8::Container* 
array_element_null_map,
+                              const std::vector<segment_v2::TermInfo>& 
query_tokens,
+                              PhraseMatcher matcher) {
+    std::vector<segment_v2::TermInfo> window;
+    window.reserve(query_tokens.size());
+    return for_each_data_element_tokens(
+            function, column_name, analyzer_ctx, string_col, row, 
array_offsets,
+            array_element_null_map, [&](std::vector<segment_v2::TermInfo>& 
tokens) {
+                for (auto& token : tokens) {
+                    if (window.size() == query_tokens.size()) {
+                        window.erase(window.begin());
+                    }
+                    window.emplace_back(std::move(token));
+                    if (window.size() == query_tokens.size() && 
matcher(window, query_tokens)) {
+                        return true;

Review Comment:
   [P1] Avoid rescanning the full phrase window for every token. With one 
English-analyzed STRING value containing 10,000 `alpha` tokens and a 
`MATCH_PHRASE` query of 999 `alpha` tokens followed by `beta`, each of the 
9,001 windows retries almost every candidate start. This performs about 4.5 
billion term comparisons here versus about 9.5 million in the previous single 
pass, even though the row is only roughly 60 KB. The fallback can therefore 
stall on a feasible nonmatching query. Keep incremental phrase state across 
tokens (or use a single candidate pass) so overlapping windows are not 
rechecked from scratch.



##########
be/src/exprs/function/match.cpp:
##########
@@ -334,39 +507,44 @@ Status FunctionMatchAll::execute_match(FunctionContext* 
context, const std::stri
         return Status::OK();
     }
 
-    auto current_src_array_offset = 0;
     for (int i = 0; i < input_rows_count; i++) {
-        auto data_tokens = analyse_data_token(column_name, analyzer_ctx, 
string_col, i,
-                                              array_offsets, 
current_src_array_offset);
-
-        // TODO: more efficient impl
-        auto find_count = 0;
-        for (auto& term_info : query_tokens) {
-            auto it = std::find_if(data_tokens.begin(), data_tokens.end(),
-                                   [&](const segment_v2::TermInfo& info) {
-                                       return info.get_single_term() == 
term_info.get_single_term();
-                                   });
-            if (it != data_tokens.end()) {
-                ++find_count;
-            } else {
-                break;
-            }
-        }
-
-        if (find_count == query_tokens.size()) {
+        std::vector<uint8_t> found(query_tokens.size(), 0);
+        size_t remaining = query_tokens.size();

Review Comment:
   [P2] Defer the MATCH_ALL found bitmap until the row yields a token. This 
vector is allocated and zeroed before array traversal, so an empty array 
immediately discards it. For 100,000 empty rows and a 10,000-term query, that 
is 100,000 allocations and about 1 GB of cumulative zeroing with no token 
comparison; the previous loop stopped after its first miss without allocating 
query-sized state. Initialize or reuse this bitmap only when a nonempty token 
set needs it.



##########
be/src/exprs/function/match.cpp:
##########
@@ -44,6 +44,165 @@ const InvertedIndexAnalyzerCtx* 
get_match_analyzer_ctx(FunctionContext* context)
     return analyzer_ctx;
 }
 
+bool match_phrase_tokens(const std::vector<segment_v2::TermInfo>& data_tokens,
+                         const std::vector<segment_v2::TermInfo>& 
query_tokens) {
+    bool matched = false;
+    auto data_it = data_tokens.begin();
+    while (data_it != data_tokens.end()) {
+        data_it = std::find_if(data_it, data_tokens.end(), [&](const 
segment_v2::TermInfo& info) {
+            return info.get_single_term() == query_tokens[0].get_single_term();
+        });
+        if (data_it != data_tokens.end()) {
+            matched = true;
+            auto data_it_next = ++data_it;
+            auto query_it = query_tokens.begin() + 1;
+            while (query_it != query_tokens.end()) {
+                if (data_it_next == data_tokens.end() ||
+                    data_it_next->get_single_term() != 
query_it->get_single_term()) {
+                    matched = false;
+                    break;
+                }
+                query_it++;
+                data_it_next++;
+            }
+
+            if (matched) {
+                break;
+            }
+        }
+    }
+
+    return matched;
+}
+
+bool match_phrase_prefix_tokens(const std::vector<segment_v2::TermInfo>& 
data_tokens,
+                                const std::vector<segment_v2::TermInfo>& 
query_tokens) {
+    if (data_tokens.size() < query_tokens.size()) {
+        return false;
+    }
+    const auto dis_count = data_tokens.size() - query_tokens.size();
+
+    for (size_t j = 0; j < dis_count + 1; j++) {
+        if (data_tokens[j].get_single_term() == 
query_tokens[0].get_single_term() ||
+            query_tokens.size() == 1) {
+            bool match = true;
+            for (size_t k = 0; k < query_tokens.size(); k++) {
+                const std::string& data_token = data_tokens[j + 
k].get_single_term();
+                const std::string& query_token = 
query_tokens[k].get_single_term();
+                if (k == query_tokens.size() - 1) {
+                    if (!data_token.starts_with(query_token)) {
+                        match = false;
+                        break;
+                    }
+                } else {
+                    if (data_token != query_token) {
+                        match = false;
+                        break;
+                    }
+                }
+            }
+            if (match) {
+                return true;
+            }
+        }
+    }
+    return false;
+}
+
+bool match_phrase_edge_tokens(const std::vector<segment_v2::TermInfo>& 
data_tokens,
+                              const std::vector<segment_v2::TermInfo>& 
query_tokens) {
+    if (data_tokens.size() < query_tokens.size()) {
+        return false;
+    }
+    const auto dis_count = data_tokens.size() - query_tokens.size();
+
+    for (size_t j = 0; j < dis_count + 1; j++) {
+        bool match = true;
+        if (query_tokens.size() == 1) {
+            if 
(data_tokens[j].get_single_term().find(query_tokens[0].get_single_term()) ==
+                std::string::npos) {
+                match = false;
+            }
+        } else {
+            for (size_t k = 0; k < query_tokens.size(); k++) {
+                const std::string& data_token = data_tokens[j + 
k].get_single_term();
+                const std::string& query_token = 
query_tokens[k].get_single_term();
+                if (k == 0) {
+                    if (!data_token.ends_with(query_token)) {
+                        match = false;
+                        break;
+                    }
+                } else if (k == query_tokens.size() - 1) {
+                    if (!data_token.starts_with(query_token)) {
+                        match = false;
+                        break;
+                    }
+                } else {
+                    if (data_token != query_token) {
+                        match = false;
+                        break;
+                    }
+                }
+            }
+        }
+        if (match) {
+            return true;
+        }
+    }
+    return false;
+}
+
+template <typename Callback>
+bool for_each_data_element_tokens(const FunctionMatchBase& function, const 
std::string& column_name,
+                                  const InvertedIndexAnalyzerCtx* analyzer_ctx,
+                                  const ColumnString* string_col, size_t row,
+                                  const ColumnArray::Offsets64* array_offsets,
+                                  const ColumnUInt8::Container* 
array_element_null_map,
+                                  Callback&& callback) {
+    const size_t begin = array_offsets ? (row == 0 ? 0 : (*array_offsets)[row 
- 1]) : row;
+    const size_t end = array_offsets ? (*array_offsets)[row] : row + 1;
+    int32_t unused_array_offset = 0;
+    for (size_t element = begin; element < end; ++element) {
+        if (array_element_null_map && (*array_element_null_map)[element]) {
+            continue;
+        }
+        auto tokens = function.analyse_data_token(column_name, analyzer_ctx, 
string_col, element,
+                                                  nullptr, 
unused_array_offset);
+        if (callback(tokens)) {
+            return true;

Review Comment:
   [P2] Skip zero-token elements before calling membership matchers. An 
English-analyzed array can contain 100,000 empty strings, each yielding no 
tokens. With a 1,000-term query, `MATCH_ANY` and `MATCH_ALL` still run their 
entire query loop for every empty element, about 100 million useless 
iterations; the previous row-level loops checked the empty token vector once 
(and ALL stopped at the first miss). Continue when `tokens.empty()` so analyzed 
empty elements do not multiply query work. Keyword empty strings still produce 
a token and should remain matchable.



##########
be/src/exprs/function/match.cpp:
##########
@@ -44,6 +44,165 @@ const InvertedIndexAnalyzerCtx* 
get_match_analyzer_ctx(FunctionContext* context)
     return analyzer_ctx;
 }
 
+bool match_phrase_tokens(const std::vector<segment_v2::TermInfo>& data_tokens,
+                         const std::vector<segment_v2::TermInfo>& 
query_tokens) {
+    bool matched = false;
+    auto data_it = data_tokens.begin();
+    while (data_it != data_tokens.end()) {
+        data_it = std::find_if(data_it, data_tokens.end(), [&](const 
segment_v2::TermInfo& info) {
+            return info.get_single_term() == query_tokens[0].get_single_term();
+        });
+        if (data_it != data_tokens.end()) {
+            matched = true;
+            auto data_it_next = ++data_it;
+            auto query_it = query_tokens.begin() + 1;
+            while (query_it != query_tokens.end()) {
+                if (data_it_next == data_tokens.end() ||
+                    data_it_next->get_single_term() != 
query_it->get_single_term()) {
+                    matched = false;
+                    break;
+                }
+                query_it++;
+                data_it_next++;
+            }
+
+            if (matched) {
+                break;
+            }
+        }
+    }
+
+    return matched;
+}
+
+bool match_phrase_prefix_tokens(const std::vector<segment_v2::TermInfo>& 
data_tokens,
+                                const std::vector<segment_v2::TermInfo>& 
query_tokens) {
+    if (data_tokens.size() < query_tokens.size()) {
+        return false;
+    }
+    const auto dis_count = data_tokens.size() - query_tokens.size();
+
+    for (size_t j = 0; j < dis_count + 1; j++) {
+        if (data_tokens[j].get_single_term() == 
query_tokens[0].get_single_term() ||
+            query_tokens.size() == 1) {
+            bool match = true;
+            for (size_t k = 0; k < query_tokens.size(); k++) {
+                const std::string& data_token = data_tokens[j + 
k].get_single_term();
+                const std::string& query_token = 
query_tokens[k].get_single_term();
+                if (k == query_tokens.size() - 1) {
+                    if (!data_token.starts_with(query_token)) {
+                        match = false;
+                        break;
+                    }
+                } else {
+                    if (data_token != query_token) {
+                        match = false;
+                        break;
+                    }
+                }
+            }
+            if (match) {
+                return true;
+            }
+        }
+    }
+    return false;
+}
+
+bool match_phrase_edge_tokens(const std::vector<segment_v2::TermInfo>& 
data_tokens,
+                              const std::vector<segment_v2::TermInfo>& 
query_tokens) {
+    if (data_tokens.size() < query_tokens.size()) {
+        return false;
+    }
+    const auto dis_count = data_tokens.size() - query_tokens.size();
+
+    for (size_t j = 0; j < dis_count + 1; j++) {
+        bool match = true;
+        if (query_tokens.size() == 1) {
+            if 
(data_tokens[j].get_single_term().find(query_tokens[0].get_single_term()) ==
+                std::string::npos) {
+                match = false;
+            }
+        } else {
+            for (size_t k = 0; k < query_tokens.size(); k++) {
+                const std::string& data_token = data_tokens[j + 
k].get_single_term();
+                const std::string& query_token = 
query_tokens[k].get_single_term();
+                if (k == 0) {
+                    if (!data_token.ends_with(query_token)) {
+                        match = false;
+                        break;
+                    }
+                } else if (k == query_tokens.size() - 1) {
+                    if (!data_token.starts_with(query_token)) {
+                        match = false;
+                        break;
+                    }
+                } else {
+                    if (data_token != query_token) {
+                        match = false;
+                        break;
+                    }
+                }
+            }
+        }
+        if (match) {
+            return true;
+        }
+    }
+    return false;
+}
+
+template <typename Callback>
+bool for_each_data_element_tokens(const FunctionMatchBase& function, const 
std::string& column_name,
+                                  const InvertedIndexAnalyzerCtx* analyzer_ctx,
+                                  const ColumnString* string_col, size_t row,
+                                  const ColumnArray::Offsets64* array_offsets,
+                                  const ColumnUInt8::Container* 
array_element_null_map,
+                                  Callback&& callback) {
+    const size_t begin = array_offsets ? (row == 0 ? 0 : (*array_offsets)[row 
- 1]) : row;
+    const size_t end = array_offsets ? (*array_offsets)[row] : row + 1;
+    int32_t unused_array_offset = 0;
+    for (size_t element = begin; element < end; ++element) {
+        if (array_element_null_map && (*array_element_null_map)[element]) {
+            continue;
+        }
+        auto tokens = function.analyse_data_token(column_name, analyzer_ctx, 
string_col, element,
+                                                  nullptr, 
unused_array_offset);
+        if (callback(tokens)) {
+            return true;
+        }
+    }
+    return false;
+}
+
+using PhraseMatcher = bool (*)(const std::vector<segment_v2::TermInfo>&,
+                               const std::vector<segment_v2::TermInfo>&);
+
+bool match_phrase_data_tokens(const FunctionMatchBase& function, const 
std::string& column_name,
+                              const InvertedIndexAnalyzerCtx* analyzer_ctx,
+                              const ColumnString* string_col, size_t row,
+                              const ColumnArray::Offsets64* array_offsets,
+                              const ColumnUInt8::Container* 
array_element_null_map,
+                              const std::vector<segment_v2::TermInfo>& 
query_tokens,
+                              PhraseMatcher matcher) {
+    std::vector<segment_v2::TermInfo> window;
+    window.reserve(query_tokens.size());
+    return for_each_data_element_tokens(

Review Comment:
   [P2] Allocate the phrase window only when a row yields tokens. 
`match_phrase_data_tokens` runs for every row and this 
`reserve(query_tokens.size())` executes before it discovers that an array is 
empty. With 100,000 empty array rows and a 10,000-term query, the fallback 
makes 100,000 large query-sized heap allocations and frees them without 
examining a token; the previous empty-row path needed no such buffer. Grow the 
window as tokens arrive, or reuse capacity across rows.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to