This is an automated email from the ASF dual-hosted git repository.
mrhhsg pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/doris.git
The following commit(s) were added to refs/heads/master by this push:
new d4e51f9489f [fix](be) Preserve regexp anchors across repeated searches
(#68428)
d4e51f9489f is described below
commit d4e51f9489f27d8a3ba863259379a0e083ec1306
Author: Jerry Hu <[email protected]>
AuthorDate: Mon Sep 28 11:51:23 2026 +0800
[fix](be) Preserve regexp anchors across repeated searches (#68428)
### What problem does this PR solve?
Issue Number: None
Repeated regular-expression searches rebuilt the unprocessed suffix as a
new input. This changed the meaning of beginning anchors: for example,
`^a` could match at every suffix position in `aaa` instead of only at
the start of the original string.
This change keeps the original input stable and advances RE2's search
start position for `regexp_count`, `regexp_extract_all`,
`regexp_extract_all_array`, and `split_by_regexp`. It also avoids
rebuilding temporary suffix strings while advancing repeated matches.
For zero-length matches, progress is derived from the actual offset
returned by RE2 rather than the previous search start, and an empty
match at the end of the subject terminates the loop. This prevents
word-boundary patterns on long words from repeatedly rescanning the same
suffix.
Empty matches advance by one decoded UTF-8 character so the next search
cannot begin on a continuation byte. Malformed UTF-8 falls back to
one-byte progress, preserving forward progress without skipping
unrelated bytes.
### Release note
Fix incorrect repeated matches for anchored regular expressions in
`regexp_count`, `regexp_extract_all`, `regexp_extract_all_array`, and
`split_by_regexp`.
### Check List (For Author)
- Test:
- Unit Test: `./run-be-ut.sh --run
--filter='FunctionLikeTest.regexp_extract_all:FunctionLikeTest.regexp_extract_all_array:function_string_test.function_regexp_count_mixed_const_test'`
(ASAN_UT, 3 passed; valid and malformed UTF-8 cases included)
- Regression test: generated the expected output with `-forceGenOut`,
then reran `test_regexp_repeated_search_anchor` without `-forceGenOut`
(both runs: 1 suite, 0 failures)
- Build: RELEASE BE and FE build passed as part of the local regression
workflow
- Style: `build-support/clang-format.sh`,
`build-support/check-format.sh`, `build-support/check-build-hygiene.sh`,
and `git diff --check`
- Static analysis: `build-support/run-clang-tidy.sh --base HEAD
--build-dir be/build_Release` was attempted, but the local toolchain
stopped on existing diagnostics including missing `stddef.h` resolution
and an unmatched `NOLINTEND` in `be/src/core/types.h`
- Behavior changed: Yes. Beginning anchors are evaluated relative to the
original input, and empty matches advance by one UTF-8 character during
repeated searches.
- Does this need documentation: No
---
be/src/exprs/function/function_regexp.cpp | 48 ++++++++++++++--------
be/src/exprs/function/function_split_by_regexp.cpp | 18 ++++----
be/test/exprs/function/function_like_test.cpp | 48 ++++++++++++++++++++--
be/test/exprs/function/function_string_test.cpp | 4 ++
.../test_regexp_repeated_search_anchor.out | 10 +++++
.../test_regexp_repeated_search_anchor.groovy | 38 +++++++++++++++++
6 files changed, 139 insertions(+), 27 deletions(-)
diff --git a/be/src/exprs/function/function_regexp.cpp
b/be/src/exprs/function/function_regexp.cpp
index 206f51ce0e7..f7f1a952953 100644
--- a/be/src/exprs/function/function_regexp.cpp
+++ b/be/src/exprs/function/function_regexp.cpp
@@ -19,6 +19,7 @@
#include <re2/re2.h>
#include <re2/stringpiece.h>
#include <stddef.h>
+#include <unicode/utf8.h>
#include <boost/regex.hpp>
#include <memory>
@@ -55,6 +56,24 @@
namespace doris {
+static bool advance_re2_search_position(const char* data, size_t size,
+ const re2::StringPiece& match, size_t&
pos) {
+ const size_t match_pos = match.data() - data;
+ if (match.empty()) {
+ if (match_pos == size) {
+ return false;
+ }
+ size_t next_pos = match_pos;
+ UChar32 character;
+ U8_NEXT(data, next_pos, size, character);
+ // Doris strings can contain malformed UTF-8, so make byte-wise
progress on decode failure.
+ pos = character < 0 ? match_pos + 1 : next_pos;
+ } else {
+ pos = match_pos + match.size();
+ }
+ return true;
+}
+
// Helper structure to hold either RE2 or Boost.Regex
struct RegexpExtractEngine {
std::unique_ptr<re2::RE2> re2_regex;
@@ -145,28 +164,26 @@ struct RegexpExtractEngine {
return; // No capturing groups
}
+ const re2::StringPiece input(data, size);
size_t pos = 0;
while (pos < size) {
- const char* str_pos = data + pos;
- size_t str_size = size - pos;
std::vector<re2::StringPiece> matches(max_matches);
- bool success = re2_regex->Match(re2::StringPiece(str_pos,
str_size), 0, str_size,
- re2::RE2::UNANCHORED,
matches.data(), max_matches);
+ bool success = re2_regex->Match(input, pos, size,
re2::RE2::UNANCHORED,
+ matches.data(), max_matches);
if (!success) {
break;
}
+ const bool can_continue = advance_re2_search_position(data,
size, matches[0], pos);
if (matches[0].empty()) {
- pos += 1;
+ if (!can_continue) {
+ break;
+ }
continue;
}
// Extract first capturing group
if (matches.size() > 1 && !matches[1].empty()) {
results.emplace_back(matches[1].data(), matches[1].size());
}
- // Move position forward
- auto offset = std::string(str_pos, str_size)
- .find(std::string(matches[0].data(),
matches[0].size()));
- pos += offset + matches[0].size();
}
} else if (is_boost()) {
const char* search_start = data;
@@ -224,23 +241,22 @@ struct RegexpCountImpl {
const auto str = str_col.value_at(index_now);
int count = 0;
size_t pos = 0;
+ const re2::StringPiece input(str.data, str.size);
while (pos < str.size) {
- auto str_pos = str.data + pos;
- auto str_size = str.size - pos;
- re2::StringPiece str_sp_current = re2::StringPiece(str_pos,
str_size);
re2::StringPiece match;
- bool success = re->Match(str_sp_current, 0, str_size,
re2::RE2::UNANCHORED, &match, 1);
+ bool success = re->Match(input, pos, str.size,
re2::RE2::UNANCHORED, &match, 1);
if (!success) {
break;
}
+ const bool can_continue = advance_re2_search_position(str.data,
str.size, match, pos);
if (match.empty()) {
- pos += 1;
+ if (!can_continue) {
+ break;
+ }
continue;
}
count++;
- size_t match_start = match.data() - str_sp_current.data();
- pos += match_start + match.size();
}
return count;
diff --git a/be/src/exprs/function/function_split_by_regexp.cpp
b/be/src/exprs/function/function_split_by_regexp.cpp
index 292bcb64559..5e469a640bf 100644
--- a/be/src/exprs/function/function_split_by_regexp.cpp
+++ b/be/src/exprs/function/function_split_by_regexp.cpp
@@ -43,6 +43,7 @@ public:
bool get(const char*& token_begin, const char*& token_end);
private:
+ const char* _begin;
const char* _pos;
const char* _end;
@@ -52,12 +53,12 @@ private:
re2::RE2* _re2 = nullptr;
unsigned _number_of_subpatterns = 0;
- unsigned match(const char* subject, size_t subject_size,
std::vector<Match>& matches,
- unsigned limit) const;
+ unsigned match(const char* subject, size_t subject_size, size_t start_pos,
+ std::vector<Match>& matches, unsigned limit) const;
};
-unsigned RegexpSplit::match(const char* subject, size_t subject_size,
std::vector<Match>& matches,
- unsigned limit) const {
+unsigned RegexpSplit::match(const char* subject, size_t subject_size, size_t
start_pos,
+ std::vector<Match>& matches, unsigned limit) const
{
matches.clear();
if (limit == 0) {
@@ -67,8 +68,8 @@ unsigned RegexpSplit::match(const char* subject, size_t
subject_size, std::vecto
limit = std::min(limit, _number_of_subpatterns + 1);
std::vector<re2::StringPiece> pieces(limit);
- if (!_re2->Match({subject, subject_size}, 0, subject_size,
re2::RE2::UNANCHORED, pieces.data(),
- limit)) {
+ if (!_re2->Match({subject, subject_size}, start_pos, subject_size,
re2::RE2::UNANCHORED,
+ pieces.data(), limit)) {
return 0;
} else {
matches.resize(limit);
@@ -95,6 +96,7 @@ void RegexpSplit::init(re2::RE2* re2, int32_t max_splits) {
// Called for each next string.
void RegexpSplit::set(const char* pos, const char* end) {
+ _begin = pos;
_pos = pos;
_end = end;
_splits = 0;
@@ -134,12 +136,12 @@ bool RegexpSplit::get(const char*& token_begin, const
char*& token_end) {
}
}
- if (!match(_pos, _end - _pos, _matches, _number_of_subpatterns + 1) ||
+ if (!match(_begin, _end - _begin, _pos - _begin, _matches,
_number_of_subpatterns + 1) ||
!_matches[0].length) {
token_end = _end;
_pos = _end + 1;
} else {
- token_end = _pos + _matches[0].offset;
+ token_end = _begin + _matches[0].offset;
_pos = token_end + _matches[0].length;
++_splits;
}
diff --git a/be/test/exprs/function/function_like_test.cpp
b/be/test/exprs/function/function_like_test.cpp
index 1543d519b99..645b517e716 100644
--- a/be/test/exprs/function/function_like_test.cpp
+++ b/be/test/exprs/function/function_like_test.cpp
@@ -555,12 +555,18 @@ TEST(FunctionLikeTest, regexp_extract_or_null) {
TEST(FunctionLikeTest, regexp_extract_all) {
std::string func_name = "regexp_extract_all";
+ const std::string word_boundary_input(10000, 'a');
+ const std::string invalid_utf8_input {static_cast<char>(0xC3), 'A'};
DataSet data_set = {
{{std::string("x=a3&x=18abc&x=2&y=3&x=4&x=17bcd"),
std::string("x=([0-9]+)([a-z]+)")},
std::string("['18','17']")},
{{std::string("x=a3&x=18abc&x=2&y=3&x=4"),
std::string("^x=([a-z]+)([0-9]+)")},
std::string("['a']")},
+ {{std::string("aaa"), std::string("(^a)")}, std::string("['a']")},
+ {{word_boundary_input, std::string("(\\b)")}, std::string("")},
+ {{std::string("é"), std::string("(?:^)|(\\C)")}, std::string("")},
+ {{invalid_utf8_input, std::string("(?:^)|(\\C)")},
std::string("['A']")},
{{std::string("http://a.m.baidu.com/i41915173660.htm"),
std::string("i([0-9]+)")},
std::string("['41915173660']")},
{{std::string("http://a.m.baidu.com/i41915i73660.htm"),
std::string("i([0-9]+)")},
@@ -585,7 +591,8 @@ TEST(FunctionLikeTest, regexp_extract_all) {
}
}
-TEST(FunctionLikeTest, regexp_extract_all_array) {
+// Keep the cases together so they share the same function lifecycle setup.
+TEST(FunctionLikeTest, regexp_extract_all_array) { //
NOLINT(readability-function-size)
std::string func_name = "regexp_extract_all_array";
auto str_type = std::make_shared<DataTypeString>();
auto return_type = make_nullable(
@@ -633,10 +640,14 @@ TEST(FunctionLikeTest, regexp_extract_all_array) {
static_cast<void>(func->close(fn_ctx,
FunctionContext::FRAGMENT_LOCAL));
};
- run_case("x=a3&x=18abc&x=2&y=3&x=4&x=17bcd", "x=([0-9]+)([a-z]+)",
"[\"18\", \"17\"]");
+ run_case("x=a3&x=18abc&x=2&y=3&x=4&x=17bcd", "x=([0-9]+)([a-z]+)",
R"(["18", "17"])");
run_case("x=a3&x=18abc&x=2&y=3&x=4", "^x=([a-z]+)([0-9]+)", "[\"a\"]");
+ run_case("aaa", "(^a)", "[\"a\"]");
+ run_case(std::string(10000, 'a'), "(\\b)", "[]");
+ run_case("é", "(?:^)|(\\C)", "[]");
+ run_case(std::string {static_cast<char>(0xC3), 'A'}, "(?:^)|(\\C)",
"[\"A\"]");
run_case("http://a.m.baidu.com/i41915173660.htm", "i([0-9]+)",
"[\"41915173660\"]");
- run_case("http://a.m.baidu.com/i41915i73660.htm", "i([0-9]+)",
"[\"41915\", \"73660\"]");
+ run_case("http://a.m.baidu.com/i41915i73660.htm", "i([0-9]+)",
R"(["41915", "73660"])");
run_case("hitdecisiondlist", "(i)(.*?)(e)", "[\"i\"]");
run_case("no_match_here", "x=([0-9]+)", "[]");
run_case("abc", "([a-z]+)", "[\"abc\"]");
@@ -735,6 +746,37 @@ TEST(FunctionLikeTest, regexp_extract_all_array) {
}
}
+TEST(FunctionLikeTest, split_by_regexp_preserves_original_anchor) {
+ const std::string input = "aaa";
+ const std::string pattern = "^a";
+ auto string_type = std::make_shared<DataTypeString>();
+ auto return_type =
+
std::make_shared<DataTypeArray>(make_nullable(std::make_shared<DataTypeString>()));
+
+ auto input_column = ColumnString::create();
+ input_column->insert_data(input.data(), input.size());
+ auto pattern_column = ColumnString::create();
+ pattern_column->insert_data(pattern.data(), pattern.size());
+
+ Block block;
+ block.insert({std::move(input_column), string_type, "str"});
+ block.insert({ColumnConst::create(std::move(pattern_column), 1),
string_type, "pattern"});
+ block.insert({nullptr, return_type, "result"});
+
+ ColumnsWithTypeAndName arg_cols = {block.get_by_position(0),
block.get_by_position(1)};
+ auto function =
SimpleFunctionFactory::instance().get_function("split_by_regexp", arg_cols,
+
return_type);
+ ASSERT_TRUE(function != nullptr);
+
+ FunctionUtils fn_utils({}, {string_type, string_type}, false);
+ auto* context = fn_utils.get_fn_ctx();
+ ASSERT_EQ(Status::OK(), function->execute(context, block, {0, 1}, 2, 1));
+
+ const auto& result_column = block.get_by_position(2).column;
+ ASSERT_TRUE(result_column.get() != nullptr);
+ EXPECT_EQ(R"(["", "aa"])", return_type->to_string(*result_column, 0));
+}
+
TEST(FunctionLikeTest, regexp_replace) {
std::string func_name = "regexp_replace";
diff --git a/be/test/exprs/function/function_string_test.cpp
b/be/test/exprs/function/function_string_test.cpp
index 3a8debab7ad..f49d15b0167 100644
--- a/be/test/exprs/function/function_string_test.cpp
+++ b/be/test/exprs/function/function_string_test.cpp
@@ -4182,6 +4182,10 @@ TEST(function_string_test,
function_regexp_count_mixed_const_test) {
{{std::string("a1b2346c3d"), std::string("\\d+")},
std::int32_t(3)},
{{std::string("abcd"), std::string("")}, std::int32_t(0)},
{{std::string("book keeper"), std::string("oo|ee")},
std::int32_t(2)},
+ {{std::string("aaa"), std::string("^a")}, std::int32_t(1)},
+ {{std::string(10000, 'a'), std::string("\\b")}, std::int32_t(0)},
+ {{std::string("é"), std::string("^|\\C")}, std::int32_t(0)},
+ {{std::string {static_cast<char>(0xC3), 'A'},
std::string("^|\\C")}, std::int32_t(1)},
{{Null(), std::string("\\d+")}, Null()},
{{std::string("abcd"), Null()}, Null()},
};
diff --git
a/regression-test/data/query_p0/sql_functions/string_functions/test_regexp_repeated_search_anchor.out
b/regression-test/data/query_p0/sql_functions/string_functions/test_regexp_repeated_search_anchor.out
new file mode 100644
index 00000000000..67932a4de30
--- /dev/null
+++
b/regression-test/data/query_p0/sql_functions/string_functions/test_regexp_repeated_search_anchor.out
@@ -0,0 +1,10 @@
+-- This file is automatically generated. You should know what you did if you
want to edit this
+-- !anchor_functions --
+1 ['a'] ["a"] ["", "aa"] 2
+
+-- !word_boundary_functions --
+0 []
+
+-- !multibyte_empty_match_functions --
+0 []
+
diff --git
a/regression-test/suites/query_p0/sql_functions/string_functions/test_regexp_repeated_search_anchor.groovy
b/regression-test/suites/query_p0/sql_functions/string_functions/test_regexp_repeated_search_anchor.groovy
new file mode 100644
index 00000000000..ef16b5c8464
--- /dev/null
+++
b/regression-test/suites/query_p0/sql_functions/string_functions/test_regexp_repeated_search_anchor.groovy
@@ -0,0 +1,38 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements. See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership. The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License. You may obtain a copy of the License at
+//
+// http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied. See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+suite("test_regexp_repeated_search_anchor") {
+ qt_anchor_functions """
+ SELECT regexp_count('aaa', '^a'),
+ regexp_extract_all('aaa', '(^a)'),
+ regexp_extract_all_array('aaa', '(^a)'),
+ split_by_regexp('aaa', '^a'),
+ size(split_by_regexp('aaa', '^a'))
+ """
+
+ qt_word_boundary_functions """
+ SELECT regexp_count(repeat('a', 10000), '\\\\b'),
+ regexp_extract_all(repeat('a', 10000), '(\\\\b)'),
+ regexp_extract_all_array(repeat('a', 10000), '(\\\\b)')
+ """
+
+ qt_multibyte_empty_match_functions """
+ SELECT regexp_count('é', '^|\\\\C'),
+ regexp_extract_all('é', '(?:^)|(\\\\C)'),
+ regexp_extract_all_array('é', '(?:^)|(\\\\C)')
+ """
+}
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]