diff --git a/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java b/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java index e2adbf11..67ea9ccb 100644 --- a/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java +++ b/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java @@ -676,7 +676,10 @@ public static ReggieMatcher compilePikeVm(String pattern, String encodedNames) { } Map nameMap = decodeNameMap(encodedNames); int groupCount = countGroups(pattern); - NFA nfa = new ThompsonBuilder().build(ast, groupCount); + // lazyAware: lazy quantifiers must compile with lazy (shortest-first) priority, or the + // extracted capture spans take the greedy split and diverge from the JDK (group spans of + // both match() and findMatch*()). Every other PikeVM compile site builds lazy-aware. + NFA nfa = new ThompsonBuilder(true).build(ast, groupCount); PIKEVM_NFA_CACHE.putIfAbsent(pattern, new PikeVMEntry(nfa, nameMap)); return PIKEVM_NFA_CACHE.get(pattern).newMatcher(pattern); } catch (RegexParser.ParseException e) { diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMLazyGroupParityTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMLazyGroupParityTest.java new file mode 100644 index 00000000..730c543a --- /dev/null +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMLazyGroupParityTest.java @@ -0,0 +1,150 @@ +/* + * Copyright 2026-Present Datadog, Inc. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package com.datadoghq.reggie.runtime; + +import static org.junit.jupiter.api.Assertions.assertEquals; + +import java.util.ArrayList; +import java.util.List; +import java.util.regex.Matcher; +import java.util.regex.Pattern; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.MethodSource; + +/** + * Differential parity for lazy-quantifier capture-group spans on the PikeVM capture route ({@link + * RuntimeCompiler#compilePikeVm}, the fallback the annotation processor emits for patterns its + * native strategies refuse). + * + *

The bug: compilePikeVm built its NFA with a non-lazy-aware {@code ThompsonBuilder}, so lazy + * quantifiers ({@code +?}, {@code *?}, {@code ??}) were compiled with greedy priority and the + * extracted group spans took the longest split instead of the lazy (shortest) one — e.g. for {@code + * ^(.+?)\.([^/]+)} on {@code net/http.(*ServeMux).ServeHTTP} the JDK yields group1 = {@code + * net/http} but the PikeVM yielded {@code net/http.(*ServeMux)}. Boolean matches()/find() are + * priority-independent and were never affected. + */ +class PikeVMLazyGroupParityTest { + static List patterns() { + return List.of( + "^(.+?)\\.([^/]+)", + "^(a+?)b", + "(.+?)-(.+?)-", + "^(\\w+?)/(\\w+?)$", + "a(.*?)b", + "^(.+?)([^/]+)"); + } + + static List inputs() { + return List.of( + "net/http.(*ServeMux).ServeHTTP", "aaab", "a-b-c-d", "foo/bar", "xaaxbbx", "pkg/file.go"); + } + + private static String groups(MatchResult result) { + if (result == null) { + return null; + } + StringBuilder sb = new StringBuilder("["); + for (int i = 0; i <= result.groupCount(); i++) { + if (i > 0) { + sb.append(','); + } + sb.append(result.group(i)); + } + return sb.append(']').toString(); + } + + private static String jdkGroups(Matcher matcher) { + StringBuilder sb = new StringBuilder("["); + for (int i = 0; i <= matcher.groupCount(); i++) { + if (i > 0) { + sb.append(','); + } + sb.append(matcher.group(i)); + } + return sb.append(']').toString(); + } + + @ParameterizedTest + @MethodSource("patterns") + void matchGroupParityOnPikeVmRoute(String pattern) { + Pattern jdk = Pattern.compile(pattern); + for (String input : inputs()) { + Matcher jm = jdk.matcher(input); + ReggieMatcher reggie = RuntimeCompiler.compilePikeVm(pattern, ""); + MatchResult result = reggie.match(input); + if (!jm.matches()) { + assertEquals(null, result, "unexpected match for /" + pattern + "/ on \"" + input + "\""); + continue; + } + assertEquals( + jdkGroups(jm), + groups(result), + "match() group divergence for /" + pattern + "/ on \"" + input + "\""); + } + } + + @ParameterizedTest + @MethodSource("patterns") + void findGroupParityOnPikeVmRoute(String pattern) { + Pattern jdk = Pattern.compile(pattern); + for (String input : inputs()) { + Matcher jm = jdk.matcher(input); + List expected = new ArrayList<>(); + while (jm.find()) { + expected.add(spans(jm)); + } + ReggieMatcher reggie = RuntimeCompiler.compilePikeVm(pattern, ""); + int from = 0; + List actual = new ArrayList<>(); + while (true) { + MatchResult result = reggie.findMatchFrom(input, from); + if (result == null) { + break; + } + actual.add(spans(result)); + from = result.end() == result.start() ? result.end() + 1 : result.end(); + if (from > input.length()) { + break; + } + } + assertEquals( + expected, actual, "find() group divergence for /" + pattern + "/ on \"" + input + "\""); + } + } + + /** Every group's half-open span {@code [start,end)} — spans, not just group strings. */ + private static String spans(MatchResult result) { + StringBuilder sb = new StringBuilder(); + for (int i = 0; i <= result.groupCount(); i++) { + if (i > 0) { + sb.append(','); + } + sb.append('[').append(result.start(i)).append(',').append(result.end(i)).append(')'); + } + return sb.toString(); + } + + private static String spans(Matcher matcher) { + StringBuilder sb = new StringBuilder(); + for (int i = 0; i <= matcher.groupCount(); i++) { + if (i > 0) { + sb.append(','); + } + sb.append('[').append(matcher.start(i)).append(',').append(matcher.end(i)).append(')'); + } + return sb.toString(); + } +}