From 637303c4b5075b779700edd9bd2acfd4553c4a4b Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Sat, 27 Jun 2026 17:22:18 +0200 Subject: [PATCH 01/27] feat: merge fix/optimized-nfa-backref-perconfig onto rebased main Squash-merge of 78 commits from fix/optimized-nfa-backref-perconfig: - Per-config captures for backref NFAs (Tasks 6-7) - 64-bit NFA contentHashCode + verify-on-hit for structural cache - Cache key NUL separator in RuntimeCompiler.cacheKeyFor() - PikeVM per-call compilation (no cached volatile field) - B16 guard: needsFallback check before PIKEVM_CAPTURE early return - epsilonGroup flag on DFA.GroupAction for zero-width group tracking - CRLF/end-anchor fixes (\Z/$ NEL/LS/PS) across all bytecode generators - BackrefBacktrackMatcher and per-config worklist for backref NFAs - Fix requiresBacktrackingForGroups false positive on anchor-only suffix Co-Authored-By: Claude Sonnet 4.6 --- .../main/java/com/datadoghq/reggie/runtime/PikeVMMatcher.java | 1 + 1 file changed, 1 insertion(+) diff --git a/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/PikeVMMatcher.java b/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/PikeVMMatcher.java index efc56ff4..034ede21 100644 --- a/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/PikeVMMatcher.java +++ b/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/PikeVMMatcher.java @@ -128,6 +128,7 @@ public final class PikeVMMatcher extends ReggieMatcher { private final LazyDFACache matchesDfa; private final NfaStep matchesStep; + /** Construct a PikeVMMatcher over the given NFA and pattern string. */ public PikeVMMatcher(NFA nfa, String pattern) { super(pattern); From 172e164e7d21d2ed574b3d93c741a1da3cd1ed35 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Sat, 27 Jun 2026 18:03:17 +0200 Subject: [PATCH 02/27] style: spotlessApply --- .../main/java/com/datadoghq/reggie/runtime/PikeVMMatcher.java | 1 - 1 file changed, 1 deletion(-) diff --git a/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/PikeVMMatcher.java b/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/PikeVMMatcher.java index 034ede21..efc56ff4 100644 --- a/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/PikeVMMatcher.java +++ b/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/PikeVMMatcher.java @@ -128,7 +128,6 @@ public final class PikeVMMatcher extends ReggieMatcher { private final LazyDFACache matchesDfa; private final NfaStep matchesStep; - /** Construct a PikeVMMatcher over the given NFA and pattern string. */ public PikeVMMatcher(NFA nfa, String pattern) { super(pattern); From 0734fe5476a74c54c51c35a8ec2059622f33c6bb Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Sun, 28 Jun 2026 09:42:19 +0200 Subject: [PATCH 03/27] test: add failing routing tests for A1+A2 group-span divergences Co-Authored-By: Claude Sonnet 4.6 --- .../reggie/runtime/PikeVMRoutingTest.java | 73 +++++++++++++++++++ 1 file changed, 73 insertions(+) diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java index fa1a9d46..22bcd04f 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java @@ -132,4 +132,77 @@ void ncgWrappedInteractingAlts_captureCorrect() { assertEquals("a", r.group(1), "group(1) must be 'a'"); assertEquals("bcd", r.group(2), "group(2) must be 'bcd'"); } + + // ── A2: capturing group absent from some alternation branch ──────────────── + + @Test + void groupAbsentFromAlt_literalThenGroup_routesToPikevm() throws Exception { + // Group 1 `(.)` only in alt 2; DFA_UNROLLED_WITH_GROUPS binds group 1 when alt 1 wins. + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("[a][1-b]|(.)"), + "[a][1-b]|(.) must route to PIKEVM_CAPTURE (A2: group absent from alt 1)"); + } + + @Test + void groupAbsentFromAlt_dotThenGroup_routesToPikevm() throws Exception { + // Group 1 `(_)` only in alt 2. + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("_.|(_)"), + "_.|(_) must route to PIKEVM_CAPTURE (A2: group absent from alt 2)"); + } + + @Test + void groupAbsentFromAlt_groupFirstAlt_routesToPikevm() throws Exception { + // Group 1 `(1)` only in alt 1; DFA binds group 1 when alt 2 wins. + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("(1)c|10"), + "(1)c|10 must route to PIKEVM_CAPTURE (A2: group absent from alt 2)"); + } + + // ── A2 regression: patterns that MUST stay on DFA_UNROLLED_WITH_GROUPS ──── + + @Test + void singleGroupWrappingAlts_staysOnDfa() throws Exception { + // (fo|foo): group wraps the whole alternation; both inner branches have no capturing group. + // A2 guard must NOT fire — guard already tested by foOrFoo_routesToDfaWithGroups(). + // This test guards the simple (ab)c case as a minimal sanity check. + assertEquals( + PatternAnalyzer.MatchingStrategy.DFA_UNROLLED_WITH_GROUPS, + StrategyCorrectnessMetaTest.routeOf("(ab)c"), + "(ab)c must stay on DFA_UNROLLED_WITH_GROUPS (no alternation, non-nullable body)"); + } + + // ── A1: capturing group body starts with nullable first element ──────────── + + @Test + void nullableFirstElem_optionalPrefix_routesToPikevm() throws Exception { + // Group body `a?.*` starts with `a?` (min=0); TDFA fires group-start too early. + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("-{1}(a?.*)"), + "-{1}(a?.*) must route to PIKEVM_CAPTURE (A1: nullable first element a?)"); + } + + @Test + void nullableFirstElem_zeroQuantifier_routesToPikevm() throws Exception { + // Group body `0{0}[^_]{1,}` starts with `0{0}` (min=0,max=0); TDFA group-start fires too early. + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("(0{0}[^_]{1,})-"), + "(0{0}[^_]{1,})- must route to PIKEVM_CAPTURE (A1: nullable first element 0{0})"); + } + + // ── A1 regression: patterns that MUST stay on DFA_UNROLLED_WITH_GROUPS ──── + + @Test + void nonNullableGroupBody_staysOnDfa() throws Exception { + // Group body `abc` starts with literal `a` (non-nullable); A1 guard must NOT fire. + assertEquals( + PatternAnalyzer.MatchingStrategy.DFA_UNROLLED_WITH_GROUPS, + StrategyCorrectnessMetaTest.routeOf("(abc)"), + "(abc) must stay on DFA_UNROLLED_WITH_GROUPS (non-nullable first element)"); + } } From 2e76df67a4bc22e7e088305a5c9d3dd36b9fc1cc Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Sun, 28 Jun 2026 09:51:45 +0200 Subject: [PATCH 04/27] feat: route A2 (group absent from alternation branch) to PIKEVM_CAPTURE Co-Authored-By: Claude Sonnet 4.6 --- .../analysis/FallbackPatternDetector.java | 95 +++++++++++++++++++ .../codegen/analysis/PatternAnalyzer.java | 13 +++ .../reggie/runtime/PikeVMRoutingTest.java | 6 +- 3 files changed, 112 insertions(+), 2 deletions(-) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java index 2d303df4..29812ecc 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java @@ -1408,6 +1408,101 @@ private static boolean containsNullableCapturingGroup(RegexNode node) { return false; } + /** + * Returns true if any {@link AlternationNode} in {@code ast} contains an alternative that is + * missing a capturing group number present in another alternative of the same alternation, AND at + * least one of the alternatives in that alternation consumes more than one character (minimum + * match length > 1). Single-character-per-alt patterns (e.g. {@code (a)|b}, {@code (b)|b}) are + * correctly handled by the TDFA 1B alternation-binding fix and do not require PIKEVM routing. + * + *

Examples that fire: {@code [a][1-b]|(.)}, {@code _.|(_)}, {@code (1)c|10} — at least one alt + * is a multi-char sequence. Examples that do NOT fire: {@code (a)|b}, {@code (b)|b}, {@code + * b|(b)}, {@code .|([^c])} — all single-char alts. {@code (a|b)} does NOT fire (alternation + * inside the group; inner branches add no separate group). Patterns with no alternation (e.g. + * {@code (ab)c}) do not fire. + */ + public static boolean hasCapturingGroupAbsentFromSomeAlternative(RegexNode ast) { + if (ast instanceof AlternationNode) { + AlternationNode alt = (AlternationNode) ast; + List alts = alt.alternatives; + @SuppressWarnings("unchecked") + Set[] groups = new Set[alts.size()]; + Set union = new HashSet<>(); + for (int i = 0; i < alts.size(); i++) { + groups[i] = new HashSet<>(); + collectGroupsInSubtree(alts.get(i), groups[i]); + union.addAll(groups[i]); + } + if (!union.isEmpty()) { + boolean anyAbsent = false; + for (Set altGroups : groups) { + if (!altGroups.containsAll(union)) { + anyAbsent = true; + break; + } + } + if (anyAbsent) { + // Only flag when at least one alternative consumes more than one character. Single-char + // alternations are handled correctly by the TDFA 1B fix. + for (RegexNode alternative : alts) { + if (altMinLength(alternative) > 1) return true; + } + } + } + for (RegexNode alternative : alts) { + if (hasCapturingGroupAbsentFromSomeAlternative(alternative)) return true; + } + return false; + } + if (ast instanceof ConcatNode) { + for (RegexNode child : ((ConcatNode) ast).children) { + if (hasCapturingGroupAbsentFromSomeAlternative(child)) return true; + } + return false; + } + if (ast instanceof GroupNode) { + return hasCapturingGroupAbsentFromSomeAlternative(((GroupNode) ast).child); + } + if (ast instanceof QuantifierNode) { + return hasCapturingGroupAbsentFromSomeAlternative(((QuantifierNode) ast).child); + } + return false; + } + + /** + * Returns the minimum number of characters that {@code node} must consume (ignoring zero-width + * anchors and assertions). Used to distinguish single-character alternation alternatives from + * multi-character ones. + */ + private static int altMinLength(RegexNode node) { + if (node instanceof LiteralNode) { + return ((LiteralNode) node).ch == 0 ? 0 : 1; // epsilon is 0; normal literal is 1 + } + if (node instanceof CharClassNode) return 1; + if (node instanceof AnchorNode || node instanceof AssertionNode) return 0; // zero-width + if (node instanceof GroupNode) return altMinLength(((GroupNode) node).child); + if (node instanceof QuantifierNode) { + QuantifierNode q = (QuantifierNode) node; + if (q.min == 0) return 0; + return q.min * altMinLength(q.child); + } + if (node instanceof ConcatNode) { + int total = 0; + for (RegexNode c : ((ConcatNode) node).children) { + total += altMinLength(c); + } + return total; + } + if (node instanceof AlternationNode) { + int min = Integer.MAX_VALUE; + for (RegexNode a : ((AlternationNode) node).alternatives) { + min = Math.min(min, altMinLength(a)); + } + return min == Integer.MAX_VALUE ? 0 : min; + } + return 0; + } + /** * Returns true if any capturing GroupNode is directly wrapped by a QuantifierNode with min=0 AND * the group's content is itself nullable (can match the empty string). Example: {@code diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java index b8c8a6e2..e965b306 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java @@ -977,6 +977,19 @@ && containsAnyQuantifier(ast) null, needsPosixSemantics); } + // A2: capturing group absent from some alternation branch — TDFA binds the absent + // group to a wrong span when the branch that lacks the group wins. + if (FallbackPatternDetector.hasCapturingGroupAbsentFromSomeAlternative(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + return new MatchingStrategyResult( + MatchingStrategy.PIKEVM_CAPTURE, + null, + null, + false, + requiredLiterals, + null, + needsPosixSemantics); + } // Pure-regular, anchor-free: C2 priority-ordered TDFA gives correct spans. int stateCount = dfa.getStateCount(); if (stateCount < DFA_UNROLLED_STATE_LIMIT) { diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java index 22bcd04f..8acfade4 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java @@ -169,10 +169,12 @@ void singleGroupWrappingAlts_staysOnDfa() throws Exception { // (fo|foo): group wraps the whole alternation; both inner branches have no capturing group. // A2 guard must NOT fire — guard already tested by foOrFoo_routesToDfaWithGroups(). // This test guards the simple (ab)c case as a minimal sanity check. + // (ab)c has no alternation so it routes to SPECIALIZED_FIXED_SEQUENCE, not + // DFA_UNROLLED_WITH_GROUPS. assertEquals( - PatternAnalyzer.MatchingStrategy.DFA_UNROLLED_WITH_GROUPS, + PatternAnalyzer.MatchingStrategy.SPECIALIZED_FIXED_SEQUENCE, StrategyCorrectnessMetaTest.routeOf("(ab)c"), - "(ab)c must stay on DFA_UNROLLED_WITH_GROUPS (no alternation, non-nullable body)"); + "(ab)c must not be routed to PIKEVM_CAPTURE by the A2 guard (no alternation)"); } // ── A1: capturing group body starts with nullable first element ──────────── From 4701bc7968dfce1b8baa65d2dc86957640148c2a Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Sun, 28 Jun 2026 10:09:26 +0200 Subject: [PATCH 05/27] feat: route A1 (nullable first element in group body) to PIKEVM_CAPTURE Co-Authored-By: Claude Sonnet 4.6 --- .../analysis/FallbackPatternDetector.java | 43 +++++++++++++++++++ .../codegen/analysis/PatternAnalyzer.java | 39 +++++++++++++++++ ...DfaUnrolledGroupAndFindRegressionTest.java | 22 ++++++---- .../reggie/runtime/PikeVMRoutingTest.java | 6 +-- 4 files changed, 99 insertions(+), 11 deletions(-) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java index 29812ecc..d025063e 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java @@ -1469,6 +1469,49 @@ public static boolean hasCapturingGroupAbsentFromSomeAlternative(RegexNode ast) return false; } + /** + * Returns true if any capturing {@link GroupNode} in {@code ast} has a body whose leading element + * is nullable (i.e. can match the empty string). This identifies patterns where the TDFA fires + * the group-start tag at a state that is epsilon-reachable from before the group, causing the + * recorded start to equal the overall match start rather than the group's actual start. + * + *

Examples: {@code -{1}(a?.*)} — group body starts with {@code a?} (min=0, nullable). {@code + * (0{0}[^_]{1,})-} — group body starts with {@code 0{0}} (max=0, nullable). + */ + public static boolean hasCapturingGroupWithNullableFirstElement(RegexNode ast) { + if (ast instanceof GroupNode) { + GroupNode g = (GroupNode) ast; + if (g.capturing) { + RegexNode body = g.child; + boolean nullableFirst; + if (body instanceof ConcatNode) { + List children = ((ConcatNode) body).children; + nullableFirst = !children.isEmpty() && isNullable(children.get(0)); + } else { + nullableFirst = isNullable(body); + } + if (nullableFirst) return true; + } + return hasCapturingGroupWithNullableFirstElement(g.child); + } + if (ast instanceof AlternationNode) { + for (RegexNode alt : ((AlternationNode) ast).alternatives) { + if (hasCapturingGroupWithNullableFirstElement(alt)) return true; + } + return false; + } + if (ast instanceof ConcatNode) { + for (RegexNode child : ((ConcatNode) ast).children) { + if (hasCapturingGroupWithNullableFirstElement(child)) return true; + } + return false; + } + if (ast instanceof QuantifierNode) { + return hasCapturingGroupWithNullableFirstElement(((QuantifierNode) ast).child); + } + return false; + } + /** * Returns the minimum number of characters that {@code node} must consume (ignoring zero-width * anchors and assertions). Used to distinguish single-character alternation alternatives from diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java index e965b306..cc289a85 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java @@ -977,6 +977,19 @@ && containsAnyQuantifier(ast) null, needsPosixSemantics); } + // A1: group body starts with a nullable first element — TDFA fires the group-start + // tag at an epsilon-reachable state, recording match-start instead of group-start. + if (FallbackPatternDetector.hasCapturingGroupWithNullableFirstElement(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + return new MatchingStrategyResult( + MatchingStrategy.PIKEVM_CAPTURE, + null, + null, + false, + requiredLiterals, + null, + needsPosixSemantics); + } // A2: capturing group absent from some alternation branch — TDFA binds the absent // group to a wrong span when the branch that lacks the group wins. if (FallbackPatternDetector.hasCapturingGroupAbsentFromSomeAlternative(ast) @@ -1059,6 +1072,32 @@ && containsAnyQuantifier(ast) null, needsPosixSemantics); } + // A1: group body starts with a nullable first element — TDFA fires the group-start + // tag at an epsilon-reachable state, recording match-start instead of group-start. + if (FallbackPatternDetector.hasCapturingGroupWithNullableFirstElement(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + return new MatchingStrategyResult( + MatchingStrategy.PIKEVM_CAPTURE, + null, + null, + false, + requiredLiterals, + null, + needsPosixSemantics); + } + // A2: capturing group absent from some alternation branch — TDFA binds the absent + // group to a wrong span when the branch that lacks the group wins. + if (FallbackPatternDetector.hasCapturingGroupAbsentFromSomeAlternative(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + return new MatchingStrategyResult( + MatchingStrategy.PIKEVM_CAPTURE, + null, + null, + false, + requiredLiterals, + null, + needsPosixSemantics); + } // Class E: two interacting variable-length capturing alternations (e.g. (a|ab)(c|bcd)). The // first alternation's branches share a prefix, so its capture span is ambiguous until the // second alternation resolves it — which the single-register TDFA cannot track diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/DfaUnrolledGroupAndFindRegressionTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/DfaUnrolledGroupAndFindRegressionTest.java index 7bbdc3a4..21696133 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/DfaUnrolledGroupAndFindRegressionTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/DfaUnrolledGroupAndFindRegressionTest.java @@ -79,32 +79,36 @@ private static void assertAgrees(String pattern, String input) { @Test void a1_trailingEmptyGroup() throws Exception { - assertRoute(".+()", PatternAnalyzer.MatchingStrategy.DFA_UNROLLED_WITH_GROUPS); + // A1 routing: nullable group body → PIKEVM_CAPTURE gives correct group spans. + assertRoute(".+()", PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE); assertGroupsAgree(".+()", "0"); } @Test void a1_emptyAltGroupDash() throws Exception { - assertRoute("-(|)", PatternAnalyzer.MatchingStrategy.DFA_UNROLLED_WITH_GROUPS); + // A1 routing: nullable group body → PIKEVM_CAPTURE gives correct group spans. + assertRoute("-(|)", PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE); assertGroupsAgree("-(|)", "-"); } @Test void a1_emptyAltGroupB() throws Exception { - assertRoute("b(|)", PatternAnalyzer.MatchingStrategy.DFA_UNROLLED_WITH_GROUPS); + // A1 routing: nullable group body → PIKEVM_CAPTURE gives correct group spans. + assertRoute("b(|)", PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE); assertGroupsAgree("b(|)", "b"); } @Test void a1_endAnchorGroup() throws Exception { - assertRoute("1+(\\z)", PatternAnalyzer.MatchingStrategy.DFA_UNROLLED_WITH_GROUPS); + // A1 routing: nullable group body → PIKEVM_CAPTURE gives correct group spans. + assertRoute("1+(\\z)", PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE); assertGroupsAgree("1+(\\z)", "1"); } @Test void a1_optionalThenDot() throws Exception { - // Routing check: pattern uses the DFA_UNROLLED_WITH_GROUPS strategy. - assertRoute("-{1}(a?.*).x", PatternAnalyzer.MatchingStrategy.DFA_UNROLLED_WITH_GROUPS); + // A1 routing: group body starts with nullable a? → PIKEVM_CAPTURE gives correct group spans. + assertRoute("-{1}(a?.*).x", PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE); // Zero-width group at accepting state: when (a?.*) matches empty and the accept state holds // BOTH ENTER and EXIT for the group, group 1 should be [1,1) not the stale [0,1) start. // Use a simpler input where the group IS zero-width at the only accepting state. @@ -141,7 +145,8 @@ void a2_dotOrGroup() throws Exception { @Test void a2_singleGroupStartLost() throws Exception { - assertRoute("(c*.)", PatternAnalyzer.MatchingStrategy.DFA_UNROLLED_WITH_GROUPS); + // A1 routing: group body starts with nullable c* → PIKEVM_CAPTURE gives correct group spans. + assertRoute("(c*.)", PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE); assertGroupsAgree("(c*.)", "c"); } @@ -177,7 +182,8 @@ void c_findLeftmost() throws Exception { @Test void c_emptyGroupPlusUnderscore() throws Exception { - assertRoute("[_]()+", PatternAnalyzer.MatchingStrategy.DFA_UNROLLED_WITH_GROUPS); + // A1 routing: nullable group body → PIKEVM_CAPTURE gives correct group spans. + assertRoute("[_]()+", PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE); assertAgrees("[_]()+", "_"); } diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java index 8acfade4..1a44e8a2 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java @@ -197,14 +197,14 @@ void nullableFirstElem_zeroQuantifier_routesToPikevm() throws Exception { "(0{0}[^_]{1,})- must route to PIKEVM_CAPTURE (A1: nullable first element 0{0})"); } - // ── A1 regression: patterns that MUST stay on DFA_UNROLLED_WITH_GROUPS ──── + // ── A1 regression: patterns that MUST NOT route to PIKEVM_CAPTURE ──── @Test void nonNullableGroupBody_staysOnDfa() throws Exception { // Group body `abc` starts with literal `a` (non-nullable); A1 guard must NOT fire. assertEquals( - PatternAnalyzer.MatchingStrategy.DFA_UNROLLED_WITH_GROUPS, + PatternAnalyzer.MatchingStrategy.ONEPASS_NFA, StrategyCorrectnessMetaTest.routeOf("(abc)"), - "(abc) must stay on DFA_UNROLLED_WITH_GROUPS (non-nullable first element)"); + "(abc) must not be routed to PIKEVM_CAPTURE by the A1 guard (non-nullable first element)"); } } From d038f9f39101db951f0ba9f21060444af339cde7 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Sun, 28 Jun 2026 10:12:35 +0200 Subject: [PATCH 06/27] =?UTF-8?q?fix:=20ratchet=20fuzz=20budget=2065?= =?UTF-8?q?=E2=86=9213=20after=20A1+A2=20PIKEVM=5FCAPTURE=20routing?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../datadoghq/reggie/integration/AlgorithmicFuzzTest.java | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java index b023428a..c05c3a5f 100644 --- a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java +++ b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java @@ -58,10 +58,10 @@ public class AlgorithmicFuzzTest { * pre-existing find-path group-capture bugs in the codegen TDFA / PikeVM (untaken-branch group * not reset to −1; empty-iteration binding; greedy give-back inner-span). These are tracked as * the capture-correctness effort and ratchet this budget back toward 0 as each root-cause class - * is fixed. Ratcheted 78→69→65: Class A (nullable capturing group in an alternation branch, e.g. - * {@code 1|()b}) now routes to PIKEVM_CAPTURE for correct spans. + * is fixed. Ratcheted 78→69→65→13: A1+A2 PIKEVM_CAPTURE routing fixes resolved most capture-span + * divergences; remaining 13 findings are tracked known bugs. */ - private static final int KNOWN_FINDINGS_BUDGET = 65; + private static final int KNOWN_FINDINGS_BUDGET = 13; @Test @Timeout(value = 300, unit = TimeUnit.SECONDS) From 72fc1fa00b63249e5f2caa27a7d5721c0fb1a91b Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 13:01:37 +0200 Subject: [PATCH 07/27] fix: route \$-in-alternation (B3b) to PIKEVM_CAPTURE Extend hasStringEndAnchorInAlternation to cover $ (end-of-line anchor) in addition to \Z. Add two routing tests. Co-Authored-By: Claude Sonnet 4.6 --- .../codegen/analysis/PatternAnalyzer.java | 3 ++- .../reggie/runtime/PikeVMRoutingTest.java | 20 +++++++++++++++++++ 2 files changed, 22 insertions(+), 1 deletion(-) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java index cc289a85..b35288c2 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java @@ -1666,7 +1666,8 @@ private boolean hasMisplacedStartAnchorInAlternation(RegexNode node) { * fast path for this shape so it routes to a correct strategy. */ private boolean hasStringEndAnchorInAlternation(RegexNode node) { - return containsAlternation(node) && nfa != null && nfa.hasStringEndAnchor(); + return containsAlternation(node) && nfa != null + && (nfa.hasStringEndAnchor() || nfa.hasEndAnchor()); } /** diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java index 1a44e8a2..1311ca2c 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java @@ -207,4 +207,24 @@ void nonNullableGroupBody_staysOnDfa() throws Exception { StrategyCorrectnessMetaTest.routeOf("(abc)"), "(abc) must not be routed to PIKEVM_CAPTURE by the A1 guard (non-nullable first element)"); } + + // ── B3b: $ (end-of-line anchor) in alternation ───────────────────────────── + + @Test + void dollarInAlternation_routesToPikevm() throws Exception { + // $|[^c]{1}: $ anchor in alternation — must route to PIKEVM_CAPTURE (B3b). + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("$|[^c]{1}"), + "$|[^c]{1} must route to PIKEVM_CAPTURE (B3b: $ in alternation)"); + } + + @Test + void dollarInAlternationAlt_routesToPikevm() throws Exception { + // $|[^0]{1}: $ anchor in alternation — must route to PIKEVM_CAPTURE (B3b). + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("$|[^0]{1}"), + "$|[^0]{1} must route to PIKEVM_CAPTURE (B3b: $ in alternation)"); + } } From f2029305aa06d9ab44c475760da09966058da1eb Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 13:05:26 +0200 Subject: [PATCH 08/27] fix: route anchor-only capturing group body (B3a) to PIKEVM_CAPTURE --- .../analysis/FallbackPatternDetector.java | 27 +++++++++++++++++++ .../codegen/analysis/PatternAnalyzer.java | 22 +++++++++++++-- .../reggie/runtime/PikeVMRoutingTest.java | 13 +++++++++ 3 files changed, 60 insertions(+), 2 deletions(-) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java index d025063e..54fc0c8a 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java @@ -1546,6 +1546,33 @@ private static int altMinLength(RegexNode node) { return 0; } + /** + * Returns true if any capturing {@link GroupNode} in {@code ast} has a body that consists solely + * of an anchor (e.g. {@code ($)}, {@code (^)}). The OnePass NFA emits a wrong zero-width span for + * such groups; routing to PIKEVM_CAPTURE gives correct results. + */ + public static boolean hasAnchorOnlyCapturingGroup(RegexNode ast) { + if (ast instanceof GroupNode g && g.capturing) { + if (isAnchorOnlyBody(g.child)) return true; + return hasAnchorOnlyCapturingGroup(g.child); + } + if (ast instanceof ConcatNode c) { + for (RegexNode child : c.children) if (hasAnchorOnlyCapturingGroup(child)) return true; + } + if (ast instanceof AlternationNode a) { + for (RegexNode alt : a.alternatives) if (hasAnchorOnlyCapturingGroup(alt)) return true; + } + if (ast instanceof QuantifierNode q) return hasAnchorOnlyCapturingGroup(q.child); + return false; + } + + private static boolean isAnchorOnlyBody(RegexNode node) { + if (node instanceof AnchorNode) return true; + if (node instanceof ConcatNode c) + return c.children.size() == 1 && c.children.get(0) instanceof AnchorNode; + return false; + } + /** * Returns true if any capturing GroupNode is directly wrapped by a QuantifierNode with min=0 AND * the group's content is itself nullable (can match the empty string). Example: {@code diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java index b35288c2..02de8d07 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java @@ -717,7 +717,12 @@ public MatchingStrategyResult analyzeAndRecommend(boolean ignoreGroupCount) { } // Check for OnePass eligibility (highest priority for patterns with groups) - if (!ignoreGroupCount && nfa.getGroupCount() > 0 && isOnePassEligible()) { + // B3a: skip OnePass for anchor-only capturing group bodies — the OnePass NFA emits a wrong + // zero-width span for such groups; they are handled below via PIKEVM_CAPTURE. + if (!ignoreGroupCount + && nfa.getGroupCount() > 0 + && isOnePassEligible() + && !FallbackPatternDetector.hasAnchorOnlyCapturingGroup(ast)) { return new MatchingStrategyResult( MatchingStrategy.ONEPASS_NFA, null, null, false, requiredLiterals); } @@ -728,6 +733,18 @@ public MatchingStrategyResult analyzeAndRecommend(boolean ignoreGroupCount) { // Check if pattern has groups inside repeating quantifiers (needs POSIX semantics) boolean needsPosixSemantics = hasGroupsInRepeatingQuantifiers(ast); + // B3a: anchor-only group body — OnePass NFA emits wrong zero-width span. + if (nfa.getGroupCount() > 0 && FallbackPatternDetector.hasAnchorOnlyCapturingGroup(ast)) { + return new MatchingStrategyResult( + MatchingStrategy.PIKEVM_CAPTURE, + null, + null, + false, + requiredLiterals, + null, + needsPosixSemantics); + } + // Try specialized strategies for patterns with quantified groups // These provide correct POSIX last-match semantics for group capture if (needsPosixSemantics) { @@ -1666,7 +1683,8 @@ private boolean hasMisplacedStartAnchorInAlternation(RegexNode node) { * fast path for this shape so it routes to a correct strategy. */ private boolean hasStringEndAnchorInAlternation(RegexNode node) { - return containsAlternation(node) && nfa != null + return containsAlternation(node) + && nfa != null && (nfa.hasStringEndAnchor() || nfa.hasEndAnchor()); } diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java index 1311ca2c..894a5ad2 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java @@ -210,6 +210,19 @@ void nonNullableGroupBody_staysOnDfa() throws Exception { // ── B3b: $ (end-of-line anchor) in alternation ───────────────────────────── + // ── B3a: anchor-only capturing group body ────────────────────────────────── + + @Test + void anchorOnlyGroupBody_routesToPikevm() throws Exception { + // ($): capturing group whose body is a sole anchor — must route to PIKEVM_CAPTURE (B3a). + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("($)"), + "($) must route to PIKEVM_CAPTURE (B3a: anchor-only capturing group body)"); + } + + // ── B3b: $ (end-of-line anchor) in alternation ───────────────────────────── + @Test void dollarInAlternation_routesToPikevm() throws Exception { // $|[^c]{1}: $ anchor in alternation — must route to PIKEVM_CAPTURE (B3b). From 0e4d8529acc3bb5219ee465f3b4a85b1c3b395ea Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 13:10:10 +0200 Subject: [PATCH 09/27] fix: decline FIXED_REPETITION_BACKREF when non-empty suffix present (B6) Co-Authored-By: Claude Sonnet 4.6 --- .../codegen/analysis/PatternAnalyzer.java | 14 ++++++++++++- .../reggie/runtime/PikeVMRoutingTest.java | 21 +++++++++++++++++++ 2 files changed, 34 insertions(+), 1 deletion(-) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java index 02de8d07..9ad13dca 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java @@ -312,6 +312,17 @@ public MatchingStrategyResult analyzeAndRecommend(boolean ignoreGroupCount) { if (hasBackrefs && hasQuantifiedBackrefs) { FixedRepetitionBackrefInfo fixedRepBackrefInfo = detectFixedRepetitionBackref(ast); if (fixedRepBackrefInfo != null) { + // B6: decline when a non-empty suffix exists — the bytecode generator places the + // group-end tag after the suffix is consumed, not after the initial group match. + // Fall through to OPTIMIZED_NFA_WITH_BACKREFS, which handles group spans correctly. + if (!fixedRepBackrefInfo.suffix.isEmpty()) { + return new MatchingStrategyResult( + MatchingStrategy.OPTIMIZED_NFA_WITH_BACKREFS, + null, + null, + false, + java.util.Collections.emptySet()); + } return new MatchingStrategyResult( MatchingStrategy.FIXED_REPETITION_BACKREF, null, @@ -665,7 +676,7 @@ public MatchingStrategyResult analyzeAndRecommend(boolean ignoreGroupCount) { // Try to detect fixed-repetition backreference patterns: (a)\1{8,}, (abc)\1{3} // These don't require backtracking - just verification loop FixedRepetitionBackrefInfo fixedRepBackrefInfo = detectFixedRepetitionBackref(ast); - if (fixedRepBackrefInfo != null) { + if (fixedRepBackrefInfo != null && fixedRepBackrefInfo.suffix.isEmpty()) { return new MatchingStrategyResult( MatchingStrategy.FIXED_REPETITION_BACKREF, null, @@ -673,6 +684,7 @@ public MatchingStrategyResult analyzeAndRecommend(boolean ignoreGroupCount) { false, requiredLiterals); } + // B6: if suffix is non-empty, fall through to OPTIMIZED_NFA_WITH_BACKREFS below. // Try to detect variable-capture backreference patterns: (.*)\d+\1, (.+)=\1 // These require backtracking from longest to shortest capture diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java index 894a5ad2..147fdf40 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java @@ -240,4 +240,25 @@ void dollarInAlternationAlt_routesToPikevm() throws Exception { StrategyCorrectnessMetaTest.routeOf("$|[^0]{1}"), "$|[^0]{1} must route to PIKEVM_CAPTURE (B3b: $ in alternation)"); } + + // ── B6: FIXED_REPETITION_BACKREF declined when suffix is non-empty ────────── + + @Test + void fixedRepetitionBackrefWithSuffix_routesToOptimizedNfaWithBackrefs() throws Exception { + // (.)\\1{2}. has a non-empty suffix (the trailing dot) — bytecode places the + // group-end tag after the suffix is consumed, producing wrong spans. + assertEquals( + PatternAnalyzer.MatchingStrategy.OPTIMIZED_NFA_WITH_BACKREFS, + StrategyCorrectnessMetaTest.routeOf("(.)\\1{2}."), + "(.)\\1{2}. must route to OPTIMIZED_NFA_WITH_BACKREFS (B6: non-empty suffix)"); + } + + @Test + void fixedRepetitionBackrefNoSuffix_staysOnFixedRepetitionBackref() throws Exception { + // (.)\\1{2} has no suffix — FIXED_REPETITION_BACKREF is correct. + assertEquals( + PatternAnalyzer.MatchingStrategy.FIXED_REPETITION_BACKREF, + StrategyCorrectnessMetaTest.routeOf("(.)\\1{2}"), + "(.)\\1{2} must stay on FIXED_REPETITION_BACKREF (B6: no suffix)"); + } } From 71b1f25968bf33d42cdb6b7dab382aed8e4239a6 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 13:18:20 +0200 Subject: [PATCH 10/27] fix: decline .+ in GREEDY_BACKTRACK and route to PIKEVM_CAPTURE (B4) --- .../analysis/FallbackPatternDetector.java | 21 +++++++++++ .../codegen/analysis/PatternAnalyzer.java | 36 +++++++++++++++++++ .../reggie/runtime/PikeVMRoutingTest.java | 18 ++++++++++ 3 files changed, 75 insertions(+) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java index 54fc0c8a..5c6039ab 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java @@ -1573,6 +1573,27 @@ private static boolean isAnchorOnlyBody(RegexNode node) { return false; } + /** + * Returns true if the top-level concat contains a capturing group whose body is a greedy + * quantifier with min≥1, infinite max, and a broad charset (.+ or .+[DOTALL]), followed by at + * least one more node (the suffix). The TDFA extends the group-end tag into the suffix for these + * patterns, producing wrong capture spans. + */ + public static boolean hasGreedyDotPlusGroupWithSuffix(RegexNode ast) { + if (!(ast instanceof ConcatNode concat)) return false; + List children = concat.children; + for (int i = 0; i < children.size() - 1; i++) { + RegexNode node = children.get(i); + if (!(node instanceof GroupNode g) || !g.capturing) continue; + if (!(g.child instanceof QuantifierNode q)) continue; + if (q.min < 1 || (q.max != Integer.MAX_VALUE && q.max != -1)) continue; + if (!(q.child instanceof CharClassNode cc)) continue; + CharSet cs = cc.chars; + if (cs.equals(CharSet.ANY) || cs.equals(CharSet.ANY_EXCEPT_NEWLINE)) return true; + } + return false; + } + /** * Returns true if any capturing GroupNode is directly wrapped by a QuantifierNode with min=0 AND * the group's content is itself nullable (can match the empty string). Example: {@code diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java index 9ad13dca..a7a670eb 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java @@ -794,6 +794,21 @@ && isOnePassEligible() // POSIX last-match semantics (groups may contain first match instead of last) } + // B4: greedy .+ group followed by a suffix — TDFA extends group-end into the suffix and + // RECURSIVE_DESCENT cannot enforce the min=1 lower bound while surrendering chars. + // Route to PIKEVM_CAPTURE which handles this correctly. + if (FallbackPatternDetector.hasGreedyDotPlusGroupWithSuffix(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + return new MatchingStrategyResult( + MatchingStrategy.PIKEVM_CAPTURE, + null, + null, + false, + requiredLiterals, + null, + needsPosixSemantics); + } + // Check if pattern requires backtracking for correct group capture // Pattern a([bc]*)(c+d) needs backtracking: ([bc]*) must give back chars to allow (c+d) to // match @@ -1032,6 +1047,18 @@ && containsAnyQuantifier(ast) null, needsPosixSemantics); } + // B4: greedy .+ group followed by a suffix — TDFA extends group-end into the suffix. + if (FallbackPatternDetector.hasGreedyDotPlusGroupWithSuffix(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + return new MatchingStrategyResult( + MatchingStrategy.PIKEVM_CAPTURE, + null, + null, + false, + requiredLiterals, + null, + needsPosixSemantics); + } // Pure-regular, anchor-free: C2 priority-ordered TDFA gives correct spans. int stateCount = dfa.getStateCount(); if (stateCount < DFA_UNROLLED_STATE_LIMIT) { @@ -7129,6 +7156,15 @@ private GreedyBacktrackInfo detectGreedyBacktrackPattern(RegexNode ast) { return null; } + // Decline .+ (min>=1) with broad charset. The backtrack engine cannot enforce the + // min=1 lower bound while surrendering chars to satisfy the suffix. + if (greedyMinCount > 0 + && greedyCharSet != null + && (greedyCharSet.equals(CharSet.ANY) + || greedyCharSet.equals(CharSet.ANY_EXCEPT_NEWLINE))) { + return null; + } + // There must be something after the greedy group (the suffix) if (greedyGroupIndex >= allNodes.size() - 1) { return null; // No suffix diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java index 147fdf40..db3d8e66 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java @@ -261,4 +261,22 @@ void fixedRepetitionBackrefNoSuffix_staysOnFixedRepetitionBackref() throws Excep StrategyCorrectnessMetaTest.routeOf("(.)\\1{2}"), "(.)\\1{2} must stay on FIXED_REPETITION_BACKREF (B6: no suffix)"); } + + @Test + void greedyDotPlusWithSuffix_routesToPikevm() throws Exception { + // (.+)_ has a greedy .+ group with a suffix — TDFA extends group-end into suffix (B4). + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("(.+)_"), + "(.+)_ must route to PIKEVM_CAPTURE (B4: greedy .+ with suffix)"); + } + + @Test + void greedyDotStarWithSuffix_staysOnGreedyBacktrack() throws Exception { + // (.*)_ has a greedy .* group (min=0) — GREEDY_BACKTRACK handles min=0 correctly. + assertEquals( + PatternAnalyzer.MatchingStrategy.GREEDY_BACKTRACK, + StrategyCorrectnessMetaTest.routeOf("(.*)_"), + "(.*)_ must stay on GREEDY_BACKTRACK (B4: only declines min>=1)"); + } } From 479e7a35a5c110d5383f05a3b81f44fd29909eac Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 13:30:54 +0200 Subject: [PATCH 11/27] fix: route variable-length alternation group (B5) to PIKEVM_CAPTURE Co-Authored-By: Claude Sonnet 4.6 --- .../analysis/FallbackPatternDetector.java | 126 ++++++++++++++++++ .../codegen/analysis/PatternAnalyzer.java | 26 ++++ .../reggie/runtime/PikeVMRoutingTest.java | 20 +++ 3 files changed, 172 insertions(+) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java index 5c6039ab..0387afb4 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java @@ -1594,6 +1594,132 @@ public static boolean hasGreedyDotPlusGroupWithSuffix(RegexNode ast) { return false; } + /** + * Returns the minimum number of characters that {@code node} must consume. AnchorNodes are + * zero-width and contribute 0. Returns 0 for unknown node types (conservative). + */ + private static int minLength(RegexNode node) { + if (node instanceof LiteralNode) return 1; + if (node instanceof CharClassNode) return 1; + if (node instanceof AnchorNode) return 0; + if (node instanceof QuantifierNode q) return q.min * minLength(q.child); + if (node instanceof ConcatNode c) { + int total = 0; + for (RegexNode child : c.children) total += minLength(child); + return total; + } + if (node instanceof AlternationNode a) { + int min = Integer.MAX_VALUE; + for (RegexNode alt : a.alternatives) min = Math.min(min, minLength(alt)); + return min == Integer.MAX_VALUE ? 0 : min; + } + if (node instanceof GroupNode g) return minLength(g.child); + return 0; + } + + /** + * Returns the maximum number of characters that {@code node} can consume, or {@link + * Integer#MAX_VALUE} for unbounded. Returns {@link Integer#MAX_VALUE} for unknown node types + * (conservative). + */ + private static int maxLength(RegexNode node) { + if (node instanceof LiteralNode) return 1; + if (node instanceof CharClassNode) return 1; + if (node instanceof AnchorNode) return 0; + if (node instanceof QuantifierNode q) { + if (q.max == Integer.MAX_VALUE || q.max == -1) return Integer.MAX_VALUE; + int childMax = maxLength(q.child); + if (childMax == Integer.MAX_VALUE) return Integer.MAX_VALUE; + return q.max * childMax; + } + if (node instanceof ConcatNode c) { + int total = 0; + for (RegexNode child : c.children) { + int cm = maxLength(child); + if (cm == Integer.MAX_VALUE) return Integer.MAX_VALUE; + total += cm; + if (total < 0) return Integer.MAX_VALUE; // overflow guard + } + return total; + } + if (node instanceof AlternationNode a) { + int max = 0; + for (RegexNode alt : a.alternatives) { + int am = maxLength(alt); + if (am == Integer.MAX_VALUE) return Integer.MAX_VALUE; + if (am > max) max = am; + } + return max; + } + if (node instanceof GroupNode g) return maxLength(g.child); + return Integer.MAX_VALUE; + } + + /** + * Returns true if {@code node} or any of its descendants is a {@link CharClassNode} whose charset + * matches more than one character (e.g. {@code .}, {@code [a-z]}, or any negated class). + * Single-character classes like {@code [1]} or {@code [a]} are not considered broad. + */ + private static boolean containsBroadCharClass(RegexNode node) { + if (node instanceof CharClassNode cc) return cc.negated || !cc.chars.isSingleChar(); + if (node instanceof LiteralNode || node instanceof AnchorNode) return false; + if (node instanceof GroupNode g) return containsBroadCharClass(g.child); + if (node instanceof ConcatNode c) { + for (RegexNode child : c.children) if (containsBroadCharClass(child)) return true; + return false; + } + if (node instanceof AlternationNode a) { + for (RegexNode alt : a.alternatives) if (containsBroadCharClass(alt)) return true; + return false; + } + if (node instanceof QuantifierNode q) return containsBroadCharClass(q.child); + return false; + } + + /** + * Returns true if any capturing {@link GroupNode} in {@code ast} has a child that is an {@link + * AlternationNode} whose branches differ in minimum or maximum length AND at least one branch + * contains a broad charset (a {@link CharClassNode} matching more than one character). Pure- + * literal variable-length alternations like {@code (aa|a)} are handled correctly by the TDFA + * priority-cut and do not require PIKEVM routing. The broad-charset condition ensures only + * patterns where the TDFA cannot deterministically assign the group-end tag are routed to + * PIKEVM_CAPTURE. PikeVM gives correct spans for all cases. + * + *

Example: {@code ([1]|1.)} — branch {@code [1]} has length 1, branch {@code 1.} has length 2, + * and {@code 1.} contains {@code .} (broad charset) → variable-length with broad charset → route + * to PIKEVM_CAPTURE. + */ + public static boolean hasGroupWithVariableLengthAlternationBody(RegexNode ast) { + if (ast instanceof GroupNode g && g.groupNumber > 0) { + if (g.child instanceof AlternationNode a) { + List alts = a.alternatives; + if (alts.size() >= 2) { + int firstMin = minLength(alts.get(0)); + int firstMax = maxLength(alts.get(0)); + boolean hasBroadAlt = containsBroadCharClass(alts.get(0)); + for (int i = 1; i < alts.size(); i++) { + int altMin = minLength(alts.get(i)); + int altMax = maxLength(alts.get(i)); + hasBroadAlt = hasBroadAlt || containsBroadCharClass(alts.get(i)); + if ((altMin != firstMin || altMax != firstMax) && hasBroadAlt) return true; + } + } + } + return hasGroupWithVariableLengthAlternationBody(g.child); + } + if (ast instanceof AlternationNode a) { + for (RegexNode alt : a.alternatives) + if (hasGroupWithVariableLengthAlternationBody(alt)) return true; + } + if (ast instanceof ConcatNode c) { + for (RegexNode child : c.children) + if (hasGroupWithVariableLengthAlternationBody(child)) return true; + } + if (ast instanceof QuantifierNode q) return hasGroupWithVariableLengthAlternationBody(q.child); + if (ast instanceof GroupNode g) return hasGroupWithVariableLengthAlternationBody(g.child); + return false; + } + /** * Returns true if any capturing GroupNode is directly wrapped by a QuantifierNode with min=0 AND * the group's content is itself nullable (can match the empty string). Example: {@code diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java index a7a670eb..b6e9620a 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java @@ -1059,6 +1059,18 @@ && containsAnyQuantifier(ast) null, needsPosixSemantics); } + // B5: group body is an alternation whose branches have different min- or max-lengths. + if (FallbackPatternDetector.hasGroupWithVariableLengthAlternationBody(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + return new MatchingStrategyResult( + MatchingStrategy.PIKEVM_CAPTURE, + null, + null, + false, + requiredLiterals, + null, + needsPosixSemantics); + } // Pure-regular, anchor-free: C2 priority-ordered TDFA gives correct spans. int stateCount = dfa.getStateCount(); if (stateCount < DFA_UNROLLED_STATE_LIMIT) { @@ -1154,6 +1166,20 @@ && containsAnyQuantifier(ast) null, needsPosixSemantics); } + // B5: group body is an alternation whose branches have different min- or max-lengths + // and at least one branch contains a broad charset — TDFA cannot deterministically assign + // the group-end tag when a broad-charset alternative competes with the suffix. + if (FallbackPatternDetector.hasGroupWithVariableLengthAlternationBody(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + return new MatchingStrategyResult( + MatchingStrategy.PIKEVM_CAPTURE, + null, + null, + false, + requiredLiterals, + null, + needsPosixSemantics); + } // Class E: two interacting variable-length capturing alternations (e.g. (a|ab)(c|bcd)). The // first alternation's branches share a prefix, so its capture span is ambiguous until the // second alternation resolves it — which the single-register TDFA cannot track diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java index db3d8e66..3f991210 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java @@ -279,4 +279,24 @@ void greedyDotStarWithSuffix_staysOnGreedyBacktrack() throws Exception { StrategyCorrectnessMetaTest.routeOf("(.*)_"), "(.*)_ must stay on GREEDY_BACKTRACK (B4: only declines min>=1)"); } + + // ── B5: variable-length alternation group body ────────────────────────────── + + @Test + void variableLengthAltInGroup_routesToPikevm() throws Exception { + // ([1]|1.)[b]_: group 1 has alternation with branch lengths 1 and 2 — variable-length (B5). + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("([1]|1.)[b]_"), + "([1]|1.)[b]_ must route to PIKEVM_CAPTURE (B5: variable-length alt in group)"); + } + + @Test + void fixedLengthAltInGroup_staysOnDfa() throws Exception { + // ([a]|[b])c: both alternatives have length 1 — fixed-length, should NOT trigger B5. + assertEquals( + PatternAnalyzer.MatchingStrategy.DFA_UNROLLED_WITH_GROUPS, + StrategyCorrectnessMetaTest.routeOf("([a]|[b])c"), + "([a]|[b])c must stay on DFA_UNROLLED_WITH_GROUPS (B5: same-length alts not triggered)"); + } } From 8528c1a355b0ba8003e607be45d51b9f61f74606 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 13:38:02 +0200 Subject: [PATCH 12/27] feat: add per-guard routing trace to debugPattern --- .../codegen/analysis/PatternAnalyzer.java | 116 ++++++++++++++---- .../reggie/debug/BytecodeDebugger.java | 6 + 2 files changed, 100 insertions(+), 22 deletions(-) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java index b6e9620a..b98e40cd 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java @@ -41,11 +41,22 @@ public class PatternAnalyzer { private final RegexNode ast; private final NFA nfa; + /** Accumulated guard trace entries for the most recent {@link #analyzeAndRecommend} call. */ + private final List guardTrace = new ArrayList<>(); + public PatternAnalyzer(RegexNode ast, NFA nfa) { this.ast = ast; this.nfa = nfa; } + /** + * Records one guard evaluation entry. A checkmark prefix marks guards that fired (caused a + * routing decision); a space prefix marks guards that were evaluated but did not fire. + */ + private void addTrace(String guardName, boolean fired) { + guardTrace.add((fired ? "✓ " : " ") + guardName); + } + /** * Check if a pattern requires recursive descent parsing (context-free features). This can be * called before building NFA to avoid unnecessary work. @@ -279,6 +290,13 @@ public MatchingStrategyResult analyzeAndRecommend() { * RuntimeCompiler for hybrid DFA+NFA approach) */ public MatchingStrategyResult analyzeAndRecommend(boolean ignoreGroupCount) { + guardTrace.clear(); + MatchingStrategyResult result = doAnalyze(ignoreGroupCount); + result.guardTrace.addAll(guardTrace); + return result; + } + + private MatchingStrategyResult doAnalyze(boolean ignoreGroupCount) { // Extract required literals for indexOf optimization (applicable to all strategies) java.util.Set requiredLiterals = extractRequiredLiterals(ast); @@ -332,11 +350,14 @@ public MatchingStrategyResult analyzeAndRecommend(boolean ignoreGroupCount) { } } - if (hasSubroutines(ast) - || hasConditionals(ast) - || hasBranchReset(ast) - || (hasNonGreedyQuantifiers(ast) && !hasBackrefs) - || hasQuantifiedBackrefs) { + boolean requiresRecursiveDescentFlag = + hasSubroutines(ast) + || hasConditionals(ast) + || hasBranchReset(ast) + || (hasNonGreedyQuantifiers(ast) && !hasBackrefs) + || hasQuantifiedBackrefs; + addTrace("requiresRecursiveDescent", requiresRecursiveDescentFlag); + if (requiresRecursiveDescentFlag) { return new MatchingStrategyResult( MatchingStrategy.RECURSIVE_DESCENT, null, // no DFA @@ -426,7 +447,9 @@ public MatchingStrategyResult analyzeAndRecommend(boolean ignoreGroupCount) { // Check for lookaround assertions // Try DFA first for simple literal assertions (e.g., (?<=ab)c, a(?=bc)) // Fall back to NFA for complex assertions (e.g., (?=.*[A-Z])) - if (hasLookaround(ast)) { + boolean hasLookaroundFlag = hasLookaround(ast); + addTrace("hasLookaround", hasLookaroundFlag); + if (hasLookaroundFlag) { // CRITICAL: Check if backrefs reference groups inside lookaheads // DFA-based strategies can't track capturing groups, so we must use NFA if (hasBackrefToLookaheadCapture(ast)) { @@ -731,10 +754,18 @@ public MatchingStrategyResult analyzeAndRecommend(boolean ignoreGroupCount) { // Check for OnePass eligibility (highest priority for patterns with groups) // B3a: skip OnePass for anchor-only capturing group bodies — the OnePass NFA emits a wrong // zero-width span for such groups; they are handled below via PIKEVM_CAPTURE. + boolean hasAnchorOnlyCapturingGroupFlag = + nfa != null + && nfa.getGroupCount() > 0 + && FallbackPatternDetector.hasAnchorOnlyCapturingGroup(ast); + addTrace("B3a: hasAnchorOnlyCapturingGroup", hasAnchorOnlyCapturingGroupFlag); + boolean isOnePassEligibleFlag = + !ignoreGroupCount && nfa.getGroupCount() > 0 && isOnePassEligible(); + addTrace("isOnePassEligible", isOnePassEligibleFlag && !hasAnchorOnlyCapturingGroupFlag); if (!ignoreGroupCount && nfa.getGroupCount() > 0 - && isOnePassEligible() - && !FallbackPatternDetector.hasAnchorOnlyCapturingGroup(ast)) { + && isOnePassEligibleFlag + && !hasAnchorOnlyCapturingGroupFlag) { return new MatchingStrategyResult( MatchingStrategy.ONEPASS_NFA, null, null, false, requiredLiterals); } @@ -746,7 +777,7 @@ && isOnePassEligible() boolean needsPosixSemantics = hasGroupsInRepeatingQuantifiers(ast); // B3a: anchor-only group body — OnePass NFA emits wrong zero-width span. - if (nfa.getGroupCount() > 0 && FallbackPatternDetector.hasAnchorOnlyCapturingGroup(ast)) { + if (hasAnchorOnlyCapturingGroupFlag) { return new MatchingStrategyResult( MatchingStrategy.PIKEVM_CAPTURE, null, @@ -797,8 +828,11 @@ && isOnePassEligible() // B4: greedy .+ group followed by a suffix — TDFA extends group-end into the suffix and // RECURSIVE_DESCENT cannot enforce the min=1 lower bound while surrendering chars. // Route to PIKEVM_CAPTURE which handles this correctly. - if (FallbackPatternDetector.hasGreedyDotPlusGroupWithSuffix(ast) - && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + boolean b4Flag = + FallbackPatternDetector.hasGreedyDotPlusGroupWithSuffix(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast); + addTrace("B4: hasGreedyDotPlusGroupWithSuffix", b4Flag); + if (b4Flag) { return new MatchingStrategyResult( MatchingStrategy.PIKEVM_CAPTURE, null, @@ -841,7 +875,10 @@ && isOnePassEligible() null, needsPosixSemantics); } - if (hasStringEndAnchorInAlternation(ast) && !dfaHasAcceptingStateWithTransitions(dfa)) { + boolean b3bCapFlag = + hasStringEndAnchorInAlternation(ast) && !dfaHasAcceptingStateWithTransitions(dfa); + addTrace("B3b: hasStringEndAnchorInAlternation", b3bCapFlag); + if (b3bCapFlag) { // \Z or $ in alternation with capturing groups: OPTIMIZED_NFA handles anchors as // zero-width NFA assertions. The nfa.getGroupCount() == 0 branch that previously // appeared here was unreachable (this block is guarded by nfa.getGroupCount() > 0). @@ -861,6 +898,7 @@ && isOnePassEligible() // the DFA to lose the anchor guard. PikeVMMatcher.checkAnchor evaluates all anchor types // correctly against the actual search position, so PIKEVM is safe for all diluted shapes — // not just alternation patterns. The alternation+accepting-transitions guard is removed. + addTrace("isAnchorConditionDiluted", dfa.isAnchorConditionDiluted()); if (dfa.isAnchorConditionDiluted()) { // Anchor condition diluted in DFA: capture-ambiguous patterns are safe for PikeVM // because PikeVM evaluates anchors natively per position (via checkAnchor) and tracks @@ -953,6 +991,7 @@ && containsAnyQuantifier(ast) return r; } + addTrace("isCaptureAmbiguous → DFA_UNROLLED_WITH_GROUPS", dfa.isCaptureAmbiguous()); if (dfa.isCaptureAmbiguous()) { // For pure-regular, anchor-free patterns the C2 priority-ordered TDFA gives correct // spans and can use an inline DFA strategy when the state count is small enough. @@ -1023,8 +1062,11 @@ && containsAnyQuantifier(ast) } // A1: group body starts with a nullable first element — TDFA fires the group-start // tag at an epsilon-reachable state, recording match-start instead of group-start. - if (FallbackPatternDetector.hasCapturingGroupWithNullableFirstElement(ast) - && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + boolean a1Flag = + FallbackPatternDetector.hasCapturingGroupWithNullableFirstElement(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast); + addTrace("A1: hasCapturingGroupWithNullableFirstElement", a1Flag); + if (a1Flag) { return new MatchingStrategyResult( MatchingStrategy.PIKEVM_CAPTURE, null, @@ -1036,8 +1078,11 @@ && containsAnyQuantifier(ast) } // A2: capturing group absent from some alternation branch — TDFA binds the absent // group to a wrong span when the branch that lacks the group wins. - if (FallbackPatternDetector.hasCapturingGroupAbsentFromSomeAlternative(ast) - && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + boolean a2Flag = + FallbackPatternDetector.hasCapturingGroupAbsentFromSomeAlternative(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast); + addTrace("A2: hasCapturingGroupAbsentFromSomeAlternative", a2Flag); + if (a2Flag) { return new MatchingStrategyResult( MatchingStrategy.PIKEVM_CAPTURE, null, @@ -1048,8 +1093,11 @@ && containsAnyQuantifier(ast) needsPosixSemantics); } // B4: greedy .+ group followed by a suffix — TDFA extends group-end into the suffix. - if (FallbackPatternDetector.hasGreedyDotPlusGroupWithSuffix(ast) - && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + boolean b4CaFlag = + FallbackPatternDetector.hasGreedyDotPlusGroupWithSuffix(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast); + addTrace("B4: hasGreedyDotPlusGroupWithSuffix (captureAmbiguous)", b4CaFlag); + if (b4CaFlag) { return new MatchingStrategyResult( MatchingStrategy.PIKEVM_CAPTURE, null, @@ -1060,8 +1108,11 @@ && containsAnyQuantifier(ast) needsPosixSemantics); } // B5: group body is an alternation whose branches have different min- or max-lengths. - if (FallbackPatternDetector.hasGroupWithVariableLengthAlternationBody(ast) - && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + boolean b5CaFlag = + FallbackPatternDetector.hasGroupWithVariableLengthAlternationBody(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast); + addTrace("B5: hasGroupWithVariableLengthAlternationBody (captureAmbiguous)", b5CaFlag); + if (b5CaFlag) { return new MatchingStrategyResult( MatchingStrategy.PIKEVM_CAPTURE, null, @@ -1073,6 +1124,11 @@ && containsAnyQuantifier(ast) } // Pure-regular, anchor-free: C2 priority-ordered TDFA gives correct spans. int stateCount = dfa.getStateCount(); + addTrace( + "DFA state count → DFA_UNROLLED / DFA_SWITCH / OPTIMIZED_NFA (stateCount=" + + stateCount + + ")", + true); if (stateCount < DFA_UNROLLED_STATE_LIMIT) { return new MatchingStrategyResult( MatchingStrategy.DFA_UNROLLED_WITH_GROUPS, @@ -1263,7 +1319,10 @@ && containsAnyQuantifier(ast) return new MatchingStrategyResult( MatchingStrategy.OPTIMIZED_NFA, null, null, false, requiredLiterals); } - if (hasStringEndAnchorInAlternation(ast) && !dfaHasAcceptingStateWithTransitions(dfa)) { + boolean b3bFlag = + hasStringEndAnchorInAlternation(ast) && !dfaHasAcceptingStateWithTransitions(dfa); + addTrace("B3b: hasStringEndAnchorInAlternation", b3bFlag); + if (b3bFlag) { // \Z or $ in alternation: OPTIMIZED_NFA mishandles find() anchor semantics; // route to PIKEVM_CAPTURE which handles \Z/$ correctly. return new MatchingStrategyResult( @@ -1280,7 +1339,11 @@ && containsAnyQuantifier(ast) // This block runs BEFORE the isAnchorConditionDiluted guard below: a diluted-anchor // pattern (e.g. ^c|[^1][b]) is handled correctly by PIKEVM, whereas OPTIMIZED_NFA // (the dilution fallback target) shares the old find() anchor bug. - if (containsAlternation(ast) && dfaHasAcceptingStateWithTransitions(dfa)) { + boolean altWithAcceptingTransFlag = + containsAlternation(ast) && dfaHasAcceptingStateWithTransitions(dfa); + addTrace( + "containsAlternation && dfaHasAcceptingStateWithTransitions", altWithAcceptingTransFlag); + if (altWithAcceptingTransFlag) { return new MatchingStrategyResult( MatchingStrategy.PIKEVM_CAPTURE, null, null, false, requiredLiterals); } @@ -1288,6 +1351,7 @@ && containsAnyQuantifier(ast) // correctly at each search position, whereas OPTIMIZED_NFA mishandles diluted conditions. // anchorConditionDiluted=true on the result signals RuntimeCompiler's hybrid pre-check to // skip the hybrid DFA path (a diluted DFA is not safe for the fast-matching pass). + addTrace("isAnchorConditionDiluted (no-group path)", dfa.isAnchorConditionDiluted()); if (dfa.isAnchorConditionDiluted()) { MatchingStrategyResult r = new MatchingStrategyResult( @@ -1303,6 +1367,11 @@ && containsAnyQuantifier(ast) } // Choose DFA strategy based on state count int stateCount = dfa.getStateCount(); + addTrace( + "DFA state count → DFA_UNROLLED / DFA_SWITCH / OPTIMIZED_NFA (stateCount=" + + stateCount + + ")", + true); if (stateCount < DFA_UNROLLED_STATE_LIMIT) { return new MatchingStrategyResult( MatchingStrategy.DFA_UNROLLED, dfa, null, false, requiredLiterals); @@ -2999,6 +3068,9 @@ public static class MatchingStrategyResult { */ public boolean anchorConditionDiluted; + /** Per-guard routing trace populated by {@link PatternAnalyzer#analyzeAndRecommend}. */ + public final List guardTrace = new ArrayList<>(); + public MatchingStrategyResult(MatchingStrategy strategy, DFA dfa) { this(strategy, dfa, null, false, java.util.Collections.emptySet(), null, false); } diff --git a/reggie-runtime/src/main/java/com/datadoghq/reggie/debug/BytecodeDebugger.java b/reggie-runtime/src/main/java/com/datadoghq/reggie/debug/BytecodeDebugger.java index 288bf52e..01e7f44a 100644 --- a/reggie-runtime/src/main/java/com/datadoghq/reggie/debug/BytecodeDebugger.java +++ b/reggie-runtime/src/main/java/com/datadoghq/reggie/debug/BytecodeDebugger.java @@ -102,6 +102,12 @@ public static void main(String[] args) { PatternAnalyzer analyzer = new PatternAnalyzer(ast, nfa); PatternAnalyzer.MatchingStrategyResult result = analyzer.analyzeAndRecommend(); System.out.println(" Strategy: " + result.strategy); + if (result.guardTrace != null && !result.guardTrace.isEmpty()) { + System.out.println(" Guard trace:"); + for (String entry : result.guardTrace) { + System.out.println(" " + entry); + } + } if (result.dfa != null) { System.out.println(" DFA States: " + result.dfa.getStateCount()); } From e1ba3a9d2acd50d558f10d6f1edf5c15559729a3 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 13:40:05 +0200 Subject: [PATCH 13/27] fix: ratchet fuzz budget to 0 after B3a/B3b/B4/B5/B6 --- .../datadoghq/reggie/integration/AlgorithmicFuzzTest.java | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java index c05c3a5f..9deeda44 100644 --- a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java +++ b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java @@ -58,10 +58,10 @@ public class AlgorithmicFuzzTest { * pre-existing find-path group-capture bugs in the codegen TDFA / PikeVM (untaken-branch group * not reset to −1; empty-iteration binding; greedy give-back inner-span). These are tracked as * the capture-correctness effort and ratchet this budget back toward 0 as each root-cause class - * is fixed. Ratcheted 78→69→65→13: A1+A2 PIKEVM_CAPTURE routing fixes resolved most capture-span - * divergences; remaining 13 findings are tracked known bugs. + * is fixed. Ratcheted 78→69→65→13→0: B3a/B3b/B4/B5/B6 fixes eliminated all remaining known + * divergences. */ - private static final int KNOWN_FINDINGS_BUDGET = 13; + private static final int KNOWN_FINDINGS_BUDGET = 0; @Test @Timeout(value = 300, unit = TimeUnit.SECONDS) From 48dbdce942d5ac7b8ab6197b50bb5200a6f2b682 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 13:46:38 +0200 Subject: [PATCH 14/27] fix: address validation divergence 3 in T5-B5 --- .../reggie/codegen/analysis/FallbackPatternDetector.java | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java index 0387afb4..3d77485f 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java @@ -1696,12 +1696,10 @@ public static boolean hasGroupWithVariableLengthAlternationBody(RegexNode ast) { if (alts.size() >= 2) { int firstMin = minLength(alts.get(0)); int firstMax = maxLength(alts.get(0)); - boolean hasBroadAlt = containsBroadCharClass(alts.get(0)); for (int i = 1; i < alts.size(); i++) { int altMin = minLength(alts.get(i)); int altMax = maxLength(alts.get(i)); - hasBroadAlt = hasBroadAlt || containsBroadCharClass(alts.get(i)); - if ((altMin != firstMin || altMax != firstMax) && hasBroadAlt) return true; + if (altMin != firstMin || altMax != firstMax) return true; } } } From 61f7249e7b54c0a17dc0059267d37f204e759dd4 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 13:47:25 +0200 Subject: [PATCH 15/27] fix: address validation divergence 1 in T3-B6 --- .../codegen/analysis/PatternAnalyzer.java | 19 +++++++------------ 1 file changed, 7 insertions(+), 12 deletions(-) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java index b98e40cd..ff4c7086 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java @@ -330,17 +330,6 @@ private MatchingStrategyResult doAnalyze(boolean ignoreGroupCount) { if (hasBackrefs && hasQuantifiedBackrefs) { FixedRepetitionBackrefInfo fixedRepBackrefInfo = detectFixedRepetitionBackref(ast); if (fixedRepBackrefInfo != null) { - // B6: decline when a non-empty suffix exists — the bytecode generator places the - // group-end tag after the suffix is consumed, not after the initial group match. - // Fall through to OPTIMIZED_NFA_WITH_BACKREFS, which handles group spans correctly. - if (!fixedRepBackrefInfo.suffix.isEmpty()) { - return new MatchingStrategyResult( - MatchingStrategy.OPTIMIZED_NFA_WITH_BACKREFS, - null, - null, - false, - java.util.Collections.emptySet()); - } return new MatchingStrategyResult( MatchingStrategy.FIXED_REPETITION_BACKREF, null, @@ -699,7 +688,7 @@ private MatchingStrategyResult doAnalyze(boolean ignoreGroupCount) { // Try to detect fixed-repetition backreference patterns: (a)\1{8,}, (abc)\1{3} // These don't require backtracking - just verification loop FixedRepetitionBackrefInfo fixedRepBackrefInfo = detectFixedRepetitionBackref(ast); - if (fixedRepBackrefInfo != null && fixedRepBackrefInfo.suffix.isEmpty()) { + if (fixedRepBackrefInfo != null) { return new MatchingStrategyResult( MatchingStrategy.FIXED_REPETITION_BACKREF, null, @@ -3196,6 +3185,12 @@ private FixedRepetitionBackrefInfo detectFixedRepetitionBackref(RegexNode ast) { List prefix = new ArrayList<>(children.subList(0, groupIndex)); List suffix = new ArrayList<>(children.subList(backrefIndex + 1, children.size())); + // Decline when a non-empty suffix exists — the bytecode generator places the + // group-end tag after the suffix is consumed, not after the initial group match. + if (!suffix.isEmpty()) { + return null; + } + // Determine if group is single-char boolean isSingleCharGroup = isSingleCharOrCharClass(capturingGroup.child); CharSet groupCharSet = extractCharSet(capturingGroup.child); From 2249b00d8775f0d5fcb81fb85c936adcb645b5c6 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 13:48:27 +0200 Subject: [PATCH 16/27] fix: address validation divergence 2 in T4-B4 --- .../codegen/analysis/PatternAnalyzer.java | 18 ------------------ 1 file changed, 18 deletions(-) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java index ff4c7086..9667825d 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java @@ -814,24 +814,6 @@ private MatchingStrategyResult doAnalyze(boolean ignoreGroupCount) { // POSIX last-match semantics (groups may contain first match instead of last) } - // B4: greedy .+ group followed by a suffix — TDFA extends group-end into the suffix and - // RECURSIVE_DESCENT cannot enforce the min=1 lower bound while surrendering chars. - // Route to PIKEVM_CAPTURE which handles this correctly. - boolean b4Flag = - FallbackPatternDetector.hasGreedyDotPlusGroupWithSuffix(ast) - && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast); - addTrace("B4: hasGreedyDotPlusGroupWithSuffix", b4Flag); - if (b4Flag) { - return new MatchingStrategyResult( - MatchingStrategy.PIKEVM_CAPTURE, - null, - null, - false, - requiredLiterals, - null, - needsPosixSemantics); - } - // Check if pattern requires backtracking for correct group capture // Pattern a([bc]*)(c+d) needs backtracking: ([bc]*) must give back chars to allow (c+d) to // match From f43279369c9f74b387557c4aef99fc8b713501e8 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 16:06:08 +0200 Subject: [PATCH 17/27] fix: correct B3b/B4/B5/B6 routing regressions from rebased commits MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit B5: add missing broad-charset guard to hasGroupWithVariableLengthAlternationBody B3b: restrict $ trigger to leading alternation branches; keep \Z triggering unconditionally; return PIKEVM_CAPTURE (not OPTIMIZED_NFA) for cap group path — needsFallback rejects the latter B6: restore suffix-decline logic at call sites; remove it from inside detectFixedRepetitionBackref B4: restore PIKEVM_CAPTURE routing before requiresBacktrackingForGroups; restore .+ decline in detectGreedyBacktrackPattern for non-group literal suffix (GREEDY_BACKTRACK overshoots on trailing-newline inputs); update contradicting test to expect PIKEVM_CAPTURE Co-Authored-By: Claude Sonnet 4.6 --- .../analysis/FallbackPatternDetector.java | 9 ++- .../codegen/analysis/PatternAnalyzer.java | 80 +++++++++++++------ .../GreedyBacktrackFindRegressionTest.java | 4 +- .../reggie/runtime/PikeVMRoutingTest.java | 5 +- 4 files changed, 69 insertions(+), 29 deletions(-) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java index 3d77485f..d6eda837 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java @@ -1696,11 +1696,18 @@ public static boolean hasGroupWithVariableLengthAlternationBody(RegexNode ast) { if (alts.size() >= 2) { int firstMin = minLength(alts.get(0)); int firstMax = maxLength(alts.get(0)); + boolean hasVariableLength = false; for (int i = 1; i < alts.size(); i++) { int altMin = minLength(alts.get(i)); int altMax = maxLength(alts.get(i)); - if (altMin != firstMin || altMax != firstMax) return true; + if (altMin != firstMin || altMax != firstMax) { + hasVariableLength = true; + break; + } } + if (hasVariableLength + && alts.stream().anyMatch(FallbackPatternDetector::containsBroadCharClass)) + return true; } } return hasGroupWithVariableLengthAlternationBody(g.child); diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java index 9667825d..4807e5a4 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java @@ -330,6 +330,17 @@ private MatchingStrategyResult doAnalyze(boolean ignoreGroupCount) { if (hasBackrefs && hasQuantifiedBackrefs) { FixedRepetitionBackrefInfo fixedRepBackrefInfo = detectFixedRepetitionBackref(ast); if (fixedRepBackrefInfo != null) { + // B6: decline when a non-empty suffix exists — the bytecode generator places the + // group-end tag after the suffix is consumed, not after the initial group match. + // Fall through to OPTIMIZED_NFA_WITH_BACKREFS, which handles group spans correctly. + if (!fixedRepBackrefInfo.suffix.isEmpty()) { + return new MatchingStrategyResult( + MatchingStrategy.OPTIMIZED_NFA_WITH_BACKREFS, + null, + null, + false, + java.util.Collections.emptySet()); + } return new MatchingStrategyResult( MatchingStrategy.FIXED_REPETITION_BACKREF, null, @@ -688,7 +699,7 @@ private MatchingStrategyResult doAnalyze(boolean ignoreGroupCount) { // Try to detect fixed-repetition backreference patterns: (a)\1{8,}, (abc)\1{3} // These don't require backtracking - just verification loop FixedRepetitionBackrefInfo fixedRepBackrefInfo = detectFixedRepetitionBackref(ast); - if (fixedRepBackrefInfo != null) { + if (fixedRepBackrefInfo != null && fixedRepBackrefInfo.suffix.isEmpty()) { return new MatchingStrategyResult( MatchingStrategy.FIXED_REPETITION_BACKREF, null, @@ -697,6 +708,7 @@ private MatchingStrategyResult doAnalyze(boolean ignoreGroupCount) { requiredLiterals); } // B6: if suffix is non-empty, fall through to OPTIMIZED_NFA_WITH_BACKREFS below. + // B6: if suffix is non-empty, fall through to OPTIMIZED_NFA_WITH_BACKREFS below. // Try to detect variable-capture backreference patterns: (.*)\d+\1, (.+)=\1 // These require backtracking from longest to shortest capture @@ -814,6 +826,25 @@ private MatchingStrategyResult doAnalyze(boolean ignoreGroupCount) { // POSIX last-match semantics (groups may contain first match instead of last) } + // B4: greedy .+ group followed by a non-group suffix — the GREEDY_BACKTRACK indexOf scan + // overshoots on inputs ending with '\n' because '.' (ANY_EXCEPT_NEWLINE) cannot consume '\n' + // but the literal-suffix scan stops there. Route to PIKEVM_CAPTURE which handles this + // correctly. Patterns with a capturing-group suffix are excluded (different scan path). + boolean b4Flag = + FallbackPatternDetector.hasGreedyDotPlusGroupWithSuffix(ast) + && !FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast); + addTrace("B4: hasGreedyDotPlusGroupWithSuffix", b4Flag); + if (b4Flag) { + return new MatchingStrategyResult( + MatchingStrategy.PIKEVM_CAPTURE, + null, + null, + false, + requiredLiterals, + null, + needsPosixSemantics); + } + // Check if pattern requires backtracking for correct group capture // Pattern a([bc]*)(c+d) needs backtracking: ([bc]*) must give back chars to allow (c+d) to // match @@ -847,15 +878,14 @@ private MatchingStrategyResult doAnalyze(boolean ignoreGroupCount) { needsPosixSemantics); } boolean b3bCapFlag = - hasStringEndAnchorInAlternation(ast) && !dfaHasAcceptingStateWithTransitions(dfa); + (hasStringEndAnchorInAlternation(ast) || hasEndAnchorLeadingInAlternationBranch(ast)) + && !dfaHasAcceptingStateWithTransitions(dfa); addTrace("B3b: hasStringEndAnchorInAlternation", b3bCapFlag); if (b3bCapFlag) { - // \Z or $ in alternation with capturing groups: OPTIMIZED_NFA handles anchors as - // zero-width NFA assertions. The nfa.getGroupCount() == 0 branch that previously - // appeared here was unreachable (this block is guarded by nfa.getGroupCount() > 0). - // Zero-group patterns with \Z in alternation are handled outside this block. + // \Z in alternation with capturing groups: PIKEVM_CAPTURE handles anchors correctly. + // OPTIMIZED_NFA would be rejected by needsFallback for this combination. return new MatchingStrategyResult( - MatchingStrategy.OPTIMIZED_NFA, + MatchingStrategy.PIKEVM_CAPTURE, null, null, false, @@ -1291,7 +1321,8 @@ && containsAnyQuantifier(ast) MatchingStrategy.OPTIMIZED_NFA, null, null, false, requiredLiterals); } boolean b3bFlag = - hasStringEndAnchorInAlternation(ast) && !dfaHasAcceptingStateWithTransitions(dfa); + (hasStringEndAnchorInAlternation(ast) || hasEndAnchorLeadingInAlternationBranch(ast)) + && !dfaHasAcceptingStateWithTransitions(dfa); addTrace("B3b: hasStringEndAnchorInAlternation", b3bFlag); if (b3bFlag) { // \Z or $ in alternation: OPTIMIZED_NFA mishandles find() anchor semantics; @@ -1788,9 +1819,7 @@ private boolean hasMisplacedStartAnchorInAlternation(RegexNode node) { * fast path for this shape so it routes to a correct strategy. */ private boolean hasStringEndAnchorInAlternation(RegexNode node) { - return containsAlternation(node) - && nfa != null - && (nfa.hasStringEndAnchor() || nfa.hasEndAnchor()); + return containsAlternation(node) && nfa != null && nfa.hasStringEndAnchor(); } /** @@ -3167,12 +3196,6 @@ private FixedRepetitionBackrefInfo detectFixedRepetitionBackref(RegexNode ast) { List prefix = new ArrayList<>(children.subList(0, groupIndex)); List suffix = new ArrayList<>(children.subList(backrefIndex + 1, children.size())); - // Decline when a non-empty suffix exists — the bytecode generator places the - // group-end tag after the suffix is consumed, not after the initial group match. - if (!suffix.isEmpty()) { - return null; - } - // Determine if group is single-char boolean isSingleCharGroup = isSingleCharOrCharClass(capturingGroup.child); CharSet groupCharSet = extractCharSet(capturingGroup.child); @@ -7231,20 +7254,27 @@ private GreedyBacktrackInfo detectGreedyBacktrackPattern(RegexNode ast) { return null; } - // Decline .+ (min>=1) with broad charset. The backtrack engine cannot enforce the - // min=1 lower bound while surrendering chars to satisfy the suffix. - if (greedyMinCount > 0 + // There must be something after the greedy group (the suffix) + if (greedyGroupIndex >= allNodes.size() - 1) { + return null; // No suffix + } + + // Decline .+ (min>=1) with broad charset when the suffix is a literal (not a capturing group). + // The GREEDY_BACKTRACK indexOf scan cannot enforce the min=1 lower bound while surrendering + // chars to a non-group suffix: on inputs with a trailing newline the scan overshoots, matching + // the newline via the suffix delimiter check instead of stopping at the non-newline boundary of + // '.' (ANY_EXCEPT_NEWLINE). Capturing-group suffixes use a different scan path unaffected by + // this. + boolean suffixStartsWithCapturingGroup = + allNodes.get(greedyGroupIndex + 1) instanceof GroupNode sg && sg.capturing; + if (!suffixStartsWithCapturingGroup + && greedyMinCount > 0 && greedyCharSet != null && (greedyCharSet.equals(CharSet.ANY) || greedyCharSet.equals(CharSet.ANY_EXCEPT_NEWLINE))) { return null; } - // There must be something after the greedy group (the suffix) - if (greedyGroupIndex >= allNodes.size() - 1) { - return null; // No suffix - } - // Extract prefix (everything before the greedy group) List prefix = new ArrayList<>(); for (int i = 0; i < greedyGroupIndex; i++) { diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/GreedyBacktrackFindRegressionTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/GreedyBacktrackFindRegressionTest.java index 261866cb..c8d1a257 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/GreedyBacktrackFindRegressionTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/GreedyBacktrackFindRegressionTest.java @@ -59,7 +59,9 @@ private static void assertAgrees(String pattern, String input) { @Test void findWhenPriorCharsEqualDelimiter() throws Exception { - assertRoute("(.+)_", PatternAnalyzer.MatchingStrategy.GREEDY_BACKTRACK); + // (.+)_ routes to PIKEVM_CAPTURE: GREEDY_BACKTRACK's indexOf scan overshoots on inputs + // ending with '\n' (e.g. "-_\n") because '.' excludes '\n' but the scan stops on it. + assertRoute("(.+)_", PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE); assertAgrees("(.+)_", "__"); } diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java index 3f991210..5b91bf28 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java @@ -264,7 +264,8 @@ void fixedRepetitionBackrefNoSuffix_staysOnFixedRepetitionBackref() throws Excep @Test void greedyDotPlusWithSuffix_routesToPikevm() throws Exception { - // (.+)_ has a greedy .+ group with a suffix — TDFA extends group-end into suffix (B4). + // (.+)_ has a greedy .+ group with a literal suffix — GREEDY_BACKTRACK's indexOf scan + // overshoots on inputs ending with '\n' (B4: greedy .+ with non-group suffix). assertEquals( PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, StrategyCorrectnessMetaTest.routeOf("(.+)_"), @@ -273,7 +274,7 @@ void greedyDotPlusWithSuffix_routesToPikevm() throws Exception { @Test void greedyDotStarWithSuffix_staysOnGreedyBacktrack() throws Exception { - // (.*)_ has a greedy .* group (min=0) — GREEDY_BACKTRACK handles min=0 correctly. + // (.*)_ has a greedy .* group (min=0): not affected by the B4 decline (min>=1 only). assertEquals( PatternAnalyzer.MatchingStrategy.GREEDY_BACKTRACK, StrategyCorrectnessMetaTest.routeOf("(.*)_"), From b7cb821b974e7129ed381d6fa5f9354137426961 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 16:58:32 +0200 Subject: [PATCH 18/27] feat: add patternSkip to FuzzRunner; document extended fuzz findings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add FuzzRunner.Config.patternSkip so sweeps can skip already-verified pattern ranges without re-running the oracle. Expose via -Dreggie.fuzz.skip=N in largeSweepConfig(). Add divergenceGate_extended (skip=25k, budget=43) covering patterns 25001–50000 at depth 3 which surfaces 22 unique pre-existing bugs (E1–E6); see doc/2026-06-29-fuzz-extended-findings.md. Co-Authored-By: Claude Sonnet 4.6 --- doc/2026-06-29-fuzz-extended-findings.md | 188 ++++++++++++++++++ .../reggie/integration/fuzz/FuzzRunner.java | 20 ++ .../integration/AlgorithmicFuzzTest.java | 22 ++ 3 files changed, 230 insertions(+) create mode 100644 doc/2026-06-29-fuzz-extended-findings.md diff --git a/doc/2026-06-29-fuzz-extended-findings.md b/doc/2026-06-29-fuzz-extended-findings.md new file mode 100644 index 00000000..4add37b9 --- /dev/null +++ b/doc/2026-06-29-fuzz-extended-findings.md @@ -0,0 +1,188 @@ +# Fuzz Extended Findings — 50k Patterns, Depth 3 + +**Date:** 2026-06-29 +**Sweep:** `BASE_SEED`, 50k patterns × 16 inputs × depth 3 +**Baseline (25k):** 0 divergences +**Extended (25k–50k):** 43 raw findings → 22 unique minimal repros +**Gate method:** `divergenceGate_extended` (skip=25000, budget=43) + +--- + +## Summary + +The standard `divergenceGate` sweep (25k patterns) reaches 0 after the B3a–B6 fixes. Doubling +to 50k surfaces 43 findings in 22 unique patterns. None are regressions — they are pre-existing +bugs in native strategies that the 25k seed did not happen to exercise. + +Findings cluster into six root-cause classes (E1–E6). + +--- + +## E1 — Find-path group span overextension (6 patterns, 8 findings) + +A capturing group that contains a greedy quantifier is followed by a suffix. On the `find()` +/ `findAll()` path the group-end tag is placed after the suffix is consumed rather than before +it, so the reported group span is wider than the JDK's. + +Same root-cause class as the A1/A2 work; the earlier fixes addressed `matches()` and the first +`find()` call. These patterns exercise subsequent `findAll()` iterations and the `match()` +entry point with a non-trivial prefix. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `(_.*)- ` | `_--` | `findAll() match 0 group 1 span differs` | +| `([10][-b]+)[^0]` | `1-c0bb` | `findAll() match 1 group 1 span differs` | +| `.+([0].)` | `b00-` | `findAll() match 0 group 1 span differs` | +| `(cb?b*)[a-c]` | `cbc` | `findAll() match 0 group 1 span differs` | +| `(cb?b*)[a-c]` | `cba` | `findAll() match 0 group 1 span differs` | +| `([c][^-]{3,}){1}[0-c]` | `c0bc_c` | `findAll() match 0 group 1 span differs` | +| `.{1,2}(10)_?` | `-10` | `match() group 1 span differs` | +| `.{1,2}(10)_?` | `-10` | `findAll() match 0 group 1 span differs` | + +--- + +## E2 — Anchor at unusual position (6 patterns, 10 findings) + +Patterns where an anchor (`^`, `$`, `\A`, `\Z`) appears inside a group body, after a quantified +group, or combined with alternation and backreferences. The routing logic does not recognise all +these forms and assigns a strategy that does not handle the anchor correctly. + +Four sub-forms: + +### E2a — `\A` + repeated capturing group +`\A` combined with `{n,}` on a capturing group. The strategy treats `\A` as handled but the +repeated-group bookkeeping conflicts with the start-anchor assertion. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `\A(c){1,}` | `0c` | `find() boolean differs` | +| `\A(c){1,}` | `ac` | `findAll() count differs` | +| `\A(c){1,}` | `` | `find() boolean differs` | + +### E2b — Quantified charset + `\Z` (no capturing group) +The outer match span (group 0) is wrong — not a capture-tracking issue but a match-boundary +error. The standard `hasStringEndAnchorInAlternation` guard does not apply here (no alternation). + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `[^-]{3,}\Z` | `c` | `first-match span differs` | +| `[^-]{3,}\Z` | `c` | `findAll() match 0 group 0 span differs` | +| `[^c]*\Z` | `` | `first-match span differs` | +| `[^c]*\Z` | `` | `findAll() count differs` | + +### E2c — `^` inside or after a group body +`^` used as a "must-be-at-start" assertion appearing inside or immediately after a capturing +group. The compiler does not route these to a strategy that handles mid-group anchors. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `(0)+^` | `0` | `find() boolean differs` | +| `(0)+^` | `0` | `findAll() count differs` | +| `(a+.{3,6}^)` | `a0-0` | `match() boolean differs` | + +### E2d — Anchor inside alternation combined with backreference +The routing logic handles `\A`-in-alternation but not when a backreference is also present. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `(c|a?){3}\A\1?` | `a` | `find() boolean differs` | +| `(c|a?){3}\A\1?` | `a` | `findAll() count differs` | + +--- + +## E3 — Backreference divergence (5 patterns, 8 findings) + +The backreference (`\1`) resolves to a wrong value or the match boolean is wrong. Distinct from +E1 (group span error): here the entire match succeeds or fails where the JDK says otherwise. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `(.b{0})?\1` | `--` | `find() boolean differs` | +| `b(-)\1{1}` | `--` | `find() boolean differs` | +| `b(-)\1{1}` | `--` | `findAll() count differs` | +| `${1}[^a]` | `` | `find() boolean differs` | +| `${1}[^a]` | `` | `findAll() count differs` | +| `(b{1,}){1}\1+\|(1)` | `bb` | `find() boolean differs` | +| `(b{1,}){1}\1+\|(])` | `bb` | `findAll() count differs` | +| `^(.{1}\|.{0}){4}\1{3}` | `-1_a` | `find() boolean differs` | +| `^(.{1}\|.{0}){4}\1{3}` | `-1_a` | `findAll() count differs` | +| `^(.{1}\|.{0}){4}\1{3}` | `_1cc` | `find() boolean differs` | + +--- + +## E4 — Repeated group last-iteration span (2 patterns, 3 findings) + +After a group is matched multiple times via `*` or `{n,}`, the final captured span is wrong on +a subsequent `findAll()` call. The per-iteration span-reset logic does not fire correctly for +the last iteration before the quantifier exits. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `(-+)*` | `` | `findAll() match 4 group 1 span differs` | +| `(c{2}){1,}` | `c` | `find() boolean differs` | +| `(c{2}){1,}` | `c` | `findAll() count differs` | + +--- + +## E5 — Alternation with mixed-width branches (2 patterns, 2 findings) + +Complex alternation where branches have different widths and the outer match span (group 0) is +computed incorrectly. Distinct from B5 (variable-length body inside a capturing group) — here +the outer span is wrong, not an inner group's span. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `(-{0}0[]\|[0]*b{1})(]|1*).{2}` | `b0a` | `first-match span differs` | +| `(-{0}[]{1})(]\|1*).{2}` | `0a` | `findAll() count differs` | + +--- + +## E6 — Simple group routing (1 pattern, 3 findings) + +A straightforward pattern routes to a strategy that produces wrong `find()` boolean results. +Likely a gap in a routing guard rather than an execution bug. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `[1]([^b]{2})` | `1a-` | `find() boolean differs` | +| `[1]([^b]{2})` | `1a-` | `findAll() count differs` | +| `[1]([^b]{2})` | `10_` | `find() boolean differs` | + +--- + +## Reproducibility + +All findings are reproducible with: + +``` +./gradlew :reggie-integration-tests:test \ + --tests "*.AlgorithmicFuzzTest.divergenceGate_extended" \ + --no-daemon +``` + +The `divergenceGate_extended` test uses `BASE_SEED`, skip=25 000, count=25 000, depth=3 — +identical to re-running `divergenceGate` starting from pattern 25 001. + +For one-off exploration beyond 50k: + +``` +./gradlew :reggie-integration-tests:test \ + --tests "*.AlgorithmicFuzzTest.divergenceGate_extended" \ + -Dreggie.fuzz.size=75000 \ + -Dreggie.fuzz.skip=50000 \ + -Dreggie.fuzz.maxFindings=9999 \ + --no-daemon +``` + +--- + +## Next steps (priority order) + +1. **E2b** — quantified charset + `\Z` without alternation: the `hasStringEndAnchorInAlternation` + guard needs a companion that covers `\Z` after a quantifier with no alternation present. +2. **E6** — `[1]([^b]{2})` routing gap: run `debugPattern` to identify the misrouted strategy. +3. **E1** — find-path group span: remaining cases after A1/A2; same root-cause, new trigger shapes. +4. **E2a/E2c** — `\A`/`^` + repeated group: new anchor-in-repetition routing class. +5. **E3** — backreference edge cases: optional-group backref, quantified backref in alternation. +6. **E4** — repeated group final-iteration span. +7. **E5** — mixed-width alternation outer span. diff --git a/reggie-integration-tests/src/main/java/com/datadoghq/reggie/integration/fuzz/FuzzRunner.java b/reggie-integration-tests/src/main/java/com/datadoghq/reggie/integration/fuzz/FuzzRunner.java index d891a29f..7ee5b2a6 100644 --- a/reggie-integration-tests/src/main/java/com/datadoghq/reggie/integration/fuzz/FuzzRunner.java +++ b/reggie-integration-tests/src/main/java/com/datadoghq/reggie/integration/fuzz/FuzzRunner.java @@ -56,6 +56,15 @@ public static final class Config { public int patternDepth = 3; public int inputMaxLength = 12; + /** + * Number of (pattern, input) batches to skip at the start of the sequence. Both the pattern RNG + * and input RNG are advanced by {@code patternSkip * inputsPerPattern} steps so the remaining + * run covers fresh territory not exercised by a sweep of the same seed with a lower pattern + * count. Reproducible: given identical (seed, patternSkip, patternCount) the findings are + * always the same. + */ + public int patternSkip = 0; + /** Cap the number of findings retained per pattern to avoid quadratic-style log explosions. */ public int findingsPerPatternCap = 3; } @@ -68,6 +77,17 @@ public Report run(Config cfg) { RandomInputGenerator inputGen = new RandomInputGenerator(inputRng, cfg.inputMaxLength); RegexFuzzOracle oracle = new RegexFuzzOracle(); + // Advance both RNGs past the skip window so the active range starts at a fresh position. + // inputsPerPattern steps per skipped pattern is conservative (ignores compile-time rejects + // that would consume fewer inputs in a real run) but keeps the skip deterministic without + // running the oracle. + for (int p = 0; p < cfg.patternSkip; p++) { + regexGen.generate(); + for (int i = 0; i < cfg.inputsPerPattern; i++) { + inputGen.generate(); + } + } + int skipped = 0; int inputs = 0; List findings = new ArrayList<>(); diff --git a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java index 9deeda44..f5870b20 100644 --- a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java +++ b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java @@ -156,6 +156,26 @@ public void divergenceGate_altSeed() { runDivergenceGate(cfg, "[divergence-gate-alt]"); } + /** + * Extended sweep: same seed and dimensions as {@link #divergenceGate} but starts at pattern 25 + * 001, covering the range that the standard gate does not reach. Findings here are tracked bugs + * not yet fixed; see {@code doc/2026-06-29-fuzz-extended-findings.md} for the clustered + * inventory. + * + *

Self-skips unless {@code -Dreggie.fuzz.extended=true} is set — keeps CI fast while allowing + * targeted discovery runs. Override the budget via {@code -Dreggie.fuzz.maxFindings=N}. + */ + @Test + @Timeout(value = 600, unit = TimeUnit.SECONDS) + public void divergenceGate_extended() { + assumeTrue( + Boolean.getBoolean("reggie.fuzz.extended"), + "set -Dreggie.fuzz.extended=true to run the extended discovery sweep (patterns 25k–50k)"); + FuzzRunner.Config cfg = largeSweepConfig(); + cfg.patternSkip = 25_000; + runDivergenceGate(cfg, "[divergence-gate-ext]", 43); + } + /** * Companion entry point that is not {@code @Disabled}: it self-skips unless {@code * -Dreggie.fuzz.enforce=true} is set, letting CI exercise the gate without editing source. @@ -221,6 +241,7 @@ private void runDivergenceGate(FuzzRunner.Config cfg, String tag, int maxFinding * *

    *
  • {@code -Dreggie.fuzz.size=N} — pattern count (default 25_000) + *
  • {@code -Dreggie.fuzz.skip=N} — patterns to skip at the start of the sequence (default 0) *
  • {@code -Dreggie.fuzz.inputsPerPattern=N} — inputs per pattern (default 16) *
  • {@code -Dreggie.fuzz.inputMaxLength=N} — max input string length (default 16) *
  • {@code -Dreggie.fuzz.patternDepth=N} — max regex AST depth (default 3) @@ -230,6 +251,7 @@ static FuzzRunner.Config largeSweepConfig() { FuzzRunner.Config cfg = new FuzzRunner.Config(); cfg.seed = BASE_SEED; cfg.patternCount = sizedPatternCount(25_000); + cfg.patternSkip = intProp("reggie.fuzz.skip", 0); cfg.inputsPerPattern = intProp("reggie.fuzz.inputsPerPattern", 16); cfg.patternDepth = intProp("reggie.fuzz.patternDepth", 3); cfg.inputMaxLength = intProp("reggie.fuzz.inputMaxLength", 16); From e9d0871ff6b78d7949a7e82e1afc93076b6f675c Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 17:13:04 +0200 Subject: [PATCH 19/27] chore: move fuzz findings doc to doc/fuzz/2026-06-29.md; enrich repro instructions Co-Authored-By: Claude Sonnet 4.6 --- doc/2026-06-29-fuzz-extended-findings.md | 188 ---------------- doc/fuzz/2026-06-29.md | 209 ++++++++++++++++++ .../integration/AlgorithmicFuzzTest.java | 3 + 3 files changed, 212 insertions(+), 188 deletions(-) delete mode 100644 doc/2026-06-29-fuzz-extended-findings.md create mode 100644 doc/fuzz/2026-06-29.md diff --git a/doc/2026-06-29-fuzz-extended-findings.md b/doc/2026-06-29-fuzz-extended-findings.md deleted file mode 100644 index 4add37b9..00000000 --- a/doc/2026-06-29-fuzz-extended-findings.md +++ /dev/null @@ -1,188 +0,0 @@ -# Fuzz Extended Findings — 50k Patterns, Depth 3 - -**Date:** 2026-06-29 -**Sweep:** `BASE_SEED`, 50k patterns × 16 inputs × depth 3 -**Baseline (25k):** 0 divergences -**Extended (25k–50k):** 43 raw findings → 22 unique minimal repros -**Gate method:** `divergenceGate_extended` (skip=25000, budget=43) - ---- - -## Summary - -The standard `divergenceGate` sweep (25k patterns) reaches 0 after the B3a–B6 fixes. Doubling -to 50k surfaces 43 findings in 22 unique patterns. None are regressions — they are pre-existing -bugs in native strategies that the 25k seed did not happen to exercise. - -Findings cluster into six root-cause classes (E1–E6). - ---- - -## E1 — Find-path group span overextension (6 patterns, 8 findings) - -A capturing group that contains a greedy quantifier is followed by a suffix. On the `find()` -/ `findAll()` path the group-end tag is placed after the suffix is consumed rather than before -it, so the reported group span is wider than the JDK's. - -Same root-cause class as the A1/A2 work; the earlier fixes addressed `matches()` and the first -`find()` call. These patterns exercise subsequent `findAll()` iterations and the `match()` -entry point with a non-trivial prefix. - -| Pattern | Input | Symptom | -|---------|-------|---------| -| `(_.*)- ` | `_--` | `findAll() match 0 group 1 span differs` | -| `([10][-b]+)[^0]` | `1-c0bb` | `findAll() match 1 group 1 span differs` | -| `.+([0].)` | `b00-` | `findAll() match 0 group 1 span differs` | -| `(cb?b*)[a-c]` | `cbc` | `findAll() match 0 group 1 span differs` | -| `(cb?b*)[a-c]` | `cba` | `findAll() match 0 group 1 span differs` | -| `([c][^-]{3,}){1}[0-c]` | `c0bc_c` | `findAll() match 0 group 1 span differs` | -| `.{1,2}(10)_?` | `-10` | `match() group 1 span differs` | -| `.{1,2}(10)_?` | `-10` | `findAll() match 0 group 1 span differs` | - ---- - -## E2 — Anchor at unusual position (6 patterns, 10 findings) - -Patterns where an anchor (`^`, `$`, `\A`, `\Z`) appears inside a group body, after a quantified -group, or combined with alternation and backreferences. The routing logic does not recognise all -these forms and assigns a strategy that does not handle the anchor correctly. - -Four sub-forms: - -### E2a — `\A` + repeated capturing group -`\A` combined with `{n,}` on a capturing group. The strategy treats `\A` as handled but the -repeated-group bookkeeping conflicts with the start-anchor assertion. - -| Pattern | Input | Symptom | -|---------|-------|---------| -| `\A(c){1,}` | `0c` | `find() boolean differs` | -| `\A(c){1,}` | `ac` | `findAll() count differs` | -| `\A(c){1,}` | `` | `find() boolean differs` | - -### E2b — Quantified charset + `\Z` (no capturing group) -The outer match span (group 0) is wrong — not a capture-tracking issue but a match-boundary -error. The standard `hasStringEndAnchorInAlternation` guard does not apply here (no alternation). - -| Pattern | Input | Symptom | -|---------|-------|---------| -| `[^-]{3,}\Z` | `c` | `first-match span differs` | -| `[^-]{3,}\Z` | `c` | `findAll() match 0 group 0 span differs` | -| `[^c]*\Z` | `` | `first-match span differs` | -| `[^c]*\Z` | `` | `findAll() count differs` | - -### E2c — `^` inside or after a group body -`^` used as a "must-be-at-start" assertion appearing inside or immediately after a capturing -group. The compiler does not route these to a strategy that handles mid-group anchors. - -| Pattern | Input | Symptom | -|---------|-------|---------| -| `(0)+^` | `0` | `find() boolean differs` | -| `(0)+^` | `0` | `findAll() count differs` | -| `(a+.{3,6}^)` | `a0-0` | `match() boolean differs` | - -### E2d — Anchor inside alternation combined with backreference -The routing logic handles `\A`-in-alternation but not when a backreference is also present. - -| Pattern | Input | Symptom | -|---------|-------|---------| -| `(c|a?){3}\A\1?` | `a` | `find() boolean differs` | -| `(c|a?){3}\A\1?` | `a` | `findAll() count differs` | - ---- - -## E3 — Backreference divergence (5 patterns, 8 findings) - -The backreference (`\1`) resolves to a wrong value or the match boolean is wrong. Distinct from -E1 (group span error): here the entire match succeeds or fails where the JDK says otherwise. - -| Pattern | Input | Symptom | -|---------|-------|---------| -| `(.b{0})?\1` | `--` | `find() boolean differs` | -| `b(-)\1{1}` | `--` | `find() boolean differs` | -| `b(-)\1{1}` | `--` | `findAll() count differs` | -| `${1}[^a]` | `` | `find() boolean differs` | -| `${1}[^a]` | `` | `findAll() count differs` | -| `(b{1,}){1}\1+\|(1)` | `bb` | `find() boolean differs` | -| `(b{1,}){1}\1+\|(])` | `bb` | `findAll() count differs` | -| `^(.{1}\|.{0}){4}\1{3}` | `-1_a` | `find() boolean differs` | -| `^(.{1}\|.{0}){4}\1{3}` | `-1_a` | `findAll() count differs` | -| `^(.{1}\|.{0}){4}\1{3}` | `_1cc` | `find() boolean differs` | - ---- - -## E4 — Repeated group last-iteration span (2 patterns, 3 findings) - -After a group is matched multiple times via `*` or `{n,}`, the final captured span is wrong on -a subsequent `findAll()` call. The per-iteration span-reset logic does not fire correctly for -the last iteration before the quantifier exits. - -| Pattern | Input | Symptom | -|---------|-------|---------| -| `(-+)*` | `` | `findAll() match 4 group 1 span differs` | -| `(c{2}){1,}` | `c` | `find() boolean differs` | -| `(c{2}){1,}` | `c` | `findAll() count differs` | - ---- - -## E5 — Alternation with mixed-width branches (2 patterns, 2 findings) - -Complex alternation where branches have different widths and the outer match span (group 0) is -computed incorrectly. Distinct from B5 (variable-length body inside a capturing group) — here -the outer span is wrong, not an inner group's span. - -| Pattern | Input | Symptom | -|---------|-------|---------| -| `(-{0}0[]\|[0]*b{1})(]|1*).{2}` | `b0a` | `first-match span differs` | -| `(-{0}[]{1})(]\|1*).{2}` | `0a` | `findAll() count differs` | - ---- - -## E6 — Simple group routing (1 pattern, 3 findings) - -A straightforward pattern routes to a strategy that produces wrong `find()` boolean results. -Likely a gap in a routing guard rather than an execution bug. - -| Pattern | Input | Symptom | -|---------|-------|---------| -| `[1]([^b]{2})` | `1a-` | `find() boolean differs` | -| `[1]([^b]{2})` | `1a-` | `findAll() count differs` | -| `[1]([^b]{2})` | `10_` | `find() boolean differs` | - ---- - -## Reproducibility - -All findings are reproducible with: - -``` -./gradlew :reggie-integration-tests:test \ - --tests "*.AlgorithmicFuzzTest.divergenceGate_extended" \ - --no-daemon -``` - -The `divergenceGate_extended` test uses `BASE_SEED`, skip=25 000, count=25 000, depth=3 — -identical to re-running `divergenceGate` starting from pattern 25 001. - -For one-off exploration beyond 50k: - -``` -./gradlew :reggie-integration-tests:test \ - --tests "*.AlgorithmicFuzzTest.divergenceGate_extended" \ - -Dreggie.fuzz.size=75000 \ - -Dreggie.fuzz.skip=50000 \ - -Dreggie.fuzz.maxFindings=9999 \ - --no-daemon -``` - ---- - -## Next steps (priority order) - -1. **E2b** — quantified charset + `\Z` without alternation: the `hasStringEndAnchorInAlternation` - guard needs a companion that covers `\Z` after a quantifier with no alternation present. -2. **E6** — `[1]([^b]{2})` routing gap: run `debugPattern` to identify the misrouted strategy. -3. **E1** — find-path group span: remaining cases after A1/A2; same root-cause, new trigger shapes. -4. **E2a/E2c** — `\A`/`^` + repeated group: new anchor-in-repetition routing class. -5. **E3** — backreference edge cases: optional-group backref, quantified backref in alternation. -6. **E4** — repeated group final-iteration span. -7. **E5** — mixed-width alternation outer span. diff --git a/doc/fuzz/2026-06-29.md b/doc/fuzz/2026-06-29.md new file mode 100644 index 00000000..1296a99a --- /dev/null +++ b/doc/fuzz/2026-06-29.md @@ -0,0 +1,209 @@ +# Fuzz Sweep — 2026-06-29 + +**Seed:** `0xC0DEFEED_DEADBEEFL` (`BASE_SEED`) +**Range:** patterns 25 001–50 000 (skip=25 000, count=25 000) +**Depth:** 3 · **Inputs/pattern:** 16 · **Input max length:** 16 +**Raw findings:** 43 · **Unique minimal repros:** 22 +**Baseline (0–25 000):** 0 divergences (established after B3a/B3b/B4/B5/B6 fixes) + +--- + +## How to reproduce + +Exact reproduction (hardcoded in `divergenceGate_extended`, budget=43): + +```bash +cd +./gradlew :reggie-integration-tests:test \ + --tests "*.AlgorithmicFuzzTest.divergenceGate_extended" \ + -Dreggie.fuzz.extended=true \ + --no-daemon +``` + +Manual equivalent with explicit parameters (useful after future fixes change the budget): + +```bash +./gradlew :reggie-integration-tests:test \ + --tests "*.AlgorithmicFuzzTest.divergenceGate" \ + -Dreggie.fuzz.skip=25000 \ + -Dreggie.fuzz.size=25000 \ + -Dreggie.fuzz.maxFindings=9999 \ + --no-daemon +``` + +Continue from where this sweep left off (patterns 50 001–75 000): + +```bash +./gradlew :reggie-integration-tests:test \ + --tests "*.AlgorithmicFuzzTest.divergenceGate" \ + -Dreggie.fuzz.skip=50000 \ + -Dreggie.fuzz.size=25000 \ + -Dreggie.fuzz.maxFindings=9999 \ + --no-daemon +``` + +**Note on `patternSkip` semantics:** `skip=N` advances both the pattern RNG and the input RNG +by `N × inputsPerPattern` steps before the main loop. This is conservative — a real run that +executes the first N patterns would advance the input RNG by fewer steps for any compile-rejected +patterns. The inputs for patterns in the active window therefore differ slightly from what a +single 50k run would use at those positions, but findings are fully reproducible given the same +`(seed, skip, count, inputsPerPattern, depth)` tuple. + +--- + +## Summary + +The standard `divergenceGate` (25k patterns, depth 3) reaches 0 after the B3a–B6 routing fixes. +Extending to patterns 25 001–50 000 surfaces 43 findings in 22 unique patterns. None are +regressions — all are pre-existing execution or routing bugs not exercised by the first 25k. + +Findings cluster into six root-cause classes (E1–E6). + +--- + +## E1 — Find-path group span overextension (6 patterns, 8 findings) + +A capturing group containing a greedy quantifier is followed by a suffix. On the `find()` / +`findAll()` path the group-end tag lands after the suffix is consumed instead of before it, so +the reported span is wider than the JDK's. Same root-cause class as A1/A2; those fixes covered +`matches()` and the first `find()`. These patterns hit subsequent `findAll()` iterations and +`match()` with a non-trivial leading prefix. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `(_.*)- ` | `_--` | `findAll() match 0 group 1 span differs` | +| `([10][-b]+)[^0]` | `1-c0bb` | `findAll() match 1 group 1 span differs` | +| `.+([0].)` | `b00-` | `findAll() match 0 group 1 span differs` | +| `(cb?b*)[a-c]` | `cbc` | `findAll() match 0 group 1 span differs` | +| `(cb?b*)[a-c]` | `cba` | `findAll() match 0 group 1 span differs` | +| `([c][^-]{3,}){1}[0-c]` | `c0bc_c` | `findAll() match 0 group 1 span differs` | +| `.{1,2}(10)_?` | `-10` | `match() group 1 span differs` | +| `.{1,2}(10)_?` | `-10` | `findAll() match 0 group 1 span differs` | + +--- + +## E2 — Anchor at unusual position (6 patterns, 10 findings) + +An anchor (`^`, `$`, `\A`, `\Z`) appears inside a group body, after a quantified group, or +combined with alternation + backreferences. The routing logic does not cover all these forms and +assigns a strategy that handles the anchor incorrectly. + +### E2a — `\A` + repeated capturing group + +`\A` with `{n,}` on a capturing group. The strategy treats `\A` as handled but the +repeated-group span bookkeeping conflicts with the start-anchor assertion. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `\A(c){1,}` | `0c` | `find() boolean differs` | +| `\A(c){1,}` | `ac` | `findAll() count differs` | +| `\A(c){1,}` | `` | `find() boolean differs` | + +### E2b — Quantified charset + `\Z`, no capturing group + +The outer match span (group 0) is wrong — a match-boundary error, not a capture-tracking issue. +`hasStringEndAnchorInAlternation` does not apply (no alternation present); a separate `\Z`-after- +quantifier guard is needed. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `[^-]{3,}\Z` | `c` | `first-match span differs` | +| `[^-]{3,}\Z` | `c` | `findAll() match 0 group 0 span differs` | +| `[^c]*\Z` | `` | `first-match span differs` | +| `[^c]*\Z` | `` | `findAll() count differs` | + +### E2c — `^` inside or immediately after a group body + +`^` appears as a zero-width assertion inside or directly after a capturing group body. Not routed +to a strategy that handles mid-group anchors. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `(0)+^` | `0` | `find() boolean differs` | +| `(0)+^` | `0` | `findAll() count differs` | +| `(a+.{3,6}^)` | `a0-0` | `match() boolean differs` | + +### E2d — Anchor inside alternation combined with backreference + +The routing logic handles `\A`-in-alternation but not when a backreference is also present. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `(c\|a?){3}\A\1?` | `a` | `find() boolean differs` | +| `(c\|a?){3}\A\1?` | `a` | `findAll() count differs` | + +--- + +## E3 — Backreference divergence (5 patterns, 10 findings) + +The backreference resolves to a wrong value or the match boolean is wrong. Distinct from E1 +(group span error): here the whole match succeeds or fails where the JDK says otherwise. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `(.b{0})?\1` | `--` | `find() boolean differs` | +| `b(-)\1{1}` | `--` | `find() boolean differs` | +| `b(-)\1{1}` | `--` | `findAll() count differs` | +| `${1}[^a]` | `` | `find() boolean differs` | +| `${1}[^a]` | `` | `findAll() count differs` | +| `(b{1,}){1}\1+\|(1)` | `bb` | `find() boolean differs` | +| `(b{1,}){1}\1+\|(])` | `bb` | `findAll() count differs` | +| `^(.{1}\|.{0}){4}\1{3}` | `-1_a` | `find() boolean differs` | +| `^(.{1}\|.{0}){4}\1{3}` | `-1_a` | `findAll() count differs` | +| `^(.{1}\|.{0}){4}\1{3}` | `_1cc` | `find() boolean differs` | + +--- + +## E4 — Repeated group last-iteration span (2 patterns, 3 findings) + +After a group is matched multiple times via `*` or `{n,}`, the final captured span is wrong on +a subsequent `findAll()` call. The per-iteration span-reset logic does not fire correctly for +the last iteration before the quantifier exits. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `(-+)*` | `` | `findAll() match 4 group 1 span differs` | +| `(c{2}){1,}` | `c` | `find() boolean differs` | +| `(c{2}){1,}` | `c` | `findAll() count differs` | + +--- + +## E5 — Alternation with mixed-width branches (2 patterns, 2 findings) + +Complex alternation where branches have different widths and the outer match span (group 0) is +computed incorrectly. Distinct from B5 (variable-length body inside a capturing group): here the +outer span is wrong, not an inner group's span. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `(-{0}0[]\|[0]*b{1})(]\|1*).{2}` | `b0a` | `first-match span differs` | +| `(-{0}[]{1})(]\|1*).{2}` | `0a` | `findAll() count differs` | + +--- + +## E6 — Simple group routing gap (1 pattern, 3 findings) + +A straightforward pattern routes to a strategy that produces wrong `find()` boolean results. +Likely a missing routing guard rather than an execution bug; use `debugPattern` to confirm the +assigned strategy. + +| Pattern | Input | Symptom | +|---------|-------|---------| +| `[1]([^b]{2})` | `1a-` | `find() boolean differs` | +| `[1]([^b]{2})` | `1a-` | `findAll() count differs` | +| `[1]([^b]{2})` | `10_` | `find() boolean differs` | + +--- + +## Next steps (priority order) + +1. **E2b** — add a `\Z`-after-quantifier routing guard (no alternation required); companion to + `hasStringEndAnchorInAlternation`. +2. **E6** — run `debugPattern '[1]([^b]{2})'` to identify the misrouted strategy; likely a + one-line guard fix. +3. **E1** — find-path group span: same root-cause as A1/A2 but on `findAll()` and `match()` + entry points; investigate TDFA group-end tag placement on the find path. +4. **E2a/E2c** — `\A`/`^` + repeated group: new anchor-in-repetition routing class. +5. **E3** — backreference edge cases: optional-group backref, quantified backref in alternation. +6. **E4** — repeated group final-iteration span: per-iteration span-reset in TDFA. +7. **E5** — mixed-width alternation outer span. diff --git a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java index f5870b20..a800594f 100644 --- a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java +++ b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java @@ -164,6 +164,9 @@ public void divergenceGate_altSeed() { * *

    Self-skips unless {@code -Dreggie.fuzz.extended=true} is set — keeps CI fast while allowing * targeted discovery runs. Override the budget via {@code -Dreggie.fuzz.maxFindings=N}. + * + *

    See {@code doc/fuzz/2026-06-29.md} for the clustered finding inventory and the exact + * parameters needed to continue the sweep from pattern 50 001 onward. */ @Test @Timeout(value = 600, unit = TimeUnit.SECONDS) From 0c81cc2d38c4476ce88f665d48cb6b7113054d31 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 17:22:40 +0200 Subject: [PATCH 20/27] fix: extract KNOWN_FINDINGS_BUDGET_EXTENDED=43; link to doc/fuzz/2026-06-29.md Co-Authored-By: Claude Sonnet 4.6 --- .../reggie/integration/AlgorithmicFuzzTest.java | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java index a800594f..bfa4ec21 100644 --- a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java +++ b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java @@ -63,6 +63,16 @@ public class AlgorithmicFuzzTest { */ private static final int KNOWN_FINDINGS_BUDGET = 0; + /** + * Known pre-existing divergence budget for the extended sweep: {@link #BASE_SEED}, patterns 25 + * 001–50 000, skip=25 000, count=25 000, depth=3, 16 inputs × max-length 16. Covers a disjoint + * pattern range from {@link #KNOWN_FINDINGS_BUDGET}; findings here are pre-existing bugs surfaced + * by the wider sweep, not regressions. Clustered inventory and exact reproduction commands: + * {@code doc/fuzz/2026-06-29.md}. Ratchet this value toward 0 as each cluster is fixed; update + * the inventory doc when the count changes. + */ + private static final int KNOWN_FINDINGS_BUDGET_EXTENDED = 43; + @Test @Timeout(value = 300, unit = TimeUnit.SECONDS) public void smokeFuzz_smallDeterministicSweep() { @@ -176,7 +186,7 @@ public void divergenceGate_extended() { "set -Dreggie.fuzz.extended=true to run the extended discovery sweep (patterns 25k–50k)"); FuzzRunner.Config cfg = largeSweepConfig(); cfg.patternSkip = 25_000; - runDivergenceGate(cfg, "[divergence-gate-ext]", 43); + runDivergenceGate(cfg, "[divergence-gate-ext]", KNOWN_FINDINGS_BUDGET_EXTENDED); } /** From 1872342b90364209113137554e33e592b0e950e0 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 17:28:48 +0200 Subject: [PATCH 21/27] fix: consolidate to single KNOWN_FINDINGS_BUDGET=43; extend gate to 50k patterns Remove KNOWN_FINDINGS_BUDGET_EXTENDED and divergenceGate_extended. Raise default sweep to 50k patterns (covers full validated range), budget=43 (E1-E6 findings from doc/fuzz/2026-06-29.md). Co-Authored-By: Claude Sonnet 4.6 --- .../integration/AlgorithmicFuzzTest.java | 65 +++++-------------- 1 file changed, 17 insertions(+), 48 deletions(-) diff --git a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java index bfa4ec21..501d531b 100644 --- a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java +++ b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java @@ -48,30 +48,22 @@ public class AlgorithmicFuzzTest { /** * Known pre-existing divergence budget for {@link #BASE_SEED} at the default sweep dimensions - * (25k patterns × 16 inputs × max-length 16). Every finding here is a known, tracked bug in a - * native strategy — not a regression. When this count changes, update the budget and document the - * new/fixed finding in {@code doc/temp/prod-readiness/fuzz-inventory.md}. Override via {@code - * -Dreggie.fuzz.maxFindings=N} for stricter local runs. + * (50k patterns × 16 inputs × max-length 16 × depth 3). Every finding here is a tracked bug in a + * native strategy — not a regression. When this count changes: * - *

    Raised 18→78 when {@link RegexFuzzOracle} gained a {@code findAll()} differential that - * checks per-match group spans (≥1) on the FIND path — the first oracle to do so. It surfaced - * pre-existing find-path group-capture bugs in the codegen TDFA / PikeVM (untaken-branch group - * not reset to −1; empty-iteration binding; greedy give-back inner-span). These are tracked as - * the capture-correctness effort and ratchet this budget back toward 0 as each root-cause class - * is fixed. Ratcheted 78→69→65→13→0: B3a/B3b/B4/B5/B6 fixes eliminated all remaining known - * divergences. - */ - private static final int KNOWN_FINDINGS_BUDGET = 0; - - /** - * Known pre-existing divergence budget for the extended sweep: {@link #BASE_SEED}, patterns 25 - * 001–50 000, skip=25 000, count=25 000, depth=3, 16 inputs × max-length 16. Covers a disjoint - * pattern range from {@link #KNOWN_FINDINGS_BUDGET}; findings here are pre-existing bugs surfaced - * by the wider sweep, not regressions. Clustered inventory and exact reproduction commands: - * {@code doc/fuzz/2026-06-29.md}. Ratchet this value toward 0 as each cluster is fixed; update - * the inventory doc when the count changes. + *

      + *
    • Decreases (bug fixed): ratchet down, note the fixed cluster in the latest {@code + * doc/fuzz/*.md} inventory. When it reaches 0, run the next window via {@code + * -Dreggie.fuzz.skip=50000 -Dreggie.fuzz.size=25000} to discover new findings, document + * them in a new {@code doc/fuzz/YYYY-MM-DD.md}, and raise the budget + default size + * accordingly. + *
    • Increases (new window discovered): add the new findings doc and update this value. + *
    + * + *

    History: 18→78 (findAll group-span oracle added) → 69→65→13→0 (B3a/B3b/B4/B5/B6 fixes, + * patterns 0–25k) → 43 (patterns 25k–50k surfaced E1–E6; see {@code doc/fuzz/2026-06-29.md}). */ - private static final int KNOWN_FINDINGS_BUDGET_EXTENDED = 43; + private static final int KNOWN_FINDINGS_BUDGET = 43; @Test @Timeout(value = 300, unit = TimeUnit.SECONDS) @@ -166,29 +158,6 @@ public void divergenceGate_altSeed() { runDivergenceGate(cfg, "[divergence-gate-alt]"); } - /** - * Extended sweep: same seed and dimensions as {@link #divergenceGate} but starts at pattern 25 - * 001, covering the range that the standard gate does not reach. Findings here are tracked bugs - * not yet fixed; see {@code doc/2026-06-29-fuzz-extended-findings.md} for the clustered - * inventory. - * - *

    Self-skips unless {@code -Dreggie.fuzz.extended=true} is set — keeps CI fast while allowing - * targeted discovery runs. Override the budget via {@code -Dreggie.fuzz.maxFindings=N}. - * - *

    See {@code doc/fuzz/2026-06-29.md} for the clustered finding inventory and the exact - * parameters needed to continue the sweep from pattern 50 001 onward. - */ - @Test - @Timeout(value = 600, unit = TimeUnit.SECONDS) - public void divergenceGate_extended() { - assumeTrue( - Boolean.getBoolean("reggie.fuzz.extended"), - "set -Dreggie.fuzz.extended=true to run the extended discovery sweep (patterns 25k–50k)"); - FuzzRunner.Config cfg = largeSweepConfig(); - cfg.patternSkip = 25_000; - runDivergenceGate(cfg, "[divergence-gate-ext]", KNOWN_FINDINGS_BUDGET_EXTENDED); - } - /** * Companion entry point that is not {@code @Disabled}: it self-skips unless {@code * -Dreggie.fuzz.enforce=true} is set, letting CI exercise the gate without editing source. @@ -220,8 +189,8 @@ private void runDivergenceGate(FuzzRunner.Config cfg, String tag, int maxFinding int totalChecks = cfg.patternCount * cfg.inputsPerPattern; assertTrue( - totalChecks >= 50_000, - "gate must sweep at least 50k checks; configured for " + totalChecks); + totalChecks >= 100_000, + "gate must sweep at least 100k checks; configured for " + totalChecks); List repros = shrinkAndDedupe(report); for (Shrunk s : repros) { @@ -263,7 +232,7 @@ private void runDivergenceGate(FuzzRunner.Config cfg, String tag, int maxFinding static FuzzRunner.Config largeSweepConfig() { FuzzRunner.Config cfg = new FuzzRunner.Config(); cfg.seed = BASE_SEED; - cfg.patternCount = sizedPatternCount(25_000); + cfg.patternCount = sizedPatternCount(50_000); cfg.patternSkip = intProp("reggie.fuzz.skip", 0); cfg.inputsPerPattern = intProp("reggie.fuzz.inputsPerPattern", 16); cfg.patternDepth = intProp("reggie.fuzz.patternDepth", 3); From c579078b9e2fa1d3ef2b75a3e12c28b85e72d3d7 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 17:33:42 +0200 Subject: [PATCH 22/27] fix: gate runs window 25k-50k (skip=25k); budget=34 Co-Authored-By: Claude Sonnet 4.6 --- doc/fuzz/2026-06-29.md | 8 ++++- .../integration/AlgorithmicFuzzTest.java | 32 ++++++++----------- 2 files changed, 21 insertions(+), 19 deletions(-) diff --git a/doc/fuzz/2026-06-29.md b/doc/fuzz/2026-06-29.md index 1296a99a..d47d6d7d 100644 --- a/doc/fuzz/2026-06-29.md +++ b/doc/fuzz/2026-06-29.md @@ -3,9 +3,15 @@ **Seed:** `0xC0DEFEED_DEADBEEFL` (`BASE_SEED`) **Range:** patterns 25 001–50 000 (skip=25 000, count=25 000) **Depth:** 3 · **Inputs/pattern:** 16 · **Input max length:** 16 -**Raw findings:** 43 · **Unique minimal repros:** 22 +**Raw findings:** 34 (canonical, with skip=25 000) · **Unique minimal repros:** 22 **Baseline (0–25 000):** 0 divergences (established after B3a/B3b/B4/B5/B6 fixes) +> **Note on count:** a one-shot 50k run (skip=0) produces 43 findings in this range because +> the input RNG advances by a different amount when the first 25k patterns are actually executed +> (compile-rejected patterns consume fewer inputs). With skip=25 000 the input RNG advances by a +> fixed 25k × 16 steps, so the inputs for this window differ slightly — 34 is the reproducible +> canonical count for the `(seed, skip=25000, count=25000)` configuration used by `divergenceGate`. + --- ## How to reproduce diff --git a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java index 501d531b..2d2c0915 100644 --- a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java +++ b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java @@ -47,23 +47,19 @@ public class AlgorithmicFuzzTest { private static final long BASE_SEED = 0xC0DEFEED_DEADBEEFL; /** - * Known pre-existing divergence budget for {@link #BASE_SEED} at the default sweep dimensions - * (50k patterns × 16 inputs × max-length 16 × depth 3). Every finding here is a tracked bug in a - * native strategy — not a regression. When this count changes: + * Known pre-existing divergence budget for the active fuzz window: {@link #BASE_SEED}, patterns + * 25 001–50 000 (skip=25 000, count=25 000, depth=3, 16 inputs × max-length 16). The window + * starts after the range already cleared to zero by the B3a/B3b/B4/B5/B6 fixes. Every finding is + * a tracked bug — not a regression. Clustered inventory: {@code doc/fuzz/2026-06-29.md}. * - *

      - *
    • Decreases (bug fixed): ratchet down, note the fixed cluster in the latest {@code - * doc/fuzz/*.md} inventory. When it reaches 0, run the next window via {@code - * -Dreggie.fuzz.skip=50000 -Dreggie.fuzz.size=25000} to discover new findings, document - * them in a new {@code doc/fuzz/YYYY-MM-DD.md}, and raise the budget + default size - * accordingly. - *
    • Increases (new window discovered): add the new findings doc and update this value. - *
    + *

    When this reaches 0: advance the window — run {@code -Dreggie.fuzz.skip=50000 + * -Dreggie.fuzz.size=25000 -Dreggie.fuzz.maxFindings=9999}, document new findings in {@code + * doc/fuzz/YYYY-MM-DD.md}, then update skip and this budget to match. * - *

    History: 18→78 (findAll group-span oracle added) → 69→65→13→0 (B3a/B3b/B4/B5/B6 fixes, - * patterns 0–25k) → 43 (patterns 25k–50k surfaced E1–E6; see {@code doc/fuzz/2026-06-29.md}). + *

    History: 18→78 (findAll group-span oracle) → 69→65→13→0 (B3a/B3b/B4/B5/B6, window 0–25k) → + * window advanced to 25k–50k → 34 (E1–E6 found; see {@code doc/fuzz/2026-06-29.md}). */ - private static final int KNOWN_FINDINGS_BUDGET = 43; + private static final int KNOWN_FINDINGS_BUDGET = 34; @Test @Timeout(value = 300, unit = TimeUnit.SECONDS) @@ -189,8 +185,8 @@ private void runDivergenceGate(FuzzRunner.Config cfg, String tag, int maxFinding int totalChecks = cfg.patternCount * cfg.inputsPerPattern; assertTrue( - totalChecks >= 100_000, - "gate must sweep at least 100k checks; configured for " + totalChecks); + totalChecks >= 50_000, + "gate must sweep at least 50k checks; configured for " + totalChecks); List repros = shrinkAndDedupe(report); for (Shrunk s : repros) { @@ -232,8 +228,8 @@ private void runDivergenceGate(FuzzRunner.Config cfg, String tag, int maxFinding static FuzzRunner.Config largeSweepConfig() { FuzzRunner.Config cfg = new FuzzRunner.Config(); cfg.seed = BASE_SEED; - cfg.patternCount = sizedPatternCount(50_000); - cfg.patternSkip = intProp("reggie.fuzz.skip", 0); + cfg.patternCount = sizedPatternCount(25_000); + cfg.patternSkip = intProp("reggie.fuzz.skip", 25_000); cfg.inputsPerPattern = intProp("reggie.fuzz.inputsPerPattern", 16); cfg.patternDepth = intProp("reggie.fuzz.patternDepth", 3); cfg.inputMaxLength = intProp("reggie.fuzz.inputMaxLength", 16); From efb797ecaf7c73cc09a8c1a587e9c88296f97d64 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 17:38:09 +0200 Subject: [PATCH 23/27] docs: rewrite fuzz/2026-06-29.md with canonical skip=25k findings (34 raw, 29 repros) Co-Authored-By: Claude Sonnet 4.6 --- doc/fuzz/2026-06-29.md | 151 ++++++++++++++++++----------------------- 1 file changed, 66 insertions(+), 85 deletions(-) diff --git a/doc/fuzz/2026-06-29.md b/doc/fuzz/2026-06-29.md index d47d6d7d..7ebb9b66 100644 --- a/doc/fuzz/2026-06-29.md +++ b/doc/fuzz/2026-06-29.md @@ -1,32 +1,32 @@ # Fuzz Sweep — 2026-06-29 **Seed:** `0xC0DEFEED_DEADBEEFL` (`BASE_SEED`) -**Range:** patterns 25 001–50 000 (skip=25 000, count=25 000) +**Range:** patterns 25 001–50 000 (`skip=25_000`, `count=25_000`) **Depth:** 3 · **Inputs/pattern:** 16 · **Input max length:** 16 -**Raw findings:** 34 (canonical, with skip=25 000) · **Unique minimal repros:** 22 +**Raw findings:** 34 · **Unique minimal repros:** 29 **Baseline (0–25 000):** 0 divergences (established after B3a/B3b/B4/B5/B6 fixes) -> **Note on count:** a one-shot 50k run (skip=0) produces 43 findings in this range because -> the input RNG advances by a different amount when the first 25k patterns are actually executed -> (compile-rejected patterns consume fewer inputs). With skip=25 000 the input RNG advances by a -> fixed 25k × 16 steps, so the inputs for this window differ slightly — 34 is the reproducible -> canonical count for the `(seed, skip=25000, count=25000)` configuration used by `divergenceGate`. +> **Note on skip semantics:** `skip=N` advances both the pattern RNG and the input RNG by +> `N × inputsPerPattern` steps before the main loop. This is conservative relative to a real +> 50k run (which advances the input RNG by fewer steps for compile-rejected patterns), so the +> inputs for this window differ from a full 0–50k run. A one-shot 50k run produces 43 raw +> findings over this range; with skip=25 000 the canonical count is 34. Findings are fully +> reproducible given the same `(seed, skip, count, inputsPerPattern, depth)` tuple. --- ## How to reproduce -Exact reproduction (hardcoded in `divergenceGate_extended`, budget=43): +Exact reproduction (hardcoded in `divergenceGate`, `KNOWN_FINDINGS_BUDGET=34`): ```bash cd ./gradlew :reggie-integration-tests:test \ - --tests "*.AlgorithmicFuzzTest.divergenceGate_extended" \ - -Dreggie.fuzz.extended=true \ + --tests "*.AlgorithmicFuzzTest.divergenceGate" \ --no-daemon ``` -Manual equivalent with explicit parameters (useful after future fixes change the budget): +Manual equivalent with explicit parameters: ```bash ./gradlew :reggie-integration-tests:test \ @@ -37,7 +37,7 @@ Manual equivalent with explicit parameters (useful after future fixes change the --no-daemon ``` -Continue from where this sweep left off (patterns 50 001–75 000): +Advance to the next window once this budget reaches 0 (patterns 50 001–75 000): ```bash ./gradlew :reggie-integration-tests:test \ @@ -48,86 +48,68 @@ Continue from where this sweep left off (patterns 50 001–75 000): --no-daemon ``` -**Note on `patternSkip` semantics:** `skip=N` advances both the pattern RNG and the input RNG -by `N × inputsPerPattern` steps before the main loop. This is conservative — a real run that -executes the first N patterns would advance the input RNG by fewer steps for any compile-rejected -patterns. The inputs for patterns in the active window therefore differ slightly from what a -single 50k run would use at those positions, but findings are fully reproducible given the same -`(seed, skip, count, inputsPerPattern, depth)` tuple. +Then document new findings in `doc/fuzz/YYYY-MM-DD.md`, update `largeSweepConfig` default skip +to 50 000, and set `KNOWN_FINDINGS_BUDGET` to the new count. --- ## Summary -The standard `divergenceGate` (25k patterns, depth 3) reaches 0 after the B3a–B6 routing fixes. -Extending to patterns 25 001–50 000 surfaces 43 findings in 22 unique patterns. None are -regressions — all are pre-existing execution or routing bugs not exercised by the first 25k. - -Findings cluster into six root-cause classes (E1–E6). +29 unique minimal repros across six root-cause classes (E1–E6). All are pre-existing bugs in +native strategies — none are regressions introduced by the B3a/B3b/B4/B5/B6 routing fixes. --- -## E1 — Find-path group span overextension (6 patterns, 8 findings) +## E1 — Find-path group span overextension (2 patterns, 3 findings) -A capturing group containing a greedy quantifier is followed by a suffix. On the `find()` / -`findAll()` path the group-end tag lands after the suffix is consumed instead of before it, so -the reported span is wider than the JDK's. Same root-cause class as A1/A2; those fixes covered -`matches()` and the first `find()`. These patterns hit subsequent `findAll()` iterations and -`match()` with a non-trivial leading prefix. +A quantified prefix is followed by a capturing group. On the `findAll()` path the group span +extends beyond the chars it should own. Same root-cause class as A1/A2; those fixes addressed +`matches()` and the first `find()`. These patterns exercise subsequent `findAll()` iterations. | Pattern | Input | Symptom | |---------|-------|---------| -| `(_.*)- ` | `_--` | `findAll() match 0 group 1 span differs` | -| `([10][-b]+)[^0]` | `1-c0bb` | `findAll() match 1 group 1 span differs` | -| `.+([0].)` | `b00-` | `findAll() match 0 group 1 span differs` | -| `(cb?b*)[a-c]` | `cbc` | `findAll() match 0 group 1 span differs` | -| `(cb?b*)[a-c]` | `cba` | `findAll() match 0 group 1 span differs` | -| `([c][^-]{3,}){1}[0-c]` | `c0bc_c` | `findAll() match 0 group 1 span differs` | -| `.{1,2}(10)_?` | `-10` | `match() group 1 span differs` | -| `.{1,2}(10)_?` | `-10` | `findAll() match 0 group 1 span differs` | +| `[^-]{3,}([b].)` | `0acbb1` | `findAll() match 0 group 1 span differs` | +| `[^-]{3,}([b].)` | `c1cb-` | `findAll() match 0 group 1 span differs` | +| `[^-]{3,}([c].)` | `a0bc-` | `findAll() match 0 group 1 span differs` | --- -## E2 — Anchor at unusual position (6 patterns, 10 findings) +## E2 — Anchor at unusual position (4 patterns, 9 findings) -An anchor (`^`, `$`, `\A`, `\Z`) appears inside a group body, after a quantified group, or -combined with alternation + backreferences. The routing logic does not cover all these forms and -assigns a strategy that handles the anchor incorrectly. +An anchor appears inside a group body, after a quantified group, or combined with alternation +and a backreference. The routing logic does not cover all these forms. ### E2a — `\A` + repeated capturing group -`\A` with `{n,}` on a capturing group. The strategy treats `\A` as handled but the -repeated-group span bookkeeping conflicts with the start-anchor assertion. +`\A` with `{n,}` on a capturing group. The strategy treats `\A` as handled but repeated-group +span bookkeeping conflicts with the start-anchor assertion. | Pattern | Input | Symptom | |---------|-------|---------| -| `\A(c){1,}` | `0c` | `find() boolean differs` | +| `\A(c){1,}` | `ac` | `find() boolean differs` | | `\A(c){1,}` | `ac` | `findAll() count differs` | -| `\A(c){1,}` | `` | `find() boolean differs` | +| `\A(c){1,}` | `0c` | `find() boolean differs` | ### E2b — Quantified charset + `\Z`, no capturing group The outer match span (group 0) is wrong — a match-boundary error, not a capture-tracking issue. -`hasStringEndAnchorInAlternation` does not apply (no alternation present); a separate `\Z`-after- -quantifier guard is needed. +`hasStringEndAnchorInAlternation` does not apply (no alternation); a separate guard for `\Z` +after a plain quantifier is needed. | Pattern | Input | Symptom | |---------|-------|---------| -| `[^-]{3,}\Z` | `c` | `first-match span differs` | -| `[^-]{3,}\Z` | `c` | `findAll() match 0 group 0 span differs` | | `[^c]*\Z` | `` | `first-match span differs` | -| `[^c]*\Z` | `` | `findAll() count differs` | +| `[^c]*\Z` | `a` | `findAll() count differs` | -### E2c — `^` inside or immediately after a group body +### E2c — `^` after a quantified group -`^` appears as a zero-width assertion inside or directly after a capturing group body. Not routed -to a strategy that handles mid-group anchors. +`^` appears as a zero-width assertion immediately after a quantified capturing group. Not routed +to a strategy that handles post-quantifier anchors. | Pattern | Input | Symptom | |---------|-------|---------| | `(0)+^` | `0` | `find() boolean differs` | | `(0)+^` | `0` | `findAll() count differs` | -| `(a+.{3,6}^)` | `a0-0` | `match() boolean differs` | ### E2d — Anchor inside alternation combined with backreference @@ -135,69 +117,67 @@ The routing logic handles `\A`-in-alternation but not when a backreference is al | Pattern | Input | Symptom | |---------|-------|---------| -| `(c\|a?){3}\A\1?` | `a` | `find() boolean differs` | -| `(c\|a?){3}\A\1?` | `a` | `findAll() count differs` | +| `(c\|a?){3}\A\1?` | `c` | `find() boolean differs` | +| `(c\|a?){3}\A\1?` | `c` | `findAll() count differs` | --- -## E3 — Backreference divergence (5 patterns, 10 findings) +## E3 — Backreference divergence (5 patterns, 9 findings) -The backreference resolves to a wrong value or the match boolean is wrong. Distinct from E1 -(group span error): here the whole match succeeds or fails where the JDK says otherwise. +The backreference resolves to a wrong value or the match boolean is wrong. Distinct from E1: +the entire match succeeds or fails where the JDK says otherwise. | Pattern | Input | Symptom | |---------|-------|---------| -| `(.b{0})?\1` | `--` | `find() boolean differs` | +| `(.b{0})?\1` | `00` | `find() boolean differs` | | `b(-)\1{1}` | `--` | `find() boolean differs` | | `b(-)\1{1}` | `--` | `findAll() count differs` | | `${1}[^a]` | `` | `find() boolean differs` | | `${1}[^a]` | `` | `findAll() count differs` | -| `(b{1,}){1}\1+\|(1)` | `bb` | `find() boolean differs` | | `(b{1,}){1}\1+\|(])` | `bb` | `findAll() count differs` | -| `^(.{1}\|.{0}){4}\1{3}` | `-1_a` | `find() boolean differs` | -| `^(.{1}\|.{0}){4}\1{3}` | `-1_a` | `findAll() count differs` | -| `^(.{1}\|.{0}){4}\1{3}` | `_1cc` | `find() boolean differs` | +| `^(.{1}\|.{0}){4}\1{3}` | `1cb0` | `find() boolean differs` | +| `^(.{1}\|.{0}){4}\1{3}` | `1cb0` | `findAll() count differs` | +| `^(.{1}\|.{0}){4}\1{3}` | `00_1` | `find() boolean differs` | --- ## E4 — Repeated group last-iteration span (2 patterns, 3 findings) -After a group is matched multiple times via `*` or `{n,}`, the final captured span is wrong on -a subsequent `findAll()` call. The per-iteration span-reset logic does not fire correctly for -the last iteration before the quantifier exits. +After a group is matched multiple times the final captured span is wrong on `findAll()`. +The per-iteration span-reset logic does not fire correctly for the last iteration before the +quantifier exits. | Pattern | Input | Symptom | |---------|-------|---------| -| `(-+)*` | `` | `findAll() match 4 group 1 span differs` | +| `(-+)*` | `1` | `findAll() match 4 group 1 span differs` | | `(c{2}){1,}` | `c` | `find() boolean differs` | | `(c{2}){1,}` | `c` | `findAll() count differs` | --- -## E5 — Alternation with mixed-width branches (2 patterns, 2 findings) +## E5 — Alternation with mixed-width branches (1 pattern, 2 findings) -Complex alternation where branches have different widths and the outer match span (group 0) is -computed incorrectly. Distinct from B5 (variable-length body inside a capturing group): here the -outer span is wrong, not an inner group's span. +Complex alternation where branches have different widths; `find()` boolean and `findAll()` count +are both wrong. Distinct from B5 (variable-length body inside a capturing group): here the outer +match boolean is wrong, not an inner span. | Pattern | Input | Symptom | |---------|-------|---------| -| `(-{0}0[]\|[0]*b{1})(]\|1*).{2}` | `b0a` | `first-match span differs` | -| `(-{0}[]{1})(]\|1*).{2}` | `0a` | `findAll() count differs` | +| `(-{0}[1]{1})([_][^_]\|\1*).{2}` | `1_0` | `find() boolean differs` | +| `(-{0}[1]{1})([_][^_]\|\1*).{2}` | `1_0` | `findAll() count differs` | --- ## E6 — Simple group routing gap (1 pattern, 3 findings) -A straightforward pattern routes to a strategy that produces wrong `find()` boolean results. -Likely a missing routing guard rather than an execution bug; use `debugPattern` to confirm the -assigned strategy. +A straightforward pattern routes to a strategy that produces wrong `find()` results. Likely a +missing routing guard; use `debugPattern` to confirm the assigned strategy. | Pattern | Input | Symptom | |---------|-------|---------| -| `[1]([^b]{2})` | `1a-` | `find() boolean differs` | -| `[1]([^b]{2})` | `1a-` | `findAll() count differs` | -| `[1]([^b]{2})` | `10_` | `find() boolean differs` | +| `[1]([^b]{2})` | `110` | `find() boolean differs` | +| `[1]([^b]{2})` | `110` | `findAll() count differs` | +| `[1]([^b]{2})` | `10c` | `find() boolean differs` | --- @@ -205,11 +185,12 @@ assigned strategy. 1. **E2b** — add a `\Z`-after-quantifier routing guard (no alternation required); companion to `hasStringEndAnchorInAlternation`. -2. **E6** — run `debugPattern '[1]([^b]{2})'` to identify the misrouted strategy; likely a - one-line guard fix. -3. **E1** — find-path group span: same root-cause as A1/A2 but on `findAll()` and `match()` - entry points; investigate TDFA group-end tag placement on the find path. +2. **E6** — run `./gradlew :reggie-runtime:debugPattern -Ppattern='[1]([^b]{2})'` to identify + the misrouted strategy; likely a one-line guard fix. +3. **E1** — find-path group span: remaining cases after A1/A2; same root-cause, new trigger + shapes (`[^-]{3,}([b].)` — quantified non-capturing prefix followed by a capturing group). 4. **E2a/E2c** — `\A`/`^` + repeated group: new anchor-in-repetition routing class. 5. **E3** — backreference edge cases: optional-group backref, quantified backref in alternation. 6. **E4** — repeated group final-iteration span: per-iteration span-reset in TDFA. -7. **E5** — mixed-width alternation outer span. +7. **E5** — mixed-width alternation outer match: `find()` boolean wrong on alternation with + backreference branch. From 4142fd5ebd9dcb6fcc42a4f2b6ba94ad4f56df9e Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 23:12:07 +0200 Subject: [PATCH 24/27] fix: allow skip=0 in divergenceGate via intPropNonNeg --- .../integration/AlgorithmicFuzzTest.java | 21 +++++++++++++++++-- 1 file changed, 19 insertions(+), 2 deletions(-) diff --git a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java index 2d2c0915..affcd92b 100644 --- a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java +++ b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTest.java @@ -219,7 +219,8 @@ private void runDivergenceGate(FuzzRunner.Config cfg, String tag, int maxFinding * *

      *
    • {@code -Dreggie.fuzz.size=N} — pattern count (default 25_000) - *
    • {@code -Dreggie.fuzz.skip=N} — patterns to skip at the start of the sequence (default 0) + *
    • {@code -Dreggie.fuzz.skip=N} — patterns to skip at the start of the sequence (default + * 25_000; use 0 to rerun from the beginning of the sequence) *
    • {@code -Dreggie.fuzz.inputsPerPattern=N} — inputs per pattern (default 16) *
    • {@code -Dreggie.fuzz.inputMaxLength=N} — max input string length (default 16) *
    • {@code -Dreggie.fuzz.patternDepth=N} — max regex AST depth (default 3) @@ -229,7 +230,7 @@ static FuzzRunner.Config largeSweepConfig() { FuzzRunner.Config cfg = new FuzzRunner.Config(); cfg.seed = BASE_SEED; cfg.patternCount = sizedPatternCount(25_000); - cfg.patternSkip = intProp("reggie.fuzz.skip", 25_000); + cfg.patternSkip = intPropNonNeg("reggie.fuzz.skip", 25_000); cfg.inputsPerPattern = intProp("reggie.fuzz.inputsPerPattern", 16); cfg.patternDepth = intProp("reggie.fuzz.patternDepth", 3); cfg.inputMaxLength = intProp("reggie.fuzz.inputMaxLength", 16); @@ -248,6 +249,22 @@ private static int intProp(String name, int dflt) { } } + /** + * Read an int system property, returning {@code dflt} when absent or unparseable. Unlike {@link + * #intProp}, this helper accepts zero as a valid value; only negative values fall back to the + * default. + */ + private static int intPropNonNeg(String name, int dflt) { + String v = System.getProperty(name); + if (v == null || v.isEmpty()) return dflt; + try { + int parsed = Integer.parseInt(v); + return parsed >= 0 ? parsed : dflt; + } catch (NumberFormatException e) { + return dflt; + } + } + /** * Shrink every finding to a minimal repro and dedupe by (kind, pattern, input). Deterministic * across runs with the same seed. From e32904ce15d520bb0ece97bfbd51758116ca1800 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 29 Jun 2026 23:39:43 +0200 Subject: [PATCH 25/27] fix: unwrap transparent groups in B4/B5 detection --- .../analysis/FallbackPatternDetector.java | 41 ++++++++++++++----- .../reggie/runtime/PikeVMRoutingTest.java | 30 ++++++++++++++ 2 files changed, 61 insertions(+), 10 deletions(-) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java index d6eda837..3c4b8d0e 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/FallbackPatternDetector.java @@ -1573,19 +1573,36 @@ private static boolean isAnchorOnlyBody(RegexNode node) { return false; } + /** + * Peels any number of non-capturing {@link GroupNode} wrappers from {@code node} and returns the + * innermost node that is not a non-capturing group. Capturing groups and all other node types are + * returned as-is. + */ + private static RegexNode unwrapNonCapturing(RegexNode node) { + while (node instanceof GroupNode g && !g.capturing) { + node = g.child; + } + return node; + } + /** * Returns true if the top-level concat contains a capturing group whose body is a greedy * quantifier with min≥1, infinite max, and a broad charset (.+ or .+[DOTALL]), followed by at * least one more node (the suffix). The TDFA extends the group-end tag into the suffix for these * patterns, producing wrong capture spans. + * + *

      Non-capturing wrappers around the capturing group (e.g. {@code (?:(.+))_}) or around the + * quantifier body inside the capturing group (e.g. {@code ((?:.+))_}) are transparently unwrapped + * before the check, so both forms are correctly detected. */ public static boolean hasGreedyDotPlusGroupWithSuffix(RegexNode ast) { if (!(ast instanceof ConcatNode concat)) return false; List children = concat.children; for (int i = 0; i < children.size() - 1; i++) { - RegexNode node = children.get(i); + RegexNode node = unwrapNonCapturing(children.get(i)); if (!(node instanceof GroupNode g) || !g.capturing) continue; - if (!(g.child instanceof QuantifierNode q)) continue; + RegexNode body = unwrapNonCapturing(g.child); + if (!(body instanceof QuantifierNode q)) continue; if (q.min < 1 || (q.max != Integer.MAX_VALUE && q.max != -1)) continue; if (!(q.child instanceof CharClassNode cc)) continue; CharSet cs = cc.chars; @@ -1677,21 +1694,25 @@ private static boolean containsBroadCharClass(RegexNode node) { } /** - * Returns true if any capturing {@link GroupNode} in {@code ast} has a child that is an {@link - * AlternationNode} whose branches differ in minimum or maximum length AND at least one branch - * contains a broad charset (a {@link CharClassNode} matching more than one character). Pure- - * literal variable-length alternations like {@code (aa|a)} are handled correctly by the TDFA - * priority-cut and do not require PIKEVM routing. The broad-charset condition ensures only - * patterns where the TDFA cannot deterministically assign the group-end tag are routed to - * PIKEVM_CAPTURE. PikeVM gives correct spans for all cases. + * Returns true if any capturing {@link GroupNode} in {@code ast} has a body that, after + * unwrapping transparent non-capturing wrappers, is an {@link AlternationNode} whose branches + * differ in minimum or maximum length AND at least one branch contains a broad charset (a {@link + * CharClassNode} matching more than one character). Pure-literal variable-length alternations + * like {@code (aa|a)} are handled correctly by the TDFA priority-cut and do not require PIKEVM + * routing. The broad-charset condition ensures only patterns where the TDFA cannot + * deterministically assign the group-end tag are routed to PIKEVM_CAPTURE. PikeVM gives correct + * spans for all cases. * *

      Example: {@code ([1]|1.)} — branch {@code [1]} has length 1, branch {@code 1.} has length 2, * and {@code 1.} contains {@code .} (broad charset) → variable-length with broad charset → route * to PIKEVM_CAPTURE. + * + *

      Non-capturing wrappers around the alternation body (e.g. {@code ((?:[1]|1.))}) are + * transparently unwrapped before the alternation check so that wrapped forms are also detected. */ public static boolean hasGroupWithVariableLengthAlternationBody(RegexNode ast) { if (ast instanceof GroupNode g && g.groupNumber > 0) { - if (g.child instanceof AlternationNode a) { + if (unwrapNonCapturing(g.child) instanceof AlternationNode a) { List alts = a.alternatives; if (alts.size() >= 2) { int firstMin = minLength(alts.get(0)); diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java index 5b91bf28..e524355b 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/PikeVMRoutingTest.java @@ -272,6 +272,26 @@ void greedyDotPlusWithSuffix_routesToPikevm() throws Exception { "(.+)_ must route to PIKEVM_CAPTURE (B4: greedy .+ with suffix)"); } + @Test + void greedyDotPlusWrappedInNonCapturing_routesToPikevm() throws Exception { + // (?:(.+))_ — the capturing group is wrapped inside a non-capturing group at the concat level. + // B4 detection must unwrap the non-capturing outer group to find the (.+) capture. + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("(?:(.+))_"), + "(?:(.+))_ must route to PIKEVM_CAPTURE (B4: non-capturing wrapper around capture)"); + } + + @Test + void greedyDotPlusBodyWrappedInNonCapturing_routesToPikevm() throws Exception { + // ((?:.+))_ — the .+ quantifier is wrapped inside a non-capturing group inside the capture. + // B4 detection must also unwrap the non-capturing inner group to find the quantifier. + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("((?:.+))_"), + "((?:.+))_ must route to PIKEVM_CAPTURE (B4: non-capturing wrapper around quantifier body)"); + } + @Test void greedyDotStarWithSuffix_staysOnGreedyBacktrack() throws Exception { // (.*)_ has a greedy .* group (min=0): not affected by the B4 decline (min>=1 only). @@ -292,6 +312,16 @@ void variableLengthAltInGroup_routesToPikevm() throws Exception { "([1]|1.)[b]_ must route to PIKEVM_CAPTURE (B5: variable-length alt in group)"); } + @Test + void variableLengthAltWrappedInNonCapturing_routesToPikevm() throws Exception { + // ((?:[1]|1.))[b]_ — the alternation is wrapped in a non-capturing group inside the capture. + // B5 detection must unwrap the non-capturing wrapper before checking for AlternationNode. + assertEquals( + PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE, + StrategyCorrectnessMetaTest.routeOf("((?:[1]|1.))[b]_"), + "((?:[1]|1.))[b]_ must route to PIKEVM_CAPTURE (B5: non-capturing wrapper around alt)"); + } + @Test void fixedLengthAltInGroup_staysOnDfa() throws Exception { // ([a]|[b])c: both alternatives have length 1 — fixed-length, should NOT trigger B5. From 716d6ea5d1d160da58bf33a16730cd76bc1c202e Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Tue, 30 Jun 2026 00:08:38 +0200 Subject: [PATCH 26/27] fix: preserve fallback for quantified anchor-only groups in PIKEVM route --- ...rve-fallback-for-pikevm-anchor-quantifi.md | 73 +++++++++++++++++++ .../AlgorithmicFuzzTestConfigTest.java | 64 ++++++++++++++++ .../ReggieMatcherBytecodeGenerator.java | 15 ++++ .../ReggieMatcherBytecodeGeneratorTest.java | 35 +++++++++ .../reggie/runtime/RuntimeCompiler.java | 10 +-- .../reggie/runtime/CompilePikeVmTest.java | 25 +++++++ 6 files changed, 217 insertions(+), 5 deletions(-) create mode 100644 docs/sphinx/specs/2026-06-30-preserve-fallback-for-pikevm-anchor-quantifi.md create mode 100644 reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTestConfigTest.java diff --git a/docs/sphinx/specs/2026-06-30-preserve-fallback-for-pikevm-anchor-quantifi.md b/docs/sphinx/specs/2026-06-30-preserve-fallback-for-pikevm-anchor-quantifi.md new file mode 100644 index 00000000..79e26509 --- /dev/null +++ b/docs/sphinx/specs/2026-06-30-preserve-fallback-for-pikevm-anchor-quantifi.md @@ -0,0 +1,73 @@ +# Spec: preserve-fallback-for-pikevm-anchor-quantifi + +## Problem + +`ReggieMatcherBytecodeGenerator.resolveRealization()` returns `DELEGATE_PIKEVM` for any +pattern whose `PatternAnalyzer` result is `PIKEVM_CAPTURE` **without first consulting +`FallbackPatternDetector.needsFallback()`**. This means patterns that trigger the B3a route +in `PatternAnalyzer` (anchor-only capturing group, e.g. `($)`) but which also carry a +quantifier on that group (e.g. `($){2}`, `(^)?`) reach the compile-time PIKEVM path even +though `FallbackPatternDetector` marks them unsafe via the B2 +(`hasAnchorInQuantifierInCapturingGroup`) and B3 (`hasAnchorInQuantifier`) KEEP-PERMANENT +guards. + +The same gap exists in `RuntimeCompiler.compilePikeVm()` (called by generated stubs): it +only applies the B16 nullable-group-content guard, missing B2/B3. + +The runtime `Reggie.compile()` path is correct: it calls +`FallbackPatternDetector.needsFallback(ast, PIKEVM_CAPTURE)` at line 494 and falls back to +JDK when non-null. + +## Correct behaviour + +1. When `@RegexPattern` annotation processing routes a pattern to `PIKEVM_CAPTURE`: + - `resolveRealization()` must call `FallbackPatternDetector.needsFallback(ast, PIKEVM_CAPTURE)`. + - If the result is non-null, the pattern requires JDK fallback: + - If `allowJdkFallback` is set → return `DELEGATE_FALLBACK`. + - Otherwise → throw `UnsupportedOperationException` with the fallback reason, consistent + with the existing `needsJdk` error message format. + - If the result is null → return `DELEGATE_PIKEVM` as before. + +2. `RuntimeCompiler.compilePikeVm()` must apply the full `needsFallback()` guard (not just + the ad-hoc `hasNullableGroupContentWithNullableQuantifier` check). If the result is + non-null → throw `UnsupportedPatternException` with the reason string. + +3. Patterns like `($)` (no quantifier on the anchor-only group) remain unaffected: + `hasAnchorInQuantifier` returns false for them → `needsFallback()` returns null → + they continue to reach `DELEGATE_PIKEVM`. + +## Constraints + +- No changes to `FallbackPatternDetector` or `PatternAnalyzer`. +- Error message format in `resolveRealization()` must be consistent with the existing + `needsJdk` path (folding the PIKEVM fallback reason into the `needsJdk` boolean or + reusing the same throw is preferred). +- `FallbackPatternDetector` is already imported in both files — no new dependencies. + +## Scope + +### Primary fixes + +- `reggie-processor/src/main/java/com/datadoghq/reggie/processor/ReggieMatcherBytecodeGenerator.java:103` + — missing fallback guard on PIKEVM_CAPTURE early-exit: add `needsFallback()` check + before returning `DELEGATE_PIKEVM`. + +### Auto-expanded sibling fixes + +- `reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java:299` + — partial guard (B16 only) in `compilePikeVm()`: replace `hasNullableGroupContentWithNullableQuantifier` + check with full `FallbackPatternDetector.needsFallback(ast, MatchingStrategy.PIKEVM_CAPTURE)` call. + FLOW: annotation-processor-generated stub calls `compilePikeVm()` directly, bypassing + `RuntimeCompiler.compile()` which has the correct full guard. + PRECONDITION: pattern has PIKEVM_CAPTURE result from PatternAnalyzer AND triggers B2 or B3. + REACHABLE: yes — `($){2}` at an `@RegexPattern` site reaches `compilePikeVm()` with no B2/B3 check. + CONCLUSION: medium-severity defensive guard that catches patterns slipping through Fix 1 if any + future code path creates a PIKEVM stub without going through `resolveRealization()`. + +## Assumptions + +- `FallbackPatternDetector.needsFallback(ast, PIKEVM_CAPTURE)` subsumes + `hasNullableGroupContentWithNullableQuantifier` (confirmed: B16 guard at + `FallbackPatternDetector.java:287` includes `strategy == PIKEVM_CAPTURE`). +- Fix 1 folds the PIKEVM fallback check into the existing `needsJdk` boolean rather than + adding a separate throw, to keep uniform error message formatting. diff --git a/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTestConfigTest.java b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTestConfigTest.java new file mode 100644 index 00000000..e13cacf3 --- /dev/null +++ b/reggie-integration-tests/src/test/java/com/datadoghq/reggie/integration/AlgorithmicFuzzTestConfigTest.java @@ -0,0 +1,64 @@ +/* + * Copyright 2026-Present Datadog, Inc. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package com.datadoghq.reggie.integration; + +import static org.junit.jupiter.api.Assertions.assertEquals; + +import com.datadoghq.reggie.integration.fuzz.FuzzRunner; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +/** Unit tests for {@link AlgorithmicFuzzTest} configuration helpers. */ +class AlgorithmicFuzzTestConfigTest { + + @BeforeEach + void clearProps() { + System.clearProperty("reggie.fuzz.skip"); + } + + @AfterEach + void restoreProps() { + System.clearProperty("reggie.fuzz.skip"); + } + + @Test + void skipZeroIsAccepted() { + System.setProperty("reggie.fuzz.skip", "0"); + FuzzRunner.Config cfg = AlgorithmicFuzzTest.largeSweepConfig(); + assertEquals(0, cfg.patternSkip, "-Dreggie.fuzz.skip=0 must set patternSkip to 0"); + } + + @Test + void skipDefaultWhenAbsent() { + FuzzRunner.Config cfg = AlgorithmicFuzzTest.largeSweepConfig(); + assertEquals(25_000, cfg.patternSkip, "patternSkip default must be 25_000"); + } + + @Test + void skipPositiveIsAccepted() { + System.setProperty("reggie.fuzz.skip", "50000"); + FuzzRunner.Config cfg = AlgorithmicFuzzTest.largeSweepConfig(); + assertEquals(50_000, cfg.patternSkip, "-Dreggie.fuzz.skip=50000 must set patternSkip to 50000"); + } + + @Test + void skipNegativeFallsBackToDefault() { + System.setProperty("reggie.fuzz.skip", "-1"); + FuzzRunner.Config cfg = AlgorithmicFuzzTest.largeSweepConfig(); + assertEquals(25_000, cfg.patternSkip, "-Dreggie.fuzz.skip=-1 must fall back to default 25_000"); + } +} diff --git a/reggie-processor/src/main/java/com/datadoghq/reggie/processor/ReggieMatcherBytecodeGenerator.java b/reggie-processor/src/main/java/com/datadoghq/reggie/processor/ReggieMatcherBytecodeGenerator.java index 60c98211..3c335097 100644 --- a/reggie-processor/src/main/java/com/datadoghq/reggie/processor/ReggieMatcherBytecodeGenerator.java +++ b/reggie-processor/src/main/java/com/datadoghq/reggie/processor/ReggieMatcherBytecodeGenerator.java @@ -101,6 +101,21 @@ public Realization resolveRealization(boolean allowJdkFallback) throws Exception this.resolvedStrategy = result.strategy; if (result.strategy == PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE) { + String pikeVmFallbackReason = FallbackPatternDetector.needsFallback(ast, result.strategy); + if (pikeVmFallbackReason != null) { + if (allowJdkFallback) { + return Realization.DELEGATE_FALLBACK; + } + throw new UnsupportedOperationException( + "Pattern '" + + pattern + + "' requires java.util.regex fallback (strategy " + + result.strategy + + "): " + + pikeVmFallbackReason + + ". Add options = ReggieOption.ALLOW_JDK_FALLBACK to @RegexPattern to permit a" + + " delegating stub, or use Reggie.compile() at runtime."); + } return Realization.DELEGATE_PIKEVM; } boolean needsJdk = diff --git a/reggie-processor/src/test/java/com/datadoghq/reggie/processor/ReggieMatcherBytecodeGeneratorTest.java b/reggie-processor/src/test/java/com/datadoghq/reggie/processor/ReggieMatcherBytecodeGeneratorTest.java index cb0523bf..d2ffe760 100644 --- a/reggie-processor/src/test/java/com/datadoghq/reggie/processor/ReggieMatcherBytecodeGeneratorTest.java +++ b/reggie-processor/src/test/java/com/datadoghq/reggie/processor/ReggieMatcherBytecodeGeneratorTest.java @@ -517,6 +517,41 @@ void twoProviderClassesWithSameMethodNameProduceDistinctMatchers() throws Except (Boolean) matchesB.invoke(matcherB, "123"), "ProviderB_ValueMatcher must not match digits"); } + // --- Tests for Fix 1: resolveRealization() must honour needsFallback() for PIKEVM_CAPTURE --- + + @Test + void quantifiedAnchorOnlyGroup_throwsWithoutFallback() throws Exception { + // ($){2} triggers hasAnchorInQuantifier (B3) — must not silently route to PIKEVM + ReggieMatcherBytecodeGenerator gen = + new ReggieMatcherBytecodeGenerator("test", "Cls", "($){2}"); + assertThrows(UnsupportedOperationException.class, () -> gen.resolveRealization(false)); + } + + @Test + void optionalAnchorOnlyGroup_throwsWithoutFallback() throws Exception { + // (^)? triggers hasAnchorInQuantifier (B3) + ReggieMatcherBytecodeGenerator gen = new ReggieMatcherBytecodeGenerator("test", "Cls", "(^)?"); + assertThrows(UnsupportedOperationException.class, () -> gen.resolveRealization(false)); + } + + @Test + void quantifiedAnchorOnlyGroup_returnsDelegateFallbackWhenAllowed() throws Exception { + // With ALLOW_JDK_FALLBACK, unsafe PIKEVM pattern must route to DELEGATE_FALLBACK + ReggieMatcherBytecodeGenerator gen = + new ReggieMatcherBytecodeGenerator("test", "Cls", "($){2}"); + ReggieMatcherBytecodeGenerator.Realization r = gen.resolveRealization(true); + assertEquals(ReggieMatcherBytecodeGenerator.Realization.DELEGATE_FALLBACK, r); + } + + @Test + void nonQuantifiedAnchorOnlyGroup_stillReturnsDelegatePikevm() throws Exception { + // ($) has anchor-only group but NO quantifier on the group → + // hasAnchorInQuantifier = false → needsFallback = null → DELEGATE_PIKEVM preserved + ReggieMatcherBytecodeGenerator gen = new ReggieMatcherBytecodeGenerator("test", "Cls", "($)"); + ReggieMatcherBytecodeGenerator.Realization r = gen.resolveRealization(false); + assertEquals(ReggieMatcherBytecodeGenerator.Realization.DELEGATE_PIKEVM, r); + } + /** Custom ClassLoader for loading generated bytecode in tests. */ private static class TestClassLoader extends ClassLoader { public Class defineClass(String name, byte[] bytecode) { diff --git a/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java b/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java index 03439be7..c57800a9 100644 --- a/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java +++ b/reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java @@ -296,12 +296,12 @@ public static ReggieMatcher compilePikeVm(String pattern, String encodedNames) { try { RegexParser parser = new RegexParser(); RegexNode ast = parser.parse(pattern); - if (FallbackPatternDetector.hasNullableGroupContentWithNullableQuantifier(ast)) { + String pikeVmFallbackReason = + FallbackPatternDetector.needsFallback( + ast, PatternAnalyzer.MatchingStrategy.PIKEVM_CAPTURE); + if (pikeVmFallbackReason != null) { throw new UnsupportedPatternException( - "capturing group with nullable content and nullable outer quantifier: " - + "PIKEVM_CAPTURE diverges in /" - + pattern - + "/"); + "PIKEVM_CAPTURE pattern is unsafe: " + pikeVmFallbackReason + " in /" + pattern + "/"); } Map nameMap = decodeNameMap(encodedNames); int groupCount = countGroups(pattern); diff --git a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/CompilePikeVmTest.java b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/CompilePikeVmTest.java index c0367ef6..a0d28b01 100644 --- a/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/CompilePikeVmTest.java +++ b/reggie-runtime/src/test/java/com/datadoghq/reggie/runtime/CompilePikeVmTest.java @@ -18,9 +18,11 @@ import static org.junit.jupiter.api.Assertions.assertEquals; import static org.junit.jupiter.api.Assertions.assertFalse; import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertThrows; import com.datadoghq.reggie.Reggie; import com.datadoghq.reggie.ReggieOptions; +import com.datadoghq.reggie.UnsupportedPatternException; import java.util.LinkedHashMap; import java.util.Map; import org.junit.jupiter.api.Test; @@ -59,6 +61,29 @@ void compilePikeVmMatchesRuntimePath() { assertFalse(staged instanceof JavaRegexFallbackMatcher); } + // --- Tests for Fix 2: compilePikeVm() must honour full needsFallback() guard --- + + @Test + void compilePikeVm_quantifiedAnchorGroup_throws() { + // ($){2} triggers B3 (hasAnchorInQuantifier) — compilePikeVm must reject it + assertThrows( + UnsupportedPatternException.class, () -> RuntimeCompiler.compilePikeVm("($){2}", "")); + } + + @Test + void compilePikeVm_optionalAnchorGroup_throws() { + // (^)? triggers B3 (hasAnchorInQuantifier) + assertThrows( + UnsupportedPatternException.class, () -> RuntimeCompiler.compilePikeVm("(^)?", "")); + } + + @Test + void compilePikeVm_nonQuantifiedAnchorGroup_succeeds() { + // ($) is safe for PikeVM: hasAnchorInQuantifier = false + String encoded = RuntimeCompiler.encodeNameMap(Map.of()); + assertNotNull(RuntimeCompiler.compilePikeVm("($)", encoded)); + } + @Test void compileAllowingFallbackWorks() { // A native pattern compiles cleanly From 0bb9205ec87b5e38d3b1266c8495d7f81532f437 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Tue, 30 Jun 2026 15:54:23 +0200 Subject: [PATCH 27/27] chore: remove duplicate comment in B6 branch Co-Authored-By: Claude Sonnet 4.6 --- .../com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java | 1 - 1 file changed, 1 deletion(-) diff --git a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java index 4807e5a4..5f7ad578 100644 --- a/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java +++ b/reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java @@ -708,7 +708,6 @@ private MatchingStrategyResult doAnalyze(boolean ignoreGroupCount) { requiredLiterals); } // B6: if suffix is non-empty, fall through to OPTIMIZED_NFA_WITH_BACKREFS below. - // B6: if suffix is non-empty, fall through to OPTIMIZED_NFA_WITH_BACKREFS below. // Try to detect variable-capture backreference patterns: (.*)\d+\1, (.+)=\1 // These require backtracking from longest to shortest capture