From 1aa964bbc07c19b31f3a0fd3a09fc3b7e0e28377 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Mon, 28 Sep 2026 12:37:52 +0200 Subject: [PATCH] cairn: install distilled knowledgebase from dd-backend-check investigations --- .knowledgebase/.cairn | 1 + .../dead-bitstate-lazy-fastconsume.md | 41 +++++++++ .../memories/dead-classwriter-subclass.md | 27 ++++++ .knowledgebase/memories/dead-env-gotchas.md | 52 +++++++++++ .../memories/dead-group-boundary-fusion.md | 36 ++++++++ .../memories/dead-handler-extent-heuristic.md | 29 ++++++ .../dead-real-generator-differential-test.md | 29 ++++++ .../memories/dead-sourceinterpreter-typing.md | 30 ++++++ .../memories/find-api-surface-sufficient.md | 35 +++++++ .../memories/find-asm-classwriter-final.md | 31 +++++++ .../find-asm-toobytearray-sizecheck.md | 28 ++++++ .../memories/find-backend-prod-regex-cost.md | 35 +++++++ .../memories/find-brace-literal-rejected.md | 27 ++++++ .../memories/find-case-insensitive-viable.md | 28 ++++++ .../find-computeextras-early-return-bug.md | 29 ++++++ .../memories/find-deadline-coverage-gaps.md | 47 ++++++++++ .../find-descriptor-interpreter-design.md | 39 ++++++++ .../memories/find-fallback-optin-refuse.md | 40 ++++++++ .../find-groupcount-charclass-parens.md | 25 +++++ .../memories/find-jit-hugemethodlimit.md | 52 +++++++++++ .../memories/find-l2-dead-code-verdict.md | 37 ++++++++ .../memories/find-l2-splitter-shipped.md | 55 +++++++++++ .../find-logs-backend-re2j-migration.md | 49 ++++++++++ .../memories/find-matchtime-heap-scaling.md | 39 ++++++++ .../memories/find-nul-pattern-truncation.md | 37 ++++++++ .../find-openj9-deadline-rejection.md | 33 +++++++ .../memories/find-re2j-dynamic-safety-gap.md | 33 +++++++ .../memories/find-refusal-set-parity.md | 47 ++++++++++ ...nd-remaining-incompatibilities-resolved.md | 78 ++++++++++++++++ .../find-repo-method-size-landscape.md | 34 +++++++ .../memories/find-rust-engine-crossover.md | 39 ++++++++ .../find-semver-exponential-compile.md | 35 +++++++ .../find-sourceinterpreter-pathologies.md | 35 +++++++ .../find-try-handler-chunk-constraint.md | 30 ++++++ .../find-unanchored-find-prefilter.md | 43 +++++++++ .../memories/find-unicode-property-gaps.md | 31 +++++++ .../memories/find-visitmaxs-zero-blindspot.md | 37 ++++++++ .../memories/hyp-bitstate-blowup-root.md | 35 +++++++ AGENTS.md | 91 +++++++++++++++++++ 39 files changed, 1479 insertions(+) create mode 100644 .knowledgebase/.cairn create mode 100644 .knowledgebase/memories/dead-bitstate-lazy-fastconsume.md create mode 100644 .knowledgebase/memories/dead-classwriter-subclass.md create mode 100644 .knowledgebase/memories/dead-env-gotchas.md create mode 100644 .knowledgebase/memories/dead-group-boundary-fusion.md create mode 100644 .knowledgebase/memories/dead-handler-extent-heuristic.md create mode 100644 .knowledgebase/memories/dead-real-generator-differential-test.md create mode 100644 .knowledgebase/memories/dead-sourceinterpreter-typing.md create mode 100644 .knowledgebase/memories/find-api-surface-sufficient.md create mode 100644 .knowledgebase/memories/find-asm-classwriter-final.md create mode 100644 .knowledgebase/memories/find-asm-toobytearray-sizecheck.md create mode 100644 .knowledgebase/memories/find-backend-prod-regex-cost.md create mode 100644 .knowledgebase/memories/find-brace-literal-rejected.md create mode 100644 .knowledgebase/memories/find-case-insensitive-viable.md create mode 100644 .knowledgebase/memories/find-computeextras-early-return-bug.md create mode 100644 .knowledgebase/memories/find-deadline-coverage-gaps.md create mode 100644 .knowledgebase/memories/find-descriptor-interpreter-design.md create mode 100644 .knowledgebase/memories/find-fallback-optin-refuse.md create mode 100644 .knowledgebase/memories/find-groupcount-charclass-parens.md create mode 100644 .knowledgebase/memories/find-jit-hugemethodlimit.md create mode 100644 .knowledgebase/memories/find-l2-dead-code-verdict.md create mode 100644 .knowledgebase/memories/find-l2-splitter-shipped.md create mode 100644 .knowledgebase/memories/find-logs-backend-re2j-migration.md create mode 100644 .knowledgebase/memories/find-matchtime-heap-scaling.md create mode 100644 .knowledgebase/memories/find-nul-pattern-truncation.md create mode 100644 .knowledgebase/memories/find-openj9-deadline-rejection.md create mode 100644 .knowledgebase/memories/find-re2j-dynamic-safety-gap.md create mode 100644 .knowledgebase/memories/find-refusal-set-parity.md create mode 100644 .knowledgebase/memories/find-remaining-incompatibilities-resolved.md create mode 100644 .knowledgebase/memories/find-repo-method-size-landscape.md create mode 100644 .knowledgebase/memories/find-rust-engine-crossover.md create mode 100644 .knowledgebase/memories/find-semver-exponential-compile.md create mode 100644 .knowledgebase/memories/find-sourceinterpreter-pathologies.md create mode 100644 .knowledgebase/memories/find-try-handler-chunk-constraint.md create mode 100644 .knowledgebase/memories/find-unanchored-find-prefilter.md create mode 100644 .knowledgebase/memories/find-unicode-property-gaps.md create mode 100644 .knowledgebase/memories/find-visitmaxs-zero-blindspot.md create mode 100644 .knowledgebase/memories/hyp-bitstate-blowup-root.md diff --git a/.knowledgebase/.cairn b/.knowledgebase/.cairn new file mode 100644 index 00000000..3e6a3b04 --- /dev/null +++ b/.knowledgebase/.cairn @@ -0,0 +1 @@ +{"generator": "cairn", "retired_threshold": 10} diff --git a/.knowledgebase/memories/dead-bitstate-lazy-fastconsume.md b/.knowledgebase/memories/dead-bitstate-lazy-fastconsume.md new file mode 100644 index 00000000..95298f3f --- /dev/null +++ b/.knowledgebase/memories/dead-bitstate-lazy-fastconsume.md @@ -0,0 +1,41 @@ +--- +id: dead-bitstate-lazy-fastconsume +title: BitState lazy char-class loop fast-consume measured flat on the real corpus — rolled back; lazy-loop consumption is not the lazy families' cost center +kind: dead-end +tags: [bitstate, lazy-loop, perf, measured-negative, rollback] +applies_to: [reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/BitStateMatcher.java] +source: backend-ready/dead-bitstate-lazy-fastconsume +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + +# BitState lazy char-class loop fast-consume measured flat on the real corpus — rolled back + +## What was tried +A lazy mirror of the existing greedy fast-consume path in BitStateMatcher: a lazy +single-char-class loop (`c*?`) whose continuation closure contains no anchor, no accepting +state, and at least one consuming transition gets a tight first pass — consume the whole +`c`-run in a scan, mark the same visited cells, push exit jobs only at positions where the +continuation's first-char set can match input[p] (descending push = ascending pop = Perl +lazy priority). Fully implemented and gate-verified (fuzz seeds, real-corpus parity probes +0 divergences, full suite). + +## Why it was rolled back +- Box JMH matched sweep: flat to slightly negative vs the pre-change build; controls flat. +- Per-pattern C2-converged profiling showed the path barely engages on the corpus lazy + families: with an all-optional tail (`\s*((?[^@]*)@)?(?.*?)(:?\d+)?...`) + the lazy loop exits EMPTY almost immediately (the zero-width guard correctly disables the + fast path) — its cost is unanchored-seed DFS overhead + greedy give-back, not loop + consumption. Where it does engage (narrow continuation first-set), loop consumption is a + small share of the pattern's total cost. +- Local single-JVM probe "wins" were noise. Rule reinforced: **local single-JVM probe + deltas are not perf evidence — box JMH + per-pattern C2-converged profiling only.** + +## What this rules out / redirects +- Lazy-loop stack round-trips are NOT the lazy families' cost center. +- The real lever: a priority-correct (lazy leftmost-first) AND fast captureless DFA find in + the LazyDFA/RD lane — the chain lane proves leftmost-first lazy is achievable for simple + shapes; the fast-consume design above is the template if revisited (greedy fast-consume + fields + ctor shape detection in BitStateMatcher). diff --git a/.knowledgebase/memories/dead-classwriter-subclass.md b/.knowledgebase/memories/dead-classwriter-subclass.md new file mode 100644 index 00000000..1ba2547a --- /dev/null +++ b/.knowledgebase/memories/dead-classwriter-subclass.md @@ -0,0 +1,27 @@ +--- +id: dead-classwriter-subclass +title: ASM ClassWriter.visit/visitMethod are final (9.10) — method emission cannot be intercepted via subclassing; wrap with a ClassVisitor instead +kind: dead-end +tags: [asm, classwriter, final, refuted-approach] +applies_to: [] +source: multi-64k/dead-classwriter-subclass +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# ClassWriter subclassing is impossible — wrap a ClassVisitor + +First design: `SplittingClassWriter extends ClassWriter`, override `visitMethod` to return a +buffering MethodVisitor. Died on compilation: ASM 9.10 declares `ClassWriter.visit` and +`visitMethod` `public final`. No interception is possible via subclassing. + +The interception point must be a **wrapping `ClassVisitor`** (the standard ASM +instrumentation shape) delegating to the real ClassWriter, which stays the terminal pipeline +stage and `toByteArray()` producer. Chunks are emitted directly on the real writer; the +splitter takes `(node, realWriter, chunk0Visitor)`. + +Any future ASM-pipeline work in this repo should start from a ClassVisitor wrapper, not a +ClassWriter subclass. diff --git a/.knowledgebase/memories/dead-env-gotchas.md b/.knowledgebase/memories/dead-env-gotchas.md new file mode 100644 index 00000000..73ea562e --- /dev/null +++ b/.knowledgebase/memories/dead-env-gotchas.md @@ -0,0 +1,52 @@ +--- +id: dead-env-gotchas +title: Build/test/harness gotchas for java-reggie and consumer-repo regex audits — stale-jar trap, gradle caching, piped-build chains, JDK-differential tests, JDK matcher statefulness (resolved — don't re-debug) +kind: dead-end +tags: [environment, git, gradle, classpath] +applies_to: [] +source: dd_backend_fit/dead-env-gotchas +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# Environment gotchas hit (resolved — don't re-debug) + +- Both consumer repos had stale `.git/index.lock` from crashed git processes → + pull failed with "an editor opened by git commit" message; fix: `rm .git/index.lock`. +- Large consumer repos (e.g. logs-backend): remote may carry dead refspecs + (plain fetch dies with "couldn't find remote ref …"; workaround + `git fetch origin && git merge --ff-only FETCH_HEAD`) and + filesystem-wide greps time out — ALWAYS `git ls-files '*.java' | xargs grep …`. +- reggie-runtime jar does NOT bundle ASM (declared `implementation`): any + standalone harness needs asm/asm-commons/asm-util 9.10.1 on the classpath + (from ~/.gradle/caches), else every compile fails with + NoClassDefFoundError MethodTooLargeException (misleading). +- Harness must emit results incrementally + per-entry timing; buffered output + + one hung pattern (semver) loses 15 min of work. + +Reggie build discipline (don't repeat): +- STALE-JAR TRAP: requesting only the runtime jar after editing reggie-codegen sources can + be an up-to-date NO-OP — the fat runtime jar embeds the codegen classes and did not + repackage. After codegen changes run :reggie-codegen:compileJava explicitly and + sanity-check via a changed generated-class hash. +- `./gradlew jar ... | grep BUILD` in an && chain does NOT stop on BUILD FAILED — grep + exits 0 on match and the following test/verify runs against the STALE jar. Check exit + codes explicitly; never pipe-build-then-run in one chain. +- Gradle test tasks are cached: `./gradlew test` returning BUILD SUCCESSFUL + in <1s means UP-TO-DATE, not re-run. For real verification use + `./gradlew cleanTest test`. +- git commit signing via ~/.ssh/datadog_git_commit_signing can fail + ("agent refused operation") until the key passphrase is cached; retry after + unlocking (user unlocked on demand 2026-09-16). +- Writing regression tests: do NOT hand-derive expectations (repeated + test-assertion bugs from misread semantics); write JDK-differential tests + (compute expected values from java.util.regex at runtime). +- JDK Matcher is STATEFUL: matcher.matches() then matcher.find() continues + from the end of the previous match — use a fresh Matcher per operation in + differential tests (ReggieMatcher is stateless per call). +- jstack sampling root-causes compile hangs; run the victim via nohup+disown + in one tool call, sample in the next (the tool kills the process group + when the command ends). diff --git a/.knowledgebase/memories/dead-group-boundary-fusion.md b/.knowledgebase/memories/dead-group-boundary-fusion.md new file mode 100644 index 00000000..41e8ab2d --- /dev/null +++ b/.knowledgebase/memories/dead-group-boundary-fusion.md @@ -0,0 +1,36 @@ +--- +id: dead-group-boundary-fusion +title: Group-boundary state fusion in the capture-tracking DFS is net-negative — group writes sit on loop edges off the hot path, while the extra ops array taxes every push/pop +kind: dead-end +tags: [fusion, group-boundary, dfs, capture, perf, measured-negative, reverted] +applies_to: [] +source: backend-ready/dead-group-boundary-fusion +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + +# Group-boundary state fusion in the capture-tracking DFS is net-negative + +## What was tried +Fusing group enter/exit-only states into incoming edges so capture writes ride on edges +instead of dedicated marker frames. Two variants built and fully tested: +- EAGER: ops applied at push — pays ~150 wasted capture clones for never-popped + marker-stop frames on greedy paths. +- DEFERRED: ops row on the frame, applied at pop (sound: push pos == pop pos of the + bypassed state); post-write dedup strictly better. + +## Why it was refuted (measured) +- Realistic 313-char accept shape: +1.4–2.4us; reject shape: +6.2–7.8us; only degenerate + 5-char inputs win ~0.1us. +- Instrumentation: fusion removes only ~10 pops on the 313-char accept — the pre-release + loop body has NO boundary states (group writes sit on loop EDGES, off the winning path) — + while the extra ops array taxes every push/pop (~0.9ns x ~1700 frames); reject paths pay + on 5–10x more frames. +- State-id bit-packing of ops (~0.3–0.5ns/pop) also estimated net-negative for the same + reason. + +## Verdict / redirect +Trades a negligible win on degenerate short inputs for microsecond-class losses on +realistic shapes. Short-input closure cost needs a hybrid engine, not fusion. diff --git a/.knowledgebase/memories/dead-handler-extent-heuristic.md b/.knowledgebase/memories/dead-handler-extent-heuristic.md new file mode 100644 index 00000000..578975f9 --- /dev/null +++ b/.knowledgebase/memories/dead-handler-extent-heuristic.md @@ -0,0 +1,29 @@ +--- +id: dead-handler-extent-heuristic +title: Exception-handler noCut zone must be {region start, region end−1, handler entry} only — a handler-extent heuristic (walk to first terminal) over-conservatively no-cuts the whole method for fall-through handlers +kind: dead-end +tags: [exception-handler, no-cut-zone, refuted-approach] +applies_to: [] +source: multi-64k/dead-handler-extent-heuristic +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# noCut = positions separating {region start, end−1, handler entry} — not the handler extent + +First attempt at protecting try/handler regions: noCut zone = protected region ∪ handler +*extent*, where the extent was computed by walking forward from the handler label to its first +`athrow`/`return`/`goto`/switch. Two flaws: + +1. **Over-conservative**: a fall-through handler (catch sets a flag, then continues into the + method tail) has no terminal → the walk ran to the end of the method → the entire method + became no-cut → `chooseCuts` declined (observed: tryRegion test declined then verbatim + threw). +2. **Wrong model**: the handler body extending across a cut is *fine* — control flows through + the chunk tail chain; only the handler *entry label* must live in the region's chunk. + +Correct rule: region + handler entry must share a chunk; bodies may tail-chain. noCut = +positions separating {start, end−1, handler}. diff --git a/.knowledgebase/memories/dead-real-generator-differential-test.md b/.knowledgebase/memories/dead-real-generator-differential-test.md new file mode 100644 index 00000000..b8c6ac8a --- /dev/null +++ b/.knowledgebase/memories/dead-real-generator-differential-test.md @@ -0,0 +1,29 @@ +--- +id: dead-real-generator-differential-test +title: No constructible pattern both overflows a generated method and compiles within the 10s deadline — an end-to-end overflow differential test is infeasible; splitter semantics must be covered at unit level +kind: dead-end +tags: [testing, differential, probe, ci-flake] +applies_to: [] +source: multi-64k/dead-real-generator-differential-test +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# End-to-end overflow differential test is infeasible + +Test-plan idea: "a pattern whose NFA step method overflows 64KB today compiles, +loads, and matches java.util.regex differentially." No such pattern could be constructed: + +- Huge literal capture alternations route to BitStateMatcher (compact) or LiteralAlternation + trie strategies — method size never overflows. +- Pushing counts up (4000 alternatives) hits OOM in the analysis phase, before codegen. +- Compiles that do succeed take 8.5–13.5s against the 10s total-compile deadline — too close + for a stable CI test; the probe test was deleted rather than kept flaky. + +Consequence: splitter semantics must be covered at unit level (load real classes so the JVM +verifier validates recomputed frames, check exact values) plus the runtime suite through the +wired-in pipeline. Don't spend more time hunting a triggering pattern within the current +strategy/limit envelope. diff --git a/.knowledgebase/memories/dead-sourceinterpreter-typing.md b/.knowledgebase/memories/dead-sourceinterpreter-typing.md new file mode 100644 index 00000000..28560ff8 --- /dev/null +++ b/.knowledgebase/memories/dead-sourceinterpreter-typing.md @@ -0,0 +1,30 @@ +--- +id: dead-sourceinterpreter-typing +title: SourceInterpreter is unusable as slot-typing basis for large generated methods — copyOperation masks producers (ASTORE) and source sets grow quadratically at loop-carried merges (78k-insn method hung the test worker >120s) +kind: dead-end +tags: [asm, sourceinterpreter, refuted-approach, performance] +applies_to: [] +source: multi-64k/dead-sourceinterpreter-typing +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + +# SourceInterpreter pathologies make it unusable for slot-typing large generated methods + +The original typing plan derived live-slot types from `SourceInterpreter` frames (per-source +descriptors + unmask recursion for ASTORE/xLOAD). Two fatal flaws: + +1. Quadratic set growth on loop-carried locals at merge points — a 78k-insn realistic test + method hung the test worker (>120s, jstack pinned `Analyzer.merge`, Analyzer.java:642). + Generated NFA-style methods are exactly this shape: each merge unions the previous + iteration's accumulated set with new defs → O(k) set unions per merge → O(n²) total. +2. ASTORE source masking forced an unmask-recursion layer (operand sources at the store, local + sources at loads) — `copyOperation` attributes values to the *copy instruction*, so a + local's type cannot be derived directly from its source insns. Built, working, then deleted + with the interpreter swap. + +Consequence: use a descriptor-flowing interpreter instead (see the DescriptorInterpreter +memory). SourceInterpreter remains fine for small methods or one-shot analyses; the pathology +needs loop-carried locals surviving merge points. diff --git a/.knowledgebase/memories/find-api-surface-sufficient.md b/.knowledgebase/memories/find-api-surface-sufficient.md new file mode 100644 index 00000000..e9650b95 --- /dev/null +++ b/.knowledgebase/memories/find-api-surface-sufficient.md @@ -0,0 +1,35 @@ +--- +id: find-api-surface-sufficient +title: ReggieMatcher runtime API covers every JDK Matcher op observed in the dd backends; grok SPI + shadow infra is the insertion seam (top-level reggie.ReggieMatcher facade exposes only matches/find) +kind: finding +tags: [api, compatibility, grok, spi, shadow] +applies_to: [reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/ReggieMatcher.java, reggie-runtime/src/main/java/com/datadoghq/reggie/ReggieMatcher.java] +source: dd_backend_fit/find-api-surface-sufficient +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# API surface no longer a blocker; grok SPI + shadow infra is the insertion seam + +Trunk `com.datadoghq.reggie.runtime.ReggieMatcher` covers every JDK Matcher op +observed in both repos: matches/find/findFrom, match/findMatch→MatchResult +(indexed+named groups, spans), matchInto/findMatchInto (alloc-free), +replaceFirst/replaceAll (literal + Function — covers +appendReplacement loops and results() streams), split(input[,limit]), findAll, +cursor() with appendReplacement/appendTail. + +Caveat: top-level `com.datadoghq.reggie.ReggieMatcher` facade exposes ONLY +matches/find — consumers must type against runtime.ReggieMatcher. + +re2j maps 1:1 (`Pattern.matcher(input).find()/matches()`, groupCount, named +groups). logs-backend grok pipeline: production runs `JdkRegexPatternSupplier` +(re2j supplier is non-production); `GrokPatternBundle` supports a SHADOW +supplier with configurable shadow_ratio + `RegexMatcherShadowActor` — A/B +validation infra already built for a `ReggieRegexPatternSupplier`. The re2j +supplier's JDK→RE2 syntax-conversion hack (`(?P<` rewrite) becomes unnecessary. +Remaining JDK-adjacent needs: Pattern.quote (keep on JDK), asPredicate adapter, +Jackson Pattern deserialization (pb YAMLInsightProvider), Map adapter type. diff --git a/.knowledgebase/memories/find-asm-classwriter-final.md b/.knowledgebase/memories/find-asm-classwriter-final.md new file mode 100644 index 00000000..0317aeca --- /dev/null +++ b/.knowledgebase/memories/find-asm-classwriter-final.md @@ -0,0 +1,31 @@ +--- +id: find-asm-classwriter-final +title: ASM interception shape — ClassWriter.visit/visitMethod are final in 9.10, so pipeline instrumentation must wrap a ClassVisitor delegating to the terminal ClassWriter +kind: finding +tags: [asm, classwriter, classvisitor, interception] +applies_to: [] +source: multi-64k/find-asm-classwriter-final +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# Wrap a ClassVisitor — ClassWriter methods are final + +`javap` on asm-9.10.1.jar shows: +`public final void visit(int,int,String,String,String,String[])` and +`public final MethodVisitor visitMethod(int,String,String,String,String[])` in +`org.objectweb.asm.ClassWriter`. A `ClassWriter` subclass therefore cannot intercept method +emission ("overridden method is final" compile errors). + +Consequence: the interception point must be a **wrapping `ClassVisitor`** (the standard ASM +instrumentation shape) delegating to the real ClassWriter, which stays the terminal pipeline +stage and `toByteArray()` producer. In the (now dropped) splitter patch, reggie's generators +had parameters typed `ClassWriter` and were mechanically widened to `ClassVisitor` +(~32 files); the only non-visitor use, `RecursiveDescentBytecodeGenerator.generate()`'s +internal `cw.toByteArray()`, was restructured to real-writer + front-visitor. Both pipeline +entry points (`RuntimeCompiler.generateBytecode`, `ReggieMatcherBytecodeGenerator`) construct +`new ClassWriter(COMPUTE_FRAMES|COMPUTE_MAXS)` wrapped by the front visitor — that wiring was +removed when the splitter was dropped and must be re-added on revival (see the L2 memories). diff --git a/.knowledgebase/memories/find-asm-toobytearray-sizecheck.md b/.knowledgebase/memories/find-asm-toobytearray-sizecheck.md new file mode 100644 index 00000000..16dc2191 --- /dev/null +++ b/.knowledgebase/memories/find-asm-toobytearray-sizecheck.md @@ -0,0 +1,28 @@ +--- +id: find-asm-toobytearray-sizecheck +title: ASM MethodTooLargeException fires at toByteArray(), NOT at visitMaxs() — any method-size probe must call toByteArray() on its scratch writer; probe results that skip it silently over-report "fits" +kind: finding +tags: [asm, methodtoolargeexception, tobytearray, probe] +applies_to: [] +source: multi-64k/find-asm-toobytearray-sizecheck +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# MethodTooLargeException fires at toByteArray(), not at visitMaxs() + +Empirically verified with a scratch probe (25,000 × {NOP; ICONST_1; POP} ≈ 75,001 code bytes +into a `ClassWriter(0)`): `visitMaxs` returns normally, `visitEnd` returns normally, +`toByteArray()` throws `MethodTooLargeException` with `getCodeSize()=75001`. The COMPUTE_FRAMES +| COMPUTE_MAXS writers used by Reggie behave the same. + +Consequences: +1. **A size probe must call `toByteArray()`** on its scratch writer. A probe that only emits + + visitMaxs silently reports 72–144KB methods as "fits" — the flush then takes the verbatim + path and the entry point's toByteArray throws later. +2. The real writer's guard surfaces at the entry point's `toByteArray()` → inside + `RuntimeCompiler`'s existing try/catch → L3 fallback preserved (a chunk-margin bug cannot + escape to a corrupt class, only to the fallback). diff --git a/.knowledgebase/memories/find-backend-prod-regex-cost.md b/.knowledgebase/memories/find-backend-prod-regex-cost.md new file mode 100644 index 00000000..0eee9537 --- /dev/null +++ b/.knowledgebase/memories/find-backend-prod-regex-cost.md @@ -0,0 +1,35 @@ +--- +id: find-backend-prod-regex-cost +title: Backend regex CPU is concentrated in logs-processing grok-on-JDK matching; savings model = cost x share x (1-1/f) with every 1% of logs-processing CPU in JDK regex worth ~$50-60K/yr +kind: finding +tags: [prod-measurement, continuous-profiler, cloud-cost, savings-estimate, logs-processing, grok] +applies_to: [] +source: backend-ready/find-backend-prod-regex-cost +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + +# Backend regex CPU is concentrated in logs-processing grok-on-JDK matching + +## Measured distribution (prod, continuous profiles + CCM) +- logs-processing: java.util.regex self = 5.60% of service CPU, essentially ALL match-time + (compile 0.003%); driver ~100% grok (fsmatic GrokModule under Matcher entries); +0.55% + InterruptibleCharSequence.charAt input plumbing. com.google.re2j = 0.00% — a RE2J + migration had NOT landed in prod despite expectations. +- prof-analyzer: java.util.regex = 1.00% (JFR/pprof parsing). +- apm-processing: java.util.regex = 0.38% only — heavy regex there is RUST dd_sds/ + regex-automata (~11%), NOT replaceable by a JDK-semantics engine. +- Continuous profiles are not reachable via the Datadog MCP toolset; pulled via the + dodo-cli flamegraph endpoint (env:prod, family java, cpu-time). + +## Savings model +cost x share x (1 - 1/f), where f = reggie-vs-JDK speedup on the fleet mix. Initial +f=3–6x (repo benchmark corpus) overstated it; the real-corpus smoke (see +find-rust-engine-crossover) revised the honest central f to ~2x. Rule of thumb: every 1% +of logs-processing CPU in JDK regex = ~$50–60K/yr. Savings materialize only if HPA scales +replicas down (fleet ~50% of requests, diurnal autoscaling visible). + +Full write-up: Datadog notebook 15576172; repo doc/investigations/multi-64k/evidence/ +ev-reggie-cpu-savings-estimate.md. diff --git a/.knowledgebase/memories/find-brace-literal-rejected.md b/.knowledgebase/memories/find-brace-literal-rejected.md new file mode 100644 index 00000000..add14716 --- /dev/null +++ b/.knowledgebase/memories/find-brace-literal-rejected.md @@ -0,0 +1,27 @@ +--- +id: find-brace-literal-rejected +title: JDK treats a bare '}' (not part of a valid quantifier) as a literal — reggie rejected it with PatternSyntaxException (fixed: parseAtom treats it as literal; mustache {{…}} patterns depend on this) +kind: finding +tags: [p1, compat, parser, brace, mustache, syntax] +applies_to: [reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/parsing/**] +source: dd_backend_fit/find-brace-literal-rejected +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# JDK treats a bare '}' as a literal; reggie rejected it (now fixed) + +JDK accepts a bare `}` (not part of a valid quantifier) as a literal; reggie used to +throw `PatternSyntaxException: Unexpected metacharacter '}'`. The syntax-level rejection +preceded the fallback decision, so `compileAllowingFallback` did NOT rescue these. + +13 patterns affected: profiling-backend 1 (`\{([\w.]+)}`), logs-backend 12 — +notably EVERY `{{ … }}` mustache-template parsing pattern (urlencode, is_match, +is_exact_match, #if/eval, local_time…) and `~|~\{closure}|…`. Also bare +`}` alone. No pattern-rewrite workaround exists for consumers. + +FIXED (commit 0727796): `}` (and an un-quantifier `{`) parse as a literal; invalid +quantifier specs still error with JDK parity. diff --git a/.knowledgebase/memories/find-case-insensitive-viable.md b/.knowledgebase/memories/find-case-insensitive-viable.md new file mode 100644 index 00000000..70e7068e --- /dev/null +++ b/.knowledgebase/memories/find-case-insensitive-viable.md @@ -0,0 +1,28 @@ +--- +id: find-case-insensitive-viable +title: JDK CASE_INSENSITIVE is safe to map for pure-ASCII patterns; input-side non-ASCII NEVER diverges (İ/ı/K/ſ verified) — the only divergence is non-ASCII pattern letters (reggie folds é->éÉ, JDK without UNICODE_CASE does not) +kind: finding +tags: [flags, case-insensitive, migration] +applies_to: [reggie-runtime/src/main/java/com/datadoghq/reggie/compat/JdkPatternCompatibility.java] +source: dd_backend_fit/find-case-insensitive-viable +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# JDK CASE_INSENSITIVE is safe to map for pure-ASCII patterns + +`JdkPatternCompatibility.toReggieFlags` maps MULTILINE/DOTALL/LITERAL and (since commit +dbd51e3, `toReggieFlags(pattern, flags)`) JDK CASE_INSENSITIVE for pure-ASCII patterns; +previously it THREW on JDK CASE_INSENSITIVE (Unicode vs ASCII folding). All 15 unique +case-insensitive literal patterns across both repos (incl. pb +BillingMetricsSender VALID_COMMIT_SHA / REGEX_NON_EMPTY_HOST_TAG) compiled +natively via inline `(?i)` prefix with ZERO divergences on the probe set. +One lb site uses `IGNORE_CASE_AND_MULTILINE` (custom constant = CI|MULTILINE, Mongo query path). + +Folding equivalence, empirically established: input-side non-ASCII NEVER diverges +(İ U+0130, ı U+0131, Kelvin K U+212A, long-s ſ U+017F all verified against literals AND +char classes — neither engine folds non-ASCII input). The ONLY divergence is non-ASCII +pattern letters: Reggie folds é->[éÉ], JDK without UNICODE_CASE does not. diff --git a/.knowledgebase/memories/find-computeextras-early-return-bug.md b/.knowledgebase/memories/find-computeextras-early-return-bug.md new file mode 100644 index 00000000..d4e1e5f7 --- /dev/null +++ b/.knowledgebase/memories/find-computeextras-early-return-bug.md @@ -0,0 +1,29 @@ +--- +id: find-computeextras-early-return-bug +title: computeExtras early-return could drop live non-param extras at cuts → chunk-signature corruption (VerifyError or silent miscompilation) — fix lives only in the dropped L2 splitter patch (reports/l2-splitter-v3-dropped.patch) +kind: finding +tags: [bug, liveness, extras, chunk-signature, corruption, post-ship-fix] +applies_to: [] +source: multi-64k/find-computeextras-early-return-bug +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# computeExtras early-return could drop live non-param extras + +Found by code reading during the splitter battery analysis. `computeExtras` had: +`if (live.nextSetBit(0) >= 0 && live.nextSetBit(0) < paramSlots) return List.of();` — an +optimization meaning "all live slots are params" that actually only checked the FIRST live +slot. A method with live params AND live non-param locals at a cut (the common real shape) +would return empty extras → chunk signatures missing live state → VerifyError at load or +silent miscompilation. It never fired in the shipped tests: the ()I tests have paramSlots=0 +(early return dead) and the (II)I crossing test carries everything in params (no extras +needed). The existing `for (int s = live.nextSetBit(paramSlots); …)` loop already handles the +all-params case correctly — the early return was pure liability. + +Status: the fix (delete the early return; regression test `splitsWithLiveParamsAndLocalAcrossCut`, +(II)J with a long accumulator live across cuts) exists only in the preserved dropped-L2 patch +`reports/l2-splitter-v3-dropped.patch` — re-apply it on any revival. diff --git a/.knowledgebase/memories/find-deadline-coverage-gaps.md b/.knowledgebase/memories/find-deadline-coverage-gaps.md new file mode 100644 index 00000000..271e9db6 --- /dev/null +++ b/.knowledgebase/memories/find-deadline-coverage-gaps.md @@ -0,0 +1,47 @@ +--- +id: find-deadline-coverage-gaps +title: The 10s total-compile deadline originally bound only SubsetConstructor — two verified holes (codegen-emission OOM in ~4.2s that a time deadline cannot close; BitState-route compile 22.3s past the deadline) closed in a5ef0fb via work-charged bypass BFS + 65,535-insn per-method emission budget + compile-scope OOM catch +kind: finding +tags: [deadline, dos, oom, bitstate, compile-time, coverage-gap, verified-on-baseline, resolved-by-fix] +applies_to: [reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/codegen/NFABytecodeGenerator.java, reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java, reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/automaton/SubsetConstructor.java] +source: multi-64k/find-deadline-coverage-gaps +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# Deadline coverage holes (closed in a5ef0fb) + +The total deadline binds ONLY via SubsetConstructor: RuntimeCompiler.compileWithDeadline sets +SubsetConstructor.TOTAL_COMPILE_DEADLINE_NANOS (ThreadLocal, clamp-only-tightens) and periodic +checks inside buildDFA loops. Two bypass classes existed: +- Routes that never determinize (BitState/NFA-cascade: parse -> NFA -> bit-parallel codegen) + consulted the deadline ZERO times → hole 2 (concatAlt x6000 `(wa0|wb0)(wa1|wb1)...`, len + 87,780, 12,000 capture groups → COMPILED in 22.3s as BitStateMatcher, 2.2x past the deadline). +- Phases outside determinization (parse, NFA build, analysis passes, codegen emission) had no + checks → hole 1 (lookahead x1000 `(?=lk0)lk0...` len 13,780 → OOM in ~4.2s; jstack pinned + NFABytecodeGenerator.generateMatchIntoMethod:8379 — codegen EMISSION, outside + SubsetConstructor's OOM-catch). Key property: a TIME deadline cannot close hole 1 — the + pattern dies inside the envelope; only a memory/size cap can. + +RESOLVED (a5ef0fb): +- Time: bypass BFS charged via chargeWork(8)/dequeued state — work budget (200M) and deadline + bind it. calt-6000: 22.3s -> 2.4s BitStateMatcher (scale-independent: x12000 -> 3.6s). +- Memory: per-method emission budget in NFABytecodeGenerator + (MAX_EMITTED_INSNS_PER_METHOD = 65_535, BoundedMethodVisitor on all 12 method-emission + sites) — oversized methods abort DURING emission (before ASM's maxs/frames pass) and + surface as the standard MethodTooLargeException graceful path. lookahead-1000: 6g OOM + in 4.2s -> graceful UnsupportedPatternException in 286ms (JDK matcher with + ALLOW_JDK_FALLBACK). Plus compile-scope OutOfMemoryError catch in compileInternal + as the general backstop. +- Deliberately NOT added: deadline checks at the PikeVM/BitState early returns — the + pass-through policy is designed semantics (TotalCompileDeadlineTest asserts PikeVM + exactly), and post-fix the finish work is genuinely cheap. +- Design note: the 65,535-insn cap can never false-trip vs the 64KB method limit + (would require <1.01 bytes/insn; cascade emission averages ~3B); it fires exactly + where MethodTooLarge would at toByteArray, ~100x cheaper. +- Tests: GroupBypassWorkBudgetTest (codegen, incl. legit-shape guard), + DeadlineCoverageHolesTest (runtime, both holes, heap-safe — see the match-time heap + memory). diff --git a/.knowledgebase/memories/find-descriptor-interpreter-design.md b/.knowledgebase/memories/find-descriptor-interpreter-design.md new file mode 100644 index 00000000..902739a1 --- /dev/null +++ b/.knowledgebase/memories/find-descriptor-interpreter-design.md @@ -0,0 +1,39 @@ +--- +id: find-descriptor-interpreter-design +title: DescriptorInterpreter for slot-typing at chunk cuts — one descriptor per value, copy ops propagate unchanged, merge = equal-or-UNKNOWN (O(1), no classloading); SimpleVerifier LUB merging is impossible because generated class names are not loadable in the codegen module +kind: finding +tags: [asm, interpreter, type-descriptor, soundness, api-gotchas] +applies_to: [] +source: multi-64k/find-descriptor-interpreter-design +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# DescriptorInterpreter: descriptor-flowing slot typing with O(1) merges + +Replacement interpreter (built in the dropped L2 splitter patch, +`reggie-codegen/codegen/DescriptorInterpreter.java`): every abstract value is a `DV` wrapping +one type-descriptor string (or `UNKNOWN`). + +- **copy ops propagate the descriptor unchanged** — the ASTORE-masking problem (see the + SourceInterpreter memory) disappears by construction; a local's type at any point is directly + the frame value. +- **merge = equal descriptors or UNKNOWN** — no class loading, no set growth; O(n) total. +- **UNKNOWN sources**: divergent merges (the original verifier would compute a LUB — + reconstructible only with a class hierarchy), `ACONST_NULL` (null-typed local used as a chunk + param typed Object breaks verification of later uses), uninitialized locals (`newEmptyValue`), + unresolvable ops. +- **Soundness**: chunk parameter types must be exactly what the verifier would accept at the + cut. Strict equality is conservative: a divergent slot that is *live* at a cut → the cut is + rejected, never guessed. `SimpleVerifier` LUB merging was ruled out: it needs classloading, + and generated class names (ReggieMatcher$xxx) are not loadable in the codegen module. + +ASM 9.10 Interpreter API gotchas (all fixed in the file): +- interpreter methods are `public`, not `protected`; +- `newExceptionValue(TryCatchBlockNode, Frame, Type)` — takes a `Type`, not a value; +- values must implement `org.objectweb.asm.tree.analysis.Value` (Frame); +- `MethodNode.exceptions` is `List` (not String[]); methods carrying an + annotation default are refused by the splitter (verbatim handles them). diff --git a/.knowledgebase/memories/find-fallback-optin-refuse.md b/.knowledgebase/memories/find-fallback-optin-refuse.md new file mode 100644 index 00000000..9bd0a1ab --- /dev/null +++ b/.knowledgebase/memories/find-fallback-optin-refuse.md @@ -0,0 +1,40 @@ +--- +id: find-fallback-optin-refuse +title: JDK fallback is opt-in (ALLOW_JDK_FALLBACK) — by default reggie REFUSES patterns it cannot compile natively with UnsupportedPatternException; every degradation path funnels through RuntimeCompiler.fallbackOrThrow's single gate +kind: finding +tags: [fallback, jdk-fallback, unsupported-pattern-exception, refuse-by-default, safety-posture, redos, linear-time] +applies_to: [reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java, reggie-runtime/src/main/java/com/datadoghq/reggie/ReggieOptions.java] +source: multi-64k/find-fallback-optin-refuse +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# Fallback is opt-in: default posture is fail-fast refusal, never silent degradation + +Code-verified on feat/dd_backend_check @ a5ef0fb: +- RuntimeCompiler.fallbackOrThrow (line ~787): `if (!options.has(ReggieOption.ALLOW_JDK_FALLBACK)) + throw new UnsupportedPatternException(reason);` — every degradation path (MethodTooLarge, + deadline, StateExplosion, compile-scope OOM) funnels here. +- Exactly ONE construction site of JavaRegexFallbackMatcher in reggie-runtime main sources, + inside fallbackOrThrow behind that gate — no bypass path. +- ReggieOptions.builder().allowJdkFallback() is an explicit enable(); the option is off by + default. Empirically: default Reggie.compile on declined patterns throws + UnsupportedPatternException (e.g. lookahead×1000 → 286ms refusal). + +Consequences: +- Default safety posture = fail-fast refusal, NOT silent degradation to backtracking + java.util.regex. The README's engine-comparison table ("O(n) guaranteed / ReDoS safe", + footnote "applies to Reggie.compile(), default, native engine only — throws + UnsupportedPatternException") is grounded in exactly this gate. +- For JDK-regex replacement: default behavior never silently changes match semantics — + a declined pattern is an explicit exception the service must route. +- For RE2J replacement: the residual exposure is the REFUSE rate (patterns reggie won't + compile natively become explicit errors where RE2J would still serve them linearly) — a + census question, not a silent-safety one. Opt-in ALLOW_JDK_FALLBACK re-introduces + backtracking semantics only when a service explicitly chooses it. +- Optional hardening: PikeVMMatcher(NFA, String) is a runtime NFA interpreter constructible + from the already-built NFA at the MethodTooLarge catch point — a linear-time terminal + fallback that would reduce the refuse rate without regressing the guarantee. diff --git a/.knowledgebase/memories/find-groupcount-charclass-parens.md b/.knowledgebase/memories/find-groupcount-charclass-parens.md new file mode 100644 index 00000000..23da14f8 --- /dev/null +++ b/.knowledgebase/memories/find-groupcount-charclass-parens.md @@ -0,0 +1,25 @@ +--- +id: find-groupcount-charclass-parens +title: groupCount must count only parens outside character classes — a naive textual paren count breaks capture-slot layout (fixed; 3 consumer patterns hit it) +kind: finding +tags: [p1, bug, groupcount, character-class, parser] +applies_to: [reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/**, reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/parsing/**] +source: dd_backend_fit/find-groupcount-charclass-parens +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# groupCount must not count parens inside character classes (fixed) + +A naive textual paren count feeding groupCount/capture-slot layout counts parens inside +`[...]` classes. Verified with findMatch before the fix: `[()\s]` → reggie groupCount=1 +(JDK 0); `([\[\]{}()*+?.\\^$|])` → reggie 2 (JDK 1); `[{}()\[\].+*?^$\\|]` → reggie 1 +(JDK 0). Matching booleans/spans were correct in these cases — group +numbering/extraction was what was wrong. + +FIXED (commit fdd6df2): the textual scan now skips `[...]` classes. 3 consumer patterns +hit this (2 logs-backend, both String-method split/match patterns over punctuation +classes). diff --git a/.knowledgebase/memories/find-jit-hugemethodlimit.md b/.knowledgebase/memories/find-jit-hugemethodlimit.md new file mode 100644 index 00000000..1a1ce537 --- /dev/null +++ b/.knowledgebase/memories/find-jit-hugemethodlimit.md @@ -0,0 +1,52 @@ +--- +id: find-jit-hugemethodlimit +title: Generated DFA-switch matcher methods above HotSpot's HugeMethodLimit (8000 bytecodes) run interpreted forever — the "slow DFA lane" is a JIT blindspot, not an algorithmic cost +kind: finding +tags: [jit, hugemethodlimit, dfa-switch, codegen, interpreted, mechanism] +applies_to: [reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/codegen/DFASwitchBytecodeGenerator.java, reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java] +source: backend-ready/find-jit-hugemethodlimit +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + +# The "slow DFA lane" is a JIT blindspot (HugeMethodLimit), not an algorithm + +## Mechanism (proven by flag-only A/B on identical generated code) +Generated DFA_SWITCH methods (matchesAtStart=21.5KB, matchInto=17.9KB, match=17.5KB, +matchesBounded=15.9KB, matches=13.8KB, findMatchEnd=9KB on the measured pattern) all exceed +HotSpot's HugeMethodLimit (default 8000 bytecodes): methods above it are NEVER JIT-compiled, +C1 or C2 — they run interpreted. Same pattern, same generated class, only the JVM flag +changed: +- dfa.find: 19,538ns -> 941ns with -XX:-DontCompileHugeMethods (20.8x; 120.6 -> 5.8 ns/char) +- dfa.findMatchFrom: 31,762ns -> 1,787ns (17.8x; 196 -> 11.0 ns/char) +The compiled code is fast as-is; the 140–420ns/char "DFA cost" attributed to it elsewhere +was interpreted-execution cost. + +## Root causes in DFASwitchBytecodeGenerator +- STATE_SPLIT_THRESHOLD=100 buckets per-state case logic into $ng_step_N helpers sized + against the 64KB JVM hard limit (~30KB helpers) — 3.75x over the 8KB JIT limit. +- The accept-state check block (per accept state: sequential state==id compare + full + anchor-condition emission, ~400B/accept for $-anchor families) is emitted INLINE in the + main method — alternation-heavy patterns (dozens of accept states) blow past 8KB even + when transitions are bucketed. +- matchesAtStart/findMatchEnd for anchored shapes also inline the anchor prologue. + +## Corpus census (513 real patterns, bytecode dump + javap size walk) +Only 5 classes exceed 8000 bytecodes: 3 DFA_SWITCH (9.0–9.1KB), 1 OPTIMIZED_NFA_WITH_ +BACKREFS (40KB), 1 OPTIMIZED_NFA_WITH_LOOKAROUND (19.9KB). All DFA_UNROLLED (86) and CHAIN +(44) classes are under 8KB. The 5 contribute ~0 to JMH sweeps (canonical inputs match none; +no-match is R1-prefiltered) — their cost is a REAL-TRAFFIC tail (a prod line paying +interpreted rates): a shadow-rollout robustness item, not a sweep item. + +## Consequences +- Hybrid re-admission of start-anchored patterns is blocked ONLY by this: a JIT-able + dfa-half fixes the find() regression. +- compileHybrid picks PikeVM as nfa-half even when the original routed BITSTATE_CAPTURE + (skips the routeBitState upgrade): measured 45.5us vs 13.5us BitState on the capture + path. Fix candidate: nfa-half = BitState for those. + +## Method note +Dump the generated classes with -Dreggie.debug.bytecode=, walk max javap offsets per +method; A/B timing with/without -XX:-DontCompileHugeMethods isolates JIT-blindspot cost. diff --git a/.knowledgebase/memories/find-l2-dead-code-verdict.md b/.knowledgebase/memories/find-l2-dead-code-verdict.md new file mode 100644 index 00000000..8e5b379c --- /dev/null +++ b/.knowledgebase/memories/find-l2-dead-code-verdict.md @@ -0,0 +1,37 @@ +--- +id: find-l2-dead-code-verdict +title: The L2 generic splitter fired on nothing real (12/12 constructible overflow families declined, +9% compile time on every pattern) — DROPPED from the tree; liveness requires a v2 wide-switch-cascade driver or NFA L1 bucketing +kind: finding +tags: [effectiveness, dead-code, battery, nfa-cascade, wide-switch, disposition] +applies_to: [] +source: multi-64k/find-l2-dead-code-verdict +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# L2 splitter verdict: declines every real overflow, costs every compile — dropped + +Battery (ev-split-coverage-probe): every CONSTRUCTIBLE overflow family routes to +NFABytecodeGenerator's per-config cascade or equally switch-dominated shapes — 12 oversized +methods across 6 families, **12/12 declined, 0 rescued** (v1 + both post-ship fixes). The only +linear-code family, `(abcdefgh)\1{N}`, is loop-compact and never overflows. Combined with the +benchmark evidence (0 splits across the entire corpus, APT + runtime paths): **no real +generator pattern can be split by v1** — the split path fires only on synthetic shapes (unit +tests). + +Costs of keeping it wired in: +- +9% (point estimate) compile-time on EVERY pattern (MethodNode buffer + replay passthrough). +- 2–12× slower failure on oversized patterns (analysis runs, then declines; bounded by the + total-compile deadline → same L3 outcome as baseline, just slower; capalt2k went 1.1s → 9.5s, + nearly the 10s deadline). + +What would make it live: (a) a v2 driver lowering wide-switch cascades (every overflow family +needs exactly this), or (b) NFABytecodeGenerator L1 bucketing (shrinks those methods instead — +REMOVES the need for L2 there). P(b) for the known families ≈ 0.4-0.5 — roughly even odds; the +disposition is invariant to this (v1 fires on nothing either way). + +**DISPOSITION: DROPPED** — work preserved as patches (`reports/l2-splitter-v3-dropped.patch`, +`reports/l2-splitter-v1-benchmark.patch`, apply from 2fc44cc). diff --git a/.knowledgebase/memories/find-l2-splitter-shipped.md b/.knowledgebase/memories/find-l2-splitter-shipped.md new file mode 100644 index 00000000..deca6387 --- /dev/null +++ b/.knowledgebase/memories/find-l2-splitter-shipped.md @@ -0,0 +1,55 @@ +--- +id: find-l2-splitter-shipped +title: L2 splitter implementation notes for revival (DROPPED — patches preserved at reports/l2-splitter-v3-dropped.patch, apply from 2fc44cc): SplittingClassVisitor/MethodSplitter/DescriptorInterpreter design and the subtleties only tests found +kind: finding +tags: [implementation, splitter, pipeline, wiring, tests] +applies_to: [] +source: multi-64k/find-l2-splitter-shipped +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# L2 splitter implementation notes (dropped — revival reference) + +Built on feat/dd-backend-check (on top of 2fc44cc), then DROPPED (see the L2 verdict memory). +Never landed as a commit; the working tree was reverted to 2fc44cc. Revive via `git apply` +from 2fc44cc: `reports/l2-splitter-v3-dropped.patch` (v3 = final, includes both post-ship +fixes; v3 = v2 + 2 files, v2 = v1 + 3 files), `reports/l2-splitter-v1-benchmark.patch` +(A/B-benchmarked tree). All subtleties below remain accurate for any future revival. + +- **`SplittingClassVisitor.java`** (reggie-codegen/codegen): wraps the real ClassWriter; + returns a `MethodNode`-backed `BufferedMethod` for every splittable method header + (`isSplittableHeader`: never ``/``/abstract/native). Flush at visitMaxs: + conservative upper-bound sizer (jumps as wide forms, ldc_w, padded switches — a fitting + conservative bound is *guaranteed* to fit) → verbatim replay via `MethodNode.accept`; + else exact probe (scratch `ClassWriter(0)`, fresh labels, **toByteArray**) → verbatim; + else `MethodSplitter.trySplit`. All refusals → verbatim → `MethodTooLargeException` → L3. +- **`MethodSplitter.java`**: instruction indexing, CFG/successors (incl. exception edges), + slot-level backward liveness (ring-buffer worklist), DV frames (Analyzer), noCut zones, + greedy cut scan at CHUNK_BUDGET=60,000 conservative bytes, `validateChunks` (per-chunk + conservative size incl. callSeq estimates; try-triples whole; param slot count ≤255), + emission (chunk0 keeps original header; `$sK` chunks private synthetic, same static-ness; + per-chunk slot remap — params identity, extras become trailing params, rest shifted; + tail = push params + invokestatic/invokespecial + matching xreturn; LDC-Integer replays + via `BytecodeUtil.pushInt` per repo convention). +- **`DescriptorInterpreter.java`** — see find-descriptor-interpreter-design. +- Wiring: `RuntimeCompiler` + `ReggieMatcherBytecodeGenerator` both wrap their writer + (dual-path rule satisfied at a single shared implementation). `asm-tree` + + `asm-analysis` 9.10.1 deps added to reggie-codegen. +- Design doc `doc/plans/method-size-splitting.md` updated with all corrections. + +## Subtleties found only by tests +- `validateChunks` must not add a tail-call estimate for the final boundary — `insnCount` + is the end boundary, not a chunk entry (first crash: IndexOutOfBounds in `extrasAt`); the + final chunk has no tail call. +- Crossing GOTO → inline call+return; crossing conditional → jump to a local end-of-chunk + label block; crossing switch case → per-case local label blocks. Forward jumps into a + *later chunk's interior* are forbidden by noCut by construction. +- `visitMaxs(0,0)` blind spot (see that memory) — every real generator method silently + declined until analyze() repaired node.maxLocals/maxStack. +- Wide-switch methods (per-config NFA cascade) stay unsplittable in v1 — documented in + doc/plans/method-size-splitting.md. +- computeExtras early-return fix (see that memory). diff --git a/.knowledgebase/memories/find-logs-backend-re2j-migration.md b/.knowledgebase/memories/find-logs-backend-re2j-migration.md new file mode 100644 index 00000000..a887cb04 --- /dev/null +++ b/.knowledgebase/memories/find-logs-backend-re2j-migration.md @@ -0,0 +1,49 @@ +--- +id: find-logs-backend-re2j-migration +title: logs-backend grok regex layer is a deliberate engine abstraction (RegexPatternSupplier + 13 suppliers + shadow A/B infra + fleet census job) mid-migration to RE2J — a ReggieRegexPatternSupplier drops into the seam with zero service redesign +kind: finding +tags: [logs-backend, re2j, migration, grok, insertion-point, shadow-testing, supplier, engine-abstraction] +applies_to: [] +source: multi-64k/find-logs-backend-re2j-migration +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# logs-backend grok regex layer: engine abstraction + active RE2J migration = reggie's insertion point + +The grok-parsing regex layer (processing-parsing, com.fsmatic.shared.parse.grok.regex) is a +deliberate engine-abstraction: RegexPatternSupplier interface + a startup-registered global +factory (RegexPatternSuppliers.getDefault, used by SafeGrokPattern/GrokPatternBundle) + 13 +supplier implementations: Jdk, Re2j, Re2jWithJdkFallback, Caching, LiteralPrefiltered, +Ipv4Rewritten, Shadow (+ RegexMatcherShadowActor). Migration tooling is ACTIVE: +- ShadowRegexPatternSupplier: main + shadow supplier, configurable shadow ratio + parser + timeout, shadow matches delegated to an actor — offline parity evaluation infra. +- InspectGrokParserMatchRulesWithRE2Job (event-jobs-logs): loads ALL track pipelines from + Mongo and tries every grok matchRule against RE2-J, emitting success/failed count metrics + — a fleet-wide compatibility census, already run, metrics already flowing. + +Implications: +1. **Reggie drops into an existing seam**: a ReggieRegexPatternSupplier behind the existing + factory needs zero service redesign; the shadow infra + inspection-job pattern can be + reused verbatim for reggie parity validation (swap the shadow supplier, add a + ReggieRegistry-style job, compare event_jobs_logs.inspect_re2j.* baselines). +2. **The migration's hard part is already reggie's strength**: the RE2J migration stalls on + RE2-incompatible patterns (lookaround/backrefs — RE2J cannot run them). Re2jWithJdk- + FallbackRegexPatternSupplier exists for exactly that: RE2J-first, then a WARN-logged + SILENT fallback to backtracking java.util.regex. Reggie handles the incompatible syntax + natively and linearly (PikeVM/BitState routes) — no backtracking fallback needed — and + refuses-by-default where it cannot guarantee semantics (see the fallback-opt-in memory). + NOTE: Re2jWithJdkFallback is currently defined-but-unreferenced in main code; which + supplier production actually registers was not found in-repo (benchmarks register Jdk; + production registration likely in a service bootstrap not present in these trees) — + confirm with logs-backend owners before any migration plan. +3. profiling-backend's UserDefinedRegex (RE2J for user-supplied request regexes, ReDoS + immunity) is the second insertion point — direct swap of the engine class, same 512-char + cap and 400-mapping, gaining lookaround/backref support with linear-time guarantees. + +Open: which supplier is registered in production today (Jdk vs Re2j vs Re2jWithJdkFallback)? +What did the InspectGrokParserMatchRulesWithRE2Job census report (the % of fleet grok +patterns RE2J cannot even compile — reggie's ceiling there)? diff --git a/.knowledgebase/memories/find-matchtime-heap-scaling.md b/.knowledgebase/memories/find-matchtime-heap-scaling.md new file mode 100644 index 00000000..f3eda922 --- /dev/null +++ b/.knowledgebase/memories/find-matchtime-heap-scaling.md @@ -0,0 +1,39 @@ +--- +id: find-matchtime-heap-scaling +title: Native matcher MATCH-time memory scales with pattern group count (per-position thread state × group arrays) — 12k-group pattern OOMs a 512m JVM while compiling fine; heap-safe tests assert semantics on the fallback path only +kind: finding +tags: [match-time, memory, pikevm, groups, test-infrastructure, gradle, scaling] +applies_to: [] +source: multi-64k/find-matchtime-heap-scaling +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# Match-time memory scales with group count — 12k groups OOMs a 512m JVM + +While writing DeadlineCoverageHolesTest, the first run killed the Gradle test +executor: `OutOfMemoryError at PikeVMMatcher.java:193` (java heap space) — NOT the +compile path. The 6g probe runs had compiled 12k-24k-group patterns to native +BitState/PikeVM matchers in seconds without ever calling matches(); the test then called +`matches()` on a native matcher for the 12k-group pattern, and MATCH-time allocation +(per-position thread state × group arrays in the NFA simulation) exceeded the test JVM heap. +Gradle test JVMs default to ~512m here (reggie-runtime/build.gradle sets no maxHeapSize; only +`-Xss8m` + add-opens), while the same compile+match at 6g heap completes fine. + +Properties: +- Compile is bounded (a5ef0fb) — this is a separate, PRE-EXISTING match-time property: the + compile fixes bound compilation cost, not matching memory. +- The exposure is (pattern group count × matching state) — services compiling+matching native + matchers for pathological group counts need match-time headroom, or a match-scope cap + (none exists today). +- Test-design consequence (applied in DeadlineCoverageHolesTest): heap-safe tests — semantic + match assertions only on the fallback path (JavaRegexFallbackMatcher delegates to + java.util.regex, heap-light); native-route coverage via codegen unit tests + big-heap probe + evidence. +- Capacity-planning input, not a correctness blocker (RE2J also grows memory with pattern + complexity). Possible follow-up: a match-scope guard analogous to the compile emission + budget (e.g. cap on groups×states at matcher construction) — only if extreme-group + service patterns exist in the fleet corpus. diff --git a/.knowledgebase/memories/find-nul-pattern-truncation.md b/.knowledgebase/memories/find-nul-pattern-truncation.md new file mode 100644 index 00000000..49284e93 --- /dev/null +++ b/.knowledgebase/memories/find-nul-pattern-truncation.md @@ -0,0 +1,37 @@ +--- +id: find-nul-pattern-truncation +title: A pattern-side NUL literal used to truncate the pattern (epsilon-sentinel collision) so matchers matched everything — fixed via ast.EpsilonNode; input-side NUL audited clean (2,152 differential checks, SWAR paths NUL-free) +kind: finding +tags: [p0, bug, nul, c-string, truncation, pattern] +applies_to: [reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/parsing/**] +source: dd_backend_fit/find-nul-pattern-truncation +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# Pattern-side NUL used to truncate the pattern (fixed); input-side NUL audited clean + +The bug: `Reggie.compile("\0")` (single NUL char as pattern) yielded a matcher whose +find()/matches() returned TRUE on every input. Root cause: `LiteralNode((char)0)` collided +with the epsilon sentinel — a pattern-side C-string-truncation semantics. `\x00` hex +escape behaved the same; `\x41`/`\x{48}` were correct. + +Impact sites found in consumers: +- logs-backend production: `domains/event-platform/libs/processing/processing-common/ + src/main/java/com/dd/ciapp/processors/Utils.java:9` — + `str.replaceAll("\u0000", "")` would strip ENTIRE strings if migrated. +- profiling-backend test: `TraceProcessorIntegrationTest.java:83,112` + `split("\0")` on proto string-cell tables. + +FIXED (commit cbb6ca5): new `ast.EpsilonNode extends LiteralNode`; ~15 epsilon-check +sites converted to instanceof EpsilonNode. Raw NUL/\x00/\0 now match NUL like JDK; +replaceAll("\u0000","")/split("\0") semantics verified. + +Input-side NUL separately audited (commit 542a76c): 2,152 differential checks (38 +patterns x 30 inputs, matches/find+span+groups/findFrom/findAll + matchesBounded over +String/StringBuilder/StringBuffer/CharBuffer all regions), zero divergences. SWAR +findFirstByte/findFirstHexDigit are XOR/range-mask based — NUL is never a sentinel; both +String coders covered (UTF-16 inputs force non-byte paths). diff --git a/.knowledgebase/memories/find-openj9-deadline-rejection.md b/.knowledgebase/memories/find-openj9-deadline-rejection.md new file mode 100644 index 00000000..8ec95d42 --- /dev/null +++ b/.knowledgebase/memories/find-openj9-deadline-rejection.md @@ -0,0 +1,33 @@ +--- +id: find-openj9-deadline-rejection +title: OpenJ9 compiles deadline-prone patterns ~an order slower than HotSpot — the 10s total-compile deadline rejects HotSpot-compilable patterns on Semeru 21 (37/83 JMH forks failed identically in baseline and candidate), and JMH on OpenJ9 is compromised anyway; benchmark on Temurin +kind: finding +tags: [openj9, deadline, compile-deadline, j9, deployment] +applies_to: [] +source: multi-64k/find-openj9-deadline-rejection +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# OpenJ9 compile-speed cliff under the total-compile deadline + +First A/B attempt ran the same benchmark selection on IBM Semeru OpenJ9 21.0.12 (sdkman +"current" on workspace-jb). 37/83 benchmark forks failed identically in BASELINE and +CANDIDATE at warmup iteration 1: RuntimeException "Failed to compile pattern: +(?:a+b+|b+a+){75}" from RuntimeCompiler.compileWithDeadline — i.e., the default 10s total +-compile deadline (commit b0c5615) expires on OpenJ9 for a bounded-quantifier pattern that +compiles well within the deadline on Temurin/HotSpot 21.0.12 (all 83 entries ran green +after the JVM switch). JMH additionally warns "Not a HotSpot compiler command compatible VM +— compiler hints are disabled" on OpenJ9, so JMH microbenchmarks on it are compromised anyway. + +Implications: +- Deployment relevance: dd-trace/Reggie on IBM J9-derived runtimes would fall back to + java.util.regex for patterns that compile fine on HotSpot — a silent performance cliff + gated by JVM flavor, not pattern class. +- Benchmarking on workspace-jb must explicitly set JAVA_HOME to Temurin + (/usr/local/sdkman/candidates/java/21.0.12-tem); sdkman "current" points at Semeru OpenJ9. + +Open: quantify the HotSpot↔OpenJ9 compile-speed ratio for deadline-prone pattern families. diff --git a/.knowledgebase/memories/find-re2j-dynamic-safety-gap.md b/.knowledgebase/memories/find-re2j-dynamic-safety-gap.md new file mode 100644 index 00000000..ce2926b3 --- /dev/null +++ b/.knowledgebase/memories/find-re2j-dynamic-safety-gap.md @@ -0,0 +1,33 @@ +--- +id: find-re2j-dynamic-safety-gap +title: Every re2j site in the dd backends compiles runtime-supplied patterns, so replacing re2j requires bounded compile (work budget + deadline + state caps, default-on graceful rejection) — RE2J's program-size caps were the safety reggie lacked; now delivered +kind: finding +tags: [re2j, untrusted, dos, dynamic-patterns, blocker] +applies_to: [reggie-runtime/src/main/java/com/datadoghq/reggie/Reggie.java, reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/ReggieNativeCompileBudget.java, reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/ReggieCompiledPatternCompiler.java] +source: dd_backend_fit/find-re2j-dynamic-safety-gap +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# re2j sites take runtime-supplied patterns — replacement requires bounded compile (now delivered) + +Every re2j call site in both repos compiles RUNTIME-SUPPLIED patterns: +- pb: `prof-viz-java/.../UserDefinedRegex.java` — request-supplied, 512-char + cap, re2j chosen for ReDoS immunity, surfaces 400 on syntax error. +- lb (6 files): quantization rules (anchored `\A(?:…)\z`), service-resolution + remapping, Stringer event filters, intake attachments, grok + Re2jRegexPatternSupplier (DOTALL, non-production). + +API port is trivial (wrappers/SPI), and reggie accepts a superset of RE2 syntax +(\A/\z supported on trunk, DOTALL flag supported). The original blocker: re2j guarantees +bounded compile via program-size caps, while reggie had exponential-compile patterns +(find-semver-exponential-compile) → CPU-DoS vector on request-supplied input. + +RESOLVED: bounded compile landed — NFA state cap, DFA work budget + wall-clock deadline, +total-compile deadline, per-method emission budget, compile-scope OOM catch (see the +deadline/coverage memories). Any remaining dynamic-site adoption is now a consumer-side +policy question (pattern length caps exist in UserDefinedRegex), not a compile-boundedness +question. diff --git a/.knowledgebase/memories/find-refusal-set-parity.md b/.knowledgebase/memories/find-refusal-set-parity.md new file mode 100644 index 00000000..39bb9048 --- /dev/null +++ b/.knowledgebase/memories/find-refusal-set-parity.md @@ -0,0 +1,47 @@ +--- +id: find-refusal-set-parity +title: Reggie and rust refusal sets on the real corpus are disjoint; reggie's refusals are JDK-fidelity guards (8/513, reducible to 7 via PikeVM re-route), not resource limits — with allowJdkFallback there are zero functional refusals +kind: finding +tags: [refusals, coverage, migration, jdk-fallback, alternation-priority] +applies_to: [reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java] +source: backend-ready/find-refusal-set-parity +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + +# Refusal-set parity: reggie vs rust on the 513-pattern corpus + +## Measured (plain compile, no fallback options) +- reggie refuses 10, rust refuses 14 (12 rust-only + 2 shared) — the sets are DISJOINT, so + replacing one engine with the other DOES introduce new refusals. +- reggie-only refusals (8/513 = 1.6%): 4x alternation-priority conflict (DFA longest-match + vs NFA first-alternative), 2x anchor-condition dilution in DFA construction, 1x + nullable-capture divergence, 1x anchor-inside-quantifier. All are CORRECTNESS guards, not + resource limits. +- Shared: camelCase splitter (lookaround alternation), backref+case-insensitive — no delta. +- rust-only refusals reggie serves natively: 12, incl. the counted-quantifier {0,256} + semver family via counted-loop lowering (rust hits a real NFA-size wall there). + +## Deployment seam +With allowJdkFallback (the natural seam mode) ALL reggie-only refusals route to +java.util.regex — zero functional refusals — and the R1 PrefilteringMatcher wraps the +fallback matcher too, so no-match inputs keep the literal prefilter. Hard-refuse +deployment would strand 1.6% of rules. For the actual grok-on-JDK seam the relevant parity +is vs the JDK, not vs rust. + +## Lift analysis +- The alternation-priority-conflict guard is OVER-CONSERVATIVE for 4 of the 8: re-routing + to PikeVM (which does first-alternative priority AND is linear-time, preserving ReDoS + resistance) verified 0 divergences vs the JDK oracle over 325 inputs/pattern. Lift + landed: corpus refusals 10 -> 7 (native 506/513 = 98.6%). +- Honest remains-refused (PikeVM itself refuses or diverges): kind:message and multiline + \n|$ blocks (anchor-in-quantifier), (^|\S)@[]/] (14 real divergences), nullable-capture + .*?\{\{ family. These need backtracking semantics to be byte-identical to Java — reggie + refuses rather than return subtly-wrong spans. + +## Root principle +rust accepts all 8 because it promises rust semantics; reggie's contract is JDK-identical +spans (grok field extraction). ReDoS resistance was never the gate — reggie's native +engines are all linear-time; the only honest blocker is semantic fidelity. diff --git a/.knowledgebase/memories/find-remaining-incompatibilities-resolved.md b/.knowledgebase/memories/find-remaining-incompatibilities-resolved.md new file mode 100644 index 00000000..314e4d56 --- /dev/null +++ b/.knowledgebase/memories/find-remaining-incompatibilities-resolved.md @@ -0,0 +1,78 @@ +--- +id: find-remaining-incompatibilities-resolved +title: Residual incompatibilities all fixed (commits 60fda73..2fc44cc): optional-group span (two bug families in chain bytecode + TDFA routing), UNICODE_CHARACTER_CLASS (?U) with empirically derived JDK sets, input-side NUL audit, total-compile deadline, huge-charset codegen — corpus 1110/1150 strict-native, 0 divergences +kind: finding +tags: [residuals, span-divergence, unicode-class, nul-audit, deadline, huge-charset] +applies_to: [reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/codegen/**, reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/RuntimeCompiler.java, reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/PatternAnalyzer.java] +source: dd_backend_fit/find-remaining-incompatibilities-resolved +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# Residual incompatibilities resolved: span fidelity, (?U) sets, NUL audit, compile deadline, huge-charset codegen + +## 1. Optional-group span divergence — 60fda73 +TWO bugs in the same family (backtrack must leave an unwound capture unmatched): +- DETERMINISTIC_CHAIN_BYTECODE: emitCapture writes capStart unconditionally at entry, capEnd + at completion; the OPT skip path and ALT_CHAIN downstreamFail restored pos but not capture + slots. Fix: emitSaveCaptures/emitRestoreCaptures snapshot every group slot before the + with-path, restore at every unwind point (LIT_ALT/LOOP_ALT safe by structure). +- Tagged TDFA (DFA_UNROLLED_WITH_GROUPS): B17 routing guard (PatternAnalyzer + hasNfaBypassCharsetOverlap) diverts to PikeVM when a bypass-able group's body charset + overlaps the bypass path (even disjoint-body sequential optionals!) OR the enter marker is + reachable via a consuming path (dfa.isCaptureAmbiguous() only samples the start closure + + accepting targets, so head shapes like x(?:(a):)?b were reported unambiguous while + state-entry group actions recorded stale starts). Alternation bypasses (b|(b)) stay on the + TDFA (C2.4/C2.4B handles them — pinned by routing tests). +Verification: 14,256-check differential fuzz (divergences 102→0); corpus 2→1 divergences. +KEY DEBUG LESSON: -Dreggie.debug.trace= dumps generated bytecode (only when the +compile succeeds; too-large patterns fail before the trace stage). + +## 2. UNICODE_CHARACTER_CLASS — d6e49b9 +(?U) inline + ReggieFlags.UNICODE_CHARACTER_CLASS switch \w/\d/\s (+complements, in classes) +and POSIX \p{...} to their JDK Unicode definitions. Set membership DERIVED EMPIRICALLY over +the whole BMP (probe: candidate predicate vs Pattern.compile(pat, U) over 65536 code points): +- \d == Nd; \w == isAlphabetic ∪ Nd ∪ M(Mn+Mc+Me) ∪ Pc ∪ Join_Control(200C/200D) (UTS#18) +- \s == isSpaceChar ∪ [\t-\r] ∪ {NEL} — the JDK set EXCLUDES U+001C-001F (unlike the + Unicode White_Space property) +- \p{Alpha}=isAlphabetic, Alnum=Alpha∪Nd, Lower/Upper=isLower/UpperCase, Punct=P*, + Blank=Zs∪{\t}, Cntrl=Cc, Space=\s-set; ASCII stays ASCII +Loud rejects under (?U): \b/\B (Unicode word boundary needs every engine evaluator +mode-aware — out of scope) and \p{Graph}/\p{Print}/\p{XDigit} (JDK sets not reproduced). +(?u) UNICODE_CASE accepted as no-op (reggie folding is unconditionally Unicode). +Drive-by: pre-existing default-\s missed \x0B (vertical tab) — fixed. +Corpus: divergences 1→0 (the [^\w:\-\.\/] lb site). JdkPatternCompatibility maps the flag. + +## 3. Input-side NUL audit — 542a76c +2,152 differential checks (38 patterns × 30 inputs, matches/find+span+groups/findFrom/ +findAll + matchesBounded over String/StringBuilder/StringBuffer/CharBuffer all regions), +zero divergences. SWAR findFirstByte/findFirstHexDigit are XOR/range-mask based — NUL is +never a sentinel; both String coders covered (UTF-16 inputs force non-byte paths). Closed. + +## 4. Total-compile deadline — b0c5615 +-Dreggie.compile.totalDeadlineMs (default 10s, 0 disables; read per-compile). ThreadLocal +deadline clamped into every determinization (SubsetConstructor) so tight knobs bind; checked +between phases → fallbackOrThrow. CRITICAL DESIGN NUANCE discovered empirically: when the +deadline expires but analysis selected an NFA-backed strategy (PikeVM/BitState — add +PIKEVM_CAPTURE to isNfaBacked's gate!), the compile FINISHES: those patterns are exactly the +ones java.util.regex CANNOT match in bounded time (verified: JDK matcher hangs on +((a|b){0,256}){12} input — catastrophic backtracking), so a JDK fallback would move the DoS +from compile time to match time. 12x bomb: 15.2s → 10.03s default / 2.05s at 2s knob. + +## 5. Huge-charset codegen — 2fc44cc +Root cause NOT the charset size per se: subset construction emits one DFA transition map +entry per PARTITION PIECE of a large class (~420 for \p{IsAlphabetic}); every DFA-unrolled +emitter unrolled a full range cascade PER PIECE → 68KB methods. Fix: (a) merge transitions +sharing target (+equal guard/tagOps) into one check (pieces are disjoint → merge-safe), +(b) merged sets ≥100 ranges match via static boolean[65536] built in from a compact +encoded-ranges string constant (faster too). Emitter by emitter: matches (generateStateCode), +matchesAtStart, matchesBounded (bounded CharSequence guards), greedy/group-action paths, +tagOps path. Corpus: the .*[\p{IsAlphabetic}].* lb site strict-native (1110/1150, +1, zero +regressions, zero divergences). + +## Final corpus state (results_after_incompatibilities.jsonl) +profiling-backend 100%/100%, logs-backend 97% explicit / 94% string-method, 1110/1150 +strict-native (was 1109), DIVERGENCES: 0. Full gradle suite green after every commit. diff --git a/.knowledgebase/memories/find-repo-method-size-landscape.md b/.knowledgebase/memories/find-repo-method-size-landscape.md new file mode 100644 index 00000000..56355e38 --- /dev/null +++ b/.knowledgebase/memories/find-repo-method-size-landscape.md @@ -0,0 +1,34 @@ +--- +id: find-repo-method-size-landscape +title: JVM caps a method at 65,535 bytes of bytecode (u2 code_length — over-limit methods cannot even be encoded); reggie's layering: L1 per-generator structural lowering (DFASwitch bucketing, charset tables, BitState), L2 generic splitter (dropped), L3 MethodTooLargeException → fallbackOrThrow JDK delegation +kind: finding +tags: [jvm, method-limit, codegen, architecture, layered-design] +applies_to: [] +source: multi-64k/find-repo-method-size-landscape +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# Method-size landscape and the L1/L2/L3 layering + +JVM caps one method's bytecode at 65,535 bytes (code_length is a `u2` — an over-limit method +cannot even be encoded; post-hoc repair of a serialized class is impossible). Reggie generates +specialized matcher classes per pattern (36+ strategies in `reggie-codegen/codegen/`), so some +patterns overflow one method. + +Layering: +- **L3 (last resort)**: `RuntimeCompiler.compile()` catches `MethodTooLargeException` → + `fallbackOrThrow` → JDK `java.util.regex` delegation with a warning (reggie-runtime, + ~line 1042). Comment names NFABytecodeGenerator as a generator without splitting. +- **L1 (per-generator structural lowering)**: precedent `DFASwitchBytecodeGenerator` + STATE_SPLIT_THRESHOLD=100 — bucket helpers `$ng_step_N`/`$gt_step_N`, ~30KB each; huge-charset + boolean[] lookup tables (2fc44cc) and BitState (compact) alternatives absorb the known + overflow families. +- **L2 (generic in-pipeline splitter)**: built, measured as firing on nothing real, and + DROPPED (see the L2 verdict memory; patches preserved). Failure of L2 must degrade to + verbatim emission → today's exception → L3 (never worse than status quo). + +Design doc: doc/plans/method-size-splitting.md. diff --git a/.knowledgebase/memories/find-rust-engine-crossover.md b/.knowledgebase/memories/find-rust-engine-crossover.md new file mode 100644 index 00000000..4ca2de42 --- /dev/null +++ b/.knowledgebase/memories/find-rust-engine-crossover.md @@ -0,0 +1,39 @@ +--- +id: find-rust-engine-crossover +title: On the real logs-backend pattern mix rust regex is 3-5x faster than reggie at median (11-54x totals, surviving FFI marshal); honest fleet speedup of reggie vs JDK is ~2x, capped by unanchored find (fixed later by R1/R2) +kind: finding +tags: [rust-regex, regex-automata, real-corpus, benchmark, ffm-jni, crossover] +applies_to: [reggie-benchmark/src/main/java/com/datadoghq/reggie/benchmark/RealCorpusScanBenchmark.java, reggie-benchmark/src/main/java/com/datadoghq/reggie/benchmark/engines/RustRegexEngine.java] +source: backend-ready/find-rust-engine-crossover +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + +# reggie vs JDK vs Rust regex on the real logs-backend mix + +## Setup +513 real logs-backend pattern literals x 6 realistic log lines; engines jdk / reggie / +rust (regex 1.13.1 = regex-automata meta engine = the dd_sds family); FFM with per-call +UTF-8 marshalling; mode-split by JDK oracle boolean. 491 common patterns, ZERO divergences +across 2,946 pairs. Coverage: jdk 0 / reggie 10 / rust 12 refusals (see +find-refusal-set-parity). + +## Results +- MATCH mode (327 pairs, grok-parsing shape where prod 5.6% CPU lives): jdk 198us, + reggie 180us, rust 16us; reggie/jdk median 0.48x (2.1x faster); reggie/rust median 2.77x. +- NOMATCH mode (2,619 pairs, rule-filter scans): jdk 11.2ms, reggie 5.2ms, rust 97us; + reggie/jdk median 0.68x; reggie/rust 4.9x median, 54x totals. +- Worst reggie shapes: .*-prefixed unanchored patterns (e.g. '(.*) \((.*)\)': jdk 213us, + reggie 1.7ms, rust 316ns) — no literal prefilter + per-start-position restarts. +- Rust counted-quantifier wall is real: semver {0,256} exceeds the default 10MB NFA limit + AND 256MB raised; {0,128} builds in 394ms. + +## Consequences +1. Rust (regex-automata) is 3–5x faster than reggie at median on the real mix, surviving + FFI marshal — replacing dd_sds rust with reggie would RAISE CPU; the case for it is + architectural only (cross-runtime FFI tax). +2. The savings model's f (see find-backend-prod-regex-cost) revised 3–6x -> ~2x central + (matched-pairs median) — the cap is the no-match scan, addressed by the literal + prefilter + single-pass scan arc (find-unanchored-find-prefilter). diff --git a/.knowledgebase/memories/find-semver-exponential-compile.md b/.knowledgebase/memories/find-semver-exponential-compile.md new file mode 100644 index 00000000..2969f1b0 --- /dev/null +++ b/.knowledgebase/memories/find-semver-exponential-compile.md @@ -0,0 +1,35 @@ +--- +id: find-semver-exponential-compile +title: Bounded quantifiers {n,m} scaled compile exponentially (semver pattern never finished; 64x bound = 68.6s, 128x+ effectively infinite) — request-supplied patterns were a CPU-DoS; bounded by work budget + deadline + NFA state cap +kind: finding +tags: [p0, bug, bitstate, quantifier, compile, dos, semver] +applies_to: [reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/codegen/**, reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/ReggieNativeCompileBudget.java, reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/ReggieCompiledPatternCompiler.java] +source: dd_backend_fit/find-semver-exponential-compile +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# Exponential compile on bounded quantifiers {n,m} (bounded, correct now) + +The canonical semver pattern (logs-backend +`domains/rum/libs/rum-commons-domain/src/main/java/com/dd/rum/utils/SemVerParser.java`, +~300 chars, three `{0,256}`-style quantifiers + named groups) NEVER finished +compiling under `Reggie.compile()` — killed after 10+ min; JDK compiles instantly. + +Measured scaling (BitStateMatcher, synthetic semver variants, Scale.java): +bound 2→121ms, 4→19ms, 8→51ms, 16→215ms, 32→2.3s, 64→68.6s, 128/256 effectively +infinite. Exponential in the {n,m} bound. + +Security angle: request-supplied patterns can nest `{n,m}` to CPU-DoS the +compiler. re2j protects via program-size caps; reggie's +`ReggieNativeCompileBudget` caps only source *length* (default 16,384 chars in +`ReggieCompiledPatternCompiler`), and plain `RuntimeCompiler.compile` applies +no budget at all. + +RESOLVED: root cause was analysis-side, not BitState (see the bitstate-blowup-root +memory); bounded by NFA state cap (1M), DFA work budget + deadline, and the +total-compile deadline — semver ~850ms, bombs get a graceful +UnsupportedPatternException instead of a hang or OOM. diff --git a/.knowledgebase/memories/find-sourceinterpreter-pathologies.md b/.knowledgebase/memories/find-sourceinterpreter-pathologies.md new file mode 100644 index 00000000..7d9904f1 --- /dev/null +++ b/.knowledgebase/memories/find-sourceinterpreter-pathologies.md @@ -0,0 +1,35 @@ +--- +id: find-sourceinterpreter-pathologies +title: SourceInterpreter's two measured pathologies — copyOperation masks value producers (local 1's source is the ASTORE, not the LDC) and quadratic source-set growth at loop-carried merges (Analyzer.merge pinned in a >120s hang) +kind: finding +tags: [asm, sourceinterpreter, quadratic, astore-masking, analysis] +applies_to: [] +source: multi-64k/find-sourceinterpreter-pathologies +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# SourceInterpreter pathologies for slot-typing at cut points + +Two independent problems, both empirically verified: + +**1. copyOperation masks producers.** `SourceInterpreter.copyOperation` attributes a value to +the *copy instruction*: after `ldc "abc"; astore 1`, local 1's source set is `[VarInsnNode#58]` +(the ASTORE), not the LDC. A local's type at a cut therefore cannot be derived directly from +its source insns — stores must be unmasked (primitive stores from the opcode; ASTORE by +recursing into the stored operand's sources; xLOAD stack sources by recursing into the loaded +local's sources). That unmasking machinery was built, then deleted when the interpreter was +replaced. + +**2. Quadratic set growth on loop-carried locals.** Source sets grow ~1 node per loop +iteration at merge points (each `Lend`-style merge unions the previous iteration's accumulated +set with new defs), so Frame merges perform O(k) set unions → O(n²) total. A 78k-insn test +method (6,001 if/else iterations) took >120s, hung the Gradle test worker; jstack pinned +`Analyzer.merge` (Analyzer.java:642). ASM's `MethodNode.accept`-time behavior is fine — it's +only the *analysis* that degrades. + +Ruled out: SourceInterpreter as the typing basis for any splitter/analysis over generated +NFA-style methods (see the dead-end and DescriptorInterpreter memories). diff --git a/.knowledgebase/memories/find-try-handler-chunk-constraint.md b/.knowledgebase/memories/find-try-handler-chunk-constraint.md new file mode 100644 index 00000000..6fea1eb6 --- /dev/null +++ b/.knowledgebase/memories/find-try-handler-chunk-constraint.md @@ -0,0 +1,30 @@ +--- +id: find-try-handler-chunk-constraint +title: Try-region cut rule: the protected region [start,end) AND the handler ENTRY label must share one chunk; handler bodies may tail-chain across cuts — noCut = positions separating {start, end−1, handler} +kind: finding +tags: [jvm, exception-table, try-catch, cut-points, verifier] +applies_to: [] +source: multi-64k/find-try-handler-chunk-constraint +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# Try regions: region + handler entry share a chunk; bodies may tail-chain + +The per-method exception table means each `tryCatchBlock(start, end, handler, type)` is +re-emitted into the chunk that contains it — so the protected region `[start, end)` and the +**handler entry label** must be in one chunk. The handler **body** is ordinary code and may +extend past chunk boundaries via normal tail chaining (handler-entry → … → cut → chunk tail +call); only the label binding matters. + +Cut-safety rule derived: a cut is forbidden iff it separates the triple +{start, end−1, handler} (noCut zone `[min(start,handler)+1 .. max(end-1, handler)]`). + +The initial heuristic — bounding the no-cut zone by the handler's *extent* (walk to first +athrow/return/goto) — was wrong: fall-through handlers (`catch { acc = -999 }` then continue +into the method tail) never hit a terminal, so the walk consumed the whole remainder of the +method and `chooseCuts` declined the split (seen: conservative=66018, probe failed, split +declined, verbatim threw). See the handler-extent dead-end memory. diff --git a/.knowledgebase/memories/find-unanchored-find-prefilter.md b/.knowledgebase/memories/find-unanchored-find-prefilter.md new file mode 100644 index 00000000..f70bab9a --- /dev/null +++ b/.knowledgebase/memories/find-unanchored-find-prefilter.md @@ -0,0 +1,43 @@ +--- +id: find-unanchored-find-prefilter +title: Unanchored-find no-match cost was reggie's fleet cap — the delivered arc (required-literal prefilter R1, single-pass unanchored scan R2, 1-char intrinsic) cut the real-corpus no-match sweep 24x (15.6ms -> 636us) and put reggie ahead of rust no-match +kind: finding +tags: [prefilter, literal-extraction, memchr, single-pass, unanchored-find, R1, R2] +applies_to: [reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/analysis/RequiredLiteralAnalyzer.java, reggie-runtime/src/main/java/com/datadoghq/reggie/runtime/PrefilteringMatcher.java] +source: backend-ready/hyp-unanchored-find-prefilter +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + +# Unanchored-find no-match cost was reggie's fleet cap — prefilter + single-pass scan arc delivered 24x + +## Problem +Real-corpus smoke (find-rust-engine-crossover) showed reggie's fleet-mix speedup vs JDK +was capped at ~2x by the no-match scan: unanchored find() restarted matching at every +start position with no cheap rejection. Two fixes, as rust/JDK BnM do it: +- R1: memchr-style required-literal prefilter for unanchored find() — a SOUND extractor + producing language-level facts (required substring in every match), audited against the + JDK oracle (0 violations over the 513-pattern corpus), rejecting absent-literal inputs + via SIMD-intrinsic'd String.indexOf before the engine runs. +- R2: single-pass unanchored scan (start-state self-loop closure, RE2 style) instead of + per-start-position restarts for no-literal .*-prefix shapes — kills the 1.7ms-class + patterns. + +## Delivered and measured +Full arc: R1 -> R2 -> R2b/R1b -> 1-char intrinsic. +- R1 alone: no-match sweep 33.9ms -> 3.7ms (9.1x); 287/513 patterns covered (63% of + no-match pairs instant-reject); reggie went from 10.6x behind rust to 1.17x behind, + 12x faster than JDK; matched-pair times unchanged. +- Full arc: no-match sweep 15.6ms -> 636us (24x); reggie 1.79x ahead of rust no-match at + the R2b checkpoint; final 4-engine matrix 0.25us/pair vs rust 0.66. +- Acceptance gate: RealCorpusScanBenchmark over the committed corpus. + +## Design constraints that made R1 sound +- Extractor facts must hold for EVERY match (soundness): use only exact runs, boundary + merges between exact runs, prefix/suffix runs of class quantifiers — never + alternation-branch guesses without LCP proof. +- The prefilter wraps the JDK fallback matcher too (allowJdkFallback mode), so no-match + inputs keep fast rejection on every engine. +- 1-char facts need the intrinsic path (String.indexOf is SIMD-intrinsic'd) to pay. diff --git a/.knowledgebase/memories/find-unicode-property-gaps.md b/.knowledgebase/memories/find-unicode-property-gaps.md new file mode 100644 index 00000000..74aa0161 --- /dev/null +++ b/.knowledgebase/memories/find-unicode-property-gaps.md @@ -0,0 +1,31 @@ +--- +id: find-unicode-property-gaps +title: POSIX/Unicode property classes (\p{Alnum}..\p{IsAlphabetic}) implemented, incl. (?U) UNICODE_CHARACTER_CLASS; remaining loud rejects under (?U): \b/\B and \p{Graph}/\p{Print}/\p{XDigit} (JDK sets not reproduced) +kind: finding +tags: [p1, compat, unicode, posix-classes] +applies_to: [reggie-codegen/src/main/java/com/datadoghq/reggie/codegen/parsing/**] +source: dd_backend_fit/find-unicode-property-gaps +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# POSIX/Unicode property classes: implemented; remaining (?U) gaps are \b/\B and Graph/Print/XDigit + +Was: rejected on trunk — `\p{Alnum}`, `\p{Alpha}`, `\p{ASCII}`, `\p{Cntrl}`, +`\p{Lower}`, `\p{IsAlphabetic}` (and negations) unsupported; README documented only +`\p{L}`, `\p{N}`. 7 sites, all logs-backend. + +RESOLVED: POSIX ASCII classes (Alnum/Alpha/Digit/Lower/Upper/Blank/XDigit/ASCII/ +Cntrl/Space/Graph/Print/Punct) + IsAlphabetic/IsLetter/IsDigit Unicode-aware landed +(f52461a); UNICODE_CHARACTER_CLASS (?U) landed with set membership DERIVED EMPIRICALLY +over the whole BMP (d6e49b9 — see the incompatibilities-resolved memory for the derived +sets). `\p{IsAlphabetic}` standalone strict-rejects on the 64KB method limit only for +huge-charset shapes — fixed by bitmap charset codegen (2fc44cc). + +Remaining gaps (loud reject, by design): under (?U), `\b`/`\B` (Unicode word boundary +needs every engine evaluator mode-aware — out of scope) and `\p{Graph}`/`\p{Print}`/ +`\p{XDigit}` (JDK sets not reproduced). Script/name properties still unimplemented +(README). diff --git a/.knowledgebase/memories/find-visitmaxs-zero-blindspot.md b/.knowledgebase/memories/find-visitmaxs-zero-blindspot.md new file mode 100644 index 00000000..c03212c9 --- /dev/null +++ b/.knowledgebase/memories/find-visitmaxs-zero-blindspot.md @@ -0,0 +1,37 @@ +--- +id: find-visitmaxs-zero-blindspot +title: COMPUTE_MAXS generators legally declare visitMaxs(0,0) (~230 call sites, 21 generator files) — ASM Analyzer sizes its frame from node.maxLocals=0 and fails at instruction 0; any MethodNode-based analysis must repair maxLocals/maxStack first (fix in the dropped L2 patch) +kind: finding +tags: [visitmaxs, computemaxes, asm-analyzer, bug, post-ship-fix, maxlocals] +applies_to: [] +source: multi-64k/find-visitmaxs-zero-blindspot +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# visitMaxs(0,0) generators blind MethodNode-based analysis + +Discovered by compiling a 40k-char literal through the real pipeline: the splitter declined +with `AnalyzerException at instruction 0: "Trying to set an inexistant local variable 0"`. + +- Generators stream into COMPUTE_MAXS writers and legally declare `visitMaxs(0, 0)` — **~230 + call sites across 21 generator files** (NFABytecodeGenerator alone: 22). +- The buffered `MethodNode` therefore carries `maxLocals = 0`, and ASM's `Analyzer` sizes its + initial Frame from `node.maxLocals` → refuses at instruction 0 → `analyze()` returned false → + every real generator method silently declined. Any MethodNode-based analysis over generator + output hits this. +- Unit suites pass anyway when the test harness declares real maxima (`visitMaxs(64, 64)`) — + tests never exercise the 0/0 contract the generators actually use. + +Fix (in the preserved dropped-L2 patch, MethodSplitter.analyze()): before analysis, +`node.maxLocals = max(varAccessed)+1 (≥ paramSlots)` and +`node.maxStack = max(node.maxStack, 64)`. The real writer recomputes both at serialization, so +the repair affects the analysis only. A still-too-small repaired stack floor surfaces as +AnalyzerException → decline → verbatim — never corrupt. Regression test +`splitsOversizedMethodDeclaredWithZeroMaxes`. + +Open: should the floor 64 be derived (paramSlots + heuristic) instead of constant? Not urgent — +failure mode is decline, not corruption. diff --git a/.knowledgebase/memories/hyp-bitstate-blowup-root.md b/.knowledgebase/memories/hyp-bitstate-blowup-root.md new file mode 100644 index 00000000..b00011d8 --- /dev/null +++ b/.knowledgebase/memories/hyp-bitstate-blowup-root.md @@ -0,0 +1,35 @@ +--- +id: hyp-bitstate-blowup-root +title: Exponential BitState compile root cause was analysis-side, NOT BitState codegen — per-transition flattenClosure, unmemoized canReachGroupExit, exponential {n,m} unrolling, residual quadratic per-state work +kind: finding +tags: [root-cause, bitstate, dfa, determinization] +applies_to: [] +source: dd_backend_fit/hyp-bitstate-blowup-root +status: active +recorded: 2026-09-28 +valid_at: 2026-09-28 +--- + + + +# Exponential compile root cause: analysis-side, NOT BitState codegen + +CONFIRMED via jstack sampling during hang + fix-by-fix measurement. The +"exponential BitState compile" was actually four stacked problems, none of them +the BitState codegen itself: + +1. SubsetConstructor.buildDFA called flattenClosure(anchoredClosures) inside the + per-(DFA-state, charset) transition loop — O(NFA states x closure size) work + rebuilt every transition. Hoisting it: semver never-finishes -> ~850ms. + (jstack leaf frame: SubsetConstructor.flattenClosure under + PatternAnalyzer.doAnalyze -> buildDFA.) +2. canReachGroupExit (called from isGroupActuallyEntered <- computeTagOperations) + ran an unmemoized closure-crawling recursion PER TRANSITION, iterating + O(|closure| x |trans| x |closure|) per invocation. Memoized reverse BFS from + group EXIT markers: 165s -> ~26s on the 12x (a|b){0,256} bomb. +3. ThompsonBuilder.buildCountedQuantifier unrolls nested {n,m} copies + exponentially (8^20 states for 20x{0,8}) — OOM before any budget sees it. + Fixed with the NFA state cap (100K was too low — semver legitimately builds + ~0.3-0.6M states; 1M default verified under -Xmx512m). +4. Residual: per-state work is still quadratic up to the 10K DFA-state cap; + the work budget + wall-clock deadline bound it (12x bomb ~10-15s, correct). diff --git a/AGENTS.md b/AGENTS.md index e36be2e9..ca83e807 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,4 +1,95 @@ # Reggie - Agent Development Guide + + +## Cairn knowledgebase — consult before designing, implementing, or reviewing + +If `.knowledgebase/memories/` exists in this repo (sentinel: `.knowledgebase/.cairn`), +it holds committed institutional memory: generally valid knowledge distilled from +past cairn investigations — confirmed findings and refuted dead ends, one flat +memory per file. The investigation trails themselves are personal (in ~/.cairn); +the repo carries only the distilled knowledge. Use it instead of re-deriving: + +- **Before feature/design work on an area**: query it — run the cairn skill's + script: `python3 /investigation.py brief [--paths ] [--diff]` (the cairn-memory skill — its directory is listed in your available skills; the `/cairn` plugin command wraps the same script on + Claude Code). Matched memories are design + constraints; dead-end memories are approaches already refuted here. +- **Before/while implementing or reviewing**: check the diff against matched + memories; never re-introduce a refuted dead end. +- **Cite** consumed memory ids (`kb/`) or their source nodes + (`/`) in designs/PRs (`cairn cite `) so the + memory's payoff stays measurable. +- **Resuming interrupted work** (fresh session or after context compaction): + read the active investigation's wrap-up note (`notes/-wrapup.md`, kept + fresh at every checkpoint) first — settled findings, refuted dead-ends, + open questions, next step — then query for specifics. The wrap-up is the + point-in-time refill artifact; the memories are canonical. +- Memories live at `.knowledgebase/memories/.md` (frontmatter: id, title, + kind, tags, applies_to, source — provenance back into the ~/.cairn trail). + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + > Single source of truth for AI agents working in this repo. `CLAUDE.md` is a redirect stub — > all edits go here, never there.