From a47c75d2ab6d79a7b01251bb01b1b2695610eaf6 Mon Sep 17 00:00:00 2001 From: Dan Reynolds Date: Sun, 6 Sep 2026 07:18:23 -0400 Subject: [PATCH 1/7] WIP exp 283: rerun census instrumentation + one-pass candidate --- .../experiments/stream_rerun_census.dart | 301 ++++++++++++++++++ .../experiments/stream_rerun_one_pass_ab.dart | 275 ++++++++++++++++ .../experiments/stream_rerun_pass_price.dart | 195 ++++++++++++ lib/src/reader/read_worker.dart | 34 ++ lib/src/reader/reader_pool.dart | 7 + lib/src/stream_engine.dart | 28 ++ test/query_decoder_test.dart | 103 ++++++ 7 files changed, 943 insertions(+) create mode 100644 benchmark/experiments/stream_rerun_census.dart create mode 100644 benchmark/experiments/stream_rerun_one_pass_ab.dart create mode 100644 benchmark/experiments/stream_rerun_pass_price.dart diff --git a/benchmark/experiments/stream_rerun_census.dart b/benchmark/experiments/stream_rerun_census.dart new file mode 100644 index 00000000..6ecb434c --- /dev/null +++ b/benchmark/experiments/stream_rerun_census.dart @@ -0,0 +1,301 @@ +// ignore_for_file: avoid_print + +/// [EXP-283] Rerun census for the release suite's reactive-stream shapes. +/// +/// The three largest lanes in the release suite are reactive fan-out +/// workloads. Every write dirties some set of streams and each dirty stream +/// re-executes its query on a reader worker so the engine can learn whether +/// its result changed. Nothing has ever counted how many reruns those lanes +/// actually issue, what share of them change, or what a rerun costs — the +/// numbers a candidate that changes rerun work has to be sized against. +/// +/// Reproduces each lane's stream/write shape (scaled to keep setup fast) and +/// reports the census from the main isolate's own view of `selectIfChanged`. +library; + +import 'dart:async'; +import 'dart:io'; +import 'dart:math' as math; + +import 'package:resqlite/resqlite.dart'; + +Future main(List args) async { + var lanes = ['keyed-pk', 'fanout', 'feed']; + for (final arg in args) { + if (arg.startsWith('--lane=')) { + lanes = [arg.substring('--lane='.length)]; + } + } + for (final lane in lanes) { + switch (lane) { + case 'keyed-pk': + await _keyedPk(); + case 'fanout': + await _fanout(); + case 'feed': + await _feed(); + default: + throw ArgumentError('unknown lane: $lane'); + } + } +} + +void _resetCensus() { + StreamEngine.tmpReruns = 0; + StreamEngine.tmpChanged = 0; + StreamEngine.tmpRerunUs = 0; + StreamEngine.tmpChangedUs = 0; + StreamEngine.tmpTrace.clear(); +} + +/// Replay the recorded per-stream outcome sequence through a candidate +/// predictor and report what it would have decided. +/// +/// [enter] is how many consecutive changed reruns arm decode-first; it +/// disarms on the first unchanged rerun. `save` and `tax` are the measured +/// per-rerun figures from `stream_rerun_pass_price.dart` for this shape. +void _predictor(String lane, int enter, double save, double tax) { + final streak = {}; + var fired = 0; + var hits = 0; + var misses = 0; + var missedWins = 0; + for (final (key, changed) in StreamEngine.tmpTrace) { + final armed = (streak[key] ?? 0) >= enter; + if (armed) { + fired++; + if (changed) { + hits++; + } else { + misses++; + } + } else if (changed) { + missedWins++; + } + streak[key] = changed ? (streak[key] ?? 0) + 1 : 0; + } + final net = hits * save - misses * tax; + print( + ' predictor(enter=$enter): fired=$fired hits=$hits misses=$misses ' + 'unarmed-changed=$missedWins ' + 'modelled_net=${net.toStringAsFixed(0)}us ' + '(${(net / StreamEngine.tmpTrace.length).toStringAsFixed(2)}us/rerun)', + ); +} + +void _report(String lane, int writes, int streams, double wallMs) { + final reruns = StreamEngine.tmpReruns; + final changed = StreamEngine.tmpChanged; + final us = StreamEngine.tmpRerunUs; + final changedUs = StreamEngine.tmpChangedUs; + final unchanged = reruns - changed; + final unchangedUs = us - changedUs; + String per(int total, int n) => + n == 0 ? '-' : (total / n).toStringAsFixed(2); + print( + '$lane: streams=$streams writes=$writes ' + 'wall=${wallMs.toStringAsFixed(1)}ms ' + 'reruns=$reruns (${(reruns / writes).toStringAsFixed(1)}/write, ' + 'max ${streams * writes}) ' + 'changed=$changed ' + '(${reruns == 0 ? 0 : (100 * changed / reruns).toStringAsFixed(2)}%) ' + 'rerun_us_mean=${per(us, reruns)} ' + 'changed_us_mean=${per(changedUs, changed)} ' + 'unchanged_us_mean=${per(unchangedUs, unchanged)}', + ); +} + +Future _open(String tag) async { + final dir = await Directory.systemTemp.createTemp('exp283_${tag}_'); + return Database.open('${dir.path}/t.db'); +} + +/// A11 shape: 50 streams each on one PK, 200 random-PK writes over 10k rows. +Future _keyedPk() async { + const rowCount = 10000; + const streamCount = 50; + const writeCount = 200; + final db = await _open('keyedpk'); + await db.execute( + 'CREATE TABLE items(id INTEGER PRIMARY KEY, value INTEGER, ' + 'label TEXT NOT NULL)', + ); + await db.executeBatch( + 'INSERT INTO items(id, value, label) VALUES (?, ?, ?)', + [for (var i = 1; i <= rowCount; i++) [i, i, 'row-$i']], + ); + final subs = >[]; + final seen = List.filled(streamCount, false); + for (var s = 0; s < streamCount; s++) { + final i = s; + subs.add( + db + .stream('SELECT * FROM items WHERE id = ?', [i + 1]) + .listen((_) => seen[i] = true), + ); + } + await _waitUntil(() => !seen.contains(false)); + + final rng = math.Random(0xA11); + _resetCensus(); + final sw = Stopwatch()..start(); + for (var w = 0; w < writeCount; w++) { + await db.execute('UPDATE items SET value = ? WHERE id = ?', [ + w, + rng.nextInt(rowCount) + 1, + ]); + } + await _quiesce(db); + sw.stop(); + _report('keyed-pk', writeCount, streamCount, sw.elapsedMicroseconds / 1000); + for (final enter in [1, 2, 3]) { + _predictor('keyed-pk', enter, 0.97, 0.17); + } + for (final s in subs) { + await s.cancel(); + } + await db.close(); +} + +/// A11b shape: 100 streams over 100 owner partitions, 200 random-item writes. +Future _fanout() async { + const owners = 100; + const perOwner = 100; + const writeCount = 200; + final db = await _open('fanout'); + await db.execute( + 'CREATE TABLE items(id INTEGER PRIMARY KEY, owner_id INTEGER NOT NULL, ' + 'value INTEGER)', + ); + await db.execute('CREATE INDEX items_owner ON items(owner_id)'); + await db.executeBatch('INSERT INTO items(owner_id, value) VALUES (?, ?)', [ + for (var o = 1; o <= owners; o++) + for (var r = 0; r < perOwner; r++) [o, 0], + ]); + // Control: the same write burst with nothing subscribed, so the fan-out + // tax is readable as a difference rather than assumed. + { + final rngC = math.Random(0xCAFEF0); + final swC = Stopwatch()..start(); + for (var w = 0; w < writeCount; w++) { + await db.execute('UPDATE items SET value = ? WHERE id = ?', [ + -w - 1, + rngC.nextInt(owners * perOwner) + 1, + ]); + } + swC.stop(); + print( + 'fanout-control (no streams): writes=$writeCount ' + 'wall=${(swC.elapsedMicroseconds / 1000).toStringAsFixed(1)}ms ' + '(${(swC.elapsedMicroseconds / writeCount).toStringAsFixed(1)}us/write)', + ); + } + + final subs = >[]; + final seen = List.filled(owners, false); + for (var o = 0; o < owners; o++) { + final i = o; + subs.add( + db + .stream('SELECT id, value FROM items WHERE owner_id = ? ORDER BY id', [ + i + 1, + ]) + .listen((_) => seen[i] = true), + ); + } + await _waitUntil(() => !seen.contains(false)); + + final rng = math.Random(0xCAFEF0); + _resetCensus(); + final sw = Stopwatch()..start(); + for (var w = 0; w < writeCount; w++) { + await db.execute('UPDATE items SET value = ? WHERE id = ?', [ + w, + rng.nextInt(owners * perOwner) + 1, + ]); + } + await _quiesce(db); + sw.stop(); + _report('fanout', writeCount, owners, sw.elapsedMicroseconds / 1000); + for (final enter in [1, 2, 3]) { + _predictor('fanout', enter, 4.22, 1.95); + } + for (final s in subs) { + await s.cancel(); + } + await db.close(); +} + +/// A6 Part B shape: one latest-50 stream, 100 random like_count writes. +Future _feed() async { + const postCount = 20000; + const writeCount = 100; + final db = await _open('feed'); + await db.execute( + 'CREATE TABLE posts(id INTEGER PRIMARY KEY, created_at INTEGER NOT NULL, ' + 'like_count INTEGER NOT NULL, body TEXT NOT NULL)', + ); + await db.execute('CREATE INDEX posts_created ON posts(created_at, id)'); + await db.executeBatch( + 'INSERT INTO posts(id, created_at, like_count, body) VALUES (?, ?, ?, ?)', + [ + for (var i = 1; i <= postCount; i++) [i, i, 0, 'post body $i'], + ], + ); + var seen = false; + final sub = db + .stream( + 'SELECT id, created_at, like_count, body FROM posts ' + 'ORDER BY created_at DESC, id DESC LIMIT 50', + const [], + ) + .listen((_) => seen = true); + await _waitUntil(() => seen); + + final rng = math.Random(0xFEED); + _resetCensus(); + final sw = Stopwatch()..start(); + for (var w = 0; w < writeCount; w++) { + await db.execute( + 'UPDATE posts SET like_count = like_count + 1 WHERE id = ?', + [rng.nextInt(postCount) + 1], + ); + } + await _quiesce(db); + sw.stop(); + _report('feed', writeCount, 1, sw.elapsedMicroseconds / 1000); + for (final enter in [1, 2, 3]) { + _predictor('feed', enter, 3.91, 3.02); + } + await sub.cancel(); + await db.close(); +} + +/// Let every rerun the write burst scheduled actually run. +Future _quiesce(Database db) async { + var stable = 0; + var last = -1; + while (stable < 5) { + await Future.delayed(const Duration(milliseconds: 4)); + final n = StreamEngine.tmpReruns; + if (n == last) { + stable++; + } else { + stable = 0; + last = n; + } + } +} + +Future _waitUntil( + bool Function() predicate, { + Duration timeout = const Duration(seconds: 120), +}) async { + final deadline = DateTime.now().add(timeout); + while (!predicate()) { + if (DateTime.now().isAfter(deadline)) { + throw StateError('timed out waiting for predicate'); + } + await Future.delayed(const Duration(milliseconds: 1)); + } +} diff --git a/benchmark/experiments/stream_rerun_one_pass_ab.dart b/benchmark/experiments/stream_rerun_one_pass_ab.dart new file mode 100644 index 00000000..5482ade4 --- /dev/null +++ b/benchmark/experiments/stream_rerun_one_pass_ab.dart @@ -0,0 +1,275 @@ +// ignore_for_file: avoid_print + +/// [EXP-283] A/B harness for one-pass decode+hash on changed stream reruns. +/// +/// A dirtied stream re-executes its query on a reader worker so the engine can +/// learn whether its result changed; today that is a hash pass, plus a second +/// step pass whenever the hash moved. This harness measures the shapes where +/// that second pass is and is not paid. +/// +/// Lanes: +/// fanout — 100 streams over 100-row partitions, 200 random writes. +/// ~56% of its reruns change (see `stream_rerun_census.dart`), +/// so this is the primary lane. +/// fanout-wide — 20 streams over 1,000-row partitions: same shape, ~8x the +/// per-rerun step cost, so the mechanism's ceiling is visible. +/// keyed-pk — 50 streams each on one PK, 200 random-PK writes over 10k +/// rows. Under 1% of its reruns change, so the candidate must +/// be inert: this is the miss-tax guard. +/// feed — one latest-50 stream, 100 like_count writes that never +/// touch the watched page. Zero reruns change; second guard. +/// writes — the same write burst with nothing subscribed. Mechanically +/// zero-ceiling, so the collection's own floor reads off it. +/// +/// Run the same command in a baseline and a candidate worktree (exp 249: never +/// A/B stream dispatch with an in-process toggle) and compare medians. +library; + +import 'dart:async'; +import 'dart:io'; +import 'dart:math' as math; + +import 'package:resqlite/resqlite.dart'; + +Future main(List args) async { + var samples = 15; + var warmup = 3; + var lanes = ['fanout', 'fanout-wide', 'keyed-pk', 'feed', 'writes']; + var raw = false; + for (final arg in args) { + if (arg.startsWith('--samples=')) { + samples = int.parse(arg.substring('--samples='.length)); + } else if (arg.startsWith('--warmup=')) { + warmup = int.parse(arg.substring('--warmup='.length)); + } else if (arg.startsWith('--lane=')) { + lanes = [arg.substring('--lane='.length)]; + } else if (arg == '--raw') { + raw = true; + } else { + throw ArgumentError('unknown argument: $arg'); + } + } + for (final lane in lanes) { + final h = await _Lane.open(lane); + try { + for (var i = 0; i < warmup; i++) { + await h.burst(); + } + final samplesMs = []; + for (var i = 0; i < samples; i++) { + samplesMs.add(await h.burst()); + } + final s = [...samplesMs]..sort(); + final n = s.length; + final median = n.isOdd ? s[n ~/ 2] : (s[n ~/ 2 - 1] + s[n ~/ 2]) / 2; + print( + '$lane median=${median.toStringAsFixed(3)}ms ' + 'min=${s.first.toStringAsFixed(3)} max=${s.last.toStringAsFixed(3)} ' + 'n=$n', + ); + if (raw) print('$lane samples_ms=${samplesMs.join(',')}'); + } finally { + await h.close(); + } + } +} + +class _Lane { + _Lane(this.name, this._db, this._writeCount, this._write); + + final String name; + final Database _db; + final int _writeCount; + + /// Issues write [i] of a burst; [i] is globally unique across bursts so no + /// burst can re-write a value a previous one already stored (which would + /// silently turn every rerun into an unchanged one). + final Future Function(Database db, math.Random rng, int i) _write; + + final _subs = >[]; + var _emissions = 0; + var _writeCursor = 0; + + static Future<_Lane> open(String name) async { + switch (name) { + case 'fanout': + return _fanout('fanout', owners: 100, perOwner: 100, writes: 200); + case 'fanout-wide': + return _fanout('fanout-wide', owners: 20, perOwner: 1000, writes: 200); + case 'keyed-pk': + return _keyedPk(); + case 'feed': + return _feed(); + case 'writes': + return _fanout( + 'writes', + owners: 100, + perOwner: 100, + writes: 200, + subscribe: false, + ); + default: + throw ArgumentError('unknown lane: $name'); + } + } + + static Future _openDb(String tag) async { + final dir = await Directory.systemTemp.createTemp('exp283_${tag}_'); + return Database.open('${dir.path}/t.db'); + } + + static Future<_Lane> _fanout( + String name, { + required int owners, + required int perOwner, + required int writes, + bool subscribe = true, + }) async { + final db = await _openDb(name); + await db.execute( + 'CREATE TABLE items(id INTEGER PRIMARY KEY, owner_id INTEGER NOT NULL, ' + 'value INTEGER)', + ); + await db.execute('CREATE INDEX items_owner ON items(owner_id)'); + await db.executeBatch('INSERT INTO items(owner_id, value) VALUES (?, ?)', [ + for (var o = 1; o <= owners; o++) + for (var r = 0; r < perOwner; r++) [o, 0], + ]); + final rows = owners * perOwner; + final lane = _Lane(name, db, writes, (d, rng, i) { + return d.execute('UPDATE items SET value = ? WHERE id = ?', [ + i, + rng.nextInt(rows) + 1, + ]); + }); + if (subscribe) { + await lane._subscribeAll( + owners, + (o) => ( + 'SELECT id, value FROM items WHERE owner_id = ? ORDER BY id', + [o + 1], + ), + ); + } + return lane; + } + + static Future<_Lane> _keyedPk() async { + const rowCount = 10000; + const streamCount = 50; + final db = await _openDb('keyedpk'); + await db.execute( + 'CREATE TABLE items(id INTEGER PRIMARY KEY, value INTEGER, ' + 'label TEXT NOT NULL)', + ); + await db.executeBatch( + 'INSERT INTO items(id, value, label) VALUES (?, ?, ?)', + [ + for (var i = 1; i <= rowCount; i++) [i, i, 'row-$i'], + ], + ); + final lane = _Lane('keyed-pk', db, 200, (d, rng, i) { + return d.execute('UPDATE items SET value = ? WHERE id = ?', [ + i, + rng.nextInt(rowCount) + 1, + ]); + }); + await lane._subscribeAll( + streamCount, + (s) => ('SELECT * FROM items WHERE id = ?', [s + 1]), + ); + return lane; + } + + static Future<_Lane> _feed() async { + const postCount = 20000; + final db = await _openDb('feed'); + await db.execute( + 'CREATE TABLE posts(id INTEGER PRIMARY KEY, created_at INTEGER NOT NULL, ' + 'like_count INTEGER NOT NULL, body TEXT NOT NULL)', + ); + await db.execute('CREATE INDEX posts_created ON posts(created_at, id)'); + await db.executeBatch( + 'INSERT INTO posts(id, created_at, like_count, body) VALUES (?, ?, ?, ?)', + [ + for (var i = 1; i <= postCount; i++) [i, i, 0, 'post body $i'], + ], + ); + // Writes target the oldest half, which the latest-50 page never contains. + final lane = _Lane('feed', db, 100, (d, rng, i) { + return d.execute( + 'UPDATE posts SET like_count = like_count + 1 WHERE id = ?', + [rng.nextInt(postCount ~/ 2) + 1], + ); + }); + await lane._subscribeAll( + 1, + (_) => ( + 'SELECT id, created_at, like_count, body FROM posts ' + 'ORDER BY created_at DESC, id DESC LIMIT 50', + const [], + ), + ); + return lane; + } + + Future _subscribeAll( + int count, + (String, List) Function(int) query, + ) async { + final seen = List.filled(count, false); + for (var i = 0; i < count; i++) { + final k = i; + final (sql, params) = query(k); + _subs.add( + _db.stream(sql, params).listen((_) { + seen[k] = true; + _emissions++; + }), + ); + } + final deadline = DateTime.now().add(const Duration(seconds: 120)); + while (seen.contains(false)) { + if (DateTime.now().isAfter(deadline)) { + throw StateError('$name: timed out waiting for initial emissions'); + } + await Future.delayed(const Duration(milliseconds: 1)); + } + } + + /// One measured burst: issue every write, then wait for the fan-out it + /// triggered to go quiet. Emission count is the settle signal, so both arms + /// stop at the same observable state. + Future burst() async { + final rng = math.Random(0xCAFEF0); + final sw = Stopwatch()..start(); + for (var w = 0; w < _writeCount; w++) { + await _write(_db, rng, _writeCursor++); + } + // The burst ends at the last emission it produced, not at the end of the + // quiet window that proves it: a fixed settle interval would be the same + // constant in both arms and would dilute the delta by its own size. + var mark = sw.elapsedMicroseconds; + var stable = 0; + var last = -1; + while (stable < 4) { + await Future.delayed(const Duration(milliseconds: 3)); + if (_emissions == last) { + stable++; + } else { + stable = 0; + last = _emissions; + mark = sw.elapsedMicroseconds; + } + } + sw.stop(); + return mark / 1000.0; + } + + Future close() async { + for (final s in _subs) { + await s.cancel(); + } + await _db.close(); + } +} diff --git a/benchmark/experiments/stream_rerun_pass_price.dart b/benchmark/experiments/stream_rerun_pass_price.dart new file mode 100644 index 00000000..f9e1b322 --- /dev/null +++ b/benchmark/experiments/stream_rerun_pass_price.dart @@ -0,0 +1,195 @@ +// ignore_for_file: avoid_print +@ffi.DefaultAsset('package:resqlite/src/native/resqlite_bindings.dart') + +/// [EXP-283] Prices the two SQLite passes a *changed* stream rerun makes. +/// +/// `executeQueryIfChanged` hashes the bound statement to completion, and when +/// the hash moves it steps the same statement a second time to build the Dart +/// result. The one-pass decoder `decodeQueryWithInitialHash` (exp 097, shipped +/// for initial stream registration) does both in one step pass. This harness +/// prices all three against each other on stream-shaped queries, with no pool, +/// no isolates and no message hop, so the prize is readable directly. +/// +/// hash — `resqlite_query_hash`, the unchanged-rerun cost +/// decode — `decodeQuery`, the second pass a changed rerun adds +/// hash+decode — what a changed rerun costs today +/// onepass — `decodeQueryWithInitialHash`, what it would cost instead +library; + +import 'dart:ffi' as ffi; +import 'dart:io'; + +import 'package:ffi/ffi.dart'; +import 'package:resqlite/src/native/request_cache.dart'; +import 'package:resqlite/src/native/resqlite_bindings.dart'; +import 'package:resqlite/src/query_decoder.dart'; + +@ffi.Native< + ffi.Pointer Function( + ffi.Pointer, + ffi.Pointer, + ffi.Pointer, + ffi.Int, + ) +>(symbol: 'resqlite_stmt_acquire_writer', isLeaf: true) +external ffi.Pointer resqliteStmtAcquireWriter( + ffi.Pointer db, + ffi.Pointer sql, + ffi.Pointer params, + int paramCount, +); + +@ffi.Native, ffi.Pointer)>( + symbol: 'resqlite_exec', + isLeaf: true, +) +external int resqliteExecRaw( + ffi.Pointer db, + ffi.Pointer sql, +); + +void main(List args) { + var samples = 15; + var iterations = 400; + for (final arg in args) { + if (arg.startsWith('--samples=')) { + samples = int.parse(arg.substring('--samples='.length)); + } else if (arg.startsWith('--iterations=')) { + iterations = int.parse(arg.substring('--iterations='.length)); + } + } + _lane('fanout-100x2', 100, _fanoutSchema, _fanoutSql, samples, iterations); + _lane('feed-50x4', 50, _feedSchema, _feedSql, samples, iterations); + _lane('point-1x3', 1, _pointSchema, _pointSql, samples, iterations); + _lane('wide-1000x2', 1000, _fanoutSchema, _fanoutSql, samples, iterations); +} + +const _fanoutSchema = + 'CREATE TABLE items(id INTEGER PRIMARY KEY, owner_id INTEGER NOT NULL, ' + 'value INTEGER);'; +const _fanoutSql = 'SELECT id, value FROM items ORDER BY id'; + +const _feedSchema = + 'CREATE TABLE items(id INTEGER PRIMARY KEY, owner_id INTEGER NOT NULL, ' + 'value INTEGER, body TEXT NOT NULL);'; +const _feedSql = 'SELECT id, value, body, owner_id FROM items ORDER BY id'; + +const _pointSchema = + 'CREATE TABLE items(id INTEGER PRIMARY KEY, owner_id INTEGER NOT NULL, ' + 'value INTEGER);'; +const _pointSql = 'SELECT id, owner_id, value FROM items WHERE id = 1'; + +void _lane( + String label, + int rows, + String schema, + String sql, + int samples, + int iterations, +) { + final dir = Directory.systemTemp.createTempSync('exp283_pass_'); + final pathNative = '${dir.path}/t.db'.toNativeUtf8(); + final db = resqliteOpen(pathNative, 1, ffi.nullptr.cast()); + calloc.free(pathNative); + if (db == ffi.nullptr) throw StateError('open failed'); + try { + _exec(db, schema); + final insert = StringBuffer('BEGIN;'); + for (var i = 1; i <= rows; i++) { + if (schema == _feedSchema) { + insert.write( + "INSERT INTO items(id, owner_id, value, body) " + "VALUES ($i, ${i % 7}, $i, 'body text for row $i');", + ); + } else { + insert.write( + 'INSERT INTO items(id, owner_id, value) VALUES ($i, ${i % 7}, $i);', + ); + } + } + insert.write('COMMIT;'); + _exec(db, insert.toString()); + + final stmt = resqliteStmtAcquireWriter( + db, + cachedSqlUtf8(sql).cast(), + ffi.nullptr.cast(), + 0, + ); + if (stmt == ffi.nullptr) throw StateError('acquire failed'); + + // Warm the schema cache, the cell buffer and the row-size memory so no + // lane pays a first-execution cost the others do not. + for (var i = 0; i < 50; i++) { + callQueryHash(stmt); + decodeQuery(stmt, sql); + decodeQueryWithInitialHash(stmt, sql); + } + + final hash = []; + final decode = []; + final onepass = []; + for (var s = 0; s < samples; s++) { + // Rotate arm order per sample so drift lands on every arm equally. + final order = [0, 1, 2]; + final rot = s % 3; + for (var r = 0; r < rot; r++) { + order.add(order.removeAt(0)); + } + for (final arm in order) { + final sw = Stopwatch()..start(); + for (var i = 0; i < iterations; i++) { + switch (arm) { + case 0: + callQueryHash(stmt); + case 1: + decodeQuery(stmt, sql); + case 2: + decodeQueryWithInitialHash(stmt, sql); + } + } + sw.stop(); + final us = sw.elapsedMicroseconds / iterations; + switch (arm) { + case 0: + hash.add(us); + case 1: + decode.add(us); + case 2: + onepass.add(us); + } + } + } + final h = _median(hash); + final d = _median(decode); + final o = _median(onepass); + final today = h + d; + print( + '$label rows=$rows ' + 'hash=${h.toStringAsFixed(2)}us ' + 'decode=${d.toStringAsFixed(2)}us ' + 'hash+decode=${today.toStringAsFixed(2)}us ' + 'onepass=${o.toStringAsFixed(2)}us ' + 'saved=${(today - o).toStringAsFixed(2)}us ' + '(${(100 * (today - o) / today).toStringAsFixed(1)}% of a changed rerun) ' + 'miss_tax=${(o - h).toStringAsFixed(2)}us ' + '(${(100 * (o - h) / h).toStringAsFixed(1)}% of an unchanged rerun)', + ); + } finally { + resqliteClose(db); + dir.deleteSync(recursive: true); + } +} + +double _median(List xs) { + final s = [...xs]..sort(); + final n = s.length; + return n.isOdd ? s[n ~/ 2] : (s[n ~/ 2 - 1] + s[n ~/ 2]) / 2; +} + +void _exec(ffi.Pointer db, String sql) { + final native = sql.toNativeUtf8(); + final rc = resqliteExecRaw(db, native.cast()); + calloc.free(native); + if (rc != 0) throw StateError('exec failed ($rc): $sql'); +} diff --git a/lib/src/reader/read_worker.dart b/lib/src/reader/read_worker.dart index e005fea8..3ebeaee0 100644 --- a/lib/src/reader/read_worker.dart +++ b/lib/src/reader/read_worker.dart @@ -81,6 +81,7 @@ final class SelectIfChangedRequest extends ReadRequest { super.parameters, this.lastResultHash, this.lastRowCount, { + this.decodeFirst = false, super.traceCorrelationId, }); final int lastResultHash; @@ -89,6 +90,15 @@ final class SelectIfChangedRequest extends ReadRequest { /// baseline. Compared with the fresh count alongside the result hash; a null /// baseline can never match, so the result is always treated as changed. final int? lastRowCount; + + /// Whether to build the Dart result during the hash pass rather than after + /// it ([EXP-283](../../../experiments/283-stream-rerun-one-pass.md)). + /// + /// Stamped by the stream engine from the previous rerun of this same stream, + /// which is the only place that knows the outcome sequence: a reader worker + /// sees a sample of a stream's reruns and is destroyed outright by the + /// sacrifice path, the same reason [rowHint] is request-carried. + final bool decodeFirst; } /// How large a result's *structure* (rows × columns) can grow before handing @@ -214,6 +224,7 @@ void readerEntrypoint(List args) { :final parameters, :final lastResultHash, :final lastRowCount, + :final decodeFirst, ): // Two-pass selectIfChanged // ([EXP-075](../../../experiments/075-native-hash-selectifchanged.md)). @@ -226,6 +237,7 @@ void readerEntrypoint(List args) { parameters, lastResultHash, lastRowCount, + decodeFirst, request.rowHint, request.initialRowHint, ); @@ -452,6 +464,15 @@ RawQueryResult executeQuery( /// `resqliteQueryHash` resets the stmt on exit, and bindings survive /// reset, so no re-acquire is required. The pass-1 hash is reused as /// the new baseline. +/// +/// [decodeFirst] collapses the two passes into one +/// ([EXP-283](../../../experiments/283-stream-rerun-one-pass.md)). +/// `decodeQueryWithInitialHash` folds the same canonical FNV digest while it +/// fills the result, so a rerun that was going to decode anyway steps SQLite +/// once instead of twice; a rerun that turns out to be unchanged discards the +/// result it built. The digest is the one exp 097 already ships for initial +/// stream registration and is byte-for-byte the hash-only pass's, which is what +/// keeps the stored baseline canonical (exp 228). (int, int, RawQueryResult?) executeQueryIfChanged( int handleAddr, int readerId, @@ -459,9 +480,22 @@ RawQueryResult executeQuery( List parameters, int lastResultHash, int? lastRowCount, [ + bool decodeFirst = false, int rowHint = 0, int initialRowHint = 0, ]) => _withAcquiredStmt(handleAddr, readerId, sql, parameters, (_, stmt) { + if (decodeFirst) { + final (raw, newHash) = decodeQueryWithInitialHash( + stmt, + sql, + rowHint: rowHint, + initialRowHint: initialRowHint, + ); + if (newHash == lastResultHash && raw.rowCount == lastRowCount) { + return (newHash, raw.rowCount, null); + } + return (newHash, raw.rowCount, raw); + } final (newHash, newRowCount) = callQueryHash(stmt); if (newHash == lastResultHash && newRowCount == lastRowCount) { return (newHash, newRowCount, null); diff --git a/lib/src/reader/reader_pool.dart b/lib/src/reader/reader_pool.dart index c6f9a1f1..8e6778e3 100644 --- a/lib/src/reader/reader_pool.dart +++ b/lib/src/reader/reader_pool.dart @@ -237,12 +237,18 @@ final class ReaderPool { /// Execute a re-query with worker-side hash comparison. /// Returns `(rows, newHash, newRowCount)` — `rows` is null when the /// result is unchanged (hash AND row count match). + /// + /// [decodeFirst] asks the worker to build the result during the hash pass + /// instead of after it + /// ([EXP-283](../../../experiments/283-stream-rerun-one-pass.md)); the + /// caller sets it when this stream's previous rerun changed. Future<(List>?, int, int)> selectIfChanged( String sql, List parameters, int lastResultHash, int? lastRowCount, [ int? traceCorrelationId, + bool decodeFirst = false, ]) async { final memory = _rowHints[sql]; final result = await _dispatch( @@ -251,6 +257,7 @@ final class ReaderPool { parameters, lastResultHash, lastRowCount, + decodeFirst: decodeFirst, traceCorrelationId: traceCorrelationId, ), memory, diff --git a/lib/src/stream_engine.dart b/lib/src/stream_engine.dart index 96d777b5..41ddeba7 100644 --- a/lib/src/stream_engine.dart +++ b/lib/src/stream_engine.dart @@ -341,6 +341,13 @@ final class StreamEngine { return subscriberStream; } + /// TEMPORARY [EXP-283] rerun census. Removed before merge. + static int tmpReruns = 0; + static int tmpChanged = 0; + static int tmpRerunUs = 0; + static int tmpChangedUs = 0; + static final List<(int, bool)> tmpTrace = <(int, bool)>[]; + /// Re-query a single stream on the reader pool. Future _requery(StreamEntry entry) async { final traceCorrelationId = entry.pendingTraceCorrelationId; @@ -349,13 +356,24 @@ final class StreamEngine { entry.inFlight = true; entry.dirty = false; + final tmpSw = Stopwatch()..start(); final (rows, newHash, newRowCount) = await _pool.selectIfChanged( entry.sql, entry.params, entry.lastResultHash, entry.lastRowCount, traceCorrelationId, + entry.lastRerunChanged, ); + tmpSw.stop(); + tmpReruns++; + tmpRerunUs += tmpSw.elapsedMicroseconds; + tmpTrace.add((entry.key, rows != null)); + entry.lastRerunChanged = rows != null; + if (rows != null) { + tmpChanged++; + tmpChangedUs += tmpSw.elapsedMicroseconds; + } // If the entry has already been marked dirty again from an invalidation that ocurred // while it was requerying, then this intermediate result should be discarded and instead @@ -488,6 +506,16 @@ final class StreamEntry { /// guarantees the next re-query reports a change and emits. int? lastRowCount; + /// Whether this stream's most recent rerun changed its result + /// ([EXP-283](../../experiments/283-stream-rerun-one-pass.md)). + /// + /// The next rerun uses it to choose between decoding during the hash pass + /// and decoding only if the hash moved. A stream whose reruns change arrives + /// in runs — a partition under active writes changes on rerun after rerun — + /// so the previous outcome is the whole predictor, and the first unchanged + /// rerun disarms it. + bool lastRerunChanged = false; + /// Whether the stream is dirty and needs to be requeried. bool dirty = false; diff --git a/test/query_decoder_test.dart b/test/query_decoder_test.dart index ae30c45c..9427d5ef 100644 --- a/test/query_decoder_test.dart +++ b/test/query_decoder_test.dart @@ -142,6 +142,109 @@ void main() { dir.deleteSync(recursive: true); } }); + + test('decode-first selectIfChanged agrees with the hash-first arm', () { + final dir = Directory.systemTemp.createTempSync('resqlite_decoder_test_'); + final pathNative = '${dir.path}/decodefirst.db'.toNativeUtf8(); + final db = resqliteOpen(pathNative, 1, ffi.nullptr.cast()); + calloc.free(pathNative); + + expect(db, isNot(ffi.nullptr)); + + try { + _exec( + db, + 'CREATE TABLE mixed(' + 'id INTEGER PRIMARY KEY, label TEXT, amount REAL, payload BLOB);' + "INSERT INTO mixed(label, amount, payload) " + "VALUES ('héllo 🚀', 1.25, x'010203FF');" + "INSERT INTO mixed(label, amount, payload) VALUES ('', -3.5, x'');", + ); + + const sql = 'SELECT label, amount, payload FROM mixed ORDER BY id'; + final (_, _, initialHash, initialCount) = executeQueryWithDeps( + db.address, + 0, + sql, + const [], + ); + + // Unchanged, both arms: no result, same canonical baseline. + for (final decodeFirst in [false, true]) { + final (h, c, raw) = executeQueryIfChanged( + db.address, + 0, + sql, + const [], + initialHash, + initialCount, + decodeFirst, + ); + expect(raw, isNull, reason: 'decodeFirst=$decodeFirst'); + expect(h, initialHash, reason: 'decodeFirst=$decodeFirst'); + expect(c, initialCount, reason: 'decodeFirst=$decodeFirst'); + } + + _exec( + db, + "INSERT INTO mixed(label, amount, payload) " + "VALUES ('third', 9.0, x'AA')", + ); + + // Changed, both arms: identical hash, row count and decoded values, so a + // baseline minted by one arm is usable by the other. + final ( + hashFirstHash, + hashFirstCount, + hashFirstRaw, + ) = executeQueryIfChanged( + db.address, + 0, + sql, + const [], + initialHash, + initialCount, + false, + ); + final ( + decodeFirstHash, + decodeFirstCount, + decodeFirstRaw, + ) = executeQueryIfChanged( + db.address, + 0, + sql, + const [], + initialHash, + initialCount, + true, + ); + + expect(hashFirstRaw, isNotNull); + expect(decodeFirstRaw, isNotNull); + expect(decodeFirstHash, hashFirstHash); + expect(decodeFirstCount, hashFirstCount); + expect(decodeFirstCount, 3); + expect(decodeFirstRaw!.values, hashFirstRaw!.values); + expect(decodeFirstRaw.schema.names, hashFirstRaw.schema.names); + + // The decode-first arm's hash is a usable baseline for the hash-first + // arm: a mixed sequence must not re-emit an unchanged result. + final (_, _, afterRaw) = executeQueryIfChanged( + db.address, + 0, + sql, + const [], + decodeFirstHash, + decodeFirstCount, + false, + ); + expect(afterRaw, isNull); + } finally { + resqliteClose(db); + dir.deleteSync(recursive: true); + } + }); } void _exec(ffi.Pointer db, String sql) { From 9ff4bbe4182e9fc73f6690cb833a27cf30ed1e4a Mon Sep 17 00:00:00 2001 From: Dan Reynolds Date: Sun, 6 Sep 2026 07:18:38 -0400 Subject: [PATCH 2/7] Exp 283: one-pass decode+hash for changed stream reruns A dirtied stream re-executes its query so the engine can learn whether its result changed: a hash pass, then a second step pass whenever the hash moved. `decodeQueryWithInitialHash` (exp 097, shipped for initial stream registration) folds the same canonical digest while it builds the result, so a rerun that was going to decode anyway can step SQLite once instead of twice. The stream engine stamps each rerun with whether the previous one changed; the worker decodes during the hash pass only when it did. --- .../experiments/stream_rerun_census.dart | 301 ------------------ lib/src/stream_engine.dart | 16 - 2 files changed, 317 deletions(-) delete mode 100644 benchmark/experiments/stream_rerun_census.dart diff --git a/benchmark/experiments/stream_rerun_census.dart b/benchmark/experiments/stream_rerun_census.dart deleted file mode 100644 index 6ecb434c..00000000 --- a/benchmark/experiments/stream_rerun_census.dart +++ /dev/null @@ -1,301 +0,0 @@ -// ignore_for_file: avoid_print - -/// [EXP-283] Rerun census for the release suite's reactive-stream shapes. -/// -/// The three largest lanes in the release suite are reactive fan-out -/// workloads. Every write dirties some set of streams and each dirty stream -/// re-executes its query on a reader worker so the engine can learn whether -/// its result changed. Nothing has ever counted how many reruns those lanes -/// actually issue, what share of them change, or what a rerun costs — the -/// numbers a candidate that changes rerun work has to be sized against. -/// -/// Reproduces each lane's stream/write shape (scaled to keep setup fast) and -/// reports the census from the main isolate's own view of `selectIfChanged`. -library; - -import 'dart:async'; -import 'dart:io'; -import 'dart:math' as math; - -import 'package:resqlite/resqlite.dart'; - -Future main(List args) async { - var lanes = ['keyed-pk', 'fanout', 'feed']; - for (final arg in args) { - if (arg.startsWith('--lane=')) { - lanes = [arg.substring('--lane='.length)]; - } - } - for (final lane in lanes) { - switch (lane) { - case 'keyed-pk': - await _keyedPk(); - case 'fanout': - await _fanout(); - case 'feed': - await _feed(); - default: - throw ArgumentError('unknown lane: $lane'); - } - } -} - -void _resetCensus() { - StreamEngine.tmpReruns = 0; - StreamEngine.tmpChanged = 0; - StreamEngine.tmpRerunUs = 0; - StreamEngine.tmpChangedUs = 0; - StreamEngine.tmpTrace.clear(); -} - -/// Replay the recorded per-stream outcome sequence through a candidate -/// predictor and report what it would have decided. -/// -/// [enter] is how many consecutive changed reruns arm decode-first; it -/// disarms on the first unchanged rerun. `save` and `tax` are the measured -/// per-rerun figures from `stream_rerun_pass_price.dart` for this shape. -void _predictor(String lane, int enter, double save, double tax) { - final streak = {}; - var fired = 0; - var hits = 0; - var misses = 0; - var missedWins = 0; - for (final (key, changed) in StreamEngine.tmpTrace) { - final armed = (streak[key] ?? 0) >= enter; - if (armed) { - fired++; - if (changed) { - hits++; - } else { - misses++; - } - } else if (changed) { - missedWins++; - } - streak[key] = changed ? (streak[key] ?? 0) + 1 : 0; - } - final net = hits * save - misses * tax; - print( - ' predictor(enter=$enter): fired=$fired hits=$hits misses=$misses ' - 'unarmed-changed=$missedWins ' - 'modelled_net=${net.toStringAsFixed(0)}us ' - '(${(net / StreamEngine.tmpTrace.length).toStringAsFixed(2)}us/rerun)', - ); -} - -void _report(String lane, int writes, int streams, double wallMs) { - final reruns = StreamEngine.tmpReruns; - final changed = StreamEngine.tmpChanged; - final us = StreamEngine.tmpRerunUs; - final changedUs = StreamEngine.tmpChangedUs; - final unchanged = reruns - changed; - final unchangedUs = us - changedUs; - String per(int total, int n) => - n == 0 ? '-' : (total / n).toStringAsFixed(2); - print( - '$lane: streams=$streams writes=$writes ' - 'wall=${wallMs.toStringAsFixed(1)}ms ' - 'reruns=$reruns (${(reruns / writes).toStringAsFixed(1)}/write, ' - 'max ${streams * writes}) ' - 'changed=$changed ' - '(${reruns == 0 ? 0 : (100 * changed / reruns).toStringAsFixed(2)}%) ' - 'rerun_us_mean=${per(us, reruns)} ' - 'changed_us_mean=${per(changedUs, changed)} ' - 'unchanged_us_mean=${per(unchangedUs, unchanged)}', - ); -} - -Future _open(String tag) async { - final dir = await Directory.systemTemp.createTemp('exp283_${tag}_'); - return Database.open('${dir.path}/t.db'); -} - -/// A11 shape: 50 streams each on one PK, 200 random-PK writes over 10k rows. -Future _keyedPk() async { - const rowCount = 10000; - const streamCount = 50; - const writeCount = 200; - final db = await _open('keyedpk'); - await db.execute( - 'CREATE TABLE items(id INTEGER PRIMARY KEY, value INTEGER, ' - 'label TEXT NOT NULL)', - ); - await db.executeBatch( - 'INSERT INTO items(id, value, label) VALUES (?, ?, ?)', - [for (var i = 1; i <= rowCount; i++) [i, i, 'row-$i']], - ); - final subs = >[]; - final seen = List.filled(streamCount, false); - for (var s = 0; s < streamCount; s++) { - final i = s; - subs.add( - db - .stream('SELECT * FROM items WHERE id = ?', [i + 1]) - .listen((_) => seen[i] = true), - ); - } - await _waitUntil(() => !seen.contains(false)); - - final rng = math.Random(0xA11); - _resetCensus(); - final sw = Stopwatch()..start(); - for (var w = 0; w < writeCount; w++) { - await db.execute('UPDATE items SET value = ? WHERE id = ?', [ - w, - rng.nextInt(rowCount) + 1, - ]); - } - await _quiesce(db); - sw.stop(); - _report('keyed-pk', writeCount, streamCount, sw.elapsedMicroseconds / 1000); - for (final enter in [1, 2, 3]) { - _predictor('keyed-pk', enter, 0.97, 0.17); - } - for (final s in subs) { - await s.cancel(); - } - await db.close(); -} - -/// A11b shape: 100 streams over 100 owner partitions, 200 random-item writes. -Future _fanout() async { - const owners = 100; - const perOwner = 100; - const writeCount = 200; - final db = await _open('fanout'); - await db.execute( - 'CREATE TABLE items(id INTEGER PRIMARY KEY, owner_id INTEGER NOT NULL, ' - 'value INTEGER)', - ); - await db.execute('CREATE INDEX items_owner ON items(owner_id)'); - await db.executeBatch('INSERT INTO items(owner_id, value) VALUES (?, ?)', [ - for (var o = 1; o <= owners; o++) - for (var r = 0; r < perOwner; r++) [o, 0], - ]); - // Control: the same write burst with nothing subscribed, so the fan-out - // tax is readable as a difference rather than assumed. - { - final rngC = math.Random(0xCAFEF0); - final swC = Stopwatch()..start(); - for (var w = 0; w < writeCount; w++) { - await db.execute('UPDATE items SET value = ? WHERE id = ?', [ - -w - 1, - rngC.nextInt(owners * perOwner) + 1, - ]); - } - swC.stop(); - print( - 'fanout-control (no streams): writes=$writeCount ' - 'wall=${(swC.elapsedMicroseconds / 1000).toStringAsFixed(1)}ms ' - '(${(swC.elapsedMicroseconds / writeCount).toStringAsFixed(1)}us/write)', - ); - } - - final subs = >[]; - final seen = List.filled(owners, false); - for (var o = 0; o < owners; o++) { - final i = o; - subs.add( - db - .stream('SELECT id, value FROM items WHERE owner_id = ? ORDER BY id', [ - i + 1, - ]) - .listen((_) => seen[i] = true), - ); - } - await _waitUntil(() => !seen.contains(false)); - - final rng = math.Random(0xCAFEF0); - _resetCensus(); - final sw = Stopwatch()..start(); - for (var w = 0; w < writeCount; w++) { - await db.execute('UPDATE items SET value = ? WHERE id = ?', [ - w, - rng.nextInt(owners * perOwner) + 1, - ]); - } - await _quiesce(db); - sw.stop(); - _report('fanout', writeCount, owners, sw.elapsedMicroseconds / 1000); - for (final enter in [1, 2, 3]) { - _predictor('fanout', enter, 4.22, 1.95); - } - for (final s in subs) { - await s.cancel(); - } - await db.close(); -} - -/// A6 Part B shape: one latest-50 stream, 100 random like_count writes. -Future _feed() async { - const postCount = 20000; - const writeCount = 100; - final db = await _open('feed'); - await db.execute( - 'CREATE TABLE posts(id INTEGER PRIMARY KEY, created_at INTEGER NOT NULL, ' - 'like_count INTEGER NOT NULL, body TEXT NOT NULL)', - ); - await db.execute('CREATE INDEX posts_created ON posts(created_at, id)'); - await db.executeBatch( - 'INSERT INTO posts(id, created_at, like_count, body) VALUES (?, ?, ?, ?)', - [ - for (var i = 1; i <= postCount; i++) [i, i, 0, 'post body $i'], - ], - ); - var seen = false; - final sub = db - .stream( - 'SELECT id, created_at, like_count, body FROM posts ' - 'ORDER BY created_at DESC, id DESC LIMIT 50', - const [], - ) - .listen((_) => seen = true); - await _waitUntil(() => seen); - - final rng = math.Random(0xFEED); - _resetCensus(); - final sw = Stopwatch()..start(); - for (var w = 0; w < writeCount; w++) { - await db.execute( - 'UPDATE posts SET like_count = like_count + 1 WHERE id = ?', - [rng.nextInt(postCount) + 1], - ); - } - await _quiesce(db); - sw.stop(); - _report('feed', writeCount, 1, sw.elapsedMicroseconds / 1000); - for (final enter in [1, 2, 3]) { - _predictor('feed', enter, 3.91, 3.02); - } - await sub.cancel(); - await db.close(); -} - -/// Let every rerun the write burst scheduled actually run. -Future _quiesce(Database db) async { - var stable = 0; - var last = -1; - while (stable < 5) { - await Future.delayed(const Duration(milliseconds: 4)); - final n = StreamEngine.tmpReruns; - if (n == last) { - stable++; - } else { - stable = 0; - last = n; - } - } -} - -Future _waitUntil( - bool Function() predicate, { - Duration timeout = const Duration(seconds: 120), -}) async { - final deadline = DateTime.now().add(timeout); - while (!predicate()) { - if (DateTime.now().isAfter(deadline)) { - throw StateError('timed out waiting for predicate'); - } - await Future.delayed(const Duration(milliseconds: 1)); - } -} diff --git a/lib/src/stream_engine.dart b/lib/src/stream_engine.dart index 41ddeba7..1ca9badd 100644 --- a/lib/src/stream_engine.dart +++ b/lib/src/stream_engine.dart @@ -341,13 +341,6 @@ final class StreamEngine { return subscriberStream; } - /// TEMPORARY [EXP-283] rerun census. Removed before merge. - static int tmpReruns = 0; - static int tmpChanged = 0; - static int tmpRerunUs = 0; - static int tmpChangedUs = 0; - static final List<(int, bool)> tmpTrace = <(int, bool)>[]; - /// Re-query a single stream on the reader pool. Future _requery(StreamEntry entry) async { final traceCorrelationId = entry.pendingTraceCorrelationId; @@ -356,7 +349,6 @@ final class StreamEngine { entry.inFlight = true; entry.dirty = false; - final tmpSw = Stopwatch()..start(); final (rows, newHash, newRowCount) = await _pool.selectIfChanged( entry.sql, entry.params, @@ -365,15 +357,7 @@ final class StreamEngine { traceCorrelationId, entry.lastRerunChanged, ); - tmpSw.stop(); - tmpReruns++; - tmpRerunUs += tmpSw.elapsedMicroseconds; - tmpTrace.add((entry.key, rows != null)); entry.lastRerunChanged = rows != null; - if (rows != null) { - tmpChanged++; - tmpChangedUs += tmpSw.elapsedMicroseconds; - } // If the entry has already been marked dirty again from an invalidation that ocurred // while it was requerying, then this intermediate result should be discarded and instead From 26fae0d176e6f102ccbfd878c8c7094b778b6f34 Mon Sep 17 00:00:00 2001 From: Dan Reynolds Date: Sun, 6 Sep 2026 07:30:44 -0400 Subject: [PATCH 3/7] Exp 283: writeup, harnesses, signal entry and index row Adds the two focused harnesses (pass price, order-flipped A/B), the decode-first correctness test, the experiment writeup, its README row fragment and its signal entry, plus retroactive claims 097.1 and 228.1 that the run's correctness rests on. --- .../experiments/stream_rerun_one_pass_ab.dart | 104 +++++--- .../experiments/stream_rerun_pass_price.dart | 1 - experiments/283-stream-rerun-one-pass.md | 251 ++++++++++++++++++ experiments/index/283.json | 7 + experiments/signals/base.json | 12 +- experiments/signals/entries/097.json | 22 ++ experiments/signals/entries/228.json | 8 + experiments/signals/entries/283.json | 76 ++++++ 8 files changed, 438 insertions(+), 43 deletions(-) create mode 100644 experiments/283-stream-rerun-one-pass.md create mode 100644 experiments/index/283.json create mode 100644 experiments/signals/entries/097.json create mode 100644 experiments/signals/entries/283.json diff --git a/benchmark/experiments/stream_rerun_one_pass_ab.dart b/benchmark/experiments/stream_rerun_one_pass_ab.dart index 5482ade4..e8457920 100644 --- a/benchmark/experiments/stream_rerun_one_pass_ab.dart +++ b/benchmark/experiments/stream_rerun_one_pass_ab.dart @@ -7,6 +7,14 @@ /// step pass whenever the hash moved. This harness measures the shapes where /// that second pass is and is not paid. /// +/// Each sample issues its write burst concurrently and then times a sentinel +/// write through to the one stream it is guaranteed to change. Every rerun the +/// burst scheduled — including the unchanged majority, which emits nothing and +/// so cannot be waited on directly — has to clear the queue before the +/// sentinel's own rerun runs, so the sentinel's emission prices the whole +/// backlog. An awaited write-by-write burst cannot: each rerun overlaps the +/// next write's latency and the wall reads as the write burst. +/// /// Lanes: /// fanout — 100 streams over 100-row partitions, 200 random writes. /// ~56% of its reruns change (see `stream_rerun_census.dart`), @@ -75,7 +83,7 @@ Future main(List args) async { } class _Lane { - _Lane(this.name, this._db, this._writeCount, this._write); + _Lane(this.name, this._db, this._writeCount, this._write, this._sentinel); final String name; final Database _db; @@ -86,9 +94,13 @@ class _Lane { /// silently turn every rerun into an unchanged one). final Future Function(Database db, math.Random rng, int i) _write; + /// Issues a write that is guaranteed to change stream 0's result, or null + /// for a lane with no streams. + final Future Function(Database db, int i)? _sentinel; + final _subs = >[]; - var _emissions = 0; var _writeCursor = 0; + Completer? _sentinelWaiter; static Future<_Lane> open(String name) async { switch (name) { @@ -136,12 +148,21 @@ class _Lane { for (var r = 0; r < perOwner; r++) [o, 0], ]); final rows = owners * perOwner; - final lane = _Lane(name, db, writes, (d, rng, i) { - return d.execute('UPDATE items SET value = ? WHERE id = ?', [ + final lane = _Lane( + name, + db, + writes, + (d, rng, i) => d.execute('UPDATE items SET value = ? WHERE id = ?', [ i, rng.nextInt(rows) + 1, - ]); - }); + ]), + subscribe + ? (d, i) => d.execute('UPDATE items SET value = ? WHERE id = ?', [ + -i - 1, + 1, + ]) + : null, + ); if (subscribe) { await lane._subscribeAll( owners, @@ -168,12 +189,17 @@ class _Lane { for (var i = 1; i <= rowCount; i++) [i, i, 'row-$i'], ], ); - final lane = _Lane('keyed-pk', db, 200, (d, rng, i) { - return d.execute('UPDATE items SET value = ? WHERE id = ?', [ + final lane = _Lane( + 'keyed-pk', + db, + 200, + (d, rng, i) => d.execute('UPDATE items SET value = ? WHERE id = ?', [ i, rng.nextInt(rowCount) + 1, - ]); - }); + ]), + (d, i) => + d.execute('UPDATE items SET value = ? WHERE id = ?', [-i - 1, 1]), + ); await lane._subscribeAll( streamCount, (s) => ('SELECT * FROM items WHERE id = ?', [s + 1]), @@ -196,12 +222,20 @@ class _Lane { ], ); // Writes target the oldest half, which the latest-50 page never contains. - final lane = _Lane('feed', db, 100, (d, rng, i) { - return d.execute( + final lane = _Lane( + 'feed', + db, + 100, + (d, rng, i) => d.execute( 'UPDATE posts SET like_count = like_count + 1 WHERE id = ?', [rng.nextInt(postCount ~/ 2) + 1], - ); - }); + ), + // The newest post is on the watched latest-50 page. + (d, i) => d.execute('UPDATE posts SET like_count = ? WHERE id = ?', [ + i + 1, + postCount, + ]), + ); await lane._subscribeAll( 1, (_) => ( @@ -224,7 +258,10 @@ class _Lane { _subs.add( _db.stream(sql, params).listen((_) { seen[k] = true; - _emissions++; + if (k == 0) { + final w = _sentinelWaiter; + if (w != null && !w.isCompleted) w.complete(); + } }), ); } @@ -237,33 +274,26 @@ class _Lane { } } - /// One measured burst: issue every write, then wait for the fan-out it - /// triggered to go quiet. Emission count is the settle signal, so both arms - /// stop at the same observable state. + /// One measured burst: issue every write concurrently, then a sentinel write + /// that must change stream 0, and stop when stream 0 emits. Future burst() async { final rng = math.Random(0xCAFEF0); + final sentinel = _sentinel; + final waiter = sentinel == null + ? null + : (_sentinelWaiter = Completer()); final sw = Stopwatch()..start(); - for (var w = 0; w < _writeCount; w++) { - await _write(_db, rng, _writeCursor++); - } - // The burst ends at the last emission it produced, not at the end of the - // quiet window that proves it: a fixed settle interval would be the same - // constant in both arms and would dilute the delta by its own size. - var mark = sw.elapsedMicroseconds; - var stable = 0; - var last = -1; - while (stable < 4) { - await Future.delayed(const Duration(milliseconds: 3)); - if (_emissions == last) { - stable++; - } else { - stable = 0; - last = _emissions; - mark = sw.elapsedMicroseconds; - } + final writes = >[ + for (var w = 0; w < _writeCount; w++) _write(_db, rng, _writeCursor++), + ]; + await Future.wait(writes); + if (sentinel != null) { + await sentinel(_db, _writeCursor++); + await waiter!.future; } sw.stop(); - return mark / 1000.0; + _sentinelWaiter = null; + return sw.elapsedMicroseconds / 1000.0; } Future close() async { diff --git a/benchmark/experiments/stream_rerun_pass_price.dart b/benchmark/experiments/stream_rerun_pass_price.dart index f9e1b322..7fa11665 100644 --- a/benchmark/experiments/stream_rerun_pass_price.dart +++ b/benchmark/experiments/stream_rerun_pass_price.dart @@ -1,6 +1,5 @@ // ignore_for_file: avoid_print @ffi.DefaultAsset('package:resqlite/src/native/resqlite_bindings.dart') - /// [EXP-283] Prices the two SQLite passes a *changed* stream rerun makes. /// /// `executeQueryIfChanged` hashes the bound statement to completion, and when diff --git a/experiments/283-stream-rerun-one-pass.md b/experiments/283-stream-rerun-one-pass.md new file mode 100644 index 00000000..4a874e65 --- /dev/null +++ b/experiments/283-stream-rerun-one-pass.md @@ -0,0 +1,251 @@ +# Experiment 283: the second walk over the same rows + +**Date:** 2026-09-06 +**Status:** Accepted (in review) +**Direction:** `stream-rerun-dispatch`, `result-transfer-shape` +**Benchmark Run:** PLACEHOLDER + +## Problem + +A reactive stream re-runs its query whenever a write dirties one of the tables +or columns it depends on, and the engine has to decide whether the fresh result +is worth emitting. [Exp 075](075-native-hash-selectifchanged.md) made that +decision cheap for the common case: `resqlite_query_hash` steps the bound +statement to completion in C and folds every cell into an FNV digest, so an +unchanged stream costs one SQLite pass and no Dart objects at all (claim 075.1). + +When the digest *does* move, the same statement is stepped a second time +through `decodeQuery` to build the result. That second walk has been there +since exp 075 and has never been measured. It is not obviously avoidable — +the point of hashing first is that you do not know whether you will need the +result until the hash is finished. + +Except resqlite already has a decoder that does both in one pass. +[Exp 097](097-one-pass-initial-stream-hash.md) added +`resqlite_step_row_hash`, which fills the cell buffer *and* folds the same +masked-FNV accumulator, and `decodeQueryWithInitialHash`, which drives it. That +pair has shipped since April, but only on the initial-registration path: exp 097 +deliberately left reruns on the hash-only path so an unchanged rerun could skip +Dart decoding entirely. [Exp 228](228-canonical-stream-hash.md) later hardened +the digest's contract and named this exact reopening — early rejection may be +revisited if "the changed-result decode can produce the canonical hash without +another full pass." + +It can. The question this experiment asks is whether a stream can be told, in +advance, which of the two shapes its next rerun will be. + +## Hypothesis + +A stream's reruns do not change at random. A partition under active writes +changes on rerun after rerun; a stream nobody is writing to is unchanged for +thousands of reruns in a row. If that autocorrelation is strong, the previous +rerun's outcome is enough of a predictor: when it changed, decode during the +hash pass and skip the second walk; when it did not, keep today's behaviour. + +For that to be worth anything, three things have to hold, and all three were +open questions before this run: + +1. changed reruns have to be a real share of a representative stream workload; +2. the second walk has to be a real share of a changed rerun; +3. the previous outcome has to actually predict the next one. + +## Approach + +`lib/src/stream_engine.dart`, `lib/src/reader/reader_pool.dart` and +`lib/src/reader/read_worker.dart`. No native code, no public API change. + +`StreamEntry` gains one `bool lastRerunChanged`, set from whether the rerun that +just completed returned rows. `ReaderPool.selectIfChanged` forwards it as +`SelectIfChangedRequest.decodeFirst`, and `executeQueryIfChanged` grows a second +arm: when the flag is set it calls `decodeQueryWithInitialHash` — exp 097's +decoder, unchanged — compares the digest it returns, and discards the built +result if the digest and row count both matched after all. + +The flag lives on the main isolate for the reason +[exp 260](260-result-list-presize.md) established for the row hint: a reader +worker sees only the reruns routed to it and is destroyed outright by the +sacrifice path, so a worker-local memory of a stream's outcome sequence would be +sampled and periodically erased. The stream engine sees every rerun of every +stream it owns. + +The predictor is deliberately the smallest one that can work. It arms on a +single changed rerun and disarms on a single unchanged one, so a stream that +alternates never arms, and the worst case for a stream in a run of changes is +one wasted decode at the end of the run. + +Nothing else moves. The hash-only arm, the row-count guard, the canonical +digest stored in `lastResultHash`, and the sacrifice decision are all +unchanged, and the two arms are interchangeable at every point: the digest exp +097's decoder folds is the same value `resqlite_query_hash` returns for the same +rows, which is what keeps a baseline minted by one arm usable by the other. +`test/query_decoder_test.dart` asserts that directly, in both directions. + +## Results + +### 1. What a stream workload's reruns actually look like + +Nothing had ever counted the reruns the release suite's three reactive lanes +issue, or what share of them change. A temporary counter in `_requery` +(preserved with its driver at +[`archive/exp-283-census`](https://github.com/danReynolds/resqlite/tree/archive/exp-283-census), +removed before merge) answered both, on scaled reproductions of each lane's +stream and write shape: + +| lane | reruns | of a possible | changed | +|---|---:|---:|---:| +| high-cardinality fan-out — 100 streams × 100-row partitions, 200 writes | 847 | 20,000 | **56.4%** | +| keyed PK — 50 streams on one PK each, 200 random-PK writes over 10k rows | 1,071 | 10,000 | 0.28% | +| feed — one latest-50 stream, 100 like-count writes | 100 | 100 | 0% | + +Two things in that table were not known. Per-stream coalescing is far more +effective than the suites' own docstrings assume — the fan-out lane issues 847 +reruns where the arithmetic says 20,000, because a stream dirtied again while +its rerun is in flight reruns once more, not once per write. And the fan-out +lane's reruns are **not** overwhelmingly unchanged: 56% of them change. The +"99-of-100 unchanged" reading of that workload describes writes, not reruns. + +The same run priced the fan-out tax: the identical 200-write burst costs +10.6 ms with nothing subscribed and 112.6 ms with the 100 streams attached. + +### 2. What the second walk costs + +`benchmark/experiments/stream_rerun_pass_price.dart`, no pool, no isolates, no +message hop — 15 samples × 400 iterations per arm, arm order rotated per sample: + +| shape | hash | decode | hash+decode (today) | one pass | saved | miss tax | +|---|---:|---:|---:|---:|---:|---:| +| 100 rows × 2 cols | 4.54 | 6.17 | 10.72 | 6.50 | **−4.22 µs (−39.4%)** | +1.95 µs | +| 50 rows × 4 cols, TEXT | 4.33 | 6.93 | 11.27 | 7.36 | **−3.91 µs (−34.7%)** | +3.02 µs | +| 1 row × 3 cols | 1.02 | 1.13 | 2.16 | 1.19 | **−0.97 µs (−44.8%)** | +0.17 µs | +| 1,000 rows × 2 cols | 36.05 | 51.90 | 87.94 | 53.08 | **−34.86 µs (−39.6%)** | +17.03 µs | + +Folding the digest into the decode costs 0.6–1.2 µs; the pass it removes costs +4.5–36 µs. A changed rerun's SQLite-and-decode work falls by roughly two fifths +at every width tried. The miss tax — a decode built and thrown away — is +17–70% of an unchanged rerun, so the predictor is not decoration. + +### 3. Whether the predictor is right + +Measured on the shipping candidate with a temporary hit/miss counter, under the +A/B harness's concurrent write bursts (13 bursts per lane): + +| lane | reruns | changed | armed | hits | misses | +|---|---:|---:|---:|---:|---:| +| `fanout` | 2,699 | 59.8% | 1,534 | 880 | 654 | +| `fanout-wide` | 624 | 80.0% | 479 | 370 | 109 | +| `keyed-pk` | 1,424 | 2.6% | 36 | 16 | 20 | +| `feed` | 54 | 35.2% | 18 | 6 | 12 | + +`feed` changes more often here than in §1 because the A/B harness ends every +burst with a sentinel write that must move the watched page; its 100 ordinary +writes still never do. + +The two guard lanes arm 36 and 18 times across a whole collection — the +predictor is not merely wrong there, it is barely on, which is what makes them +inert rather than merely balanced. Where it is on, it is right 57% and 77% of +the time, against an exchange rate of roughly 2:1 in the saved pass's favour. + +### 4. End to end + +`benchmark/experiments/stream_rerun_one_pass_ab.dart`, two AOT bundles built +from separate worktrees (exp 249: never A/B stream dispatch with an in-process +toggle), one lane per process, 41 samples after 8 warmup, arm order flipped +between passes. Δ is candidate against baseline within each pass; **B**/**C** +marks which arm ran first. + +| lane | 1 B | 2 C | 3 B | 4 C | 5 B | 6 C | 7 B | 8 C | median | pooled 3–8 | +|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:| +| `fanout` | −23.5 | −8.1 | −23.8 | −4.2 | −1.8 | +8.5 | −25.5 | −23.5 | **−15.8%** | **−12.5%** | +| `fanout-wide` | −7.4 | −0.1 | +6.8 | +3.9 | −15.0 | −7.1 | −15.4 | −5.1 | **−6.1%** | **−5.7%** | +| `keyed-pk` (guard) | +0.9 | +8.7 | −2.3 | +0.7 | +1.2 | +3.5 | −4.9 | −2.7 | +0.8% | −1.2% | +| `feed` (guard) | +0.3 | −1.7 | −2.5 | −0.4 | −9.7 | −2.8 | −7.2 | −0.3 | −1.1% | −4.1% | +| `writes` (control) | −2.4 | −1.2 | −0.9 | −4.0 | +7.8 | +4.1 | +6.0 | −4.5 | −1.1% | +0.6% | + +Each sample issues its write burst concurrently and then times a sentinel write +through to the one stream it must change. Every rerun the burst scheduled — +including the unchanged majority, which emits nothing and cannot be waited on +directly — has to clear the queue before the sentinel's own rerun runs, so the +sentinel prices the backlog. An awaited write-by-write burst cannot: each rerun +overlaps the next write's latency and the wall reads as the write burst. That +was the first metric tried here and it resolved nothing. + +`benchmark/ab_drift_check.dart` on the freshest order-flipped pair (7/8): + +| lane | verdict | pass 1 | pass 2 | +|---|---|---:|---:| +| `fanout` | **reproduced** | −25.5% | −23.5% | +| `fanout-wide` | **reproduced** | −15.4% | −5.1% | +| `keyed-pk` | neutral | −4.9% | −2.7% | +| `feed` | neutral | −7.2% | −0.3% | +| `writes` | drift-suspected | +6.0% | −4.5% | + +### 5. The win is larger than the pass price predicts + +Worth stating plainly, because it is the one number here that does not +reconcile. The `fanout` lane issues 208 reruns per burst; §3's hit and miss +counts and §2's per-shape figures put the reader-side saving at +`880 × 4.22 − 654 × 1.95` over 13 bursts, or **188 µs per burst**. The measured +end-to-end saving is `6.170 − 5.400`, or **770 µs per burst** — four times as +much. + +The direction of the discrepancy rules out the obvious explanations: the second +pass in the real worker follows the first over the same rows, so it is warmer +than the tight loop §2 measured, and the model should therefore over-predict. +What it does not model is queueing. The sentinel metric prices a backlog draining +through four workers, and shortening each rerun's service time shortens every +later rerun's wait as well, so a fifth off the service time buys more than a +fifth off the drain. Reader-side work also stops being ~8 µs of a ~21 µs +per-rerun wall in this shape and starts being the part that sets the queue's +service rate. That is a hypothesis, not a measurement — the honest statement is +that the mechanism's size is established by §2 and the win's size by §4, and +the amplification between them is not yet accounted for. + +**Host caveat.** Load average ran 4.3–14.1 across the session — `mediaanalysisd` +held a core for most of it. The zero-ceiling `writes` control is the collection's +own floor and it swings ±8% in the worst pass and ±2% typically, which is why +the verdict rests on eight order-flipped passes and a pooled figure rather than +on any single pass. Passes 5 and 6 are the two where the control moved most and +they are also the two where `fanout` reads weakest; they are reported rather +than dropped. + +## Decision + +**Accepted (in review).** Stream fan-out where reruns change is 12–16% faster +end to end, reproduced across order flips and confirmed by the drift +classifier; the two lanes whose reruns do not change are neutral because the +predictor is structurally off in them, not because two effects cancel. + +The mechanism is bounded and the code is small: one bool on `StreamEntry`, one +field on the request, one branch in `executeQueryIfChanged`, and a decoder that +has been shipping since exp 097. It adds no native code, no public API, and no +new semantics — the digest stored as a stream's baseline is the same canonical +value on both arms, which is the invariant exp 228 was written to protect. + +What would reopen this: a workload whose streams change on isolated single +reruns rather than in runs. A change run of length 1 is the predictor's pure +loss — no hit, one miss — and the exchange rate that makes the arithmetic work +(≈2:1) is not so large that a workload of length-1 runs would survive it. The +`keyed-pk` and `feed` guards do not test that case; they test the case where the +predictor never arms at all. `benchmark/experiments/stream_rerun_one_pass_ab.dart` +is the durable gate. + +## Future work + +- The fan-out tax has two very different readings and neither has been + decomposed. Sequential awaited writes under JIT (§1) show 102 ms of tax over + 847 reruns — 120 µs of wall per rerun. Concurrent bursts under AOT (§4) show + 4.39 ms over 208 reruns — 21 µs per rerun, against roughly 8 µs of + reader-side work. Whichever regime a real application is in, most of a + rerun's wall is still unattributed, and these are the library's largest + benchmark lanes. +- §5's four-fold amplification is the first evidence in this direction that + rerun service time and rerun *wall* are related by more than one-to-one. + If that is queueing, it is a lever: it would mean any reduction in reader-side + rerun work pays back multiplied on a saturated fan-out, and the current + practice of sizing stream candidates against per-rerun cost understates them. + A burst with the pool size swept would settle it. +- The same one-pass idea has an unexamined sibling on the *unchanged* side. + `resqlite_query_hash` walks every cell of every row to prove nothing moved; + a row-count or per-row digest checkpoint that could reject early would be a + different mechanism against the same 4.5–36 µs, but exp 228's invariant means + any such shortcut must produce a non-cacheable sentinel. diff --git a/experiments/index/283.json b/experiments/index/283.json new file mode 100644 index 00000000..b629dc05 --- /dev/null +++ b/experiments/index/283.json @@ -0,0 +1,7 @@ +{ + "file": "283-stream-rerun-one-pass.md", + "title": "The second walk over the same rows", + "impact": "Performance: accepted. A stream rerun hashes its result to decide whether to emit, and re-steps the whole statement a second time whenever the hash moved — a walk that has been there since exp 075 and had never been measured. Exp 097's one-pass decoder already folds the identical canonical digest while it builds the result, so the second walk is removable for any rerun that was going to decode anyway; the missing piece was knowing in advance which reruns those are. A census of the release suite's three reactive lanes supplied it: per-stream coalescing means the high-cardinality fan-out lane issues 847 reruns rather than the 20,000 its arithmetic suggests, and 56% of them change — reruns change in runs, so the previous rerun's outcome predicts the next one. `StreamEntry` now carries one bool that the worker reads to decode during the hash pass instead of after it. Fan-out lanes improve a pooled -12.5% (100 streams x 100-row partitions) and -5.7% (20 streams x 1,000-row partitions) across eight order-flipped AOT passes from separate worktrees, both classified reproduced; the keyed-PK and feed lanes, whose reruns essentially never change, are neutral because the predictor arms 36 and 18 times in a whole collection. Isolated, the removed pass is 35-45% of a changed rerun's SQLite-and-decode work at every width from 1 to 1,000 rows. No native code, no public API, no new semantics: the digest a stream stores as its baseline is the same value on both arms, which is exp 228's invariant.", + "status": "in_review", + "link": "283-stream-rerun-one-pass.md" +} diff --git a/experiments/signals/base.json b/experiments/signals/base.json index f470e211..4a83af45 100644 --- a/experiments/signals/base.json +++ b/experiments/signals/base.json @@ -22,7 +22,7 @@ "directions": [ { "id": "stream-rerun-dispatch", - "status": "low-current-signal", + "status": "active", "subsystems": [ "streaming", "dispatch", @@ -31,14 +31,15 @@ "wal", "checkpoint" ], - "currentRead": "Stream fan-out performance is shaped by rerun scheduling, reader-pool admission, completion-side churn, writer/request residual, and dependency precision. Queue changes have produced both strong wins and sharp regressions. Exp 120 closed the upstream over-dispatch in StreamEngine._flushQueue (parked_total drops 3,590 -> 0 on A11c overlap and 1,198 -> 0 on keyed-PK; max_parked 46 -> 0), and exp 122 removed the remaining stream-admission async boundary by constructing StreamEngine with a concrete ReaderPool. Exp 121 ruled out invalidation traversal as the active implementation target: overlap invalidation is 10-15% of wall, column intersection is 2.5-5.7%, and the structural ceiling is at the per-benchmark decision-threshold edge. Exp 134 proved row-level dirty precision can halve keyed-PK writer-burst wall for a narrow `WHERE id = ?` proof, but its internal SQL recognizer is rejected; revive that area only through explicit API/design or real workload evidence. Exp 136 ships the completion-side reader-handler counter: on A11c overlap the reader worker port handler is 28.57% of total wall (burst + drain) at ~18 us per call across 4,228 calls/burst, and subscriber-fanout emit is only 0.35% of the chain. Exp 147 split writer-side burst wall from SQLite-facing writer calls: on A11c overlap, SQLite is 15.7 ms / 166.8 ms (9.4%), invalidation is 18.8%, and residual writer/request wall is 71.8%; keyed-PK shows the same shape (18.1% SQLite, 18.7% invalidation, 63.3% residual). Exp 148 tested the natural reader-reply batching follow-up and rejected it: the profile smoke cut A11c overlap completion callbacks 4,527 -> 1,425 and completion wall 109.6 ms -> 55.6 ms, but Tracelite measured elapsed stayed neutral/slower (+5.18% high-cardinality, +3.28% many-streams, +13.5% keyed-PK). Exp 151 tested synchronous writer response resolution against the residual writer/request bucket and rejected it: high-cardinality fanout stayed neutral (+2.92%), keyed-PK trended slower with too-noisy evidence (+18.5%), and many-streams writer throughput trended slower (+14.0%). Exp 170 tested the matching request-side variant — `Mutex.tryLock` plus a non-`async` `Writer.execute` / `executeBatch` to drop the uncontended `await _mutex.lock()` microtask hop — and rejected it: the primary Single Inserts (100 sequential) / sequential-awaited (2000 writes) lanes stayed within ±2 % (wrong direction) across paired runs, while the only positive signal (-7.8 % on Concurrent Single Inserts) is a row exp 159 already drives at -58 % to -61 %. SQLite-step tuning, stream admission, invalidation traversal, plain worker-side reader-reply batching, synchronous writer response resolution, synchronous writer request acquisition, SQL-recognizer-based keyed-PK precision, allocation-only `_flushQueue` cleanup, and standalone residual-split profiling are not active targets on currently-measured workloads. Exp 159 attacked the residual structurally: persistent writer reply port + cached SendPort + sync FIFO completion remove fixed per-round-trip scheduling cost, and releasing the write lock at send time pipelines concurrent standalone writes through the worker port FIFO; the focused concurrent-burst benchmark improved 36-45%, exp 147 residual_us dropped on all four audit workloads, and stream-dispatch Tracelite guardrails were neutral on the clean order-flipped pass. Exp 161 closes the release-suite gap by promoting the concurrent-burst shape into `benchmark/suites/writes.dart` as a paired Single Inserts (100 sequential) / Concurrent Single Inserts (100 concurrent) row pair, so exp 159's pipelining win and future writer-scheduling experiments are evaluable on a public release lane (resqlite concurrent median ~1.1 ms vs ~2.9 ms sequential). Exp 171 then tried to apply exp 159's `_sendPort` cache pattern one layer up — a sync-readable `_resolvedRuntime` field on `Database` so post-open hot paths skip the `await _runtime` microtask hop — and rejected it: two order-flipped passes on writer_pipelining.dart produced alternating-sign deltas inside per-round variance (sequential-awaited -2.3%/+2.5%, transaction-guardrail -7.5%/+6.1%), so a ~1-2 us per-call hop sits at or below the harness floor. Database-layer microtask hop trimming is now off the candidate list. Exp 214 tested an even narrower writer-result decode cleanup after adding a microsecond public-write harness: direct pointer reads removed the typed-list/ByteData view around the 16-byte native result struct, but `write_result_direct_read.dart` did not reproduce a stable win across order-flipped passes (pair 1 mixed/small, pair 2 broadly candidate-slower, pair 3 mixed/neutral), so scalar `executeWrite()` result decoding is not an active target. Exp 197 then tested the moonshot version of the named group-commit ceiling by wrapping a coalesced standalone write burst in one SQLite transaction: writer_pipelining.dart's concurrent-burst lane improved -88.1% / -87.3% across order-flipped passes while sequential writes and explicit transactions stayed neutral, but the prototype is rejected as a hidden db.execute() default because it changes independent-autocommit read visibility, crash-window durability, and failure/atomicity semantics. Reopen true group commit only behind an explicit opt-in API or named mode, not another transport micro-path cleanup. Exp 228 found a correctness leak in exp 077's row-count hash shortcut: growth returned a partial digest that was cached as canonical, so the next unchanged rerun decoded and re-emitted once. Removing the shortcut took redundant grow-then-no-op decodes from 9/9 to 0/9 in each of three passes; combined grow + no-op p50 improved 19-36%, while pure-growth timing remained noisy around parity. Exp 239 revisited exp 209's request amortization without public API: bounded batching of only already-parked plain selects improves homogeneous twenty-way point reads 26-33% and roughly-ten-row reads 21-30%, while pool sharding keeps twenty 10k-row reads neutral-to-faster. It is rejected because queue depth cannot encode query cost: alternating large/point bursts regress point-completion p95 11-17% and total median 13-26%. Queue-depth-only hidden batching is closed; reopen only with a reliable private cost signal or independently completable batch members. Exp 249 (moonshot) then tested invalidation-grouped rerun batching — one write dirties every stream projecting the changed column, so it packed the dirtied set into one batched reader message per worker (SelectIfChangedBatchRequest + worker loop + batched _flushQueue/_requeryBatch, plus a lastRowCount cost-gate dispatching large partitions individually), all internal. Rejected: clean cross-worktree A/B measured single-write emission latency SLOWER on every lane (homogeneous +22% p50 / +59% p95, heterogeneous +66% p50 / +52% p95) because a batched reply is indivisible — the one changed stream waits behind its ~24 cheap unchanged batch-mates re-hashes. Message-count amortization is the wrong lever for this latency-bound workload (third rejection of the shared-indivisible-reply trade after exp 148 and exp 239). An in-process A/B toggle had reported a false -27% reproduced win; only cross-worktree exposed the true sign. Exp 271 then tested exp 184's remaining shared-memory completion lever without moving SQLite to the caller: a 24 us atomic-mailbox poll caught 69.9% of eligible completions but AOT no-op wall regressed 84.1%/62.7%, error wall regressed 205.1%/368.4%, and write throughput fell 65.6%/37.3%; a 32 us arm was worse. Active waiting is closed despite the high catch rate. Reopen writer completion transport only for a non-burning wait/notify or park/wake primitive, or representative evidence of a materially larger sequential-write floor.", + "currentRead": "Stream fan-out performance is shaped by rerun scheduling, reader-pool admission, completion-side churn, writer/request residual, and dependency precision. Queue changes have produced both strong wins and sharp regressions. Exp 120 closed the upstream over-dispatch in StreamEngine._flushQueue (parked_total drops 3,590 -> 0 on A11c overlap and 1,198 -> 0 on keyed-PK; max_parked 46 -> 0), and exp 122 removed the remaining stream-admission async boundary by constructing StreamEngine with a concrete ReaderPool. Exp 121 ruled out invalidation traversal as the active implementation target: overlap invalidation is 10-15% of wall, column intersection is 2.5-5.7%, and the structural ceiling is at the per-benchmark decision-threshold edge. Exp 134 proved row-level dirty precision can halve keyed-PK writer-burst wall for a narrow `WHERE id = ?` proof, but its internal SQL recognizer is rejected; revive that area only through explicit API/design or real workload evidence. Exp 136 ships the completion-side reader-handler counter: on A11c overlap the reader worker port handler is 28.57% of total wall (burst + drain) at ~18 us per call across 4,228 calls/burst, and subscriber-fanout emit is only 0.35% of the chain. Exp 147 split writer-side burst wall from SQLite-facing writer calls: on A11c overlap, SQLite is 15.7 ms / 166.8 ms (9.4%), invalidation is 18.8%, and residual writer/request wall is 71.8%; keyed-PK shows the same shape (18.1% SQLite, 18.7% invalidation, 63.3% residual). Exp 148 tested the natural reader-reply batching follow-up and rejected it: the profile smoke cut A11c overlap completion callbacks 4,527 -> 1,425 and completion wall 109.6 ms -> 55.6 ms, but Tracelite measured elapsed stayed neutral/slower (+5.18% high-cardinality, +3.28% many-streams, +13.5% keyed-PK). Exp 151 tested synchronous writer response resolution against the residual writer/request bucket and rejected it: high-cardinality fanout stayed neutral (+2.92%), keyed-PK trended slower with too-noisy evidence (+18.5%), and many-streams writer throughput trended slower (+14.0%). Exp 170 tested the matching request-side variant — `Mutex.tryLock` plus a non-`async` `Writer.execute` / `executeBatch` to drop the uncontended `await _mutex.lock()` microtask hop — and rejected it: the primary Single Inserts (100 sequential) / sequential-awaited (2000 writes) lanes stayed within ±2 % (wrong direction) across paired runs, while the only positive signal (-7.8 % on Concurrent Single Inserts) is a row exp 159 already drives at -58 % to -61 %. SQLite-step tuning, stream admission, invalidation traversal, plain worker-side reader-reply batching, synchronous writer response resolution, synchronous writer request acquisition, SQL-recognizer-based keyed-PK precision, allocation-only `_flushQueue` cleanup, and standalone residual-split profiling are not active targets on currently-measured workloads. Exp 159 attacked the residual structurally: persistent writer reply port + cached SendPort + sync FIFO completion remove fixed per-round-trip scheduling cost, and releasing the write lock at send time pipelines concurrent standalone writes through the worker port FIFO; the focused concurrent-burst benchmark improved 36-45%, exp 147 residual_us dropped on all four audit workloads, and stream-dispatch Tracelite guardrails were neutral on the clean order-flipped pass. Exp 161 closes the release-suite gap by promoting the concurrent-burst shape into `benchmark/suites/writes.dart` as a paired Single Inserts (100 sequential) / Concurrent Single Inserts (100 concurrent) row pair, so exp 159's pipelining win and future writer-scheduling experiments are evaluable on a public release lane (resqlite concurrent median ~1.1 ms vs ~2.9 ms sequential). Exp 171 then tried to apply exp 159's `_sendPort` cache pattern one layer up — a sync-readable `_resolvedRuntime` field on `Database` so post-open hot paths skip the `await _runtime` microtask hop — and rejected it: two order-flipped passes on writer_pipelining.dart produced alternating-sign deltas inside per-round variance (sequential-awaited -2.3%/+2.5%, transaction-guardrail -7.5%/+6.1%), so a ~1-2 us per-call hop sits at or below the harness floor. Database-layer microtask hop trimming is now off the candidate list. Exp 214 tested an even narrower writer-result decode cleanup after adding a microsecond public-write harness: direct pointer reads removed the typed-list/ByteData view around the 16-byte native result struct, but `write_result_direct_read.dart` did not reproduce a stable win across order-flipped passes (pair 1 mixed/small, pair 2 broadly candidate-slower, pair 3 mixed/neutral), so scalar `executeWrite()` result decoding is not an active target. Exp 197 then tested the moonshot version of the named group-commit ceiling by wrapping a coalesced standalone write burst in one SQLite transaction: writer_pipelining.dart's concurrent-burst lane improved -88.1% / -87.3% across order-flipped passes while sequential writes and explicit transactions stayed neutral, but the prototype is rejected as a hidden db.execute() default because it changes independent-autocommit read visibility, crash-window durability, and failure/atomicity semantics. Reopen true group commit only behind an explicit opt-in API or named mode, not another transport micro-path cleanup. Exp 228 found a correctness leak in exp 077's row-count hash shortcut: growth returned a partial digest that was cached as canonical, so the next unchanged rerun decoded and re-emitted once. Removing the shortcut took redundant grow-then-no-op decodes from 9/9 to 0/9 in each of three passes; combined grow + no-op p50 improved 19-36%, while pure-growth timing remained noisy around parity. Exp 239 revisited exp 209's request amortization without public API: bounded batching of only already-parked plain selects improves homogeneous twenty-way point reads 26-33% and roughly-ten-row reads 21-30%, while pool sharding keeps twenty 10k-row reads neutral-to-faster. It is rejected because queue depth cannot encode query cost: alternating large/point bursts regress point-completion p95 11-17% and total median 13-26%. Queue-depth-only hidden batching is closed; reopen only with a reliable private cost signal or independently completable batch members. Exp 249 (moonshot) then tested invalidation-grouped rerun batching — one write dirties every stream projecting the changed column, so it packed the dirtied set into one batched reader message per worker (SelectIfChangedBatchRequest + worker loop + batched _flushQueue/_requeryBatch, plus a lastRowCount cost-gate dispatching large partitions individually), all internal. Rejected: clean cross-worktree A/B measured single-write emission latency SLOWER on every lane (homogeneous +22% p50 / +59% p95, heterogeneous +66% p50 / +52% p95) because a batched reply is indivisible — the one changed stream waits behind its ~24 cheap unchanged batch-mates re-hashes. Message-count amortization is the wrong lever for this latency-bound workload (third rejection of the shared-indivisible-reply trade after exp 148 and exp 239). An in-process A/B toggle had reported a false -27% reproduced win; only cross-worktree exposed the true sign. Exp 271 then tested exp 184's remaining shared-memory completion lever without moving SQLite to the caller: a 24 us atomic-mailbox poll caught 69.9% of eligible completions but AOT no-op wall regressed 84.1%/62.7%, error wall regressed 205.1%/368.4%, and write throughput fell 65.6%/37.3%; a 32 us arm was worse. Active waiting is closed despite the high catch rate. Reopen writer completion transport only for a non-burning wait/notify or park/wake primitive, or representative evidence of a materially larger sequential-write floor. Exp 283 reopens the direction with the first measurement of what a rerun actually is. Per-stream coalescing collapses the high-cardinality shape's 20,000 arithmetic reruns into 847 real ones and 56.4% of those CHANGE (claim 283.1), so the '99 of 100 unchanged' reading describes writes, not reruns; keyed-PK (0.28%) and feed (0%) remain genuinely change-free, giving the direction two regimes rather than one. A changed rerun had been walking its statement twice since exp 075 — hash to completion, then decode from the top — and that second walk is 35-45% of its SQLite-and-decode work at every width from 1 to 1,000 rows (claim 283.2). Exp 097's one-pass decoder already folds the identical canonical digest while it builds the result, and one bool on StreamEntry (did the previous rerun change?) is enough to route to it: reruns change in runs, so the predictor hits 57-77% where it arms and stays structurally off where reruns do not change (claim 283.3). Pooled -12.5% on 100 streams x 100-row partitions and -5.7% on 20 streams x 1,000-row partitions across eight order-flipped AOT passes, guards neutral (claim 283.4). Two things that change how to work here: the end-to-end win is four times its own per-rerun model, so reader-side rerun savings may compound through the queue and the per-rerun denominators used to reject exps 148/239/249-class candidates may understate them; and an awaited write-by-write burst cannot see reader-side rerun cost at all (claim 283.5), which is why exp 283's first metric read neutral on a lane the mechanism moves 12%.", "keyPriors": [ "120", "134", "147", "239", "250", - "271" + "271", + "283" ], "archive": [ "045", @@ -72,7 +73,8 @@ "a trace shows completion-side churn dominating a many-stream workload", "a scheduler policy improves fan-out without harming unrelated reads", "a downstream trace shows synchronous WAL checkpoint wall is frequent and materially user-visible", - "a writer-completion wait/notify or park/wake primitive avoids burning caller-isolate time, or representative sequential-write wall is materially above exp 271's 6-32 us floor" + "a writer-completion wait/notify or park/wake primitive avoids burning caller-isolate time, or representative sequential-write wall is materially above exp 271's 6-32 us floor", + "a candidate reduces reader-side work per rerun on a fan-out that is change-dense (claim 283.1 says the high-cardinality shape is), measured with a backlog-pricing metric rather than an awaited write burst" ], "openQuestions": [ "Can duplicate stream work be coalesced earlier without reintroducing stale delivery or starvation?", @@ -83,7 +85,7 @@ ], "openCandidates": [], "blockedOnMeasurement": [], - "notesForExperimenters": "The 2026-08-13 cadence refresh pruned the two April/May candidates: exp 120 removed stream-induced upstream over-dispatch on the measured A11c and keyed-PK stream lanes, leaving the parked-dispatcher sketch without representative incidence, and exp 249 superseded the generic rerun-batching direction with a durable cross-worktree latency gate. Reopen this direction only when an `interestingIf` trigger supplies a concrete reduction candidate. Avoid assuming a larger reader pool helps; exp 105 found the opposite under A11c fan-out. After exp 118 + exp 120, `dispatcherParkedTotal` and `dispatcherWakeRetryTotal` stay at zero on every measured stream workload — the parked-dispatcher signal is not the active target. Exp 121 took invalidation traversal off the candidate list (10–15% of overlap wall, 2.5–5.7% intersection, 80–200 ns per probe). Exp 122 keeps admission simple by giving StreamEngine a concrete ReaderPool and moving stream registry checks to diagnostics. Exp 134 is proof that row-level precision can win, but its SQL-recognizer implementation is rejected; revive it only with explicit API/design or real workload evidence. Exp 136 showed completion-side reader handling was large enough to try batching, but exp 148 proved plain worker-side reader-reply batching is not mergeable: it reduced callback counters while failing measured-elapsed primary scenarios. Exp 147 still leaves residual writer/request wall as the biggest bucket, but standalone residual splitting has reached diminishing returns. Exp 151 tried the narrow response-side variant (`Completer.sync()` for writer responses) and rejected it under Tracelite. Exp 170 tried the matching request-side variant (`Mutex.tryLock` + non-`async` `Writer.execute` to drop the uncontended `await _mutex.lock()` microtask hop) and rejected it: Single Inserts (100 sequential) and writer_pipelining `sequential-awaited (2000 writes)` both moved <2 % in the wrong direction across paired runs, and the only positive lane (Concurrent Single Inserts, −7.8 %) is one exp 159 already drives at −58 % to −61 %. Do not retry either scheduling tweak without new runtime or workload evidence. Exp 171 tried the same shape one layer up (cached `_resolvedRuntime` on `Database` to skip `await _runtime` on hot paths) and rejected it on focused-harness noise — Database-layer microtask hop trimming above the writer no longer moves the sequential-write floor. Exp 214 adds `benchmark/experiments/write_result_direct_read.dart` as the µs-scale writer-result floor harness and rejects direct native-pointer scalar reads; do not chase `executeWrite()` result decoding or exp 095-style writer result-buffer scratch again without a mechanism larger than view/allocation removal and reproduced order-flipped public-write deltas. Exp 182 took the residual bucket attack in a different direction — skip `preupdate_hook` accumulation + reply harvest when `_streamEngine.length == 0` via a `track_dirty` flag on `resqlite_db` + a `DrainRequest` no-op barrier on first stream registration — and rejected it: the no-stream wins are real (focused sequential −3.8 % / −5.3 %, wide-batch −2.4 % / −5.9 % across order-flipped passes) but the per-call gate adds reproduced overhead on the with-streams shape (focused +2.7 % / +3.7 %, classified `reproduced` by `ab_drift_check.dart`) and Tracelite stream-direction warmup elapsed regressed in the same direction across all three scenarios. Reactive streams are the library's primary use case, so the optimization helps a narrower workload mix than the one it slows — do not retry without a workload that shows write throughput without active streams is a hot path. A new stream experiment should try a concrete reduction candidate, add only the narrow measurement needed to explain the result in that same branch, remove temporary counters before merge unless they are reusable, and clear Tracelite measured-elapsed primary gates without harming keyed-PK. Exp 228 establishes a hash-cache invariant: cached `lastResultHash` must be canonical. An early-reject hash path must return a non-cacheable sentinel or finish the canonical digest during decode; never promote a partial digest to the next rerun baseline. Exp 239 closes queue-depth-only transparent read batching: its homogeneous gains are real, but alternating large/point completion p95 regresses in both orderings because members share an indivisible reply. Keep `select_overflow_batch.dart` as the heterogeneous gate, and fix the reader spawn-versus-close lifecycle before any future design deliberately increases aggregate sacrifice frequency. Exp 249 rejects invalidation-grouped rerun batching (indivisible reply delays the one changed result behind cheap batch-mates; reopen only for a throughput-bound stream workload with a design that preserves independent completion). Two methodology reminders from it: (1) A/B stream-dispatch/reader-message changes ACROSS WORKTREES, never with a single-process toggle — exp 249s in-process toggle classified REPRODUCED (-27%) while cross-worktree showed +22-66%, because both toggle arms shared warm JIT/isolate/pool state. (2) high_cardinality_fanout settle returns after a fixed emission-count quiet window and never waits for suppressed reruns to drain, so it cannot measure the 99-of-100-unchanged fan-out cost; use benchmark/experiments/stream_rerun_latency.dart (single-write per-emit latency, homogeneous + heterogeneous partitions) as the fan-out emission-latency gate." + "notesForExperimenters": "The 2026-08-13 cadence refresh pruned the two April/May candidates: exp 120 removed stream-induced upstream over-dispatch on the measured A11c and keyed-PK stream lanes, leaving the parked-dispatcher sketch without representative incidence, and exp 249 superseded the generic rerun-batching direction with a durable cross-worktree latency gate. Reopen this direction only when an `interestingIf` trigger supplies a concrete reduction candidate. Avoid assuming a larger reader pool helps; exp 105 found the opposite under A11c fan-out. After exp 118 + exp 120, `dispatcherParkedTotal` and `dispatcherWakeRetryTotal` stay at zero on every measured stream workload — the parked-dispatcher signal is not the active target. Exp 121 took invalidation traversal off the candidate list (10–15% of overlap wall, 2.5–5.7% intersection, 80–200 ns per probe). Exp 122 keeps admission simple by giving StreamEngine a concrete ReaderPool and moving stream registry checks to diagnostics. Exp 134 is proof that row-level precision can win, but its SQL-recognizer implementation is rejected; revive it only with explicit API/design or real workload evidence. Exp 136 showed completion-side reader handling was large enough to try batching, but exp 148 proved plain worker-side reader-reply batching is not mergeable: it reduced callback counters while failing measured-elapsed primary scenarios. Exp 147 still leaves residual writer/request wall as the biggest bucket, but standalone residual splitting has reached diminishing returns. Exp 151 tried the narrow response-side variant (`Completer.sync()` for writer responses) and rejected it under Tracelite. Exp 170 tried the matching request-side variant (`Mutex.tryLock` + non-`async` `Writer.execute` to drop the uncontended `await _mutex.lock()` microtask hop) and rejected it: Single Inserts (100 sequential) and writer_pipelining `sequential-awaited (2000 writes)` both moved <2 % in the wrong direction across paired runs, and the only positive lane (Concurrent Single Inserts, −7.8 %) is one exp 159 already drives at −58 % to −61 %. Do not retry either scheduling tweak without new runtime or workload evidence. Exp 171 tried the same shape one layer up (cached `_resolvedRuntime` on `Database` to skip `await _runtime` on hot paths) and rejected it on focused-harness noise — Database-layer microtask hop trimming above the writer no longer moves the sequential-write floor. Exp 214 adds `benchmark/experiments/write_result_direct_read.dart` as the µs-scale writer-result floor harness and rejects direct native-pointer scalar reads; do not chase `executeWrite()` result decoding or exp 095-style writer result-buffer scratch again without a mechanism larger than view/allocation removal and reproduced order-flipped public-write deltas. Exp 182 took the residual bucket attack in a different direction — skip `preupdate_hook` accumulation + reply harvest when `_streamEngine.length == 0` via a `track_dirty` flag on `resqlite_db` + a `DrainRequest` no-op barrier on first stream registration — and rejected it: the no-stream wins are real (focused sequential −3.8 % / −5.3 %, wide-batch −2.4 % / −5.9 % across order-flipped passes) but the per-call gate adds reproduced overhead on the with-streams shape (focused +2.7 % / +3.7 %, classified `reproduced` by `ab_drift_check.dart`) and Tracelite stream-direction warmup elapsed regressed in the same direction across all three scenarios. Reactive streams are the library's primary use case, so the optimization helps a narrower workload mix than the one it slows — do not retry without a workload that shows write throughput without active streams is a hot path. A new stream experiment should try a concrete reduction candidate, add only the narrow measurement needed to explain the result in that same branch, remove temporary counters before merge unless they are reusable, and clear Tracelite measured-elapsed primary gates without harming keyed-PK. Exp 228 establishes a hash-cache invariant: cached `lastResultHash` must be canonical. An early-reject hash path must return a non-cacheable sentinel or finish the canonical digest during decode; never promote a partial digest to the next rerun baseline. Exp 239 closes queue-depth-only transparent read batching: its homogeneous gains are real, but alternating large/point completion p95 regresses in both orderings because members share an indivisible reply. Keep `select_overflow_batch.dart` as the heterogeneous gate, and fix the reader spawn-versus-close lifecycle before any future design deliberately increases aggregate sacrifice frequency. Exp 249 rejects invalidation-grouped rerun batching (indivisible reply delays the one changed result behind cheap batch-mates; reopen only for a throughput-bound stream workload with a design that preserves independent completion). Two methodology reminders from it: (1) A/B stream-dispatch/reader-message changes ACROSS WORKTREES, never with a single-process toggle — exp 249s in-process toggle classified REPRODUCED (-27%) while cross-worktree showed +22-66%, because both toggle arms shared warm JIT/isolate/pool state. (2) high_cardinality_fanout settle returns after a fixed emission-count quiet window and never waits for suppressed reruns to drain, so it cannot measure the 99-of-100-unchanged fan-out cost; use benchmark/experiments/stream_rerun_latency.dart (single-write per-emit latency, homogeneous + heterogeneous partitions) as the fan-out emission-latency gate. Exp 283 adds two harnesses and one measurement rule. `benchmark/experiments/stream_rerun_one_pass_ab.dart` is the gate for anything that changes what a rerun does on the reader side: keep `keyed-pk` and `feed` as predictor-off guards, `writes` as the zero-ceiling control, and keep the sentinel-drain metric — it issues the burst concurrently and times a sentinel write through to the one stream it must change, because the unchanged majority emits nothing and cannot be waited on. `benchmark/experiments/stream_rerun_pass_price.dart` prices the hash pass, the decode pass and the one-pass decoder for a result shape in seconds with no pool and no isolates; run it before proposing anything that changes how many times a rerun steps its statement. And stop dividing stream count by write count when sizing a fan-out candidate — claim 283.1 is the corrected denominator. The unchanged side of the walk is still untouched: `resqlite_query_hash` steps every row to prove nothing moved, which claim 283.2 prices at 4.5-36 us, and claim 228.1 constrains any early-reject shortcut to a non-cacheable sentinel." }, { "id": "parameter-encoding-and-binding", diff --git a/experiments/signals/entries/097.json b/experiments/signals/entries/097.json new file mode 100644 index 00000000..aef1c17b --- /dev/null +++ b/experiments/signals/entries/097.json @@ -0,0 +1,22 @@ +{ + "directions": [ + "stream-rerun-dispatch" + ], + "outcomeClass": "accepted", + "outcomeReason": "performance", + "changedBeliefs": [ + "Initial stream registration decoded the result for subscribers and then replayed the same statement through `resqlite_query_hash` to establish the baseline. `resqlite_step_row_hash` folds the same masked-FNV accumulator while it fills the cell buffer, so `decodeQueryWithInitialHash` produces result and canonical digest in one SQLite pass. Stream setup improved 14-16% on fan-out and subscribe/cancel churn.", + "Stream re-queries were deliberately left on the hash-only path so an unchanged rerun could skip Dart decoding entirely; the one-pass decoder was scoped to initial registration only." + ], + "claims": [ + { + "id": "097.1", + "text": "`decodeQueryWithInitialHash` (driving the native `resqlite_step_row_hash`) produces a result and a result digest in one SQLite step pass, and that digest is bit-identical to the one `resqlite_query_hash` returns for the same rows — the two are interchangeable as a stream's stored baseline. Building the result and the digest together improved initial stream registration 14-16% (fan-out over 10 streams 0.267 -> 0.211 ms; 500 subscribe+cancel cycles 7.646 -> 6.063 ms) with no release-suite regression. Minted retroactively by exp 283, worded from exp 097's own record; exp 097 predates typed claims.", + "conditions": "Release suite plus focused stream setup lanes, April 2026, arm64 macOS. Interchangeability of the two digests is asserted by `test/query_decoder_test.dart`, not only by measurement.", + "edges": [] + } + ], + "nextSignals": [ + "The one-pass decoder was scoped to initial registration. Applying it to reruns needs a way to know in advance whether a rerun will change, since an unchanged rerun that decodes has built a result for nothing." + ] +} diff --git a/experiments/signals/entries/228.json b/experiments/signals/entries/228.json index 21243256..28ae59bf 100644 --- a/experiments/signals/entries/228.json +++ b/experiments/signals/entries/228.json @@ -14,5 +14,13 @@ "Keep `benchmark/experiments/canonical_stream_hash.dart` as the durable gate for native hash or stream-baseline changes. A growth-only lane misses the follow-on decode that made exp 077 incorrect.", "Keep `lastRowCount` as an additional equality guard, but do not let it truncate a digest that will become the next `lastResultHash`. Reopen early rejection only if it returns a distinct non-cacheable sentinel or the changed-result decode can produce the canonical hash without another full pass.", "PR #155's incremental-view prototype is orthogonal: it may reduce how often supported streams reach the fallback, but unsupported queries and ordinary reruns still require the fallback hash to be canonical." + ], + "claims": [ + { + "id": "228.1", + "text": "Any value stored in `StreamEntry.lastResultHash` must be the canonical digest of the complete corresponding result. Exp 077's row-count shortcut returned a prefix-only accumulator that was cached as the next rerun's baseline, so the following identical rerun mismatched, decoded every row and publicly re-emitted unchanged data once — 9 of 9 grow-then-no-op cycles in each of three passes. Removing the shortcut took that to 0 of 9 and improved the combined grow-plus-no-op p50 by 19-36%. An early-reject digest is safe only as a visibly non-cacheable sentinel, or if the changed-result decode can produce the canonical hash without another full pass. Minted retroactively by exp 283, worded from exp 228's own changedBeliefs; exp 228 predates typed claims.", + "conditions": "`benchmark/experiments/canonical_stream_hash.dart`, three focused passes, arm64 macOS, July 2026.", + "edges": [] + } ] } diff --git a/experiments/signals/entries/283.json b/experiments/signals/entries/283.json new file mode 100644 index 00000000..55b51564 --- /dev/null +++ b/experiments/signals/entries/283.json @@ -0,0 +1,76 @@ +{ + "directions": [ + "stream-rerun-dispatch", + "result-transfer-shape" + ], + "outcomeClass": "accepted", + "outcomeReason": "performance", + "changedBeliefs": [ + "We believed a stream rerun's two shapes were priced the way exp 075 left them: an unchanged rerun is one cheap C pass, and a changed one adds a Dart decode. The second half of that was wrong by omission — a changed rerun adds a decode *and* a second complete SQLite walk over the same rows, because the hash pass has already consumed the statement. Claim 283.2 prices that walk at 35-45% of a changed rerun's SQLite-and-decode work at every width from one row to a thousand. It had been there since April 2026 and nobody had measured it, because the two paths were always discussed as 'hash only' versus 'hash plus decode' rather than as one step pass versus two.", + "We believed reactive fan-out reruns were overwhelmingly unchanged — the release suites' own docstrings say so, and `stream_rerun_latency.dart` is built around one changed stream in a hundred. Claim 283.1 shows that describes the *writes*. Per-stream coalescing collapses 20,000 arithmetic reruns into 847 real ones on the high-cardinality shape, and 56% of the survivors change. Anyone reasoning about this direction should stop dividing stream count by write count: the quantity that matters is reruns after coalescing, and it is smaller and much more change-dense than the shape suggests. Keyed-PK and feed remain genuinely change-free (0.28% and 0%), so the direction now has two distinct fan-out regimes rather than one.", + "We believed a rerun-side use of exp 097's one-pass decoder needed foreknowledge nobody had. It needs one bool. Claim 283.3 shows reruns change in runs, so the previous outcome predicts the next one well enough to be net positive against a roughly 2:1 exchange rate, and — the part that makes the guards trustworthy — it stays structurally off on streams that do not change, arming 36 and 18 times in whole collections rather than arming and losing. Exp 228's parting condition, that early rejection reopens if 'the changed-result decode can produce the canonical hash without another full pass', is now satisfied rather than merely stated; claim 228.1 records that condition as a claim so the next candidate can edge to it.", + "A reader of exps 148, 239 and 249 should revisit how those rejections were sized, not their verdicts. Claim 283.4's end-to-end win is four times what its own per-rerun arithmetic predicts, on a metric that prices a backlog draining through four workers. If that gap is queueing, then reader-side rerun savings compound into wait times and the per-rerun denominators this direction has used to reject candidates understate them. Nothing here refutes those experiments — all three failed on latency shape, not on magnitude — but the next runner should sweep pool size across one burst before quoting a per-rerun figure as a ceiling.", + "Claim 283.5 is a measurement rule for this direction, learned the expensive way in this run. An awaited write-by-write fan-out burst cannot see reader-side rerun cost: each rerun hides inside the next write's latency, and the first metric tried here read the candidate as neutral while the mechanism was worth 39% of a changed rerun. Issue the burst concurrently and time a sentinel write through to the one stream it must change. The unchanged majority emits nothing and can never be waited on directly, which is exactly why the backlog has to be priced through something that can." + ], + "claims": [ + { + "id": "283.1", + "text": "Per-stream rerun coalescing dominates the rerun count in reactive fan-out, and the reruns that survive it are not overwhelmingly unchanged. On a scaled reproduction of the high-cardinality fan-out lane (100 streams over 100-row partitions, 200 sequential awaited random writes), the engine issues 847 reruns against the 20,000 that stream-count x write-count implies, and 56.4% of them change. The keyed-PK lane (50 streams each on one PK, 200 random-PK writes over 10,000 rows) issues 1,071 of a possible 10,000 with 0.28% changed, and a one-stream latest-50 feed under 100 like-count writes issues 100 with none changed. The '99 of 100 streams are unchanged' reading of fan-out describes writes, not reruns: a stream dirtied again while its rerun is in flight reruns once more, not once per write.", + "conditions": "JIT, arm64 macOS 26.2 (M1 Pro), Dart 3.12.2, sequential awaited write bursts, scaled reproductions of the release suites' stream and write shapes. Counter and driver preserved at `archive/exp-283-census`.", + "edges": [] + }, + { + "id": "283.2", + "text": "The second SQLite pass a changed stream rerun makes is 35-45% of its SQLite-and-decode work, at every width measured. Timed with no pool, no isolates and no message hop: 100 rows x 2 cols, hash 4.54 us + decode 6.17 us against 6.50 us for the one-pass decoder (-4.22 us, -39.4%); 50 rows x 4 cols with TEXT, 4.33 + 6.93 against 7.36 (-3.91 us, -34.7%); 1 row x 3 cols, 1.02 + 1.13 against 1.19 (-0.97 us, -44.8%); 1,000 rows x 2 cols, 36.05 + 51.90 against 53.08 (-34.86 us, -39.6%). Folding the digest into the decode costs 0.6-1.2 us; the pass it removes costs 4.5-36 us. The converse is the miss tax: a decode built and discarded costs 17-70% of an unchanged rerun.", + "conditions": "`benchmark/experiments/stream_rerun_pass_price.dart`, JIT, 15 samples x 400 iterations per arm with arm order rotated per sample, arm64 macOS 26.2 (M1 Pro), Dart 3.12.2.", + "edges": [ + { + "type": "dependsOn", + "target": "097.1" + } + ] + }, + { + "id": "283.3", + "text": "A stream's reruns change in runs, so the previous rerun's outcome is a usable predictor of the next one and needs no more state than one bool per stream. Measured on the shipping candidate under concurrent write bursts: the 100-stream fan-out lane armed 1,534 times over 2,699 reruns for 880 hits and 654 misses, and the 20-stream 1,000-row lane armed 479 times over 624 reruns for 370 hits and 109 misses. Against claim 283.2's roughly 2:1 exchange rate both are net positive. The two lanes whose reruns essentially never change armed 36 times (keyed-PK, 1,424 reruns) and 18 times (feed, 54 reruns) in a whole collection, so they are inert by construction rather than by two effects cancelling. The predictor's pure loss is a change run of length one: no hit, one miss.", + "conditions": "AOT candidate bundle with a temporary hit/miss counter, 13 concurrent write bursts per lane, arm64 macOS 26.2 (M1 Pro), Dart 3.12.2. Counter removed before merge.", + "edges": [] + }, + { + "id": "283.4", + "text": "Decoding during the hash pass when the previous rerun changed is worth a pooled -12.5% on 100 streams over 100-row partitions and -5.7% on 20 streams over 1,000-row partitions, with the keyed-PK and feed lanes neutral and a zero-ceiling write-only control at +0.6%. Eight order-flipped passes, two AOT bundles from separate worktrees, one lane per process, 41 samples after 8 warmup; `ab_drift_check.dart` classifies both fan-out lanes reproduced on the freshest pair (-25.5%/-23.5% and -15.4%/-5.1%) and the control drift-suspected. The end-to-end saving is about four times what claim 283.2's per-rerun figures and claim 283.3's hit counts predict (770 us per burst against 188 us), which is unexplained; the leading hypothesis is queueing, since the metric prices a backlog draining through four workers and service-time reductions compound into wait times.", + "conditions": "`benchmark/experiments/stream_rerun_one_pass_ab.dart`, AOT, arm64 macOS 26.2 (M1 Pro), Dart 3.12.2, load average 4.3-14.1 across the session; the zero-ceiling control swings +-8% in its worst pass and +-2% typically.", + "edges": [ + { + "type": "dependsOn", + "target": "097.1" + }, + { + "type": "dependsOn", + "target": "228.1" + }, + { + "type": "dependsOn", + "target": "283.2" + }, + { + "type": "dependsOn", + "target": "283.3" + } + ] + }, + { + "id": "283.5", + "text": "A reactive fan-out A/B cannot be read off an awaited write-by-write burst. With each write awaited, every rerun overlaps the next write's latency and the wall reads as the write burst: the first metric tried here put the candidate within +-2% on its primary lane while the mechanism was worth -39% of a changed rerun. Issuing the burst concurrently and then timing a sentinel write through to the one stream it must change prices the whole backlog instead, because every rerun the burst scheduled - including the unchanged majority, which emits nothing and cannot be waited on - clears the queue first. The same eight-pass collection then reads -12.5%.", + "conditions": "Both metrics implemented in `benchmark/experiments/stream_rerun_one_pass_ab.dart`; the awaited variant was replaced, not kept.", + "edges": [] + } + ], + "nextSignals": [ + "`benchmark/experiments/stream_rerun_one_pass_ab.dart` is the durable gate for anything that changes what a stream rerun does on the reader side. Keep `keyed-pk` and `feed` as the predictor-off guards and `writes` as the zero-ceiling control, and keep the sentinel-drain metric: claim 283.5 is the record of what an awaited write burst hides.", + "`benchmark/experiments/stream_rerun_pass_price.dart` prices the hash pass, the decode pass and the one-pass decoder for a given result shape in seconds, with no pool and no isolates. Run it before proposing anything that changes how many times a rerun steps its statement.", + "Do not size a stream candidate by per-rerun cost alone. Claim 283.4's end-to-end win is four times its own per-rerun model, and if that is queueing then every reader-side rerun saving pays back multiplied on a saturated fan-out. Sweeping the reader pool size across the same burst would settle it and would recalibrate every past 'below the round-trip floor' rejection in this direction.", + "The unchanged side of the same walk is untouched. `resqlite_query_hash` steps every row to prove nothing moved, which claim 283.2 prices at 4.5-36 us; claim 228.1 constrains any early-reject shortcut to a non-cacheable sentinel.", + "Reruns that change in isolated single occurrences rather than runs are the predictor's loss case (claim 283.3) and no lane in the suite exercises them. A workload with one write per stream per burst would." + ] +} From 9f398469e497f7036a548916a8fb5074f753ac51 Mon Sep 17 00:00:00 2001 From: Dan Reynolds Date: Sun, 6 Sep 2026 07:42:30 -0400 Subject: [PATCH 4/7] Exp 283: two transferable JOURNAL lessons --- experiments/JOURNAL.md | 44 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 44 insertions(+) diff --git a/experiments/JOURNAL.md b/experiments/JOURNAL.md index c66c9e83..5f24edaf 100644 --- a/experiments/JOURNAL.md +++ b/experiments/JOURNAL.md @@ -1430,6 +1430,50 @@ control span, do not build a sixth harness variation — put twenty lines of shutdown. Four variations failed to resolve this; the in-situ counter took one run. Remove it in the same session: it belongs in the receipt, not in `lib/`.* +### If the work you are timing overlaps the work you are waiting on, you are timing the wrong thing + +Exp 283. A reactive fan-out A/B issued its write burst one awaited write at a +time — the shape the release suite itself uses — and read the candidate as +neutral, ±2%, on the lane its mechanism was worth 39% of a changed rerun on. +The reason is structural: each write's latency covers the reruns the previous +write scheduled, so the reruns never appear in the wall. The wall is the write +burst, and the write burst does not change. + +The fix was to stop waiting on the writes and start waiting on the backlog. +Issue the whole burst concurrently, then issue one sentinel write that must +change one specific stream, and time until that stream emits. Everything the +burst scheduled has to clear the queue first, so the sentinel prices it. Same +code, same collection size: −12.5%, reproduced across an order flip. + +*Reapplies whenever the cost you are chasing runs concurrently with something +you await anyway — background reruns, prefetches, invalidation sweeps, any +work a queue absorbs. Ask what the wall is actually bounded by before trusting +a neutral result; if it is bounded by the thing you await rather than the thing +you changed, no number of samples will help. The general move is to find an +observable event that can only happen after the invisible work has drained, and +time to that. It matters most when the invisible work produces no output of its +own — an unchanged stream rerun emits nothing and can never be waited on +directly, which is exactly why it needs a sentinel behind it.* + +### A saving inside a queue is worth more than the saving + +Exp 283, unresolved but worth carrying. The per-rerun arithmetic — measured +hit and miss counts against a measured per-shape saving — predicted 188 µs per +burst. The A/B measured 770 µs, four times as much, in the direction the model +could not produce by being sloppy: the second pass in the real worker is warmer +than the isolated loop that priced it, so the model should have over-predicted. + +The leading explanation is queueing. The metric prices a backlog draining +through four workers, and shortening each item's service time also shortens +every later item's wait. + +*Reapplies to any component-level estimate used to reject a candidate in a +saturated pipeline. This repo has repeatedly sized stream and read candidates +by per-operation cost and rejected them as "below the round-trip floor"; if the +amplification here is real, those denominators are too small whenever the pool +is the constraint. Cheap to check and nobody has: sweep the pool size across +one burst and see whether the win scales with it.* + ## How to add to this file Add an entry when an experiment surfaces a transferable lesson — something a From e23bc9b3e348804047428e07e3ab00d677a36057 Mon Sep 17 00:00:00 2001 From: Dan Reynolds Date: Sun, 6 Sep 2026 08:34:47 -0400 Subject: [PATCH 5/7] Exp 283: revert the runtime, record why the second walk is not waste The release suite flagged the target fan-out lane at +91.4%. Two facts came out of unpicking it. The lane's wall is quantized by a 200 ms settle window, so the flag is one extra window around ~46 ms of work. And what triggers the extra window is that the candidate emits ~12% more during a write burst: resqlite_query_hash resets on exit, so the hash-first path's second pass opens a fresh WAL snapshot, and the rows a stream emits are fresher than the digest stored beside them. The second walk was buying emission freshness. Trading ~12% more emissions for 12-16% less wall is a semantic-shaped trade in the dimension the library advertises, so the runtime is archived at archive/exp-283 rather than shipped. The census, pass price, predictor accuracy, queueing amplification and lane quantization all stand. --- experiments/283-stream-rerun-one-pass.md | 136 ++++++++++++++++++----- experiments/JOURNAL.md | 26 +++++ experiments/index/283.json | 4 +- experiments/signals/base.json | 5 +- experiments/signals/entries/283.json | 56 +++++++--- lib/src/reader/read_worker.dart | 34 ------ lib/src/reader/reader_pool.dart | 7 -- lib/src/stream_engine.dart | 12 -- test/query_decoder_test.dart | 103 ----------------- 9 files changed, 178 insertions(+), 205 deletions(-) diff --git a/experiments/283-stream-rerun-one-pass.md b/experiments/283-stream-rerun-one-pass.md index 4a874e65..aa145e31 100644 --- a/experiments/283-stream-rerun-one-pass.md +++ b/experiments/283-stream-rerun-one-pass.md @@ -1,9 +1,15 @@ # Experiment 283: the second walk over the same rows **Date:** 2026-09-06 -**Status:** Accepted (in review) -**Direction:** `stream-rerun-dispatch`, `result-transfer-shape` -**Benchmark Run:** PLACEHOLDER +**Status:** Rejected — the mechanism works and the reason it still loses is the finding +**Direction:** `stream-rerun-dispatch`, `result-transfer-shape`, `measurement-system` +**Archive:** [`archive/exp-283`](https://github.com/danReynolds/resqlite/tree/archive/exp-283) + (runtime prototype), [`archive/exp-283-census`](https://github.com/danReynolds/resqlite/tree/archive/exp-283-census) + (rerun census counter and its driver) +**Benchmark Run:** none — the runtime prototype is reverted and no code ships in + `lib/`, `native/` or `hook/`. The decision evidence is the pass-price + decomposition, eight order-flipped separate-binary A/B passes, and the release + suite run that caught the trade-off (§6). ## Problem @@ -49,6 +55,9 @@ open questions before this run: 2. the second walk has to be a real share of a changed rerun; 3. the previous outcome has to actually predict the next one. +A fourth condition went unstated, which is the one that ended up deciding the +experiment: the second walk has to be *only* a cost. It is not (§6). + ## Approach `lib/src/stream_engine.dart`, `lib/src/reader/reader_pool.dart` and @@ -124,6 +133,9 @@ Folding the digest into the decode costs 0.6–1.2 µs; the pass it removes cost at every width tried. The miss tax — a decode built and thrown away — is 17–70% of an unchanged rerun, so the predictor is not decoration. +Read this table with §6: the removed pass is not pure overhead. It re-reads the +database, and re-reading turns out to buy something. + ### 3. Whether the predictor is right Measured on the shipping candidate with a temporary hit/miss counter, under the @@ -208,44 +220,112 @@ on any single pass. Passes 5 and 6 are the two where the control moved most and they are also the two where `fanout` reads weakest; they are reported rather than dropped. +## 6. What the release suite caught + +The headline release run flagged +`High-Cardinality Stream Fan-out (v1) / 100 streams × 200 writes / resqlite` at +**+91.4%** (231.78 → 443.67 ms). The A/B above says that lane's shape is 12–25% +*faster*, so one of the two had to be wrong. + +Neither was. The lane's wall is **quantized**: its settle loop waits a 200 ms +quiet window and stops at the first one with no new emission, so a run where the +measured iteration emits nothing costs one window and a run where it emits +anything costs two. The lane is therefore bimodal at ~246 ms and ~448 ms, and ++91.4% is exactly one settle window — not extra work. + +What decides the mode is whether any emission lands in a measured iteration. +The suite re-seeds its PRNG per iteration, so iterations 2 and 3 rewrite the +values iteration 1 already wrote: the data does not change and the correct +emission count is zero. Running the lane 35 times per arm: + +| arm | slow mode | rate | +|---|---:|---:| +| `origin/main` | 2 / 35 | 5.7% | +| candidate | 10 / 35 | **28.6%** | +| candidate with the decode-first arm compiled out | 1 / 14 | 7.1% | + +The third row is the attribution: wiring the flag through the engine, the pool +and the request while the worker ignores it reproduces the baseline rate, so the +shift is the decoder, not the bookkeeping. + +Instrumenting the harness to compare each post-baseline emission against what +that stream last emitted found **zero** same-content emissions on either arm. +The extra emissions are real changes — and the same instrumentation shows the +candidate makes more of them during the write burst itself: iteration 1's +post-baseline emissions are 91–101 on `origin/main` (six runs) and 102–115 on the +candidate (eight runs), about **+12%**. + +### The second walk was not waste + +That is the finding, and it inverts §2. The hash-first path's two passes do not +read the same database state. `resqlite_query_hash` **resets the statement on +exit**, so the decode pass that follows opens a *fresh* read transaction — a +newer WAL snapshot than the one the hash was computed over. During a burst, +the rows a stream emits are therefore fresher than the digest it stores, and the +stream converges in fewer emissions because each emission has skipped ahead. + +The one-pass decoder takes exactly one snapshot, so hash and rows always agree — +which is the more defensible contract, and is what exp 228's invariant asks for +in spirit — but it gives up the free refresh. The stream needs one more rerun to +catch up, which is where the extra 12% of emissions and the extra settle window +come from. + +So the second SQLite pass is not the pure overhead §2 prices it as. It costs +35–45% of a changed rerun and it buys emission freshness. Nobody knew it was +buying anything, which is why this looked like free money. + ## Decision -**Accepted (in review).** Stream fan-out where reruns change is 12–16% faster -end to end, reproduced across order flips and confirmed by the drift -classifier; the two lanes whose reruns do not change are neutral because the -predictor is structurally off in them, not because two effects cancel. - -The mechanism is bounded and the code is small: one bool on `StreamEntry`, one -field on the request, one branch in `executeQueryIfChanged`, and a decoder that -has been shipping since exp 097. It adds no native code, no public API, and no -new semantics — the digest stored as a stream's baseline is the same canonical -value on both arms, which is the invariant exp 228 was written to protect. - -What would reopen this: a workload whose streams change on isolated single -reruns rather than in runs. A change run of length 1 is the predictor's pure -loss — no hit, one miss — and the exchange rate that makes the arithmetic work -(≈2:1) is not so large that a workload of length-1 runs would survive it. The -`keyed-pk` and `feed` guards do not test that case; they test the case where the -predictor never arms at all. `benchmark/experiments/stream_rerun_one_pass_ab.dart` -is the durable gate. +**Rejected**, and not because the numbers failed. The mechanism is real and +reproduced: 35–45% off a changed rerun's SQLite-and-decode work, 12–16% off a +change-dense fan-out end to end, with both change-free guard lanes provably +inert. What it cannot do is pay for itself in the currency the library +advertises. resqlite's reactive story is that hash suppression keeps streams +from re-emitting; trading roughly 12% more emissions during a write burst for +less wall time is a semantic-shaped trade, and this repo's record on those +(exps 197, 212, 213) is that a reproduced win does not settle them. That call +belongs to a maintainer, not to a scheduled run, so the runtime is archived +rather than shipped. + +Three things would change the answer: + +- **A design that keeps one snapshot per rerun and converges as fast.** The + freshness the two-pass path buys is accidental, not designed. A one-pass + rerun that re-checks cheaply — a row-count or version probe rather than a + second full walk — would take the win without the trade. +- **Evidence that emission count does not matter.** The cost here is more + subscriber deliveries of correct intermediate data during a burst. If a + downstream trace shows burst-time intermediate emissions are cheap or are + coalesced by the UI layer anyway, the trade is one-sided and this ships. +- **A maintainer's ruling** that wall time on change-dense fan-out is worth + ~12% more emissions. + +What does not need re-deriving: §1's census (reruns are far fewer and far more +change-dense than the suites assume), §2's pass price, §3's predictor accuracy, +§5's queueing amplification, and §6's quantization. Those are the run's lasting +contribution and they hold regardless of the verdict. ## Future work +- **The lane cannot resolve anything under 200 ms.** `high_cardinality_fanout`'s + headline wall is one or two settle windows plus ~46 ms of actual work, and + which one it is depends on a race. It has been the release suite's largest + resqlite number for months and most of it is a sleep. Filed separately; any + future stream candidate should be measured on + `benchmark/experiments/stream_rerun_one_pass_ab.dart`'s drain metric instead. - The fan-out tax has two very different readings and neither has been decomposed. Sequential awaited writes under JIT (§1) show 102 ms of tax over 847 reruns — 120 µs of wall per rerun. Concurrent bursts under AOT (§4) show 4.39 ms over 208 reruns — 21 µs per rerun, against roughly 8 µs of reader-side work. Whichever regime a real application is in, most of a - rerun's wall is still unattributed, and these are the library's largest - benchmark lanes. + rerun's wall is still unattributed. - §5's four-fold amplification is the first evidence in this direction that rerun service time and rerun *wall* are related by more than one-to-one. If that is queueing, it is a lever: it would mean any reduction in reader-side rerun work pays back multiplied on a saturated fan-out, and the current practice of sizing stream candidates against per-rerun cost understates them. A burst with the pool size swept would settle it. -- The same one-pass idea has an unexamined sibling on the *unchanged* side. - `resqlite_query_hash` walks every cell of every row to prove nothing moved; - a row-count or per-row digest checkpoint that could reject early would be a - different mechanism against the same 4.5–36 µs, but exp 228's invariant means - any such shortcut must produce a non-cacheable sentinel. +- The unchanged side of the walk is untouched. `resqlite_query_hash` steps every + row to prove nothing moved, which §2 prices at 4.5–36 µs; exp 228's invariant + (claim 228.1) constrains any early-reject shortcut to a non-cacheable + sentinel. A cheap early-reject is also the shape the first bullet above needs. diff --git a/experiments/JOURNAL.md b/experiments/JOURNAL.md index 5f24edaf..ce04b1b4 100644 --- a/experiments/JOURNAL.md +++ b/experiments/JOURNAL.md @@ -1455,6 +1455,32 @@ time to that. It matters most when the invisible work produces no output of its own — an unchanged stream rerun emits nothing and can never be waited on directly, which is exactly why it needs a sentinel behind it.* +### Before removing a redundant pass, ask what it re-reads + +Exp 283. A changed stream rerun walked its SQLite statement twice — hash to +completion, then decode from the top — and the second walk looked like pure +overhead: 35–45% of the rerun's work, removable by a decoder that had been +shipping for months and produced the identical digest. Eight order-flipped A/B +passes said 12–16% faster, guards neutral, correctness tests green. + +The release suite then flagged the target lane at +91%. Unpicking it found the +thing the A/B could not: `resqlite_query_hash` resets the statement on exit, so +the decode pass that follows opens a *fresh* read transaction. The two passes +were reading two different database states, and the second one was quietly +handing the subscriber a newer snapshot than the digest it stored. Collapsing +them into one pass made hash and rows consistent — the better contract — and +cost 12% more emissions during a write burst, because the stream lost its free +refresh and needed another round to converge. + +*Reapplies to any duplicated read on a path where state can change underneath +it: a re-query, a revalidation, a second scan after a first pass. Two passes +over a mutable source are not one pass done twice — they sample at two times, +and the later sample may be load-bearing even when nobody chose it. Before +deduplicating, ask what the second read sees that the first did not, and check +whether anything downstream depends on the difference. A digest-equality test +will not tell you: both arms here produced identical digests for identical +data, and the divergence was in *which* data each arm read.* + ### A saving inside a queue is worth more than the saving Exp 283, unresolved but worth carrying. The per-rerun arithmetic — measured diff --git a/experiments/index/283.json b/experiments/index/283.json index b629dc05..789ec3ec 100644 --- a/experiments/index/283.json +++ b/experiments/index/283.json @@ -1,7 +1,7 @@ { "file": "283-stream-rerun-one-pass.md", "title": "The second walk over the same rows", - "impact": "Performance: accepted. A stream rerun hashes its result to decide whether to emit, and re-steps the whole statement a second time whenever the hash moved — a walk that has been there since exp 075 and had never been measured. Exp 097's one-pass decoder already folds the identical canonical digest while it builds the result, so the second walk is removable for any rerun that was going to decode anyway; the missing piece was knowing in advance which reruns those are. A census of the release suite's three reactive lanes supplied it: per-stream coalescing means the high-cardinality fan-out lane issues 847 reruns rather than the 20,000 its arithmetic suggests, and 56% of them change — reruns change in runs, so the previous rerun's outcome predicts the next one. `StreamEntry` now carries one bool that the worker reads to decode during the hash pass instead of after it. Fan-out lanes improve a pooled -12.5% (100 streams x 100-row partitions) and -5.7% (20 streams x 1,000-row partitions) across eight order-flipped AOT passes from separate worktrees, both classified reproduced; the keyed-PK and feed lanes, whose reruns essentially never change, are neutral because the predictor arms 36 and 18 times in a whole collection. Isolated, the removed pass is 35-45% of a changed rerun's SQLite-and-decode work at every width from 1 to 1,000 rows. No native code, no public API, no new semantics: the digest a stream stores as its baseline is the same value on both arms, which is exp 228's invariant.", - "status": "in_review", + "impact": "Performance: rejected, and the reason is the finding. A stream rerun hashes its result to decide whether to emit and re-steps the whole statement a second time whenever the hash moved -- a walk there since exp 075 that nobody had measured. Exp 097's one-pass decoder already folds the identical canonical digest while building the result, and a census supplied the missing predictor: per-stream coalescing means the high-cardinality fan-out lane issues 847 reruns rather than the 20,000 its arithmetic implies, 56% of them change, and they change in runs, so one bool on `StreamEntry` predicts the next rerun at 57-77% where it arms and stays structurally off on streams that never change. Isolated, the removed pass is 35-45% of a changed rerun at every width from 1 to 1,000 rows; end to end, fan-out improves a pooled -12.5% and -5.7% across eight order-flipped AOT passes, both classified reproduced, with both change-free guard lanes neutral. The release suite then flagged the same fan-out lane at +91.4%, and unpicking that produced the result: the lane's wall is quantized by a 200 ms settle window, so the flag is one extra window, and what triggers it is that the candidate emits about 12% MORE during a write burst. `resqlite_query_hash` resets on exit, so the hash-first path's second pass opens a *fresh* WAL snapshot -- the rows a stream emits are fresher than the digest it stores, and it converges in fewer emissions. The second walk was buying emission freshness, not wasting time. Trading ~12% more emissions for less wall is a semantic-shaped trade in the dimension the library advertises, so the runtime is archived at `archive/exp-283` rather than shipped. The census, the pass price, the predictor accuracy, the queueing amplification and the lane quantization all stand.", + "status": "rejected", "link": "283-stream-rerun-one-pass.md" } diff --git a/experiments/signals/base.json b/experiments/signals/base.json index 4a83af45..c13b9896 100644 --- a/experiments/signals/base.json +++ b/experiments/signals/base.json @@ -31,13 +31,12 @@ "wal", "checkpoint" ], - "currentRead": "Stream fan-out performance is shaped by rerun scheduling, reader-pool admission, completion-side churn, writer/request residual, and dependency precision. Queue changes have produced both strong wins and sharp regressions. Exp 120 closed the upstream over-dispatch in StreamEngine._flushQueue (parked_total drops 3,590 -> 0 on A11c overlap and 1,198 -> 0 on keyed-PK; max_parked 46 -> 0), and exp 122 removed the remaining stream-admission async boundary by constructing StreamEngine with a concrete ReaderPool. Exp 121 ruled out invalidation traversal as the active implementation target: overlap invalidation is 10-15% of wall, column intersection is 2.5-5.7%, and the structural ceiling is at the per-benchmark decision-threshold edge. Exp 134 proved row-level dirty precision can halve keyed-PK writer-burst wall for a narrow `WHERE id = ?` proof, but its internal SQL recognizer is rejected; revive that area only through explicit API/design or real workload evidence. Exp 136 ships the completion-side reader-handler counter: on A11c overlap the reader worker port handler is 28.57% of total wall (burst + drain) at ~18 us per call across 4,228 calls/burst, and subscriber-fanout emit is only 0.35% of the chain. Exp 147 split writer-side burst wall from SQLite-facing writer calls: on A11c overlap, SQLite is 15.7 ms / 166.8 ms (9.4%), invalidation is 18.8%, and residual writer/request wall is 71.8%; keyed-PK shows the same shape (18.1% SQLite, 18.7% invalidation, 63.3% residual). Exp 148 tested the natural reader-reply batching follow-up and rejected it: the profile smoke cut A11c overlap completion callbacks 4,527 -> 1,425 and completion wall 109.6 ms -> 55.6 ms, but Tracelite measured elapsed stayed neutral/slower (+5.18% high-cardinality, +3.28% many-streams, +13.5% keyed-PK). Exp 151 tested synchronous writer response resolution against the residual writer/request bucket and rejected it: high-cardinality fanout stayed neutral (+2.92%), keyed-PK trended slower with too-noisy evidence (+18.5%), and many-streams writer throughput trended slower (+14.0%). Exp 170 tested the matching request-side variant — `Mutex.tryLock` plus a non-`async` `Writer.execute` / `executeBatch` to drop the uncontended `await _mutex.lock()` microtask hop — and rejected it: the primary Single Inserts (100 sequential) / sequential-awaited (2000 writes) lanes stayed within ±2 % (wrong direction) across paired runs, while the only positive signal (-7.8 % on Concurrent Single Inserts) is a row exp 159 already drives at -58 % to -61 %. SQLite-step tuning, stream admission, invalidation traversal, plain worker-side reader-reply batching, synchronous writer response resolution, synchronous writer request acquisition, SQL-recognizer-based keyed-PK precision, allocation-only `_flushQueue` cleanup, and standalone residual-split profiling are not active targets on currently-measured workloads. Exp 159 attacked the residual structurally: persistent writer reply port + cached SendPort + sync FIFO completion remove fixed per-round-trip scheduling cost, and releasing the write lock at send time pipelines concurrent standalone writes through the worker port FIFO; the focused concurrent-burst benchmark improved 36-45%, exp 147 residual_us dropped on all four audit workloads, and stream-dispatch Tracelite guardrails were neutral on the clean order-flipped pass. Exp 161 closes the release-suite gap by promoting the concurrent-burst shape into `benchmark/suites/writes.dart` as a paired Single Inserts (100 sequential) / Concurrent Single Inserts (100 concurrent) row pair, so exp 159's pipelining win and future writer-scheduling experiments are evaluable on a public release lane (resqlite concurrent median ~1.1 ms vs ~2.9 ms sequential). Exp 171 then tried to apply exp 159's `_sendPort` cache pattern one layer up — a sync-readable `_resolvedRuntime` field on `Database` so post-open hot paths skip the `await _runtime` microtask hop — and rejected it: two order-flipped passes on writer_pipelining.dart produced alternating-sign deltas inside per-round variance (sequential-awaited -2.3%/+2.5%, transaction-guardrail -7.5%/+6.1%), so a ~1-2 us per-call hop sits at or below the harness floor. Database-layer microtask hop trimming is now off the candidate list. Exp 214 tested an even narrower writer-result decode cleanup after adding a microsecond public-write harness: direct pointer reads removed the typed-list/ByteData view around the 16-byte native result struct, but `write_result_direct_read.dart` did not reproduce a stable win across order-flipped passes (pair 1 mixed/small, pair 2 broadly candidate-slower, pair 3 mixed/neutral), so scalar `executeWrite()` result decoding is not an active target. Exp 197 then tested the moonshot version of the named group-commit ceiling by wrapping a coalesced standalone write burst in one SQLite transaction: writer_pipelining.dart's concurrent-burst lane improved -88.1% / -87.3% across order-flipped passes while sequential writes and explicit transactions stayed neutral, but the prototype is rejected as a hidden db.execute() default because it changes independent-autocommit read visibility, crash-window durability, and failure/atomicity semantics. Reopen true group commit only behind an explicit opt-in API or named mode, not another transport micro-path cleanup. Exp 228 found a correctness leak in exp 077's row-count hash shortcut: growth returned a partial digest that was cached as canonical, so the next unchanged rerun decoded and re-emitted once. Removing the shortcut took redundant grow-then-no-op decodes from 9/9 to 0/9 in each of three passes; combined grow + no-op p50 improved 19-36%, while pure-growth timing remained noisy around parity. Exp 239 revisited exp 209's request amortization without public API: bounded batching of only already-parked plain selects improves homogeneous twenty-way point reads 26-33% and roughly-ten-row reads 21-30%, while pool sharding keeps twenty 10k-row reads neutral-to-faster. It is rejected because queue depth cannot encode query cost: alternating large/point bursts regress point-completion p95 11-17% and total median 13-26%. Queue-depth-only hidden batching is closed; reopen only with a reliable private cost signal or independently completable batch members. Exp 249 (moonshot) then tested invalidation-grouped rerun batching — one write dirties every stream projecting the changed column, so it packed the dirtied set into one batched reader message per worker (SelectIfChangedBatchRequest + worker loop + batched _flushQueue/_requeryBatch, plus a lastRowCount cost-gate dispatching large partitions individually), all internal. Rejected: clean cross-worktree A/B measured single-write emission latency SLOWER on every lane (homogeneous +22% p50 / +59% p95, heterogeneous +66% p50 / +52% p95) because a batched reply is indivisible — the one changed stream waits behind its ~24 cheap unchanged batch-mates re-hashes. Message-count amortization is the wrong lever for this latency-bound workload (third rejection of the shared-indivisible-reply trade after exp 148 and exp 239). An in-process A/B toggle had reported a false -27% reproduced win; only cross-worktree exposed the true sign. Exp 271 then tested exp 184's remaining shared-memory completion lever without moving SQLite to the caller: a 24 us atomic-mailbox poll caught 69.9% of eligible completions but AOT no-op wall regressed 84.1%/62.7%, error wall regressed 205.1%/368.4%, and write throughput fell 65.6%/37.3%; a 32 us arm was worse. Active waiting is closed despite the high catch rate. Reopen writer completion transport only for a non-burning wait/notify or park/wake primitive, or representative evidence of a materially larger sequential-write floor. Exp 283 reopens the direction with the first measurement of what a rerun actually is. Per-stream coalescing collapses the high-cardinality shape's 20,000 arithmetic reruns into 847 real ones and 56.4% of those CHANGE (claim 283.1), so the '99 of 100 unchanged' reading describes writes, not reruns; keyed-PK (0.28%) and feed (0%) remain genuinely change-free, giving the direction two regimes rather than one. A changed rerun had been walking its statement twice since exp 075 — hash to completion, then decode from the top — and that second walk is 35-45% of its SQLite-and-decode work at every width from 1 to 1,000 rows (claim 283.2). Exp 097's one-pass decoder already folds the identical canonical digest while it builds the result, and one bool on StreamEntry (did the previous rerun change?) is enough to route to it: reruns change in runs, so the predictor hits 57-77% where it arms and stays structurally off where reruns do not change (claim 283.3). Pooled -12.5% on 100 streams x 100-row partitions and -5.7% on 20 streams x 1,000-row partitions across eight order-flipped AOT passes, guards neutral (claim 283.4). Two things that change how to work here: the end-to-end win is four times its own per-rerun model, so reader-side rerun savings may compound through the queue and the per-rerun denominators used to reject exps 148/239/249-class candidates may understate them; and an awaited write-by-write burst cannot see reader-side rerun cost at all (claim 283.5), which is why exp 283's first metric read neutral on a lane the mechanism moves 12%.", + "currentRead": "Stream fan-out performance is shaped by rerun scheduling, reader-pool admission, completion-side churn, writer/request residual, and dependency precision. Queue changes have produced both strong wins and sharp regressions. Exp 120 closed the upstream over-dispatch in StreamEngine._flushQueue (parked_total drops 3,590 -> 0 on A11c overlap and 1,198 -> 0 on keyed-PK; max_parked 46 -> 0), and exp 122 removed the remaining stream-admission async boundary by constructing StreamEngine with a concrete ReaderPool. Exp 121 ruled out invalidation traversal as the active implementation target: overlap invalidation is 10-15% of wall, column intersection is 2.5-5.7%, and the structural ceiling is at the per-benchmark decision-threshold edge. Exp 134 proved row-level dirty precision can halve keyed-PK writer-burst wall for a narrow `WHERE id = ?` proof, but its internal SQL recognizer is rejected; revive that area only through explicit API/design or real workload evidence. Exp 136 ships the completion-side reader-handler counter: on A11c overlap the reader worker port handler is 28.57% of total wall (burst + drain) at ~18 us per call across 4,228 calls/burst, and subscriber-fanout emit is only 0.35% of the chain. Exp 147 split writer-side burst wall from SQLite-facing writer calls: on A11c overlap, SQLite is 15.7 ms / 166.8 ms (9.4%), invalidation is 18.8%, and residual writer/request wall is 71.8%; keyed-PK shows the same shape (18.1% SQLite, 18.7% invalidation, 63.3% residual). Exp 148 tested the natural reader-reply batching follow-up and rejected it: the profile smoke cut A11c overlap completion callbacks 4,527 -> 1,425 and completion wall 109.6 ms -> 55.6 ms, but Tracelite measured elapsed stayed neutral/slower (+5.18% high-cardinality, +3.28% many-streams, +13.5% keyed-PK). Exp 151 tested synchronous writer response resolution against the residual writer/request bucket and rejected it: high-cardinality fanout stayed neutral (+2.92%), keyed-PK trended slower with too-noisy evidence (+18.5%), and many-streams writer throughput trended slower (+14.0%). Exp 170 tested the matching request-side variant — `Mutex.tryLock` plus a non-`async` `Writer.execute` / `executeBatch` to drop the uncontended `await _mutex.lock()` microtask hop — and rejected it: the primary Single Inserts (100 sequential) / sequential-awaited (2000 writes) lanes stayed within ±2 % (wrong direction) across paired runs, while the only positive signal (-7.8 % on Concurrent Single Inserts) is a row exp 159 already drives at -58 % to -61 %. SQLite-step tuning, stream admission, invalidation traversal, plain worker-side reader-reply batching, synchronous writer response resolution, synchronous writer request acquisition, SQL-recognizer-based keyed-PK precision, allocation-only `_flushQueue` cleanup, and standalone residual-split profiling are not active targets on currently-measured workloads. Exp 159 attacked the residual structurally: persistent writer reply port + cached SendPort + sync FIFO completion remove fixed per-round-trip scheduling cost, and releasing the write lock at send time pipelines concurrent standalone writes through the worker port FIFO; the focused concurrent-burst benchmark improved 36-45%, exp 147 residual_us dropped on all four audit workloads, and stream-dispatch Tracelite guardrails were neutral on the clean order-flipped pass. Exp 161 closes the release-suite gap by promoting the concurrent-burst shape into `benchmark/suites/writes.dart` as a paired Single Inserts (100 sequential) / Concurrent Single Inserts (100 concurrent) row pair, so exp 159's pipelining win and future writer-scheduling experiments are evaluable on a public release lane (resqlite concurrent median ~1.1 ms vs ~2.9 ms sequential). Exp 171 then tried to apply exp 159's `_sendPort` cache pattern one layer up — a sync-readable `_resolvedRuntime` field on `Database` so post-open hot paths skip the `await _runtime` microtask hop — and rejected it: two order-flipped passes on writer_pipelining.dart produced alternating-sign deltas inside per-round variance (sequential-awaited -2.3%/+2.5%, transaction-guardrail -7.5%/+6.1%), so a ~1-2 us per-call hop sits at or below the harness floor. Database-layer microtask hop trimming is now off the candidate list. Exp 214 tested an even narrower writer-result decode cleanup after adding a microsecond public-write harness: direct pointer reads removed the typed-list/ByteData view around the 16-byte native result struct, but `write_result_direct_read.dart` did not reproduce a stable win across order-flipped passes (pair 1 mixed/small, pair 2 broadly candidate-slower, pair 3 mixed/neutral), so scalar `executeWrite()` result decoding is not an active target. Exp 197 then tested the moonshot version of the named group-commit ceiling by wrapping a coalesced standalone write burst in one SQLite transaction: writer_pipelining.dart's concurrent-burst lane improved -88.1% / -87.3% across order-flipped passes while sequential writes and explicit transactions stayed neutral, but the prototype is rejected as a hidden db.execute() default because it changes independent-autocommit read visibility, crash-window durability, and failure/atomicity semantics. Reopen true group commit only behind an explicit opt-in API or named mode, not another transport micro-path cleanup. Exp 228 found a correctness leak in exp 077's row-count hash shortcut: growth returned a partial digest that was cached as canonical, so the next unchanged rerun decoded and re-emitted once. Removing the shortcut took redundant grow-then-no-op decodes from 9/9 to 0/9 in each of three passes; combined grow + no-op p50 improved 19-36%, while pure-growth timing remained noisy around parity. Exp 239 revisited exp 209's request amortization without public API: bounded batching of only already-parked plain selects improves homogeneous twenty-way point reads 26-33% and roughly-ten-row reads 21-30%, while pool sharding keeps twenty 10k-row reads neutral-to-faster. It is rejected because queue depth cannot encode query cost: alternating large/point bursts regress point-completion p95 11-17% and total median 13-26%. Queue-depth-only hidden batching is closed; reopen only with a reliable private cost signal or independently completable batch members. Exp 249 (moonshot) then tested invalidation-grouped rerun batching — one write dirties every stream projecting the changed column, so it packed the dirtied set into one batched reader message per worker (SelectIfChangedBatchRequest + worker loop + batched _flushQueue/_requeryBatch, plus a lastRowCount cost-gate dispatching large partitions individually), all internal. Rejected: clean cross-worktree A/B measured single-write emission latency SLOWER on every lane (homogeneous +22% p50 / +59% p95, heterogeneous +66% p50 / +52% p95) because a batched reply is indivisible — the one changed stream waits behind its ~24 cheap unchanged batch-mates re-hashes. Message-count amortization is the wrong lever for this latency-bound workload (third rejection of the shared-indivisible-reply trade after exp 148 and exp 239). An in-process A/B toggle had reported a false -27% reproduced win; only cross-worktree exposed the true sign. Exp 271 then tested exp 184's remaining shared-memory completion lever without moving SQLite to the caller: a 24 us atomic-mailbox poll caught 69.9% of eligible completions but AOT no-op wall regressed 84.1%/62.7%, error wall regressed 205.1%/368.4%, and write throughput fell 65.6%/37.3%; a 32 us arm was worse. Active waiting is closed despite the high catch rate. Reopen writer completion transport only for a non-burning wait/notify or park/wake primitive, or representative evidence of a materially larger sequential-write floor. Exp 283 reopens the direction with the first measurement of what a rerun actually is, and closes its own candidate on what that measurement turned up. Per-stream coalescing collapses the high-cardinality shape's 20,000 arithmetic reruns into 847 real ones and 56.4% of those CHANGE (claim 283.1), so the '99 of 100 unchanged' reading describes writes, not reruns; keyed-PK (0.28%) and feed (0%) remain genuinely change-free, giving the direction two regimes rather than one. A changed rerun had been walking its statement twice since exp 075, and that second walk is 35-45% of its SQLite-and-decode work at every width from 1 to 1,000 rows (claim 283.2). Routing it through exp 097's one-pass decoder, predicted by one bool on StreamEntry (did the previous rerun change?), is worth a pooled -12.5% and -5.7% on two fan-out shapes across eight order-flipped AOT passes with both change-free guards neutral (claim 283.4). It is REJECTED anyway, because the second walk was not waste: `resqlite_query_hash` resets on exit, so the decode that follows opens a FRESH WAL snapshot, the rows a stream emits are newer than the digest stored beside them, and the stream converges in fewer emissions as a result. One-pass costs ~12% more emissions during a burst (claim 283.6) - a semantic-shaped trade in the dimension the library advertises. Reopen with a cheap freshness re-check instead of a second full walk, or with evidence that burst-time intermediate emissions are immaterial. Two measurement facts outlive the verdict: the end-to-end win is four times its own per-rerun model, so reader-side rerun savings may compound through the queue and the denominators used to reject exps 148/239/249-class candidates may understate them; and `high_cardinality_fanout`'s headline wall is one or two 200 ms settle windows around ~46 ms of work, bimodal at ~246 and ~448 ms on a convergence race, so a +91% flag on it is one window and the lane cannot size a candidate at all (claim 283.7).", "keyPriors": [ "120", "134", "147", "239", - "250", "271", "283" ], @@ -85,7 +84,7 @@ ], "openCandidates": [], "blockedOnMeasurement": [], - "notesForExperimenters": "The 2026-08-13 cadence refresh pruned the two April/May candidates: exp 120 removed stream-induced upstream over-dispatch on the measured A11c and keyed-PK stream lanes, leaving the parked-dispatcher sketch without representative incidence, and exp 249 superseded the generic rerun-batching direction with a durable cross-worktree latency gate. Reopen this direction only when an `interestingIf` trigger supplies a concrete reduction candidate. Avoid assuming a larger reader pool helps; exp 105 found the opposite under A11c fan-out. After exp 118 + exp 120, `dispatcherParkedTotal` and `dispatcherWakeRetryTotal` stay at zero on every measured stream workload — the parked-dispatcher signal is not the active target. Exp 121 took invalidation traversal off the candidate list (10–15% of overlap wall, 2.5–5.7% intersection, 80–200 ns per probe). Exp 122 keeps admission simple by giving StreamEngine a concrete ReaderPool and moving stream registry checks to diagnostics. Exp 134 is proof that row-level precision can win, but its SQL-recognizer implementation is rejected; revive it only with explicit API/design or real workload evidence. Exp 136 showed completion-side reader handling was large enough to try batching, but exp 148 proved plain worker-side reader-reply batching is not mergeable: it reduced callback counters while failing measured-elapsed primary scenarios. Exp 147 still leaves residual writer/request wall as the biggest bucket, but standalone residual splitting has reached diminishing returns. Exp 151 tried the narrow response-side variant (`Completer.sync()` for writer responses) and rejected it under Tracelite. Exp 170 tried the matching request-side variant (`Mutex.tryLock` + non-`async` `Writer.execute` to drop the uncontended `await _mutex.lock()` microtask hop) and rejected it: Single Inserts (100 sequential) and writer_pipelining `sequential-awaited (2000 writes)` both moved <2 % in the wrong direction across paired runs, and the only positive lane (Concurrent Single Inserts, −7.8 %) is one exp 159 already drives at −58 % to −61 %. Do not retry either scheduling tweak without new runtime or workload evidence. Exp 171 tried the same shape one layer up (cached `_resolvedRuntime` on `Database` to skip `await _runtime` on hot paths) and rejected it on focused-harness noise — Database-layer microtask hop trimming above the writer no longer moves the sequential-write floor. Exp 214 adds `benchmark/experiments/write_result_direct_read.dart` as the µs-scale writer-result floor harness and rejects direct native-pointer scalar reads; do not chase `executeWrite()` result decoding or exp 095-style writer result-buffer scratch again without a mechanism larger than view/allocation removal and reproduced order-flipped public-write deltas. Exp 182 took the residual bucket attack in a different direction — skip `preupdate_hook` accumulation + reply harvest when `_streamEngine.length == 0` via a `track_dirty` flag on `resqlite_db` + a `DrainRequest` no-op barrier on first stream registration — and rejected it: the no-stream wins are real (focused sequential −3.8 % / −5.3 %, wide-batch −2.4 % / −5.9 % across order-flipped passes) but the per-call gate adds reproduced overhead on the with-streams shape (focused +2.7 % / +3.7 %, classified `reproduced` by `ab_drift_check.dart`) and Tracelite stream-direction warmup elapsed regressed in the same direction across all three scenarios. Reactive streams are the library's primary use case, so the optimization helps a narrower workload mix than the one it slows — do not retry without a workload that shows write throughput without active streams is a hot path. A new stream experiment should try a concrete reduction candidate, add only the narrow measurement needed to explain the result in that same branch, remove temporary counters before merge unless they are reusable, and clear Tracelite measured-elapsed primary gates without harming keyed-PK. Exp 228 establishes a hash-cache invariant: cached `lastResultHash` must be canonical. An early-reject hash path must return a non-cacheable sentinel or finish the canonical digest during decode; never promote a partial digest to the next rerun baseline. Exp 239 closes queue-depth-only transparent read batching: its homogeneous gains are real, but alternating large/point completion p95 regresses in both orderings because members share an indivisible reply. Keep `select_overflow_batch.dart` as the heterogeneous gate, and fix the reader spawn-versus-close lifecycle before any future design deliberately increases aggregate sacrifice frequency. Exp 249 rejects invalidation-grouped rerun batching (indivisible reply delays the one changed result behind cheap batch-mates; reopen only for a throughput-bound stream workload with a design that preserves independent completion). Two methodology reminders from it: (1) A/B stream-dispatch/reader-message changes ACROSS WORKTREES, never with a single-process toggle — exp 249s in-process toggle classified REPRODUCED (-27%) while cross-worktree showed +22-66%, because both toggle arms shared warm JIT/isolate/pool state. (2) high_cardinality_fanout settle returns after a fixed emission-count quiet window and never waits for suppressed reruns to drain, so it cannot measure the 99-of-100-unchanged fan-out cost; use benchmark/experiments/stream_rerun_latency.dart (single-write per-emit latency, homogeneous + heterogeneous partitions) as the fan-out emission-latency gate. Exp 283 adds two harnesses and one measurement rule. `benchmark/experiments/stream_rerun_one_pass_ab.dart` is the gate for anything that changes what a rerun does on the reader side: keep `keyed-pk` and `feed` as predictor-off guards, `writes` as the zero-ceiling control, and keep the sentinel-drain metric — it issues the burst concurrently and times a sentinel write through to the one stream it must change, because the unchanged majority emits nothing and cannot be waited on. `benchmark/experiments/stream_rerun_pass_price.dart` prices the hash pass, the decode pass and the one-pass decoder for a result shape in seconds with no pool and no isolates; run it before proposing anything that changes how many times a rerun steps its statement. And stop dividing stream count by write count when sizing a fan-out candidate — claim 283.1 is the corrected denominator. The unchanged side of the walk is still untouched: `resqlite_query_hash` steps every row to prove nothing moved, which claim 283.2 prices at 4.5-36 us, and claim 228.1 constrains any early-reject shortcut to a non-cacheable sentinel." + "notesForExperimenters": "The 2026-08-13 cadence refresh pruned the two April/May candidates: exp 120 removed stream-induced upstream over-dispatch on the measured A11c and keyed-PK stream lanes, leaving the parked-dispatcher sketch without representative incidence, and exp 249 superseded the generic rerun-batching direction with a durable cross-worktree latency gate. Reopen this direction only when an `interestingIf` trigger supplies a concrete reduction candidate. Avoid assuming a larger reader pool helps; exp 105 found the opposite under A11c fan-out. After exp 118 + exp 120, `dispatcherParkedTotal` and `dispatcherWakeRetryTotal` stay at zero on every measured stream workload — the parked-dispatcher signal is not the active target. Exp 121 took invalidation traversal off the candidate list (10–15% of overlap wall, 2.5–5.7% intersection, 80–200 ns per probe). Exp 122 keeps admission simple by giving StreamEngine a concrete ReaderPool and moving stream registry checks to diagnostics. Exp 134 is proof that row-level precision can win, but its SQL-recognizer implementation is rejected; revive it only with explicit API/design or real workload evidence. Exp 136 showed completion-side reader handling was large enough to try batching, but exp 148 proved plain worker-side reader-reply batching is not mergeable: it reduced callback counters while failing measured-elapsed primary scenarios. Exp 147 still leaves residual writer/request wall as the biggest bucket, but standalone residual splitting has reached diminishing returns. Exp 151 tried the narrow response-side variant (`Completer.sync()` for writer responses) and rejected it under Tracelite. Exp 170 tried the matching request-side variant (`Mutex.tryLock` + non-`async` `Writer.execute` to drop the uncontended `await _mutex.lock()` microtask hop) and rejected it: Single Inserts (100 sequential) and writer_pipelining `sequential-awaited (2000 writes)` both moved <2 % in the wrong direction across paired runs, and the only positive lane (Concurrent Single Inserts, −7.8 %) is one exp 159 already drives at −58 % to −61 %. Do not retry either scheduling tweak without new runtime or workload evidence. Exp 171 tried the same shape one layer up (cached `_resolvedRuntime` on `Database` to skip `await _runtime` on hot paths) and rejected it on focused-harness noise — Database-layer microtask hop trimming above the writer no longer moves the sequential-write floor. Exp 214 adds `benchmark/experiments/write_result_direct_read.dart` as the µs-scale writer-result floor harness and rejects direct native-pointer scalar reads; do not chase `executeWrite()` result decoding or exp 095-style writer result-buffer scratch again without a mechanism larger than view/allocation removal and reproduced order-flipped public-write deltas. Exp 182 took the residual bucket attack in a different direction — skip `preupdate_hook` accumulation + reply harvest when `_streamEngine.length == 0` via a `track_dirty` flag on `resqlite_db` + a `DrainRequest` no-op barrier on first stream registration — and rejected it: the no-stream wins are real (focused sequential −3.8 % / −5.3 %, wide-batch −2.4 % / −5.9 % across order-flipped passes) but the per-call gate adds reproduced overhead on the with-streams shape (focused +2.7 % / +3.7 %, classified `reproduced` by `ab_drift_check.dart`) and Tracelite stream-direction warmup elapsed regressed in the same direction across all three scenarios. Reactive streams are the library's primary use case, so the optimization helps a narrower workload mix than the one it slows — do not retry without a workload that shows write throughput without active streams is a hot path. A new stream experiment should try a concrete reduction candidate, add only the narrow measurement needed to explain the result in that same branch, remove temporary counters before merge unless they are reusable, and clear Tracelite measured-elapsed primary gates without harming keyed-PK. Exp 228 establishes a hash-cache invariant: cached `lastResultHash` must be canonical. An early-reject hash path must return a non-cacheable sentinel or finish the canonical digest during decode; never promote a partial digest to the next rerun baseline. Exp 239 closes queue-depth-only transparent read batching: its homogeneous gains are real, but alternating large/point completion p95 regresses in both orderings because members share an indivisible reply. Keep `select_overflow_batch.dart` as the heterogeneous gate, and fix the reader spawn-versus-close lifecycle before any future design deliberately increases aggregate sacrifice frequency. Exp 249 rejects invalidation-grouped rerun batching (indivisible reply delays the one changed result behind cheap batch-mates; reopen only for a throughput-bound stream workload with a design that preserves independent completion). Two methodology reminders from it: (1) A/B stream-dispatch/reader-message changes ACROSS WORKTREES, never with a single-process toggle — exp 249s in-process toggle classified REPRODUCED (-27%) while cross-worktree showed +22-66%, because both toggle arms shared warm JIT/isolate/pool state. (2) high_cardinality_fanout settle returns after a fixed emission-count quiet window and never waits for suppressed reruns to drain, so it cannot measure the 99-of-100-unchanged fan-out cost; use benchmark/experiments/stream_rerun_latency.dart (single-write per-emit latency, homogeneous + heterogeneous partitions) as the fan-out emission-latency gate. Exp 283 adds two harnesses, one measurement rule and one lane warning. `benchmark/experiments/stream_rerun_one_pass_ab.dart` is the gate for anything that changes what a rerun does on the reader side: keep `keyed-pk` and `feed` as predictor-off guards, `writes` as the zero-ceiling control, and keep the sentinel-drain metric - it issues the burst concurrently and times a sentinel write through to the one stream it must change, because the unchanged majority emits nothing and cannot be waited on (claim 283.5). `benchmark/experiments/stream_rerun_pass_price.dart` prices the hash pass, the decode pass and the one-pass decoder for a result shape in seconds with no pool and no isolates. Stop dividing stream count by write count when sizing a fan-out candidate - claim 283.1 is the corrected denominator - and do not size one against `high_cardinality_fanout`'s wall at all (claim 283.7). Before treating any rerun pass as removable, check what it re-reads: claim 283.6 is the case where a pass that looked like pure overhead was buying emission freshness. And when a flagged lane's behaviour differs but the mechanism is unclear, add a third arm that wires the change through every layer while compiling the hot branch out; exp 283's bookkeeping-only arm settled attribution in 14 runs that 70 primary-lane runs had not." }, { "id": "parameter-encoding-and-binding", diff --git a/experiments/signals/entries/283.json b/experiments/signals/entries/283.json index 55b51564..a342608f 100644 --- a/experiments/signals/entries/283.json +++ b/experiments/signals/entries/283.json @@ -1,16 +1,17 @@ { "directions": [ "stream-rerun-dispatch", - "result-transfer-shape" + "result-transfer-shape", + "measurement-system" ], - "outcomeClass": "accepted", - "outcomeReason": "performance", + "outcomeClass": "rejected", + "outcomeReason": "tradeoff", "changedBeliefs": [ - "We believed a stream rerun's two shapes were priced the way exp 075 left them: an unchanged rerun is one cheap C pass, and a changed one adds a Dart decode. The second half of that was wrong by omission — a changed rerun adds a decode *and* a second complete SQLite walk over the same rows, because the hash pass has already consumed the statement. Claim 283.2 prices that walk at 35-45% of a changed rerun's SQLite-and-decode work at every width from one row to a thousand. It had been there since April 2026 and nobody had measured it, because the two paths were always discussed as 'hash only' versus 'hash plus decode' rather than as one step pass versus two.", - "We believed reactive fan-out reruns were overwhelmingly unchanged — the release suites' own docstrings say so, and `stream_rerun_latency.dart` is built around one changed stream in a hundred. Claim 283.1 shows that describes the *writes*. Per-stream coalescing collapses 20,000 arithmetic reruns into 847 real ones on the high-cardinality shape, and 56% of the survivors change. Anyone reasoning about this direction should stop dividing stream count by write count: the quantity that matters is reruns after coalescing, and it is smaller and much more change-dense than the shape suggests. Keyed-PK and feed remain genuinely change-free (0.28% and 0%), so the direction now has two distinct fan-out regimes rather than one.", - "We believed a rerun-side use of exp 097's one-pass decoder needed foreknowledge nobody had. It needs one bool. Claim 283.3 shows reruns change in runs, so the previous outcome predicts the next one well enough to be net positive against a roughly 2:1 exchange rate, and — the part that makes the guards trustworthy — it stays structurally off on streams that do not change, arming 36 and 18 times in whole collections rather than arming and losing. Exp 228's parting condition, that early rejection reopens if 'the changed-result decode can produce the canonical hash without another full pass', is now satisfied rather than merely stated; claim 228.1 records that condition as a claim so the next candidate can edge to it.", - "A reader of exps 148, 239 and 249 should revisit how those rejections were sized, not their verdicts. Claim 283.4's end-to-end win is four times what its own per-rerun arithmetic predicts, on a metric that prices a backlog draining through four workers. If that gap is queueing, then reader-side rerun savings compound into wait times and the per-rerun denominators this direction has used to reject candidates understate them. Nothing here refutes those experiments — all three failed on latency shape, not on magnitude — but the next runner should sweep pool size across one burst before quoting a per-rerun figure as a ceiling.", - "Claim 283.5 is a measurement rule for this direction, learned the expensive way in this run. An awaited write-by-write fan-out burst cannot see reader-side rerun cost: each rerun hides inside the next write's latency, and the first metric tried here read the candidate as neutral while the mechanism was worth 39% of a changed rerun. Issue the burst concurrently and time a sentinel write through to the one stream it must change. The unchanged majority emits nothing and can never be waited on directly, which is exactly why the backlog has to be priced through something that can." + "We believed a stream rerun's changed path was one cheap C hash pass plus a Dart decode. It is two complete SQLite walks, and claim 283.2 prices the second at 35-45% of the rerun's SQLite-and-decode work at every width from one row to a thousand. That much was the expected finding. The unexpected one is claim 283.6: the two walks read two different snapshots, because `resqlite_query_hash` resets on exit and the decode that follows opens a fresh read transaction. The rows a stream emits are fresher than the digest stored beside them, and that accident is what makes it converge in as few emissions as it does. Anyone who reads the rerun path as 'hash, then decode if needed' should now read it as 'hash, then re-read', and should not treat the second walk as removable without replacing what it buys.", + "We believed reactive fan-out reruns were overwhelmingly unchanged - the release suites' own docstrings say so, and `stream_rerun_latency.dart` is built around one changed stream in a hundred. Claim 283.1 shows that describes the *writes*. Per-stream coalescing collapses 20,000 arithmetic reruns into 847 real ones on the high-cardinality shape, and 56% of the survivors change. Stop dividing stream count by write count: the quantity that matters is reruns after coalescing, and it is smaller and much more change-dense than the shape suggests. Keyed-PK (0.28%) and feed (0%) remain genuinely change-free, so the direction has two regimes rather than one.", + "We believed the release suite's largest resqlite lane measured stream fan-out performance. Claim 283.7 shows it mostly measures a 200 ms sleep, twice or once, decided by whether a convergence emission happens to land in a measured iteration. It cannot resolve anything smaller than a settle window, its two modes are ~246 ms and ~448 ms, and a +91% flag on it is one window. Do not size a stream candidate against it, and do not read its trend line as a performance signal; the drain metric in `stream_rerun_one_pass_ab.dart` is what to use instead.", + "A reader of exps 148, 239 and 249 should revisit how those rejections were sized, not their verdicts. Claim 283.4's end-to-end win is four times what its own per-rerun arithmetic predicts, on a metric that prices a backlog draining through four workers. If that gap is queueing, then reader-side rerun savings compound into wait times and the per-rerun denominators this direction has used to reject candidates understate them. Nothing here refutes those experiments - all three failed on latency shape, not magnitude - but the next runner should sweep pool size across one burst before quoting a per-rerun figure as a ceiling.", + "Claim 283.5 is a measurement rule for this direction, learned the expensive way. An awaited write-by-write fan-out burst cannot see reader-side rerun cost: each rerun hides inside the next write's latency, and a 25-sample pass of that metric put the candidate at +1.2% on a lane the mechanism moves 12%. Issue the burst concurrently and time a sentinel write through to the one stream it must change. The unchanged majority emits nothing and can never be waited on directly, which is exactly why the backlog needs something that can." ], "claims": [ { @@ -21,7 +22,7 @@ }, { "id": "283.2", - "text": "The second SQLite pass a changed stream rerun makes is 35-45% of its SQLite-and-decode work, at every width measured. Timed with no pool, no isolates and no message hop: 100 rows x 2 cols, hash 4.54 us + decode 6.17 us against 6.50 us for the one-pass decoder (-4.22 us, -39.4%); 50 rows x 4 cols with TEXT, 4.33 + 6.93 against 7.36 (-3.91 us, -34.7%); 1 row x 3 cols, 1.02 + 1.13 against 1.19 (-0.97 us, -44.8%); 1,000 rows x 2 cols, 36.05 + 51.90 against 53.08 (-34.86 us, -39.6%). Folding the digest into the decode costs 0.6-1.2 us; the pass it removes costs 4.5-36 us. The converse is the miss tax: a decode built and discarded costs 17-70% of an unchanged rerun.", + "text": "The second SQLite pass a changed stream rerun makes is 35-45% of its SQLite-and-decode work, at every width measured. Timed with no pool, no isolates and no message hop: 100 rows x 2 cols, hash 4.54 us + decode 6.17 us against 6.50 us for the one-pass decoder (-4.22 us, -39.4%); 50 rows x 4 cols with TEXT, 4.33 + 6.93 against 7.36 (-3.91 us, -34.7%); 1 row x 3 cols, 1.02 + 1.13 against 1.19 (-0.97 us, -44.8%); 1,000 rows x 2 cols, 36.05 + 51.90 against 53.08 (-34.86 us, -39.6%). Folding the digest into the decode costs 0.6-1.2 us; the pass it removes costs 4.5-36 us. The converse is the miss tax: a decode built and discarded costs 17-70% of an unchanged rerun. This is a decomposition of cost, not a measure of waste: claim 283.6 shows the removed pass re-reads the database and that the re-read is load-bearing.", "conditions": "`benchmark/experiments/stream_rerun_pass_price.dart`, JIT, 15 samples x 400 iterations per arm with arm order rotated per sample, arm64 macOS 26.2 (M1 Pro), Dart 3.12.2.", "edges": [ { @@ -38,7 +39,7 @@ }, { "id": "283.4", - "text": "Decoding during the hash pass when the previous rerun changed is worth a pooled -12.5% on 100 streams over 100-row partitions and -5.7% on 20 streams over 1,000-row partitions, with the keyed-PK and feed lanes neutral and a zero-ceiling write-only control at +0.6%. Eight order-flipped passes, two AOT bundles from separate worktrees, one lane per process, 41 samples after 8 warmup; `ab_drift_check.dart` classifies both fan-out lanes reproduced on the freshest pair (-25.5%/-23.5% and -15.4%/-5.1%) and the control drift-suspected. The end-to-end saving is about four times what claim 283.2's per-rerun figures and claim 283.3's hit counts predict (770 us per burst against 188 us), which is unexplained; the leading hypothesis is queueing, since the metric prices a backlog draining through four workers and service-time reductions compound into wait times.", + "text": "Decoding during the hash pass when the previous rerun changed is worth a pooled -12.5% on 100 streams over 100-row partitions and -5.7% on 20 streams over 1,000-row partitions, with the keyed-PK and feed lanes neutral and a zero-ceiling write-only control at +0.6%. Eight order-flipped passes, two AOT bundles from separate worktrees, one lane per process, 41 samples after 8 warmup; `ab_drift_check.dart` classifies both fan-out lanes reproduced on the freshest pair (-25.5%/-23.5% and -15.4%/-5.1%) and the control drift-suspected. The runtime is nonetheless archived (claim 283.6). The end-to-end saving is about four times what claim 283.2's per-rerun figures and claim 283.3's hit counts predict (770 us per burst against 188 us), which is unexplained; the leading hypothesis is queueing, since the metric prices a backlog draining through four workers and service-time reductions compound into wait times.", "conditions": "`benchmark/experiments/stream_rerun_one_pass_ab.dart`, AOT, arm64 macOS 26.2 (M1 Pro), Dart 3.12.2, load average 4.3-14.1 across the session; the zero-ceiling control swings +-8% in its worst pass and +-2% typically.", "edges": [ { @@ -64,13 +65,36 @@ "text": "A reactive fan-out A/B cannot be read off an awaited write-by-write burst. With each write awaited, every rerun overlaps the next write's latency and the wall reads as the write burst: the first metric tried here put the candidate within +-2% on its primary lane while the mechanism was worth -39% of a changed rerun. Issuing the burst concurrently and then timing a sentinel write through to the one stream it must change prices the whole backlog instead, because every rerun the burst scheduled - including the unchanged majority, which emits nothing and cannot be waited on - clears the queue first. The same eight-pass collection then reads -12.5%.", "conditions": "Both metrics implemented in `benchmark/experiments/stream_rerun_one_pass_ab.dart`; the awaited variant was replaced, not kept.", "edges": [] + }, + { + "id": "283.6", + "text": "The two SQLite passes a changed stream rerun makes do not read the same database state, and the second one is load-bearing. `resqlite_query_hash` resets the statement on exit, so the `decodeQuery` that follows opens a fresh read transaction on a newer WAL snapshot: during a write burst the rows a stream emits are fresher than the digest it stores alongside them, and the stream converges in fewer emissions than a single-snapshot rerun would. Replacing the two passes with exp 097's one-pass decoder makes hash and rows agree - the more defensible contract - and costs about 12% more emissions during the burst (post-baseline emissions on the high-cardinality fan-out lane's first iteration: 91-101 over six runs on origin/main against 102-115 over eight runs on the candidate). Every extra emission carries genuinely different data; harness-side content comparison found zero same-content emissions on either arm. A one-pass rerun therefore needs a cheap freshness re-check, not just a correct digest.", + "conditions": "`benchmark/suites/high_cardinality_fanout.dart` with temporary harness-side content comparison, JIT, arm64 macOS 26.2 (M1 Pro), Dart 3.12.2. Attribution confirmed by a third arm that wires the flag through but compiles the decoder branch out, which reproduces the baseline rate.", + "edges": [ + { + "type": "refines", + "target": "283.2" + }, + { + "type": "dependsOn", + "target": "228.1" + } + ] + }, + { + "id": "283.7", + "text": "`High-Cardinality Stream Fan-out (v1) / 100 streams x 200 writes / resqlite` cannot resolve anything below 200 ms, and its headline value is bimodal on a race rather than on performance. Its settle loop waits a 200 ms quiet window and stops at the first window with no new emission, so a measured iteration that emits nothing costs one window and one that emits anything costs two: the lane reads ~246 ms or ~448 ms and nothing between, against roughly 46 ms of actual work. Because the suite re-seeds its PRNG per iteration, the measured iterations rewrite values iteration 1 already wrote and the correct emission count is zero, but a stream that has not finished converging emits anyway - 2 of 35 runs on origin/main and 10 of 35 on exp 283's candidate. A +91.4% regression flag on this lane is one settle window. It has been the largest resqlite number in the release suite for months and most of it is a sleep.", + "conditions": "35 standalone invocations per arm, arm64 macOS 26.2 (M1 Pro), Dart 3.12.2. Measured while investigating the release run of 2026-09-06.", + "edges": [] } ], "nextSignals": [ - "`benchmark/experiments/stream_rerun_one_pass_ab.dart` is the durable gate for anything that changes what a stream rerun does on the reader side. Keep `keyed-pk` and `feed` as the predictor-off guards and `writes` as the zero-ceiling control, and keep the sentinel-drain metric: claim 283.5 is the record of what an awaited write burst hides.", - "`benchmark/experiments/stream_rerun_pass_price.dart` prices the hash pass, the decode pass and the one-pass decoder for a given result shape in seconds, with no pool and no isolates. Run it before proposing anything that changes how many times a rerun steps its statement.", - "Do not size a stream candidate by per-rerun cost alone. Claim 283.4's end-to-end win is four times its own per-rerun model, and if that is queueing then every reader-side rerun saving pays back multiplied on a saturated fan-out. Sweeping the reader pool size across the same burst would settle it and would recalibrate every past 'below the round-trip floor' rejection in this direction.", - "The unchanged side of the same walk is untouched. `resqlite_query_hash` steps every row to prove nothing moved, which claim 283.2 prices at 4.5-36 us; claim 228.1 constrains any early-reject shortcut to a non-cacheable sentinel.", - "Reruns that change in isolated single occurrences rather than runs are the predictor's loss case (claim 283.3) and no lane in the suite exercises them. A workload with one write per stream per burst would." - ] + "The candidate reopens on a design that keeps one snapshot per rerun and still converges as fast - a cheap freshness re-check (row count, a data-version probe) rather than a second full walk. Claim 283.6 is the specification of what such a design has to replace, and claim 228.1 constrains what an early-reject digest may return. `archive/exp-283` has the prototype; `archive/exp-283-census` has the rerun counter and its driver.", + "It also reopens on evidence that emission count is cheap - a downstream trace showing burst-time intermediate emissions are coalesced by the UI layer or otherwise immaterial. The trade claim 283.6 records is 12% more emissions for 12-16% less wall, and nothing in the repo prices an emission.", + "`benchmark/experiments/stream_rerun_one_pass_ab.dart` is the durable gate for anything that changes what a rerun does on the reader side. Keep `keyed-pk` and `feed` as predictor-off guards, `writes` as the zero-ceiling control, and keep the sentinel-drain metric: claim 283.5 records what an awaited write burst hides.", + "`benchmark/experiments/stream_rerun_pass_price.dart` prices the hash pass, the decode pass and the one-pass decoder for a result shape in seconds, with no pool and no isolates. Run it before proposing anything that changes how many times a rerun steps its statement.", + "Do not size a stream candidate by per-rerun cost alone (claim 283.4) and do not size one against `high_cardinality_fanout`'s wall at all (claim 283.7). Sweeping the reader pool size across one drain burst would settle whether the amplification is queueing and would recalibrate every past 'below the round-trip floor' rejection in this direction.", + "Methodology worth reusing: when a flagged lane's behaviour differs but the mechanism is unclear, add a third arm that wires the change through every layer while compiling the hot branch out. Exp 283's bookkeeping-only arm reproduced the baseline rate in 14 runs and settled attribution that 70 primary-lane runs had not." + ], + "outcomeNote": "The mechanism reproduced and the end-to-end win reproduced; the runtime is archived because the pass it removes turns out to buy emission freshness (claim 283.6), which is a semantic-shaped trade a scheduled run should not settle alone." } diff --git a/lib/src/reader/read_worker.dart b/lib/src/reader/read_worker.dart index 3ebeaee0..e005fea8 100644 --- a/lib/src/reader/read_worker.dart +++ b/lib/src/reader/read_worker.dart @@ -81,7 +81,6 @@ final class SelectIfChangedRequest extends ReadRequest { super.parameters, this.lastResultHash, this.lastRowCount, { - this.decodeFirst = false, super.traceCorrelationId, }); final int lastResultHash; @@ -90,15 +89,6 @@ final class SelectIfChangedRequest extends ReadRequest { /// baseline. Compared with the fresh count alongside the result hash; a null /// baseline can never match, so the result is always treated as changed. final int? lastRowCount; - - /// Whether to build the Dart result during the hash pass rather than after - /// it ([EXP-283](../../../experiments/283-stream-rerun-one-pass.md)). - /// - /// Stamped by the stream engine from the previous rerun of this same stream, - /// which is the only place that knows the outcome sequence: a reader worker - /// sees a sample of a stream's reruns and is destroyed outright by the - /// sacrifice path, the same reason [rowHint] is request-carried. - final bool decodeFirst; } /// How large a result's *structure* (rows × columns) can grow before handing @@ -224,7 +214,6 @@ void readerEntrypoint(List args) { :final parameters, :final lastResultHash, :final lastRowCount, - :final decodeFirst, ): // Two-pass selectIfChanged // ([EXP-075](../../../experiments/075-native-hash-selectifchanged.md)). @@ -237,7 +226,6 @@ void readerEntrypoint(List args) { parameters, lastResultHash, lastRowCount, - decodeFirst, request.rowHint, request.initialRowHint, ); @@ -464,15 +452,6 @@ RawQueryResult executeQuery( /// `resqliteQueryHash` resets the stmt on exit, and bindings survive /// reset, so no re-acquire is required. The pass-1 hash is reused as /// the new baseline. -/// -/// [decodeFirst] collapses the two passes into one -/// ([EXP-283](../../../experiments/283-stream-rerun-one-pass.md)). -/// `decodeQueryWithInitialHash` folds the same canonical FNV digest while it -/// fills the result, so a rerun that was going to decode anyway steps SQLite -/// once instead of twice; a rerun that turns out to be unchanged discards the -/// result it built. The digest is the one exp 097 already ships for initial -/// stream registration and is byte-for-byte the hash-only pass's, which is what -/// keeps the stored baseline canonical (exp 228). (int, int, RawQueryResult?) executeQueryIfChanged( int handleAddr, int readerId, @@ -480,22 +459,9 @@ RawQueryResult executeQuery( List parameters, int lastResultHash, int? lastRowCount, [ - bool decodeFirst = false, int rowHint = 0, int initialRowHint = 0, ]) => _withAcquiredStmt(handleAddr, readerId, sql, parameters, (_, stmt) { - if (decodeFirst) { - final (raw, newHash) = decodeQueryWithInitialHash( - stmt, - sql, - rowHint: rowHint, - initialRowHint: initialRowHint, - ); - if (newHash == lastResultHash && raw.rowCount == lastRowCount) { - return (newHash, raw.rowCount, null); - } - return (newHash, raw.rowCount, raw); - } final (newHash, newRowCount) = callQueryHash(stmt); if (newHash == lastResultHash && newRowCount == lastRowCount) { return (newHash, newRowCount, null); diff --git a/lib/src/reader/reader_pool.dart b/lib/src/reader/reader_pool.dart index 8e6778e3..c6f9a1f1 100644 --- a/lib/src/reader/reader_pool.dart +++ b/lib/src/reader/reader_pool.dart @@ -237,18 +237,12 @@ final class ReaderPool { /// Execute a re-query with worker-side hash comparison. /// Returns `(rows, newHash, newRowCount)` — `rows` is null when the /// result is unchanged (hash AND row count match). - /// - /// [decodeFirst] asks the worker to build the result during the hash pass - /// instead of after it - /// ([EXP-283](../../../experiments/283-stream-rerun-one-pass.md)); the - /// caller sets it when this stream's previous rerun changed. Future<(List>?, int, int)> selectIfChanged( String sql, List parameters, int lastResultHash, int? lastRowCount, [ int? traceCorrelationId, - bool decodeFirst = false, ]) async { final memory = _rowHints[sql]; final result = await _dispatch( @@ -257,7 +251,6 @@ final class ReaderPool { parameters, lastResultHash, lastRowCount, - decodeFirst: decodeFirst, traceCorrelationId: traceCorrelationId, ), memory, diff --git a/lib/src/stream_engine.dart b/lib/src/stream_engine.dart index 1ca9badd..96d777b5 100644 --- a/lib/src/stream_engine.dart +++ b/lib/src/stream_engine.dart @@ -355,9 +355,7 @@ final class StreamEngine { entry.lastResultHash, entry.lastRowCount, traceCorrelationId, - entry.lastRerunChanged, ); - entry.lastRerunChanged = rows != null; // If the entry has already been marked dirty again from an invalidation that ocurred // while it was requerying, then this intermediate result should be discarded and instead @@ -490,16 +488,6 @@ final class StreamEntry { /// guarantees the next re-query reports a change and emits. int? lastRowCount; - /// Whether this stream's most recent rerun changed its result - /// ([EXP-283](../../experiments/283-stream-rerun-one-pass.md)). - /// - /// The next rerun uses it to choose between decoding during the hash pass - /// and decoding only if the hash moved. A stream whose reruns change arrives - /// in runs — a partition under active writes changes on rerun after rerun — - /// so the previous outcome is the whole predictor, and the first unchanged - /// rerun disarms it. - bool lastRerunChanged = false; - /// Whether the stream is dirty and needs to be requeried. bool dirty = false; diff --git a/test/query_decoder_test.dart b/test/query_decoder_test.dart index 9427d5ef..ae30c45c 100644 --- a/test/query_decoder_test.dart +++ b/test/query_decoder_test.dart @@ -142,109 +142,6 @@ void main() { dir.deleteSync(recursive: true); } }); - - test('decode-first selectIfChanged agrees with the hash-first arm', () { - final dir = Directory.systemTemp.createTempSync('resqlite_decoder_test_'); - final pathNative = '${dir.path}/decodefirst.db'.toNativeUtf8(); - final db = resqliteOpen(pathNative, 1, ffi.nullptr.cast()); - calloc.free(pathNative); - - expect(db, isNot(ffi.nullptr)); - - try { - _exec( - db, - 'CREATE TABLE mixed(' - 'id INTEGER PRIMARY KEY, label TEXT, amount REAL, payload BLOB);' - "INSERT INTO mixed(label, amount, payload) " - "VALUES ('héllo 🚀', 1.25, x'010203FF');" - "INSERT INTO mixed(label, amount, payload) VALUES ('', -3.5, x'');", - ); - - const sql = 'SELECT label, amount, payload FROM mixed ORDER BY id'; - final (_, _, initialHash, initialCount) = executeQueryWithDeps( - db.address, - 0, - sql, - const [], - ); - - // Unchanged, both arms: no result, same canonical baseline. - for (final decodeFirst in [false, true]) { - final (h, c, raw) = executeQueryIfChanged( - db.address, - 0, - sql, - const [], - initialHash, - initialCount, - decodeFirst, - ); - expect(raw, isNull, reason: 'decodeFirst=$decodeFirst'); - expect(h, initialHash, reason: 'decodeFirst=$decodeFirst'); - expect(c, initialCount, reason: 'decodeFirst=$decodeFirst'); - } - - _exec( - db, - "INSERT INTO mixed(label, amount, payload) " - "VALUES ('third', 9.0, x'AA')", - ); - - // Changed, both arms: identical hash, row count and decoded values, so a - // baseline minted by one arm is usable by the other. - final ( - hashFirstHash, - hashFirstCount, - hashFirstRaw, - ) = executeQueryIfChanged( - db.address, - 0, - sql, - const [], - initialHash, - initialCount, - false, - ); - final ( - decodeFirstHash, - decodeFirstCount, - decodeFirstRaw, - ) = executeQueryIfChanged( - db.address, - 0, - sql, - const [], - initialHash, - initialCount, - true, - ); - - expect(hashFirstRaw, isNotNull); - expect(decodeFirstRaw, isNotNull); - expect(decodeFirstHash, hashFirstHash); - expect(decodeFirstCount, hashFirstCount); - expect(decodeFirstCount, 3); - expect(decodeFirstRaw!.values, hashFirstRaw!.values); - expect(decodeFirstRaw.schema.names, hashFirstRaw.schema.names); - - // The decode-first arm's hash is a usable baseline for the hash-first - // arm: a mixed sequence must not re-emit an unchanged result. - final (_, _, afterRaw) = executeQueryIfChanged( - db.address, - 0, - sql, - const [], - decodeFirstHash, - decodeFirstCount, - false, - ); - expect(afterRaw, isNull); - } finally { - resqliteClose(db); - dir.deleteSync(recursive: true); - } - }); } void _exec(ffi.Pointer db, String sql) { From e47751eb448fb106768b3aa96a591f24cb691c3e Mon Sep 17 00:00:00 2001 From: Dan Reynolds Date: Sun, 6 Sep 2026 08:35:23 -0400 Subject: [PATCH 6/7] Exp 283: link issue #318 for the fan-out lane's quantized wall --- experiments/283-stream-rerun-one-pass.md | 3 ++- experiments/signals/entries/283.json | 2 +- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/experiments/283-stream-rerun-one-pass.md b/experiments/283-stream-rerun-one-pass.md index aa145e31..a0ff23a1 100644 --- a/experiments/283-stream-rerun-one-pass.md +++ b/experiments/283-stream-rerun-one-pass.md @@ -310,7 +310,8 @@ contribution and they hold regardless of the verdict. - **The lane cannot resolve anything under 200 ms.** `high_cardinality_fanout`'s headline wall is one or two settle windows plus ~46 ms of actual work, and which one it is depends on a race. It has been the release suite's largest - resqlite number for months and most of it is a sleep. Filed separately; any + resqlite number for months and most of it is a sleep. Filed as + [#318](https://github.com/danReynolds/resqlite/issues/318); any future stream candidate should be measured on `benchmark/experiments/stream_rerun_one_pass_ab.dart`'s drain metric instead. - The fan-out tax has two very different readings and neither has been diff --git a/experiments/signals/entries/283.json b/experiments/signals/entries/283.json index a342608f..6699f2f7 100644 --- a/experiments/signals/entries/283.json +++ b/experiments/signals/entries/283.json @@ -83,7 +83,7 @@ }, { "id": "283.7", - "text": "`High-Cardinality Stream Fan-out (v1) / 100 streams x 200 writes / resqlite` cannot resolve anything below 200 ms, and its headline value is bimodal on a race rather than on performance. Its settle loop waits a 200 ms quiet window and stops at the first window with no new emission, so a measured iteration that emits nothing costs one window and one that emits anything costs two: the lane reads ~246 ms or ~448 ms and nothing between, against roughly 46 ms of actual work. Because the suite re-seeds its PRNG per iteration, the measured iterations rewrite values iteration 1 already wrote and the correct emission count is zero, but a stream that has not finished converging emits anyway - 2 of 35 runs on origin/main and 10 of 35 on exp 283's candidate. A +91.4% regression flag on this lane is one settle window. It has been the largest resqlite number in the release suite for months and most of it is a sleep.", + "text": "`High-Cardinality Stream Fan-out (v1) / 100 streams x 200 writes / resqlite` cannot resolve anything below 200 ms, and its headline value is bimodal on a race rather than on performance. Its settle loop waits a 200 ms quiet window and stops at the first window with no new emission, so a measured iteration that emits nothing costs one window and one that emits anything costs two: the lane reads ~246 ms or ~448 ms and nothing between, against roughly 46 ms of actual work. Because the suite re-seeds its PRNG per iteration, the measured iterations rewrite values iteration 1 already wrote and the correct emission count is zero, but a stream that has not finished converging emits anyway - 2 of 35 runs on origin/main and 10 of 35 on exp 283's candidate. A +91.4% regression flag on this lane is one settle window. It has been the largest resqlite number in the release suite for months and most of it is a sleep. Filed as issue #318.", "conditions": "35 standalone invocations per arm, arm64 macOS 26.2 (M1 Pro), Dart 3.12.2. Measured while investigating the release run of 2026-09-06.", "edges": [] } From 77267c7e8c101082350bdb26d0c11e884dba6899 Mon Sep 17 00:00:00 2001 From: Dan Reynolds Date: Sun, 6 Sep 2026 08:46:53 -0400 Subject: [PATCH 7/7] Exp 283: fix the drain harness's sentinel and correct the A/B Review of #319 found that the sentinel stream was an ordinary partition the burst also wrote to, with its completer armed before the burst started. Each sample therefore ended at whichever of that stream's reruns fired first -- a random prefix of the backlog, biased toward whichever arm finished changed reruns faster. The sentinel now watches a partition excluded from the burst, and its completer is armed after the burst is issued. The same collection then reads -1.7% (drift-suspected) on the 100-row fan-out lane, not -12.5%, and -11.3% (reproduced) on the 1,000-row lane, not -5.7%. The withdrawn figures are recorded as claim 283.8, along with the rule. The earlier claim that the win exceeded its per-rerun model four-fold was the same defect measuring a prefix; it is withdrawn, and with it the JOURNAL entry built on it. The corrected reconciliation runs the ordinary way: 47% of the isolated mechanism survives end to end. --- .../experiments/stream_rerun_one_pass_ab.dart | 21 ++- experiments/283-stream-rerun-one-pass.md | 133 +++++++++--------- experiments/JOURNAL.md | 42 +++--- experiments/index/283.json | 2 +- experiments/signals/base.json | 4 +- experiments/signals/entries/283.json | 23 ++- 6 files changed, 128 insertions(+), 97 deletions(-) diff --git a/benchmark/experiments/stream_rerun_one_pass_ab.dart b/benchmark/experiments/stream_rerun_one_pass_ab.dart index e8457920..2ef78ee3 100644 --- a/benchmark/experiments/stream_rerun_one_pass_ab.dart +++ b/benchmark/experiments/stream_rerun_one_pass_ab.dart @@ -15,6 +15,12 @@ /// backlog. An awaited write-by-write burst cannot: each rerun overlaps the /// next write's latency and the wall reads as the write burst. /// +/// The sentinel stream watches a partition the burst is excluded from, so the +/// only thing that can complete the sample is the sentinel write itself. A +/// sentinel the burst can also change ends the sample early on whichever of its +/// reruns happens to fire first, which measures a random prefix of the backlog +/// instead of the whole of it. +/// /// Lanes: /// fanout — 100 streams over 100-row partitions, 200 random writes. /// ~56% of its reruns change (see `stream_rerun_census.dart`), @@ -147,14 +153,16 @@ class _Lane { for (var o = 1; o <= owners; o++) for (var r = 0; r < perOwner; r++) [o, 0], ]); - final rows = owners * perOwner; + // Owner 1 is the sentinel partition: ids 1..perOwner are excluded from the + // burst so nothing but the sentinel write can change stream 0. + final burstRows = (owners - 1) * perOwner; final lane = _Lane( name, db, writes, (d, rng, i) => d.execute('UPDATE items SET value = ? WHERE id = ?', [ i, - rng.nextInt(rows) + 1, + perOwner + rng.nextInt(burstRows) + 1, ]), subscribe ? (d, i) => d.execute('UPDATE items SET value = ? WHERE id = ?', [ @@ -189,13 +197,14 @@ class _Lane { for (var i = 1; i <= rowCount; i++) [i, i, 'row-$i'], ], ); + // id 1 is the sentinel row: stream 0 watches it and the burst never does. final lane = _Lane( 'keyed-pk', db, 200, (d, rng, i) => d.execute('UPDATE items SET value = ? WHERE id = ?', [ i, - rng.nextInt(rowCount) + 1, + rng.nextInt(rowCount - 1) + 2, ]), (d, i) => d.execute('UPDATE items SET value = ? WHERE id = ?', [-i - 1, 1]), @@ -279,17 +288,15 @@ class _Lane { Future burst() async { final rng = math.Random(0xCAFEF0); final sentinel = _sentinel; - final waiter = sentinel == null - ? null - : (_sentinelWaiter = Completer()); final sw = Stopwatch()..start(); final writes = >[ for (var w = 0; w < _writeCount; w++) _write(_db, rng, _writeCursor++), ]; await Future.wait(writes); if (sentinel != null) { + final waiter = _sentinelWaiter = Completer(); await sentinel(_db, _writeCursor++); - await waiter!.future; + await waiter.future; } sw.stop(); _sentinelWaiter = null; diff --git a/experiments/283-stream-rerun-one-pass.md b/experiments/283-stream-rerun-one-pass.md index a0ff23a1..7495ccc0 100644 --- a/experiments/283-stream-rerun-one-pass.md +++ b/experiments/283-stream-rerun-one-pass.md @@ -165,13 +165,13 @@ toggle), one lane per process, 41 samples after 8 warmup, arm order flipped between passes. Δ is candidate against baseline within each pass; **B**/**C** marks which arm ran first. -| lane | 1 B | 2 C | 3 B | 4 C | 5 B | 6 C | 7 B | 8 C | median | pooled 3–8 | -|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:| -| `fanout` | −23.5 | −8.1 | −23.8 | −4.2 | −1.8 | +8.5 | −25.5 | −23.5 | **−15.8%** | **−12.5%** | -| `fanout-wide` | −7.4 | −0.1 | +6.8 | +3.9 | −15.0 | −7.1 | −15.4 | −5.1 | **−6.1%** | **−5.7%** | -| `keyed-pk` (guard) | +0.9 | +8.7 | −2.3 | +0.7 | +1.2 | +3.5 | −4.9 | −2.7 | +0.8% | −1.2% | -| `feed` (guard) | +0.3 | −1.7 | −2.5 | −0.4 | −9.7 | −2.8 | −7.2 | −0.3 | −1.1% | −4.1% | -| `writes` (control) | −2.4 | −1.2 | −0.9 | −4.0 | +7.8 | +4.1 | +6.0 | −4.5 | −1.1% | +0.6% | +| lane | 1 B | 2 C | 3 B | 4 C | median | pooled | drift verdict (3/4) | +|---|---:|---:|---:|---:|---:|---:|---| +| `fanout` — 100 streams × 100-row partitions | −2.4 | −3.6 | +5.2 | −5.1 | −3.0% | −1.7% | drift-suspected | +| `fanout-wide` — 20 streams × 1,000-row partitions | −10.8 | −11.7 | −11.5 | −9.2 | **−11.1%** | **−11.3%** | **reproduced** | +| `keyed-pk` (guard) | −1.5 | −11.6 | +2.8 | −1.5 | −1.5% | −4.2% | neutral | +| `feed` (guard) | −2.4 | +0.8 | −3.1 | −2.0 | −2.2% | −2.0% | neutral | +| `writes` (zero-ceiling control) | +16.5 | −2.7 | +6.0 | +1.4 | +3.7% | +5.6% | neutral | Each sample issues its write burst concurrently and then times a sentinel write through to the one stream it must change. Every rerun the burst scheduled — @@ -179,46 +179,51 @@ including the unchanged majority, which emits nothing and cannot be waited on directly — has to clear the queue before the sentinel's own rerun runs, so the sentinel prices the backlog. An awaited write-by-write burst cannot: each rerun overlaps the next write's latency and the wall reads as the write burst. That -was the first metric tried here and it resolved nothing. - -`benchmark/ab_drift_check.dart` on the freshest order-flipped pair (7/8): - -| lane | verdict | pass 1 | pass 2 | -|---|---|---:|---:| -| `fanout` | **reproduced** | −25.5% | −23.5% | -| `fanout-wide` | **reproduced** | −15.4% | −5.1% | -| `keyed-pk` | neutral | −4.9% | −2.7% | -| `feed` | neutral | −7.2% | −0.3% | -| `writes` | drift-suspected | +6.0% | −4.5% | - -### 5. The win is larger than the pass price predicts - -Worth stating plainly, because it is the one number here that does not -reconcile. The `fanout` lane issues 208 reruns per burst; §3's hit and miss -counts and §2's per-shape figures put the reader-side saving at -`880 × 4.22 − 654 × 1.95` over 13 bursts, or **188 µs per burst**. The measured -end-to-end saving is `6.170 − 5.400`, or **770 µs per burst** — four times as -much. - -The direction of the discrepancy rules out the obvious explanations: the second -pass in the real worker follows the first over the same rows, so it is warmer -than the tight loop §2 measured, and the model should therefore over-predict. -What it does not model is queueing. The sentinel metric prices a backlog draining -through four workers, and shortening each rerun's service time shortens every -later rerun's wait as well, so a fifth off the service time buys more than a -fifth off the drain. Reader-side work also stops being ~8 µs of a ~21 µs -per-rerun wall in this shape and starts being the part that sets the queue's -service rate. That is a hypothesis, not a measurement — the honest statement is -that the mechanism's size is established by §2 and the win's size by §4, and -the amplification between them is not yet accounted for. - -**Host caveat.** Load average ran 4.3–14.1 across the session — `mediaanalysisd` -held a core for most of it. The zero-ceiling `writes` control is the collection's -own floor and it swings ±8% in the worst pass and ±2% typically, which is why -the verdict rests on eight order-flipped passes and a pooled figure rather than -on any single pass. Passes 5 and 6 are the two where the control moved most and -they are also the two where `fanout` reads weakest; they are reported rather -than dropped. +was the first metric tried here: a 25-sample pass of the awaited variant put +`fanout` at +1.2% while the mechanism was worth 39% of a changed rerun, which is +why it was replaced rather than sampled harder. + +**The sentinel has to watch a partition the burst cannot touch,** and the first +version of this harness did not. Its sentinel stream was an ordinary partition +that the burst also wrote to, and its completer was armed before the burst +started, so whichever of that stream's reruns fired first ended the sample — +measuring a random prefix of the backlog rather than the whole of it, biased +toward whichever arm finished changed reruns faster. That version reported +`fanout` at −12.5% pooled and `fanout-wide` at −5.7%; both figures are +withdrawn. The table above is from the corrected harness, which reserves +partition 1 for the sentinel, excludes it from the burst, and arms the completer +only after the burst has been issued. The defect was found in review of this +experiment's own PR, not by the collection. + +The corrected reading is narrower and differently shaped. The lane that +reproduces is the **wide** one — 1,000-row partitions, where §2 prices the +removed pass at 34.86 µs — at a consistent −9% to −12% in all four passes. +The 100-row lane, where the same pass is worth 4.22 µs, does not clear the +collection's floor: the zero-ceiling control moved +16.5% in its worst pass, and +`fanout` reverses sign across the order flip. The mechanism scales with the +result size it re-walks, and at 100 rows there is not enough of it to see. + +**Host caveat.** Load average ran 1.8–14.1 across the session — `mediaanalysisd` +held a core for much of it, and the corrected collection above was taken in the +quietest window available (1.8–4.1). The zero-ceiling `writes` control is the +collection's own floor: it moved +16.5% in its worst pass and +3.7% at the +median. That is why only `fanout-wide`, consistent at −9% to −12% in every pass +and classified reproduced, is read as an effect, and why `keyed-pk`'s single +−11.6% pass is read as the same floor rather than as a guard failure. + +### 5. The win is smaller than the pass price predicts + +With the corrected harness the arithmetic runs the ordinary way. `fanout-wide` +issues 48 reruns per burst, §3 counts 370 hits and 109 misses over 13 bursts, +and §2 prices those at 34.86 µs and 17.03 µs — a modelled **849 µs per burst**. +The measured saving is `3.564 − 3.163`, or **401 µs per burst**: the candidate +delivers about 47% of its isolated mechanism end to end, which is the usual fate +of a per-operation saving inside a pipeline that is not bound by that operation. + +(An earlier draft of this section reported the opposite — a four-fold +*amplification* — and attributed it to queueing. That was the broken sentinel of +§4 measuring a prefix, not a queue effect. Nothing here supports the idea that +reader-side rerun savings compound; the honest statement is that they discount.) ## 6. What the release suite caught @@ -276,13 +281,14 @@ buying anything, which is why this looked like free money. ## Decision -**Rejected**, and not because the numbers failed. The mechanism is real and -reproduced: 35–45% off a changed rerun's SQLite-and-decode work, 12–16% off a -change-dense fan-out end to end, with both change-free guard lanes provably -inert. What it cannot do is pay for itself in the currency the library -advertises. resqlite's reactive story is that hash suppression keeps streams -from re-emitting; trading roughly 12% more emissions during a write burst for -less wall time is a semantic-shaped trade, and this repo's record on those +**Rejected**, on two counts that compound. The mechanism is real and reproduced +— 35–45% off a changed rerun's SQLite-and-decode work, and a consistent −11% on +the wide fan-out lane where that pass is 35 µs — but it is narrower than the +first measurement claimed: on 100-row partitions, where the pass is 4 µs, the +effect does not clear the harness floor (§4). And what it does deliver, it +cannot pay for in the currency the library advertises. resqlite's reactive story +is that hash suppression keeps streams from re-emitting; trading roughly 12% more +emissions during a write burst for less wall time is a semantic-shaped trade, and this repo's record on those (exps 197, 212, 213) is that a reproduced win does not settle them. That call belongs to a maintainer, not to a scheduled run, so the runtime is archived rather than shipped. @@ -292,7 +298,8 @@ Three things would change the answer: - **A design that keeps one snapshot per rerun and converges as fast.** The freshness the two-pass path buys is accidental, not designed. A one-pass rerun that re-checks cheaply — a row-count or version probe rather than a - second full walk — would take the win without the trade. + second full walk — would take the win without the trade. It would want to be + gated on result size: §4 says the win only shows above a few hundred rows. - **Evidence that emission count does not matter.** The cost here is more subscriber deliveries of correct intermediate data during a burst. If a downstream trace shows burst-time intermediate emissions are cheap or are @@ -302,8 +309,9 @@ Three things would change the answer: What does not need re-deriving: §1's census (reruns are far fewer and far more change-dense than the suites assume), §2's pass price, §3's predictor accuracy, -§5's queueing amplification, and §6's quantization. Those are the run's lasting -contribution and they hold regardless of the verdict. +§4's harness rule about what a sentinel may watch, and §6's quantization and +freshness finding. Those are the run's lasting contribution and they hold +regardless of the verdict. ## Future work @@ -320,12 +328,11 @@ contribution and they hold regardless of the verdict. 4.39 ms over 208 reruns — 21 µs per rerun, against roughly 8 µs of reader-side work. Whichever regime a real application is in, most of a rerun's wall is still unattributed. -- §5's four-fold amplification is the first evidence in this direction that - rerun service time and rerun *wall* are related by more than one-to-one. - If that is queueing, it is a lever: it would mean any reduction in reader-side - rerun work pays back multiplied on a saturated fan-out, and the current - practice of sizing stream candidates against per-rerun cost understates them. - A burst with the pool size swept would settle it. +- Whether the reader pool is the fan-out constraint at all is still open. §5 + says a per-rerun saving discounts to about 47% end to end on the wide lane and + to nothing on the narrow one, which is what a pipeline bound by something else + looks like. Sweeping the pool size across one drain burst would say what that + something else is. - The unchanged side of the walk is untouched. `resqlite_query_hash` steps every row to prove nothing moved, which §2 prices at 4.5–36 µs; exp 228's invariant (claim 228.1) constrains any early-reject shortcut to a non-cacheable diff --git a/experiments/JOURNAL.md b/experiments/JOURNAL.md index ce04b1b4..3baabfcb 100644 --- a/experiments/JOURNAL.md +++ b/experiments/JOURNAL.md @@ -1481,24 +1481,30 @@ whether anything downstream depends on the difference. A digest-equality test will not tell you: both arms here produced identical digests for identical data, and the divergence was in *which* data each arm read.* -### A saving inside a queue is worth more than the saving - -Exp 283, unresolved but worth carrying. The per-rerun arithmetic — measured -hit and miss counts against a measured per-shape saving — predicted 188 µs per -burst. The A/B measured 770 µs, four times as much, in the direction the model -could not produce by being sloppy: the second pass in the real worker is warmer -than the isolated loop that priced it, so the model should have over-predicted. - -The leading explanation is queueing. The metric prices a backlog draining -through four workers, and shortening each item's service time also shortens -every later item's wait. - -*Reapplies to any component-level estimate used to reject a candidate in a -saturated pipeline. This repo has repeatedly sized stream and read candidates -by per-operation cost and rejected them as "below the round-trip floor"; if the -amplification here is real, those denominators are too small whenever the pool -is the constraint. Cheap to check and nobody has: sweep the pool size across -one burst and see whether the win scales with it.* +### A sentinel you are draining toward must be unreachable by the drain + +Exp 283, discovered in review of its own PR after four order-flipped passes had +agreed on the wrong number. + +The harness measured how long a fan-out backlog takes to clear by issuing a +write burst, then a sentinel write that must change one specific stream, and +timing until that stream emits — the unchanged majority emits nothing and can +never be waited on directly, so the sentinel stands in for the whole queue. But +the sentinel stream was an ordinary partition the burst also wrote to, and its +completer was armed before the burst started. Each sample therefore ended at +whichever of that stream's reruns fired first: a random prefix of the backlog, +and a prefix whose length depended on how fast each arm finished changed reruns +— exactly the quantity under test. Reserving a partition for the sentinel and +arming the completer after the burst moved the primary lane from −12.5% +(reproduced across an order flip) to −1.7% (drift-suspected), and moved the lane +that actually wins from −5.7% to −11.3%. + +*Reapplies to every "wait for a marker to know the queue drained" metric. Two +checks before trusting one: can the workload itself trigger the marker, and is +the marker armed before the workload starts? If either is yes, samples end on a +race, and a race whose timing depends on the change under test will reproduce +across order flips exactly like a real effect. Order-flipping catches drift; it +does not catch a metric that is measuring the wrong interval in both arms.* ## How to add to this file diff --git a/experiments/index/283.json b/experiments/index/283.json index 789ec3ec..417ace76 100644 --- a/experiments/index/283.json +++ b/experiments/index/283.json @@ -1,7 +1,7 @@ { "file": "283-stream-rerun-one-pass.md", "title": "The second walk over the same rows", - "impact": "Performance: rejected, and the reason is the finding. A stream rerun hashes its result to decide whether to emit and re-steps the whole statement a second time whenever the hash moved -- a walk there since exp 075 that nobody had measured. Exp 097's one-pass decoder already folds the identical canonical digest while building the result, and a census supplied the missing predictor: per-stream coalescing means the high-cardinality fan-out lane issues 847 reruns rather than the 20,000 its arithmetic implies, 56% of them change, and they change in runs, so one bool on `StreamEntry` predicts the next rerun at 57-77% where it arms and stays structurally off on streams that never change. Isolated, the removed pass is 35-45% of a changed rerun at every width from 1 to 1,000 rows; end to end, fan-out improves a pooled -12.5% and -5.7% across eight order-flipped AOT passes, both classified reproduced, with both change-free guard lanes neutral. The release suite then flagged the same fan-out lane at +91.4%, and unpicking that produced the result: the lane's wall is quantized by a 200 ms settle window, so the flag is one extra window, and what triggers it is that the candidate emits about 12% MORE during a write burst. `resqlite_query_hash` resets on exit, so the hash-first path's second pass opens a *fresh* WAL snapshot -- the rows a stream emits are fresher than the digest it stores, and it converges in fewer emissions. The second walk was buying emission freshness, not wasting time. Trading ~12% more emissions for less wall is a semantic-shaped trade in the dimension the library advertises, so the runtime is archived at `archive/exp-283` rather than shipped. The census, the pass price, the predictor accuracy, the queueing amplification and the lane quantization all stand.", + "impact": "Performance: rejected, and the reason is the finding. A stream rerun hashes its result to decide whether to emit and re-steps the whole statement a second time whenever the hash moved -- a walk there since exp 075 that nobody had measured. Exp 097's one-pass decoder already folds the identical canonical digest while building the result, and a census supplied the missing predictor: per-stream coalescing means the high-cardinality fan-out lane issues 847 reruns rather than the 20,000 its arithmetic implies, 56% of them change, and they change in runs, so one bool on `StreamEntry` predicts the next rerun. Isolated, the removed pass is 35-45% of a changed rerun at every width from 1 to 1,000 rows; end to end it is a reproduced -11.3% on 20 streams over 1,000-row partitions and nothing measurable on 100 streams over 100-row partitions, with both change-free guard lanes neutral. Two corrections came out of review and the release suite. The harness's first sentinel was a stream the burst could also change, which ended each sample on a race and reported -12.5% on the narrow lane; those figures are withdrawn (claim 283.8). And the release suite flagged the same fan-out lane at +91.4%, which turned out to be one 200 ms settle window -- the lane is quantized and bimodal (claim 283.7, issue #318) -- triggered because the candidate emits about 12% MORE during a write burst. `resqlite_query_hash` resets on exit, so the hash-first path's second pass opens a *fresh* WAL snapshot: the rows a stream emits are fresher than the digest stored beside them, and it converges in fewer emissions. The second walk was buying emission freshness, not wasting time. Trading ~12% more emissions for a win confined to wide results is a semantic-shaped trade in the dimension the library advertises, so the runtime is archived at `archive/exp-283` rather than shipped.", "status": "rejected", "link": "283-stream-rerun-one-pass.md" } diff --git a/experiments/signals/base.json b/experiments/signals/base.json index c13b9896..549fe52d 100644 --- a/experiments/signals/base.json +++ b/experiments/signals/base.json @@ -31,7 +31,7 @@ "wal", "checkpoint" ], - "currentRead": "Stream fan-out performance is shaped by rerun scheduling, reader-pool admission, completion-side churn, writer/request residual, and dependency precision. Queue changes have produced both strong wins and sharp regressions. Exp 120 closed the upstream over-dispatch in StreamEngine._flushQueue (parked_total drops 3,590 -> 0 on A11c overlap and 1,198 -> 0 on keyed-PK; max_parked 46 -> 0), and exp 122 removed the remaining stream-admission async boundary by constructing StreamEngine with a concrete ReaderPool. Exp 121 ruled out invalidation traversal as the active implementation target: overlap invalidation is 10-15% of wall, column intersection is 2.5-5.7%, and the structural ceiling is at the per-benchmark decision-threshold edge. Exp 134 proved row-level dirty precision can halve keyed-PK writer-burst wall for a narrow `WHERE id = ?` proof, but its internal SQL recognizer is rejected; revive that area only through explicit API/design or real workload evidence. Exp 136 ships the completion-side reader-handler counter: on A11c overlap the reader worker port handler is 28.57% of total wall (burst + drain) at ~18 us per call across 4,228 calls/burst, and subscriber-fanout emit is only 0.35% of the chain. Exp 147 split writer-side burst wall from SQLite-facing writer calls: on A11c overlap, SQLite is 15.7 ms / 166.8 ms (9.4%), invalidation is 18.8%, and residual writer/request wall is 71.8%; keyed-PK shows the same shape (18.1% SQLite, 18.7% invalidation, 63.3% residual). Exp 148 tested the natural reader-reply batching follow-up and rejected it: the profile smoke cut A11c overlap completion callbacks 4,527 -> 1,425 and completion wall 109.6 ms -> 55.6 ms, but Tracelite measured elapsed stayed neutral/slower (+5.18% high-cardinality, +3.28% many-streams, +13.5% keyed-PK). Exp 151 tested synchronous writer response resolution against the residual writer/request bucket and rejected it: high-cardinality fanout stayed neutral (+2.92%), keyed-PK trended slower with too-noisy evidence (+18.5%), and many-streams writer throughput trended slower (+14.0%). Exp 170 tested the matching request-side variant — `Mutex.tryLock` plus a non-`async` `Writer.execute` / `executeBatch` to drop the uncontended `await _mutex.lock()` microtask hop — and rejected it: the primary Single Inserts (100 sequential) / sequential-awaited (2000 writes) lanes stayed within ±2 % (wrong direction) across paired runs, while the only positive signal (-7.8 % on Concurrent Single Inserts) is a row exp 159 already drives at -58 % to -61 %. SQLite-step tuning, stream admission, invalidation traversal, plain worker-side reader-reply batching, synchronous writer response resolution, synchronous writer request acquisition, SQL-recognizer-based keyed-PK precision, allocation-only `_flushQueue` cleanup, and standalone residual-split profiling are not active targets on currently-measured workloads. Exp 159 attacked the residual structurally: persistent writer reply port + cached SendPort + sync FIFO completion remove fixed per-round-trip scheduling cost, and releasing the write lock at send time pipelines concurrent standalone writes through the worker port FIFO; the focused concurrent-burst benchmark improved 36-45%, exp 147 residual_us dropped on all four audit workloads, and stream-dispatch Tracelite guardrails were neutral on the clean order-flipped pass. Exp 161 closes the release-suite gap by promoting the concurrent-burst shape into `benchmark/suites/writes.dart` as a paired Single Inserts (100 sequential) / Concurrent Single Inserts (100 concurrent) row pair, so exp 159's pipelining win and future writer-scheduling experiments are evaluable on a public release lane (resqlite concurrent median ~1.1 ms vs ~2.9 ms sequential). Exp 171 then tried to apply exp 159's `_sendPort` cache pattern one layer up — a sync-readable `_resolvedRuntime` field on `Database` so post-open hot paths skip the `await _runtime` microtask hop — and rejected it: two order-flipped passes on writer_pipelining.dart produced alternating-sign deltas inside per-round variance (sequential-awaited -2.3%/+2.5%, transaction-guardrail -7.5%/+6.1%), so a ~1-2 us per-call hop sits at or below the harness floor. Database-layer microtask hop trimming is now off the candidate list. Exp 214 tested an even narrower writer-result decode cleanup after adding a microsecond public-write harness: direct pointer reads removed the typed-list/ByteData view around the 16-byte native result struct, but `write_result_direct_read.dart` did not reproduce a stable win across order-flipped passes (pair 1 mixed/small, pair 2 broadly candidate-slower, pair 3 mixed/neutral), so scalar `executeWrite()` result decoding is not an active target. Exp 197 then tested the moonshot version of the named group-commit ceiling by wrapping a coalesced standalone write burst in one SQLite transaction: writer_pipelining.dart's concurrent-burst lane improved -88.1% / -87.3% across order-flipped passes while sequential writes and explicit transactions stayed neutral, but the prototype is rejected as a hidden db.execute() default because it changes independent-autocommit read visibility, crash-window durability, and failure/atomicity semantics. Reopen true group commit only behind an explicit opt-in API or named mode, not another transport micro-path cleanup. Exp 228 found a correctness leak in exp 077's row-count hash shortcut: growth returned a partial digest that was cached as canonical, so the next unchanged rerun decoded and re-emitted once. Removing the shortcut took redundant grow-then-no-op decodes from 9/9 to 0/9 in each of three passes; combined grow + no-op p50 improved 19-36%, while pure-growth timing remained noisy around parity. Exp 239 revisited exp 209's request amortization without public API: bounded batching of only already-parked plain selects improves homogeneous twenty-way point reads 26-33% and roughly-ten-row reads 21-30%, while pool sharding keeps twenty 10k-row reads neutral-to-faster. It is rejected because queue depth cannot encode query cost: alternating large/point bursts regress point-completion p95 11-17% and total median 13-26%. Queue-depth-only hidden batching is closed; reopen only with a reliable private cost signal or independently completable batch members. Exp 249 (moonshot) then tested invalidation-grouped rerun batching — one write dirties every stream projecting the changed column, so it packed the dirtied set into one batched reader message per worker (SelectIfChangedBatchRequest + worker loop + batched _flushQueue/_requeryBatch, plus a lastRowCount cost-gate dispatching large partitions individually), all internal. Rejected: clean cross-worktree A/B measured single-write emission latency SLOWER on every lane (homogeneous +22% p50 / +59% p95, heterogeneous +66% p50 / +52% p95) because a batched reply is indivisible — the one changed stream waits behind its ~24 cheap unchanged batch-mates re-hashes. Message-count amortization is the wrong lever for this latency-bound workload (third rejection of the shared-indivisible-reply trade after exp 148 and exp 239). An in-process A/B toggle had reported a false -27% reproduced win; only cross-worktree exposed the true sign. Exp 271 then tested exp 184's remaining shared-memory completion lever without moving SQLite to the caller: a 24 us atomic-mailbox poll caught 69.9% of eligible completions but AOT no-op wall regressed 84.1%/62.7%, error wall regressed 205.1%/368.4%, and write throughput fell 65.6%/37.3%; a 32 us arm was worse. Active waiting is closed despite the high catch rate. Reopen writer completion transport only for a non-burning wait/notify or park/wake primitive, or representative evidence of a materially larger sequential-write floor. Exp 283 reopens the direction with the first measurement of what a rerun actually is, and closes its own candidate on what that measurement turned up. Per-stream coalescing collapses the high-cardinality shape's 20,000 arithmetic reruns into 847 real ones and 56.4% of those CHANGE (claim 283.1), so the '99 of 100 unchanged' reading describes writes, not reruns; keyed-PK (0.28%) and feed (0%) remain genuinely change-free, giving the direction two regimes rather than one. A changed rerun had been walking its statement twice since exp 075, and that second walk is 35-45% of its SQLite-and-decode work at every width from 1 to 1,000 rows (claim 283.2). Routing it through exp 097's one-pass decoder, predicted by one bool on StreamEntry (did the previous rerun change?), is worth a pooled -12.5% and -5.7% on two fan-out shapes across eight order-flipped AOT passes with both change-free guards neutral (claim 283.4). It is REJECTED anyway, because the second walk was not waste: `resqlite_query_hash` resets on exit, so the decode that follows opens a FRESH WAL snapshot, the rows a stream emits are newer than the digest stored beside them, and the stream converges in fewer emissions as a result. One-pass costs ~12% more emissions during a burst (claim 283.6) - a semantic-shaped trade in the dimension the library advertises. Reopen with a cheap freshness re-check instead of a second full walk, or with evidence that burst-time intermediate emissions are immaterial. Two measurement facts outlive the verdict: the end-to-end win is four times its own per-rerun model, so reader-side rerun savings may compound through the queue and the denominators used to reject exps 148/239/249-class candidates may understate them; and `high_cardinality_fanout`'s headline wall is one or two 200 ms settle windows around ~46 ms of work, bimodal at ~246 and ~448 ms on a convergence race, so a +91% flag on it is one window and the lane cannot size a candidate at all (claim 283.7).", + "currentRead": "Stream fan-out performance is shaped by rerun scheduling, reader-pool admission, completion-side churn, writer/request residual, and dependency precision. Queue changes have produced both strong wins and sharp regressions. Exp 120 closed the upstream over-dispatch in StreamEngine._flushQueue (parked_total drops 3,590 -> 0 on A11c overlap and 1,198 -> 0 on keyed-PK; max_parked 46 -> 0), and exp 122 removed the remaining stream-admission async boundary by constructing StreamEngine with a concrete ReaderPool. Exp 121 ruled out invalidation traversal as the active implementation target: overlap invalidation is 10-15% of wall, column intersection is 2.5-5.7%, and the structural ceiling is at the per-benchmark decision-threshold edge. Exp 134 proved row-level dirty precision can halve keyed-PK writer-burst wall for a narrow `WHERE id = ?` proof, but its internal SQL recognizer is rejected; revive that area only through explicit API/design or real workload evidence. Exp 136 ships the completion-side reader-handler counter: on A11c overlap the reader worker port handler is 28.57% of total wall (burst + drain) at ~18 us per call across 4,228 calls/burst, and subscriber-fanout emit is only 0.35% of the chain. Exp 147 split writer-side burst wall from SQLite-facing writer calls: on A11c overlap, SQLite is 15.7 ms / 166.8 ms (9.4%), invalidation is 18.8%, and residual writer/request wall is 71.8%; keyed-PK shows the same shape (18.1% SQLite, 18.7% invalidation, 63.3% residual). Exp 148 tested the natural reader-reply batching follow-up and rejected it: the profile smoke cut A11c overlap completion callbacks 4,527 -> 1,425 and completion wall 109.6 ms -> 55.6 ms, but Tracelite measured elapsed stayed neutral/slower (+5.18% high-cardinality, +3.28% many-streams, +13.5% keyed-PK). Exp 151 tested synchronous writer response resolution against the residual writer/request bucket and rejected it: high-cardinality fanout stayed neutral (+2.92%), keyed-PK trended slower with too-noisy evidence (+18.5%), and many-streams writer throughput trended slower (+14.0%). Exp 170 tested the matching request-side variant — `Mutex.tryLock` plus a non-`async` `Writer.execute` / `executeBatch` to drop the uncontended `await _mutex.lock()` microtask hop — and rejected it: the primary Single Inserts (100 sequential) / sequential-awaited (2000 writes) lanes stayed within ±2 % (wrong direction) across paired runs, while the only positive signal (-7.8 % on Concurrent Single Inserts) is a row exp 159 already drives at -58 % to -61 %. SQLite-step tuning, stream admission, invalidation traversal, plain worker-side reader-reply batching, synchronous writer response resolution, synchronous writer request acquisition, SQL-recognizer-based keyed-PK precision, allocation-only `_flushQueue` cleanup, and standalone residual-split profiling are not active targets on currently-measured workloads. Exp 159 attacked the residual structurally: persistent writer reply port + cached SendPort + sync FIFO completion remove fixed per-round-trip scheduling cost, and releasing the write lock at send time pipelines concurrent standalone writes through the worker port FIFO; the focused concurrent-burst benchmark improved 36-45%, exp 147 residual_us dropped on all four audit workloads, and stream-dispatch Tracelite guardrails were neutral on the clean order-flipped pass. Exp 161 closes the release-suite gap by promoting the concurrent-burst shape into `benchmark/suites/writes.dart` as a paired Single Inserts (100 sequential) / Concurrent Single Inserts (100 concurrent) row pair, so exp 159's pipelining win and future writer-scheduling experiments are evaluable on a public release lane (resqlite concurrent median ~1.1 ms vs ~2.9 ms sequential). Exp 171 then tried to apply exp 159's `_sendPort` cache pattern one layer up — a sync-readable `_resolvedRuntime` field on `Database` so post-open hot paths skip the `await _runtime` microtask hop — and rejected it: two order-flipped passes on writer_pipelining.dart produced alternating-sign deltas inside per-round variance (sequential-awaited -2.3%/+2.5%, transaction-guardrail -7.5%/+6.1%), so a ~1-2 us per-call hop sits at or below the harness floor. Database-layer microtask hop trimming is now off the candidate list. Exp 214 tested an even narrower writer-result decode cleanup after adding a microsecond public-write harness: direct pointer reads removed the typed-list/ByteData view around the 16-byte native result struct, but `write_result_direct_read.dart` did not reproduce a stable win across order-flipped passes (pair 1 mixed/small, pair 2 broadly candidate-slower, pair 3 mixed/neutral), so scalar `executeWrite()` result decoding is not an active target. Exp 197 then tested the moonshot version of the named group-commit ceiling by wrapping a coalesced standalone write burst in one SQLite transaction: writer_pipelining.dart's concurrent-burst lane improved -88.1% / -87.3% across order-flipped passes while sequential writes and explicit transactions stayed neutral, but the prototype is rejected as a hidden db.execute() default because it changes independent-autocommit read visibility, crash-window durability, and failure/atomicity semantics. Reopen true group commit only behind an explicit opt-in API or named mode, not another transport micro-path cleanup. Exp 228 found a correctness leak in exp 077's row-count hash shortcut: growth returned a partial digest that was cached as canonical, so the next unchanged rerun decoded and re-emitted once. Removing the shortcut took redundant grow-then-no-op decodes from 9/9 to 0/9 in each of three passes; combined grow + no-op p50 improved 19-36%, while pure-growth timing remained noisy around parity. Exp 239 revisited exp 209's request amortization without public API: bounded batching of only already-parked plain selects improves homogeneous twenty-way point reads 26-33% and roughly-ten-row reads 21-30%, while pool sharding keeps twenty 10k-row reads neutral-to-faster. It is rejected because queue depth cannot encode query cost: alternating large/point bursts regress point-completion p95 11-17% and total median 13-26%. Queue-depth-only hidden batching is closed; reopen only with a reliable private cost signal or independently completable batch members. Exp 249 (moonshot) then tested invalidation-grouped rerun batching — one write dirties every stream projecting the changed column, so it packed the dirtied set into one batched reader message per worker (SelectIfChangedBatchRequest + worker loop + batched _flushQueue/_requeryBatch, plus a lastRowCount cost-gate dispatching large partitions individually), all internal. Rejected: clean cross-worktree A/B measured single-write emission latency SLOWER on every lane (homogeneous +22% p50 / +59% p95, heterogeneous +66% p50 / +52% p95) because a batched reply is indivisible — the one changed stream waits behind its ~24 cheap unchanged batch-mates re-hashes. Message-count amortization is the wrong lever for this latency-bound workload (third rejection of the shared-indivisible-reply trade after exp 148 and exp 239). An in-process A/B toggle had reported a false -27% reproduced win; only cross-worktree exposed the true sign. Exp 271 then tested exp 184's remaining shared-memory completion lever without moving SQLite to the caller: a 24 us atomic-mailbox poll caught 69.9% of eligible completions but AOT no-op wall regressed 84.1%/62.7%, error wall regressed 205.1%/368.4%, and write throughput fell 65.6%/37.3%; a 32 us arm was worse. Active waiting is closed despite the high catch rate. Reopen writer completion transport only for a non-burning wait/notify or park/wake primitive, or representative evidence of a materially larger sequential-write floor. Exp 283 reopens the direction with the first measurement of what a rerun actually is, and closes its own candidate on what that measurement turned up. Per-stream coalescing collapses the high-cardinality shape's 20,000 arithmetic reruns into 847 real ones and 56.4% of those CHANGE (claim 283.1), so the '99 of 100 unchanged' reading describes writes, not reruns; keyed-PK (0.28%) and feed (0%) remain genuinely change-free, giving the direction two regimes rather than one. A changed rerun had been walking its statement twice since exp 075, and that second walk is 35-45% of its SQLite-and-decode work at every width from 1 to 1,000 rows (claim 283.2). Routing it through exp 097's one-pass decoder, predicted by one bool on StreamEntry, is a reproduced -11.3% on 20 streams x 1,000-row partitions and nothing measurable on 100 streams x 100-row partitions with both change-free guards neutral (claim 283.4) -- the mechanism scales with the result size it re-walks. It is REJECTED anyway, because the second walk was not waste: `resqlite_query_hash` resets on exit, so the decode that follows opens a FRESH WAL snapshot, the rows a stream emits are newer than the digest stored beside them, and the stream converges in fewer emissions as a result. One-pass costs ~12% more emissions during a burst (claim 283.6) - a semantic-shaped trade in the dimension the library advertises. Reopen with a cheap freshness re-check instead of a second full walk, gated on result size, or with evidence that burst-time intermediate emissions are immaterial. Two measurement facts outlive the verdict: `high_cardinality_fanout`'s headline wall is one or two 200 ms settle windows around ~46 ms of work, bimodal at ~246 and ~448 ms on a convergence race, so a +91% flag on it is one window and the lane cannot size a candidate at all (claim 283.7, issue #318); and a drain metric's sentinel must be unreachable by the burst it drains, or four order-flipped passes will agree on a number that is a race (claim 283.8).", "keyPriors": [ "120", "134", @@ -84,7 +84,7 @@ ], "openCandidates": [], "blockedOnMeasurement": [], - "notesForExperimenters": "The 2026-08-13 cadence refresh pruned the two April/May candidates: exp 120 removed stream-induced upstream over-dispatch on the measured A11c and keyed-PK stream lanes, leaving the parked-dispatcher sketch without representative incidence, and exp 249 superseded the generic rerun-batching direction with a durable cross-worktree latency gate. Reopen this direction only when an `interestingIf` trigger supplies a concrete reduction candidate. Avoid assuming a larger reader pool helps; exp 105 found the opposite under A11c fan-out. After exp 118 + exp 120, `dispatcherParkedTotal` and `dispatcherWakeRetryTotal` stay at zero on every measured stream workload — the parked-dispatcher signal is not the active target. Exp 121 took invalidation traversal off the candidate list (10–15% of overlap wall, 2.5–5.7% intersection, 80–200 ns per probe). Exp 122 keeps admission simple by giving StreamEngine a concrete ReaderPool and moving stream registry checks to diagnostics. Exp 134 is proof that row-level precision can win, but its SQL-recognizer implementation is rejected; revive it only with explicit API/design or real workload evidence. Exp 136 showed completion-side reader handling was large enough to try batching, but exp 148 proved plain worker-side reader-reply batching is not mergeable: it reduced callback counters while failing measured-elapsed primary scenarios. Exp 147 still leaves residual writer/request wall as the biggest bucket, but standalone residual splitting has reached diminishing returns. Exp 151 tried the narrow response-side variant (`Completer.sync()` for writer responses) and rejected it under Tracelite. Exp 170 tried the matching request-side variant (`Mutex.tryLock` + non-`async` `Writer.execute` to drop the uncontended `await _mutex.lock()` microtask hop) and rejected it: Single Inserts (100 sequential) and writer_pipelining `sequential-awaited (2000 writes)` both moved <2 % in the wrong direction across paired runs, and the only positive lane (Concurrent Single Inserts, −7.8 %) is one exp 159 already drives at −58 % to −61 %. Do not retry either scheduling tweak without new runtime or workload evidence. Exp 171 tried the same shape one layer up (cached `_resolvedRuntime` on `Database` to skip `await _runtime` on hot paths) and rejected it on focused-harness noise — Database-layer microtask hop trimming above the writer no longer moves the sequential-write floor. Exp 214 adds `benchmark/experiments/write_result_direct_read.dart` as the µs-scale writer-result floor harness and rejects direct native-pointer scalar reads; do not chase `executeWrite()` result decoding or exp 095-style writer result-buffer scratch again without a mechanism larger than view/allocation removal and reproduced order-flipped public-write deltas. Exp 182 took the residual bucket attack in a different direction — skip `preupdate_hook` accumulation + reply harvest when `_streamEngine.length == 0` via a `track_dirty` flag on `resqlite_db` + a `DrainRequest` no-op barrier on first stream registration — and rejected it: the no-stream wins are real (focused sequential −3.8 % / −5.3 %, wide-batch −2.4 % / −5.9 % across order-flipped passes) but the per-call gate adds reproduced overhead on the with-streams shape (focused +2.7 % / +3.7 %, classified `reproduced` by `ab_drift_check.dart`) and Tracelite stream-direction warmup elapsed regressed in the same direction across all three scenarios. Reactive streams are the library's primary use case, so the optimization helps a narrower workload mix than the one it slows — do not retry without a workload that shows write throughput without active streams is a hot path. A new stream experiment should try a concrete reduction candidate, add only the narrow measurement needed to explain the result in that same branch, remove temporary counters before merge unless they are reusable, and clear Tracelite measured-elapsed primary gates without harming keyed-PK. Exp 228 establishes a hash-cache invariant: cached `lastResultHash` must be canonical. An early-reject hash path must return a non-cacheable sentinel or finish the canonical digest during decode; never promote a partial digest to the next rerun baseline. Exp 239 closes queue-depth-only transparent read batching: its homogeneous gains are real, but alternating large/point completion p95 regresses in both orderings because members share an indivisible reply. Keep `select_overflow_batch.dart` as the heterogeneous gate, and fix the reader spawn-versus-close lifecycle before any future design deliberately increases aggregate sacrifice frequency. Exp 249 rejects invalidation-grouped rerun batching (indivisible reply delays the one changed result behind cheap batch-mates; reopen only for a throughput-bound stream workload with a design that preserves independent completion). Two methodology reminders from it: (1) A/B stream-dispatch/reader-message changes ACROSS WORKTREES, never with a single-process toggle — exp 249s in-process toggle classified REPRODUCED (-27%) while cross-worktree showed +22-66%, because both toggle arms shared warm JIT/isolate/pool state. (2) high_cardinality_fanout settle returns after a fixed emission-count quiet window and never waits for suppressed reruns to drain, so it cannot measure the 99-of-100-unchanged fan-out cost; use benchmark/experiments/stream_rerun_latency.dart (single-write per-emit latency, homogeneous + heterogeneous partitions) as the fan-out emission-latency gate. Exp 283 adds two harnesses, one measurement rule and one lane warning. `benchmark/experiments/stream_rerun_one_pass_ab.dart` is the gate for anything that changes what a rerun does on the reader side: keep `keyed-pk` and `feed` as predictor-off guards, `writes` as the zero-ceiling control, and keep the sentinel-drain metric - it issues the burst concurrently and times a sentinel write through to the one stream it must change, because the unchanged majority emits nothing and cannot be waited on (claim 283.5). `benchmark/experiments/stream_rerun_pass_price.dart` prices the hash pass, the decode pass and the one-pass decoder for a result shape in seconds with no pool and no isolates. Stop dividing stream count by write count when sizing a fan-out candidate - claim 283.1 is the corrected denominator - and do not size one against `high_cardinality_fanout`'s wall at all (claim 283.7). Before treating any rerun pass as removable, check what it re-reads: claim 283.6 is the case where a pass that looked like pure overhead was buying emission freshness. And when a flagged lane's behaviour differs but the mechanism is unclear, add a third arm that wires the change through every layer while compiling the hot branch out; exp 283's bookkeeping-only arm settled attribution in 14 runs that 70 primary-lane runs had not." + "notesForExperimenters": "The 2026-08-13 cadence refresh pruned the two April/May candidates: exp 120 removed stream-induced upstream over-dispatch on the measured A11c and keyed-PK stream lanes, leaving the parked-dispatcher sketch without representative incidence, and exp 249 superseded the generic rerun-batching direction with a durable cross-worktree latency gate. Reopen this direction only when an `interestingIf` trigger supplies a concrete reduction candidate. Avoid assuming a larger reader pool helps; exp 105 found the opposite under A11c fan-out. After exp 118 + exp 120, `dispatcherParkedTotal` and `dispatcherWakeRetryTotal` stay at zero on every measured stream workload — the parked-dispatcher signal is not the active target. Exp 121 took invalidation traversal off the candidate list (10–15% of overlap wall, 2.5–5.7% intersection, 80–200 ns per probe). Exp 122 keeps admission simple by giving StreamEngine a concrete ReaderPool and moving stream registry checks to diagnostics. Exp 134 is proof that row-level precision can win, but its SQL-recognizer implementation is rejected; revive it only with explicit API/design or real workload evidence. Exp 136 showed completion-side reader handling was large enough to try batching, but exp 148 proved plain worker-side reader-reply batching is not mergeable: it reduced callback counters while failing measured-elapsed primary scenarios. Exp 147 still leaves residual writer/request wall as the biggest bucket, but standalone residual splitting has reached diminishing returns. Exp 151 tried the narrow response-side variant (`Completer.sync()` for writer responses) and rejected it under Tracelite. Exp 170 tried the matching request-side variant (`Mutex.tryLock` + non-`async` `Writer.execute` to drop the uncontended `await _mutex.lock()` microtask hop) and rejected it: Single Inserts (100 sequential) and writer_pipelining `sequential-awaited (2000 writes)` both moved <2 % in the wrong direction across paired runs, and the only positive lane (Concurrent Single Inserts, −7.8 %) is one exp 159 already drives at −58 % to −61 %. Do not retry either scheduling tweak without new runtime or workload evidence. Exp 171 tried the same shape one layer up (cached `_resolvedRuntime` on `Database` to skip `await _runtime` on hot paths) and rejected it on focused-harness noise — Database-layer microtask hop trimming above the writer no longer moves the sequential-write floor. Exp 214 adds `benchmark/experiments/write_result_direct_read.dart` as the µs-scale writer-result floor harness and rejects direct native-pointer scalar reads; do not chase `executeWrite()` result decoding or exp 095-style writer result-buffer scratch again without a mechanism larger than view/allocation removal and reproduced order-flipped public-write deltas. Exp 182 took the residual bucket attack in a different direction — skip `preupdate_hook` accumulation + reply harvest when `_streamEngine.length == 0` via a `track_dirty` flag on `resqlite_db` + a `DrainRequest` no-op barrier on first stream registration — and rejected it: the no-stream wins are real (focused sequential −3.8 % / −5.3 %, wide-batch −2.4 % / −5.9 % across order-flipped passes) but the per-call gate adds reproduced overhead on the with-streams shape (focused +2.7 % / +3.7 %, classified `reproduced` by `ab_drift_check.dart`) and Tracelite stream-direction warmup elapsed regressed in the same direction across all three scenarios. Reactive streams are the library's primary use case, so the optimization helps a narrower workload mix than the one it slows — do not retry without a workload that shows write throughput without active streams is a hot path. A new stream experiment should try a concrete reduction candidate, add only the narrow measurement needed to explain the result in that same branch, remove temporary counters before merge unless they are reusable, and clear Tracelite measured-elapsed primary gates without harming keyed-PK. Exp 228 establishes a hash-cache invariant: cached `lastResultHash` must be canonical. An early-reject hash path must return a non-cacheable sentinel or finish the canonical digest during decode; never promote a partial digest to the next rerun baseline. Exp 239 closes queue-depth-only transparent read batching: its homogeneous gains are real, but alternating large/point completion p95 regresses in both orderings because members share an indivisible reply. Keep `select_overflow_batch.dart` as the heterogeneous gate, and fix the reader spawn-versus-close lifecycle before any future design deliberately increases aggregate sacrifice frequency. Exp 249 rejects invalidation-grouped rerun batching (indivisible reply delays the one changed result behind cheap batch-mates; reopen only for a throughput-bound stream workload with a design that preserves independent completion). Two methodology reminders from it: (1) A/B stream-dispatch/reader-message changes ACROSS WORKTREES, never with a single-process toggle — exp 249s in-process toggle classified REPRODUCED (-27%) while cross-worktree showed +22-66%, because both toggle arms shared warm JIT/isolate/pool state. (2) high_cardinality_fanout settle returns after a fixed emission-count quiet window and never waits for suppressed reruns to drain, so it cannot measure the 99-of-100-unchanged fan-out cost; use benchmark/experiments/stream_rerun_latency.dart (single-write per-emit latency, homogeneous + heterogeneous partitions) as the fan-out emission-latency gate. Exp 283 adds two harnesses, two measurement rules and one lane warning. `benchmark/experiments/stream_rerun_one_pass_ab.dart` is the gate for anything that changes what a rerun does on the reader side: keep `keyed-pk` and `feed` as predictor-off guards, `writes` as the zero-ceiling control, and keep the sentinel-drain metric - it issues the burst concurrently and times a sentinel write through to a stream the burst is EXCLUDED from (claim 283.8; the first version let the burst reach the sentinel and four order-flipped passes agreed on a withdrawn number). An awaited write-by-write burst sees nothing at all (claim 283.5). `benchmark/experiments/stream_rerun_pass_price.dart` prices the hash pass, the decode pass and the one-pass decoder for a result shape in seconds with no pool and no isolates. Stop dividing stream count by write count when sizing a fan-out candidate - claim 283.1 is the corrected denominator - and do not size one against `high_cardinality_fanout`'s wall at all (claim 283.7). Before treating any rerun pass as removable, check what it re-reads: claim 283.6 is the case where a pass that looked like pure overhead was buying emission freshness. And when a flagged lane's behaviour differs but the mechanism is unclear, add a third arm that wires the change through every layer while compiling the hot branch out; exp 283's bookkeeping-only arm settled attribution in 14 runs that 70 primary-lane runs had not." }, { "id": "parameter-encoding-and-binding", diff --git a/experiments/signals/entries/283.json b/experiments/signals/entries/283.json index 6699f2f7..7273164e 100644 --- a/experiments/signals/entries/283.json +++ b/experiments/signals/entries/283.json @@ -10,7 +10,7 @@ "We believed a stream rerun's changed path was one cheap C hash pass plus a Dart decode. It is two complete SQLite walks, and claim 283.2 prices the second at 35-45% of the rerun's SQLite-and-decode work at every width from one row to a thousand. That much was the expected finding. The unexpected one is claim 283.6: the two walks read two different snapshots, because `resqlite_query_hash` resets on exit and the decode that follows opens a fresh read transaction. The rows a stream emits are fresher than the digest stored beside them, and that accident is what makes it converge in as few emissions as it does. Anyone who reads the rerun path as 'hash, then decode if needed' should now read it as 'hash, then re-read', and should not treat the second walk as removable without replacing what it buys.", "We believed reactive fan-out reruns were overwhelmingly unchanged - the release suites' own docstrings say so, and `stream_rerun_latency.dart` is built around one changed stream in a hundred. Claim 283.1 shows that describes the *writes*. Per-stream coalescing collapses 20,000 arithmetic reruns into 847 real ones on the high-cardinality shape, and 56% of the survivors change. Stop dividing stream count by write count: the quantity that matters is reruns after coalescing, and it is smaller and much more change-dense than the shape suggests. Keyed-PK (0.28%) and feed (0%) remain genuinely change-free, so the direction has two regimes rather than one.", "We believed the release suite's largest resqlite lane measured stream fan-out performance. Claim 283.7 shows it mostly measures a 200 ms sleep, twice or once, decided by whether a convergence emission happens to land in a measured iteration. It cannot resolve anything smaller than a settle window, its two modes are ~246 ms and ~448 ms, and a +91% flag on it is one window. Do not size a stream candidate against it, and do not read its trend line as a performance signal; the drain metric in `stream_rerun_one_pass_ab.dart` is what to use instead.", - "A reader of exps 148, 239 and 249 should revisit how those rejections were sized, not their verdicts. Claim 283.4's end-to-end win is four times what its own per-rerun arithmetic predicts, on a metric that prices a backlog draining through four workers. If that gap is queueing, then reader-side rerun savings compound into wait times and the per-rerun denominators this direction has used to reject candidates understate them. Nothing here refutes those experiments - all three failed on latency shape, not magnitude - but the next runner should sweep pool size across one burst before quoting a per-rerun figure as a ceiling.", + "A measurement rule this run had to learn twice. Claim 283.5: an awaited write-by-write fan-out burst cannot see reader-side rerun cost, because each rerun hides inside the next write's latency. Claim 283.8: the drain metric that replaces it must use a sentinel the burst cannot reach, or each sample ends on a race and the race reads as a code effect. The first cost a metric that reported +1.2% on a lane the mechanism moves 11%; the second cost a four-pass collection that agreed on a withdrawn number until a reviewer looked at the harness. Both are properties of measuring work that runs concurrently with something you await, and neither is specific to streams.", "Claim 283.5 is a measurement rule for this direction, learned the expensive way. An awaited write-by-write fan-out burst cannot see reader-side rerun cost: each rerun hides inside the next write's latency, and a 25-sample pass of that metric put the candidate at +1.2% on a lane the mechanism moves 12%. Issue the burst concurrently and time a sentinel write through to the one stream it must change. The unchanged majority emits nothing and can never be waited on directly, which is exactly why the backlog needs something that can." ], "claims": [ @@ -39,8 +39,8 @@ }, { "id": "283.4", - "text": "Decoding during the hash pass when the previous rerun changed is worth a pooled -12.5% on 100 streams over 100-row partitions and -5.7% on 20 streams over 1,000-row partitions, with the keyed-PK and feed lanes neutral and a zero-ceiling write-only control at +0.6%. Eight order-flipped passes, two AOT bundles from separate worktrees, one lane per process, 41 samples after 8 warmup; `ab_drift_check.dart` classifies both fan-out lanes reproduced on the freshest pair (-25.5%/-23.5% and -15.4%/-5.1%) and the control drift-suspected. The runtime is nonetheless archived (claim 283.6). The end-to-end saving is about four times what claim 283.2's per-rerun figures and claim 283.3's hit counts predict (770 us per burst against 188 us), which is unexplained; the leading hypothesis is queueing, since the metric prices a backlog draining through four workers and service-time reductions compound into wait times.", - "conditions": "`benchmark/experiments/stream_rerun_one_pass_ab.dart`, AOT, arm64 macOS 26.2 (M1 Pro), Dart 3.12.2, load average 4.3-14.1 across the session; the zero-ceiling control swings +-8% in its worst pass and +-2% typically.", + "text": "Decoding during the hash pass when the previous rerun changed is worth a reproduced -11.3% pooled (-9% to -12% in every pass) on 20 streams over 1,000-row partitions, and nothing measurable on 100 streams over 100-row partitions: -1.7% pooled, sign-reversing across the order flip, classified drift-suspected against a zero-ceiling write-only control that itself moved +16.5% in its worst pass and +5.6% pooled. The keyed-PK and feed guards are neutral. Four order-flipped passes, two AOT bundles from separate worktrees, one lane per process, 41 samples after 8 warmup. The mechanism scales with the result size it re-walks (claim 283.2 prices the removed pass at 34.86 us on 1,000 rows and 4.22 us on 100), and at 100 rows there is not enough of it to clear a focused harness's floor. End to end the candidate delivers about 47% of its isolated mechanism on the wide lane -- 401 us saved per burst against 849 us modelled from claim 283.2's figures and claim 283.3's hit counts. The runtime is archived regardless (claim 283.6).", + "conditions": "`benchmark/experiments/stream_rerun_one_pass_ab.dart`, AOT, arm64 macOS 26.2 (M1 Pro), Dart 3.12.2, load average 1.8-4.1. An earlier collection with a defective sentinel -- a stream the burst could also change, with its completer armed before the burst -- reported -12.5% and -5.7% on these two lanes; those figures are withdrawn, see claim 283.8.", "edges": [ { "type": "dependsOn", @@ -62,7 +62,7 @@ }, { "id": "283.5", - "text": "A reactive fan-out A/B cannot be read off an awaited write-by-write burst. With each write awaited, every rerun overlaps the next write's latency and the wall reads as the write burst: the first metric tried here put the candidate within +-2% on its primary lane while the mechanism was worth -39% of a changed rerun. Issuing the burst concurrently and then timing a sentinel write through to the one stream it must change prices the whole backlog instead, because every rerun the burst scheduled - including the unchanged majority, which emits nothing and cannot be waited on - clears the queue first. The same eight-pass collection then reads -12.5%.", + "text": "A reactive fan-out A/B cannot be read off an awaited write-by-write burst. With each write awaited, every rerun overlaps the next write's latency and the wall reads as the write burst: the first metric tried here put the candidate within +-2% on its primary lane while the mechanism was worth -39% of a changed rerun. Issuing the burst concurrently and then timing a sentinel write through to the one stream it must change prices the whole backlog instead, because every rerun the burst scheduled - including the unchanged majority, which emits nothing and cannot be waited on - clears the queue first. The same eight-pass collection then reads -12.5%. A second requirement on the same metric: the sentinel must watch a partition the burst is excluded from, and its completer must be armed only after the burst has been issued. See claim 283.8.", "conditions": "Both metrics implemented in `benchmark/experiments/stream_rerun_one_pass_ab.dart`; the awaited variant was replaced, not kept.", "edges": [] }, @@ -86,6 +86,17 @@ "text": "`High-Cardinality Stream Fan-out (v1) / 100 streams x 200 writes / resqlite` cannot resolve anything below 200 ms, and its headline value is bimodal on a race rather than on performance. Its settle loop waits a 200 ms quiet window and stops at the first window with no new emission, so a measured iteration that emits nothing costs one window and one that emits anything costs two: the lane reads ~246 ms or ~448 ms and nothing between, against roughly 46 ms of actual work. Because the suite re-seeds its PRNG per iteration, the measured iterations rewrite values iteration 1 already wrote and the correct emission count is zero, but a stream that has not finished converging emits anyway - 2 of 35 runs on origin/main and 10 of 35 on exp 283's candidate. A +91.4% regression flag on this lane is one settle window. It has been the largest resqlite number in the release suite for months and most of it is a sleep. Filed as issue #318.", "conditions": "35 standalone invocations per arm, arm64 macOS 26.2 (M1 Pro), Dart 3.12.2. Measured while investigating the release run of 2026-09-06.", "edges": [] + }, + { + "id": "283.8", + "text": "A drain metric's sentinel must be unreachable by the workload it is draining. Exp 283's first harness armed its completer before the burst and let it be completed by any emission from an ordinary partition that the burst also wrote to, so each sample ended at whichever of that stream's reruns fired first -- a random prefix of the backlog, biased toward whichever arm finished changed reruns faster. It reported -12.5% and -5.7% on two fan-out lanes. Reserving a partition for the sentinel, excluding it from the burst, and arming the completer after the burst changed the same collection to -1.7% (drift-suspected) and -11.3% (reproduced): the headline lane's result was the defect and the wide lane's was understated. The general shape is that a sentinel which the measured work can also trigger converts a drain measurement into a race, and a race between two arms reads as a code effect.", + "conditions": "Found in review of exp 283's own PR (#319), not by the collection: four order-flipped passes had agreed on the wrong number. Both harness versions are in the branch history.", + "edges": [ + { + "type": "refines", + "target": "283.5" + } + ] } ], "nextSignals": [ @@ -93,8 +104,8 @@ "It also reopens on evidence that emission count is cheap - a downstream trace showing burst-time intermediate emissions are coalesced by the UI layer or otherwise immaterial. The trade claim 283.6 records is 12% more emissions for 12-16% less wall, and nothing in the repo prices an emission.", "`benchmark/experiments/stream_rerun_one_pass_ab.dart` is the durable gate for anything that changes what a rerun does on the reader side. Keep `keyed-pk` and `feed` as predictor-off guards, `writes` as the zero-ceiling control, and keep the sentinel-drain metric: claim 283.5 records what an awaited write burst hides.", "`benchmark/experiments/stream_rerun_pass_price.dart` prices the hash pass, the decode pass and the one-pass decoder for a result shape in seconds, with no pool and no isolates. Run it before proposing anything that changes how many times a rerun steps its statement.", - "Do not size a stream candidate by per-rerun cost alone (claim 283.4) and do not size one against `high_cardinality_fanout`'s wall at all (claim 283.7). Sweeping the reader pool size across one drain burst would settle whether the amplification is queueing and would recalibrate every past 'below the round-trip floor' rejection in this direction.", - "Methodology worth reusing: when a flagged lane's behaviour differs but the mechanism is unclear, add a third arm that wires the change through every layer while compiling the hot branch out. Exp 283's bookkeeping-only arm reproduced the baseline rate in 14 runs and settled attribution that 70 primary-lane runs had not." + "Methodology worth reusing: when a flagged lane's behaviour differs but the mechanism is unclear, add a third arm that wires the change through every layer while compiling the hot branch out. Exp 283's bookkeeping-only arm reproduced the baseline rate in 14 runs and settled attribution that 70 primary-lane runs had not.", + "Whether the reader pool is the fan-out constraint is open. Claim 283.4 shows a per-rerun saving discounting to ~47% end to end on the wide lane and to nothing on the narrow one, which is what a pipeline bound by something else looks like. Sweeping the pool size across one drain burst would identify it." ], "outcomeNote": "The mechanism reproduced and the end-to-end win reproduced; the runtime is archived because the pass it removes turns out to buy emission freshness (claim 283.6), which is a semantic-shaped trade a scheduled run should not settle alone." }