diff --git a/.github/workflows/desktop-cold-start.yml b/.github/workflows/desktop-cold-start.yml new file mode 100644 index 0000000000..4973f47128 --- /dev/null +++ b/.github/workflows/desktop-cold-start.yml @@ -0,0 +1,46 @@ +name: Desktop Cold Start +# Desktop cold-start A/B (T3 #5971). The runner script comes from the head tree, so head_sha must +# contain scripts/perf/desktopColdStart.ts. Use base_sha == head_sha for an A/A run. +on: + workflow_dispatch: + inputs: + base_sha: { description: "Base commit (40-hex SHA)", required: true, type: string } + head_sha: { description: "Head commit (40-hex SHA)", required: true, type: string } + pairs: { description: "Launch pairs", required: false, type: string, default: "20" } +permissions: + contents: read +jobs: + cold-start: + runs-on: ${{ github.repository_owner == 'coder' && 'depot-ubuntu-22.04-16' || 'ubuntu-latest' }} + timeout-minutes: 60 + env: # Inputs reach shell steps only through env (no template expansion inside run:). + BASE_SHA: ${{ inputs.base_sha }} + HEAD_SHA: ${{ inputs.head_sha }} + PAIRS: ${{ inputs.pairs }} + steps: + - name: Validate inputs + run: | + [[ "$BASE_SHA" =~ ^[0-9a-f]{40}$ && "$HEAD_SHA" =~ ^[0-9a-f]{40}$ ]] || { echo "SHAs must be 40-char lowercase hex"; exit 1; } + [[ "$PAIRS" =~ ^([2-9]|[1-9][0-9]+)$ ]] || { echo "pairs must be an integer >= 2"; exit 1; } + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 + with: + ref: ${{ inputs.head_sha }} + fetch-depth: 0 + persist-credentials: false + - uses: ./.github/actions/setup-xum + - name: Install xvfb + run: command -v xvfb-run >/dev/null 2>&1 || { sudo apt-get update && sudo apt-get install -y xvfb; } + - name: Build head (workspace root) and base ($RUNNER_TEMP/base) trees + run: | + make build + git worktree add --detach "$RUNNER_TEMP/base" "$BASE_SHA" + cd "$RUNNER_TEMP/base" && bun install --frozen-lockfile && make build + - name: Run cold-start A/B + env: { ELECTRON_DISABLE_SANDBOX: 1 } + run: xvfb-run -a node --import tsx scripts/perf/desktopColdStart.ts --base "$RUNNER_TEMP/base" --head "$GITHUB_WORKSPACE" --pairs "$PAIRS" --json > cold-start.json + - uses: actions/upload-artifact@bbbca2ddaa5d8feaa63e36b76fdaad77386f024f # v7.0.0 + if: always() + with: + name: desktop-cold-start-${{ github.run_id }} + path: cold-start.json + if-no-files-found: error diff --git a/scripts/perf/desktopColdStart.ts b/scripts/perf/desktopColdStart.ts new file mode 100644 index 0000000000..1716922f1d --- /dev/null +++ b/scripts/perf/desktopColdStart.ts @@ -0,0 +1,153 @@ +/** + * Desktop cold-start A/B runner (T3 #5971, desktop protocol D2). Each tree is a checkout after + * `make build`. A launch is a cold Electron start of the tree's dist (XUM_E2E=1, XUM_E2E_LOAD_DIST=1, + * mock AI) that reads the renderer's `xum:app-shell-ready` mark startTime. Pairs run ABBA (see + * launchPlan) so order effects and drift hit both arms alike. Each launch gets a fresh copy of the + * seeded root at one fixed path: config.json holds absolute paths, and state one launch writes must + * not speed up the next. Each tree runs its own Electron, so an Electron bump is measured too. + * Runs under Node (tsx), not Bun: under Bun 1.3.12 Playwright never finishes the websocket + * upgrade to Electron's debugger, so every launch hangs (`bun x playwright test` uses Node too). + */ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import * as fs from "node:fs"; +import { createRequire } from "node:module"; +import * as os from "node:os"; +import * as path from "node:path"; +import { fileURLToPath } from "node:url"; +import { parseArgs } from "node:util"; +import { _electron as electron } from "playwright"; +import { setXumE2EEnv } from "../../tests/e2e/env"; +import { prepareDemoProject } from "../../tests/e2e/utils/demoProject"; +import { launchPlan, pairedRelativeDelta, type PlannedLaunch } from "./desktopColdStartStats"; + +const MARK = "xum:app-shell-ready"; +const USAGE = + "usage: node --import tsx scripts/perf/desktopColdStart.ts --base --head [--pairs 20] [--warmups 2] [--json] [--timeout-ms 120000]"; +const E2E_ENV = { E2E: "1", E2E_LOAD_DIST: "1", MOCK_AI: "1", ENABLE_TUTORIALS_IN_SANDBOX: "0" }; +type Arm = PlannedLaunch["arm"]; +// Run in the renderer: Playwright serializes them, so they must not close over module state. +const markCount = (n: string) => performance.getEntriesByName(n, "mark").length; +const markTimes = (n: string) => performance.getEntriesByName(n, "mark").map((e) => e.startTime); +type Launch = PlannedLaunch & { index: number; warmup: boolean; markMs: number; wallMs: number }; + +function usageError(message: string): never { + console.error(`${message}\n${USAGE}`); + process.exit(2); +} + +function intArg(name: string, raw: string, min: number): number { + if (!/^[0-9]+$/.test(raw) || Number(raw) < min) usageError(`--${name}: integer >= ${min}`); + return Number(raw); +} + +/** Checks that the tree is built and returns its own Electron binary. */ +function electronFor(tree: string): string { + const manifest = path.join(tree, "package.json"); + if (!fs.existsSync(manifest)) usageError(`${tree}: no package.json`); + const { main } = JSON.parse(fs.readFileSync(manifest, "utf8")) as { main?: string }; + for (const rel of [main ?? "", "dist/index.html", "node_modules/electron"]) { + if (!fs.existsSync(path.join(tree, rel))) usageError(`${tree}: missing ${rel} (make build)`); + } + return createRequire(manifest)("electron") as string; +} + +function median(values: number[]): number { + const s = [...values].sort((a, b) => a - b); + return (s[(s.length - 1) >> 1] + s[s.length >> 1]) / 2; // odd length: the same index twice +} + +async function main(): Promise { + // Only the result goes to stdout: electron logs "Downloading..." there when its binary is missing. + const out = (line: string) => process.stdout.write(`${line}\n`); + console.log = console.error; + let values; + try { + ({ values } = parseArgs({ + options: { + base: { type: "string" }, + head: { type: "string" }, + pairs: { type: "string", default: "20" }, + warmups: { type: "string", default: "2" }, + json: { type: "boolean", default: false }, + "timeout-ms": { type: "string", default: "120000" }, + }, + })); + } catch (error) { + usageError(String(error)); + } + if (!values.base || !values.head) usageError("--base and --head are required"); + const pairs = intArg("pairs", values.pairs, 2); + const warmups = intArg("warmups", values.warmups, 0); + const timeout = intArg("timeout-ms", values["timeout-ms"], 1); + const tree = { base: path.resolve(values.base), head: path.resolve(values.head) }; + const bin = { base: electronFor(tree.base), head: electronFor(tree.head) }; + const tmp = fs.mkdtempSync(path.join(os.tmpdir(), "xum-cold-start-")); + // An exit handler, not `finally`: Playwright can reject outside our await and crash the process. + process.on("exit", () => fs.rmSync(tmp, { recursive: true, force: true })); + const [root, seed] = [path.join(tmp, "root"), path.join(tmp, "seed")]; + // Enumerating process.env yields only strings. setXumE2EEnv also replaces XUM_ROOT / MUX_ROOT. + const env = { ...process.env, NODE_ENV: "production" } as Record; + env.ELECTRON_DISABLE_SECURITY_WARNINGS = "true"; + for (const [k, v] of Object.entries({ ...E2E_ENV, ROOT: root })) setXumE2EEnv(env, k, v); + const launchOnce = async (arm: Arm) => { + const start = Date.now(); + const args = process.platform === "linux" ? ["--no-sandbox", "."] : ["."]; + const executablePath = bin[arm]; + const app = await electron.launch({ executablePath, cwd: tree[arm], args, env, timeout }); + try { + // E2E mode skips the splash, so the first window is the main window. + const window = await app.firstWindow({ timeout }); + await window.waitForFunction(markCount, MARK, { timeout }); + const startTimes = await window.evaluate(markTimes, MARK); + assert(startTimes.length === 1 && startTimes[0] > 0, `${MARK} marks: [${startTimes.join()}]`); + return { markMs: startTimes[0], wallMs: Date.now() - start }; + } finally { + await app.close(); + } + }; + const launches: Launch[] = []; + prepareDemoProject(root); + fs.cpSync(root, seed, { recursive: true }); + for (const [index, { arm, pair }] of launchPlan(pairs, warmups).entries()) { + fs.rmSync(root, { recursive: true, force: true }); + fs.cpSync(seed, root, { recursive: true }); + // No retries: retrying failed launches would bias the sample. + const result = await launchOnce(arm).catch((error: unknown) => { + throw new Error(`launch ${index} (${arm}) failed: ${String(error)}`); + }); + launches.push({ index, arm, pair, warmup: pair === null, ...result }); + console.error(`launch ${index} ${arm} ${pair ?? "warmup"}: ${result.markMs.toFixed(1)} ms`); + } + + // launchPlan visits pairs in increasing order, so index i of each arm is pair i. + const measured = (arm: Arm) => + launches.filter((l) => l.arm === arm && !l.warmup).map((l) => l.markMs); + const [base, head] = [measured("base"), measured("head")]; + const stats = pairedRelativeDelta(base, head); + if (values.json) { + const side = (arm: Arm) => { + const git = spawnSync("git", ["-C", tree[arm], "rev-parse", "HEAD"], { encoding: "utf8" }); + return { tree: tree[arm], gitSha: git.status === 0 ? git.stdout.trim() : null }; + }; + const result = { base: side("base"), head: side("head"), pairs, warmupsPerArm: warmups }; + out(JSON.stringify({ ...result, launches, stats }, null, 2)); + return; + } + const pct = (fraction: number) => `${(fraction * 100).toFixed(2)}%`; + out(["index", "arm", "pair", "mark ms", "wall ms"].join("\t")); + for (const l of launches) { + out([l.index, l.arm, l.pair ?? "warmup", l.markMs.toFixed(1), l.wallMs].join("\t")); + } + const { n, meanDelta, halfWidth95, upperBound95 } = stats; + out(`\nn=${n} mean delta ${pct(meanDelta)} one-sided 95% upper bound ${pct(upperBound95)}`); + out(`half-width ${pct(halfWidth95)} (<= 2.5%: ${halfWidth95 <= 0.025 ? "yes" : "no"})`); + out(`median mark ms: base ${median(base).toFixed(1)}, head ${median(head).toFixed(1)}`); +} + +if (fs.realpathSync(process.argv[1]) === fileURLToPath(import.meta.url)) { + main().catch((error: unknown) => { + console.error(error instanceof Error ? error.message : error); + process.exit(1); + }); +} diff --git a/scripts/perf/desktopColdStartStats.test.ts b/scripts/perf/desktopColdStartStats.test.ts new file mode 100644 index 0000000000..054d3068a7 --- /dev/null +++ b/scripts/perf/desktopColdStartStats.test.ts @@ -0,0 +1,52 @@ +import { describe, expect, test } from "bun:test"; +import { launchPlan, pairedRelativeDelta } from "./desktopColdStartStats"; + +const near = (value: number) => expect.closeTo(value, 12); + +describe("launchPlan", () => { + test("alternates warm-ups, runs pairs ABBA, and measures every pair once per arm", () => { + const order = launchPlan(3).map((launch) => `${launch.arm}:${launch.pair}`); + expect(order.join(" ")).toBe( + "base:null head:null base:null head:null base:0 head:0 head:1 base:1 base:2 head:2" + ); + // With no warm-ups, each arm visits pairs 0..19 once, in order. + for (const arm of ["base", "head"] as const) { + const pairs = launchPlan(20, 0).filter((launch) => launch.arm === arm); + expect(pairs.map((launch) => launch.pair)).toEqual([...Array(20).keys()]); + } + }); +}); + +describe("pairedRelativeDelta", () => { + test("hand-computed example: mean 0, sd 0.1, one-sided t(0.95, 2) = 2.920", () => { + const stats = pairedRelativeDelta([100, 100, 100], [110, 100, 90]); + const halfWidth = (2.92 * 0.1) / Math.sqrt(3); + expect([stats.n, stats.meanDelta, stats.sdDelta]).toEqual([3, near(0), near(0.1)]); + expect([stats.halfWidth95, stats.upperBound95]).toEqual([near(halfWidth), near(halfWidth)]); + }); + + test("a slower head is positive; identical arms give zero delta and half-width", () => { + const slower = pairedRelativeDelta([100, 200, 400], [110, 220, 440]); + expect([slower.meanDelta, slower.upperBound95]).toEqual([near(0.1), near(0.1)]); + const same = pairedRelativeDelta([500, 520, 480], [500, 520, 480]); + expect([same.meanDelta, same.halfWidth95]).toEqual([0, 0]); + }); + + test("uses t = 1.729 at 20 pairs (df 19) and keeps 1.697 above df 30", () => { + const alternating = (n: number) => + pairedRelativeDelta( + Array(n).fill(100), + [...Array(n).keys()].map((i) => (i % 2 ? 99 : 101)) + ); + // d alternates +1% / -1%, so halfWidth = t * sd / sqrt(n) = t * 0.01 / sqrt(n - 1). + expect(alternating(20).halfWidth95).toBeCloseTo((1.729 * 0.01) / Math.sqrt(19), 12); + expect(alternating(40).halfWidth95).toBeCloseTo((1.697 * 0.01) / Math.sqrt(39), 12); + }); + + test("rejects mismatched, short, or non-positive input", () => { + expect(() => pairedRelativeDelta([100, 100], [100])).toThrow(); + expect(() => pairedRelativeDelta([100], [100])).toThrow(); + expect(() => pairedRelativeDelta([100, 0], [100, 100])).toThrow(); + expect(() => pairedRelativeDelta([100, 100], [100, Number.NaN])).toThrow(); + }); +}); diff --git a/scripts/perf/desktopColdStartStats.ts b/scripts/perf/desktopColdStartStats.ts new file mode 100644 index 0000000000..cb70b2afe0 --- /dev/null +++ b/scripts/perf/desktopColdStartStats.ts @@ -0,0 +1,47 @@ +/** Pure helpers for scripts/perf/desktopColdStart.ts (T3 #5971, desktop protocol D2). */ +import assert from "node:assert/strict"; + +export type Arm = "base" | "head"; +export interface PlannedLaunch { + arm: Arm; + pair: number | null; // null = warm-up +} + +// One-sided 95% (= two-sided 90%) Student t critical values for df 1..30 (index = df - 1). +// benchStats.ts holds the two-sided 95% table, which is the wrong quantile for an upper bound. +const T_ONE_SIDED_95 = [ + 6.314, 2.92, 2.353, 2.132, 2.015, 1.943, 1.895, 1.86, 1.833, 1.812, 1.796, 1.782, 1.771, 1.761, + 1.753, 1.746, 1.74, 1.734, 1.729, 1.725, 1.721, 1.717, 1.714, 1.711, 1.708, 1.706, 1.703, 1.701, + 1.699, 1.697, +]; + +/** Warm-ups alternate base, head; pairs then run ABBA (base-head, head-base, base-head, ...). */ +export function launchPlan(pairs: number, warmupsPerArm = 2): PlannedLaunch[] { + assert(Number.isInteger(pairs) && pairs >= 2, `pairs must be an integer >= 2, got ${pairs}`); + assert(Number.isInteger(warmupsPerArm) && warmupsPerArm >= 0, `bad warm-ups ${warmupsPerArm}`); + const plan: PlannedLaunch[] = []; + for (let i = 0; i < warmupsPerArm; i++) { + plan.push({ arm: "base", pair: null }, { arm: "head", pair: null }); + } + for (let pair = 0; pair < pairs; pair++) { + const order: Arm[] = pair % 2 === 0 ? ["base", "head"] : ["head", "base"]; + for (const arm of order) plan.push({ arm, pair }); + } + return plan; +} + +/** Per pair d = (head - base) / base. Fractions: 0.012 = 1.2%. */ +export function pairedRelativeDelta(base: number[], head: number[]) { + assert(base.length === head.length, `arm sizes differ: ${base.length} vs ${head.length}`); + assert(base.length >= 2, `need at least 2 pairs, got ${base.length}`); + const bad = [...base, ...head].find((value) => !(Number.isFinite(value) && value > 0)); + assert(bad === undefined, `invalid value ${bad}`); + const n = base.length; + const deltas = base.map((b, i) => (head[i] - b) / b); + const meanDelta = deltas.reduce((sum, d) => sum + d, 0) / n; + const sdDelta = Math.sqrt(deltas.reduce((sum, d) => sum + (d - meanDelta) ** 2, 0) / (n - 1)); + // Above df 30 keep t(30): slightly wide, never too narrow (same convention as benchStats.ts). + const t = T_ONE_SIDED_95[Math.min(n - 1, T_ONE_SIDED_95.length) - 1]; + const halfWidth95 = (t * sdDelta) / Math.sqrt(n); + return { n, meanDelta, sdDelta, halfWidth95, upperBound95: meanDelta + halfWidth95 }; +}