// Microbenchmark: batch-of-25 frames (22 small results + 3 x 512KB results) // through the batch-processing hot path, plus final payload serialization. // Usage: npx tsx bench/runBatch.bench.mts [iterations] // Prints JSON: { iterations, samples_ms, stats } import { performance } from "node:perf_hooks"; import { runBatchBaseline, payloadBaseline, makeFixture } from "./baseline.ts"; import { runBatch, assemblePayload, type BidiResponse, type Frame, type BatchCtx } from "../lib.ts"; const ITER = Number(process.argv[2] ?? 15); function stats(samples: number[]) { const sorted = [...samples].sort((a, b) => a - b); const mean = samples.reduce((a, b) => a + b, 0) / samples.length; const sd = Math.sqrt(samples.reduce((a, b) => a + (b - mean) ** 2, 0) / samples.length); return { n: samples.length, median: sorted[Math.floor(sorted.length / 2)], mean, sd, min: sorted[0], max: sorted[sorted.length - 1], p90: sorted[Math.floor(sorted.length * 0.9)] }; } // Warmup: JIT + fs settle { const fx = await makeFixture(); await runBatchBaseline(fx.ws, fx.ctx, fx.frames, fx.byId, 30000); const fx2 = await makeFixture(); await runBatch(fx2.ws, fx2.ctx, fx2.frames as Frame[], fx2.byId, 30000); } // Interleaved A/B pairs cancel machine drift; per-pair differences feed the // signed-rank test. Fixture construction (fs mkdir) stays outside the timer. const pairs: { base: number; cand: number }[] = []; for (let i = 0; i < ITER; i++) { const fx = await makeFixture(); const t0 = performance.now(); const base = await runBatchBaseline(fx.ws, fx.ctx, fx.frames, fx.byId, 30000); const baseText = payloadBaseline({ sessionId: "s", port: 9222, capabilities: {} }, base, [], undefined, 200); pairs.push({ base: performance.now() - t0, cand: 0 }); const fx2 = await makeFixture(); const t1 = performance.now(); const cand = await runBatch(fx2.ws, fx2.ctx, fx2.frames as Frame[], fx2.byId, 30000); const candText = assemblePayload({ sessionId: "s", port: 9222, capabilities: {} }, cand.texts, [], 200, undefined); pairs[i].cand = performance.now() - t1; if (i === 0) { const norm = (t: string, dir: string) => JSON.stringify(JSON.parse(t.replaceAll(dir, ""))); if (norm(baseText, fx.ctx.spoolDir) !== norm(candText, fx2.ctx.spoolDir)) { console.error("OUTPUT MISMATCH between baseline and candidate"); process.exit(1); } } } const baseSamples = pairs.map((p) => p.base); const candSamples = pairs.map((p) => p.cand); const diffs = pairs.map((p) => p.base - p.cand); // Wilcoxon signed-rank (normal approximation, no zero-diff correction) function wilcoxon(d: number[]): number { const nz = d.filter((x) => x !== 0).map((x, i) => ({ abs: Math.abs(x), sign: Math.sign(x), rank: 0 })); nz.sort((a, b) => a.abs - b.abs); nz.forEach((o, i) => (o.rank = i + 1)); const wPlus = nz.filter((o) => o.sign > 0).reduce((a, o) => a + o.rank, 0); const n = nz.length; const mu = (n * (n + 1)) / 4; const sigma = Math.sqrt((n * (n + 1) * (2 * n + 1)) / 24); const z = (wPlus - mu) / sigma; const phi = 0.5 * (1 + erf(Math.abs(z) / Math.SQRT2)); return 2 * (1 - phi); } function erf(x: number): number { const t = 1 / (1 + 0.3275911 * Math.abs(x)); const y = 1 - ((((1.061405429 * t - 1.453152027) * t + 1.421413741) * t - 0.284496736) * t + 0.254829592) * t * Math.exp(-x * x); return x >= 0 ? y : -y; } console.log(JSON.stringify({ iterations: ITER, samples_ms: { baseline: baseSamples, candidate: candSamples }, paired_diff_ms: { mean: diffs.reduce((a, b) => a + b, 0) / diffs.length }, stats: { baseline: stats(baseSamples), candidate: stats(candSamples) }, wilcoxon_p: wilcoxon(diffs), output_equal: true, }));